{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![](https://storage.googleapis.com/kaggle-media/competitions/Harvard%20Medical%20School/eFig2.png)","metadata":{}},{"cell_type":"markdown","source":"## Brain activity notebook series\n\n### [EEGS 10–20 system](https://www.kaggle.com/code/seshurajup/eegs-10-20-system)\nBetter understanding eegs 10-20 system\n### [Missing Eeg_ids Train.csv vs train_eegs [Resolved]](https://www.kaggle.com/code/seshurajup/missing-eeg-ids-in-train-csv-vs-train-eegs-parquet)\nExtra training eggs [Resolved] as we can ignore it\n### [EDA train.csv](https://www.kaggle.com/code/seshurajup/eda-train-csv)\nDetailed analysis of the train.csv\n### [Eegs Pairing Analysis & Features](https://www.kaggle.com/code/seshurajup/eegs-pairing-analysis-features)\nPairing features analysis and build features\n### [Eegs Target Analysis - Correct way to merge target](https://www.kaggle.com/code/seshurajup/eegs-target-analysis-correct-way-to-merge-target)\nHow to choice the target votes for training\n### [Eegs Train Split (CV)](https://www.kaggle.com/seshurajup/eegs-train-splits-cv)\ngenerate better train split without patient_id overlap\n\n#### **Upvote my work if it is useful**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nfrom IPython.display import Image, display, HTML\nsample = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1000913311.parquet\")\ntrain = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-01-15T08:51:08.330524Z","iopub.execute_input":"2024-01-15T08:51:08.330908Z","iopub.status.idle":"2024-01-15T08:51:10.585431Z","shell.execute_reply.started":"2024-01-15T08:51:08.330874Z","shell.execute_reply":"2024-01-15T08:51:10.584262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pywt\ndef maddest(d, axis=None):\n    return np.mean(np.absolute(d - np.mean(d, axis)), axis)\n\ndef denoise(x, wavelet='haar', level=1):\n    ret = {key:[] for key in x.columns}\n    \n    for pos in x.columns:\n        coeff = pywt.wavedec(x[pos], wavelet, mode=\"per\")\n        sigma = (1/0.6745) * maddest(coeff[-level])\n\n        uthresh = sigma * np.sqrt(2*np.log(len(x)))\n        coeff[1:] = (pywt.threshold(i, value=uthresh, mode='hard') for i in coeff[1:])\n\n        ret[pos]=pywt.waverec(coeff, wavelet, mode='per')\n    \n    return pd.DataFrame(ret)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T08:51:10.587335Z","iopub.execute_input":"2024-01-15T08:51:10.587747Z","iopub.status.idle":"2024-01-15T08:51:10.695246Z","shell.execute_reply.started":"2024-01-15T08:51:10.587707Z","shell.execute_reply":"2024-01-15T08:51:10.694435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LL\n![](https://www.googleapis.com/download/storage/v1/b/kaggle-forum-message-attachments/o/inbox%2F761268%2F42abf3341243af65c92c8ba2083b9a79%2FScreenshot%202024-01-14%20at%205.49.27PM.png?generation=1705234855564208&alt=media)\n\n#LP\n![](https://www.googleapis.com/download/storage/v1/b/kaggle-forum-message-attachments/o/inbox%2F761268%2Fa725bf3a965b4299d0af07f4937216f5%2FScreenshot%202024-01-14%20at%205.49.38PM.png?generation=1705234875852002&alt=media)\n\n# RP\n![](https://www.googleapis.com/download/storage/v1/b/kaggle-forum-message-attachments/o/inbox%2F761268%2F5869dccab21b6cc3ba7b31e849622d5a%2FScreenshot%202024-01-14%20at%205.50.02PM.png?generation=1705234897860474&alt=media)\n\n# RR\n![](https://www.googleapis.com/download/storage/v1/b/kaggle-forum-message-attachments/o/inbox%2F761268%2Fa13eff4393d4fc67fbf3a2bc52294553%2FScreenshot%202024-01-14%20at%205.50.12PM.png?generation=1705234918197292&alt=media)","metadata":{}},{"cell_type":"code","source":"pairing = {\n    \"Fp1\": \"F7\",\n    \"F7\": \"T3\",\n    \"T3\": \"T5\",\n    \"T5\": \"O1\",\n    \"Fp2\": \"F8\",\n    \"F8\": \"T4\",\n    \"T4\": \"T6\",\n    \"T6\": \"O2\",\n    \"Fp1\": \"F3\",\n    \"F3\": \"C3\",\n    \"C3\": \"P3\",\n    \"P3\": \"O1\",\n    \"Fp2\": \"F4\",\n    \"F4\": \"C4\",\n    \"C4\": \"P4\",\n    \"P4\": \"O2\",\n    \"Fz\": \"Cz\",\n    \"Cz\": \"Pz\"\n}\n","metadata":{"execution":{"iopub.status.busy":"2024-01-15T08:51:10.696579Z","iopub.execute_input":"2024-01-15T08:51:10.696944Z","iopub.status.idle":"2024-01-15T08:51:10.703403Z","shell.execute_reply.started":"2024-01-15T08:51:10.696903Z","shell.execute_reply":"2024-01-15T08:51:10.702632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_comparision(data):\n    data = denoise(data, wavelet='db8')\n    plt.figure(figsize=(15, 45))\n    for i, (channel1, channel2) in enumerate(pairing.items()):\n        plt.subplot(len(pairing), 1, i + 1)\n        sns.lineplot(data=data[[channel1, channel2]], alpha=0.5)\n        plt.title(f\"Comparison of {channel1} and {channel2}\")\n        plt.xlabel(\"Time\")\n        plt.ylabel(\"Values\")\n        plt.legend([channel1, channel2])\n\n    plt.tight_layout()\n    plt.show()\nplot_comparision(sample)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T08:51:10.705623Z","iopub.execute_input":"2024-01-15T08:51:10.706484Z","iopub.status.idle":"2024-01-15T08:51:17.579197Z","shell.execute_reply.started":"2024-01-15T08:51:10.706425Z","shell.execute_reply":"2024-01-15T08:51:17.577740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_difference(data):\n    data = denoise(data, wavelet='db8')\n    plt.figure(figsize=(15, 45))\n    for i, (channel1, channel2) in enumerate(pairing.items()):\n        plt.subplot(len(pairing), 1, i + 1)\n        difference = data[channel1] - data[channel2]\n        sns.lineplot(data=difference)\n        plt.title(f\"Difference between {channel1} and {channel2}\")\n        plt.xlabel(\"Time\")\n        plt.ylabel(\"Difference in Values\")\n    plt.tight_layout()\n    plt.show()\nplot_difference(sample)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T08:51:17.581286Z","iopub.execute_input":"2024-01-15T08:51:17.581764Z","iopub.status.idle":"2024-01-15T08:51:22.227789Z","shell.execute_reply.started":"2024-01-15T08:51:17.581710Z","shell.execute_reply":"2024-01-15T08:51:22.226543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! mkdir /kaggle/working/samples","metadata":{"execution":{"iopub.status.busy":"2024-01-15T08:51:22.229340Z","iopub.execute_input":"2024-01-15T08:51:22.230050Z","iopub.status.idle":"2024-01-15T08:51:23.246427Z","shell.execute_reply.started":"2024-01-15T08:51:22.230010Z","shell.execute_reply":"2024-01-15T08:51:23.244870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! for i in {01..20}; do convert -density 300 /kaggle/input/hms-harmful-brain-activity-classification/example_figures/Sample$i.pdf -quality 100 /kaggle/working/samples/Sample$i.png; done","metadata":{"execution":{"iopub.status.busy":"2024-01-15T08:51:23.248764Z","iopub.execute_input":"2024-01-15T08:51:23.249282Z","iopub.status.idle":"2024-01-15T08:53:22.815743Z","shell.execute_reply.started":"2024-01-15T08:51:23.249237Z","shell.execute_reply":"2024-01-15T08:53:22.814433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Sample_filters={\n    1: {\"seizure_vote\":20},\n    2: {\"seizure_vote\": 10, \"other_vote\": 10},\n    3: {\"seizure_vote\": 7, \"lpd_vote\": 6},\n    4: {\"seizure_vote\": 10, \"gpd_vote\": 9},\n    5: {\"lpd_vote\": 26},\n    6: {\"lpd_vote\": 15, \"other_vote\": 14},\n    7: {\"lpd_vote\": 7, \"gpd_vote\": 7},\n    8: {\"lpd_vote\": 7, \"lrda_vote\": 7},\n    9: {\"gpd_vote\": 22},\n    10: {\"gpd_vote\":10, \"other_vote\": 10},\n    11: {\"gpd_vote\":6, \"lrda_vote\": 10},\n    12: {\"gpd_vote\":9, \"grda_vote\": 9},\n    13: {\"lrda_vote\": 21},\n    14: {\"lrda_vote\": 10, \"other_vote\": 10},\n    15: {\"lrda_vote\": 6, \"seizure_vote\": 6},\n    16: {\"lrda_vote\": 7, \"grda_vote\": 7},\n    17: {\"grda_vote\": 18},\n    18: {\"grda_vote\": 11, \"other_vote\": 11},\n    19: {\"grda_vote\": 6, \"seizure_vote\": 6},\n    20: {\"grda_vote\": 7, \"lpd_vote\": 7},\n}","metadata":{"execution":{"iopub.status.busy":"2024-01-15T08:53:22.817529Z","iopub.execute_input":"2024-01-15T08:53:22.817874Z","iopub.status.idle":"2024-01-15T08:53:22.827033Z","shell.execute_reply.started":"2024-01-15T08:53:22.817842Z","shell.execute_reply":"2024-01-15T08:53:22.825925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_sample_num(num):\n    filters = Sample_filters[num]\n    filters_txt = \", \".join([f\"{label}: {freq}\" for label, freq in filters.items()])\n    display(HTML(f\"<h1>Sample {num:02d} {filters_txt}</h1>\"))\n    display(Image(f\"./samples/Sample{num:02d}.png\", embed=True, width=1024))\n    sample_query = \" & \".join([f\"{label} == {count}\" for label, count in filters.items()])\n    selected_train = train.query(sample_query).reset_index(drop=True)\n    print(\"Total Samples\", selected_train.shape)\n    if selected_train.shape[0] == 0:\n        print(f\"No samples of this Sample{num:02d}.pdf\")\n    else:\n        eeg_id = selected_train['eeg_id'][0]\n        sample = pd.read_parquet(f\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/{eeg_id}.parquet\")\n        plot_difference(sample)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T08:53:22.829911Z","iopub.execute_input":"2024-01-15T08:53:22.830263Z","iopub.status.idle":"2024-01-15T08:53:22.845964Z","shell.execute_reply.started":"2024-01-15T08:53:22.830234Z","shell.execute_reply":"2024-01-15T08:53:22.844643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(1, 21):\n    plot_sample_num(i)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T08:53:22.847593Z","iopub.execute_input":"2024-01-15T08:53:22.848126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! rm -rf /kaggle/working/samples","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pairs = []\nfor index, (channel1, channel2) in enumerate(pairing.items()):\n    pairs.append((index, channel1, channel2))\nstr(pairs)","metadata":{"execution":{"iopub.status.busy":"2024-01-15T07:59:29.402664Z","iopub.execute_input":"2024-01-15T07:59:29.403259Z","iopub.status.idle":"2024-01-15T07:59:29.412832Z","shell.execute_reply.started":"2024-01-15T07:59:29.403213Z","shell.execute_reply":"2024-01-15T07:59:29.411343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def compute_difference(data, pairing):\n    data = denoise(data, wavelet='db8')\n    differences = []\n    for (index, channel1, channel2) in pairs:\n        differences.append(data[channel1] - data[channel2])\n    return differences\nsample_dif = compute_difference(sample, pairing)\nlen(sample_dif), len(sample_dif[0])","metadata":{"execution":{"iopub.status.busy":"2024-01-15T08:05:42.533165Z","iopub.execute_input":"2024-01-15T08:05:42.533817Z","iopub.status.idle":"2024-01-15T08:05:42.584153Z","shell.execute_reply.started":"2024-01-15T08:05:42.533762Z","shell.execute_reply":"2024-01-15T08:05:42.582746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ~~uncomment below code and build pairing features as it need more than 20GB~~\n## Resolved issue with more than 20GB limit","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\nTARGETS = [x for x in df.columns if 'vote' in x]\ntrain = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','min']\n\ntmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp\n\ntmp = df.groupby('eeg_id')[['patient_id']].agg('first')\ntrain['patient_id'] = tmp\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n    \ny_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\ntrain = train.reset_index()\nprint('Train non-overlapp eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-15T08:05:48.884464Z","iopub.execute_input":"2024-01-15T08:05:48.884965Z","iopub.status.idle":"2024-01-15T08:05:49.229462Z","shell.execute_reply.started":"2024-01-15T08:05:48.884925Z","shell.execute_reply":"2024-01-15T08:05:49.227985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! rm -rf /kaggle/working/\n! mkdir pair_features","metadata":{"execution":{"iopub.status.busy":"2024-01-15T08:18:34.211756Z","iopub.execute_input":"2024-01-15T08:18:34.212231Z","iopub.status.idle":"2024-01-15T08:18:36.705304Z","shell.execute_reply.started":"2024-01-15T08:18:34.212195Z","shell.execute_reply":"2024-01-15T08:18:36.703311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i,row in tqdm(train.iterrows(), total=len(train)):\n    eeg_id = row['eeg_id']\n    sample = pd.read_parquet(f\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/{eeg_id}.parquet\")\n    sample = sample.fillna(0)\n    pair_data = compute_difference(sample, pairing)\n    egg = np.array(pair_data)\n    max_val = np.finfo(np.float16).max\n    min_val = np.finfo(np.float16).min\n    egg = np.clip(egg, min_val, max_val).T.astype(\"float16\")\n    np.save(f\"/kaggle/working/pair_features/{eeg_id}.npy\", np.array(pair_data,dtype=np.float16))","metadata":{"execution":{"iopub.status.busy":"2024-01-15T08:20:37.349787Z","iopub.execute_input":"2024-01-15T08:20:37.350209Z","iopub.status.idle":"2024-01-15T08:42:13.233732Z","shell.execute_reply.started":"2024-01-15T08:20:37.350176Z","shell.execute_reply":"2024-01-15T08:42:13.232340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}