{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":28922,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":24355}],"dockerImageVersionId":30674,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"###Imports\nimport numpy as np \nimport pandas as pd \nimport os\n\nfrom scipy import signal\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn import preprocessing\nimport pickle\nimport joblib  \n\nimport multiprocessing as mp\nfrom pathos.multiprocessing import Pool\nimport tqdm\n\nimport matplotlib.pyplot as plt\nimport plotly.express as px","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_master = pd.read_csv(\"../input/hms-harmful-brain-activity-classification/train.csv\")\n\ndisplay(train_master)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#determine the expert confidence in each classification\nvote_cols = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\ndef confidence(r):\n    \n    vote_vals = r[vote_cols].values\n    \n    confidence = max(vote_vals)/sum(vote_vals) * 100\n    \n    return confidence\n\ntrain_master['expert_confidence'] = train_master.apply(confidence, axis=1)\n\n#Examples with 100% confidence are 'textbook', everything else is not totally agreed upon\ntextbook_examples = train_master.loc[train_master['expert_confidence'] == 100]\nother_examples = train_master.loc[train_master['expert_confidence'] != 100]\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pairs = [['Fp1', 'F7'], ['F7', 'T3'], ['T3', 'T5'], \n         ['T5', 'O1'], ['Fp2', 'F8'], ['F8', 'T4'], \n         ['T4', 'T6'], ['T6', 'O2'], ['Fp1', 'F3'], \n        ['F3', 'C3'], ['C3', 'P3'], ['P3', 'O1'], \n        ['Fp2', 'F4'], ['F4', 'C4'], ['C4', 'P4'], \n         ['P4', 'O2'], ['Fz', 'Cz'], ['Cz', 'Pz']]\n\n\ndef get_features_for_entry(ir, q=2):\n    \n    r = ir[1]\n    \n    ex_eeg = pd.read_parquet(\"../input/hms-harmful-brain-activity-classification/train_eegs/{}.parquet\".format(r['eeg_id']))\n    ex_spec = pd.read_parquet(\"../input/hms-harmful-brain-activity-classification/train_spectrograms/{}.parquet\".format(r['spectrogram_id']))\n    \n    eeg_offset = int(r['eeg_label_offset_seconds'])\n    spec_offset = int(r['spectrogram_label_offset_seconds'])\n    \n    eeg = ex_eeg.iloc[eeg_offset*200:(eeg_offset+50)*200]\n    eeg_subsample = eeg.iloc[20*200:30*200]\n    times = np.arange(0, eeg_subsample.shape[0]/200, 1/200.) - 5.\n    \n    middle_50_spec = ex_spec.loc[(ex_spec.time>=spec_offset+275)\n                     &(ex_spec.time<spec_offset+325)]\n \n    \n    output_eeg_X = np.array([])\n    \n    for p in pairs:\n        row = np.nan_to_num(eeg_subsample[p[0]].values - eeg_subsample[p[1]].values)\n        #downsample\n        row = signal.decimate(row, q)\n        \n        if output_eeg_X.shape[0] == 0:\n            output_eeg_X = preprocessing.normalize([row])[0]\n        else:\n            output_eeg_X = np.vstack((output_eeg_X, preprocessing.normalize([row])[0]))\n    \n            \n    return pd.Series({\"X_eeg\": output_eeg_X, \n                      'spec_values': middle_50_spec.iloc[:, 1:].values})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result = Pool(8).imap(get_features_for_entry, textbook_examples.iterrows(), chunksize=1)\n\nrdf = pd.DataFrame(tqdm.tqdm(result, total = textbook_examples.shape[0]))\n","metadata":{"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"textbook_train_examples_dataframe.pkl\", 'wb') as f:\n    pickle.dump(rdf, f)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"textbook_examples.to_parquet(\"textbook_train_examples.parquet\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result = Pool(8).imap(get_features_for_entry, other_examples.iterrows(), chunksize=1)\n\nrdf = pd.DataFrame(tqdm.tqdm(result, total = other_examples.shape[0]))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"other_train_examples_dataframe.pkl\", 'wb') as f:\n    pickle.dump(rdf, f)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"other_examples.to_parquet(\"other_train_examples.parquet\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing Data","metadata":{}},{"cell_type":"code","source":"with open(\"other_train_examples_dataframe.pkl\", 'rb') as f:\n    other_train_df = pickle.load(f)\n    \nwith open(\"textbook_train_examples_dataframe.pkl\", 'rb') as f:\n    textbook_train_df = pickle.load(f)\n    \nother_train_examples = pd.read_parquet(\"other_train_examples.parquet\")\ntextbook_train_examples = pd.read_parquet(\"textbook_train_examples.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-04-09T21:17:54.757992Z","iopub.execute_input":"2024-04-09T21:17:54.758677Z","iopub.status.idle":"2024-04-09T21:19:32.574986Z","shell.execute_reply.started":"2024-04-09T21:17:54.758632Z","shell.execute_reply":"2024-04-09T21:19:32.572993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(other_train_df.shape)\nprint(textbook_train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-09T21:19:32.579937Z","iopub.execute_input":"2024-04-09T21:19:32.580429Z","iopub.status.idle":"2024-04-09T21:19:32.588833Z","shell.execute_reply.started":"2024-04-09T21:19:32.580388Z","shell.execute_reply":"2024-04-09T21:19:32.586543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(other_train_examples)\ndisplay(textbook_train_examples)","metadata":{"execution":{"iopub.status.busy":"2024-04-09T21:19:32.590889Z","iopub.execute_input":"2024-04-09T21:19:32.591283Z","iopub.status.idle":"2024-04-09T21:19:32.665244Z","shell.execute_reply.started":"2024-04-09T21:19:32.591250Z","shell.execute_reply":"2024-04-09T21:19:32.663224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vote_cols = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\ny_train_textbook = textbook_train_examples['expert_consensus'].values\ny_train_other = other_train_examples['expert_consensus'].values\n\n#normalize votes\ny_train_other_prob = other_train_examples[vote_cols].div(other_train_examples[vote_cols].sum(axis=1), axis=0).values\n\nprint(y_train_other.reshape(len(y_train_other), 1))\nprint(y_train_other_prob)\nprint(np.hstack((y_train_other.reshape(len(y_train_other), 1), y_train_other_prob)))","metadata":{"execution":{"iopub.status.busy":"2024-04-09T21:19:32.670685Z","iopub.execute_input":"2024-04-09T21:19:32.671301Z","iopub.status.idle":"2024-04-09T21:19:32.740414Z","shell.execute_reply.started":"2024-04-09T21:19:32.671247Z","shell.execute_reply":"2024-04-09T21:19:32.739075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#decrease size (conserve memory)\nX_train_textbook_idx, _, y_train_textbook, _ = train_test_split(textbook_train_df.index.values, \n                                                y_train_textbook,\n                                                 train_size=0.1,\n                                                stratify=y_train_textbook)\n\nX_train_other_idx, _, y_train_other_prob, _ = train_test_split(other_train_df.index.values, \n                                                np.hstack((y_train_other.reshape(len(y_train_other),1),\n                                                           y_train_other_prob)),\n                                                 train_size=0.1,\n                                                stratify=y_train_other)\n\ntextbook_train_downsample_df = textbook_train_df.loc[X_train_textbook_idx]\nother_train_downsample_df = other_train_df.loc[X_train_other_idx]\n\ndel textbook_train_df\ndel other_train_df\ndel textbook_train_examples\ndel other_train_examples","metadata":{"execution":{"iopub.status.busy":"2024-04-09T21:19:32.742083Z","iopub.execute_input":"2024-04-09T21:19:32.742580Z","iopub.status.idle":"2024-04-09T21:19:33.147448Z","shell.execute_reply.started":"2024-04-09T21:19:32.742547Z","shell.execute_reply":"2024-04-09T21:19:33.145896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.array(textbook_train_downsample_df['X_eeg'].values.tolist()).shape","metadata":{"execution":{"iopub.status.busy":"2024-04-09T21:19:33.179697Z","iopub.execute_input":"2024-04-09T21:19:33.183046Z","iopub.status.idle":"2024-04-09T21:19:34.173425Z","shell.execute_reply.started":"2024-04-09T21:19:33.182991Z","shell.execute_reply":"2024-04-09T21:19:34.171947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pairs = [['Fp1', 'F7'], ['F7', 'T3'], ['T3', 'T5'], \n         ['T5', 'O1'], ['Fp2', 'F8'], ['F8', 'T4'], \n         ['T4', 'T6'], ['T6', 'O2'], ['Fp1', 'F3'], \n        ['F3', 'C3'], ['C3', 'P3'], ['P3', 'O1'], \n        ['Fp2', 'F4'], ['F4', 'C4'], ['C4', 'P4'], \n         ['P4', 'O2'], ['Fz', 'Cz'], ['Cz', 'Pz']]","metadata":{"execution":{"iopub.status.busy":"2024-04-09T21:20:06.512207Z","iopub.execute_input":"2024-04-09T21:20:06.512730Z","iopub.status.idle":"2024-04-09T21:20:06.520987Z","shell.execute_reply.started":"2024-04-09T21:20:06.512688Z","shell.execute_reply":"2024-04-09T21:20:06.519509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idx = 10\n\nselection = np.array(textbook_train_downsample_df['X_eeg'].values.tolist())[idx]\ngrid_map = {0: [0, 0], 1: [1, 0], 2: [2, 0], 3: [3, 0], 4: [4, 0], 5: [5, 0], \n            6: [0, 1], 7: [1, 1], 8: [2, 1], 9: [3, 1], 10: [4, 1], 11: [5, 1],\n           12: [0, 2], 13: [1, 2], 14: [2, 2], 15: [3, 2], 16: [4, 2], 17: [5, 2]}\n\nfig, axs = plt.subplots(6, 3, figsize=(10, 20))\nplt.title(y_train_textbook[idx])\n\nfor i in range(len(selection)):\n    \n    ax = grid_map[i]\n\n    axs[ax[0], ax[1]].plot(np.linspace(0, selection.shape[1]/100, selection.shape[1]), selection[i])\n    axs[ax[0], ax[1]].set_title(\"{} - {}\".format(pairs[i][0], pairs[i][1]))\n    \n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-09T21:56:01.472886Z","iopub.execute_input":"2024-04-09T21:56:01.473679Z","iopub.status.idle":"2024-04-09T21:56:06.143564Z","shell.execute_reply.started":"2024-04-09T21:56:01.473635Z","shell.execute_reply":"2024-04-09T21:56:06.142212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#spectrogram\n\nheaders = ['LL_0.59', 'LL_0.78', 'LL_0.98', 'LL_1.17', 'LL_1.37', \n'LL_1.56', 'LL_1.76', 'LL_1.95', 'LL_2.15', 'LL_2.34', 'LL_2.54', \n'LL_2.73', 'LL_2.93', 'LL_3.13', 'LL_3.32', 'LL_3.52', 'LL_3.71', \n'LL_3.91', 'LL_4.1', 'LL_4.3', 'LL_4.49', 'LL_4.69', 'LL_4.88', \n'LL_5.08', 'LL_5.27', 'LL_5.47', 'LL_5.66', 'LL_5.86', 'LL_6.05', \n'LL_6.25', 'LL_6.45', 'LL_6.64', 'LL_6.84', 'LL_7.03', 'LL_7.23', \n'LL_7.42', 'LL_7.62', 'LL_7.81', 'LL_8.01', 'LL_8.2', 'LL_8.4', 'LL_8.59', \n'LL_8.79', 'LL_8.98', 'LL_9.18', 'LL_9.38', 'LL_9.57', 'LL_9.77', 'LL_9.96', \n'LL_10.16', 'LL_10.35', 'LL_10.55', 'LL_10.74', 'LL_10.94', 'LL_11.13', 'LL_11.33',\n'LL_11.52', 'LL_11.72', 'LL_11.91', 'LL_12.11', 'LL_12.3', 'LL_12.5', 'LL_12.7',\n'LL_12.89', 'LL_13.09', 'LL_13.28', 'LL_13.48', 'LL_13.67', 'LL_13.87', 'LL_14.06',\n'LL_14.26', 'LL_14.45', 'LL_14.65', 'LL_14.84', 'LL_15.04', 'LL_15.23', 'LL_15.43', \n'LL_15.63', 'LL_15.82', 'LL_16.02', 'LL_16.21', 'LL_16.41', 'LL_16.6', 'LL_16.8', 'LL_16.99',\n'LL_17.19', 'LL_17.38', 'LL_17.58', 'LL_17.77', 'LL_17.97', 'LL_18.16', 'LL_18.36', \n'LL_18.55', 'LL_18.75', 'LL_18.95', 'LL_19.14', 'LL_19.34', 'LL_19.53', 'LL_19.73', \n'LL_19.92', 'RL_0.59', 'RL_0.78', 'RL_0.98', 'RL_1.17', 'RL_1.37', 'RL_1.56', 'RL_1.76', \n'RL_1.95', 'RL_2.15', 'RL_2.34', 'RL_2.54', 'RL_2.73', 'RL_2.93', 'RL_3.13', 'RL_3.32', \n'RL_3.52', 'RL_3.71', 'RL_3.91', 'RL_4.1', 'RL_4.3', 'RL_4.49', 'RL_4.69', 'RL_4.88',\n'RL_5.08', 'RL_5.27', 'RL_5.47', 'RL_5.66', 'RL_5.86', 'RL_6.05', 'RL_6.25', 'RL_6.45', \n'RL_6.64', 'RL_6.84', 'RL_7.03', 'RL_7.23', 'RL_7.42', 'RL_7.62', 'RL_7.81', 'RL_8.01',\n'RL_8.2', 'RL_8.4', 'RL_8.59', 'RL_8.79', 'RL_8.98', 'RL_9.18', 'RL_9.38', 'RL_9.57',\n'RL_9.77', 'RL_9.96', 'RL_10.16', 'RL_10.35', 'RL_10.55', 'RL_10.74', 'RL_10.94', 'RL_11.13', \n'RL_11.33', 'RL_11.52', 'RL_11.72', 'RL_11.91', 'RL_12.11', 'RL_12.3', 'RL_12.5', 'RL_12.7', \n'RL_12.89', 'RL_13.09', 'RL_13.28', 'RL_13.48', 'RL_13.67', 'RL_13.87', 'RL_14.06', 'RL_14.26', \n'RL_14.45', 'RL_14.65', 'RL_14.84', 'RL_15.04', 'RL_15.23', 'RL_15.43', 'RL_15.63', 'RL_15.82', \n'RL_16.02', 'RL_16.21', 'RL_16.41', 'RL_16.6', 'RL_16.8', 'RL_16.99', 'RL_17.19', 'RL_17.38', \n'RL_17.58', 'RL_17.77', 'RL_17.97', 'RL_18.16', 'RL_18.36', 'RL_18.55', 'RL_18.75', 'RL_18.95',\n'RL_19.14', 'RL_19.34', 'RL_19.53', 'RL_19.73', 'RL_19.92', 'LP_0.59', 'LP_0.78', 'LP_0.98',\n'LP_1.17', 'LP_1.37', 'LP_1.56', 'LP_1.76', 'LP_1.95', 'LP_2.15', 'LP_2.34', 'LP_2.54',\n'LP_2.73', 'LP_2.93', 'LP_3.13', 'LP_3.32', 'LP_3.52', 'LP_3.71', 'LP_3.91', 'LP_4.1', \n'LP_4.3', 'LP_4.49', 'LP_4.69', 'LP_4.88', 'LP_5.08', 'LP_5.27', 'LP_5.47', 'LP_5.66', \n'LP_5.86', 'LP_6.05', 'LP_6.25', 'LP_6.45', 'LP_6.64', 'LP_6.84', 'LP_7.03', 'LP_7.23',\n'LP_7.42', 'LP_7.62', 'LP_7.81', 'LP_8.01', 'LP_8.2', 'LP_8.4', 'LP_8.59', 'LP_8.79', \n'LP_8.98', 'LP_9.18', 'LP_9.38', 'LP_9.57', 'LP_9.77', 'LP_9.96', 'LP_10.16', 'LP_10.35',\n'LP_10.55', 'LP_10.74', 'LP_10.94', 'LP_11.13', 'LP_11.33', 'LP_11.52', 'LP_11.72', 'LP_11.91',\n'LP_12.11', 'LP_12.3', 'LP_12.5', 'LP_12.7', 'LP_12.89', 'LP_13.09', 'LP_13.28', 'LP_13.48',\n'LP_13.67', 'LP_13.87', 'LP_14.06', 'LP_14.26', 'LP_14.45', 'LP_14.65', 'LP_14.84', 'LP_15.04',\n'LP_15.23', 'LP_15.43', 'LP_15.63', 'LP_15.82', 'LP_16.02', 'LP_16.21', 'LP_16.41', 'LP_16.6',\n'LP_16.8', 'LP_16.99', 'LP_17.19', 'LP_17.38', 'LP_17.58', 'LP_17.77', 'LP_17.97', 'LP_18.16', \n'LP_18.36', 'LP_18.55', 'LP_18.75', 'LP_18.95', 'LP_19.14', 'LP_19.34', 'LP_19.53', 'LP_19.73', \n'LP_19.92', 'RP_0.59', 'RP_0.78', 'RP_0.98', 'RP_1.17', 'RP_1.37', 'RP_1.56', 'RP_1.76', \n'RP_1.95', 'RP_2.15', 'RP_2.34', 'RP_2.54', 'RP_2.73', 'RP_2.93', 'RP_3.13', 'RP_3.32', \n'RP_3.52', 'RP_3.71', 'RP_3.91', 'RP_4.1', 'RP_4.3', 'RP_4.49', 'RP_4.69', 'RP_4.88',\n'RP_5.08', 'RP_5.27', 'RP_5.47', 'RP_5.66', 'RP_5.86', 'RP_6.05', 'RP_6.25', 'RP_6.45', \n'RP_6.64', 'RP_6.84', 'RP_7.03', 'RP_7.23', 'RP_7.42', 'RP_7.62', 'RP_7.81', 'RP_8.01', \n'RP_8.2', 'RP_8.4', 'RP_8.59', 'RP_8.79', 'RP_8.98', 'RP_9.18', 'RP_9.38', 'RP_9.57', 'RP_9.77',\n'RP_9.96', 'RP_10.16', 'RP_10.35', 'RP_10.55', 'RP_10.74', 'RP_10.94', 'RP_11.13', 'RP_11.33',\n'RP_11.52', 'RP_11.72', 'RP_11.91', 'RP_12.11', 'RP_12.3', 'RP_12.5', 'RP_12.7', 'RP_12.89',\n'RP_13.09', 'RP_13.28', 'RP_13.48', 'RP_13.67', 'RP_13.87', 'RP_14.06', 'RP_14.26', 'RP_14.45',\n'RP_14.65', 'RP_14.84', 'RP_15.04', 'RP_15.23', 'RP_15.43', 'RP_15.63', 'RP_15.82', 'RP_16.02',\n'RP_16.21', 'RP_16.41', 'RP_16.6', 'RP_16.8', 'RP_16.99', 'RP_17.19', 'RP_17.38', 'RP_17.58',\n'RP_17.77', 'RP_17.97', 'RP_18.16', 'RP_18.36', 'RP_18.55', 'RP_18.75', 'RP_18.95', 'RP_19.14',\n'RP_19.34', 'RP_19.53', 'RP_19.73', 'RP_19.92']\n\nprint(y_train_textbook[idx])\n\n\nfor prefix in ['LL', 'RL', 'LP', 'RP']:\n\n    selection = np.array(textbook_train_downsample_df['spec_values'].values.tolist())[idx]\n    selection = selection[:, np.where(np.char.startswith(np.array(headers), prefix))[0]]\n\n        \n    #plotting the log lets us see trends at different ooms\n    fig = px.imshow(np.log10(selection.T), 0, \n                y = [float(x.replace(\"{}_\".format(prefix), '')) for x in np.array(headers)[np.where(np.char.startswith(np.array(headers), prefix))]],\n                aspect='auto', origin=\"lower\", title=\"{} Spectrogram (50s)\".format(prefix))\n\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-09T21:56:13.595484Z","iopub.execute_input":"2024-04-09T21:56:13.595993Z","iopub.status.idle":"2024-04-09T21:56:14.398950Z","shell.execute_reply.started":"2024-04-09T21:56:13.595955Z","shell.execute_reply":"2024-04-09T21:56:14.397481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}