{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pprint import pprint\nfrom scipy.signal import spectrogram\nimport matplotlib as mpl\nfrom matplotlib import cm\nimport matplotlib.patches as patches\nimport matplotlib.pyplot as plt\nos.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-31T18:07:28.951559Z","iopub.execute_input":"2024-01-31T18:07:28.952336Z","iopub.status.idle":"2024-01-31T18:07:32.847206Z","shell.execute_reply.started":"2024-01-31T18:07:28.952303Z","shell.execute_reply":"2024-01-31T18:07:32.846204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df =pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ntest=pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\neeg = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1484166292.parquet')\nspectrogram = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/1000086677.parquet')","metadata":{"execution":{"iopub.status.busy":"2024-01-31T18:07:32.848935Z","iopub.execute_input":"2024-01-31T18:07:32.849358Z","iopub.status.idle":"2024-01-31T18:07:33.671099Z","shell.execute_reply.started":"2024-01-31T18:07:32.849331Z","shell.execute_reply":"2024-01-31T18:07:33.670092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-01-31T18:07:33.672458Z","iopub.execute_input":"2024-01-31T18:07:33.672761Z","iopub.status.idle":"2024-01-31T18:07:33.701863Z","shell.execute_reply.started":"2024-01-31T18:07:33.672735Z","shell.execute_reply":"2024-01-31T18:07:33.700838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"expert_consensus = df.expert_consensus.value_counts().reset_index()\nexpert_consensus.columns = [\"consensus\", \"frequency\"]\n\nplt.figure(figsize=(20, 10))\n\nfigure = sns.barplot(data=expert_consensus,x=\"consensus\", y=\"frequency\",errwidth=0)\nplt.title('Expert Consensus - Distribution', weight=\"bold\", size=20)\nplt.xlabel(\"Consensus\", size = 18, weight=\"bold\")\nplt.ylabel(\"Count\", size = 18, weight=\"bold\")\n\nfor i in figure.containers:\n    figure.bar_label(i,)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-31T18:07:33.704968Z","iopub.execute_input":"2024-01-31T18:07:33.705493Z","iopub.status.idle":"2024-01-31T18:07:34.329637Z","shell.execute_reply.started":"2024-01-31T18:07:33.705453Z","shell.execute_reply":"2024-01-31T18:07:34.328250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\n# **Create New Features**\nWe will create a new feature named pattern , where it based on the experts votes.\n* **idealized** = High levels of agreement\n* **proto** = 50% of expert give a lebel as 'other_vote' and 50% give one of the remaining five labels\n* **edge** = Cases where experts are approximately split between 2 of the 5 named patterns","metadata":{}},{"cell_type":"code","source":"target_col = list(df.columns[-6:])\ntrain_df = df.groupby(by=[\"eeg_id\", \"spectrogram_id\", \"patient_id\"])\\\n                    [target_col].sum().reset_index()\n\ndef get_pattern(df):\n    col_name = list(df.columns[-6:])\n    pattern =[]\n    max_treshold = 0.75\n    equal_treshold = 0.4\n    for i in range(len(df)):\n        max_vote = df.iloc[i][target_col].max()\n        total_vote = df.iloc[i][target_col].sum()\n        perc = max_vote/total_vote \n   \n        if perc >= max_treshold:\n            pattern.append('idealized')\n        elif df.iloc[i]['other_vote']/total_vote >=equal_treshold and perc>= equal_treshold:\n            pattern.append('proto')\n        elif df.iloc[i]['other_vote']==0  and perc>= equal_treshold:\n            pattern.append('edge')\n        else:\n            pattern.append('unidentified')\n    return pattern\ntrain_df['pattern'] = get_pattern(train_df)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-31T18:07:34.331291Z","iopub.execute_input":"2024-01-31T18:07:34.331681Z","iopub.status.idle":"2024-01-31T18:07:52.189675Z","shell.execute_reply.started":"2024-01-31T18:07:34.331651Z","shell.execute_reply":"2024-01-31T18:07:52.188540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pattern = train_df.pattern.value_counts().reset_index()\npattern.columns = [\"pattern\", \"frequency\"]\n\nplt.figure(figsize=(20, 10))\n\nfigure = sns.barplot(data=pattern,x=\"pattern\", y=\"frequency\",errwidth=0)\nplt.title('Pattern - Distribution', weight=\"bold\", size=20)\nplt.xlabel(\"pattern\", size = 18, weight=\"bold\")\nplt.ylabel(\"Count\", size = 18, weight=\"bold\")\n\nfor i in figure.containers:\n    figure.bar_label(i,)","metadata":{"execution":{"iopub.status.busy":"2024-01-31T18:07:52.191683Z","iopub.execute_input":"2024-01-31T18:07:52.192017Z","iopub.status.idle":"2024-01-31T18:07:52.600940Z","shell.execute_reply.started":"2024-01-31T18:07:52.191989Z","shell.execute_reply":"2024-01-31T18:07:52.599991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-01-31T18:07:52.602120Z","iopub.execute_input":"2024-01-31T18:07:52.602461Z","iopub.status.idle":"2024-01-31T18:07:52.625938Z","shell.execute_reply.started":"2024-01-31T18:07:52.602435Z","shell.execute_reply":"2024-01-31T18:07:52.624994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Relationship between the patient Id ,eeg_id and spectrogram_id\n\nunique patient_id = 1950\n\nfor each patient_id , we have more then 1 eeg_id and spectrogram_id is present.\n\nlet me take patient_id == 30631 ,","metadata":{}},{"cell_type":"code","source":"pat_id=df[df['patient_id']==30631]\nprint('the length :',len(pat_id))\n\nspec_count=pat_id['spectrogram_id'].nunique()\neeg_count=pat_id['eeg_id'].nunique()\nprint('eeg_count:',eeg_count)\nprint('spec_count:',spec_count)\n\neeg_id =pat_id[pat_id['eeg_id']==1270973624]\nunique_spec_id_wrt_eegid=eeg_id['spectrogram_id'].nunique()\nprint('total counts of spectogram_id for a unique eeg_id:',unique_spec_id_wrt_eegid)\n\nspec_id =pat_id[pat_id['spectrogram_id']==764146759]\nunique_eeg_id_wrt_specid=spec_id['eeg_id'].nunique()\nprint('total counts of eeg_id for a unique spectogram_id:',unique_eeg_id_wrt_specid)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-31T18:07:52.627150Z","iopub.execute_input":"2024-01-31T18:07:52.627930Z","iopub.status.idle":"2024-01-31T18:07:52.639807Z","shell.execute_reply.started":"2024-01-31T18:07:52.627892Z","shell.execute_reply":"2024-01-31T18:07:52.638854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The above code shows the following conclusion :\n* One spectrogram (like the case below - 764146759) can be a part of multiple EEG recordings, all being from the same patient_id.\n* One EEG Id  (like the case below - 1270973624) have single spectrogram id , all being from the same patient_id.","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Spectogram","metadata":{}},{"cell_type":"code","source":"spec_id = '1000086677'\nspec_path = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nspectrogram = pd.read_parquet(spec_path + spec_id+'.parquet')\nspectrogram","metadata":{"execution":{"iopub.status.busy":"2024-01-31T18:07:52.641222Z","iopub.execute_input":"2024-01-31T18:07:52.641574Z","iopub.status.idle":"2024-01-31T18:07:52.710501Z","shell.execute_reply.started":"2024-01-31T18:07:52.641545Z","shell.execute_reply":"2024-01-31T18:07:52.709395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Visualize the Spectrogram on Pattern = 'idealized'**","metadata":{}},{"cell_type":"code","source":"\n\ndef spec_dict(idealized_df):\n    N = 1\n    spec_dict = {\n        \"seizure_vote\": 0,\n        \"lpd_vote\": 0,\n        \"gpd_vote\": 0,\n        \"lrda_vote\":0, \n        \"grda_vote\":0,\n        \"other_vote\":0\n    }\n\n    for key in spec_dict.keys():\n        col_idx = idealized_df[key].sort_values(ascending=False).head(N).index\n        spec_dict[key] = idealized_df.loc[col_idx, \"spectrogram_id\"].values\n        \n    return spec_dict\n\n    \n\ndef idealized_visualization(vote,spec_id,pattern):\n    spec_path = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\n    spec_data = pd.read_parquet(spec_path + spec_id+'.parquet')\n\n    fig, axes = plt.subplots(1, 1, figsize=(6, 3), sharey=True)\n\n    axes.imshow(np.log(spec_data.T))\n    axes.set_title(f'{pattern} pattern {vote} id {spec_id}', size=10)\n    axes.set_xlabel('Time', size=10)\n    axes.set_ylabel('(Hz)', size=10)\n    axes.tick_params(axis='both', which='both', labelsize=10)\n\n    plt.show()\n    \n\nidealized_df = train_df[train_df['pattern']=='idealized']\nspec_dict=spec_dict(idealized_df)\nfor vote,spec_id in spec_dict.items():\n    \n    idealized_visualization(vote,str(spec_id[0]),'idealized')\n    \n\n","metadata":{"execution":{"iopub.status.busy":"2024-01-31T18:07:52.714481Z","iopub.execute_input":"2024-01-31T18:07:52.714801Z","iopub.status.idle":"2024-01-31T18:07:55.096535Z","shell.execute_reply.started":"2024-01-31T18:07:52.714773Z","shell.execute_reply":"2024-01-31T18:07:55.095468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Spectrogram Visualization on pattern = Edge","metadata":{}},{"cell_type":"code","source":"\nedge_df = train_df[train_df['pattern']=='edge']\nN = 1\nspec_dict = {\n    \"seizure_vote\": 0,\n    \"lpd_vote\": 0,\n    \"gpd_vote\": 0,\n    \"lrda_vote\":0, \n    \"grda_vote\":0,\n    \"other_vote\":0\n}\n\nfor key in spec_dict.keys():\n    col_idx = edge_df[key].sort_values(ascending=False).head(N).index\n    spec_dict[key] = edge_df.loc[col_idx, \"spectrogram_id\"].values\n# spec_dict=spec_dict(edge_df)\nfor vote,spec_id in spec_dict.items():\n    \n    idealized_visualization(vote,str(spec_id[0]),'edge')","metadata":{"execution":{"iopub.status.busy":"2024-01-31T18:07:55.098433Z","iopub.execute_input":"2024-01-31T18:07:55.098806Z","iopub.status.idle":"2024-01-31T18:07:57.480877Z","shell.execute_reply.started":"2024-01-31T18:07:55.098770Z","shell.execute_reply":"2024-01-31T18:07:57.479839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Spectrogram visualization pattern ='proto'","metadata":{}},{"cell_type":"code","source":"\nproto_df = train_df[train_df['pattern']=='proto']\nN =1\nspec_dict = {\n    \"seizure_vote\": 0,\n    \"lpd_vote\": 0,\n    \"gpd_vote\": 0,\n    \"lrda_vote\":0, \n    \"grda_vote\":0,\n    \"other_vote\":0\n}\n\nfor key in spec_dict.keys():\n    col_idx = proto_df[key].sort_values(ascending=False).head(N).index\n    spec_dict[key] = proto_df.loc[col_idx, \"spectrogram_id\"].values\n# spec_dict=spec_dict(edge_df)\nfor vote,spec_id in spec_dict.items():\n    \n    idealized_visualization(vote,str(spec_id[0]),'proto')","metadata":{"execution":{"iopub.status.busy":"2024-01-31T18:07:57.482179Z","iopub.execute_input":"2024-01-31T18:07:57.482509Z","iopub.status.idle":"2024-01-31T18:08:00.065151Z","shell.execute_reply.started":"2024-01-31T18:07:57.482482Z","shell.execute_reply":"2024-01-31T18:08:00.064068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Spectrogram Visualization Pattern = 'unidentified'","metadata":{}},{"cell_type":"code","source":"\nunidentified_df = train_df[train_df['pattern']=='unidentified']\nN =1\nspec_dict = {\n    \"seizure_vote\": 0,\n    \"lpd_vote\": 0,\n    \"gpd_vote\": 0,\n    \"lrda_vote\":0, \n    \"grda_vote\":0,\n    \"other_vote\":0\n}\n\nfor key in spec_dict.keys():\n    col_idx = unidentified_df[key].sort_values(ascending=False).head(N).index\n    spec_dict[key] = unidentified_df.loc[col_idx, \"spectrogram_id\"].values\n# spec_dict=spec_dict(edge_df)\nfor vote,spec_id in spec_dict.items():\n    \n    idealized_visualization(vote,str(spec_id[0]),'unidentified')","metadata":{"execution":{"iopub.status.busy":"2024-01-31T18:08:00.066327Z","iopub.execute_input":"2024-01-31T18:08:00.066627Z","iopub.status.idle":"2024-01-31T18:08:02.673997Z","shell.execute_reply.started":"2024-01-31T18:08:00.066600Z","shell.execute_reply":"2024-01-31T18:08:02.672992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import spectrogram info\nspect_data = np.load(\"/kaggle/input/brain-spectrograms/specs.npy\", allow_pickle=True).item()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-31T18:08:02.675360Z","iopub.execute_input":"2024-01-31T18:08:02.675690Z","iopub.status.idle":"2024-01-31T18:09:11.460509Z","shell.execute_reply.started":"2024-01-31T18:08:02.675662Z","shell.execute_reply":"2024-01-31T18:09:11.459426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spectrogram = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/1000086677.parquet')\nspect_feature_name =spectrogram.columns[1:]","metadata":{"execution":{"iopub.status.busy":"2024-01-31T18:09:11.461919Z","iopub.execute_input":"2024-01-31T18:09:11.462278Z","iopub.status.idle":"2024-01-31T18:09:11.501909Z","shell.execute_reply.started":"2024-01-31T18:09:11.462241Z","shell.execute_reply":"2024-01-31T18:09:11.500913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fe_data = {}\n\nfor spect_id, data in spect_data.items():\n    fe_data[spect_id] = {}\n    \n    for k, feature in enumerate(spect_feature_name):\n        fe_data[spect_id][f\"{feature}_mean\"] = data[:, k].mean()\n        fe_data[spect_id][f\"{feature}_min\"] = data[:, k].min()\n        fe_data[spect_id][f\"{feature}_max\"] = data[:, k].max()\n        fe_data[spect_id][f\"{feature}_std\"] = data[:, k].std()\n        \n# convert to df\nfe_data_df = pd.DataFrame.from_dict(fe_data, orient='index').reset_index()\n\n# append target labels\n# target_df = train_group\\\n#             .groupby(\"spectrogram_id\")[train_group.filter(regex='_vote$').columns]\\\n#             .sum().reset_index()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-31T18:09:11.503079Z","iopub.execute_input":"2024-01-31T18:09:11.503381Z","iopub.status.idle":"2024-01-31T18:15:28.098068Z","shell.execute_reply.started":"2024-01-31T18:09:11.503356Z","shell.execute_reply":"2024-01-31T18:15:28.097200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_df = df\\\n            .groupby(\"spectrogram_id\")[\"expert_consensus\"]\\\n            .first().reset_index()\n# encoding from string to numbers\ntarget_df['expert_consensus'] = pd.factorize(target_df['expert_consensus'])[0]\n\nfinal_df = pd.merge(left=fe_data_df, right=target_df, \n                    left_on=\"index\", right_on=\"spectrogram_id\")\n\nfinal_df","metadata":{"execution":{"iopub.status.busy":"2024-01-31T18:15:28.099549Z","iopub.execute_input":"2024-01-31T18:15:28.099851Z","iopub.status.idle":"2024-01-31T18:15:28.180299Z","shell.execute_reply.started":"2024-01-31T18:15:28.099825Z","shell.execute_reply":"2024-01-31T18:15:28.179316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n\ndtrain, dvalid = train_test_split(final_df, train_size=0.8, random_state=42)\n\nFEATURE_COLS = final_df.columns[1:-2]\nTARGET_COL = final_df.columns[-1]\n\n","metadata":{"execution":{"iopub.status.busy":"2024-01-31T18:15:28.181609Z","iopub.execute_input":"2024-01-31T18:15:28.181906Z","iopub.status.idle":"2024-01-31T18:15:28.495702Z","shell.execute_reply.started":"2024-01-31T18:15:28.181879Z","shell.execute_reply":"2024-01-31T18:15:28.494567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# This notebook still in progress","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}