{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30636,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Importing Libraries","metadata":{}},{"cell_type":"code","source":"%%capture\n\nimport numpy as np \nimport pandas as pd \nimport os\nimport shutil\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport gc\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import accuracy_score\nfrom scipy.special import kl_div\n\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom multiprocessing import Pool\nfrom sklearn.impute import SimpleImputer\n\n\nfrom glob import glob\n\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-11T16:04:52.235471Z","iopub.execute_input":"2024-04-11T16:04:52.236217Z","iopub.status.idle":"2024-04-11T16:05:07.391641Z","shell.execute_reply.started":"2024-04-11T16:04:52.236174Z","shell.execute_reply":"2024-04-11T16:05:07.390721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow tensorflow-gpu ","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:05:07.393634Z","iopub.execute_input":"2024-04-11T16:05:07.394397Z","iopub.status.idle":"2024-04-11T16:05:09.832504Z","shell.execute_reply.started":"2024-04-11T16:05:07.394353Z","shell.execute_reply":"2024-04-11T16:05:09.831354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading Training Data","metadata":{}},{"cell_type":"code","source":"train0= pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nprint(train0.shape)\ntrain0.info()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:05:09.834239Z","iopub.execute_input":"2024-04-11T16:05:09.834641Z","iopub.status.idle":"2024-04-11T16:05:10.049064Z","shell.execute_reply.started":"2024-04-11T16:05:09.834606Z","shell.execute_reply":"2024-04-11T16:05:10.048125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train0.describe()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:05:10.051893Z","iopub.execute_input":"2024-04-11T16:05:10.052612Z","iopub.status.idle":"2024-04-11T16:05:10.135836Z","shell.execute_reply.started":"2024-04-11T16:05:10.052571Z","shell.execute_reply":"2024-04-11T16:05:10.134796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train0.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:05:10.137006Z","iopub.execute_input":"2024-04-11T16:05:10.137316Z","iopub.status.idle":"2024-04-11T16:05:10.154170Z","shell.execute_reply.started":"2024-04-11T16:05:10.137289Z","shell.execute_reply":"2024-04-11T16:05:10.153112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train0.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:05:10.155623Z","iopub.execute_input":"2024-04-11T16:05:10.155930Z","iopub.status.idle":"2024-04-11T16:05:10.163750Z","shell.execute_reply.started":"2024-04-11T16:05:10.155904Z","shell.execute_reply":"2024-04-11T16:05:10.162828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfeatures = train0.drop(['eeg_id', 'spectrogram_id', 'eeg_sub_id', 'label_id', 'expert_consensus'], axis=1)\ntarget = list(train0.columns[-6:])\nprint(target)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:05:10.164982Z","iopub.execute_input":"2024-04-11T16:05:10.165471Z","iopub.status.idle":"2024-04-11T16:05:10.177972Z","shell.execute_reply.started":"2024-04-11T16:05:10.165440Z","shell.execute_reply":"2024-04-11T16:05:10.176866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGETS = target\nTARGETS","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:05:10.179401Z","iopub.execute_input":"2024-04-11T16:05:10.179723Z","iopub.status.idle":"2024-04-11T16:05:10.186222Z","shell.execute_reply.started":"2024-04-11T16:05:10.179693Z","shell.execute_reply":"2024-04-11T16:05:10.185287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train0.head(50)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:04:38.791076Z","iopub.status.idle":"2024-04-11T16:04:38.792138Z","shell.execute_reply.started":"2024-04-11T16:04:38.791822Z","shell.execute_reply":"2024-04-11T16:04:38.791851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.unique(train0['expert_consensus'])","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:04:38.794155Z","iopub.status.idle":"2024-04-11T16:04:38.794660Z","shell.execute_reply.started":"2024-04-11T16:04:38.794399Z","shell.execute_reply":"2024-04-11T16:04:38.794425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(style='darkgrid')\n\nplt.figure(figsize=(8, 6))\nax = sns.barplot(x=target, y=train0[target].sum(), color='maroon', alpha = 0.7)\n\n# data labels above the bars\nfor p in ax.patches:\n    ax.annotate(f'{p.get_height():.0f}', (p.get_x() + p.get_width() / 2., p.get_height()),\n                ha='center', va='center', fontsize=10, color='black', xytext=(0, 5),\n                textcoords='offset points')\n\nplt.title('Distribution of Votes for Each Class')\nplt.xlabel('Brain Activity Class')\nplt.ylabel('Vote Count')\nplt.xticks(rotation=45, fontsize=10)\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:05:15.226146Z","iopub.execute_input":"2024-04-11T16:05:15.226518Z","iopub.status.idle":"2024-04-11T16:05:15.801008Z","shell.execute_reply.started":"2024-04-11T16:05:15.226490Z","shell.execute_reply":"2024-04-11T16:05:15.800014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reading a single Parquet file\n\n# DataFrame with paths to all Parquet files\ndf = pd.DataFrame({'path': glob('/kaggle/input/hms-harmful-brain-activity-classification/**/*.parquet')})\n\ndf['test_type'] = df['path'].str.split('/').str[-2].str.split('_').str[-1]\ndf['id'] = df['path'].str.split('/').str[-1].str.split('.').str[0]\n\ndf_eeg = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1000913311.parquet')\ndf_eeg.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:05:20.026360Z","iopub.execute_input":"2024-04-11T16:05:20.026752Z","iopub.status.idle":"2024-04-11T16:05:20.526220Z","shell.execute_reply.started":"2024-04-11T16:05:20.026717Z","shell.execute_reply":"2024-04-11T16:05:20.525225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:05:22.682418Z","iopub.execute_input":"2024-04-11T16:05:22.682801Z","iopub.status.idle":"2024-04-11T16:05:22.693641Z","shell.execute_reply.started":"2024-04-11T16:05:22.682771Z","shell.execute_reply":"2024-04-11T16:05:22.692558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_eeg.shape   # -> [#time points, #channels]","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:04:38.803110Z","iopub.status.idle":"2024-04-11T16:04:38.803993Z","shell.execute_reply.started":"2024-04-11T16:04:38.803683Z","shell.execute_reply":"2024-04-11T16:04:38.803710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n\n# # spectrogram files\n# files = os.listdir('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/')\n# print(len(files), 'files')\n\n# spectrograms = {}\n# for i, f in enumerate(files):\n#     print(\"Processing file\", i+1, end = '\\r')\n#     spectrograms[int(f.split('.')[0])] = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/' + f).iloc[:, 1:].values\n\n# print(\"Processing complete. Total spectrograms processed:\", len(spectrograms))","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:04:38.805593Z","iopub.status.idle":"2024-04-11T16:04:38.806177Z","shell.execute_reply.started":"2024-04-11T16:04:38.805890Z","shell.execute_reply":"2024-04-11T16:04:38.805914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# spectrogram files\n\ndef read_spectrogram(file):\n    file_path = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/' + file\n    spectrogram = pd.read_parquet(file_path).iloc[:, 1:].values\n    return int(file.split('.')[0]), spectrogram\n\nfiles = os.listdir('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/')\nprint(len(files), 'files')\n\n# Use multiprocessing to read spectrogram files concurrently\nwith Pool() as pool:\n    spectrograms = dict(pool.map(read_spectrogram, files))\n\nprint(\"Processing complete. Total spectrograms processed:\", len(spectrograms))\n","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:05:26.122750Z","iopub.execute_input":"2024-04-11T16:05:26.123593Z","iopub.status.idle":"2024-04-11T16:08:42.494977Z","shell.execute_reply.started":"2024-04-11T16:05:26.123555Z","shell.execute_reply":"2024-04-11T16:08:42.493674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# One parquet file in train_spectrogram\n\ndf_eeg = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/1000086677.parquet')\n\ndf_eeg.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:08:42.497070Z","iopub.execute_input":"2024-04-11T16:08:42.497406Z","iopub.status.idle":"2024-04-11T16:08:42.558615Z","shell.execute_reply.started":"2024-04-11T16:08:42.497373Z","shell.execute_reply":"2024-04-11T16:08:42.557583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:04:38.812239Z","iopub.status.idle":"2024-04-11T16:04:38.812755Z","shell.execute_reply.started":"2024-04-11T16:04:38.812482Z","shell.execute_reply":"2024-04-11T16:04:38.812505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:04:38.814125Z","iopub.status.idle":"2024-04-11T16:04:38.814620Z","shell.execute_reply.started":"2024-04-11T16:04:38.814362Z","shell.execute_reply":"2024-04-11T16:04:38.814386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating Non-Overlapping EEG ID Train Data","metadata":{}},{"cell_type":"markdown","source":"## Creating Unique EEG Segments\n\nThe EEG data is grouped by `eeg_id`, and the first `spectrogram_id` and the earliest `spectrogram_label_offset_seconds` are selected for each `eeg_id`. This establishes the starting point of each EEG segment.\n\n### train (updated)\n\n| eeg_id | spectrogram_id | start_time |\n|--------|----------------|------------|\n| 1      | 101            | 10         |\n| 2      | 103            | 5          |\n\n## Determining the Latest Point in Each EEG Segment\n\nHere, we group the data by `eeg_id` again and find the latest `spectrogram_label_offset_seconds` for each segment. This maximum value is appended to the `train` DataFrame, representing the end point of each EEG segment.\n\n### train (updated)\n\n| eeg_id | spectrogram_id | start_time | end_time |\n|--------|----------------|------------|----------|\n| 1      | 101            | 10         | 15       |\n| 2      | 103            | 5          | 20       |\n\n## Incorporating Patient Information\n\nThe `patient_id` corresponding to each `eeg_id` is added to the `train` DataFrame, establishing a link between each EEG segment and a specific patient.\n\n### train (updated)\n\n| eeg_id | spectrogram_id | start_time | end_time | patient_id |\n|--------|----------------|------------|----------|------------|\n| 1      | 101            | 10         | 15       | 1          |\n| 2      | 103            | 5          | 20       | 2          |\n\n## Summing Target Variable Counts\n\nWe then sum up the counts of target variables (e.g., seizure votes, LPD votes) for each `eeg_id`.\n\n### train (updated)\n\n| eeg_id | spectrogram_id | start_time | end_time | patient_id | seizure_vote | lpd_vote | gpd_vote | lrda_vote | grda_vote | other_vote |\n|--------|----------------|------------|----------|------------|--------------|----------|----------|-----------|-----------|------------|\n| 1      | 101            | 10         | 15       | 1          | 6            | 0        | 0        | 0         | 0         | 0          |\n| 2      | 103            | 5          | 20       | 2          | 0            | 0        | 2        | 0         | 0         | 0          |\n\n## Normalizing Target Variable Counts\n\nThe counts are normalized to ensure they sum up to 1, converting them into probabilities.\n\n### train (updated)\n\n| eeg_id | spectrogram_id | start_time | end_time | patient_id | seizure_vote | lpd_vote | gpd_vote | lrda_vote | grda_vote | other_vote |\n|--------|----------------|------------|----------|------------|--------------|----------|----------|-----------|-----------|------------|\n| 1      | 101            | 10         | 15       | 1          | 1            | 0        | 0        | 0         | 0         | 0          |\n| 2      | 103            | 5          | 20       | 2          | 0            | 0        | 1        | 0         | 0         | 0          |\n\n## Including Expert Consensus\n\nFor each `eeg_id`, the `expert_consensus` on the EEG segment's classification is included in the dataframe.\n\n### train (updated)\n\n| eeg_id | spectrogram_id | start_time | end_time | patient_id | seizure_vote | lpd_vote | gpd_vote | lrda_vote | grda_vote | other_vote | target |\n|--------|----------------|------------|----------|------------|--------------|----------|----------|-----------|-----------|------------|--------|\n| 1      | 101            | 10         | 15       | 1          | 1            | 0        | 0        | 0         | 0         | 0          | Seizure|\n| 2      | 103            | 5          | 20       | 2          | 0            | 0        | 1        | 0         | 0         | 0          | GPD    |\n","metadata":{}},{"cell_type":"code","source":"# Creating Unique EEG Segments:\n# The EEG data is grouped by eeg_id, and the first spectrogram_id and the earliest spectrogram_label_offset_seconds is selected for each eeg_id. This establishes the starting point of each EEG segment.\n\n\ntrain = train0.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','start_time']\n\n\n# Determining the Latest Point in Each EEG Segment:\n# Here, we group the data by eeg_id again and find the latest spectrogram_label_offset_seconds for each segment. This maximum value is appended to the train DataFrame, representing the end point of each EEG segment.\n\ntmp = train0.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['end_time'] = tmp\n\n# The patient_id corresponding to each eeg_id is added to the train DataFrame, establishing a link between each EEG segment and a specific patient.\ntmp = train0.groupby('eeg_id')[['patient_id']].agg('first') \ntrain['patient_id'] = tmp\n\n# We then sum up the counts of target variables (e.g., seizure votes, LPD votes) for each eeg_id.\ntmp = train0.groupby('eeg_id')[target].agg('sum')\nfor t in target:\n    train[t] = tmp[t].values\n    \n\n# The counts are normalized to ensure they sum up to 1, converting them into probabilities.\ny_data = train[target].values\ny_data = y_data / y_data.sum(axis=1, keepdims=True)\ntrain[target] = y_data\n\n\n# For each eeg_id, the expert_consensus on the EEG segment's classification is included in the dataframe.\ntmp = train0.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\n# The eeg_id is converted into a regular column\ntrain = train.reset_index() \nprint('Shape of EEG segments without overlap in training data:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:08:42.559941Z","iopub.execute_input":"2024-04-11T16:08:42.560257Z","iopub.status.idle":"2024-04-11T16:08:42.650551Z","shell.execute_reply.started":"2024-04-11T16:08:42.560230Z","shell.execute_reply":"2024-04-11T16:08:42.649497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# start and end times of EEG segments\nplt.figure(figsize=(10, 6))\nplt.plot(train['start_time'], train.index, label='Start Time', marker='o', linestyle='', color='blue')\nplt.plot(train['end_time'], train.index, label='End Time', marker='o', linestyle='', color='red')\nplt.xlabel('Time (seconds)')\nplt.ylabel('EEG Segment Index')\nplt.title('Start and End Times of EEG Segments')\nplt.legend()\nplt.grid(True)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:08:42.653414Z","iopub.execute_input":"2024-04-11T16:08:42.653773Z","iopub.status.idle":"2024-04-11T16:08:43.247177Z","shell.execute_reply.started":"2024-04-11T16:08:42.653741Z","shell.execute_reply":"2024-04-11T16:08:43.246240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FEATS = [['Fp1','F7','T3','T5','O1'],\n#          ['Fp1','F3','C3','P3','O1'],\n#          ['Fp2','F8','T4','T6','O2'],\n#          ['Fp2','F4','C4','P4','O2']]","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:04:38.820065Z","iopub.status.idle":"2024-04-11T16:04:38.820570Z","shell.execute_reply.started":"2024-04-11T16:04:38.820303Z","shell.execute_reply":"2024-04-11T16:04:38.820329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nSPEC_COLS = pd.read_parquet(f'{PATH}1000086677.parquet').columns[1:]\nprint(SPEC_COLS.shape)\nSPEC_COLS","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:08:43.248350Z","iopub.execute_input":"2024-04-11T16:08:43.248643Z","iopub.status.idle":"2024-04-11T16:08:43.286732Z","shell.execute_reply.started":"2024-04-11T16:08:43.248617Z","shell.execute_reply":"2024-04-11T16:08:43.285800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SPEC_COLS","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:08:43.288105Z","iopub.execute_input":"2024-04-11T16:08:43.288375Z","iopub.status.idle":"2024-04-11T16:08:43.294149Z","shell.execute_reply.started":"2024-04-11T16:08:43.288350Z","shell.execute_reply":"2024-04-11T16:08:43.293173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\n\n\nspectrogram_columns = pd.read_parquet(f'{PATH}1000086677.parquet').columns[1:]\nfeatures = [f'{c}_mean_10m' for c in spectrogram_columns]\nfeatures += [f'{c}_min_10m' for c in spectrogram_columns]\nfeatures += [f'{c}_mean_20s' for c in spectrogram_columns]\nfeatures += [f'{c}_min_20s' for c in spectrogram_columns]\nfeatures += [f'{c}_mean_10s' for c in spectrogram_columns]\nfeatures += [f'{c}_min_10s' for c in spectrogram_columns]\nprint(f'We need to create {len(features)} features for {len(train)} rows... ')\n\n\n# Initializing data matrix to store new features\ndata = np.zeros((len(train), len(features)))\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:08:43.295332Z","iopub.execute_input":"2024-04-11T16:08:43.295637Z","iopub.status.idle":"2024-04-11T16:08:43.333310Z","shell.execute_reply.started":"2024-04-11T16:08:43.295612Z","shell.execute_reply":"2024-04-11T16:08:43.332463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Iterating through each row in train\n\n\nfor i, row in train.iterrows():\n    if i % 100 == 0:\n        print(i, ' ', end='')\n\n\n    window_range = int((row['start_time'] + row['end_time']) // 4)\n\n    # Calculating features for 10-minute window\n    window_10_min = spectrograms[row.spec_id][window_range:window_range + 300, :]\n    data[i, :400] = np.nanmean(window_10_min, axis=0)\n    data[i, 400:800] = np.nanmin(window_10_min, axis=0)\n\n    # Calculating features for 20-second window\n    window_20_sec = spectrograms[row.spec_id][window_range + 145:window_range + 155, :]\n    data[i, 800:1200] = np.nanmean(window_20_sec, axis=0)\n    data[i, 1200:1600] = np.nanmin(window_20_sec, axis=0)\n    \n    # Calculating features for 10-second window\n    window_10_sec = spectrograms[row.spec_id][window_range + 147:window_range + 153, :]\n    data[i, 1600:2000] = np.nanmean(window_10_sec, axis=0)\n    data[i, 2000:2400] = np.nanmin(window_10_sec, axis=0)\n    \n\n# Adding new features to train DataFrame\ntrain[features] = data\n\nprint()\nprint('New train shape:',train.shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:08:43.334307Z","iopub.execute_input":"2024-04-11T16:08:43.334550Z","iopub.status.idle":"2024-04-11T16:09:03.125788Z","shell.execute_reply.started":"2024-04-11T16:08:43.334528Z","shell.execute_reply":"2024-04-11T16:09:03.124643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:04:38.830810Z","iopub.status.idle":"2024-04-11T16:04:38.831532Z","shell.execute_reply.started":"2024-04-11T16:04:38.831269Z","shell.execute_reply":"2024-04-11T16:04:38.831295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:04:38.832710Z","iopub.status.idle":"2024-04-11T16:04:38.833641Z","shell.execute_reply.started":"2024-04-11T16:04:38.833364Z","shell.execute_reply":"2024-04-11T16:04:38.833392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:09:03.127033Z","iopub.execute_input":"2024-04-11T16:09:03.127361Z","iopub.status.idle":"2024-04-11T16:09:06.392471Z","shell.execute_reply.started":"2024-04-11T16:09:03.127331Z","shell.execute_reply":"2024-04-11T16:09:06.391457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = train[train.columns[12:]].values\ny = train[train.columns[5:11]].values","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:09:06.395631Z","iopub.execute_input":"2024-04-11T16:09:06.396275Z","iopub.status.idle":"2024-04-11T16:09:07.082536Z","shell.execute_reply.started":"2024-04-11T16:09:06.396241Z","shell.execute_reply":"2024-04-11T16:09:07.081258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique, frequency = np.unique(x, \n                              return_counts = True)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:09:07.083879Z","iopub.execute_input":"2024-04-11T16:09:07.084284Z","iopub.status.idle":"2024-04-11T16:09:10.159974Z","shell.execute_reply.started":"2024-04-11T16:09:07.084251Z","shell.execute_reply":"2024-04-11T16:09:10.158857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_FEATURES = 2400","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:12:14.053635Z","iopub.execute_input":"2024-04-11T16:12:14.054036Z","iopub.status.idle":"2024-04-11T16:12:14.058813Z","shell.execute_reply.started":"2024-04-11T16:12:14.054000Z","shell.execute_reply":"2024-04-11T16:12:14.057694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = tf.data.Dataset.from_tensor_slices((x, y))","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:09:10.168339Z","iopub.execute_input":"2024-04-11T16:09:10.168611Z","iopub.status.idle":"2024-04-11T16:09:11.529792Z","shell.execute_reply.started":"2024-04-11T16:09:10.168586Z","shell.execute_reply":"2024-04-11T16:09:11.529006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = dataset.cache()\ndataset = dataset.shuffle(160000)\ndataset = dataset.batch(16)\ndataset = dataset.prefetch(8)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:09:11.530884Z","iopub.execute_input":"2024-04-11T16:09:11.531163Z","iopub.status.idle":"2024-04-11T16:09:11.552027Z","shell.execute_reply.started":"2024-04-11T16:09:11.531138Z","shell.execute_reply":"2024-04-11T16:09:11.551144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_x, batch_y = dataset.as_numpy_iterator().next()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:09:11.553325Z","iopub.execute_input":"2024-04-11T16:09:11.553928Z","iopub.status.idle":"2024-04-11T16:09:11.849264Z","shell.execute_reply.started":"2024-04-11T16:09:11.553892Z","shell.execute_reply":"2024-04-11T16:09:11.848371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1 = dataset.take(int(len(dataset)*.8))\nval = dataset.skip(int(len(dataset)*.8)).take(int(len(dataset)*.2))","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:09:49.828587Z","iopub.execute_input":"2024-04-11T16:09:49.829236Z","iopub.status.idle":"2024-04-11T16:09:49.840460Z","shell.execute_reply.started":"2024-04-11T16:09:49.829189Z","shell.execute_reply":"2024-04-11T16:09:49.839610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, Dropout, Bidirectional, Dense, Embedding","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:09:52.387465Z","iopub.execute_input":"2024-04-11T16:09:52.387862Z","iopub.status.idle":"2024-04-11T16:09:52.416371Z","shell.execute_reply.started":"2024-04-11T16:09:52.387829Z","shell.execute_reply":"2024-04-11T16:09:52.415507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Embedding(MAX_FEATURES+1, 32))\nmodel.add(Bidirectional(LSTM(32, activation = 'tanh')))\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dense(256, activation='relu'))\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dense(6, activation='softmax'))","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:37:06.012089Z","iopub.execute_input":"2024-04-11T16:37:06.013143Z","iopub.status.idle":"2024-04-11T16:37:06.577547Z","shell.execute_reply.started":"2024-04-11T16:37:06.013101Z","shell.execute_reply":"2024-04-11T16:37:06.576652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss='KLDivergence', optimizer='Adam')","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:37:10.605351Z","iopub.execute_input":"2024-04-11T16:37:10.606062Z","iopub.status.idle":"2024-04-11T16:37:10.618080Z","shell.execute_reply.started":"2024-04-11T16:37:10.606023Z","shell.execute_reply":"2024-04-11T16:37:10.617294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:37:16.316840Z","iopub.execute_input":"2024-04-11T16:37:16.317251Z","iopub.status.idle":"2024-04-11T16:37:16.343454Z","shell.execute_reply.started":"2024-04-11T16:37:16.317214Z","shell.execute_reply":"2024-04-11T16:37:16.341067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train1, epochs=10, validation_data=val)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T16:37:19.429706Z","iopub.execute_input":"2024-04-11T16:37:19.430531Z","iopub.status.idle":"2024-04-11T17:03:22.998298Z","shell.execute_reply.started":"2024-04-11T16:37:19.430493Z","shell.execute_reply":"2024-04-11T17:03:22.997213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save(\"brain.h5\")","metadata":{"execution":{"iopub.status.busy":"2024-04-11T17:11:44.442813Z","iopub.execute_input":"2024-04-11T17:11:44.443787Z","iopub.status.idle":"2024-04-11T17:11:44.506510Z","shell.execute_reply.started":"2024-04-11T17:11:44.443748Z","shell.execute_reply":"2024-04-11T17:11:44.505650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\nimproved_model = load_model(\"brain.h5\")","metadata":{"execution":{"iopub.status.busy":"2024-04-11T17:13:52.305678Z","iopub.execute_input":"2024-04-11T17:13:52.306569Z","iopub.status.idle":"2024-04-11T17:13:52.957670Z","shell.execute_reply.started":"2024-04-11T17:13:52.306533Z","shell.execute_reply":"2024-04-11T17:13:52.956841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history2 = improved_model.fit(train1, epochs=50, validation_data=val)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T17:14:28.424270Z","iopub.execute_input":"2024-04-11T17:14:28.425054Z","iopub.status.idle":"2024-04-11T18:58:58.903800Z","shell.execute_reply.started":"2024-04-11T17:14:28.425001Z","shell.execute_reply":"2024-04-11T18:58:58.902850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"improved_model.save(\"brain2.h5\")","metadata":{"execution":{"iopub.status.busy":"2024-04-11T18:59:39.701202Z","iopub.execute_input":"2024-04-11T18:59:39.701630Z","iopub.status.idle":"2024-04-11T18:59:39.752522Z","shell.execute_reply.started":"2024-04-11T18:59:39.701594Z","shell.execute_reply":"2024-04-11T18:59:39.751714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape',test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T17:05:13.298867Z","iopub.execute_input":"2024-04-11T17:05:13.299438Z","iopub.status.idle":"2024-04-11T17:05:13.316711Z","shell.execute_reply.started":"2024-04-11T17:05:13.299392Z","shell.execute_reply":"2024-04-11T17:05:13.315826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FEATURE ENGINEER TEST\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\ndata = np.zeros((len(test),len(features)))\n    \nfor k in range(len(test)):\n    row = test.iloc[k]\n    s = int( row.spectrogram_id )\n    spec = pd.read_parquet(f'{PATH2}{s}.parquet')\n    \n    # 10 MINUTE WINDOW FEATURES\n    x = np.nanmean( spec.iloc[:,1:].values, axis=0)\n    data[k,:400] = x\n    x = np.nanmin( spec.iloc[:,1:].values, axis=0)\n    data[k,400:800] = x\n\n    # 20 SECOND WINDOW FEATURES\n    x = np.nanmean( spec.iloc[145:155,1:].values, axis=0)\n    data[k,800:1200] = x\n    x = np.nanmin( spec.iloc[145:155,1:].values, axis=0)\n    data[k,1200:1600] = x\n\ntest[features] = data\nprint('New test shape',test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T17:05:16.644923Z","iopub.execute_input":"2024-04-11T17:05:16.645653Z","iopub.status.idle":"2024-04-11T17:05:18.302304Z","shell.execute_reply.started":"2024-04-11T17:05:16.645612Z","shell.execute_reply":"2024-04-11T17:05:18.301321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"improved_model.predict(test[features])","metadata":{"execution":{"iopub.status.busy":"2024-04-11T19:04:58.324582Z","iopub.execute_input":"2024-04-11T19:04:58.325641Z","iopub.status.idle":"2024-04-11T19:04:59.409898Z","shell.execute_reply.started":"2024-04-11T19:04:58.325596Z","shell.execute_reply":"2024-04-11T19:04:59.409017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []\n\nfor i in range(5):\n    print(i, ', ', end='')\n    \n    # Make predictions\n    pred = improved_model.predict(test[features])\n    preds.append(pred)\n\n# Average the predictions from each fold\npred = np.mean(preds, axis=0)\nprint()\nprint('Test preds shape', pred.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T19:05:21.630729Z","iopub.execute_input":"2024-04-11T19:05:21.631150Z","iopub.status.idle":"2024-04-11T19:05:22.641538Z","shell.execute_reply.started":"2024-04-11T19:05:21.631117Z","shell.execute_reply":"2024-04-11T19:05:22.640742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame({'eeg_id':test.eeg_id.values})\nsub[TARGETS] = pred\nsub.to_csv('submission.csv',index=False)\nprint('Submission shape',sub.shape)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T19:05:44.227870Z","iopub.execute_input":"2024-04-11T19:05:44.228555Z","iopub.status.idle":"2024-04-11T19:05:44.257262Z","shell.execute_reply.started":"2024-04-11T19:05:44.228517Z","shell.execute_reply":"2024-04-11T19:05:44.256324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SANITY CHECK TO CONFIRM PREDICTIONS SUM TO ONE\nsub.iloc[:,-6:].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T19:06:06.380344Z","iopub.execute_input":"2024-04-11T19:06:06.381732Z","iopub.status.idle":"2024-04-11T19:06:06.391415Z","shell.execute_reply.started":"2024-04-11T19:06:06.381667Z","shell.execute_reply":"2024-04-11T19:06:06.390453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}