{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport tensorflow as tf\nfrom matplotlib import pyplot as plt\nimport keras\nfrom keras import ops\nfrom sklearn.model_selection import KFold, GroupKFold\nfrom sklearn.preprocessing import StandardScaler\nfrom keras.models import Sequential, Model\nfrom keras.layers import Input, Dense, Activation, Dropout\nfrom keras.layers import BatchNormalization\nfrom keras.utils import to_categorical\nfrom keras import regularizers, layers\nfrom keras.callbacks import History, EarlyStopping\nfrom keras.optimizers import SGD\nimport sys\nimport os, gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T05:08:02.875075Z","iopub.execute_input":"2024-11-16T05:08:02.876047Z","iopub.status.idle":"2024-11-16T05:08:17.610158Z","shell.execute_reply.started":"2024-11-16T05:08:02.875981Z","shell.execute_reply":"2024-11-16T05:08:17.609391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\ntargets = train_dataset.columns[-6:]\n\nprint(\"Training data shape:\", train_dataset.shape)\nprint(\"Training targets:\", targets)\nprint(train_dataset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T05:08:17.611721Z","iopub.execute_input":"2024-11-16T05:08:17.612210Z","iopub.status.idle":"2024-11-16T05:08:17.875609Z","shell.execute_reply.started":"2024-11-16T05:08:17.612178Z","shell.execute_reply":"2024-11-16T05:08:17.874634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_eeg_parquet = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1003529080.parquet', engine='pyarrow')\nprint(sample_eeg_parquet)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T05:08:17.877191Z","iopub.execute_input":"2024-11-16T05:08:17.878161Z","iopub.status.idle":"2024-11-16T05:08:18.074036Z","shell.execute_reply.started":"2024-11-16T05:08:17.878108Z","shell.execute_reply":"2024-11-16T05:08:18.073129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_spectrogram_parquet = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/1000086677.parquet', engine='pyarrow')\nprint(sample_spectrogram_parquet)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T05:08:18.076040Z","iopub.execute_input":"2024-11-16T05:08:18.076364Z","iopub.status.idle":"2024-11-16T05:08:18.127102Z","shell.execute_reply.started":"2024-11-16T05:08:18.076314Z","shell.execute_reply":"2024-11-16T05:08:18.126200Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nREAD_SPEC_FILES = False #Required to be false to load the single npy file\nREAD_EEG_SPEC_FILES = False\n\n# READ ALL SPECTROGRAMS\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nfiles = os.listdir(PATH)\nprint(f'There are {len(files)} spectrogram parquets')\nspectrograms = np.load('/kaggle/input/brain-spectrograms/specs.npy',allow_pickle=True).item()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T05:08:18.128327Z","iopub.execute_input":"2024-11-16T05:08:18.128809Z","iopub.status.idle":"2024-11-16T05:08:54.707016Z","shell.execute_reply.started":"2024-11-16T05:08:18.128766Z","shell.execute_reply":"2024-11-16T05:08:54.706040Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train_dataset.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'}) #Find minimum offset for each unique id\ntrain.columns = ['spec_id','min']\n\ntmp = train_dataset.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'}) #Find maximum offset for each unique id\ntrain['max'] = tmp\n\ntmp = train_dataset.groupby('eeg_id')[['patient_id']].agg('first') #Add patient id to dataframe\ntrain['patient_id'] = tmp\n\ntmp = train_dataset.groupby('eeg_id')[targets].agg('sum') #Add vote counts per id to dataframe\nfor target in targets:\n    train[target] = tmp[target].values\n\n#Normalize vote counts\ny_data = train[targets].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[targets] = y_data\n\ntmp = train_dataset.groupby('eeg_id')[['expert_consensus']].agg('first') #Add consensus vote target to dataframe\ntrain['target'] = tmp\n\ntrain = train.reset_index()\nprint('Train non-overlapp eeg_id shape:', train.shape )\n\nprint(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T05:08:54.708202Z","iopub.execute_input":"2024-11-16T05:08:54.708544Z","iopub.status.idle":"2024-11-16T05:08:54.798409Z","shell.execute_reply.started":"2024-11-16T05:08:54.708509Z","shell.execute_reply":"2024-11-16T05:08:54.797411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time\n# ENGINEER FEATURES\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# FEATURE NAMES\nSPEC_COLS = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/1000086677.parquet').columns[1:]\nFEATURES = [f'{c}_mean_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_mean_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_20s' for c in SPEC_COLS]\nprint(f'We are creating {len(FEATURES)} features for {len(train)} rows... ',end='')\n\n# print(SPEC_COLS)\ndata = np.zeros((len(train),len(FEATURES)))\nfor k in range(len(train)):\n    if k%100==0: print(k,', ',end='')\n    row = train.iloc[k]\n    r = int( (row['min'] + row['max'])//4 ) \n\n    # 10 MINUTE WINDOW FEATURES (MEANS and MINS)\n    x = np.nanmean(spectrograms[row.spec_id][r:r+300,:],axis=0)\n    data[k,:400] = x\n    x = np.nanmin(spectrograms[row.spec_id][r:r+300,:],axis=0)\n    data[k,400:800] = x\n\n    # 20 SECOND WINDOW FEATURES (MEANS and MINS)\n    x = np.nanmean(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n    data[k,800:1200] = x\n    x = np.nanmin(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n    data[k,1200:1600] = x\n\n\ntrain[FEATURES] = data\nprint(); print('New train shape:',train.shape)\n\n# FREE MEMORY\ndel spectrograms, data\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T05:08:54.799807Z","iopub.execute_input":"2024-11-16T05:08:54.800121Z","iopub.status.idle":"2024-11-16T05:09:12.840136Z","shell.execute_reply.started":"2024-11-16T05:08:54.800088Z","shell.execute_reply":"2024-11-16T05:09:12.839053Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T05:09:12.841699Z","iopub.execute_input":"2024-11-16T05:09:12.842121Z","iopub.status.idle":"2024-11-16T05:09:12.874397Z","shell.execute_reply.started":"2024-11-16T05:09:12.842075Z","shell.execute_reply":"2024-11-16T05:09:12.873453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_model():\n    model = Sequential()\n    model.add(Input(shape=(1600,)))\n    model.add(Dense(2100, activation=\"sigmoid\", name=\"HiddenLayer1\"))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.2))\n\n    model.add(Dense(1600, activation=\"sigmoid\", name=\"HiddenLayer2\"))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.2))\n\n    model.add(Dense(1600, activation=\"sigmoid\", name=\"HiddenLayer3\"))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.2))\n\n    model.add(Dense(1600, activation=\"sigmoid\", name=\"HiddenLayer4\"))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.2))\n\n    model.add(Dense(1024, activation=\"sigmoid\", name=\"HiddenLayer5\"))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.2))\n\n    model.add(Dense(1024, activation=\"sigmoid\", name=\"HiddenLayer6\"))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.2))\n\n    model.add(Dense(1024, activation=\"sigmoid\", name=\"HiddenLayer7\"))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.2))\n\n    model.add(Dense(512, activation=\"sigmoid\", name=\"HiddenLayer8\"))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.2))    \n\n    model.add(Dense(256, activation=\"sigmoid\", name=\"HiddenLayer9\"))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.2))    \n\n    model.add(Dense(32, activation=\"sigmoid\", name=\"HiddenLayer10\"))\n    model.add(BatchNormalization())\n    model.add(Dropout(0.2))\n\n    model.add(Dense(6, activation=\"softmax\", name=\"Output\"))\n\n    opt = SGD(learning_rate=1e-3, momentum=0.9, clipvalue=1)\n    model.compile(loss='sparse_categorical_crossentropy', optimizer=opt, metrics=['sparse_categorical_accuracy'])\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T05:09:12.875737Z","iopub.execute_input":"2024-11-16T05:09:12.876061Z","iopub.status.idle":"2024-11-16T05:09:12.889792Z","shell.execute_reply.started":"2024-11-16T05:09:12.876027Z","shell.execute_reply":"2024-11-16T05:09:12.888705Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_oof = []\nall_true = []\ntarget_map = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\nhistory = []\n\nmodel = create_model()\nmodel.summary()\nmodel.save(\"TestModel.keras\")\n\ntrain[FEATURES] = train[FEATURES].fillna(0)\ngroup_kfold = GroupKFold(n_splits=5)\nfor i, (train_index, valid_index) in enumerate(group_kfold.split(train[:], train[:].target, train[:].patient_id)):\n    print(f'### Fold {i+1}')\n    print(f'### train size {len(train_index)}, valid size {len(valid_index)}')\n    \n    #training data\n    data_train = train.loc[train_index,FEATURES] #extract feature data\n    label_train = train.loc[train_index,'target'].map(target_map) #extract label data\n    #label_train = to_categorical(label_train, 6)\n    \n    #validation data\n    data_valid = train.loc[valid_index,FEATURES]\n    label_valid = train.loc[valid_index,'target'].map(target_map)\n    \n    #Normalize input features\n    scaler = StandardScaler().fit(data_train)\n    data_train = scaler.transform(data_train)\n    data_valid = scaler.transform(data_valid)  \n \n    assert not np.any(np.isnan(data_train))\n    \n    \n    # Checkpoint Callback\n    checkpoint_filepath = 'epoch-{epoch:02d}-loss-{val_loss:.3f}.weights.h5'\n    checkpoint = keras.callbacks.ModelCheckpoint(\n        filepath=checkpoint_filepath,\n        save_weights_only=True,\n        monitor='val_loss',\n        mode='min',\n        save_best_only=True,\n        save_freq='epoch')\n    es = EarlyStopping(monitor='val_loss',mode='min', patience=15)\n    callbacks_list = [checkpoint, es]\n    fold_history = model.fit(x=data_train, y=label_train, verbose=1, epochs=50, validation_data=(data_valid,label_valid),callbacks=callbacks_list)\n    history.append(fold_history)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T05:09:12.893131Z","iopub.execute_input":"2024-11-16T05:09:12.893491Z","iopub.status.idle":"2024-11-16T05:15:29.856443Z","shell.execute_reply.started":"2024-11-16T05:09:12.893456Z","shell.execute_reply":"2024-11-16T05:15:29.855417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.plot(history[4].history['sparse_categorical_accuracy'])\nplt.plot(history[4].history['val_sparse_categorical_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'val'], loc='upper left')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T05:15:29.858619Z","iopub.execute_input":"2024-11-16T05:15:29.859022Z","iopub.status.idle":"2024-11-16T05:15:30.186507Z","shell.execute_reply.started":"2024-11-16T05:15:29.858977Z","shell.execute_reply":"2024-11-16T05:15:30.185340Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"validation_predictions = model.predict(data_valid)\nprint(validation_predictions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T05:15:30.187992Z","iopub.execute_input":"2024-11-16T05:15:30.188886Z","iopub.status.idle":"2024-11-16T05:15:31.700155Z","shell.execute_reply.started":"2024-11-16T05:15:30.188841Z","shell.execute_reply":"2024-11-16T05:15:31.699213Z"}},"outputs":[],"execution_count":null}]}