{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":8011206,"sourceType":"datasetVersion","datasetId":4719156},{"sourceId":8013891,"sourceType":"datasetVersion","datasetId":4721331},{"sourceId":8014144,"sourceType":"datasetVersion","datasetId":4721498},{"sourceId":168829242,"sourceType":"kernelVersion"}],"dockerImageVersionId":30674,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Install Extra Packages Offline","metadata":{"execution":{"iopub.status.busy":"2024-04-03T00:07:03.099881Z","iopub.execute_input":"2024-04-03T00:07:03.100760Z","iopub.status.idle":"2024-04-03T00:07:15.631906Z","shell.execute_reply.started":"2024-04-03T00:07:03.100722Z","shell.execute_reply":"2024-04-03T00:07:15.630730Z"}}},{"cell_type":"markdown","source":"1) tslearn","metadata":{}},{"cell_type":"code","source":"!pip install '/kaggle/input/hmsofflinepackages/tslearn-0.6.3-py3-none-any.whl'","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:21:33.181213Z","iopub.execute_input":"2024-04-03T01:21:33.181477Z","iopub.status.idle":"2024-04-03T01:22:06.653620Z","shell.execute_reply.started":"2024-04-03T01:21:33.181452Z","shell.execute_reply":"2024-04-03T01:22:06.652465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport zipfile\n\ndef zip_folder(folder_path, output_zip):\n    \"\"\"\n    Zip the contents of an entire folder (with that folder included\n    in the archive). Empty directories are included in the archive as well.\n    \"\"\"\n    with zipfile.ZipFile(output_zip, 'w', zipfile.ZIP_DEFLATED) as zipf:\n        lenDirPath = len(folder_path)\n        for root, _, files in os.walk(folder_path):\n            # Include all subdirectories, including empty ones.\n            for dirName in os.listdir(root):\n                dirPath = os.path.join(root, dirName)\n                if os.path.isdir(dirPath):\n                    zipf.write(dirPath, os.path.relpath(dirPath, folder_path))\n            # Add files\n            for file in files:\n                filePath = os.path.join(root, file)\n                zipf.write(filePath, os.path.relpath(filePath, folder_path))","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:22:06.655650Z","iopub.execute_input":"2024-04-03T01:22:06.655970Z","iopub.status.idle":"2024-04-03T01:22:06.664317Z","shell.execute_reply.started":"2024-04-03T01:22:06.655926Z","shell.execute_reply":"2024-04-03T01:22:06.663385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"2) signatory","metadata":{}},{"cell_type":"code","source":"zip_folder(\"/kaggle/input/hmsofflinepackages/signatory-1.2.6.1.9.0\", \"./signatory.zip\")\n\n!pip install \"./signatory.zip\"","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:22:06.665389Z","iopub.execute_input":"2024-04-03T01:22:06.665648Z","iopub.status.idle":"2024-04-03T01:23:55.617295Z","shell.execute_reply.started":"2024-04-03T01:22:06.665627Z","shell.execute_reply":"2024-04-03T01:23:55.616306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# standard packages\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport random\nimport pandas as pd\nimport torch","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:23:55.620838Z","iopub.execute_input":"2024-04-03T01:23:55.621610Z","iopub.status.idle":"2024-04-03T01:23:58.419174Z","shell.execute_reply.started":"2024-04-03T01:23:55.621575Z","shell.execute_reply":"2024-04-03T01:23:58.418095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from offline package installations\nfrom tslearn.preprocessing import TimeSeriesScalerMinMax\nimport signatory","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:23:58.420333Z","iopub.execute_input":"2024-04-03T01:23:58.420723Z","iopub.status.idle":"2024-04-03T01:24:00.334875Z","shell.execute_reply.started":"2024-04-03T01:23:58.420698Z","shell.execute_reply":"2024-04-03T01:24:00.333924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare Train Set","metadata":{}},{"cell_type":"markdown","source":"Load signature features and prepare for training NN","metadata":{}},{"cell_type":"code","source":"def TrainPreProcessing(DataPortion, Validation_Split):\n    \n    # Load all numpy arrays \n    EEG_Sig_Total = np.load('/kaggle/input/logsig3scaled/eeg_data_scaled.npy')   \n    NumVotes_Total = np.load('/kaggle/input/logsig3scaled/num_votes.npy')\n    Targets_Total = np.load('/kaggle/input/logsig3scaled/targets.npy')\n\n    n_features = 2470\n    \n    # Combine columns into one array: ¦2470 Features¦6 Targets¦NumVotes¦ \n    TotalDataset = np.hstack((EEG_Sig_Total, Targets_Total, NumVotes_Total))\n    \n    # We want to drop all rows with nans in them\n    nan_rows = np.isnan(TotalDataset).any(axis=1)\n    # Drop rows with NaN values\n    TotalDataset = TotalDataset[~nan_rows]\n    \n    TotalN = TotalDataset.shape[0]\n\n    # shuffle rows\n    np.random.seed(69)\n    p = np.random.permutation(TotalN)\n\n    TotalDataset = TotalDataset[p]\n\n    # Work with only a small portion of the dataset for experimentation \n    N = round(DataPortion*TotalN)\n\n    TruncatedDataset = TotalDataset[:N]\n    \n    # set proportion of train data for validation set\n    Validation_N = round(Validation_Split*N)\n\n    ValidationDataset = TruncatedDataset[:Validation_N]\n    TrainDataset = TruncatedDataset[Validation_N:]\n\n    X_val, y_val = ValidationDataset[:,:n_features], ValidationDataset[:,n_features:n_features+6]\n    X_train, y_train = TrainDataset[:,:n_features], TrainDataset[:,n_features:n_features+6]\n    N_votes_train = TrainDataset[:,-1]\n\n    return X_train, y_train, X_val, y_val, N_votes_train\n","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:24:00.336196Z","iopub.execute_input":"2024-04-03T01:24:00.336628Z","iopub.status.idle":"2024-04-03T01:24:00.345989Z","shell.execute_reply.started":"2024-04-03T01:24:00.336601Z","shell.execute_reply":"2024-04-03T01:24:00.345067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Do not need a test set from the train data since we are submitting\nX_train, y_train, X_test, y_test, N_votes_train = TrainPreProcessing(DataPortion=1, Validation_Split=0)\n\nprint(X_train.shape, y_train.shape)\nprint(X_test.shape, y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:24:00.347072Z","iopub.execute_input":"2024-04-03T01:24:00.347329Z","iopub.status.idle":"2024-04-03T01:24:20.338560Z","shell.execute_reply.started":"2024-04-03T01:24:00.347306Z","shell.execute_reply":"2024-04-03T01:24:20.337661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build MLP Model","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, Model\nfrom tensorflow.keras.losses import kullback_leibler_divergence\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n\n# Define the number of input features and output classes\nnum_features = 2470\n\nnum_classes = 6","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:24:20.339757Z","iopub.execute_input":"2024-04-03T01:24:20.340079Z","iopub.status.idle":"2024-04-03T01:24:31.658201Z","shell.execute_reply.started":"2024-04-03T01:24:20.340054Z","shell.execute_reply":"2024-04-03T01:24:31.657236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set random seeds for reproducibility\nseed_value = 69\n\n# 1. Set the seed for Python's built-in random number generator\nrandom.seed(seed_value)\n# 2. Set the seed for NumPy\nnp.random.seed(seed_value)\n# 3. Set the seed for TensorFlow\ntf.random.set_seed(seed_value)\n\n# Performance on total train dataset w/ 10% portioned for test: 0.4381\n\n\n\n# Random attempt\ndef create_model():\n    inputs = tf.keras.Input(shape=(num_features,))\n    \n    x = layers.Dense(800, activation='sigmoid')(inputs) # 800 is good\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(0.5)(x)\n    \n    x = layers.Dense(500, activation='relu')(x) # 500 is good\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(0.2)(x)  \n    \n    x = layers.Dense(500, activation='relu')(x) # 500 is good\n    x = layers.Dropout(0.2)(x)  \n    \n    x = layers.Dense(350, activation='relu')(x) # 350 is good   \n    x = layers.Dropout(0.1)(x)\n\n    outputs = layers.Dense(num_classes, activation='softmax')(x)\n    model = Model(inputs, outputs)\n    return model\n\n# Instantiate the model\nmodel = create_model()\n\n# Compile the model with KL divergence loss\nmodel.compile(optimizer='adam',\n              loss=kullback_leibler_divergence)\n\n# Print the model summary\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:24:31.659479Z","iopub.execute_input":"2024-04-03T01:24:31.660180Z","iopub.status.idle":"2024-04-03T01:24:32.448755Z","shell.execute_reply.started":"2024-04-03T01:24:31.660146Z","shell.execute_reply":"2024-04-03T01:24:32.447918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load MLP Weights","metadata":{}},{"cell_type":"code","source":"model.load_weights('/kaggle/input/praneesh-mlp-model-weights/TrainedModel.weights.h5')\n","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-04-03T01:24:32.451370Z","iopub.execute_input":"2024-04-03T01:24:32.451661Z","iopub.status.idle":"2024-04-03T01:24:32.636960Z","shell.execute_reply.started":"2024-04-03T01:24:32.451635Z","shell.execute_reply":"2024-04-03T01:24:32.636023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Process Test Set","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape:',test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:26:07.237519Z","iopub.execute_input":"2024-04-03T01:26:07.238178Z","iopub.status.idle":"2024-04-03T01:26:07.250878Z","shell.execute_reply.started":"2024-04-03T01:26:07.238145Z","shell.execute_reply":"2024-04-03T01:26:07.249853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EEG_SUB_PATH_TEMPL = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\nSP_SUB_PATH_TEMPL = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\n\ndef get_sub_eeg_sp_data(train_row):\n    \"\"\"Gets EEG and Spectogram data from a specific row in the dataset\"\"\"\n    \n    eeg_id = train_row.eeg_id\n    sp_id = train_row.spectrogram_id\n    \n    eeg_parquet = pd.read_parquet(f'{EEG_SUB_PATH_TEMPL}{eeg_id}.parquet')\n    sp = pd.read_parquet(f'{SP_SUB_PATH_TEMPL}{sp_id}.parquet')\n    \n    rows = len(eeg_parquet)\n    eeg_offset = (rows-10_000)//2\n    \n    \n    # get middle 50 seconds of eeg data\n    #eeg_offset = int(train_row.eeg_label_offset_seconds + 20) #only 10 central seconds from 50 secs were labeled, which should be seconds 20-30 in the sample\n    eeg_data = eeg_parquet.iloc[eeg_offset:eeg_offset + 10_000]\n    \n    \n    # sp_offset = int(train_row.spectrogram_label_offset_seconds )\n    \n    # get spectrogram data\n    # sp = sp_parquet.loc[(sp_parquet.time>=sp_offset)&(sp_parquet.time<sp_offset+SP_WIN)]\n    sp = sp.loc[:, sp.columns != 'time']\n    sp = {\n        \"LL\": sp.filter(regex='^LL', axis=1),\n        \"RL\": sp.filter(regex='^RL', axis=1),\n        \"RP\": sp.filter(regex='^RP', axis=1),\n        \"LP\": sp.filter(regex='^LP', axis=1)}\n    \n    # calculate eeg data\n    # print(eeg_data.keys()) # Has keys Index(['Fp1', 'F3', 'C3', 'P3', 'F7', 'T3', 'T5', 'O1', 'Fz', 'Cz', 'Pz',\n                            # 'Fp2', 'F4', 'C4', 'P4', 'F8', 'T4', 'T6', 'O2', 'EKG']\n    # assert 0 == 1\n    \n    CHAINS = {\n    'LL' : [(\"Fp1\",\"F7\"),(\"F7\",\"T3\"),(\"T3\",\"T5\"),(\"T5\",\"O1\")],\n    'RL' : [(\"Fp2\",\"F8\"),(\"F8\",\"T4\"),(\"T4\",\"T6\"),(\"T6\",\"O2\")],\n    'LP' : [(\"Fp1\",\"F3\"),(\"F3\",\"C3\"),(\"C3\",\"P3\"),(\"P3\",\"O1\")],\n    'RP' : [(\"Fp2\",\"F4\"),(\"F4\",\"C4\"),(\"C4\",\"P4\"),(\"P4\",\"O2\")],\n    'other' : [(\"Fz\",\"Cz\"), (\"Cz\", \"Pz\"), (\"EKG\")]\n}\n    \n    eeg = pd.DataFrame({})\n    for chain in CHAINS.keys():\n        for s_i, signals in enumerate(CHAINS[chain]):\n            if len(signals) == 2:\n                diff=eeg_data[signals[0]]-eeg_data[signals[1]] # Subtracts relevant fields as in the image above\n                diff.ffill(inplace = True) # forward fills in the casse of nan values\n                eeg[f\"{chain}: {signals[0]} - {signals[1]}\"] = diff\n            \n            elif len(signals) == 1:\n                sig=eeg_data[signals[0]]\n                sig.ffill(inplace = True) \n                eeg[f\"{chain}: {signals[0]}\"] = sig\n                \n                \n    \n    return eeg, sp   ","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:26:07.977494Z","iopub.execute_input":"2024-04-03T01:26:07.977865Z","iopub.status.idle":"2024-04-03T01:26:07.991256Z","shell.execute_reply.started":"2024-04-03T01:26:07.977831Z","shell.execute_reply":"2024-04-03T01:26:07.990215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess(pre_eeg):\n\n    tot_len = 53400\n    n_chunks = 10\n    chunk = int(tot_len // n_chunks)\n\n    for i in range(n_chunks):\n        if i == n_chunks-1:\n            eeg_data = pre_eeg[i*chunk:]\n        else:\n            eeg_data = pre_eeg[i*chunk:(i+1)*chunk]\n            \n        print(eeg_data.shape)\n\n        device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n        print(device)\n\n        ts_scaler = TimeSeriesScalerMinMax()\n        eeg_data = ts_scaler.fit_transform(eeg_data)\n        print(np.nanmax(eeg_data), np.nanmin(eeg_data))\n\n        eeg_data_cuda = torch.from_numpy(eeg_data).to(device)\n        exp = torch.linspace(0, 1, eeg_data_cuda.shape[1], device=device).unsqueeze(0).unsqueeze(-1).expand_as(eeg_data_cuda[:, :, 0].unsqueeze(-1))\n        eeg_data_cuda = torch.cat([eeg_data_cuda, exp], dim=2)\n        print(eeg_data_cuda.shape)\n\n        sig = signatory.logsignature(eeg_data_cuda, 3)\n        print(sig.shape)\n        eeg_data = sig.cpu().numpy()\n\n        # eeg_sig = np.concatenate(eeg_data, axis=-1)\n        return eeg_data","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:26:08.616750Z","iopub.execute_input":"2024-04-03T01:26:08.617613Z","iopub.status.idle":"2024-04-03T01:26:08.626702Z","shell.execute_reply.started":"2024-04-03T01:26:08.617578Z","shell.execute_reply":"2024-04-03T01:26:08.625730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eeg_arr = []\nfor i in tqdm(range(len(test))):\n        exp_row = test.iloc[i]\n        eeg_data, sp_dict = get_sub_eeg_sp_data(exp_row)\n        eeg_arr.append(eeg_data.to_numpy())\n\neeg_arr = np.array(eeg_arr)\n\neeg_arr","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:26:09.633904Z","iopub.execute_input":"2024-04-03T01:26:09.634799Z","iopub.status.idle":"2024-04-03T01:26:09.908127Z","shell.execute_reply.started":"2024-04-03T01:26:09.634766Z","shell.execute_reply":"2024-04-03T01:26:09.907273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame({'eeg_id':test.eeg_id.values})\nprep_eeg_sub = preprocess(eeg_arr)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:26:10.605308Z","iopub.execute_input":"2024-04-03T01:26:10.605666Z","iopub.status.idle":"2024-04-03T01:26:11.265459Z","shell.execute_reply.started":"2024-04-03T01:26:10.605638Z","shell.execute_reply":"2024-04-03T01:26:11.264485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict from processed test set","metadata":{}},{"cell_type":"code","source":"sub_pred = model.predict(prep_eeg_sub)\n\nprint(\"This is the sub pred sum:\",sub_pred.sum())\nprint(sub_pred)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:26:12.215038Z","iopub.execute_input":"2024-04-03T01:26:12.215384Z","iopub.status.idle":"2024-04-03T01:26:13.194217Z","shell.execute_reply.started":"2024-04-03T01:26:12.215357Z","shell.execute_reply":"2024-04-03T01:26:13.193242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CREATE SUBMISSION.CSV\nfrom IPython.display import display\n\nsub = pd.DataFrame({'eeg_id':test.eeg_id.values})\n\nTARGETS = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\nsub[TARGETS] = sub_pred\nsub.to_csv('submission.csv',index=False)\nprint('Submission shape',sub.shape)\ndisplay( sub.head() )\n\n# SANITY CHECK TO CONFIRM PREDICTIONS SUM TO ONE\nprint('Sub row 0 sums to:',sub.iloc[0,-6:].sum())","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:26:22.144611Z","iopub.execute_input":"2024-04-03T01:26:22.144987Z","iopub.status.idle":"2024-04-03T01:26:22.171717Z","shell.execute_reply.started":"2024-04-03T01:26:22.144958Z","shell.execute_reply":"2024-04-03T01:26:22.170708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}