{"metadata":{"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7437371,"sourceType":"datasetVersion","datasetId":4328630},{"sourceId":7437573,"sourceType":"datasetVersion","datasetId":4328761},{"sourceId":159656961,"sourceType":"kernelVersion"}],"dockerImageVersionId":30636,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.12"},"papermill":{"default_parameters":{},"duration":102.485133,"end_time":"2024-01-16T12:14:21.10455","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-01-16T12:12:38.619417","version":"2.4.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np, os\nimport matplotlib.pyplot as plt, gc\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom scipy.signal import butter, lfilter\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom scipy.signal import freqz\n\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom tqdm import tqdm","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.693827,"end_time":"2024-01-16T12:12:42.606147","exception":false,"start_time":"2024-01-16T12:12:41.91232","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-01-19T20:15:35.218962Z","iopub.execute_input":"2024-01-19T20:15:35.219660Z","iopub.status.idle":"2024-01-19T20:15:40.723873Z","shell.execute_reply.started":"2024-01-19T20:15:35.219618Z","shell.execute_reply":"2024-01-19T20:15:40.722815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nprint('Train shape', train.shape )\ndisplay( train.head() )","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.693827,"end_time":"2024-01-16T12:12:42.606147","exception":false,"start_time":"2024-01-16T12:12:41.91232","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-01-19T20:15:40.726103Z","iopub.execute_input":"2024-01-19T20:15:40.726799Z","iopub.status.idle":"2024-01-19T20:15:41.053998Z","shell.execute_reply.started":"2024-01-19T20:15:40.726736Z","shell.execute_reply":"2024-01-19T20:15:41.053213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGETS = train.columns[-6:]","metadata":{"execution":{"iopub.status.busy":"2024-01-19T20:15:41.055441Z","iopub.execute_input":"2024-01-19T20:15:41.056019Z","iopub.status.idle":"2024-01-19T20:15:41.061090Z","shell.execute_reply.started":"2024-01-19T20:15:41.055985Z","shell.execute_reply":"2024-01-19T20:15:41.059617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def butter_bandpass(lowcut, highcut, fs, order=5):\n    return butter(order, [lowcut, highcut], fs=fs, btype='band')\n\ndef butter_bandpass_filter(data, lowcut, highcut, fs, order=5):\n    b, a = butter_bandpass(lowcut, highcut, fs, order=order)\n    y = lfilter(b, a, data)\n    return y\n\n\ndef denoise_filter(x):\n    # Sample rate and desired cutoff frequencies (in Hz).\n    fs = 200.0\n    lowcut = 1.0\n    highcut = 25.0\n    \n    # Filter a noisy signal.\n    T = 50\n    nsamples = T * fs\n    t = np.arange(0, nsamples) / fs\n    y = butter_bandpass_filter(x, lowcut, highcut, fs, order=6)\n    y = (y + np.roll(y,-1)+ np.roll(y,-2)+ np.roll(y,-3))/4\n    y = y[0:-1:4]\n    \n    return y","metadata":{"execution":{"iopub.status.busy":"2024-01-19T20:15:41.064152Z","iopub.execute_input":"2024-01-19T20:15:41.064587Z","iopub.status.idle":"2024-01-19T20:15:41.074582Z","shell.execute_reply.started":"2024-01-19T20:15:41.064552Z","shell.execute_reply":"2024-01-19T20:15:41.073553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IS_TRAINING=False","metadata":{"execution":{"iopub.status.busy":"2024-01-19T20:15:41.076440Z","iopub.execute_input":"2024-01-19T20:15:41.076895Z","iopub.status.idle":"2024-01-19T20:15:41.088304Z","shell.execute_reply.started":"2024-01-19T20:15:41.076860Z","shell.execute_reply":"2024-01-19T20:15:41.087054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if IS_TRAINING:\n    eegs_data = np.load('/kaggle/input/hms-eeg-raw-dataset/eeg_specs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2024-01-19T20:15:41.089694Z","iopub.execute_input":"2024-01-19T20:15:41.090068Z","iopub.status.idle":"2024-01-19T20:17:00.583633Z","shell.execute_reply.started":"2024-01-19T20:15:41.090038Z","shell.execute_reply":"2024-01-19T20:17:00.581178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataloader","metadata":{"papermill":{"duration":0.028035,"end_time":"2024-01-16T12:13:50.519598","exception":false,"start_time":"2024-01-16T12:13:50.491563","status":"completed"},"tags":[]}},{"cell_type":"code","source":"NAMES = ['LL','LP','RP','RR']\n\nFEATS = [['Fp1','F7','T3','T5','O1'],\n         ['Fp1','F3','C3','P3','O1'],\n         ['Fp2','F8','T4','T6','O2'],\n         ['Fp2','F4','C4','P4','O2']]\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\n\nclass CustomDataset(Dataset):\n    def __init__(self, dataframe, transform=None):\n        self.dataframe = dataframe\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, idx):\n        \n        row = self.dataframe.iloc[idx]\n        eeg_id = row['eeg_id']\n        eeg_sub_id = row['eeg_sub_id']\n        eeg_key = f'{eeg_id}_{eeg_sub_id}'\n        signals = eegs_data[eeg_key]\n        labels = row[TARGETS].values.astype(np.float64) #np.array(row[-6:]).reshape(6,1)\n        labels = labels/np.sum(labels)\n        return torch.tensor(signals,dtype=torch.float64), torch.tensor(labels,dtype=torch.float64)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-01-19T20:19:14.003788Z","iopub.execute_input":"2024-01-19T20:19:14.004293Z","iopub.status.idle":"2024-01-19T20:19:14.015830Z","shell.execute_reply.started":"2024-01-19T20:19:14.004257Z","shell.execute_reply":"2024-01-19T20:19:14.014942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the 1D CNN model\nclass CNN1D(nn.Module):\n    def __init__(self,in_channels):\n        super(CNN1D, self).__init__()\n        self.hidden_channels = 128\n        self.conv1 = nn.Conv1d(in_channels, 64, 20, 10)\n        self.conv2 = nn.Conv1d(64, 64, 10, 5)\n        self.conv3 = nn.Conv1d(64, 64, 12, 4)\n#         self.conv3 = nn.Conv1d(32, 32, 98, 1)\n        self.flatten = nn.Flatten()\n#         self.pool = nn.MaxPool1d(kernel_size=2, stride=2)\n        self.fc1 = nn.Linear(640, 32)\n        self.fc2 = nn.Linear(32, 6)  # Adjust the input size based on your input dimensions\n        self.softmax = nn.Softmax(dim=1)\n        self.dropout = nn.Dropout(0.2)\n\n    def forward(self, x):\n#         print(x.shape)\n        x = F.relu(self.conv1(x))\n        x = self.dropout(x)\n#         print(x.shape)\n        x = F.relu(self.conv2(x))\n        x = self.dropout(x)\n#         print(x.shape)\n        x = F.relu(self.conv3(x))\n        x = self.dropout(x)\n#         print(x.shape)\n        x = (self.flatten(x))\n#         print(x.shape)\n#         x = x.view(-1, self.num_flat_features(x))\n#         print(x.shape)\n        x = self.fc1(x)\n        x = self.fc2(x)\n#         print(x)\n        x = self.softmax(x)\n#         print(x)\n        return x\n\n    def num_flat_features(self, x):\n        size = x.size()[1:]\n        num_features = 1\n        for s in size:\n            num_features *= s\n        return num_features","metadata":{"execution":{"iopub.status.busy":"2024-01-19T20:43:31.395842Z","iopub.execute_input":"2024-01-19T20:43:31.396492Z","iopub.status.idle":"2024-01-19T20:43:31.412655Z","shell.execute_reply.started":"2024-01-19T20:43:31.396441Z","shell.execute_reply":"2024-01-19T20:43:31.411275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create an instance of your custom dataset\ndataset = CustomDataset(dataframe=train)","metadata":{"execution":{"iopub.status.busy":"2024-01-19T20:34:46.585458Z","iopub.execute_input":"2024-01-19T20:34:46.586865Z","iopub.status.idle":"2024-01-19T20:34:46.592924Z","shell.execute_reply.started":"2024-01-19T20:34:46.586816Z","shell.execute_reply":"2024-01-19T20:34:46.591456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\nexpert_consensus = [TARS[train.iloc[i,8]] for i in range(len(train))]\n\nclass_sample_count = np.array(\n    [len(np.where(expert_consensus == t)[0]) for t in np.unique(expert_consensus)])\n\nweight = 1. / class_sample_count\nsamples_weight = np.array([weight[t] for t in expert_consensus])\nsamples_weight = torch.from_numpy(samples_weight)","metadata":{"execution":{"iopub.status.busy":"2024-01-19T20:34:46.927648Z","iopub.execute_input":"2024-01-19T20:34:46.928099Z","iopub.status.idle":"2024-01-19T20:34:51.099317Z","shell.execute_reply.started":"2024-01-19T20:34:46.928061Z","shell.execute_reply":"2024-01-19T20:34:51.097964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sampler = torch.utils.data.sampler.WeightedRandomSampler(samples_weight.type('torch.DoubleTensor'), len(samples_weight))","metadata":{"execution":{"iopub.status.busy":"2024-01-19T20:34:51.102046Z","iopub.execute_input":"2024-01-19T20:34:51.102549Z","iopub.status.idle":"2024-01-19T20:34:51.109083Z","shell.execute_reply.started":"2024-01-19T20:34:51.102504Z","shell.execute_reply":"2024-01-19T20:34:51.107752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")","metadata":{"execution":{"iopub.status.busy":"2024-01-19T20:34:51.110759Z","iopub.execute_input":"2024-01-19T20:34:51.111191Z","iopub.status.idle":"2024-01-19T20:34:51.119897Z","shell.execute_reply.started":"2024-01-19T20:34:51.111146Z","shell.execute_reply":"2024-01-19T20:34:51.118556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not os.path.exists('CNN1D_Model'):\n        os.makedirs('CNN1D_Model')","metadata":{"execution":{"iopub.status.busy":"2024-01-19T20:34:51.122045Z","iopub.execute_input":"2024-01-19T20:34:51.122607Z","iopub.status.idle":"2024-01-19T20:34:51.132301Z","shell.execute_reply.started":"2024-01-19T20:34:51.122573Z","shell.execute_reply":"2024-01-19T20:34:51.130946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a DataLoader to handle batching and shuffling\nbatch_size = 256\ntrain_dataloader = DataLoader(dataset, batch_size=batch_size, sampler=sampler)\n\n# Create an instance of the model\nmodel = CNN1D(in_channels=4).double()\n# Define KL Divergence loss\ncriterion = nn.KLDivLoss(reduction=\"batchmean\")\n# Define optimizer\noptimizer = optim.Adam(model.parameters(), lr=0.001)\nmodel.to(device)\nif IS_TRAINING:\n    model.train()    \n    epochs = 2\n    for epoch in range(epochs):\n        pbar = tqdm(train_dataloader)\n        for batch in pbar:\n            eeg_, label = batch\n            pred = model(eeg_.to(device))\n            loss = criterion(torch.log(pred), label.to(device))\n            # Backward pass and optimization\n            optimizer.zero_grad()\n            loss.backward()\n            optimizer.step()\n    #         stop\n\n        # Print loss for monitoring training progress\n            pbar.set_description('Batch loss{:.3f}'.format(loss.item()))\n#         print(f\"Epoch {epoch+1}/{epochs}, Loss: {loss.item()}\")\n\n    \n        torch.save(model.state_dict(), f'CNN1D_Model/model_{epoch}.pt')","metadata":{"execution":{"iopub.status.busy":"2024-01-19T20:43:49.717019Z","iopub.execute_input":"2024-01-19T20:43:49.717534Z","iopub.status.idle":"2024-01-19T20:51:05.646466Z","shell.execute_reply.started":"2024-01-19T20:43:49.717495Z","shell.execute_reply":"2024-01-19T20:51:05.645420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submit to Kaggle LB\n","metadata":{"papermill":{"duration":0.039743,"end_time":"2024-01-16T12:14:15.75862","exception":false,"start_time":"2024-01-16T12:14:15.718877","status":"completed"},"tags":[]}},{"cell_type":"code","source":"del train; gc.collect()\ntest = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape:',test.shape)\ntest.head()","metadata":{"papermill":{"duration":0.059958,"end_time":"2024-01-16T12:14:15.85819","exception":false,"start_time":"2024-01-16T12:14:15.798232","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-01-19T15:33:35.286818Z","iopub.execute_input":"2024-01-19T15:33:35.287182Z","iopub.status.idle":"2024-01-19T15:33:35.436726Z","shell.execute_reply.started":"2024-01-19T15:33:35.287142Z","shell.execute_reply":"2024-01-19T15:33:35.435846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_path = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\nclass CustomDataset_test(Dataset):\n    def __init__(self, dataframe, transform=None):\n        self.dataframe = dataframe\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, idx):\n        \n        row = self.dataframe.iloc[idx]\n        eeg_id = row['eeg_id']\n        parq_path = f'{test_path}{eeg_id}.parquet'\n        eeg = pd.read_parquet(parq_path)\n        rows = len(eeg)\n        offset = (rows-10_000)//2\n        eeg = eeg.iloc[offset:offset+10_000]\n        \n        signals = []\n        for k in range(4):\n            COLS = FEATS[k]\n\n            # COMPUTE PAIR DIFFERENCES AND AVERAGE\n            x = eeg[COLS[0]].values - eeg[COLS[1]].values\n            for j in range(3):\n                x += eeg[COLS[j+1]].values - eeg[COLS[j+2]].values\n            x /= 4.0\n            x = denoise_filter(x)\n            signals.append(x)\n        signals = np.array(signals)\n        \n        return torch.tensor(signals,dtype=torch.float64)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-01-19T15:34:43.119397Z","iopub.execute_input":"2024-01-19T15:34:43.120248Z","iopub.status.idle":"2024-01-19T15:34:43.129435Z","shell.execute_reply.started":"2024-01-19T15:34:43.120208Z","shell.execute_reply":"2024-01-19T15:34:43.128243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_test = CustomDataset_test(dataframe=test)","metadata":{"execution":{"iopub.status.busy":"2024-01-19T15:34:44.630228Z","iopub.execute_input":"2024-01-19T15:34:44.631056Z","iopub.status.idle":"2024-01-19T15:34:44.635011Z","shell.execute_reply.started":"2024-01-19T15:34:44.631024Z","shell.execute_reply":"2024-01-19T15:34:44.634014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_loader = DataLoader(dataset_test, batch_size=16,shuffle=False)\nmodel = CNN1D(in_channels=4).double()\nmodel.load_state_dict(torch.load('/kaggle/input/cnn-drop/model_1.pt'))\nmodel.eval()\nmodel.cpu()\npreds = []\n\nfor batch in test_loader:\n    pred = model(batch)\n    preds.append(pred.detach().numpy())\npreds = np.vstack(preds)","metadata":{"execution":{"iopub.status.busy":"2024-01-19T15:34:44.838058Z","iopub.execute_input":"2024-01-19T15:34:44.838429Z","iopub.status.idle":"2024-01-19T15:34:44.915328Z","shell.execute_reply.started":"2024-01-19T15:34:44.838399Z","shell.execute_reply":"2024-01-19T15:34:44.914355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CREATE SUBMISSION.CSV\nfrom IPython.display import display\n\nsub = pd.DataFrame({'eeg_id':test.eeg_id.values})\nsub[TARGETS] = preds\nsub.to_csv('submission.csv',index=False)\nprint('Submission shape',sub.shape)\ndisplay( sub.head() )\n\n# SANITY CHECK TO CONFIRM PREDICTIONS SUM TO ONE\nprint('Sub row 0 sums to:',sub.iloc[0,-6:].sum())","metadata":{"papermill":{"duration":0.07372,"end_time":"2024-01-16T12:14:18.05987","exception":false,"start_time":"2024-01-16T12:14:17.98615","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-01-19T15:34:55.938722Z","iopub.execute_input":"2024-01-19T15:34:55.939461Z","iopub.status.idle":"2024-01-19T15:34:55.960529Z","shell.execute_reply.started":"2024-01-19T15:34:55.939427Z","shell.execute_reply":"2024-01-19T15:34:55.959636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}