{"metadata":{"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7437371,"sourceType":"datasetVersion","datasetId":4328630},{"sourceId":7437573,"sourceType":"datasetVersion","datasetId":4328761},{"sourceId":161531087,"sourceType":"kernelVersion"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.12"},"papermill":{"default_parameters":{},"duration":102.485133,"end_time":"2024-01-16T12:14:21.10455","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-01-16T12:12:38.619417","version":"2.4.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#GROUP NOTEBOOK: https://www.kaggle.com/code/peluch/hms-hbac-kerascv-starter-notebook","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd, numpy as np, os\nimport matplotlib.pyplot as plt, gc\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom scipy.signal import butter, lfilter\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom scipy.signal import freqz\nfrom sklearn.model_selection import train_test_split\n\nimport torch as t\nfrom torch.utils.data import Dataset, DataLoader\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom tqdm import tqdm\n\nprint(\"done\")","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.693827,"end_time":"2024-01-16T12:12:42.606147","exception":false,"start_time":"2024-01-16T12:12:41.91232","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"#Get data\nALL_DATA_df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\n# Define target columns(votes for each activity pattern)\nTARGET_COLS = TRAIN.columns[-6:]\n\n#---- CELL OUTPUT ------------:\nprint('All data shape:', ALL_DATA_df.shape, '\\n' )\ndisplay(ALL_DATA_df.head() )\nprint(f\"\\nTarget columns:\")\nfor col in TARGET_COLS:\n    print(\"  -\",col)","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.693827,"end_time":"2024-01-16T12:12:42.606147","exception":false,"start_time":"2024-01-16T12:12:41.91232","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Dataloaders","metadata":{}},{"cell_type":"code","source":"#Create Dataset Class\n\nclass CustomDataset(Dataset):\n    def __init__(self, df):\n        self.df = df\n        \n    def len(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx, target_cols = TARGET_COLS):\n        #The below assumes the label is the last column in the df\n        row_series = self.df.iloc[idx]\n        feature_cols = [col for col in df.columns if col not in target_cols]\n        features_list = row_series[feature_cols].values\n        label = row[target_cols].values\n        return t.tensor(features_list, dtype=torch.float32), t.tensor(label, dtype=torch.float32)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create datasets (train, val)\n\nTRAIN_df, VAL_df = sklearn.train_test_split(ALL_DATA_df, test_size.1, randome_state = 1)\n\nTRAIN_ds = CustomDataset(TRAIN_df)\nVAL_ds = CustomDataset(VAL_df)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ORIG NOTEBOOK:","metadata":{}},{"cell_type":"code","source":"def butter_bandpass(lowcut, highcut, fs, order=5):\n    return butter(order, [lowcut, highcut], fs=fs, btype='band')\n\ndef butter_bandpass_filter(data, lowcut, highcut, fs, order=5):\n    b, a = butter_bandpass(lowcut, highcut, fs, order=order)\n    y = lfilter(b, a, data)\n    return y\n\n\ndef denoise_filter(x):\n    # Sample rate and desired cutoff frequencies (in Hz).\n    fs = 200.0\n    lowcut = 1.0\n    highcut = 25.0\n    \n    # Filter a noisy signal.\n    T = 50\n    nsamples = T * fs\n    t = np.arange(0, nsamples) / fs\n    y = butter_bandpass_filter(x, lowcut, highcut, fs, order=6)\n    y = (y + np.roll(y,-1)+ np.roll(y,-2)+ np.roll(y,-3))/4\n    y = y[0:-1:4]\n    \n    return y","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IS_TRAINING=False","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if IS_TRAINING:\n    eegs_data = np.load('/kaggle/input/hms-eeg-raw-dataset/eeg_specs.npy',allow_pickle=True).item()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataloader","metadata":{"papermill":{"duration":0.028035,"end_time":"2024-01-16T12:13:50.519598","exception":false,"start_time":"2024-01-16T12:13:50.491563","status":"completed"},"tags":[]}},{"cell_type":"code","source":"NAMES = ['LL','LP','RP','RR']\n\nFEATS = [['Fp1','F7','T3','T5','O1'],\n         ['Fp1','F3','C3','P3','O1'],\n         ['Fp2','F8','T4','T6','O2'],\n         ['Fp2','F4','C4','P4','O2']]\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\n\nclass CustomDataset(Dataset):\n    def __init__(self, dataframe, transform=None):\n        self.dataframe = dataframe\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, idx):\n        \n        row = self.dataframe.iloc[idx]\n        eeg_id = row['eeg_id']\n        eeg_sub_id = row['eeg_sub_id']\n        eeg_key = f'{eeg_id}_{eeg_sub_id}'\n        signals = eegs_data[eeg_key]\n        labels = row[TARGETS].values.astype(np.float64) #np.array(row[-6:]).reshape(6,1)\n        labels = labels/np.sum(labels)\n        return torch.tensor(signals,dtype=torch.float64), torch.tensor(labels,dtype=torch.float64)\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the 1D CNN model\nclass CNN1D(nn.Module):\n    def __init__(self,in_channels):\n        super(CNN1D, self).__init__()\n        self.hidden_channels = 128\n        self.conv1 = nn.Conv1d(in_channels, 64, 20, 10)\n        self.conv2 = nn.Conv1d(64, 64, 10, 5)\n        self.conv3 = nn.Conv1d(64, 64, 12, 4)\n#         self.conv3 = nn.Conv1d(32, 32, 98, 1)\n        self.flatten = nn.Flatten()\n#         self.pool = nn.MaxPool1d(kernel_size=2, stride=2)\n        self.fc1 = nn.Linear(640, 32)\n        self.fc2 = nn.Linear(32, 6)  # Adjust the input size based on your input dimensions\n        self.softmax = nn.Softmax(dim=1)\n        self.dropout = nn.Dropout(0.2)\n\n    def forward(self, x):\n#         print(x.shape)\n        x = F.relu(self.conv1(x))\n        x = self.dropout(x)\n#         print(x.shape)\n        x = F.relu(self.conv2(x))\n        x = self.dropout(x)\n#         print(x.shape)\n        x = F.relu(self.conv3(x))\n        x = self.dropout(x)\n#         print(x.shape)\n        x = (self.flatten(x))\n#         print(x.shape)\n#         x = x.view(-1, self.num_flat_features(x))\n#         print(x.shape)\n        x = self.fc1(x)\n        x = self.fc2(x)\n#         print(x)\n        x = self.softmax(x)\n#         print(x)\n        return x\n\n    def num_flat_features(self, x):\n        size = x.size()[1:]\n        num_features = 1\n        for s in size:\n            num_features *= s\n        return num_features","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create an instance of your custom dataset\ndataset = CustomDataset(dataframe=train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\nexpert_consensus = [TARS[train.iloc[i,8]] for i in range(len(train))]\n\nclass_sample_count = np.array(\n    [len(np.where(expert_consensus == t)[0]) for t in np.unique(expert_consensus)])\n\nweight = 1. / class_sample_count\nsamples_weight = np.array([weight[t] for t in expert_consensus])\nsamples_weight = torch.from_numpy(samples_weight)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sampler = torch.utils.data.sampler.WeightedRandomSampler(samples_weight.type('torch.DoubleTensor'), len(samples_weight))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not os.path.exists('CNN1D_Model'):\n        os.makedirs('CNN1D_Model')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a DataLoader to handle batching and shuffling\nbatch_size = 256\ntrain_dataloader = DataLoader(dataset, batch_size=batch_size, sampler=sampler)\n\n# Create an instance of the model\nmodel = CNN1D(in_channels=4).double()\n# Define KL Divergence loss\ncriterion = nn.KLDivLoss(reduction=\"batchmean\")\n# Define optimizer\noptimizer = optim.Adam(model.parameters(), lr=0.001)\nmodel.to(device)\nif IS_TRAINING:\n    model.train()    \n    epochs = 2\n    for epoch in range(epochs):\n        pbar = tqdm(train_dataloader)\n        for batch in pbar:\n            eeg_, label = batch\n            pred = model(eeg_.to(device))\n            loss = criterion(torch.log(pred), label.to(device))\n            # Backward pass and optimization\n            optimizer.zero_grad()\n            loss.backward()\n            optimizer.step()\n    #         stop\n\n        # Print loss for monitoring training progress\n            pbar.set_description('Batch loss{:.3f}'.format(loss.item()))\n#         print(f\"Epoch {epoch+1}/{epochs}, Loss: {loss.item()}\")\n\n    \n        torch.save(model.state_dict(), f'CNN1D_Model/model_{epoch}.pt')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submit to Kaggle LB\n","metadata":{"papermill":{"duration":0.039743,"end_time":"2024-01-16T12:14:15.75862","exception":false,"start_time":"2024-01-16T12:14:15.718877","status":"completed"},"tags":[]}},{"cell_type":"code","source":"del train; gc.collect()\ntest = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape:',test.shape)\ntest.head()","metadata":{"papermill":{"duration":0.059958,"end_time":"2024-01-16T12:14:15.85819","exception":false,"start_time":"2024-01-16T12:14:15.798232","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_path = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\nclass CustomDataset_test(Dataset):\n    def __init__(self, dataframe, transform=None):\n        self.dataframe = dataframe\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, idx):\n        \n        row = self.dataframe.iloc[idx]\n        eeg_id = row['eeg_id']\n        parq_path = f'{test_path}{eeg_id}.parquet'\n        eeg = pd.read_parquet(parq_path)\n        rows = len(eeg)\n        offset = (rows-10_000)//2\n        eeg = eeg.iloc[offset:offset+10_000]\n        \n        signals = []\n        for k in range(4):\n            COLS = FEATS[k]\n\n            # COMPUTE PAIR DIFFERENCES AND AVERAGE\n            x = eeg[COLS[0]].values - eeg[COLS[1]].values\n            for j in range(3):\n                x += eeg[COLS[j+1]].values - eeg[COLS[j+2]].values\n            x /= 4.0\n            x = denoise_filter(x)\n            signals.append(x)\n        signals = np.array(signals)\n        \n        return torch.tensor(signals,dtype=torch.float64)\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_test = CustomDataset_test(dataframe=test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_loader = DataLoader(dataset_test, batch_size=16,shuffle=False)\nmodel = CNN1D(in_channels=4).double()\nmodel.load_state_dict(torch.load('/kaggle/input/cnn-drop/model_1.pt'))\nmodel.eval()\nmodel.cpu()\npreds = []\n\nfor batch in test_loader:\n    pred = model(batch)\n    preds.append(pred.detach().numpy())\npreds = np.vstack(preds)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CREATE SUBMISSION.CSV\nfrom IPython.display import display\n\nsub = pd.DataFrame({'eeg_id':test.eeg_id.values})\nsub[TARGETS] = preds\nsub.to_csv('submission.csv',index=False)\nprint('Submission shape',sub.shape)\ndisplay( sub.head() )\n\n# SANITY CHECK TO CONFIRM PREDICTIONS SUM TO ONE\nprint('Sub row 0 sums to:',sub.iloc[0,-6:].sum())","metadata":{"papermill":{"duration":0.07372,"end_time":"2024-01-16T12:14:18.05987","exception":false,"start_time":"2024-01-16T12:14:17.98615","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}