{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":165784130,"sourceType":"kernelVersion"}],"dockerImageVersionId":30664,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport os\nimport glob\nfrom pathlib import Path\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import preprocessing\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as transforms\nimport torchvision.models as models\nfrom PIL import Image\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-22T06:16:12.290273Z","iopub.execute_input":"2024-03-22T06:16:12.290685Z","iopub.status.idle":"2024-03-22T06:16:19.135417Z","shell.execute_reply.started":"2024-03-22T06:16:12.290654Z","shell.execute_reply":"2024-03-22T06:16:19.134626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration and Hyperparameters","metadata":{}},{"cell_type":"code","source":"INPUT_IMAGE_SIZE = (400, 400)\nTRAIN_SIZE = 0.9\nBATCH_SIZE = 16\nLEARNING_RATE = 1e-3\nEPOCHS = 25\nSEED = 42\nDEVICE = 'cuda' if torch.cuda.is_available() else 'cpu'\nCLASSES = ['Seizure', 'LPD', 'GPD', 'LRDA','GRDA', 'Other']\nprint(f'using device: {DEVICE}')","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:16:19.136885Z","iopub.execute_input":"2024-03-22T06:16:19.137287Z","iopub.status.idle":"2024-03-22T06:16:19.168694Z","shell.execute_reply.started":"2024-03-22T06:16:19.137261Z","shell.execute_reply":"2024-03-22T06:16:19.167850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reproducibility","metadata":{}},{"cell_type":"code","source":"torch.manual_seed(SEED)","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:16:19.169789Z","iopub.execute_input":"2024-03-22T06:16:19.170142Z","iopub.status.idle":"2024-03-22T06:16:19.180907Z","shell.execute_reply.started":"2024-03-22T06:16:19.170108Z","shell.execute_reply":"2024-03-22T06:16:19.180101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing Data Files","metadata":{}},{"cell_type":"code","source":"train_eeg_files = glob.glob('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/*')\ntest_eeg_files = glob.glob('/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/*')\ntrain_spectrogram_files = glob.glob('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/*')\ntest_spectrogram_files = glob.glob('/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/*')\nprint(f'Total number of files in train_eegs: {len(train_eeg_files)}')\nprint(f'Total number of files in test_eegs: {len(test_eeg_files)}')\nprint(f'Total number of files in train_spectrograms: {len(train_spectrogram_files)}')\nprint(f'Total number of files in test_spectrograms: {len(test_spectrogram_files)}')","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:16:19.710220Z","iopub.execute_input":"2024-03-22T06:16:19.710488Z","iopub.status.idle":"2024-03-22T06:16:20.319256Z","shell.execute_reply.started":"2024-03-22T06:16:19.710464Z","shell.execute_reply":"2024-03-22T06:16:20.318258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ntest_csv = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nsample_csv = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv')\nprint(f'number of entries in train_csv {len(train_csv)}')\nprint(f'number of entries in test_csv {len(test_csv)}')\nprint(f'number of entries in sample_csv {len(sample_csv)}')","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:16:21.626309Z","iopub.execute_input":"2024-03-22T06:16:21.626652Z","iopub.status.idle":"2024-03-22T06:16:21.901051Z","shell.execute_reply.started":"2024-03-22T06:16:21.626625Z","shell.execute_reply":"2024-03-22T06:16:21.900031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating Spectrogram Dataset","metadata":{}},{"cell_type":"code","source":"columns_to_drop = ['eeg_id', \n                   'eeg_sub_id', \n                   'eeg_label_offset_seconds', \n                   'spectrogram_sub_id', \n                   'spectrogram_label_offset_seconds',\n                   'label_id',\n                   'patient_id',\n                   'expert_consensus']","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:16:24.182163Z","iopub.execute_input":"2024-03-22T06:16:24.182508Z","iopub.status.idle":"2024-03-22T06:16:24.187219Z","shell.execute_reply.started":"2024-03-22T06:16:24.182479Z","shell.execute_reply":"2024-03-22T06:16:24.186124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spectro_csv = train_csv[train_csv['spectrogram_sub_id']==0].drop(columns_to_drop, axis=1).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:16:25.967326Z","iopub.execute_input":"2024-03-22T06:16:25.967703Z","iopub.status.idle":"2024-03-22T06:16:25.984235Z","shell.execute_reply.started":"2024-03-22T06:16:25.967671Z","shell.execute_reply":"2024-03-22T06:16:25.983327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class SquarePad:\n    def __call__(self, image):\n        max_wh = max(image.size)\n        p_left, p_top = [(max_wh - s) // 2 for s in image.size]\n        p_right, p_bottom = [max_wh - (s+pad) for s, pad in zip(image.size, [p_left, p_top])]\n        padding = (p_left, p_top, p_right, p_bottom)\n        return transforms.functional.pad(image, padding, 0, 'constant')","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:16:31.719195Z","iopub.execute_input":"2024-03-22T06:16:31.719902Z","iopub.status.idle":"2024-03-22T06:16:31.725744Z","shell.execute_reply.started":"2024-03-22T06:16:31.719872Z","shell.execute_reply":"2024-03-22T06:16:31.724840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class HMS_dataset(Dataset):\n    def __init__(self,csv_file, root_dir, image_size = (400,400), targets_available=True):\n        self.hms_dataset = csv_file\n        self.root_dir = root_dir\n        self.image_size = image_size\n        self.targets_available = targets_available\n        self.transform = transforms.Compose([\n                            SquarePad(),\n                            transforms.Resize(image_size),\n                            transforms.ToTensor(),\n                            ])\n        \n    def __len__(self):\n        return len(self.hms_dataset)\n    \n    def __getitem__(self, idx):\n        if torch.is_tensor(idx):\n            idx = idx.tolist()\n        \n        file_name = os.path.join(self.root_dir, str(self.hms_dataset.iloc[idx, 0]) + '.parquet')\n        spectro = pd.read_parquet(file_name).drop('time', axis=1).fillna(0).to_numpy().T\n        spectro = np.clip(spectro, a_min=1e-6, a_max=np.inf)\n        spectro = 10*np.log10(spectro/np.max(spectro))\n        spectro = spectro/np.min(spectro)\n        spectro_pil = Image.fromarray(spectro)\n        spectro_torch = self.transform(spectro_pil)\n        if self.targets_available:\n            votes_label = self.hms_dataset.iloc[idx, 1:].to_numpy()\n            torch_labels = torch.tensor(votes_label, dtype=torch.float)\n            sample = (spectro_torch, torch_labels)\n        else:\n            sample = spectro_torch\n            \n        return sample","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:16:32.066457Z","iopub.execute_input":"2024-03-22T06:16:32.066780Z","iopub.status.idle":"2024-03-22T06:16:32.076747Z","shell.execute_reply.started":"2024-03-22T06:16:32.066754Z","shell.execute_reply":"2024-03-22T06:16:32.075864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading Dataset","metadata":{}},{"cell_type":"code","source":"root_train = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nhms_data = HMS_dataset(spectro_csv,root_train, INPUT_IMAGE_SIZE)\nprint(f'size of the whole data: {len(hms_data)}')","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:16:34.783245Z","iopub.execute_input":"2024-03-22T06:16:34.783606Z","iopub.status.idle":"2024-03-22T06:16:34.788905Z","shell.execute_reply.started":"2024-03-22T06:16:34.783578Z","shell.execute_reply":"2024-03-22T06:16:34.788001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing Input Images","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(14, 7))\n\nfor i in range(len(hms_data)):\n    signal = hms_data[i][0]\n    labels = hms_data[i][1]\n    if i == 9:\n        break\n    plt.subplot(3, 3, i+1)\n    plt.tight_layout()\n    plt.imshow(signal.squeeze(0).numpy())\n    plt.title(CLASSES[np.argmax(labels.numpy())])","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:16:36.928275Z","iopub.execute_input":"2024-03-22T06:16:36.928607Z","iopub.status.idle":"2024-03-22T06:16:40.597523Z","shell.execute_reply.started":"2024-03-22T06:16:36.928582Z","shell.execute_reply":"2024-03-22T06:16:40.596642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Splitting Data","metadata":{}},{"cell_type":"code","source":"train_size = int(TRAIN_SIZE * len(hms_data))\ntest_size = len(hms_data) - train_size\ntrain_set, test_set = torch.utils.data.random_split(hms_data, [train_size, test_size])\nprint(f'size of the training data: {len(train_set)}')\nprint(f'size of the test data: {len(test_set)}')","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:16:44.879649Z","iopub.execute_input":"2024-03-22T06:16:44.880387Z","iopub.status.idle":"2024-03-22T06:16:44.893821Z","shell.execute_reply.started":"2024-03-22T06:16:44.880353Z","shell.execute_reply":"2024-03-22T06:16:44.892893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_loader = DataLoader(dataset=train_set, batch_size=BATCH_SIZE, shuffle=True)\ntest_loader = DataLoader(dataset=test_set, batch_size=BATCH_SIZE, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:16:45.638275Z","iopub.execute_input":"2024-03-22T06:16:45.639101Z","iopub.status.idle":"2024-03-22T06:16:45.643807Z","shell.execute_reply.started":"2024-03-22T06:16:45.639067Z","shell.execute_reply":"2024-03-22T06:16:45.642881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating Model","metadata":{}},{"cell_type":"code","source":"model = models.efficientnet_v2_s()","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:16:48.496360Z","iopub.execute_input":"2024-03-22T06:16:48.497173Z","iopub.status.idle":"2024-03-22T06:16:48.966223Z","shell.execute_reply.started":"2024-03-22T06:16:48.497141Z","shell.execute_reply":"2024-03-22T06:16:48.965194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.features[0][0] = nn.Conv2d(1, 24, kernel_size=(3, 3), stride=(2, 2), padding=(1, 1), bias=False)\nmodel.classifier[1] = nn.Linear(in_features=1280, out_features=6, bias=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:16:49.416460Z","iopub.execute_input":"2024-03-22T06:16:49.416836Z","iopub.status.idle":"2024-03-22T06:16:49.423274Z","shell.execute_reply.started":"2024-03-22T06:16:49.416787Z","shell.execute_reply":"2024-03-22T06:16:49.422269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.to(DEVICE)","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:16:50.016041Z","iopub.execute_input":"2024-03-22T06:16:50.016836Z","iopub.status.idle":"2024-03-22T06:16:50.230689Z","shell.execute_reply.started":"2024-03-22T06:16:50.016781Z","shell.execute_reply":"2024-03-22T06:16:50.229857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"optimizer = torch.optim.Adam(model.parameters(), lr = LEARNING_RATE)\nloss_fn = nn.KLDivLoss(reduction='batchmean')","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:16:56.554355Z","iopub.execute_input":"2024-03-22T06:16:56.555179Z","iopub.status.idle":"2024-03-22T06:16:56.565937Z","shell.execute_reply.started":"2024-03-22T06:16:56.555148Z","shell.execute_reply":"2024-03-22T06:16:56.564789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train(model, optimizer, loss_fn, data_loader):\n    total_loss = 0\n    data_len = len(data_loader)\n    \n    for x,y in tqdm(data_loader, desc='(Training)', ncols=100):\n        preds = F.log_softmax(model(x.to(DEVICE)), dim=1)\n        y = F.softmax(y, dim=1).to(DEVICE)\n        loss = loss_fn(preds, y)\n        total_loss += loss.item()\n        \n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n        \n    \n    return (total_loss/data_len), loss\n\ndef val(model, loss_fn, data_loader):\n    total_loss = 0\n    data_len = len(data_loader)\n    \n    for x,y in tqdm(data_loader, desc='(Testing)', ncols=100):\n        with torch.no_grad():\n            preds = F.log_softmax(model(x.to(DEVICE)), dim=1)\n            y = F.softmax(y, dim=1).to(DEVICE)\n            loss = loss_fn(preds, y)\n            total_loss += loss.item()\n    \n    return (total_loss/data_len)","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:16:58.775351Z","iopub.execute_input":"2024-03-22T06:16:58.775700Z","iopub.status.idle":"2024-03-22T06:16:58.784936Z","shell.execute_reply.started":"2024-03-22T06:16:58.775673Z","shell.execute_reply":"2024-03-22T06:16:58.784057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for e in range(EPOCHS):\n    print(f'epoch {e+1}\\n')\n    avg_loss, _ = train(model,optimizer,loss_fn, train_loader)\n    if torch.cuda.is_available():\n        torch.cuda.empty_cache()\n    print(f'Training loss: {avg_loss:.3f}')\n    test_results = val(model, loss_fn, test_loader)\n    print(f'Test loss: {test_results:.3f}\\n')","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:17:02.632473Z","iopub.execute_input":"2024-03-22T06:17:02.632874Z","iopub.status.idle":"2024-03-22T06:18:12.718830Z","shell.execute_reply.started":"2024-03-22T06:17:02.632841Z","shell.execute_reply":"2024-03-22T06:18:12.717002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predicting Test Data","metadata":{}},{"cell_type":"code","source":"spectro_test_csv = test_csv.drop(['eeg_id', 'patient_id'], axis=1)\nroot_test = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\ntest_hms_data = HMS_dataset(spectro_test_csv, root_test, targets_available=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:18:17.590349Z","iopub.execute_input":"2024-03-22T06:18:17.590681Z","iopub.status.idle":"2024-03-22T06:18:17.596731Z","shell.execute_reply.started":"2024-03-22T06:18:17.590656Z","shell.execute_reply":"2024-03-22T06:18:17.595649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(test_hms_data[0].squeeze(0))","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:18:18.867317Z","iopub.execute_input":"2024-03-22T06:18:18.867664Z","iopub.status.idle":"2024-03-22T06:18:19.190397Z","shell.execute_reply.started":"2024-03-22T06:18:18.867637Z","shell.execute_reply":"2024-03-22T06:18:19.189468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with torch.no_grad():\n    test_preds = F.softmax(model(test_hms_data[0].unsqueeze_(0).to(DEVICE)), dim=1).detach().cpu().numpy()","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:18:19.557987Z","iopub.execute_input":"2024-03-22T06:18:19.558349Z","iopub.status.idle":"2024-03-22T06:18:19.689168Z","shell.execute_reply.started":"2024-03-22T06:18:19.558322Z","shell.execute_reply":"2024-03-22T06:18:19.688335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:18:20.304433Z","iopub.execute_input":"2024-03-22T06:18:20.305097Z","iopub.status.idle":"2024-03-22T06:18:20.311761Z","shell.execute_reply.started":"2024-03-22T06:18:20.305054Z","shell.execute_reply":"2024-03-22T06:18:20.310681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds.sum()","metadata":{"execution":{"iopub.status.busy":"2024-03-22T06:18:23.034628Z","iopub.execute_input":"2024-03-22T06:18:23.034979Z","iopub.status.idle":"2024-03-22T06:18:23.041379Z","shell.execute_reply.started":"2024-03-22T06:18:23.034952Z","shell.execute_reply":"2024-03-22T06:18:23.040512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submitting","metadata":{}},{"cell_type":"code","source":"i=0\nfor col in sample_csv.columns[1:]:\n    sample_csv[col] = test_preds[0][i]\n    i+=1","metadata":{"execution":{"iopub.status.busy":"2024-03-20T08:18:17.653978Z","iopub.execute_input":"2024-03-20T08:18:17.654326Z","iopub.status.idle":"2024-03-20T08:18:17.660710Z","shell.execute_reply.started":"2024-03-20T08:18:17.654300Z","shell.execute_reply":"2024-03-20T08:18:17.659744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_csv","metadata":{"execution":{"iopub.status.busy":"2024-03-20T08:18:18.943171Z","iopub.execute_input":"2024-03-20T08:18:18.943519Z","iopub.status.idle":"2024-03-20T08:18:18.958005Z","shell.execute_reply.started":"2024-03-20T08:18:18.943490Z","shell.execute_reply":"2024-03-20T08:18:18.957005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_csv.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2024-03-19T13:31:59.013503Z","iopub.execute_input":"2024-03-19T13:31:59.013783Z","iopub.status.idle":"2024-03-19T13:31:59.021433Z","shell.execute_reply.started":"2024-03-19T13:31:59.013760Z","shell.execute_reply":"2024-03-19T13:31:59.020522Z"},"trusted":true},"execution_count":null,"outputs":[]}]}