{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os, sys\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nsys.path.append(\"/kaggle/working/code\")\n\nTRAIN_EEG_DIR = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs'\nTRAIN_SPC_DIR = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms'\nTRAIN_FILE = '/kaggle/input/hms-harmful-brain-activity-classification/train.csv'\nCODE_DIR = '/kaggle/working/code'\nPROCESSED_DIR = \"/kaggle/working/preprocessed\"\nPROCESSED_SPC_DIR = \"/kaggle/working/preprocessed/spectrogram\"\n\nos.makedirs(PROCESSED_DIR, exist_ok=True)\nos.makedirs(PROCESSED_SPC_DIR, exist_ok=True)\n\nif not os.path.exists(CODE_DIR):\n    os.makedirs(CODE_DIR)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-16T11:59:14.848863Z","iopub.execute_input":"2024-02-16T11:59:14.849356Z","iopub.status.idle":"2024-02-16T11:59:14.856447Z","shell.execute_reply.started":"2024-02-16T11:59:14.849321Z","shell.execute_reply":"2024-02-16T11:59:14.855503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile {CODE_DIR}/preprocess_train.py\n\nimport os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport librosa\n\nTRAIN_EEG_DIR = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs'\nTRAIN_SPC_DIR = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms'\nPROCESSED_SPC_DIR = \"/kaggle/working/preprocessed/spectrogram\"\n\nclass TrainDataProcessor():\n    def __init__(self):\n        self.eeg_files = [f\"{TRAIN_EEG_DIR}/{file}\" for file in os.listdir(TRAIN_EEG_DIR)]\n        self.spectrogram_files =  [f\"{TRAIN_SPC_DIR}/{file}\" for file in os.listdir(TRAIN_SPC_DIR)]\n        self.vote_cols = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']    \n        return\n\n#     def delete_unnecesary_cols_train(self, df):\n#         del df['spectrogram_id']\n#         del df['eeg_sub_id']\n#         del df['spectrogram_sub_id']\n#         del df['label_id']\n#         del df['expert_consensus']\n#         del df['eeg_label_offset_seconds']\n#         del df['spectrogram_label_offset_seconds']\n#         return\n        \n    def filter_rows_without_data_train(self, df):\n        df = df[df['eeg_file'].isin(self.eeg_files)]\n        df = df[df['spectrogram_file'].isin(self.spectrogram_files)]\n        df.dropna(subset=['eeg_label_offset_seconds', 'spectrogram_label_offset_seconds'], inplace = True)\n        return\n        \n    def get_spectrogram_interval(self, df):\n        df['spec_start_idx'] = df['spectrogram_label_offset_seconds']//2\n        df['spec_end_idx'] = df['spectrogram_label_offset_seconds']//2 + 300\n        df['spec_start_idx'] = df['spec_start_idx'].astype('int')\n        df['spec_end_idx'] = df['spec_end_idx'].astype('int')\n    \n    def preprocess_train_df(self, df):\n        df['eeg_file'] = TRAIN_EEG_DIR + '/' + df['eeg_id'].astype(str) + '.parquet'\n        df['spectrogram_file'] = PROCESSED_SPC_DIR + '/' + df['spectrogram_id'].astype(str) + '.pt'\n        \n        self.get_spectrogram_interval(df)\n        self.filter_rows_without_data_train(df)\n        #self.delete_unnecesary_cols_train(df)   \n        return df\n    \n    def remove_nan_from_signal(self, df):\n        f_df = df.copy().ffill()\n        f_df.fillna(0, inplace = True)\n        b_df = df.copy().bfill()\n        b_df.fillna(0, inplace = True)\n        return (f_df + b_df)/2","metadata":{"execution":{"iopub.status.busy":"2024-02-16T11:59:18.163760Z","iopub.execute_input":"2024-02-16T11:59:18.164232Z","iopub.status.idle":"2024-02-16T11:59:18.172500Z","shell.execute_reply.started":"2024-02-16T11:59:18.164198Z","shell.execute_reply":"2024-02-16T11:59:18.170797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile {CODE_DIR}/preprocess_specs.py\n\nimport os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport librosa\nimport torch\nfrom tqdm.auto import tqdm\n\nPROCESSED_SPC_DIR = \"/kaggle/working/preprocessed/spectrogram\"\n\nclass SpectrogramDatasetBuilder: \n    def remove_nan_from_signal(self, df):\n        f_df = df.copy().ffill()\n        f_df.fillna(0, inplace = True)\n        b_df = df.copy().bfill()\n        b_df.fillna(0, inplace = True)\n        return (f_df + b_df)/2\n    \n    def specToDb(self, spc):\n        sgram = librosa.amplitude_to_db(spc, ref=np.min)\n        return spc\n    \n    def preprocess_spec(self, spec_df):\n        spec_df = self.remove_nan_from_signal(spec_df)\n        spc = spec_df.values\n        spc = self.specToDb(spc)\n        return spc\n    \n    def buildAll(self, SPC_DIR):\n        for filename in tqdm(os.listdir(SPC_DIR)):\n            \n            if not filename.endswith('.parquet'):\n                continue\n            \n            filepath = os.path.join(SPC_DIR, filename)\n            spc = pd.read_parquet(filepath)\n            \n            x1,x2,x3,x4 = spc.iloc[:,1:101], spc.iloc[:,101:201], spc.iloc[:,201:301], spc.iloc[:,301:401]\n            x = [x1,x2,x3,x4]\n            for i in range(len(x)):\n                x[i] = self.preprocess_spec(x[i])\n                x[i] = torch.FloatTensor(x[i])\n\n            x = torch.stack(x)\n            x = x.permute([1,2,0])\n            target_file = filename.replace('parquet', 'pt')\n            target_path = os.path.join(PROCESSED_SPC_DIR, target_file)\n            torch.save(x, target_path)   ","metadata":{"execution":{"iopub.status.busy":"2024-02-16T11:59:22.271868Z","iopub.execute_input":"2024-02-16T11:59:22.272255Z","iopub.status.idle":"2024-02-16T11:59:22.280447Z","shell.execute_reply.started":"2024-02-16T11:59:22.272221Z","shell.execute_reply":"2024-02-16T11:59:22.279080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\n\nfrom preprocess_train import TrainDataProcessor\ndata_preprocessor = TrainDataProcessor()\n\ntrain_df = pd.read_csv(TRAIN_FILE)\ntrain_df = data_preprocessor.preprocess_train_df(train_df)\ntrain_df.to_csv(PROCESSED_DIR + '/train.csv')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-16T11:59:26.031265Z","iopub.execute_input":"2024-02-16T11:59:26.031796Z","iopub.status.idle":"2024-02-16T11:59:26.337288Z","shell.execute_reply.started":"2024-02-16T11:59:26.031750Z","shell.execute_reply":"2024-02-16T11:59:26.335720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from preprocess_specs import SpectrogramDatasetBuilder\nspectrogram_builder = SpectrogramDatasetBuilder()\nspectrogram_builder.buildAll(TRAIN_SPC_DIR)","metadata":{"execution":{"iopub.status.busy":"2024-02-16T11:59:58.763761Z","iopub.execute_input":"2024-02-16T11:59:58.764201Z","iopub.status.idle":"2024-02-16T11:59:59.170207Z","shell.execute_reply.started":"2024-02-16T11:59:58.764172Z","shell.execute_reply":"2024-02-16T11:59:59.168964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nsample_file = os.listdir(PROCESSED_SPC_DIR)[0]\nfilepath = os.path.join(PROCESSED_SPC_DIR, sample_file)\n\nsample = torch.load(filepath)\nsample = sample[:300]\nsample.shape","metadata":{"execution":{"iopub.status.busy":"2024-02-16T12:03:28.066033Z","iopub.execute_input":"2024-02-16T12:03:28.066554Z","iopub.status.idle":"2024-02-16T12:03:28.077238Z","shell.execute_reply.started":"2024-02-16T12:03:28.066514Z","shell.execute_reply":"2024-02-16T12:03:28.075641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nimport numpy as np\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\n\n# Define custom dataset\nclass SpectrogramDataset(Dataset):\n    def __init__(self, train_df):\n        self.train = train_df     \n\n    def __len__(self):\n        return len(self.train)\n\n    def __getitem__(self, idx):\n        spc_file = self.train.spectrogram_file[idx]\n        spc = torch.load(spc_file)\n        \n        spc_start = self.train.spec_start_idx[idx]\n        spc_end = min(self.train.spec_end_idx[idx], len(spc))\n        spc = spc[spc_start:spc_end]\n        spc = spc.permute(1,2,0)\n        return spc, y\n        \n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-02-16T12:03:24.020122Z","iopub.execute_input":"2024-02-16T12:03:24.020640Z","iopub.status.idle":"2024-02-16T12:03:24.033573Z","shell.execute_reply.started":"2024-02-16T12:03:24.020600Z","shell.execute_reply":"2024-02-16T12:03:24.030948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from tqdm.auto import tqdm\n# specDataset = SpectrogramDataset(train_df)\n\n# for x,y in tqdm(specDataset):\n#     print(x.shape)\n#     break","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nclass GRUClassifier(nn.Module):\n    def __init__(self, input_size, hidden_size, output_size):\n        super(GRUClassifier, self).__init__()\n        self.hidden_size = hidden_size\n        self.gru = nn.GRU(input_size, hidden_size, batch_first=True)\n        self.fc = nn.Linear(hidden_size, output_size)\n        self.softmax = nn.LogSoftmax(dim=1)\n\n    def forward(self, input):\n        _, hidden = self.gru(input)\n        output = self.fc(hidden[-1])\n        output = self.softmax(output)\n        return output\n\n# Example usage\ninput_size = 10\nhidden_size = 20\noutput_size = 5\n\nmodel = GRUClassifier(input_size, hidden_size, output_size)\n\n# Define optimizer\noptimizer = optim.Adam(model.parameters(), lr=0.001)\n\n# Example data\ninput_data = torch.randn(1, 3, input_size)  # (batch_size, sequence_length, input_size)\ntarget_data = torch.tensor([[0.2, 0.3, 0.1, 0.2, 0.2]])  # Target probability distribution\n\n# Training loop\noptimizer.zero_grad()\noutput = model(input_data)\noutput = output.unsqueeze(0)  # Add batch dimension to output\nloss = nn.KLDivLoss()(output, target_data)\nloss.backward()\noptimizer.step()\n\nprint(\"Loss:\", loss.item())\n\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}