{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7527876,"sourceType":"datasetVersion","datasetId":4384638},{"sourceId":7618462,"sourceType":"datasetVersion","datasetId":4437272}],"dockerImageVersionId":30636,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# UPDATE\n\nV10:Data preprocessing has been added, resolving the issue of loss convergence.\n\nV12:Added L2 regularization and filtering","metadata":{}},{"cell_type":"code","source":"import os,gc\nos.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"\nimport pandas as pd, numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.utils import to_categorical\nimport matplotlib.pyplot as plt\nimport glob\nimport torch\nimport warnings\n\n# disable warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-02-13T14:23:34.713457Z","iopub.execute_input":"2024-02-13T14:23:34.714340Z","iopub.status.idle":"2024-02-13T14:23:34.720209Z","shell.execute_reply.started":"2024-02-13T14:23:34.714306Z","shell.execute_reply":"2024-02-13T14:23:34.719295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Train data","metadata":{}},{"cell_type":"code","source":"PATH = '/kaggle/input/hms-harmful-brain-activity-classification/'\ndf = pd.read_csv(PATH + 'train.csv')\nTARGETS = df.columns[-6:]\nprint('Train shape:', df.shape )\nprint('Targets', list(TARGETS))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-13T14:23:34.737084Z","iopub.execute_input":"2024-02-13T14:23:34.737329Z","iopub.status.idle":"2024-02-13T14:23:34.905675Z","shell.execute_reply.started":"2024-02-13T14:23:34.737308Z","shell.execute_reply":"2024-02-13T14:23:34.904699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Non-Overlaping EEG Id Train Data","metadata":{}},{"cell_type":"code","source":"train = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spectrogram_id','min']\n\ntmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp\n\ntmp = df.groupby('eeg_id')[['patient_id']].agg('first')\ntrain['patient_id'] = tmp\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n    \ny_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\ntrain = train.reset_index()\nprint('Train non-overlap eeg_id shape:', train.shape )\n\ntrain.head(20)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T14:23:34.907097Z","iopub.execute_input":"2024-02-13T14:23:34.907352Z","iopub.status.idle":"2024-02-13T14:23:34.991230Z","shell.execute_reply.started":"2024-02-13T14:23:34.907330Z","shell.execute_reply":"2024-02-13T14:23:34.990305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\nycol = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\ncd = {'Seizure': 'seizure_vote', 'GPD': 'gpd_vote', 'LRDA': 'lrda_vote', 'Other': 'other_vote', 'GRDA': 'grda_vote', 'LPD': 'lpd_vote'}\n\n# Extract probability column and label column\neeg_id_col = train.iloc[:, 0]     # The first column is the eeg_id column\nprob_cols = train.iloc[:, -7:-1]  # The seventh to last column is the probability column\nlabel_col = train.iloc[:, -1]     # The last column is the label column\n\n# Convert probability column to float32 \nprob_cols = prob_cols.astype(\"float32\")\n\n# Normalized probability column\nprob_cols_normalized = prob_cols.div(prob_cols.sum(axis=1), axis=0)\n\n# Reassemble into DataFrame\nnormalized_data = pd.concat([eeg_id_col, prob_cols_normalized, label_col], axis=1)\n\nnormalized_data.head(20)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T14:23:34.992390Z","iopub.execute_input":"2024-02-13T14:23:34.992721Z","iopub.status.idle":"2024-02-13T14:23:35.020483Z","shell.execute_reply.started":"2024-02-13T14:23:34.992693Z","shell.execute_reply":"2024-02-13T14:23:35.019660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normalized_data.to_csv(\"normalized_data.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T14:23:35.022528Z","iopub.execute_input":"2024-02-13T14:23:35.022817Z","iopub.status.idle":"2024-02-13T14:23:35.139478Z","shell.execute_reply.started":"2024-02-13T14:23:35.022793Z","shell.execute_reply":"2024-02-13T14:23:35.138550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EEG_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\ntrain_path = '/kaggle/input/normalized-data/normalized_data.csv'","metadata":{"execution":{"iopub.status.busy":"2024-02-13T14:23:35.140511Z","iopub.execute_input":"2024-02-13T14:23:35.140812Z","iopub.status.idle":"2024-02-13T14:23:35.144917Z","shell.execute_reply.started":"2024-02-13T14:23:35.140788Z","shell.execute_reply":"2024-02-13T14:23:35.143993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport numpy as np\nimport pandas as pd\nimport glob\nfrom scipy.signal import butter, sosfilt\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.impute import SimpleImputer\n\nimport pandas as pd\n\nclass EEGDataset(Dataset):\n    def __init__(self, csv_file, eeg_path):\n        self.csv = pd.read_csv(csv_file)  # Load CSV file\n        self.eeg_path = eeg_path\n        self.sos = self.butter_bandpass_filter_init()\n        self.FEATS = ['Fp1','T3','C3','O1','Fp2','C4','T4','O2']  # Specify the feature columns to be used\n\n    def butter_bandpass_filter_init(self):\n        lowcut = 0.5    # Set the low-frequency cutoff frequency of the bandpass filter\n        highcut = 40.0  # Set the high-frequency cutoff frequency of the bandpass filter\n        freq = 200.0    # Sampling frequency\n        order = 5       # Filter order\n\n        nyq = 0.5 * freq\n        low = lowcut / nyq\n        high = highcut / nyq\n        sos = butter(order, [low, high], analog=False, btype='band', output='sos')\n        return sos\n\n    def butter_bandpass_filter(self, data):\n        y = sosfilt(self.sos, data)\n        return y\n    \n    def __len__(self):\n        return len(self.csv)\n    \n    def __getitem__(self, idx):\n        eeg_id = self.csv.loc[idx, 'eeg_id']\n        eeg_file_path = f\"{self.eeg_path}{eeg_id}.parquet\" # Path to build EEG data files\n        # Load specified feature columns from Parquet file\n        eeg_data = pd.read_parquet(eeg_file_path)[self.FEATS].values \n        # Check if has NaN values\n        if np.isnan(eeg_data).any():\n            # If a NaN value is present, you can choose to fill it with a specific value or interpolate it\n            # Here we use SimpleImputer for simple filling processing, replacing NaN values with mean values\n            imputer = SimpleImputer(strategy='mean')\n            eeg_data = imputer.fit_transform(eeg_data)\n        eeg_data = self.butter_bandpass_filter(eeg_data)        # Filter the data\n        eeg_data = torch.tensor(eeg_data, dtype=torch.float32)  # Convert to PyTorch tensor\n                    \n        # Select data from the middle 10,000 time points\n        mid_index = eeg_data.shape[0] // 2\n        start_index = mid_index - 5000  # Shift 5000 time points to the left from the center\n        end_index = mid_index + 5000  # Shift 5000 time points to the right from the center\n        eeg_data = eeg_data[start_index:end_index]\n        # Swap dimension positions\n        eeg_data = torch.transpose(eeg_data, 0, 1)\n        \n        # Load the corresponding tag\n        labels = torch.tensor(self.csv.loc[idx, ycol].values.astype(np.float32), dtype=torch.float32)  \n        #labels = labels.unsqueeze(0).expand(eeg_data.size(0), -1)  # 调整标签的尺寸与输出相匹配\n        \n        return eeg_data, labels\n\n# Create an EEG dataset instance\ndataset = EEGDataset(train_path, EEG_PATH)\n\n\n# Create data loader\ndataloader = DataLoader(dataset, batch_size=32, shuffle=True, num_workers=4)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T14:23:35.146555Z","iopub.execute_input":"2024-02-13T14:23:35.146830Z","iopub.status.idle":"2024-02-13T14:23:35.181867Z","shell.execute_reply.started":"2024-02-13T14:23:35.146808Z","shell.execute_reply":"2024-02-13T14:23:35.181180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X, y = dataset[0]\nprint(f\"Sample {0 + 1}: X shape {X.shape}, y shape {y.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-02-13T14:23:35.182793Z","iopub.execute_input":"2024-02-13T14:23:35.183043Z","iopub.status.idle":"2024-02-13T14:23:35.199743Z","shell.execute_reply.started":"2024-02-13T14:23:35.183013Z","shell.execute_reply":"2024-02-13T14:23:35.198946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CNN+LSTM Model","metadata":{}},{"cell_type":"code","source":"import torch.nn.functional as F\n\nclass CNNLSTM(nn.Module):\n    def __init__(self, in_channels=8, num_classes=6):\n        \n        super(CNNLSTM, self).__init__()\n        self.conv1 = nn.Conv1d(in_channels, 32, kernel_size=3, stride=1, padding=1)\n        self.bn1 = nn.BatchNorm1d(32)\n        self.relu1 = nn.ReLU(inplace=True)\n        self.dropout1 = nn.Dropout(p=0.5)\n        self.pool1 = nn.MaxPool1d(kernel_size=2, stride=2)\n\n        self.conv2 = nn.Conv1d(32, 64, kernel_size=3, stride=1, padding=1)\n        self.bn2 = nn.BatchNorm1d(64)\n        self.relu2 = nn.ReLU(inplace=True)\n        self.dropout2 = nn.Dropout(p=0.75)\n        self.pool2 = nn.MaxPool1d(kernel_size=2, stride=2)\n\n        self.lstm1 = nn.LSTM(input_size=64, hidden_size=128, num_layers=1, batch_first=True, bidirectional=True)\n        self.lstm2 = nn.LSTM(input_size=256, hidden_size=128, num_layers=1, batch_first=True, bidirectional=True)\n\n        self.attention = nn.Sequential(\n            nn.Linear(128 * 2, 64),\n            nn.Tanh(),\n            nn.Linear(64, 1)\n        )\n\n        self.fc = nn.Linear(128 * 2, num_classes)\n\n    def forward(self, x):\n        x = self.conv1(x)\n        x = self.bn1(x)\n        x = self.relu1(x)\n        x = self.dropout1(x)\n        x = self.pool1(x)\n\n        x = self.conv2(x)\n        x = self.bn2(x)\n        x = self.relu2(x)\n        x = self.dropout2(x)\n        x = self.pool2(x)\n\n        # reshape for LSTM\n        batch_size, channels, seq_length = x.size()\n        x = x.permute(0, 2, 1)\n\n        # First LSTM layer\n        x, _ = self.lstm1(x)\n\n        # Second LSTM layer\n        x, _ = self.lstm2(x)\n        \n        # Attention layer\n        att_weights = F.softmax(self.attention(x), dim=1)\n        x = torch.sum(att_weights * x, dim=1)\n\n        # Fully connected layer\n        x = self.fc(x)\n\n        return x\n","metadata":{"execution":{"iopub.status.busy":"2024-02-13T14:23:35.201242Z","iopub.execute_input":"2024-02-13T14:23:35.202105Z","iopub.status.idle":"2024-02-13T14:23:35.214560Z","shell.execute_reply.started":"2024-02-13T14:23:35.202079Z","shell.execute_reply":"2024-02-13T14:23:35.213656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define some hyperparameters\ninput_channels = 8\nnum_classes = 6  \n\n# Create an EEGNet instance\nmodel = CNNLSTM(in_channels=input_channels, num_classes=num_classes)\n\n# 将模型放在GPU上\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel.to(device)\n\n# Define loss function and optimizer\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001)\n\n# 训练模型\nnum_epochs = 5\nfor epoch in range(num_epochs):\n    model.train()\n    running_loss = 0.0\n    for i, (inputs, labels) in enumerate(dataloader):\n        inputs, labels = inputs.to(device), labels.to(device)\n        optimizer.zero_grad()\n        outputs = model(inputs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        running_loss += loss.item()\n    print(f\"Epoch {epoch+1}, Loss: {running_loss / len(dataloader)}\")","metadata":{"execution":{"iopub.status.busy":"2024-02-13T14:23:35.215693Z","iopub.execute_input":"2024-02-13T14:23:35.216019Z","iopub.status.idle":"2024-02-13T14:33:08.952294Z","shell.execute_reply.started":"2024-02-13T14:23:35.215986Z","shell.execute_reply":"2024-02-13T14:33:08.951182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict test data","metadata":{}},{"cell_type":"code","source":"test_path = '/kaggle/input/hms-harmful-brain-activity-classification/test.csv'\nTEST_EEG_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\n\nclass TestEEGDataset(Dataset):\n    def __init__(self, csv_file, eeg_path):\n        self.csv = pd.read_csv(csv_file)  # Load the CSV file\n        self.eeg_path = eeg_path\n        self.sos = self.butter_bandpass_filter_init()  # Initialize the Butterworth filter parameters\n        self.FEATS = ['Fp1','T3','C3','O1','Fp2','C4','T4','O2']\n        \n    def __len__(self):\n        return len(self.csv)\n    \n    def butter_bandpass_filter_init(self):\n        lowcut = 0.5  # Set the low-frequency cutoff for bandpass filtering\n        highcut = 45.0  # Set the high-frequency cutoff for bandpass filtering\n        freq = 200.0  # Sampling frequency\n        order = 5  # Filter order\n\n        nyq = 0.5 * freq\n        low = lowcut / nyq\n        high = highcut / nyq\n        sos = butter(order, [low, high], analog=False, btype='band', output='sos')  # Create second-order sections for the Butterworth filter\n        return sos\n\n    def butter_bandpass_filter(self, data):\n        y = sosfilt(self.sos, data)\n        return y\n\n    def __getitem__(self, idx):\n        eeg_id = self.csv.loc[idx, 'eeg_id']\n        eeg_file_path = f\"{self.eeg_path}{eeg_id}.parquet\"  # Build the EEG data file path\n        eeg_data = pd.read_parquet(eeg_file_path)[self.FEATS].values  # Load EEG data from Parquet file\n\n        eeg_data = self.butter_bandpass_filter(eeg_data)  # Apply filtering to the data\n        eeg_data = torch.tensor(eeg_data, dtype=torch.float32)  # Convert to PyTorch tensor\n         \n        # Select 10,000 data points from the middle\n        mid_index = eeg_data.shape[0] // 2\n        start_index = mid_index - 5000  # Offset 5000 data points to the left from the middle\n        end_index = mid_index + 5000  # Offset 5000 data points to the right from the middle\n        eeg_data = eeg_data[start_index:end_index]\n        # Swap dimensions\n        eeg_data = torch.transpose(eeg_data, 0, 1)\n        \n        return eeg_data\n\n# 创建 EEG 数据集实例\ntestdataset = TestEEGDataset(test_path, TEST_EEG_PATH)\n\n# 创建数据加载器\ntest_dataloader = DataLoader(testdataset, batch_size=32, shuffle=True, num_workers=4)\n\n# 2. 将模型设置为评估模式\nmodel.eval()\n\npredictions = []\n# 执行推理\nwith torch.no_grad():\n    for inputs in test_dataloader:  # 注意这里不需要 labels\n        inputs = inputs.to(device)\n        outputs = model(inputs)\n        # 处理输出，得到类别概率值\n        probabilities = torch.softmax(outputs, dim=1)\n        predictions.append(probabilities.cpu().numpy())\n\n# 4. Process the prediction results, such as converting them into probabilities\npredictions = np.concatenate(predictions, axis=0)\n\n# 5. Output prediction results\nprint(predictions)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T14:33:08.955766Z","iopub.execute_input":"2024-02-13T14:33:08.956075Z","iopub.status.idle":"2024-02-13T14:33:09.190026Z","shell.execute_reply.started":"2024-02-13T14:33:08.956046Z","shell.execute_reply":"2024-02-13T14:33:09.188772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create DataFrame\ncolumns = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\nresults_df = pd.DataFrame(predictions, columns=columns)\n\n# Print DataFrame\nprint(results_df)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T14:33:09.191387Z","iopub.execute_input":"2024-02-13T14:33:09.191711Z","iopub.status.idle":"2024-02-13T14:33:09.201956Z","shell.execute_reply.started":"2024-02-13T14:33:09.191681Z","shell.execute_reply":"2024-02-13T14:33:09.200697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(test_path)\nsub = pd.DataFrame({'eeg_id':test_data.eeg_id.values})\nsub[TARGETS] = results_df\nsub.to_csv('submission.csv',index=False)\nprint('Submissionn shape',sub.shape)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-13T14:33:09.203063Z","iopub.execute_input":"2024-02-13T14:33:09.203321Z","iopub.status.idle":"2024-02-13T14:33:09.227484Z","shell.execute_reply.started":"2024-02-13T14:33:09.203296Z","shell.execute_reply":"2024-02-13T14:33:09.226656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.iloc[:,-6:].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-13T14:33:09.228563Z","iopub.execute_input":"2024-02-13T14:33:09.228852Z","iopub.status.idle":"2024-02-13T14:33:09.236930Z","shell.execute_reply.started":"2024-02-13T14:33:09.228828Z","shell.execute_reply":"2024-02-13T14:33:09.236034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}