{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7527876,"sourceType":"datasetVersion","datasetId":4384638},{"sourceId":7617842,"sourceType":"datasetVersion","datasetId":4436814}],"dockerImageVersionId":30648,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os,gc\nos.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"\nimport pandas as pd, numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom tensorflow.keras import layers, models\nfrom tensorflow.keras.utils import to_categorical\nimport matplotlib.pyplot as plt\nimport glob\nimport torch\nimport warnings\n\n# 禁用所有警告\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-17T15:41:13.681120Z","iopub.execute_input":"2024-02-17T15:41:13.681538Z","iopub.status.idle":"2024-02-17T15:41:13.688483Z","shell.execute_reply.started":"2024-02-17T15:41:13.681509Z","shell.execute_reply":"2024-02-17T15:41:13.687385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH = '/kaggle/input/hms-harmful-brain-activity-classification/'\ndf = pd.read_csv(PATH + 'train.csv')\nTARGETS = df.columns[-6:]\nprint('Train shape:', df.shape )\nprint('Targets', list(TARGETS))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-17T15:41:13.691458Z","iopub.execute_input":"2024-02-17T15:41:13.691814Z","iopub.status.idle":"2024-02-17T15:41:13.896290Z","shell.execute_reply.started":"2024-02-17T15:41:13.691784Z","shell.execute_reply":"2024-02-17T15:41:13.895098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spectrogram_id','min']\n\ntmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp\n\ntmp = df.groupby('eeg_id')[['patient_id']].agg('first')\ntrain['patient_id'] = tmp\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n    \ny_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\ntrain = train.reset_index()\nprint('Train non-overlapp eeg_id shape:', train.shape )\n\ntrain.head(20)","metadata":{"execution":{"iopub.status.busy":"2024-02-17T15:41:13.898649Z","iopub.execute_input":"2024-02-17T15:41:13.899111Z","iopub.status.idle":"2024-02-17T15:41:13.995996Z","shell.execute_reply.started":"2024-02-17T15:41:13.899077Z","shell.execute_reply":"2024-02-17T15:41:13.994912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ycol = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\ncd = {'Seizure': 'seizure_vote', 'GPD': 'gpd_vote', 'LRDA': 'lrda_vote', 'Other': 'other_vote', 'GRDA': 'grda_vote', 'LPD': 'lpd_vote'}\n\n# 提取概率列和标签列\neeg_id_col = train.iloc[:, 0]  # 第一列是 eeg_id 列\nprob_cols = train.iloc[:, -7:-1]  # 倒数第七到倒数第二列是概率列\nlabel_col = train.iloc[:, -1]  # 最后一列是标签列\n\n# 将概率列转换为 float32 类型\nprob_cols = prob_cols.astype(\"float32\")\n\n# 归一化概率列\nprob_cols_normalized = prob_cols.div(prob_cols.sum(axis=1), axis=0)\n\n# 重新组合成 DataFrame\nnormalized_data = pd.concat([eeg_id_col, prob_cols_normalized, label_col], axis=1)\n\nnormalized_data.head(20)","metadata":{"execution":{"iopub.status.busy":"2024-02-17T15:41:13.997408Z","iopub.execute_input":"2024-02-17T15:41:13.997822Z","iopub.status.idle":"2024-02-17T15:41:14.036178Z","shell.execute_reply.started":"2024-02-17T15:41:13.997783Z","shell.execute_reply":"2024-02-17T15:41:14.035238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normalized_data.to_csv(\"normalized_data.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-17T15:41:14.039427Z","iopub.execute_input":"2024-02-17T15:41:14.039850Z","iopub.status.idle":"2024-02-17T15:41:14.161247Z","shell.execute_reply.started":"2024-02-17T15:41:14.039811Z","shell.execute_reply":"2024-02-17T15:41:14.160242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EEG_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\ntrain_path = '/kaggle/input/processed/normalized_data.csv'","metadata":{"execution":{"iopub.status.busy":"2024-02-17T15:41:14.162368Z","iopub.execute_input":"2024-02-17T15:41:14.162698Z","iopub.status.idle":"2024-02-17T15:41:14.167368Z","shell.execute_reply.started":"2024-02-17T15:41:14.162657Z","shell.execute_reply":"2024-02-17T15:41:14.166408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport numpy as np\nimport pandas as pd\nimport glob\nfrom scipy.signal import butter, sosfilt\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.impute import SimpleImputer\n\nimport pandas as pd\n\nclass EEGDataset(Dataset):\n    def __init__(self, csv_file, eeg_path):\n        self.csv = pd.read_csv(csv_file)  # 加载 CSV 文件\n        self.eeg_path = eeg_path\n        self.sos = self.butter_bandpass_filter_init()\n        self.FEATS = ['Fp1','T3','C3','O1','Fp2','C4','T4','O2']  # 指定需要使用的特征列\n\n    def butter_bandpass_filter_init(self):\n        lowcut = 0.5  # 设置带通滤波的低频截止频率\n        highcut = 40.0  # 设置带通滤波的高频截止频率\n        fs = 200.0  # 采样频率\n        order = 3  # 滤波器阶数\n\n        nyq = 0.5 * fs\n        low = lowcut / nyq\n        high = highcut / nyq\n        sos = butter(order, [low, high], analog=False, btype='band', output='sos')\n        return sos\n\n    def butter_bandpass_filter(self, data):\n        y = sosfilt(self.sos, data)\n        return y\n    \n    def __len__(self):\n        return len(self.csv)\n    \n    def __getitem__(self, idx):\n        eeg_id = self.csv.loc[idx, 'eeg_id']\n        eeg_file_path = f\"{self.eeg_path}{eeg_id}.parquet\"  # 构建 EEG 数据文件的路径\n        eeg_data = pd.read_parquet(eeg_file_path)[self.FEATS].values  # 从 Parquet 文件中加载指定的特征列\n        # 检查是否存在 NaN 值\n        if np.isnan(eeg_data).any():\n            # 如果存在 NaN 值，你可以选择将其填充为特定的值或者进行插值处理\n            # 这里我们使用 SimpleImputer 进行简单的填充处理，将 NaN 值替换为均值\n            imputer = SimpleImputer(strategy='mean')\n            eeg_data = imputer.fit_transform(eeg_data)\n        eeg_data = self.butter_bandpass_filter(eeg_data)  # 对数据进行滤波\n        eeg_data = torch.tensor(eeg_data, dtype=torch.float32)  # 转换为 PyTorch 张量\n                    \n        # 选择中间 10000 个时间点的数据\n        mid_index = eeg_data.shape[0] // 2\n        start_index = mid_index - 5000  # 从中间向左偏移5000个时间点\n        end_index = mid_index + 5000  # 到中间向右偏移5000个时间点\n        eeg_data = eeg_data[start_index:end_index]\n        # 交换维度位置\n        eeg_data = torch.transpose(eeg_data, 0, 1)\n        \n        labels = torch.tensor(self.csv.loc[idx, ycol].values.astype(np.float32), dtype=torch.float32)  # 加载对应的标签\n        #labels = labels.unsqueeze(0).expand(eeg_data.size(0), -1)  # 调整标签的尺寸与输出相匹配\n        \n        return eeg_data, labels\n\n# 创建 EEG 数据集实例\ndataset = EEGDataset(train_path, EEG_PATH)\n\n\n# 创建数据加载器\ndataloader = DataLoader(dataset, batch_size=32, shuffle=True, num_workers=4)","metadata":{"execution":{"iopub.status.busy":"2024-02-17T15:41:14.168685Z","iopub.execute_input":"2024-02-17T15:41:14.168971Z","iopub.status.idle":"2024-02-17T15:41:14.206036Z","shell.execute_reply.started":"2024-02-17T15:41:14.168947Z","shell.execute_reply":"2024-02-17T15:41:14.205199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X, y = dataset[0]\nprint(f\"Sample {0 + 1}: X shape {X.shape}, y shape {y.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-02-17T15:41:14.207430Z","iopub.execute_input":"2024-02-17T15:41:14.208076Z","iopub.status.idle":"2024-02-17T15:41:14.228896Z","shell.execute_reply.started":"2024-02-17T15:41:14.208041Z","shell.execute_reply":"2024-02-17T15:41:14.228106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nclass EEGTransformerClassifier(nn.Module):\n    def __init__(self, input_channels, num_classes, num_layers=2, num_heads=4, dropout=0.35):\n        super(EEGTransformerClassifier, self).__init__()\n        \n        # Transformer部分作为特征提取器\n        self.transformer = nn.TransformerEncoder(\n            nn.TransformerEncoderLayer(d_model=input_channels, nhead=num_heads),\n            num_layers=num_layers\n        )\n        \n        # 全连接层用于分类\n        self.fc = nn.Linear(input_channels, num_classes)\n        \n        self.dropout = nn.Dropout(dropout)\n        \n    def forward(self, x):\n        # 输入形状：(batch_size, channel, timepoints)\n        # 需要转换为：(timepoints, batch_size, channel)\n        x = x.permute(2, 0, 1)\n        \n        # Transformer特征提取\n        features = self.transformer(x)\n        \n        # 取最后一个时间步的特征用于分类\n        features = features[-1]\n        \n        # 使用全连接层进行分类\n        out = self.dropout(features)\n        out = self.fc(out)\n        \n        return out","metadata":{"execution":{"iopub.status.busy":"2024-02-17T15:41:14.229732Z","iopub.execute_input":"2024-02-17T15:41:14.230001Z","iopub.status.idle":"2024-02-17T15:41:14.242724Z","shell.execute_reply.started":"2024-02-17T15:41:14.229978Z","shell.execute_reply":"2024-02-17T15:41:14.241656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN = True","metadata":{"execution":{"iopub.status.busy":"2024-02-17T15:41:14.243823Z","iopub.execute_input":"2024-02-17T15:41:14.244085Z","iopub.status.idle":"2024-02-17T15:41:14.253488Z","shell.execute_reply.started":"2024-02-17T15:41:14.244061Z","shell.execute_reply":"2024-02-17T15:41:14.252523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 定义一些超参数\ninput_channels = 8\nnum_classes = 6  # 类别数\n    \nif TRAIN == True:\n\n    # 创建EEGTransformerClassifier模型实例\n    model = EEGTransformerClassifier(input_channels=input_channels, num_classes=num_classes, num_layers=2, num_heads=4, dropout=0.35)\n\n    # 将模型放在GPU上\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    model.to(device)\n\n    # 定义损失函数和优化器\n    criterion = nn.CrossEntropyLoss()\n    optimizer = optim.Adam(model.parameters(), lr=0.001)\n\n    # 训练模型\n    num_epochs = 5\n    for epoch in range(num_epochs):\n        model.train()\n        running_loss = 0.0\n        for i, (inputs, labels) in enumerate(dataloader):\n            inputs, labels = inputs.to(device), labels.to(device)\n\n            # 前向传播\n            outputs = model(inputs)\n\n            # 计算损失\n            loss = criterion(outputs, labels)\n\n            # 反向传播和优化\n            optimizer.zero_grad()\n            loss.backward()\n            optimizer.step()\n\n            running_loss += loss.item()\n\n        print(f\"Epoch {epoch+1}, Loss: {running_loss / len(dataloader)}\")\n    # 定义路径\n    model_path = 'eeg_transformer_model.pth'\n\n    # 保存模型\n    torch.save(model.state_dict(), model_path)\nelse:\n    model_weight_path = '/kaggle/input/transformer/eeg_transformer_model.pth'\n    # 加载模型\n    model = EEGTransformerClassifier(input_channels, num_classes)\n    # 将模型放在GPU上\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    model.to(device)\n    model.load_state_dict(torch.load(model_weight_path))\n","metadata":{"execution":{"iopub.status.busy":"2024-02-17T15:41:14.260015Z","iopub.execute_input":"2024-02-17T15:41:14.260685Z","iopub.status.idle":"2024-02-17T15:41:14.291834Z","shell.execute_reply.started":"2024-02-17T15:41:14.260649Z","shell.execute_reply":"2024-02-17T15:41:14.290833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_path = '/kaggle/input/hms-harmful-brain-activity-classification/test.csv'\nTEST_EEG_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\n\nclass TestEEGDataset(Dataset):\n    def __init__(self, csv_file, eeg_path):\n        self.csv = pd.read_csv(csv_file)  # Load the CSV file\n        self.eeg_path = eeg_path\n        self.sos = self.butter_bandpass_filter_init()  # Initialize the Butterworth filter parameters\n        self.FEATS = ['Fp1','T3','C3','O1','Fp2','C4','T4','O2']\n        \n    def __len__(self):\n        return len(self.csv)\n    \n    def butter_bandpass_filter_init(self):\n        lowcut = 0.5  # Set the low-frequency cutoff for bandpass filtering\n        highcut = 45.0  # Set the high-frequency cutoff for bandpass filtering\n        fs = 200.0  # Sampling frequency\n        order = 3  # Filter order\n\n        nyq = 0.5 * fs\n        low = lowcut / nyq\n        high = highcut / nyq\n        sos = butter(order, [low, high], analog=False, btype='band', output='sos')  # Create second-order sections for the Butterworth filter\n        return sos\n\n    def butter_bandpass_filter(self, data):\n        y = sosfilt(self.sos, data)\n        return y\n\n    def __getitem__(self, idx):\n        eeg_id = self.csv.loc[idx, 'eeg_id']\n        eeg_file_path = f\"{self.eeg_path}{eeg_id}.parquet\"  # Build the EEG data file path\n        eeg_data = pd.read_parquet(eeg_file_path)[self.FEATS].values  # Load EEG data from Parquet file\n\n        eeg_data = self.butter_bandpass_filter(eeg_data)  # Apply filtering to the data\n        eeg_data = torch.tensor(eeg_data, dtype=torch.float32)  # Convert to PyTorch tensor\n         \n        # Select 10,000 data points from the middle\n        mid_index = eeg_data.shape[0] // 2\n        start_index = mid_index - 5000  # Offset 5000 data points to the left from the middle\n        end_index = mid_index + 5000  # Offset 5000 data points to the right from the middle\n        eeg_data = eeg_data[start_index:end_index]\n        # Swap dimensions\n        eeg_data = torch.transpose(eeg_data, 0, 1)\n        \n        return eeg_data\n\n# 创建 EEG 数据集实例\ntestdataset = TestEEGDataset(test_path, TEST_EEG_PATH)\n\n# 创建数据加载器\ntest_dataloader = DataLoader(testdataset, batch_size=32, shuffle=True, num_workers=4)\n\n# 2. 将模型设置为评估模式\nmodel.eval()\n\npredictions = []\n# 执行推理\nwith torch.no_grad():\n    for inputs in test_dataloader:  # 注意这里不需要 labels\n        inputs = inputs.to(device)\n        outputs = model(inputs)\n        # 处理输出，得到类别概率值\n        probabilities = torch.softmax(outputs, dim=1)\n        predictions.append(probabilities.cpu().numpy())\n\n# 4. 对预测结果进行处理，如转换为概率\npredictions = np.concatenate(predictions, axis=0)\n\n# 5. 输出预测结果\nprint(predictions)","metadata":{"execution":{"iopub.status.busy":"2024-02-17T15:41:14.293363Z","iopub.execute_input":"2024-02-17T15:41:14.293673Z","iopub.status.idle":"2024-02-17T15:41:14.517372Z","shell.execute_reply.started":"2024-02-17T15:41:14.293650Z","shell.execute_reply":"2024-02-17T15:41:14.516010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 创建 DataFrame\ncolumns = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\nresults_df = pd.DataFrame(predictions, columns=columns)\n\n# 打印 DataFrame\nprint(results_df)","metadata":{"execution":{"iopub.status.busy":"2024-02-17T15:41:14.519166Z","iopub.execute_input":"2024-02-17T15:41:14.519570Z","iopub.status.idle":"2024-02-17T15:41:14.535099Z","shell.execute_reply.started":"2024-02-17T15:41:14.519533Z","shell.execute_reply":"2024-02-17T15:41:14.534169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(test_path)\nsub = pd.DataFrame({'eeg_id':test_data.eeg_id.values})\nsub[TARGETS] = results_df\nsub.to_csv('submission.csv',index=False)\nprint('Submissionn shape',sub.shape)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-17T15:41:14.536548Z","iopub.execute_input":"2024-02-17T15:41:14.536918Z","iopub.status.idle":"2024-02-17T15:41:14.560778Z","shell.execute_reply.started":"2024-02-17T15:41:14.536886Z","shell.execute_reply":"2024-02-17T15:41:14.559627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.iloc[:,-6:].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-17T15:41:14.562288Z","iopub.execute_input":"2024-02-17T15:41:14.562837Z","iopub.status.idle":"2024-02-17T15:41:14.572383Z","shell.execute_reply.started":"2024-02-17T15:41:14.562805Z","shell.execute_reply":"2024-02-17T15:41:14.571245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}