{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7842804,"sourceType":"datasetVersion","datasetId":4598110}],"dockerImageVersionId":30665,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport torch\nimport torchvision.transforms as transforms\nfrom torchvision.models import resnext101_64x4d\nfrom torchvision.models.efficientnet import EfficientNet_V2_S_Weights\nimport torchvision.models as models\nfrom torch.utils.data import DataLoader\nimport torch.nn.functional as F\nfrom torch import nn\nfrom tqdm import tqdm\nimport math\nfrom torchvision import transforms\nimport torch\nimport numpy as np\nimport pandas as pd\nfrom torch.utils.data import IterableDataset, Dataset\nimport matplotlib.pyplot as plt\nimport time\nimport warnings\nfrom torch.utils.data import DataLoader","metadata":{"execution":{"iopub.status.busy":"2024-03-15T14:26:39.865192Z","iopub.execute_input":"2024-03-15T14:26:39.865647Z","iopub.status.idle":"2024-03-15T14:26:39.932909Z","shell.execute_reply.started":"2024-03-15T14:26:39.865618Z","shell.execute_reply":"2024-03-15T14:26:39.931975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class HMSDatasetD(Dataset):\n    def __init__(self, data, path, target_p, features, device):\n        super(HMSDatasetD, self).__init__()\n        self.data = data\n        self.path = path\n        self.target_p = target_p\n        self.features = features\n        self.device = device\n        self._loaddata()\n\n    def __getitem__(self, index):\n        return self.x[index]\n\n    def __len__(self):\n        return len(self.x)\n\n    def _loaddata(self):\n        self.x = []\n        part = 1\n        start_time = time.time()  # 记录开始时间\n        last_time = start_time\n        for i in range(len(self.data)):\n            if i / len(self.data) > part / 100:\n                this_time = time.time()\n                print(f'{part}%, elapse = {this_time - last_time}s')  # 打印进度\n                last_time = this_time\n                part += 1\n            \n            x = self.load_sample(i)\n\n            self.x.append(x)\n            # if(i == len(self.data)-1):\n                # torch.save(processed_input, 'processed_input.pth')\n        self.x = torch.stack(self.x, dim=0).to(self.device)\n\n    def load_sample(self, index):\n        eeg_id = self.data.iloc[index]['eeg_id']\n        parquet = pd.read_parquet(\n            self.path + str(eeg_id) + '.parquet', \n            columns = self.features\n        )\n        sequences = [\n            ['Fp1','F7','T3','T5','O1'],\n            ['Fp1','F3','C3','P3','O1'],\n            ['Fp2','F4','C4','P4','O2'],\n            ['Fp2','F8','T4','T6','O2']\n        ]\n        start = 0\n        parquet_data = parquet[start: start+ 10000]\n        # 创建一个新的四通道的图像数组\n        height, width = 500, 20  # 所有谱图具有相同的尺寸\n        merged_image = np.zeros((4, height, width))\n        # 创建一个空的 DataFrame，用于存储差值\n        data = pd.DataFrame()\n        # 初始化一个数组来存储四个谱图的强度矩阵之和\n        total_spec = np.zeros((500, 20))\n        # 计算差值并添加到新的 DataFrame 中\n        for j in range(4):\n            for i in range(4):  # 这里的范围是从0到3\n                col_name = f\"{sequences[j][i]} - {sequences[j][i+1]}\"\n                data[col_name] = parquet_data[sequences[j][i]] - parquet_data[sequences[j][i+1]]\n                X = data[col_name]\n                # 将零值替换为较小的正值，例如 1e-10\n                X_processed = X.apply(self.process_value)\n                fs = 200  # 采样率\n                t = np.linspace(0, len(X_processed) // fs, fs, endpoint=False)  # 生成时间序列\n                # 计算谱图\n                res = plt.specgram(X_processed, NFFT=998, Fs=fs, noverlap=530, cmap='viridis')\n                # 将强度矩阵添加到总的谱图矩阵中\n                total_spec += res[0]\n                plt.clf()\n            # 计算平均谱图的强度矩阵\n            average_spec = total_spec / 4\n            merged_image[j, :, :] = average_spec\n        output = torch.tensor(merged_image, dtype = torch.float32)\n        # processed_input[index] = output\n        return output\n\n    def process_value(self, value):\n        if value > 0:\n            return np.log10(value)\n        elif value < 0:\n            return -np.log10(-value)\n        else:\n            # 处理零值，这里选择替换为一个很小的正数\n            return -np.log(1e-10)","metadata":{"execution":{"iopub.status.busy":"2024-03-15T14:26:48.041220Z","iopub.execute_input":"2024-03-15T14:26:48.041928Z","iopub.status.idle":"2024-03-15T14:26:48.059831Z","shell.execute_reply.started":"2024-03-15T14:26:48.041901Z","shell.execute_reply":"2024-03-15T14:26:48.058944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_file = '/kaggle/input/hms-harmful-brain-activity-classification/train.csv'\n\nparquet_file = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\ntrain_data_mode = 1\ndrop_rate = 0.4\n\ntarget_p = [\n    'seizure_vote', \n    'lpd_vote', \n    'gpd_vote', \n    'lrda_vote', \n    'grda_vote', \n    'other_vote'\n]\n\nfeatures = [\n    'Fp1','F7','F3','T3','C3','T5','P3','O1','Fp2','F4','F8','C4','T4','P4','T6','O2'\n]\ndevice = 'cuda'\nbatch_size = 256\n\n","metadata":{"execution":{"iopub.status.busy":"2024-03-15T14:26:51.439579Z","iopub.execute_input":"2024-03-15T14:26:51.440236Z","iopub.status.idle":"2024-03-15T14:26:51.446046Z","shell.execute_reply.started":"2024-03-15T14:26:51.440198Z","shell.execute_reply":"2024-03-15T14:26:51.444961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MResNeXt101(nn.Module):\n    def __init__(self, input_shape, num_class, pretrained=True):\n        super(MResNeXt101, self).__init__()\n        self.conv = nn.Conv2d(in_channels = 4, out_channels = 3, kernel_size = 3)\n        self.MResNeXt101 = models.resnext101_64x4d(pretrained=pretrained)\n        self.fc = nn.Linear(1000, 6)\n\n    def forward(self, x):\n        temp = self.conv(x)\n        temp[torch.isnan(temp)] = 0        \n        y = self.MResNeXt101(temp)\n        outputs = self.fc(y)\n        return outputs\n    def predict(self, x):\n        with torch.no_grad():\n            self.eval()\n            outputs = self.forward(x)\n            return F.softmax(outputs, dim=-1)\n\nclass CustomCrossEntropyLoss(nn.Module):\n    def __init__(self):\n        super(CustomCrossEntropyLoss, self).__init__()\n    def forward(self, y_hat, y):    \n        y_hat = F.log_softmax(y_hat, dim=-1)\n        return -torch.sum(y * y_hat) / y.shape[0]   \n\n","metadata":{"execution":{"iopub.status.busy":"2024-03-15T14:26:54.304965Z","iopub.execute_input":"2024-03-15T14:26:54.305661Z","iopub.status.idle":"2024-03-15T14:26:54.314601Z","shell.execute_reply.started":"2024-03-15T14:26:54.305631Z","shell.execute_reply":"2024-03-15T14:26:54.313720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_shape = (batch_size, 4, 500, 20)\n# epoch = 0\n# model = MResNeXt101(input_shape, len(target_p)).to('cuda')\nmodel_path = '/kaggle/input/res123/resnext101_model.pth'\nmodel = MResNeXt101(input_shape, len(target_p), pretrained=False).to('cuda')\nmodel_weights = torch.load(model_path)\nmodel.load_state_dict(model_weights)\n# optimizer = torch.optim.Adam(model.parameters(), lr=0.001)\n# criterion = CustomCrossEntropyLoss()\n# trainer = Trainer(epoch, model, optimizer, criterion, device)\n# trainer.train(dataloader)\ntrain_file = '/kaggle/input/hms-harmful-brain-activity-classification/test.csv'\nparquet_file = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\n# 生成谱图\ntrain = pd.read_csv(train_file)\n\nhmsdataset = HMSDatasetD(train, parquet_file, target_p, features, device)","metadata":{"execution":{"iopub.status.busy":"2024-03-15T14:32:04.494916Z","iopub.execute_input":"2024-03-15T14:32:04.495516Z","iopub.status.idle":"2024-03-15T14:32:07.416509Z","shell.execute_reply.started":"2024-03-15T14:32:04.495486Z","shell.execute_reply":"2024-03-15T14:32:07.415507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = []\neeg_ids = []\nfor i in range(len(hmsdataset)):\n    x = hmsdataset[i]\n    preds = model.predict(x.unsqueeze(0))\n    predictions.append(preds)\n    eeg_ids.append(hmsdataset.data.iloc[i]['eeg_id'])\n\n# 合并所有预测结果\nall_predictions = torch.cat(predictions, dim=0)\n\n\npredictions_np = all_predictions.cpu().detach().numpy()\ncol_labels = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\ndf = pd.DataFrame(predictions_np, columns=col_labels)\ndf['eeg_id'] = eeg_ids\ndf = df.reindex(columns=['eeg_id'] + col_labels)\n# 保存为 CSV 文件\ndf.to_csv('submission.csv', index=False)\n# with open('predictions.csv', 'r') as file:\n    #csv_content = file.read()\n    #print(csv_content)","metadata":{"execution":{"iopub.status.busy":"2024-03-15T14:36:39.145907Z","iopub.execute_input":"2024-03-15T14:36:39.146268Z","iopub.status.idle":"2024-03-15T14:36:39.178215Z","shell.execute_reply.started":"2024-03-15T14:36:39.146237Z","shell.execute_reply":"2024-03-15T14:36:39.177433Z"},"trusted":true},"execution_count":null,"outputs":[]}]}