{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392733,"sourceType":"datasetVersion","datasetId":4297749},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782},{"sourceId":7402356,"sourceType":"datasetVersion","datasetId":4304475},{"sourceId":7403069,"sourceType":"datasetVersion","datasetId":4304949},{"sourceId":7447509,"sourceType":"datasetVersion","datasetId":4334995},{"sourceId":7450712,"sourceType":"datasetVersion","datasetId":4336944},{"sourceId":7581697,"sourceType":"datasetVersion","datasetId":4413439},{"sourceId":7581715,"sourceType":"datasetVersion","datasetId":4413451},{"sourceId":7581720,"sourceType":"datasetVersion","datasetId":4413454},{"sourceId":7585255,"sourceType":"datasetVersion","datasetId":4415285,"isSourceIdPinned":true},{"sourceId":158958765,"sourceType":"kernelVersion"},{"sourceId":159333316,"sourceType":"kernelVersion"},{"sourceId":159396114,"sourceType":"kernelVersion"}],"dockerImageVersionId":30636,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## ℹ️Update Info\n\n* **2024/02/10 forked original great work kernels**\n    * [Blend][Inference] https://www.kaggle.com/code/kitsuha/3-model-ensemble-lb-0-38\n    * [Blend][Inference] https://www.kaggle.com/code/andreasbis/hms-inference-lb-0-41\n    * [Single][Inference] https://www.kaggle.com/code/yunsuxiaozi/hms-baseline-resnet34d-512-512-inference-6-models\n    * [Single][Inference] https://www.kaggle.com/code/crackle/efficientnetb0-pytorch-starter-lb-0-40\n    \n\n* **2024/02/16**\n    * add Blend Weights. Use param by https://www.kaggle.com/code/kitsuha/3-model-ensemble-lb-0-37https://www.kaggle.com/code/kitsuha/3-model-ensemble-lb-0-37","metadata":{}},{"cell_type":"markdown","source":"```\nCV Score KL-Div for EfficientNetB2 = 0.6431805911043548\n```","metadata":{}},{"cell_type":"markdown","source":"---\n# **《《《　Model 2　》》》**\n---","metadata":{}},{"cell_type":"code","source":"#https://www.kaggle.com/code/ttahara/hms-hbac-resnet34d-baseline-training\n#https://www.kaggle.com/code/ttahara/hms-hbac-resnet34d-baseline-inference\n#necessary\nimport pandas as pd#导入csv文件的库\nimport numpy as np#进行矩阵运算的库\nimport torch #一个深度学习的库Pytorch\nimport torch.nn as nn#neural network,神经网络\nimport torch.nn.functional as F#神经网络函数库\nimport torchvision.transforms as transforms#Pytorch下面的图像处理库,用于对图像进行数据增强\n#设置随机种子\nimport random\nimport warnings#避免一些可以忽略的报错\nwarnings.filterwarnings('ignore')#filterwarnings()方法是用于设置警告过滤器的方法，它可以控制警告信息的输出方式和级别。","metadata":{"execution":{"iopub.status.busy":"2024-02-21T01:17:35.567403Z","iopub.execute_input":"2024-02-21T01:17:35.567978Z","iopub.status.idle":"2024-02-21T01:17:43.292055Z","shell.execute_reply.started":"2024-02-21T01:17:35.567905Z","shell.execute_reply":"2024-02-21T01:17:43.290831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    seed=2024\n    image_transform=transforms.Resize((512, 512))\n    num_folds=5","metadata":{"execution":{"iopub.status.busy":"2024-02-21T01:21:48.803122Z","iopub.execute_input":"2024-02-21T01:21:48.803675Z","iopub.status.idle":"2024-02-21T01:21:48.809381Z","shell.execute_reply.started":"2024-02-21T01:21:48.803644Z","shell.execute_reply":"2024-02-21T01:21:48.808370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models=[]\nfor i in range(Config.num_folds):\n    model = torch.load(f'/kaggle/input/hms-baseline-resnet34d-512-512-training-5-folds/HMS_resnet_fold{i}.pth')\n    models.append(model)\nmodel = torch.load(\"/kaggle/input/hms-baseline-resnet34d-512-512-training/HMS_resnet.pth\")\nmodels.append(model)","metadata":{"execution":{"iopub.status.busy":"2024-02-21T01:21:55.889477Z","iopub.execute_input":"2024-02-21T01:21:55.890424Z","iopub.status.idle":"2024-02-21T01:22:05.171985Z","shell.execute_reply.started":"2024-02-21T01:21:55.890381Z","shell.execute_reply":"2024-02-21T01:22:05.171124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed):\n    torch.backends.cudnn.deterministic = True#将cuda加速的随机数生成器设为确定性模式\n    torch.backends.cudnn.benchmark = True#关闭CuDNN框架的自动寻找最优卷积算法的功能，以避免不同的算法对结果产生影响\n    torch.manual_seed(seed)#pytorch的随机种子\n    np.random.seed(seed)#numpy的随机种子\n    random.seed(seed)#python内置的随机种子\nseed_everything(Config.seed)","metadata":{"execution":{"iopub.status.busy":"2024-02-21T01:22:13.519061Z","iopub.execute_input":"2024-02-21T01:22:13.519454Z","iopub.status.idle":"2024-02-21T01:22:13.526633Z","shell.execute_reply.started":"2024-02-21T01:22:13.519424Z","shell.execute_reply":"2024-02-21T01:22:13.525548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df=pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/test.csv\")\nsubmission=pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv\")\nsubmission=submission.merge(test_df,on='eeg_id',how='left')\nsubmission['path']=submission['spectrogram_id'].apply(lambda x: \"/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/\"+str(x)+\".parquet\" )\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-21T01:22:15.439169Z","iopub.execute_input":"2024-02-21T01:22:15.439541Z","iopub.status.idle":"2024-02-21T01:22:15.523378Z","shell.execute_reply.started":"2024-02-21T01:22:15.439510Z","shell.execute_reply":"2024-02-21T01:22:15.522224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths=submission['path'].values\ntest_preds=[]\nfor path in paths:\n    eps=1e-6\n    data=pd.read_parquet(path)\n    #这里最小值是0,故用-1填充.第一列是时间列,故去掉 ,行是不同列,列是时间\n    data = data.fillna(-1).values[:,1:].T\n    #选取一段时间的数据进行训练\n    data=data[:,0:300]#(400,300)\n    data=np.clip(data,np.exp(-6),np.exp(10))#最大值为89209464.0\n    data= np.log(data)#对数变换\n    #对数据进行归一化\n    data_mean=data.mean(axis=(0,1))\n    data_std=data.std(axis=(0,1))\n    data=(data-data_mean)/(data_std+eps)\n    data_tensor = torch.unsqueeze(torch.Tensor(data), dim=0)\n    data=Config.image_transform(data_tensor)\n    test_pred=[]\n    for model in models:\n        model.eval()\n        with torch.no_grad():\n            pred=F.softmax(model(data.unsqueeze(0)))[0]\n            pred=pred.detach().cpu().numpy()\n        test_pred.append(pred)\n    test_pred=np.array(test_pred).mean(axis=0)\n    test_preds.append(test_pred)\ntest_preds=np.array(test_preds)\ntest_preds","metadata":{"execution":{"iopub.status.busy":"2024-02-21T01:22:18.077753Z","iopub.execute_input":"2024-02-21T01:22:18.078172Z","iopub.status.idle":"2024-02-21T01:22:21.193542Z","shell.execute_reply.started":"2024-02-21T01:22:18.078140Z","shell.execute_reply":"2024-02-21T01:22:21.192485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub2=pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv\")\nlabels=['seizure','lpd','gpd','lrda','grda','other']\nfor i in range(len(labels)):\n    sub2[f'{labels[i]}_vote']=test_preds[:,i]\nsub2.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-16T12:10:45.815342Z","iopub.execute_input":"2024-02-16T12:10:45.81569Z","iopub.status.idle":"2024-02-16T12:10:45.833142Z","shell.execute_reply.started":"2024-02-16T12:10:45.815661Z","shell.execute_reply":"2024-02-16T12:10:45.832218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\n# **《《《　Model 3　》》》**\n---","metadata":{}},{"cell_type":"code","source":"# Importing essential libraries\nimport gc\nimport os\nimport random\nimport warnings\nimport numpy as np\nimport pandas as pd\nfrom IPython.display import display\n\n# PyTorch for deep learning\nimport timm\nimport torch\nimport torch.nn as nn  \nimport torch.optim as optim\nimport torch.nn.functional as F\n\n# torchvision for image processing and augmentation\nimport torchvision.transforms as transforms\n\n# Suppressing minor warnings to keep the output clean\nwarnings.filterwarnings('ignore', category=Warning)\n\n# Reclaim memory no longer in use.\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-21T01:22:24.679016Z","iopub.execute_input":"2024-02-21T01:22:24.679414Z","iopub.status.idle":"2024-02-21T01:22:24.826248Z","shell.execute_reply.started":"2024-02-21T01:22:24.679383Z","shell.execute_reply":"2024-02-21T01:22:24.825154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    seed=42\n    image_transform=transforms.Resize((512, 512))\n    num_folds=5\n    \n# Set the seed for reproducibility across multiple libraries\ndef set_seed(seed):\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\n    torch.manual_seed(seed)\n    np.random.seed(seed)\n    random.seed(seed)\n    \nset_seed(Config.seed)","metadata":{"execution":{"iopub.status.busy":"2024-02-21T01:22:27.309202Z","iopub.execute_input":"2024-02-21T01:22:27.310100Z","iopub.status.idle":"2024-02-21T01:22:27.317217Z","shell.execute_reply.started":"2024-02-21T01:22:27.310065Z","shell.execute_reply":"2024-02-21T01:22:27.316158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load and store the trained models for each fold into a list\nmodels = []\n\n# Load ResNet34d\nfor i in range(Config.num_folds):\n    # Create the same model architecture as during training\n    model_resnet = timm.create_model('resnet34d', pretrained=False, num_classes=6, in_chans=1)\n    \n    # Load the trained weights from the corresponding file\n    model_resnet.load_state_dict(torch.load(f'/kaggle/input/resnet34d/hms-train-resnet34d/resnet34d_fold{i}.pth', map_location=torch.device('cpu')))\n    \n    # Append the loaded model to the models list\n    models.append(model_resnet)\n\n# Reclaim memory no longer in use.\ngc.collect()\n\n# Load EfficientNetB0\nfor j in range(Config.num_folds):\n    # Create the same model architecture as during training\n    model_effnet_b0 = timm.create_model('efficientnet_b0', pretrained=False, num_classes=6, in_chans=1)\n    \n    # Load the trained weights from the corresponding file\n    model_effnet_b0.load_state_dict(torch.load(f'/kaggle/input/efficientnetb0/hms-train-efficientnetb0/efficientnet_b0_fold{j}.pth', map_location=torch.device('cpu')))\n    \n    # Append the loaded model to the models list\n    models.append(model_effnet_b0)\n    \n# Reclaim memory no longer in use.\ngc.collect()\n    \n# Load EfficientNetB1\nfor k in range(Config.num_folds):\n    # Create the same model architecture as during training\n    model_effnet_b1 = timm.create_model('efficientnet_b1', pretrained=False, num_classes=6, in_chans=1)\n    \n    # Load the trained weights from the corresponding file\n    model_effnet_b1.load_state_dict(torch.load(f'/kaggle/input/efficientnetb1/hms-train-efficientnetb1/efficientnet_b1_fold{k}.pth', map_location=torch.device('cpu')))\n    \n    # Append the loaded model to the models list\n    models.append(model_effnet_b1)\n\n# Reclaim memory no longer in use.\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-21T01:22:29.490437Z","iopub.execute_input":"2024-02-21T01:22:29.490841Z","iopub.status.idle":"2024-02-21T01:22:41.243939Z","shell.execute_reply.started":"2024-02-21T01:22:29.490811Z","shell.execute_reply":"2024-02-21T01:22:41.242511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load test data and sample submission dataframe\ntest_df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/test.csv\")\nsubmission = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv\")\n\n# Merge the submission dataframe with the test data on EEG IDs\nsubmission = submission.merge(test_df, on='eeg_id', how='left')\n\n# Generate file paths for each spectrogram based on the EEG data in the submission dataframe\nsubmission['path'] = submission['spectrogram_id'].apply(lambda x: f\"/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/{x}.parquet\")\n\n# Display the first few rows of the submission dataframe\ndisplay(submission.head())\n\n# Reclaim memory no longer in use\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-21T01:22:45.419302Z","iopub.execute_input":"2024-02-21T01:22:45.420625Z","iopub.status.idle":"2024-02-21T01:22:45.603198Z","shell.execute_reply.started":"2024-02-21T01:22:45.420586Z","shell.execute_reply":"2024-02-21T01:22:45.602122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the weights for each model\nweight_resnet34d = 0.25\nweight_effnetb0 = 0.42\nweight_effnetb1 = 0.33\n\n# Get file paths for test spectrograms\npaths = submission['path'].values\ntest_predss = []\n\n# Generate predictions for each spectrogram using all models\nfor path in paths:\n    eps = 1e-6\n    # Read and preprocess spectrogram data\n    data = pd.read_parquet(path)\n    data = data.fillna(-1).values[:, 1:].T\n    data = np.clip(data, np.exp(-6), np.exp(10))\n    data = np.log(data)\n    \n    # Normalize the data\n    data_mean = data.mean(axis=(0, 1))\n    data_std = data.std(axis=(0, 1))\n    data = (data - data_mean) / (data_std + eps)\n    data_tensor = torch.unsqueeze(torch.Tensor(data), dim=0)\n    data = Config.image_transform(data_tensor)\n\n    test_pred = []\n    \n    # Generate predictions using all models\n    for model in models:\n        model.eval()\n        with torch.no_grad():\n            pred = F.softmax(model(data.unsqueeze(0)))[0]\n            pred = pred.detach().cpu().numpy()\n        test_pred.append(pred)\n        \n    # Combine predictions from all models using weighted voting\n    weighted_pred = weight_resnet34d * np.mean(test_pred[:Config.num_folds], axis=0) + \\\n                     weight_effnetb0 * np.mean(test_pred[Config.num_folds:2*Config.num_folds], axis=0) + \\\n                     weight_effnetb1 * np.mean(test_pred[2*Config.num_folds:], axis=0)\n    \n    test_predss.append(weighted_pred)\n\n# Convert the list of predictions to a NumPy array for further processing\ntest_predss = np.array(test_predss)\n\n# Reclaim memory no longer in use\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-02-21T01:22:48.357331Z","iopub.execute_input":"2024-02-21T01:22:48.357722Z","iopub.status.idle":"2024-02-21T01:22:51.712999Z","shell.execute_reply.started":"2024-02-21T01:22:48.357692Z","shell.execute_reply":"2024-02-21T01:22:51.712001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predss","metadata":{"execution":{"iopub.status.busy":"2024-02-21T01:22:55.849214Z","iopub.execute_input":"2024-02-21T01:22:55.849581Z","iopub.status.idle":"2024-02-21T01:22:55.856176Z","shell.execute_reply.started":"2024-02-21T01:22:55.849554Z","shell.execute_reply":"2024-02-21T01:22:55.855149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\n# **《《《　Sub　》》》**\n---","metadata":{}},{"cell_type":"code","source":"test_preds","metadata":{"execution":{"iopub.status.busy":"2024-02-21T01:32:05.877439Z","iopub.execute_input":"2024-02-21T01:32:05.877810Z","iopub.status.idle":"2024-02-21T01:32:05.885511Z","shell.execute_reply.started":"2024-02-21T01:32:05.877782Z","shell.execute_reply":"2024-02-21T01:32:05.883936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission=pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv\")\nlabels=['seizure','lpd','gpd','lrda','grda','other']\nfor i in range(len(labels)):\n    submission[f'{labels[i]}_vote']=(test_preds[:, i]*0.304 + test_predss[:, i]*0.696)\nsubmission.to_csv(\"submission.csv\",index=None)\ndisplay(submission.head())","metadata":{"execution":{"iopub.status.busy":"2024-02-21T01:34:04.161535Z","iopub.execute_input":"2024-02-21T01:34:04.161986Z","iopub.status.idle":"2024-02-21T01:34:04.198269Z","shell.execute_reply.started":"2024-02-21T01:34:04.161951Z","shell.execute_reply":"2024-02-21T01:34:04.197114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SANITY CHECK TO CONFIRM PREDICTIONS SUM TO ONE\nsubmission.iloc[:,-6:].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-21T01:34:07.437000Z","iopub.execute_input":"2024-02-21T01:34:07.437772Z","iopub.status.idle":"2024-02-21T01:34:07.447807Z","shell.execute_reply.started":"2024-02-21T01:34:07.437739Z","shell.execute_reply":"2024-02-21T01:34:07.446904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}