{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":272138,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":232995,"modelId":254717}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-01T17:09:01.916399Z","iopub.execute_input":"2025-03-01T17:09:01.916706Z","iopub.status.idle":"2025-03-01T17:09:49.228178Z","shell.execute_reply.started":"2025-03-01T17:09:01.916683Z","shell.execute_reply":"2025-03-01T17:09:49.227347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install efficientnet_pytorch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T17:09:49.229230Z","iopub.execute_input":"2025-03-01T17:09:49.229581Z","iopub.status.idle":"2025-03-01T17:09:56.588129Z","shell.execute_reply.started":"2025-03-01T17:09:49.229561Z","shell.execute_reply":"2025-03-01T17:09:56.587365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, Dataset, random_split\nfrom torchvision import datasets, models, transforms\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport warnings\nimport time\nimport cv2\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T17:09:56.589714Z","iopub.execute_input":"2025-03-01T17:09:56.589954Z","iopub.status.idle":"2025-03-01T17:10:05.405948Z","shell.execute_reply.started":"2025-03-01T17:09:56.589934Z","shell.execute_reply":"2025-03-01T17:10:05.405040Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"start_time = time.time()\nBASE_DIR = \"/kaggle/input/hms-harmful-brain-activity-classification/\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T17:10:05.407198Z","iopub.execute_input":"2025-03-01T17:10:05.407557Z","iopub.status.idle":"2025-03-01T17:10:05.411145Z","shell.execute_reply.started":"2025-03-01T17:10:05.407536Z","shell.execute_reply":"2025-03-01T17:10:05.410282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"brain_activities = ['Seizure', 'GPD', 'LRDA', 'Other', 'GRDA', 'LPD']\nactivity_mapping = {activity: idx for idx, activity in enumerate(brain_activities)}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T17:10:05.412057Z","iopub.execute_input":"2025-03-01T17:10:05.412309Z","iopub.status.idle":"2025-03-01T17:10:05.437427Z","shell.execute_reply.started":"2025-03-01T17:10:05.412283Z","shell.execute_reply":"2025-03-01T17:10:05.436839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(f\"{BASE_DIR}train.csv\")\n# Split 80% Train, 20% Temp (Validation + Test)\ntrain_df, temp_df = train_test_split(df, test_size=0.2, random_state=42)\n\n# Split 10% Validation, 10% Test from Temp\nval_df, test_df = train_test_split(temp_df, test_size=0.5, random_state=42)\n\n# Save to CSV\ntrain_df.to_csv(\"train.csv\", index=False)\nval_df.to_csv(\"validation.csv\", index=False)\ntest_df.to_csv(\"test.csv\", index=False)\n\nprint(\"Splitting done! Train:\", len(train_df), \"Val:\", len(val_df), \"Test:\", len(test_df))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T17:10:05.438179Z","iopub.execute_input":"2025-03-01T17:10:05.438477Z","iopub.status.idle":"2025-03-01T17:10:06.117410Z","shell.execute_reply.started":"2025-03-01T17:10:05.438449Z","shell.execute_reply":"2025-03-01T17:10:06.116546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ChunkedBrainActivityDataset(Dataset):\n    def __init__(self, csv_file, base_dir, activity_mapping,md):\n        self.df = csv_file\n        self.base_dir = base_dir\n        self.activity_mapping = activity_mapping\n        self.resize_transform = transforms.Resize((224, 224))\n        self.md = md\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        spect_id, label, offset = self.df.iloc[idx][[\"spectrogram_id\", \"expert_consensus\", \"spectrogram_label_offset_seconds\"]]\n\n        temp_df = pd.read_parquet(f'{self.base_dir}/train_spectrograms/{spect_id}.parquet')\n        temp_df.drop(['time'], axis=1, inplace=True)\n\n        start = int(offset) // 2\n        temp_df = temp_df[start:start+300]\n        temp_df = np.log1p(temp_df)\n        temp_df /= temp_df.max()\n        temp_arr = np.nan_to_num(temp_df.to_numpy(), nan=1e-4)\n\n        # Use OpenCV to apply a colormap and convert to RGB\n        temp_arr_uint8 = np.uint8(255 * temp_arr)\n        rgb_image = cv2.applyColorMap(temp_arr_uint8, cv2.COLORMAP_JET)\n\n        # Normalize to [0, 1] and convert to tensor\n        rgb_image = rgb_image.astype(np.float32) / 255.0\n        rgb_image_tensor = torch.tensor(rgb_image).permute(2, 0, 1)  # (C, H, W)\n        rgb_image_tensor = self.resize_transform(rgb_image_tensor)\n            \n        y = self.activity_mapping[label]\n        y_tensor = torch.nn.functional.one_hot(torch.tensor(y, dtype=torch.long), num_classes=6).float()\n        \n        return rgb_image_tensor, y_tensor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T17:10:06.118328Z","iopub.execute_input":"2025-03-01T17:10:06.118653Z","iopub.status.idle":"2025-03-01T17:10:06.126128Z","shell.execute_reply.started":"2025-03-01T17:10:06.118622Z","shell.execute_reply":"2025-03-01T17:10:06.125523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Now create DataLoader with the chunked dataset\n# chunk_size = 1000  # Adjust chunk size according to memory constraints\n\ntrain_dataset = ChunkedBrainActivityDataset(csv_file=train_df, base_dir=BASE_DIR, activity_mapping=activity_mapping,md = \"lr\")\nval_dataset = ChunkedBrainActivityDataset(csv_file=val_df, base_dir=BASE_DIR, activity_mapping=activity_mapping,md = \"lr\")\ntest_dataset = ChunkedBrainActivityDataset(csv_file=test_df, base_dir=BASE_DIR, activity_mapping=activity_mapping,md = \"lr\")\n\ntrain_loader = DataLoader(train_dataset, batch_size=64, shuffle=True, num_workers=12, pin_memory=True, prefetch_factor=2)\nval_loader = DataLoader(val_dataset, batch_size=64, shuffle=False, num_workers=12, pin_memory=True, prefetch_factor=2)\ntest_loader = DataLoader(test_dataset, batch_size=64, shuffle=False, num_workers=12, pin_memory=True, prefetch_factor=2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T17:10:06.127687Z","iopub.execute_input":"2025-03-01T17:10:06.127894Z","iopub.status.idle":"2025-03-01T17:10:06.151666Z","shell.execute_reply.started":"2025-03-01T17:10:06.127876Z","shell.execute_reply":"2025-03-01T17:10:06.150938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import torch\n# import torch.nn as nn\n# import torch.optim as optim\n# from torch.utils.data import DataLoader\n\n# # Assuming your ChunkedBrainActivityDataset class and the DataLoader code\n# # for train_loader, val_loader, and test_loader are already defined.\n\n# # Define a logistic regression model for multi-class classification.\n# class LogisticRegressionModel(nn.Module):\n#     def __init__(self, input_dim, num_classes):\n#         super(LogisticRegressionModel, self).__init__()\n#         self.linear = nn.Linear(input_dim, num_classes)\n        \n#     def forward(self, x):\n#         # Flatten the input tensor: (batch, 3, 224, 224) => (batch, 3*224*224)\n#         x = x.view(x.size(0), -1)\n#         # Return the raw logits (CrossEntropyLoss applies softmax internally)\n#         return self.linear(x)\n\n# # Set device to GPU if available\n# device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# # Calculate the size of the flattened image.\n# # Your images are of shape: (3, 224, 224)\n# input_dim = 3 * 224 * 224\n# num_classes = 6\n\n# # Instantiate the model, loss function, and optimizer.\n# model = LogisticRegressionModel(input_dim, num_classes).to(device)\n\n# # CrossEntropyLoss expects integer labels, not one-hot vectors.\n# criterion = nn.CrossEntropyLoss()\n# optimizer = optim.SGD(model.parameters(), lr=0.01)\n\n# # Number of training epochs\n# num_epochs = 10\n\n# for epoch in range(num_epochs):\n#     model.train()\n#     running_loss = 0.0\n#     correct = 0\n#     total = 0\n#     for images, targets in train_loader:\n#         # Move the batch to the device\n#         images = images.to(device)\n#         targets = targets.to(device)\n#         # Convert one-hot targets to integer labels.\n#         labels = torch.argmax(targets, dim=1)\n\n#         optimizer.zero_grad()\n#         logits = model(images)\n#         loss = criterion(logits, labels)\n#         loss.backward()\n#         optimizer.step()\n\n#         running_loss += loss.item() * images.size(0)\n#         # Calculate accuracy for the batch\n#         _, preds = torch.max(logits, 1)\n#         total += labels.size(0)\n#         correct += (preds == labels).sum().item()\n        \n#     epoch_loss = running_loss / total\n#     epoch_acc = 100 * correct / total\n#     print(f\"Epoch [{epoch+1}/{num_epochs}], Loss: {epoch_loss:.4f}, Accuracy: {epoch_acc:.2f}%\")\n\n\n# model.eval() # Set the model to evaluation mode\n# correct_test = 0\n# total_test = 0\n\n# with torch.no_grad(): # Disables gradient calculation\n#     for images, targets in test_loader:\n#         images = images.to(device)\n#         targets = targets.to(device)\n#         # Convert one-hot target vectors to scalar class labels\n#         labels = torch.argmax(targets, dim=1)\n#         # Forward pass to get predictions\n#         logits = model(images)\n#         _, preds = torch.max(logits, dim=1)\n#         total_test += labels.size(0)\n#         correct_test += (preds == labels).sum().item()\n#     test_accuracy = 100 * correct_test / total_test\n#     print(f\"Test Accuracy: {test_accuracy:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T17:10:06.152462Z","iopub.execute_input":"2025-03-01T17:10:06.152647Z","iopub.status.idle":"2025-03-01T17:10:06.173581Z","shell.execute_reply.started":"2025-03-01T17:10:06.152630Z","shell.execute_reply":"2025-03-01T17:10:06.173001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom efficientnet_pytorch import EfficientNet\n\n# Define the model that uses EfficientNet as an encoder with logistic regression as the classifier\nclass EfficientNetEncoderLogisticRegression(nn.Module):\n    def __init__(self, num_classes=6):\n        super(EfficientNetEncoderLogisticRegression, self).__init__()\n        # Load the EfficientNet-B0 architecture\n        self.encoder = EfficientNet.from_name('efficientnet-b0')\n        # Save the number of output features before replacing the classifier\n        n_features = self.encoder._fc.in_features\n        # Remove the original classification head\n        self.encoder._fc = nn.Identity()\n        # Add a logistic regression layer for classification\n        self.logistic_regression = nn.Linear(n_features, num_classes)\n\n    def forward(self, x):\n        # Extract features using the pretrained encoder\n        features = self.encoder(x)\n        # Apply logistic regression to obtain class logits\n        logits = self.logistic_regression(features)\n        return logits\n\n# Set device\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Instantiate the model and move it to the appropriate device\nnum_classes = 6\nmodel = EfficientNetEncoderLogisticRegression(num_classes=num_classes).to(device)\n\n# Load the pretrained model weights\npretrained_path = \"/kaggle/input/cnn_efficientnet_v1/pytorch/default/1/HMS_model_v1_efficientnet_v2_s.pth\"  # Update with actual path\nstate_dict = torch.load(pretrained_path, map_location=device)\nmodel.load_state_dict(state_dict, strict=False)\n\n# Define the loss function and optimizer\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(model.parameters(), lr=0.001)\n\n# Training loop\nnum_epochs = 10\nfor epoch in range(num_epochs):\n    model.train()\n    running_loss = 0.0\n    correct = 0\n    total = 0\n    for images, targets in train_loader:\n        images = images.to(device)\n        targets = targets.to(device)\n        # Convert one-hot encoded target vectors to integer labels\n        labels = torch.argmax(targets, dim=1)\n        \n        optimizer.zero_grad()\n        logits = model(images)\n        loss = criterion(logits, labels)\n        loss.backward()\n        optimizer.step()\n        \n        running_loss += loss.item() * images.size(0)\n        _, preds = torch.max(logits, dim=1)\n        total += labels.size(0)\n        correct += (preds == labels).sum().item()\n    \n    epoch_loss = running_loss / total\n    epoch_acc = 100 * correct / total\n    print(f\"Epoch [{epoch+1}/{num_epochs}], Loss: {epoch_loss:.4f}, Accuracy: {epoch_acc:.2f}%\")\n\n# Evaluate on the test dataset\nmodel.eval()\ncorrect_test = 0\ntotal_test = 0\nwith torch.no_grad():\n    for images, targets in test_loader:\n        images = images.to(device)\n        targets = targets.to(device)\n        labels = torch.argmax(targets, dim=1)\n        logits = model(images)\n        _, preds = torch.max(logits, dim=1)\n        total_test += labels.size(0)\n        correct_test += (preds == labels).sum().item()\n\ntest_accuracy = 100 * correct_test / total_test\nprint(f\"Test Accuracy: {test_accuracy:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-01T17:13:21.497443Z","iopub.execute_input":"2025-03-01T17:13:21.497727Z","execution_failed":"2025-03-01T17:40:40.200Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}