{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30665,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np\nimport pandas as pd\nimport os\nfrom fastai.vision.all import *\nfrom fastai.tabular.all import *\nfrom pathlib import Path\nfrom PIL import Image\nfrom fastcore.parallel import parallel\nfrom tqdm.notebook import tqdm\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-23T02:23:06.017505Z","iopub.execute_input":"2024-03-23T02:23:06.018352Z","iopub.status.idle":"2024-03-23T02:23:16.28648Z","shell.execute_reply.started":"2024-03-23T02:23:06.018318Z","shell.execute_reply":"2024-03-23T02:23:16.285628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_path = Path('/kaggle/input/hms-harmful-brain-activity-classification')\nbase_path.ls()","metadata":{"execution":{"iopub.status.busy":"2024-03-23T02:23:20.750713Z","iopub.execute_input":"2024-03-23T02:23:20.751082Z","iopub.status.idle":"2024-03-23T02:23:20.759033Z","shell.execute_reply.started":"2024-03-23T02:23:20.751055Z","shell.execute_reply":"2024-03-23T02:23:20.758116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#read contents \n#df_train_eegs = pd.read_parquet(base_path/'train_eegs')\n# df_sample_submission = pd.read_csv((base_path/'sample_submission.csv')\n#df_train_spectograms = pd.read_parquet(base_path/'train_spectrograms')\n# df_example_figures = pd.read((base_path/'example_figures')\n# df_test_eegs = pd.read_parquet(base_path/'test_eegs')\n# df_test_spectograms = pd.pd.read_parquet(base_path/'test_spectograms')\n# df_test_csv = pd.read_csv(base_path/'test.csv')  \ndf_train_csv = pd.read_csv(base_path/'train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-03-23T03:40:01.281813Z","iopub.execute_input":"2024-03-23T03:40:01.28286Z","iopub.status.idle":"2024-03-23T03:40:01.450032Z","shell.execute_reply.started":"2024-03-23T03:40:01.282816Z","shell.execute_reply":"2024-03-23T03:40:01.448993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#this code snippet sets up a directory structure for storing spectrogram data, creating separate directories for training and testing data within the main SPEC_DIR.\n\nSPEC_DIR = \"/tmp/dataset/hms-hbac\"\nEEG_DIR = \"/tmp/dataset/hms-hbac\"\nos.makedirs(SPEC_DIR+'/train_spectrograms', exist_ok=True)\nos.makedirs(SPEC_DIR+'/test_spectrograms', exist_ok=True)\nos.makedirs(EEG_DIR+'/train_eegs', exist_ok=True)\nos.makedirs(EEG_DIR+'/test_eegs', exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-23T02:23:28.193751Z","iopub.execute_input":"2024-03-23T02:23:28.194378Z","iopub.status.idle":"2024-03-23T02:23:28.201282Z","shell.execute_reply.started":"2024-03-23T02:23:28.194346Z","shell.execute_reply":"2024-03-23T02:23:28.200167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_spec(spec_id, split=\"train\"):\n    # Read the data\n    data = pd.read_parquet(base_path / f'{split}_spectrograms' / f'{spec_id}.parquet')\n    \n    # Replace NA with 0\n    data = data.fillna(0)\n    \n    # Convert DataFrame to array\n    data = data.values[:, 1:]\n    \n    # Transpose\n    data = data.T\n    data = data.astype(\"float32\")\n    \n    # Convert array to PILImage\n    im = PILImage.create(Image.fromarray((data * 255).astype(np.uint8)))\n    im.save(os.path.join(SPEC_DIR, f'{split}_spectrograms', f'{spec_id}.png'))","metadata":{"execution":{"iopub.status.busy":"2024-03-23T02:23:31.439088Z","iopub.execute_input":"2024-03-23T02:23:31.439947Z","iopub.status.idle":"2024-03-23T02:23:31.446883Z","shell.execute_reply.started":"2024-03-23T02:23:31.439915Z","shell.execute_reply":"2024-03-23T02:23:31.445677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_csv = pd.read_csv(base_path/'train.csv')\nspec_ids = df_train_csv[\"spectrogram_id\"].unique()\nlen(spec_ids)","metadata":{"execution":{"iopub.status.busy":"2024-03-23T02:23:34.208996Z","iopub.execute_input":"2024-03-23T02:23:34.209369Z","iopub.status.idle":"2024-03-23T02:23:34.469738Z","shell.execute_reply.started":"2024-03-23T02:23:34.209339Z","shell.execute_reply":"2024-03-23T02:23:34.468675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parallel(process_spec, spec_ids, split='train', n_workers=4)\n\n# It takes a function (process_spec), a list of spec_ids, and other arguments (split='train', n_workers=4) to execute the function in parallel.\n# The n_workers argument specifies the number of worker processes to use for parallel execution.","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\nfrom PIL import Image\nfrom IPython.display import display\n\n# Specify the directory containing the images\nimages_dir = Path('/tmp/dataset/hms-hbac/train_spectrograms')\n\n# Get the first 10 image paths\nimage_paths = list(images_dir.glob('*'))[:10]\n\n# Open and display each image\nfor image_path in image_paths:\n    with Image.open(image_path) as img:\n        # Convert the image to RGB mode if it's not already in that mode\n#         img = img.convert('RGB')\n        display(img)","metadata":{"execution":{"iopub.status.busy":"2024-03-23T03:27:21.903286Z","iopub.execute_input":"2024-03-23T03:27:21.904558Z","iopub.status.idle":"2024-03-23T03:27:22.15075Z","shell.execute_reply.started":"2024-03-23T03:27:21.904523Z","shell.execute_reply":"2024-03-23T03:27:22.149714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_dir.ls()","metadata":{"execution":{"iopub.status.busy":"2024-03-23T03:28:33.443592Z","iopub.execute_input":"2024-03-23T03:28:33.443947Z","iopub.status.idle":"2024-03-23T03:28:33.475897Z","shell.execute_reply.started":"2024-03-23T03:28:33.443921Z","shell.execute_reply":"2024-03-23T03:28:33.474831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FROM https://www.kaggle.com/code/vishalbakshi/hms-hbac-fastai-tta-submit\n# https://www.kaggle.com/code/vishalbakshi/hms-hbac-fastai-stacked-images-train\n\n# df_train_csv['img_path'] = '/tmp/dataset/hms-hbac/train_spectrograms/' + df_train_csv['spectrogram_id'].astype(str) + '.png'\n    \n# cols = ['eeg_id', 'spectrogram_id', 'img_path', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n# agg_funcs = {c: 'sum' for c in cols if 'vote' in c}\n\n# unique_df = df_train_csv[cols].groupby(['eeg_id', 'spectrogram_id', 'img_path'], as_index=False).agg(agg_funcs)\n# unique_df['target'] = unique_df[[c for c in cols if 'vote' in c]].idxmax(axis=1)\n    \n# train_bool = [False for _ in range(int(0.8 * len(unique_df)))]\n# valid_bool = [True for _ in range(len(unique_df) - int(0.8 * len(unique_df)))]\n# is_valid_bool = pd.Series(train_bool + valid_bool).sample(frac=1).reset_index(drop=True)\n\n# unique_df[\"is_valid\"] = is_valid_bool\n# unique_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-03-20T13:31:04.245319Z","iopub.execute_input":"2024-03-20T13:31:04.24559Z","iopub.status.idle":"2024-03-20T13:31:04.250058Z","shell.execute_reply.started":"2024-03-20T13:31:04.245566Z","shell.execute_reply":"2024-03-20T13:31:04.249282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dblock = DataBlock(\n#             blocks=(ImageBlock, CategoryBlock),\n#             splitter=ColSplitter(),\n#             get_x=ColReader('img_path'),\n#             get_y=ColReader('target'),\n#             item_tfms=Resize(224, method='squish'))\n\n# dls = dblock.dataloaders(unique_df)","metadata":{"execution":{"iopub.status.busy":"2024-03-20T13:31:04.251309Z","iopub.execute_input":"2024-03-20T13:31:04.251586Z","iopub.status.idle":"2024-03-20T13:31:04.263707Z","shell.execute_reply.started":"2024-03-20T13:31:04.251563Z","shell.execute_reply":"2024-03-20T13:31:04.262963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dls.show_batch()","metadata":{"execution":{"iopub.status.busy":"2024-03-20T13:31:04.264719Z","iopub.execute_input":"2024-03-20T13:31:04.265073Z","iopub.status.idle":"2024-03-20T13:31:04.27289Z","shell.execute_reply.started":"2024-03-20T13:31:04.265045Z","shell.execute_reply":"2024-03-20T13:31:04.272089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Define the columns for image paths and targets\ndf_train_csv['img_path'] = '/tmp/dataset/hms-hbac/train_spectrograms/' + df_train_csv['spectrogram_id'].astype(str) + '.png'\n\ntarget_cols = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']  # Adjust these to the actual column names containing the target labels\n\n# Define a function to convert image paths to complete paths\ndef get_full_img_path(img_name):\n    return images_dir / img_name\n\n# Update the DataBlock definition\ndblock = DataBlock(\n    blocks=(ImageBlock, MultiCategoryBlock),  \n    splitter=RandomSplitter(valid_pct=0.2, seed=42),  \n    get_x=Pipeline([ColReader('img_path'), get_full_img_path]),  \n    get_y=ColReader(target_cols),  \n    item_tfms=Resize(224, method='squish')\n)\n\n# Create the DataLoaders\ndls = dblock.dataloaders(df_train_csv)\n\n# Show a sample of the resulting DataLoaders\ndls.show_batch()","metadata":{"execution":{"iopub.status.busy":"2024-03-23T03:41:39.615934Z","iopub.execute_input":"2024-03-23T03:41:39.61654Z","iopub.status.idle":"2024-03-23T03:41:57.104256Z","shell.execute_reply.started":"2024-03-23T03:41:39.6165Z","shell.execute_reply":"2024-03-23T03:41:57.103237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TRYING TO USE DIRECTLY CSV TO SOFTMAX MULTICATEGORY BLOCK\n\n# Assuming df_train_csv is already loaded\n\n# Construct the path for image files\n# df_train_csv['img_path'] = '/tmp/dataset/hms-hbac/train_spectrograms/' + df_train_csv['spectrogram_id'].astype(str) + '.png'\n\n# Define the columns and construct the target\n# cols = ['eeg_id', 'spectrogram_id', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n# targets = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n# unique_df = df_train_csv[cols]\n# # unique_df = df_train_csv[targets]\n\n# # Split the dataset into train and validation sets\n# splitter = RandomSplitter(valid_pct=0.2, seed=42)\n# splits = splitter(range_of(unique_df))\n\n# Define the DataBlock\n# dblock = DataBlock(\n#     blocks=(ImageBlock, MultiCategoryBlock),  # Use MultiCategoryBlock for softmax\n#     splitter=IndexSplitter(splits[1]),  # Use the validation split\n#     get_x=ColReader('img_path'),\n#     get_y=[ColReader(col) for col in cols if 'vote' in col],  # Use each vote column as a separate label\n#     item_tfms=Resize(224, method='squish')\n# )\n\n# dblock = DataBlock(\n#     blocks=(ImageBlock, MultiCategoryBlock),  # Use MultiCategoryBlock for softmax\n#     splitter=RandomSplitter(valid_pct=0.2, seed=42),  # Use random splitter for validation set\n#     get_x=ColReader('img_path'),  # Specify path prefix\n#     get_y=[ColReader(col) for col in cols if 'vote' in col],  # Use each vote column as a separate label\n#     item_tfms=Resize(224, method='squish')\n# )\n\n\n# dblock = DataBlock(\n#     blocks=(ImageBlock, MultiCategoryBlock),  # Use MultiCategoryBlock for softmax\n#     splitter=RandomSplitter(valid_pct=0.2, seed=42),  # Use random splitter for validation set\n#     get_x=ColReader('img_path'),  # Specify path prefix\n#     get_y=[ColReader(col) for col in targets],  # Use each vote column as a separate label\n#     item_tfms=Resize(224, method='squish\n\n# dblock = DataBlock(\n#     blocks=(ImageBlock, MultiCategoryBlock),  # Use MultiCategoryBlock for softmax\n#     splitter=RandomSplitter(valid_pct=0.2, seed=42),  # Use random splitter for validation set\n#     get_x=ColReader('img_path'),  # Specify path prefix\n#     get_y=[ColReader(col) for col in targets],  # Use each vote column as a separate label\n#     item_tfms=Resize(224, method='squish')\n# )\n# # Create the DataLoaders\n# dls = dblock.dataloaders(unique_df)\n\n# dls = DataLoaders.from_df(df, valid_pct=0.2, seed=42,\n#                            fn_col='image_path',\n#                            label_col=['jimmy', 'johny', 'jane'])\n\n# Show a sample of the resulting DataLoaders\ndls.show_batch()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.vocab","metadata":{"execution":{"iopub.status.busy":"2024-03-23T03:43:10.555035Z","iopub.execute_input":"2024-03-23T03:43:10.55552Z","iopub.status.idle":"2024-03-23T03:43:10.563771Z","shell.execute_reply.started":"2024-03-23T03:43:10.555482Z","shell.execute_reply":"2024-03-23T03:43:10.562734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-03-23T04:49:22.21005Z","iopub.execute_input":"2024-03-23T04:49:22.21083Z","iopub.status.idle":"2024-03-23T04:49:22.216134Z","shell.execute_reply.started":"2024-03-23T04:49:22.210793Z","shell.execute_reply":"2024-03-23T04:49:22.214896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = vision_learner(dls, resnet34, metrics=error_rate).to_fp16() # Convert to mixed precision\nlearn.model = learn.model.to('cuda')  # Move model to GPU\n\n# Check the shapes of predictions and targets\nxb, yb = dls.one_batch()\npreds = learn.model(xb)\nprint(\"Shape of predictions (inp):\", preds.shape)\nprint(\"Shape of targets (targ):\", yb.shape)\n\n# ERROR : https://docs.fast.ai/learner.html#recorder\n","metadata":{"execution":{"iopub.status.busy":"2024-03-23T05:03:46.717362Z","iopub.execute_input":"2024-03-23T05:03:46.718386Z","iopub.status.idle":"2024-03-23T05:03:48.187315Z","shell.execute_reply.started":"2024-03-23T05:03:46.718345Z","shell.execute_reply":"2024-03-23T05:03:48.186272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Train the Model\nlearn.fit(2)","metadata":{"execution":{"iopub.status.busy":"2024-03-23T04:22:57.497847Z","iopub.execute_input":"2024-03-23T04:22:57.498245Z","iopub.status.idle":"2024-03-23T04:22:59.542283Z","shell.execute_reply.started":"2024-03-23T04:22:57.498201Z","shell.execute_reply":"2024-03-23T04:22:59.54032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = vision_learner(dls, 'resnet34' ,metrics=accuracy)","metadata":{"execution":{"iopub.status.busy":"2024-03-23T03:57:59.030976Z","iopub.execute_input":"2024-03-23T03:57:59.031896Z","iopub.status.idle":"2024-03-23T03:58:01.341839Z","shell.execute_reply.started":"2024-03-23T03:57:59.031852Z","shell.execute_reply":"2024-03-23T03:58:01.340828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-03-23T04:12:06.581456Z","iopub.execute_input":"2024-03-23T04:12:06.582015Z","iopub.status.idle":"2024-03-23T04:20:15.690541Z","shell.execute_reply.started":"2024-03-23T04:12:06.581969Z","shell.execute_reply":"2024-03-23T04:20:15.688503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.load('/kaggle/input/hms-hbac-resnet34/pytorch/2/1/hms_hbac_resnet34_stacked')","metadata":{"execution":{"iopub.status.busy":"2024-03-20T13:31:05.134675Z","iopub.status.idle":"2024-03-20T13:31:05.135062Z","shell.execute_reply.started":"2024-03-20T13:31:05.134874Z","shell.execute_reply":"2024-03-20T13:31:05.134887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-03-20T13:31:05.136517Z","iopub.status.idle":"2024-03-20T13:31:05.136815Z","shell.execute_reply.started":"2024-03-20T13:31:05.136668Z","shell.execute_reply":"2024-03-20T13:31:05.13668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['img_path'] = '/tmp/dataset/hms-hbac/test_spectrograms/' + test_df['spectrogram_id'].astype(str) + '.png'\ntst_dl = learn.dls.test_dl(test_df)\ntst_dl.show_batch()","metadata":{"execution":{"iopub.status.busy":"2024-03-20T13:31:05.137658Z","iopub.status.idle":"2024-03-20T13:31:05.138001Z","shell.execute_reply.started":"2024-03-20T13:31:05.137817Z","shell.execute_reply":"2024-03-20T13:31:05.13783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probs,_,idxs = learn.get_preds(dl=tst_dl, with_decoded=True)\nprobs_df = pd.DataFrame(probs, columns=dls.vocab)\nprobs_df['eeg_id'] = test_df['eeg_id']\nprobs_df = probs_df[['eeg_id', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']]\nprobs_df","metadata":{"execution":{"iopub.status.busy":"2024-03-20T13:31:05.139381Z","iopub.status.idle":"2024-03-20T13:31:05.139701Z","shell.execute_reply.started":"2024-03-20T13:31:05.139544Z","shell.execute_reply":"2024-03-20T13:31:05.139557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probs_df[['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']].sum(axis=1) == 1","metadata":{"execution":{"iopub.status.busy":"2024-03-20T13:31:05.140969Z","iopub.status.idle":"2024-03-20T13:31:05.141296Z","shell.execute_reply.started":"2024-03-20T13:31:05.141133Z","shell.execute_reply":"2024-03-20T13:31:05.141146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# probs_df.to_csv('submission.csv', index=False)\n# !head submission.csv","metadata":{"execution":{"iopub.status.busy":"2024-03-20T13:31:05.142483Z","iopub.status.idle":"2024-03-20T13:31:05.14278Z","shell.execute_reply.started":"2024-03-20T13:31:05.142632Z","shell.execute_reply":"2024-03-20T13:31:05.142644Z"},"trusted":true},"execution_count":null,"outputs":[]}]}