{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":5819812,"sourceType":"datasetVersion","datasetId":3344234},{"sourceId":7781194,"sourceType":"datasetVersion","datasetId":4553461},{"sourceId":20907,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":17310},{"sourceId":20908,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":17311},{"sourceId":20909,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":17312},{"sourceId":20910,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":17313},{"sourceId":20911,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":17314}],"dockerImageVersionId":30665,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Setup","metadata":{}},{"cell_type":"code","source":"!pip uninstall timm -y\n!pip install /kaggle/input/timm-0613/timm-0.6.13-py3-none-any.whl -qq","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:22:47.895423Z","iopub.execute_input":"2024-03-27T22:22:47.895827Z","iopub.status.idle":"2024-03-27T22:23:26.364369Z","shell.execute_reply.started":"2024-03-27T22:22:47.895793Z","shell.execute_reply":"2024-03-27T22:23:26.363129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nimport timm\nfrom fastai.vision.all import *\nfrom fastcore.parallel import *\n\npath = Path('/kaggle/input/hms-harmful-brain-activity-classification')\n\npath.ls()","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-03-27T22:29:22.703824Z","iopub.execute_input":"2024-03-27T22:29:22.704239Z","iopub.status.idle":"2024-03-27T22:29:25.548839Z","shell.execute_reply.started":"2024-03-27T22:29:22.704189Z","shell.execute_reply":"2024-03-27T22:29:25.547687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Background","metadata":{}},{"cell_type":"markdown","source":"In this notebook I submit predictions from all 3-model ensembles of the 5 `swinv2_base_window12_192_22k` models I've trained.","metadata":{}},{"cell_type":"markdown","source":"## Submission Results","metadata":{}},{"cell_type":"markdown","source":"In an effort to save time (as I'm running out of days and submissions), I've only submitted 10 three-model combinations. Here are the results:\n\n|Ensemble|Public Score|\n|:-:|:-:|\n|||\n\n\n\nSummarizing the three top models:\n\n|Model Name|item method|item img size|batch_tfms|Public Score|\n|:-:|:-:|:-:|:-:|:-:|\n||||||","metadata":{}},{"cell_type":"markdown","source":"## Generate Test Data Images","metadata":{}},{"cell_type":"markdown","source":"My models are image classifiers trained on spectrogram images so I need to convert the test parquet data to images. I'm referencing the following notebooks:\n\n- [HMS - HBAC - Fastai Starter](https://www.kaggle.com/code/sonujha090/hms-hbac-fastai-starter)\n- [HMS-HBAC: KerasCV Starter Notebook](https://www.kaggle.com/code/awsaf49/hms-hbac-kerascv-starter-notebook)","metadata":{}},{"cell_type":"code","source":"# create temporary folders to hold spectrograms\nSPEC_DIR = \"/tmp/dataset/hms-hbac\"\nos.makedirs(SPEC_DIR+'/test_spectrograms', exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:29:42.129639Z","iopub.execute_input":"2024-03-27T22:29:42.130789Z","iopub.status.idle":"2024-03-27T22:29:42.135898Z","shell.execute_reply.started":"2024-03-27T22:29:42.130754Z","shell.execute_reply":"2024-03-27T22:29:42.134745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_spec(spec_id, split=\"train\"):\n    # read the data\n    data = pd.read_parquet(path/f'{split}_spectrograms'/f'{spec_id}.parquet')\n    \n    # replace NA with 0\n    data = data.fillna(0)\n    \n    # convert DataFrame to array\n    data = data.values[:, 1:]\n    \n    # transpose\n    data = data.T\n    data = data.astype(\"float32\")\n    \n    # convert array to PILImage\n    im = PILImage.create(Image.fromarray((data * 255).astype(np.uint8)))\n    im.save(f\"{SPEC_DIR}/{split}_spectrograms/{spec_id}.png\")","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:29:42.914156Z","iopub.execute_input":"2024-03-27T22:29:42.915160Z","iopub.status.idle":"2024-03-27T22:29:42.923535Z","shell.execute_reply.started":"2024-03-27T22:29:42.915117Z","shell.execute_reply":"2024-03-27T22:29:42.922374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(path/'test.csv')\ntest_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:29:44.693065Z","iopub.execute_input":"2024-03-27T22:29:44.693844Z","iopub.status.idle":"2024-03-27T22:29:44.724338Z","shell.execute_reply.started":"2024-03-27T22:29:44.693808Z","shell.execute_reply":"2024-03-27T22:29:44.723198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spec_ids = test_df['spectrogram_id'].unique()\nlen(spec_ids)","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:29:45.744596Z","iopub.execute_input":"2024-03-27T22:29:45.744949Z","iopub.status.idle":"2024-03-27T22:29:45.755872Z","shell.execute_reply.started":"2024-03-27T22:29:45.744922Z","shell.execute_reply":"2024-03-27T22:29:45.754917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"warnings.filterwarnings(\"ignore\")\nparallel(process_spec, spec_ids, split='test', n_workers=4)\nwarnings.filterwarnings(\"default\")","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:29:47.184892Z","iopub.execute_input":"2024-03-27T22:29:47.185298Z","iopub.status.idle":"2024-03-27T22:29:47.660429Z","shell.execute_reply.started":"2024-03-27T22:29:47.185265Z","shell.execute_reply":"2024-03-27T22:29:47.658778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PILImage.create(Path('/tmp/dataset/hms-hbac/test_spectrograms').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:29:48.302194Z","iopub.execute_input":"2024-03-27T22:29:48.302615Z","iopub.status.idle":"2024-03-27T22:29:48.371135Z","shell.execute_reply.started":"2024-03-27T22:29:48.302580Z","shell.execute_reply":"2024-03-27T22:29:48.370012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating the `DataLoaders` Object","metadata":{}},{"cell_type":"markdown","source":"I already have created a dataset with training images, so I'll load that into my notebook and create a training path `trn_path` to use in my `DataLoaders`.","metadata":{}},{"cell_type":"code","source":"trn_path = Path('/kaggle/input/hms-hbac-training-spectrogram-images/train_spectrograms')","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:29:51.380065Z","iopub.execute_input":"2024-03-27T22:29:51.380493Z","iopub.status.idle":"2024-03-27T22:29:51.386318Z","shell.execute_reply.started":"2024-03-27T22:29:51.380463Z","shell.execute_reply":"2024-03-27T22:29:51.385126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Each model was trained with different `item_tfms` and `batch_tfms` so I'll list them out here.","metadata":{}},{"cell_type":"code","source":"item1 = Resize((320,512), method='squish') # Model AU\nitem2 = Resize((320,512), method='squish') # Model AO\nitem3 = Resize((400,311), method='crop') # Model AB\nitem4 = Resize((320,512), method='squish') # Model BA\nitem5 = Resize((400,311), method='crop') # Model J","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:29:53.393535Z","iopub.execute_input":"2024-03-27T22:29:53.393939Z","iopub.status.idle":"2024-03-27T22:29:53.403236Z","shell.execute_reply.started":"2024-03-27T22:29:53.393897Z","shell.execute_reply":"2024-03-27T22:29:53.402194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch1 = aug_transforms(size=192, min_scale=0.75) # Model AU\nbatch2 = aug_transforms(size=192, min_scale=0.75)# Model AO\nbatch3 = RandomResizedCropGPU(size=192, min_scale=1.0)# Model AB\nbatch4 = aug_transforms(size=192, min_scale=0.75)# Model BA\nbatch5 = aug_transforms(size=192, min_scale=0.75) # Model J","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:29:59.221668Z","iopub.execute_input":"2024-03-27T22:29:59.222070Z","iopub.status.idle":"2024-03-27T22:29:59.238243Z","shell.execute_reply.started":"2024-03-27T22:29:59.222040Z","shell.execute_reply":"2024-03-27T22:29:59.237196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Ensemble","metadata":{}},{"cell_type":"markdown","source":"I'll create a list of `DataLoaders`, one for each of the five models that I am using in this ensemble.","metadata":{}},{"cell_type":"code","source":"dls_list = []\nitems = [item1, item2, item3, item4, item5]\nbatches = [batch1, batch2, batch3, batch4, batch5]\n\nfor i in range(5):\n    dls = ImageDataLoaders.from_folder(\n        trn_path, \n        valid_pct=0.2, \n        item_tfms=items[i],\n        batch_tfms=batches[i],\n        bs=16)\n    \n    dls_list.append(dls)","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:30:12.322166Z","iopub.execute_input":"2024-03-27T22:30:12.323156Z","iopub.status.idle":"2024-03-27T22:30:40.883262Z","shell.execute_reply.started":"2024-03-27T22:30:12.323117Z","shell.execute_reply":"2024-03-27T22:30:40.882123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls_list","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:30:43.129629Z","iopub.execute_input":"2024-03-27T22:30:43.130003Z","iopub.status.idle":"2024-03-27T22:30:43.137462Z","shell.execute_reply.started":"2024-03-27T22:30:43.129974Z","shell.execute_reply":"2024-03-27T22:30:43.136255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next, I'll create a list of `Learner`s, one for each model.","metadata":{}},{"cell_type":"code","source":"model_paths = [\n    Path('/kaggle/input/hms-hbac-swinv2_base_window12_192_22k/pytorch/au/1/hms_hbac_swinv2_base_window12_192_22k_AU'),\n    Path('/kaggle/input/hms-hbac-swinv2_base_window12_192_22k/pytorch/ao/1/hms_hbac_swinv2_base_window12_192_22k_AO'),\n    Path('/kaggle/input/hms-hbac-swinv2_base_window12_192_22k/pytorch/ab/1/hms_hbac_swinv2_base_window12_192_22k_AB'),\n    Path('/kaggle/input/hms-hbac-swinv2_base_window12_192_22k/pytorch/ba/1/hms_hbac_swinv2_base_window12_192_22k_BA'),\n    Path('/kaggle/input/hms-hbac-swinv2_base_window12_192_22k/pytorch/j/1/hms_hbac_swinv2_base_window12_192_22k_J')\n]","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:30:46.982294Z","iopub.execute_input":"2024-03-27T22:30:46.983144Z","iopub.status.idle":"2024-03-27T22:30:46.992286Z","shell.execute_reply.started":"2024-03-27T22:30:46.983106Z","shell.execute_reply":"2024-03-27T22:30:46.991325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learners = []\n\nfor idx, model_path in enumerate(model_paths):\n    learn = vision_learner(dls_list[idx], 'swinv2_base_window12_192_22k', pretrained=False)\n    learn.model_dir = '/kaggle/working/'\n    learn.load(model_path)\n    learners.append(learn)","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:30:52.702259Z","iopub.execute_input":"2024-03-27T22:30:52.703096Z","iopub.status.idle":"2024-03-27T22:31:22.197337Z","shell.execute_reply.started":"2024-03-27T22:30:52.703055Z","shell.execute_reply":"2024-03-27T22:31:22.196296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learners","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:31:25.021086Z","iopub.execute_input":"2024-03-27T22:31:25.021506Z","iopub.status.idle":"2024-03-27T22:31:25.028521Z","shell.execute_reply.started":"2024-03-27T22:31:25.021469Z","shell.execute_reply":"2024-03-27T22:31:25.027545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next, I'll get a `DataFrame` of probabilities for each test image from each `Learner` in my 3-model ensemble.","metadata":{}},{"cell_type":"code","source":"probs_df_list = []\n\ntst_files = get_image_files(SPEC_DIR+'/test_spectrograms')\n\nfor idx in range(5):\n    # create test DataLoader\n    tst_dl = dls_list[idx].test_dl(tst_files)\n    \n    # get TTA predictions\n    probs,_= learners[idx].tta(dl=tst_dl)\n    \n    # formatting\n    probs_df = pd.DataFrame(probs, columns=dls_list[idx].vocab)\n    probs_df['eeg_id'] = test_df['eeg_id']\n    probs_df = probs_df[['eeg_id', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']]\n    \n    probs_df_list.append(probs_df)","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:31:27.806427Z","iopub.execute_input":"2024-03-27T22:31:27.807343Z","iopub.status.idle":"2024-03-27T22:31:37.034449Z","shell.execute_reply.started":"2024-03-27T22:31:27.807298Z","shell.execute_reply":"2024-03-27T22:31:37.033253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Concatenate all `DataFrames`:","metadata":{}},{"cell_type":"code","source":"all_probs_df = pd.concat([probs_df_list[i] for i in [2, 3, 4]])\nall_probs_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:31:45.286911Z","iopub.execute_input":"2024-03-27T22:31:45.287322Z","iopub.status.idle":"2024-03-27T22:31:45.296405Z","shell.execute_reply.started":"2024-03-27T22:31:45.287288Z","shell.execute_reply":"2024-03-27T22:31:45.295393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Take the mean value of each vote by `eeg_id`","metadata":{}},{"cell_type":"code","source":"final_probs = all_probs_df.groupby('eeg_id').mean().reset_index()","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:31:47.765805Z","iopub.execute_input":"2024-03-27T22:31:47.766138Z","iopub.status.idle":"2024-03-27T22:31:47.776682Z","shell.execute_reply.started":"2024-03-27T22:31:47.766113Z","shell.execute_reply":"2024-03-27T22:31:47.775569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_probs.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:31:48.657815Z","iopub.execute_input":"2024-03-27T22:31:48.658252Z","iopub.status.idle":"2024-03-27T22:31:48.671572Z","shell.execute_reply.started":"2024-03-27T22:31:48.658206Z","shell.execute_reply":"2024-03-27T22:31:48.670656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And export it:","metadata":{}},{"cell_type":"code","source":"final_probs.to_csv('submission.csv', index=False)\n!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2024-03-27T22:31:49.841065Z","iopub.execute_input":"2024-03-27T22:31:49.841985Z","iopub.status.idle":"2024-03-27T22:31:50.901253Z","shell.execute_reply.started":"2024-03-27T22:31:49.841944Z","shell.execute_reply":"2024-03-27T22:31:50.900231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}