{"metadata":{"kernelspec":{"display_name":"Python 3 (ipykernel)","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.9.13"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install timm huggingface_hub kaggle -Uqq\n!jupyter notebook --ServerApp.iopub_data_rate_limit=1.0e10","metadata":{"execution":{"iopub.execute_input":"2024-03-19T04:17:43.246684Z","iopub.status.busy":"2024-03-19T04:17:43.246059Z","iopub.status.idle":"2024-03-19T04:17:47.607066Z","shell.execute_reply":"2024-03-19T04:17:47.606287Z","shell.execute_reply.started":"2024-03-19T04:17:43.246616Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nimport timm\nimport gc\n\nfrom fastai.vision.all import *\nfrom fastcore.parallel import *\n\n#path = Path('/kaggle/input/hms-hbac-training-spectrogram-images/train_spectrograms')\npath = Path('/notebooks/hms-hbac-training-spectrogram-images/train_spectrograms')\npath.ls()","metadata":{"execution":{"iopub.execute_input":"2024-03-19T04:30:27.087283Z","iopub.status.busy":"2024-03-19T04:30:27.087030Z","iopub.status.idle":"2024-03-19T04:30:29.029708Z","shell.execute_reply":"2024-03-19T04:30:29.029086Z","shell.execute_reply.started":"2024-03-19T04:30:27.087264Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Background","metadata":{}},{"cell_type":"markdown","source":"In this notebook, I'll train the 5 `vit_small_patch16_224` variants that performed well (consistently), and export them so I can upload them to Kaggle and use them for submission.\n\nA key difference in the trainings in this notebook from Part 1 is that I will **not** use a random `seed` to split the data into training and validation sets because I want each model to use a different validation set as I'll eventually be ensembling these models for submission.\n\nA key difference between my `convnext_small_in22k` trainings and these `vit` ones is that progressive resizing performed very well for `vit`, but did not for `convnext`. All five models for `vit` use progressive resizing.","metadata":{}},{"cell_type":"code","source":"arch = 'vit_small_patch16_224'","metadata":{"execution":{"iopub.execute_input":"2024-03-19T04:45:37.567440Z","iopub.status.busy":"2024-03-19T04:45:37.566921Z","iopub.status.idle":"2024-03-19T04:45:37.570199Z","shell.execute_reply":"2024-03-19T04:45:37.569584Z","shell.execute_reply.started":"2024-03-19T04:45:37.567417Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def progressive_resizing(fn, sizes, method, batch, arch='vit_small_patch16_224', pad=PadMode.Reflection):\n    dls = ImageDataLoaders.from_folder(\n            path, \n            valid_pct=0.2, \n            item_tfms=Resize(sizes[0], method=method, pad_mode=pad),\n            batch_tfms=batch,\n            bs=64//4)\n    \n    cbs = GradientAccumulation(64)\n    learn = vision_learner(dls, arch, metrics=error_rate, cbs=cbs).to_fp16()\n    learn.fine_tune(4, 0.01)\n\n    dls = ImageDataLoaders.from_folder(\n            path, \n            valid_pct=0.2, \n            item_tfms=Resize(sizes[1], method=method, pad_mode=pad),\n            batch_tfms=batch,\n            bs=64//4)\n\n    learn.dls = dls\n    learn.fine_tune(7, 0.01)\n\n    dls = ImageDataLoaders.from_folder(\n            path, \n            valid_pct=0.2, \n            item_tfms=Resize(sizes[2], method=method, pad_mode=pad),\n            batch_tfms=batch,\n            bs=64//4)\n\n    learn.dls = dls\n    learn.fine_tune(10, 0.01)\n    \n    print(error_rate(*learn.tta(dl=dls.valid)))\n    learn.save(fn, with_opt=False)","metadata":{"execution":{"iopub.execute_input":"2024-03-19T04:49:11.567812Z","iopub.status.busy":"2024-03-19T04:49:11.567289Z","iopub.status.idle":"2024-03-19T04:49:11.573240Z","shell.execute_reply":"2024-03-19T04:49:11.572546Z","shell.execute_reply.started":"2024-03-19T04:49:11.567791Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model BA","metadata":{}},{"cell_type":"code","source":"progressive_resizing(\n    fn=f\"/notebooks/hms_hbac_{arch}_BA\", \n    sizes=[256, 400, (320, 512)], \n    method='squish', \n    batch=aug_transforms(size=224, min_scale=0.75))","metadata":{"execution":{"iopub.execute_input":"2024-03-19T04:50:21.241681Z","iopub.status.busy":"2024-03-19T04:50:21.241383Z","iopub.status.idle":"2024-03-19T05:04:36.014798Z","shell.execute_reply":"2024-03-19T05:04:36.013944Z","shell.execute_reply.started":"2024-03-19T04:50:21.241662Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model AO","metadata":{}},{"cell_type":"code","source":"progressive_resizing(\n    fn=f\"/notebooks/hms_hbac_{arch}_AO\", \n    sizes=[256, (311,400), (320,512)], \n    method='squish', \n    batch=aug_transforms(size=224, min_scale=0.75))","metadata":{"execution":{"iopub.execute_input":"2024-03-19T05:04:36.016844Z","iopub.status.busy":"2024-03-19T05:04:36.016341Z","iopub.status.idle":"2024-03-19T05:18:34.654255Z","shell.execute_reply":"2024-03-19T05:18:34.653667Z","shell.execute_reply.started":"2024-03-19T05:04:36.016821Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model AU","metadata":{}},{"cell_type":"code","source":"progressive_resizing(\n    fn=f\"/notebooks/hms_hbac_{arch}_AU\", \n    sizes=[256, (400,311), (320,512)], \n    method='squish', \n    batch=aug_transforms(size=224, min_scale=0.75))","metadata":{"execution":{"iopub.execute_input":"2024-03-19T05:18:34.655518Z","iopub.status.busy":"2024-03-19T05:18:34.655176Z","iopub.status.idle":"2024-03-19T05:32:30.510305Z","shell.execute_reply":"2024-03-19T05:32:30.509567Z","shell.execute_reply.started":"2024-03-19T05:18:34.655496Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model BB","metadata":{}},{"cell_type":"code","source":"progressive_resizing(\n    fn=f\"/notebooks/hms_hbac_{arch}_BB\", \n    sizes=[256, 400, (320,512)], \n    method=ResizeMethod.Pad, \n    batch=aug_transforms(size=224, min_scale=0.75),\n    pad=PadMode.Zeros)","metadata":{"execution":{"iopub.execute_input":"2024-03-19T05:32:30.512156Z","iopub.status.busy":"2024-03-19T05:32:30.511928Z","iopub.status.idle":"2024-03-19T05:46:30.235170Z","shell.execute_reply":"2024-03-19T05:46:30.234167Z","shell.execute_reply.started":"2024-03-19T05:32:30.512137Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model AW","metadata":{}},{"cell_type":"code","source":"progressive_resizing(\n    fn=f\"/notebooks/hms_hbac_{arch}_AW\", \n    sizes=[256, (400,311), (320,512)], \n    method='crop', \n    batch=RandomResizedCropGPU(size=224, min_scale=1.0))","metadata":{"execution":{"iopub.execute_input":"2024-03-19T05:46:30.241125Z","iopub.status.busy":"2024-03-19T05:46:30.240970Z","iopub.status.idle":"2024-03-19T05:58:10.723114Z","shell.execute_reply":"2024-03-19T05:58:10.721990Z","shell.execute_reply.started":"2024-03-19T05:46:30.241107Z"}},"execution_count":null,"outputs":[]}]}