{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11828260,"sourceType":"datasetVersion","datasetId":7430593},{"sourceId":11839829,"sourceType":"datasetVersion","datasetId":7438816},{"sourceId":11867185,"sourceType":"datasetVersion","datasetId":7457365},{"sourceId":11922453,"sourceType":"datasetVersion","datasetId":7495659},{"sourceId":411821,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":336229,"modelId":357232}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":27.447062,"end_time":"2025-03-12T14:13:11.647927","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-03-12T14:12:44.200865","version":"2.6.0"},"colab":{"provenance":[]},"widgets":{"application/vnd.jupyter.widget-state+json":{"57220fdd59bb4f9d8360df121be5328e":{"model_module":"@jupyter-widgets/controls","model_name":"VBoxModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"VBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"VBoxView","box_style":"","children":["IPY_MODEL_e499ced412e04b7fb86f8ed650d1ff6c"],"layout":"IPY_MODEL_0b07f820d42045ddaf976a8eaed8d779"}},"643cc4eb7a694b3bb0da6c36ca1a29ed":{"model_module":"@jupyter-widgets/controls","model_name":"HTMLModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_18598ab629c64cce83fa80af8869fa1a","placeholder":"​","style":"IPY_MODEL_c4e95a948e29429ea07aca43c68744b3","value":"<center> <img\nsrc=https://www.kaggle.com/static/images/site-logo.png\nalt='Kaggle'> <br> Create an API token from <a\nhref=\"https://www.kaggle.com/settings/account\" target=\"_blank\">your Kaggle\nsettings page</a> and paste it below along with your Kaggle username. <br> </center>"}},"975cd359fe014832a92ca385dae5afc5":{"model_module":"@jupyter-widgets/controls","model_name":"TextModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"TextModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"TextView","continuous_update":true,"description":"Username:","description_tooltip":null,"disabled":false,"layout":"IPY_MODEL_cbd2ba87ece440248b89160a96188eca","placeholder":"​","style":"IPY_MODEL_96ef698decbe438da2f81f0fdf89ff1c","value":"miki42v"}},"4d4491695e5640618786195e4bbde385":{"model_module":"@jupyter-widgets/controls","model_name":"PasswordModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"PasswordModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"PasswordView","continuous_update":true,"description":"Token:","description_tooltip":null,"disabled":false,"layout":"IPY_MODEL_7084260339b9470bbcf3447c66db464b","placeholder":"​","style":"IPY_MODEL_0f3cc82bfb644100933aa2cf017d8325","value":""}},"b868c1b57ded4b48a0907accc7b859f8":{"model_module":"@jupyter-widgets/controls","model_name":"ButtonModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ButtonModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ButtonView","button_style":"","description":"Login","disabled":false,"icon":"","layout":"IPY_MODEL_0315f352bcac4505ab9703c4b6690ebe","style":"IPY_MODEL_9b55d0842b5e4e3fb24094013e00826a","tooltip":""}},"bd4eb897af844a7ba1ad86659837ff37":{"model_module":"@jupyter-widgets/controls","model_name":"HTMLModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_cc4d29ed89f944b4b59e4cf65a7c318e","placeholder":"​","style":"IPY_MODEL_ec1589582cd64764b5c2979c8455e2ac","value":"\n<b>Thank You</b></center>"}},"0b07f820d42045ddaf976a8eaed8d779":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":"center","align_self":null,"border":null,"bottom":null,"display":"flex","flex":null,"flex_flow":"column","grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"50%"}},"18598ab629c64cce83fa80af8869fa1a":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"c4e95a948e29429ea07aca43c68744b3":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"cbd2ba87ece440248b89160a96188eca":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"96ef698decbe438da2f81f0fdf89ff1c":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"7084260339b9470bbcf3447c66db464b":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"0f3cc82bfb644100933aa2cf017d8325":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"0315f352bcac4505ab9703c4b6690ebe":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"9b55d0842b5e4e3fb24094013e00826a":{"model_module":"@jupyter-widgets/controls","model_name":"ButtonStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ButtonStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","button_color":null,"font_weight":""}},"cc4d29ed89f944b4b59e4cf65a7c318e":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"ec1589582cd64764b5c2979c8455e2ac":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"904753394845431888fabac56079e4a3":{"model_module":"@jupyter-widgets/controls","model_name":"LabelModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"LabelModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"LabelView","description":"","description_tooltip":null,"layout":"IPY_MODEL_2c72a56766c74fc88b8f9173031caf2a","placeholder":"​","style":"IPY_MODEL_bf6602179fb14313911019176528df18","value":"Connecting..."}},"2c72a56766c74fc88b8f9173031caf2a":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"bf6602179fb14313911019176528df18":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"e499ced412e04b7fb86f8ed650d1ff6c":{"model_module":"@jupyter-widgets/controls","model_name":"LabelModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"LabelModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"LabelView","description":"","description_tooltip":null,"layout":"IPY_MODEL_4c74f0d2d9c148e3998b23dae237a8c3","placeholder":"​","style":"IPY_MODEL_5c943751b66c41b59c97eb96b4ec2c59","value":"Kaggle credentials successfully validated."}},"4c74f0d2d9c148e3998b23dae237a8c3":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"5c943751b66c41b59c97eb96b4ec2c59":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"ed88de1542b145b3905f975ad69f225a":{"model_module":"@jupyter-widgets/controls","model_name":"HBoxModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_5310117fb78142cfbda281856babf4bc","IPY_MODEL_741a439c0e4d46bdb1ed643bd69dd8fd","IPY_MODEL_8fe00a940056455c9c015f384e661883"],"layout":"IPY_MODEL_f7d38553139b43d0b8b3da898f03f76f"}},"5310117fb78142cfbda281856babf4bc":{"model_module":"@jupyter-widgets/controls","model_name":"HTMLModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_bbc7a28c3e184552a3b1c9f4a7e4f625","placeholder":"​","style":"IPY_MODEL_b0c567d4133943e0a55aa19ce71b70a5","value":"Downloading 1 files: 100%"}},"741a439c0e4d46bdb1ed643bd69dd8fd":{"model_module":"@jupyter-widgets/controls","model_name":"FloatProgressModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"success","description":"","description_tooltip":null,"layout":"IPY_MODEL_28cd892b41c1433b8eb3c96884a98f9f","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_23c89f7605814eca99558641c31f1f8f","value":1}},"8fe00a940056455c9c015f384e661883":{"model_module":"@jupyter-widgets/controls","model_name":"HTMLModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_f4b7ec40bc3b47ccb07b2ca099ab70dc","placeholder":"​","style":"IPY_MODEL_10ca23680e0b48a28159d461349e2168","value":" 1/1 [00:03&lt;00:00,  3.35s/it]"}},"f7d38553139b43d0b8b3da898f03f76f":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"bbc7a28c3e184552a3b1c9f4a7e4f625":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"b0c567d4133943e0a55aa19ce71b70a5":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"28cd892b41c1433b8eb3c96884a98f9f":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"23c89f7605814eca99558641c31f1f8f":{"model_module":"@jupyter-widgets/controls","model_name":"ProgressStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"f4b7ec40bc3b47ccb07b2ca099ab70dc":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"10ca23680e0b48a28159d461349e2168":{"model_module":"@jupyter-widgets/controls","model_name":"DescriptionStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}}}}},"nbformat_minor":0,"nbformat":4,"cells":[{"source":"# IMPORTANT: SOME KAGGLE DATA SOURCES ARE PRIVATE\n# RUN THIS CELL IN ORDER TO IMPORT YOUR KAGGLE DATA SOURCES.\nimport kagglehub\nkagglehub.login()\n","metadata":{"colab":{"base_uri":"https://localhost:8080/","height":84,"referenced_widgets":["57220fdd59bb4f9d8360df121be5328e","643cc4eb7a694b3bb0da6c36ca1a29ed","975cd359fe014832a92ca385dae5afc5","4d4491695e5640618786195e4bbde385","b868c1b57ded4b48a0907accc7b859f8","bd4eb897af844a7ba1ad86659837ff37","0b07f820d42045ddaf976a8eaed8d779","18598ab629c64cce83fa80af8869fa1a","c4e95a948e29429ea07aca43c68744b3","cbd2ba87ece440248b89160a96188eca","96ef698decbe438da2f81f0fdf89ff1c","7084260339b9470bbcf3447c66db464b","0f3cc82bfb644100933aa2cf017d8325","0315f352bcac4505ab9703c4b6690ebe","9b55d0842b5e4e3fb24094013e00826a","cc4d29ed89f944b4b59e4cf65a7c318e","ec1589582cd64764b5c2979c8455e2ac","904753394845431888fabac56079e4a3","2c72a56766c74fc88b8f9173031caf2a","bf6602179fb14313911019176528df18","e499ced412e04b7fb86f8ed650d1ff6c","4c74f0d2d9c148e3998b23dae237a8c3","5c943751b66c41b59c97eb96b4ec2c59"]},"id":"H4OTB2uJDiNl","outputId":"49269324-b035-4379-fa87-f2a8b28bbb5d"},"cell_type":"code","outputs":[{"output_type":"display_data","data":{"text/plain":["VBox(children=(HTML(value='<center> <img\\nsrc=https://www.kaggle.com/static/images/site-logo.png\\nalt=\\'Kaggle…"],"application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"57220fdd59bb4f9d8360df121be5328e"}},"metadata":{}},{"output_type":"stream","name":"stdout","text":["Kaggle credentials set.\n","Kaggle credentials successfully validated.\n"]}],"execution_count":1},{"source":"# IMPORTANT: RUN THIS CELL IN ORDER TO IMPORT YOUR KAGGLE DATA SOURCES,\n# THEN FEEL FREE TO DELETE THIS CELL.\n# NOTE: THIS NOTEBOOK ENVIRONMENT DIFFERS FROM KAGGLE'S PYTHON\n# ENVIRONMENT SO THERE MAY BE MISSING LIBRARIES USED BY YOUR\n# NOTEBOOK.\n\nbirdclef_2025_path = kagglehub.competition_download('birdclef-2025')\nmyso1987_birdclef_2025_sed_models_p_path = kagglehub.dataset_download('myso1987/birdclef-2025-sed-models-p')\nchandanreddy10_convnext_xcl_transformers_model_path = kagglehub.dataset_download('chandanreddy10/convnext-xcl-transformers-model')\nhideyukizushi_bird25_d_330v2_ppv15_convnextv2_nano_path = kagglehub.dataset_download('hideyukizushi/bird25-d-330v2-ppv15-convnextv2-nano')\nvineetdairashri_mapping_path = kagglehub.dataset_download('vineetdairashri/mapping')\nvineetdairashri_convnext_db_new_tensorflow2_default_1_path = kagglehub.model_download('vineetdairashri/convnext_db_new/TensorFlow2/default/1')\n\nprint('Data source import complete.')\n","metadata":{"id":"_8GvhMZkDiNq","colab":{"base_uri":"https://localhost:8080/","height":760,"referenced_widgets":["ed88de1542b145b3905f975ad69f225a","5310117fb78142cfbda281856babf4bc","741a439c0e4d46bdb1ed643bd69dd8fd","8fe00a940056455c9c015f384e661883","f7d38553139b43d0b8b3da898f03f76f","bbc7a28c3e184552a3b1c9f4a7e4f625","b0c567d4133943e0a55aa19ce71b70a5","28cd892b41c1433b8eb3c96884a98f9f","23c89f7605814eca99558641c31f1f8f","f4b7ec40bc3b47ccb07b2ca099ab70dc","10ca23680e0b48a28159d461349e2168"]},"outputId":"6a4c6318-a906-49ae-f135-ee40265dd29a"},"cell_type":"code","outputs":[{"output_type":"stream","name":"stdout","text":["Downloading from https://www.kaggle.com/api/v1/competitions/data/download-all/birdclef-2025...\n"]},{"output_type":"stream","name":"stderr","text":["100%|██████████| 11.5G/11.5G [01:52<00:00, 110MB/s]"]},{"output_type":"stream","name":"stdout","text":["Extracting files...\n"]},{"output_type":"stream","name":"stderr","text":["\n"]},{"output_type":"stream","name":"stdout","text":["Downloading from https://www.kaggle.com/api/v1/datasets/download/myso1987/birdclef-2025-sed-models-p?dataset_version_number=1...\n"]},{"output_type":"stream","name":"stderr","text":["100%|██████████| 299M/299M [00:02<00:00, 149MB/s]"]},{"output_type":"stream","name":"stdout","text":["Extracting files...\n"]},{"output_type":"stream","name":"stderr","text":["\n"]},{"output_type":"stream","name":"stdout","text":["Downloading from https://www.kaggle.com/api/v1/datasets/download/chandanreddy10/convnext-xcl-transformers-model?dataset_version_number=1...\n"]},{"output_type":"stream","name":"stderr","text":["100%|██████████| 346M/346M [00:03<00:00, 99.5MB/s]"]},{"output_type":"stream","name":"stdout","text":["Extracting files...\n"]},{"output_type":"stream","name":"stderr","text":["\n"]},{"output_type":"stream","name":"stdout","text":["Downloading from https://www.kaggle.com/api/v1/datasets/download/hideyukizushi/bird25-d-330v2-ppv15-convnextv2-nano?dataset_version_number=1...\n"]},{"output_type":"stream","name":"stderr","text":["100%|██████████| 320M/320M [00:03<00:00, 102MB/s]"]},{"output_type":"stream","name":"stdout","text":["Extracting files...\n"]},{"output_type":"stream","name":"stderr","text":["\n"]},{"output_type":"stream","name":"stdout","text":["Downloading from https://www.kaggle.com/api/v1/datasets/download/vineetdairashri/mapping?dataset_version_number=1...\n"]},{"output_type":"stream","name":"stderr","text":["100%|██████████| 1.45k/1.45k [00:00<00:00, 2.88MB/s]"]},{"output_type":"stream","name":"stdout","text":["Extracting files...\n"]},{"output_type":"stream","name":"stderr","text":["\n"]},{"output_type":"display_data","data":{"text/plain":["Downloading 1 files:   0%|          | 0/1 [00:00<?, ?it/s]"],"application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"ed88de1542b145b3905f975ad69f225a"}},"metadata":{}},{"output_type":"stream","name":"stdout","text":["Downloading from https://www.kaggle.com/api/v1/models/vineetdairashri/convnext_db_new/TensorFlow2/default/1/download/birdset_epoch1.pt...\n"]},{"output_type":"stream","name":"stderr","text":["\n","  0%|          | 0.00/335M [00:00<?, ?B/s]\u001b[A\n","  2%|▏         | 6.00M/335M [00:00<00:05, 60.8MB/s]\u001b[A\n","  6%|▌         | 19.0M/335M [00:00<00:03, 101MB/s] \u001b[A\n"," 10%|█         | 35.0M/335M [00:00<00:02, 129MB/s]\u001b[A\n"," 14%|█▍        | 48.0M/335M [00:00<00:02, 106MB/s]\u001b[A\n"," 18%|█▊        | 59.0M/335M [00:00<00:02, 104MB/s]\u001b[A\n"," 22%|██▏       | 75.0M/335M [00:00<00:02, 122MB/s]\u001b[A\n"," 26%|██▋       | 88.0M/335M [00:00<00:02, 100MB/s]\u001b[A\n"," 30%|██▉       | 99.0M/335M [00:01<00:02, 82.9MB/s]\u001b[A\n"," 32%|███▏      | 108M/335M [00:01<00:02, 82.5MB/s] \u001b[A\n"," 36%|███▋      | 122M/335M [00:01<00:02, 95.9MB/s]\u001b[A\n"," 39%|███▉      | 132M/335M [00:01<00:02, 95.5MB/s]\u001b[A\n"," 42%|████▏     | 142M/335M [00:01<00:02, 93.1MB/s]\u001b[A\n"," 47%|████▋     | 158M/335M [00:01<00:01, 108MB/s] \u001b[A\n"," 53%|█████▎    | 176M/335M [00:01<00:01, 129MB/s]\u001b[A\n"," 58%|█████▊    | 193M/335M [00:01<00:01, 140MB/s]\u001b[A\n"," 64%|██████▎   | 213M/335M [00:01<00:00, 158MB/s]\u001b[A\n"," 69%|██████▊   | 230M/335M [00:02<00:00, 162MB/s]\u001b[A\n"," 74%|███████▍  | 248M/335M [00:02<00:00, 168MB/s]\u001b[A\n"," 80%|███████▉  | 267M/335M [00:02<00:00, 175MB/s]\u001b[A\n"," 85%|████████▍ | 284M/335M [00:02<00:00, 170MB/s]\u001b[A\n"," 90%|█████████ | 303M/335M [00:02<00:00, 145MB/s]\u001b[A\n","100%|██████████| 335M/335M [00:02<00:00, 128MB/s]"]},{"output_type":"stream","name":"stdout","text":["Data source import complete.\n"]},{"output_type":"stream","name":"stderr","text":["\n"]}],"execution_count":2},{"cell_type":"markdown","source":"<h1 style=\"color: #6cb4e4;  text-align: center;  padding: 0.25em;  border-top: solid 2.5px #6cb4e4;  border-bottom: solid 2.5px #6cb4e4;  background: -webkit-repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);  background: repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);height:45px;\">\n<b>\nOnly Submission(LoadLocalTrainModel)\n</b></h1>","metadata":{"id":"0tAK4vHGDiNr"}},{"cell_type":"markdown","source":"### **ℹ️INFO** Ensemble Learning\n* This notebook is an weighted blend(convnextv2 * 0.6 + nfnet * 0.4).\n    * **GreatWork LB.850(nfnet)** https://www.kaggle.com/code/myso1987/post-processing-with-power-adjustment-for-low-rank\n    * **LocalTrainModel(convnext_DB)** Pretrained model on Bird Data which was more finetuned on 206 species\n\n","metadata":{"id":"R79vKcm8DiNu"}},{"cell_type":"markdown","source":"<h1 style=\"color: #6cb4e4;  text-align: center;  padding: 0.25em;  border-top: solid 2.5px #6cb4e4;  border-bottom: solid 2.5px #6cb4e4;  background: -webkit-repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);  background: repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);height:45px;\">\n<b>\n《《《Submission1(convnext_BD)》》》\n</b></h1>","metadata":{"id":"qgMg4ZXGDiNv"}},{"cell_type":"code","source":"from torch import nn\nimport torch\nimport torchaudio\nimport librosa\n\nfrom torchvision import transforms\nfrom torchaudio.transforms import Resample\nimport datasets\nimport warnings\nimport pandas as pd\nimport json\nimport os\nimport glob\nfrom typing import Dict, Optional\nimport random\nimport numpy as np\nimport datasets\nimport torch\nfrom torch import nn\nfrom transformers import AutoConfig, ConvNextForImageClassification","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:06:59.27171Z","iopub.execute_input":"2025-05-25T17:06:59.272078Z","iopub.status.idle":"2025-05-25T17:07:31.389737Z","shell.execute_reply.started":"2025-05-25T17:06:59.272046Z","shell.execute_reply":"2025-05-25T17:07:31.388398Z"},"id":"p3tNLN0CDiNw"},"outputs":[],"execution_count":3},{"cell_type":"code","source":"class Config:\n    train_dir = \"/kaggle/input/birdclef-2025/train_audio\"\n    seed = 42\n    train_csv = \"/kaggle/input/birdclef-2025/train.csv\"\n    train_soundscapes = \"/kaggle/input/birdclef-2025/train_soundscapes\"\n    test_soundscapes = \"/kaggle/input/birdclef-2025/test_soundscapes/\"\n    sample_submission_csv = \"/kaggle/input/birdclef-2025/sample_submission.csv\"\n    num_classes = 206\n    submission_mode = len(glob.glob(\"/kaggle/input/birdclef-2025/test_soundscapes/*.ogg\")) > 0\n\n\ncheckpoint_path = \"/kaggle/input/convnext_db_new/tensorflow2/default/1/birdset_epoch1.pt\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:31.391328Z","iopub.execute_input":"2025-05-25T17:07:31.392202Z","iopub.status.idle":"2025-05-25T17:07:31.40201Z","shell.execute_reply.started":"2025-05-25T17:07:31.392134Z","shell.execute_reply":"2025-05-25T17:07:31.400248Z"},"id":"bK54b8KRDiNw"},"outputs":[],"execution_count":4},{"cell_type":"code","source":"def set_seed(seed: int=42):\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n\n    if torch.cuda.is_available():\n        torch.cuda.manual_seed(seed)\n        torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic=True\n    torch.backends.cudnn.benchmark=False\n\n    print(f\"Setting seed : {seed}\")\nset_seed()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:31.404198Z","iopub.execute_input":"2025-05-25T17:07:31.404627Z","iopub.status.idle":"2025-05-25T17:07:31.448515Z","shell.execute_reply.started":"2025-05-25T17:07:31.404587Z","shell.execute_reply":"2025-05-25T17:07:31.44694Z"},"id":"UYfaJwXzDiNx","colab":{"base_uri":"https://localhost:8080/"},"outputId":"0276526a-08c1-4d00-e3ad-798900370904"},"outputs":[{"output_type":"stream","name":"stdout","text":["Setting seed : 42\n"]}],"execution_count":5},{"cell_type":"code","source":"import os\nimport torch\nimport torchaudio\nfrom torchaudio.transforms import Resample\n\ndef preprocess_audio(audio_path, window_size=5, sample_rate=32000):\n    audio, sr = torchaudio.load(audio_path, normalize=True)\n    if sr != sample_rate:\n        audio = Resample(orig_freq=sr, new_freq=sample_rate)(audio)\n\n    window_size_samples = window_size * sample_rate\n    one_sec_samples = sample_rate\n\n    total_samples = audio.size(1)\n    num_segments = (total_samples) // window_size_samples  # ceil division\n\n    segments = {}\n    for i in range(num_segments):\n        start = i * window_size_samples\n        end = start + window_size_samples\n        segment = audio[:, start:end]\n\n        if segment.size(1) < window_size_samples:\n            # Get last 1 second\n            repeat_part = segment[:, -one_sec_samples:] if segment.size(1) >= one_sec_samples else segment\n            repeats_needed = (window_size_samples - segment.size(1) + one_sec_samples - 1) // one_sec_samples\n            repeated = repeat_part.repeat(1, repeats_needed)\n            segment = torch.cat([segment, repeated], dim=1)[:, :window_size_samples]  # Trim extra if needed\n\n        end_second = (i + 1) * window_size\n        filename_key = f\"{os.path.splitext(os.path.basename(audio_path))[0]}_{end_second}\"\n        segment =  segment.mean(dim=0, keepdim=True)\n        segments[filename_key] = segment.squeeze(-1)\n\n    return segments","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:31.450462Z","iopub.execute_input":"2025-05-25T17:07:31.450878Z","iopub.status.idle":"2025-05-25T17:07:31.459544Z","shell.execute_reply.started":"2025-05-25T17:07:31.450836Z","shell.execute_reply":"2025-05-25T17:07:31.458059Z"},"id":"tzT0WEgTDiNy"},"outputs":[],"execution_count":6},{"cell_type":"code","source":"class PowerToDB(nn.Module):\n    def __init__(self, ref=1.0, amin=1e-10, top_db=80.0):\n        super(PowerToDB, self).__init__()\n        # Initialize parameters\n        self.ref = ref\n        self.amin = amin\n        self.top_db = top_db\n\n    def forward(self, S):\n        # Convert S to a PyTorch tensor if it is not already\n        S = torch.as_tensor(S, dtype=torch.float32)\n\n        if self.amin <= 0:\n            raise ValueError(\"amin must be strictly positive\")\n\n        if torch.is_complex(S):\n            warnings.warn(\n                \"power_to_db was called on complex input so phase \"\n                \"information will be discarded. To suppress this warning, \"\n                \"call power_to_db(S.abs()**2) instead.\",\n                stacklevel=2,\n            )\n            magnitude = S.abs()\n        else:\n            magnitude = S\n\n        # Check if ref is a callable function or a scalar\n        if callable(self.ref):\n            ref_value = self.ref(magnitude)\n        else:\n            ref_value = torch.abs(torch.tensor(self.ref, dtype=S.dtype))\n\n        # Compute the log spectrogram\n        log_spec = 10.0 * torch.log10(\n            torch.maximum(magnitude, torch.tensor(self.amin, device=magnitude.device))\n        )\n        log_spec -= 10.0 * torch.log10(\n            torch.maximum(ref_value, torch.tensor(self.amin, device=magnitude.device))\n        )\n\n        # Apply top_db threshold if necessary\n        if self.top_db is not None:\n            if self.top_db < 0:\n                raise ValueError(\"top_db must be non-negative\")\n            log_spec = torch.maximum(log_spec, log_spec.max() - self.top_db)\n\n        return log_spec","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:31.460594Z","iopub.execute_input":"2025-05-25T17:07:31.460941Z","iopub.status.idle":"2025-05-25T17:07:31.482727Z","shell.execute_reply.started":"2025-05-25T17:07:31.46091Z","shell.execute_reply":"2025-05-25T17:07:31.481523Z"},"id":"IJYJmkhHDiNz"},"outputs":[],"execution_count":7},{"cell_type":"code","source":"class ConvNextClassifier(nn.Module):\n    \"\"\"\n    ConvNext model for audio classification.\n    \"\"\"\n\n    def __init__(\n        self,\n        num_channels: int = 1,\n        num_classes: Optional[int] = None,\n        checkpoint: Optional[str] = None,\n        local_checkpoint: Optional[str] = None,\n        cache_dir: Optional[str] = None,\n        pretrain_info= None #PretrainInfoConfig = None,\n    ):\n        \"\"\"\n        Note: Either num_classes or pretrain_info must be given\n        Args:\n            num_channels: Number of input channels.\n            checkpoint: huggingface checkpoint path of any model of correct type\n            num_classes: number of classification heads to be used in the model\n            local_checkpoint: local path to checkpoint file\n            cache_dir: specified cache dir to save model files at\n            pretrain_info: hf_path and hf_name of info will be used to infer if num_classes is None\n        \"\"\"\n        super().__init__()\n\n        if pretrain_info:\n            self.hf_path = pretrain_info.hf_path\n            self.hf_name = (\n                pretrain_info.hf_name\n                if not pretrain_info.hf_pretrain_name\n                else pretrain_info.hf_pretrain_name\n            )\n            self.num_classes = len(\n                datasets.load_dataset_builder(self.hf_path, self.hf_name)\n                .info.features[\"ebird_code\"]\n                .names\n            )\n        else:\n            self.hf_path = None\n            self.hf_name = None\n            self.num_classes = num_classes\n\n        self.num_channels = num_channels\n        self.checkpoint = checkpoint\n        self.local_checkpoint = local_checkpoint\n        self.cache_dir = cache_dir\n\n        self.model = None\n\n        self._initialize_model()\n\n    def _initialize_model(self):\n        \"\"\"Initializes the ConvNext model based on specified attributes.\"\"\"\n\n        adjusted_state_dict = None\n\n        if self.checkpoint:\n            if self.local_checkpoint:\n                state_dict = torch.load(self.local_checkpoint)[\"state_dict\"]\n\n                # Update this part to handle the necessary key replacements\n                adjusted_state_dict = {}\n                for key, value in state_dict.items():\n                    # Handle 'model.model.' prefix\n                    new_key = key.replace(\"model.model.\", \"\")\n\n                    # Handle 'model._orig_mod.model.' prefix\n                    new_key = new_key.replace(\"model._orig_mod.model.\", \"\")\n\n                    # Assign the adjusted key\n                    adjusted_state_dict[new_key] = value\n\n            self.model = ConvNextForImageClassification.from_pretrained(\n                self.checkpoint,\n                num_labels=self.num_classes,\n                num_channels=self.num_channels,\n                cache_dir=self.cache_dir,\n                state_dict=adjusted_state_dict,\n                ignore_mismatched_sizes=True,\n            )\n        else:\n            config = AutoConfig.from_pretrained(\n                \"/kaggle/input/fb/pytorch/default/1/convnext-base-224-22k\",\n                num_labels=self.num_classes,\n                num_channels=self.num_channels,\n            )\n            self.model = ConvNextForImageClassification(config)\n\n    def forward(\n        self, input_values: torch.Tensor, labels: Optional[torch.Tensor] = None\n    ) -> torch.Tensor:\n        \"\"\"\n        Defines the forward pass of the ConvNext model.\n\n        Args:\n            input_values (torch.Tensor): An input batch.\n            labels (Optional[torch.Tensor]): The corresponding labels. Default is None.\n\n        Returns:\n            torch.Tensor: The output of the ConvNext model.\n        \"\"\"\n        output = self.model(input_values)\n        logits = output.logits\n\n        return logits\n\n    @torch.inference_mode()\n    def get_logits(self, dataloader, device):\n        pass\n\n    @torch.inference_mode()\n    def get_probas(self, dataloader, device):\n        pass\n\n    @torch.inference_mode()\n    def get_representations(self, dataloader, device):\n        pass","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:31.483794Z","iopub.execute_input":"2025-05-25T17:07:31.484215Z","iopub.status.idle":"2025-05-25T17:07:31.51486Z","shell.execute_reply.started":"2025-05-25T17:07:31.484153Z","shell.execute_reply":"2025-05-25T17:07:31.513469Z"},"id":"BoAGPMJIDiN0"},"outputs":[],"execution_count":8},{"cell_type":"code","source":"class ConvNextBirdSet(nn.Module):\n    \"\"\"\n    BirdSet ConvNext model trained on BirdSet XCL dataset.\n    The model expects a raw 1 channel 5s waveform with sample rate of 32kHz as an input.\n    Its preprocess function will:\n        - convert the waveform to a spectrogram: n_fft: 1024, hop_length: 320, power: 2.0\n        - melscale the spectrogram: n_mels: 128, n_stft: 513\n        - dbscale with top_db: 80\n        - normalize the spectrogram mean: -4.268, std: 4.569 (from esc-50)\n    \"\"\"\n\n    def __init__(\n        self,\n        PowerToDB,\n        num_classes=9736,\n    ):\n        super().__init__()\n        self.model = ConvNextClassifier(\n            checkpoint=\"/kaggle/input/convnext-xcl-transformers-model\",\n            num_classes=num_classes,\n        )\n        self.spectrogram_converter = torchaudio.transforms.Spectrogram(\n            n_fft=1024, hop_length=320, power=2.0\n        )\n        self.mel_converter = torchaudio.transforms.MelScale(\n            n_mels=128, n_stft=513, sample_rate=32_000\n        )\n        self.normalizer = transforms.Normalize((-4.268,), (4.569,))\n        self.powerToDB = PowerToDB(top_db=80)\n        self.config = self.model.model.config\n\n    def preprocess(self, waveform: torch.Tensor):\n        # convert waveform to spectrogram\n        spectrogram = self.spectrogram_converter(waveform)\n        spectrogram = spectrogram.to(torch.float32)\n        melspec = self.mel_converter(spectrogram)\n        dbscale = self.powerToDB(melspec)\n        normalized_dbscale = self.normalizer(dbscale)\n        # add dimension 3 from left\n        normalized_dbscale = normalized_dbscale.unsqueeze(-3)\n\n        return normalized_dbscale\n\n    def forward(self, input: torch.Tensor):\n        # spectrogram = self.preprocess(waveform)\n        return self.model(input)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:31.515671Z","iopub.execute_input":"2025-05-25T17:07:31.515968Z","iopub.status.idle":"2025-05-25T17:07:31.533425Z","shell.execute_reply.started":"2025-05-25T17:07:31.515943Z","shell.execute_reply":"2025-05-25T17:07:31.532336Z"},"id":"oDPJ3G0hDiN1"},"outputs":[],"execution_count":9},{"cell_type":"code","source":"audio_dir = glob.glob(Config.test_soundscapes + \"/*.ogg\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:31.536152Z","iopub.execute_input":"2025-05-25T17:07:31.536528Z","iopub.status.idle":"2025-05-25T17:07:31.558716Z","shell.execute_reply.started":"2025-05-25T17:07:31.536499Z","shell.execute_reply":"2025-05-25T17:07:31.557274Z"},"id":"wf3eaRhiDiN2"},"outputs":[],"execution_count":10},{"cell_type":"code","source":"import torch\nimport os\n\n# ✅ Set device\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# ✅ Define model wrapper\nclass BirdsetModule(torch.nn.Module):\n    def __init__(self, PowerToDB, num_classes=206):\n        super().__init__()\n        # Ensure ConvNextBirdSet does NOT try to use HuggingFace Hub if loading locally\n        self.model = ConvNextBirdSet(PowerToDB, num_classes=num_classes)\n\n    def forward(self, x):\n        preprocessed = self.model.preprocess(x)\n        return self.model(preprocessed)\n\n# ✅ Path to local checkpoint (update filename if needed)\ncheckpoint_path = \"/kaggle/input/convnext-xcl-transformers-model/model.pth\"\n\n# ✅ Instantiate model\nmodel = BirdsetModule(PowerToDB, num_classes=206).to(device)\n\n# ✅ Load weights safely from local path\nstate_dict = torch.load(checkpoint_path, map_location=device)\nmodel.load_state_dict(state_dict)\n\n# ✅ Set to evaluation mode\nmodel.eval()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:31.560979Z","iopub.execute_input":"2025-05-25T17:07:31.561341Z","iopub.status.idle":"2025-05-25T17:07:37.018404Z","shell.execute_reply.started":"2025-05-25T17:07:31.561311Z","shell.execute_reply":"2025-05-25T17:07:37.016364Z"},"id":"2Ln14K5DDiN3","colab":{"base_uri":"https://localhost:8080/","height":477},"outputId":"fa511cc4-3e86-42d7-fa77-a24d30f11325"},"outputs":[{"output_type":"error","ename":"HFValidationError","evalue":"Repo id must be in the form 'repo_name' or 'namespace/repo_name': '/kaggle/input/convnext-xcl-transformers-model'. Use `repo_type` argument if needed.","traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mHFValidationError\u001b[0m                         Traceback (most recent call last)","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/transformers/utils/hub.py\u001b[0m in \u001b[0;36mcached_files\u001b[0;34m(path_or_repo_id, filenames, cache_dir, force_download, resume_download, proxies, token, revision, local_files_only, subfolder, repo_type, user_agent, _raise_exceptions_for_gated_repo, _raise_exceptions_for_missing_entries, _raise_exceptions_for_connection_errors, _commit_hash, **deprecated_kwargs)\u001b[0m\n\u001b[1;32m    469\u001b[0m             \u001b[0;31m# This is slightly better for only 1 file\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 470\u001b[0;31m             hf_hub_download(\n\u001b[0m\u001b[1;32m    471\u001b[0m                 \u001b[0mpath_or_repo_id\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/huggingface_hub/utils/_validators.py\u001b[0m in \u001b[0;36m_inner_fn\u001b[0;34m(*args, **kwargs)\u001b[0m\n\u001b[1;32m    105\u001b[0m             \u001b[0;32mif\u001b[0m \u001b[0marg_name\u001b[0m \u001b[0;32min\u001b[0m \u001b[0;34m[\u001b[0m\u001b[0;34m\"repo_id\"\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m\"from_id\"\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m\"to_id\"\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 106\u001b[0;31m                 \u001b[0mvalidate_repo_id\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0marg_value\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m    107\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/huggingface_hub/utils/_validators.py\u001b[0m in \u001b[0;36mvalidate_repo_id\u001b[0;34m(repo_id)\u001b[0m\n\u001b[1;32m    153\u001b[0m     \u001b[0;32mif\u001b[0m \u001b[0mrepo_id\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mcount\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m\"/\"\u001b[0m\u001b[0;34m)\u001b[0m \u001b[0;34m>\u001b[0m \u001b[0;36m1\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 154\u001b[0;31m         raise HFValidationError(\n\u001b[0m\u001b[1;32m    155\u001b[0m             \u001b[0;34m\"Repo id must be in the form 'repo_name' or 'namespace/repo_name':\"\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;31mHFValidationError\u001b[0m: Repo id must be in the form 'repo_name' or 'namespace/repo_name': '/kaggle/input/convnext-xcl-transformers-model'. Use `repo_type` argument if needed.","\nDuring handling of the above exception, another exception occurred:\n","\u001b[0;31mHFValidationError\u001b[0m                         Traceback (most recent call last)","\u001b[0;32m<ipython-input-15-7e20f3264b7e>\u001b[0m in \u001b[0;36m<cell line: 0>\u001b[0;34m()\u001b[0m\n\u001b[1;32m     20\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     21\u001b[0m \u001b[0;31m# ✅ Instantiate model\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m---> 22\u001b[0;31m \u001b[0mmodel\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mBirdsetModule\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mPowerToDB\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mnum_classes\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;36m206\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mto\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mdevice\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m     23\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     24\u001b[0m \u001b[0;31m# ✅ Load weights safely from local path\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m<ipython-input-15-7e20f3264b7e>\u001b[0m in \u001b[0;36m__init__\u001b[0;34m(self, PowerToDB, num_classes)\u001b[0m\n\u001b[1;32m     10\u001b[0m         \u001b[0msuper\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0m__init__\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     11\u001b[0m         \u001b[0;31m# Ensure ConvNextBirdSet does NOT try to use HuggingFace Hub if loading locally\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m---> 12\u001b[0;31m         \u001b[0mself\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mmodel\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mConvNextBirdSet\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mPowerToDB\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mnum_classes\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mnum_classes\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m     13\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     14\u001b[0m     \u001b[0;32mdef\u001b[0m \u001b[0mforward\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mself\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mx\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m<ipython-input-9-d8d21a87456c>\u001b[0m in \u001b[0;36m__init__\u001b[0;34m(self, PowerToDB, num_classes)\u001b[0m\n\u001b[1;32m     16\u001b[0m     ):\n\u001b[1;32m     17\u001b[0m         \u001b[0msuper\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0m__init__\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m---> 18\u001b[0;31m         self.model = ConvNextClassifier(\n\u001b[0m\u001b[1;32m     19\u001b[0m             \u001b[0mcheckpoint\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;34m\"/kaggle/input/convnext-xcl-transformers-model\"\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     20\u001b[0m             \u001b[0mnum_classes\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mnum_classes\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m<ipython-input-8-c5845ddb6376>\u001b[0m in \u001b[0;36m__init__\u001b[0;34m(self, num_channels, num_classes, checkpoint, local_checkpoint, cache_dir, pretrain_info)\u001b[0m\n\u001b[1;32m     49\u001b[0m         \u001b[0mself\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mmodel\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0;32mNone\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     50\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m---> 51\u001b[0;31m         \u001b[0mself\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0m_initialize_model\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m     52\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     53\u001b[0m     \u001b[0;32mdef\u001b[0m \u001b[0m_initialize_model\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mself\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m<ipython-input-8-c5845ddb6376>\u001b[0m in \u001b[0;36m_initialize_model\u001b[0;34m(self)\u001b[0m\n\u001b[1;32m     72\u001b[0m                     \u001b[0madjusted_state_dict\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0mnew_key\u001b[0m\u001b[0;34m]\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mvalue\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     73\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m---> 74\u001b[0;31m             self.model = ConvNextForImageClassification.from_pretrained(\n\u001b[0m\u001b[1;32m     75\u001b[0m                 \u001b[0mself\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mcheckpoint\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     76\u001b[0m                 \u001b[0mnum_labels\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mself\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mnum_classes\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/transformers/modeling_utils.py\u001b[0m in \u001b[0;36m_wrapper\u001b[0;34m(*args, **kwargs)\u001b[0m\n\u001b[1;32m    307\u001b[0m         \u001b[0mold_dtype\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mtorch\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mget_default_dtype\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    308\u001b[0m         \u001b[0;32mtry\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 309\u001b[0;31m             \u001b[0;32mreturn\u001b[0m \u001b[0mfunc\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m*\u001b[0m\u001b[0margs\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m**\u001b[0m\u001b[0mkwargs\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m    310\u001b[0m         \u001b[0;32mfinally\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    311\u001b[0m             \u001b[0mtorch\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mset_default_dtype\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mold_dtype\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/transformers/modeling_utils.py\u001b[0m in \u001b[0;36mfrom_pretrained\u001b[0;34m(cls, pretrained_model_name_or_path, config, cache_dir, ignore_mismatched_sizes, force_download, local_files_only, token, revision, use_safetensors, weights_only, *model_args, **kwargs)\u001b[0m\n\u001b[1;32m   4210\u001b[0m             \u001b[0;32mif\u001b[0m \u001b[0;32mnot\u001b[0m \u001b[0misinstance\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mconfig\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mPretrainedConfig\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m   4211\u001b[0m                 \u001b[0;31m# We make a call to the config file first (which may be absent) to get the commit hash as soon as possible\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m-> 4212\u001b[0;31m                 resolved_config_file = cached_file(\n\u001b[0m\u001b[1;32m   4213\u001b[0m                     \u001b[0mpretrained_model_name_or_path\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m   4214\u001b[0m                     \u001b[0mCONFIG_NAME\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/transformers/utils/hub.py\u001b[0m in \u001b[0;36mcached_file\u001b[0;34m(path_or_repo_id, filename, **kwargs)\u001b[0m\n\u001b[1;32m    310\u001b[0m     \u001b[0;31m`\u001b[0m\u001b[0;31m`\u001b[0m\u001b[0;31m`\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    311\u001b[0m     \"\"\"\n\u001b[0;32m--> 312\u001b[0;31m     \u001b[0mfile\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mcached_files\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mpath_or_repo_id\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mpath_or_repo_id\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mfilenames\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0mfilename\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m**\u001b[0m\u001b[0mkwargs\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m    313\u001b[0m     \u001b[0mfile\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mfile\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0;36m0\u001b[0m\u001b[0;34m]\u001b[0m \u001b[0;32mif\u001b[0m \u001b[0mfile\u001b[0m \u001b[0;32mis\u001b[0m \u001b[0;32mnot\u001b[0m \u001b[0;32mNone\u001b[0m \u001b[0;32melse\u001b[0m \u001b[0mfile\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    314\u001b[0m     \u001b[0;32mreturn\u001b[0m \u001b[0mfile\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/transformers/utils/hub.py\u001b[0m in \u001b[0;36mcached_files\u001b[0;34m(path_or_repo_id, filenames, cache_dir, force_download, resume_download, proxies, token, revision, local_files_only, subfolder, repo_type, user_agent, _raise_exceptions_for_gated_repo, _raise_exceptions_for_missing_entries, _raise_exceptions_for_connection_errors, _commit_hash, **deprecated_kwargs)\u001b[0m\n\u001b[1;32m    520\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    521\u001b[0m         \u001b[0;31m# Now we try to recover if we can find all files correctly in the cache\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 522\u001b[0;31m         resolved_files = [\n\u001b[0m\u001b[1;32m    523\u001b[0m             \u001b[0m_get_cache_file_to_return\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mpath_or_repo_id\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mfilename\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mcache_dir\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mrevision\u001b[0m\u001b[0;34m)\u001b[0m \u001b[0;32mfor\u001b[0m \u001b[0mfilename\u001b[0m \u001b[0;32min\u001b[0m \u001b[0mfull_filenames\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    524\u001b[0m         ]\n","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/transformers/utils/hub.py\u001b[0m in \u001b[0;36m<listcomp>\u001b[0;34m(.0)\u001b[0m\n\u001b[1;32m    521\u001b[0m         \u001b[0;31m# Now we try to recover if we can find all files correctly in the cache\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    522\u001b[0m         resolved_files = [\n\u001b[0;32m--> 523\u001b[0;31m             \u001b[0m_get_cache_file_to_return\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mpath_or_repo_id\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mfilename\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mcache_dir\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mrevision\u001b[0m\u001b[0;34m)\u001b[0m \u001b[0;32mfor\u001b[0m \u001b[0mfilename\u001b[0m \u001b[0;32min\u001b[0m \u001b[0mfull_filenames\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m    524\u001b[0m         ]\n\u001b[1;32m    525\u001b[0m         \u001b[0;32mif\u001b[0m \u001b[0mall\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mfile\u001b[0m \u001b[0;32mis\u001b[0m \u001b[0;32mnot\u001b[0m \u001b[0;32mNone\u001b[0m \u001b[0;32mfor\u001b[0m \u001b[0mfile\u001b[0m \u001b[0;32min\u001b[0m \u001b[0mresolved_files\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/transformers/utils/hub.py\u001b[0m in \u001b[0;36m_get_cache_file_to_return\u001b[0;34m(path_or_repo_id, full_filename, cache_dir, revision)\u001b[0m\n\u001b[1;32m    138\u001b[0m ):\n\u001b[1;32m    139\u001b[0m     \u001b[0;31m# We try to see if we have a cached version (not up to date):\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 140\u001b[0;31m     \u001b[0mresolved_file\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mtry_to_load_from_cache\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mpath_or_repo_id\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mfull_filename\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mcache_dir\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mcache_dir\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mrevision\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mrevision\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m    141\u001b[0m     \u001b[0;32mif\u001b[0m \u001b[0mresolved_file\u001b[0m \u001b[0;32mis\u001b[0m \u001b[0;32mnot\u001b[0m \u001b[0;32mNone\u001b[0m \u001b[0;32mand\u001b[0m \u001b[0mresolved_file\u001b[0m \u001b[0;34m!=\u001b[0m \u001b[0m_CACHED_NO_EXIST\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    142\u001b[0m         \u001b[0;32mreturn\u001b[0m \u001b[0mresolved_file\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/huggingface_hub/utils/_validators.py\u001b[0m in \u001b[0;36m_inner_fn\u001b[0;34m(*args, **kwargs)\u001b[0m\n\u001b[1;32m    104\u001b[0m         ):\n\u001b[1;32m    105\u001b[0m             \u001b[0;32mif\u001b[0m \u001b[0marg_name\u001b[0m \u001b[0;32min\u001b[0m \u001b[0;34m[\u001b[0m\u001b[0;34m\"repo_id\"\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m\"from_id\"\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m\"to_id\"\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 106\u001b[0;31m                 \u001b[0mvalidate_repo_id\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0marg_value\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m    107\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    108\u001b[0m             \u001b[0;32melif\u001b[0m \u001b[0marg_name\u001b[0m \u001b[0;34m==\u001b[0m \u001b[0;34m\"token\"\u001b[0m \u001b[0;32mand\u001b[0m \u001b[0marg_value\u001b[0m \u001b[0;32mis\u001b[0m \u001b[0;32mnot\u001b[0m \u001b[0;32mNone\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/usr/local/lib/python3.11/dist-packages/huggingface_hub/utils/_validators.py\u001b[0m in \u001b[0;36mvalidate_repo_id\u001b[0;34m(repo_id)\u001b[0m\n\u001b[1;32m    152\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    153\u001b[0m     \u001b[0;32mif\u001b[0m \u001b[0mrepo_id\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mcount\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m\"/\"\u001b[0m\u001b[0;34m)\u001b[0m \u001b[0;34m>\u001b[0m \u001b[0;36m1\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 154\u001b[0;31m         raise HFValidationError(\n\u001b[0m\u001b[1;32m    155\u001b[0m             \u001b[0;34m\"Repo id must be in the form 'repo_name' or 'namespace/repo_name':\"\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    156\u001b[0m             \u001b[0;34mf\" '{repo_id}'. Use `repo_type` argument if needed.\"\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;31mHFValidationError\u001b[0m: Repo id must be in the form 'repo_name' or 'namespace/repo_name': '/kaggle/input/convnext-xcl-transformers-model'. Use `repo_type` argument if needed."]}],"execution_count":15},{"cell_type":"code","source":"def inference(model, audio_tensor, device):\n    model.eval()\n    with torch.no_grad():\n        audio_tensor = audio_tensor.to(device)\n        logits = model(audio_tensor)  # add batch dim\n        probs = torch.softmax(logits, dim=1)\n        #top_probs, top_indices = torch.topk(probs, k=top_k)\n        return probs.squeeze(0).tolist() #top_indices.squeeze(0).tolist(), top_probs.squeeze(0).tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:37.019608Z","iopub.execute_input":"2025-05-25T17:07:37.019952Z","iopub.status.idle":"2025-05-25T17:07:37.026061Z","shell.execute_reply.started":"2025-05-25T17:07:37.019922Z","shell.execute_reply":"2025-05-25T17:07:37.024557Z"},"id":"VNr5W8ndDiN3"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(\"/kaggle/input/mapping/map.txt\",\"r\") as file:\n    content = file.read()\nprint(\"Taking \",audio_dir)\nmap_text = json.loads(content)\nprobabilities=[]\nrows = []\nfor index, audio_path in enumerate(audio_dir):\n\n    audio_tensor_dict = preprocess_audio(audio_path)\n    for key, value in audio_tensor_dict.items():\n        probs = inference(model,value,device)\n        probabilities.append(probs)\n        rows.append(key)\ncolumns = list(map_text.values())\ndf = pd.DataFrame(probabilities,columns=columns)\ndf.insert(0,\"row_id\",rows)\nprint(\"Writing Submission...\")\ndf.to_csv(\"submission0.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:37.027506Z","iopub.execute_input":"2025-05-25T17:07:37.027948Z","iopub.status.idle":"2025-05-25T17:07:37.089197Z","shell.execute_reply.started":"2025-05-25T17:07:37.027904Z","shell.execute_reply":"2025-05-25T17:07:37.0878Z"},"id":"oYSwBAr8DiN3"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n\n    test_soundscapes = '/kaggle/input/birdclef-2025/test_soundscapes'\n    submission_csv = '/kaggle/input/birdclef-2025/sample_submission.csv'\n    taxonomy_csv = '/kaggle/input/birdclef-2025/taxonomy.csv'\n\n    # ------------------------------------------- #\n    # [IMPORTANT]\n    # * Melspectrogram & Audio Params\n    # ------------------------------------------- #\n    FS = 32000\n    WINDOW_SIZE = 5\n    N_FFT = 2048\n    HOP_LENGTH = 512\n    N_MELS = 512\n    FMIN = 20\n    FMAX = 16000\n    TARGET_SHAPE = (256, 256)\n\n    # ------------------------------------------- #\n    # * Model def\n    # ------------------------------------------- #\n    model_path = '/kaggle/input/bird25-d-330v2-ppv15-convnextv2-nano'\n    model_name = 'convnextv2_nano.fcmae_ft_in22k_in1k'\n    use_specific_folds = True\n    folds = [0,1]\n    in_channels = 1\n    device = 'cpu'\n\n    # Inference parameters\n    batch_size = 16\n    use_tta = False\n    tta_count = 3\n    threshold = 0.5\n\n    # util\n    debug = False\n    debug_count = 3\n\ncfg = CFG()\n\nprint(f\"Using device: {cfg.device}\")\nprint(f\"Loading taxonomy data...\")\ntaxonomy_df = pd.read_csv(cfg.taxonomy_csv)\nspecies_ids = taxonomy_df['primary_label'].tolist()\nnum_classes = len(species_ids)\nprint(f\"Number of classes: {num_classes}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:37.13048Z","iopub.execute_input":"2025-05-25T17:07:37.130849Z","iopub.status.idle":"2025-05-25T17:07:37.164373Z","shell.execute_reply.started":"2025-05-25T17:07:37.130809Z","shell.execute_reply":"2025-05-25T17:07:37.163117Z"},"id":"rtK6S7S7DiN4"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def audio2melspec(audio_data, cfg):\n    \"\"\"Convert audio data to mel spectrogram\"\"\"\n    if np.isnan(audio_data).any():\n        mean_signal = np.nanmean(audio_data)\n        audio_data = np.nan_to_num(audio_data, nan=mean_signal)\n\n    mel_spec = librosa.feature.melspectrogram(\n        y=audio_data,\n        sr=cfg.FS,\n        n_fft=cfg.N_FFT,\n        hop_length=cfg.HOP_LENGTH,\n        n_mels=cfg.N_MELS,\n        fmin=cfg.FMIN,\n        fmax=cfg.FMAX,\n        power=2.0,\n        pad_mode=\"reflect\",\n        norm='slaney',\n        htk=True,\n        center=True,\n    )\n\n    mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n    mel_spec_norm = (mel_spec_db - mel_spec_db.min()) / (mel_spec_db.max() - mel_spec_db.min() + 1e-8)\n\n    return mel_spec_norm\n\ndef process_audio_segment(audio_data, cfg):\n    \"\"\"Process audio segment to get mel spectrogram\"\"\"\n    if len(audio_data) < cfg.FS * cfg.WINDOW_SIZE:\n        audio_data = np.pad(audio_data,\n                          (0, cfg.FS * cfg.WINDOW_SIZE - len(audio_data)),\n                          mode='constant')\n\n    mel_spec = audio2melspec(audio_data, cfg)\n\n    # Resize if needed\n    if mel_spec.shape != cfg.TARGET_SHAPE:\n        mel_spec = cv2.resize(mel_spec, cfg.TARGET_SHAPE, interpolation=cv2.INTER_LINEAR)\n\n    return mel_spec.astype(np.float32)\n\ndef find_model_files(cfg):\n    \"\"\"\n    Find all .pth model files in the specified model directory\n    \"\"\"\n    model_files = []\n\n    model_dir = Path(cfg.model_path)\n\n    for path in model_dir.glob('**/*.pth'):\n        model_files.append(str(path))\n\n    return model_files\n\ndef load_models(cfg, num_classes):\n    \"\"\"\n    Load all found model files and prepare them for ensemble\n    \"\"\"\n    models = []\n\n    model_files = find_model_files(cfg)\n\n    if not model_files:\n        print(f\"Warning: No model files found under {cfg.model_path}!\")\n        return models\n\n    print(f\"Found a total of {len(model_files)} model files.\")\n\n    if cfg.use_specific_folds:\n        filtered_files = []\n        for fold in cfg.folds:\n            fold_files = [f for f in model_files if f\"fold{fold}\" in f]\n            filtered_files.extend(fold_files)\n        model_files = filtered_files\n        print(f\"Using {len(model_files)} model files for the specified folds ({cfg.folds}).\")\n\n    for model_path in model_files:\n        try:\n            print(f\"Loading model: {model_path}\")\n            checkpoint = torch.load(model_path, map_location=torch.device(cfg.device))\n\n            model = BirdCLEFModel(cfg, num_classes)\n            model.load_state_dict(checkpoint['model_state_dict'])\n            model = model.to(cfg.device)\n            model.eval()\n\n            models.append(model)\n        except Exception as e:\n            print(f\"Error loading model {model_path}: {e}\")\n\n    return models\n\ndef predict_on_spectrogram(audio_path, models, cfg, species_ids):\n    \"\"\"Process a single audio file and predict species presence for each 5-second segment\"\"\"\n    predictions = []\n    row_ids = []\n    soundscape_id = Path(audio_path).stem\n\n    try:\n        print(f\"Processing {soundscape_id}\")\n        audio_data, _ = librosa.load(audio_path, sr=cfg.FS)\n\n        total_segments = int(len(audio_data) / (cfg.FS * cfg.WINDOW_SIZE))\n\n        for segment_idx in range(total_segments):\n            start_sample = segment_idx * cfg.FS * cfg.WINDOW_SIZE\n            end_sample = start_sample + cfg.FS * cfg.WINDOW_SIZE\n            segment_audio = audio_data[start_sample:end_sample]\n\n            end_time_sec = (segment_idx + 1) * cfg.WINDOW_SIZE\n            row_id = f\"{soundscape_id}_{end_time_sec}\"\n            row_ids.append(row_id)\n\n            if cfg.use_tta:\n                all_preds = []\n\n                for tta_idx in range(cfg.tta_count):\n                    mel_spec = process_audio_segment(segment_audio, cfg)\n                    mel_spec = apply_tta(mel_spec, tta_idx)\n\n                    mel_spec = torch.tensor(mel_spec, dtype=torch.float32).unsqueeze(0).unsqueeze(0)\n                    mel_spec = mel_spec.to(cfg.device)\n\n                    if len(models) == 1:\n                        with torch.no_grad():\n                            outputs = models[0](mel_spec)\n                            probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                            all_preds.append(probs)\n                    else:\n                        segment_preds = []\n                        for model in models:\n                            with torch.no_grad():\n                                outputs = model(mel_spec)\n                                probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                                segment_preds.append(probs)\n\n                        avg_preds = np.mean(segment_preds, axis=0)\n                        all_preds.append(avg_preds)\n\n                final_preds = np.mean(all_preds, axis=0)\n            else:\n                mel_spec = process_audio_segment(segment_audio, cfg)\n\n                mel_spec = torch.tensor(mel_spec, dtype=torch.float32).unsqueeze(0).unsqueeze(0)\n                mel_spec = mel_spec.to(cfg.device)\n\n                if len(models) == 1:\n                    with torch.no_grad():\n                        outputs = models[0](mel_spec)\n                        final_preds = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                else:\n                    segment_preds = []\n                    for model in models:\n                        with torch.no_grad():\n                            outputs = model(mel_spec)\n                            probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                            segment_preds.append(probs)\n\n                    final_preds = np.mean(segment_preds, axis=0)\n\n            predictions.append(final_preds)\n\n    except Exception as e:\n        print(f\"Error processing {audio_path}: {e}\")\n\n    return row_ids, predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:37.171949Z","iopub.execute_input":"2025-05-25T17:07:37.172327Z","iopub.status.idle":"2025-05-25T17:07:37.194833Z","shell.execute_reply.started":"2025-05-25T17:07:37.1723Z","shell.execute_reply":"2025-05-25T17:07:37.193583Z"},"id":"9NFVU9wXDiN4"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def apply_tta(spec, tta_idx):\n    \"\"\"Apply test-time augmentation\"\"\"\n    if tta_idx == 0:\n        # Original spectrogram\n        return spec\n    elif tta_idx == 1:\n        # Time shift (horizontal flip)\n        return np.flip(spec, axis=1)\n    elif tta_idx == 2:\n        # Frequency shift (vertical flip)\n        return np.flip(spec, axis=0)\n    else:\n        return spec\n\ndef run_inference(cfg, models, species_ids):\n    \"\"\"Run inference on all test soundscapes\"\"\"\n    test_files = list(Path(cfg.test_soundscapes).glob('*.ogg'))\n\n    if cfg.debug:\n        print(f\"Debug mode enabled, using only {cfg.debug_count} files\")\n        test_files = test_files[:cfg.debug_count]\n\n    print(f\"Found {len(test_files)} test soundscapes\")\n\n    all_row_ids = []\n    all_predictions = []\n\n    for audio_path in tqdm(test_files):\n        row_ids, predictions = predict_on_spectrogram(str(audio_path), models, cfg, species_ids)\n        all_row_ids.extend(row_ids)\n        all_predictions.extend(predictions)\n\n    return all_row_ids, all_predictions\n\ndef create_submission(row_ids, predictions, species_ids, cfg):\n    \"\"\"Create submission dataframe\"\"\"\n    print(\"Creating submission dataframe...\")\n\n    submission_dict = {'row_id': row_ids}\n\n    for i, species in enumerate(species_ids):\n        submission_dict[species] = [pred[i] for pred in predictions]\n\n    submission_df = pd.DataFrame(submission_dict)\n\n    submission_df.set_index('row_id', inplace=True)\n\n    sample_sub = pd.read_csv(cfg.submission_csv, index_col='row_id')\n\n    missing_cols = set(sample_sub.columns) - set(submission_df.columns)\n    if missing_cols:\n        print(f\"Warning: Missing {len(missing_cols)} species columns in submission\")\n        for col in missing_cols:\n            submission_df[col] = 0.0\n\n    submission_df = submission_df[sample_sub.columns]\n\n    submission_df = submission_df.reset_index()\n\n    return submission_df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:37.195833Z","iopub.execute_input":"2025-05-25T17:07:37.196317Z","iopub.status.idle":"2025-05-25T17:07:37.216912Z","shell.execute_reply.started":"2025-05-25T17:07:37.196284Z","shell.execute_reply":"2025-05-25T17:07:37.215584Z"},"id":"jdix8u9BDiN4"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1 style=\"color: #6cb4e4;  text-align: center;  padding: 0.25em;  border-top: solid 2.5px #6cb4e4;  border-bottom: solid 2.5px #6cb4e4;  background: -webkit-repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);  background: repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);height:45px;\">\n<b>\n《《《Submission2(nfnet)》》》\n</b></h1>","metadata":{"id":"pk1YaOsfDiN5"}},{"cell_type":"code","source":"import numpy as np\nfrom typing import Union\n\ndef apply_power_to_low_ranked_cols(\n    p: np.ndarray,\n    top_k: int = 30,\n    exponent: Union[int, float] = 2,\n    inplace: bool = True\n) -> np.ndarray:\n    \"\"\"\n    Rank columns by their column‑wise maximum and raise every column whose\n    rank falls below `top_k` to a given power.\n\n    Parameters\n    ----------\n    p : np.ndarray\n        A 2‑D array of shape **(n_chunks, n_classes)**.\n\n        - **n_chunks** is the number of fixed‑length time chunks obtained\n          after slicing the input audio (or other sequential data).\n          *Example:* In the BirdCLEF `test_soundscapes` set, each file is\n          60 s long. If you extract non‑overlapping 5 s windows,\n          `n_chunks = 60 s / 5 s = 12`.\n        - **n_classes** is the number of classes being predicted.\n        - Each element `p[i, j]` is the score or probability of class *j*\n          in chunk *i*.\n\n    top_k : int, default=35\n        The highest‑ranked columns (by their maximum value) that remain\n        unchanged.\n\n    exponent : int or float, default=1.5\n        The power applied to the selected low‑ranked columns\n        (e.g. `2` squares, `0.5` takes the square root, `3` cubes).\n\n    inplace : bool, default=True\n        If `True`, modify `p` in place.\n        If `False`, operate on a copy and leave the original array intact.\n\n    Returns\n    -------\n    np.ndarray\n        The transformed array. It is the same object as `p` when\n        `inplace=True`; otherwise, it is a new array.\n\n    \"\"\"\n    if not inplace:\n        p = p.copy()\n\n    # Identify columns whose max value ranks below `top_k`\n    tail_cols = np.argsort(-p.max(axis=0))[top_k:]\n\n    # Apply the power transformation to those columns\n    p[:, tail_cols] = p[:, tail_cols] ** exponent\n    return p","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:37.292415Z","iopub.execute_input":"2025-05-25T17:07:37.292842Z","iopub.status.idle":"2025-05-25T17:07:37.305867Z","shell.execute_reply.started":"2025-05-25T17:07:37.292799Z","shell.execute_reply":"2025-05-25T17:07:37.30457Z"},"id":"3daajt9DDiN5"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nimport time\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport timm\nimport torch.nn.functional as F\nimport torchaudio\nimport torchaudio.transforms as AT\nfrom contextlib import contextmanager\nimport concurrent.futures","metadata":{"papermill":{"duration":12.984639,"end_time":"2025-03-12T14:13:00.145177","exception":false,"start_time":"2025-03-12T14:12:47.160538","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:37.307128Z","iopub.execute_input":"2025-05-25T17:07:37.307597Z","iopub.status.idle":"2025-05-25T17:07:40.998784Z","shell.execute_reply.started":"2025-05-25T17:07:37.307555Z","shell.execute_reply":"2025-05-25T17:07:40.99744Z"},"id":"0jekbUiDDiN5"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_audio_dir = '../input/birdclef-2025/test_soundscapes/'\nfile_list = [f for f in sorted(os.listdir(test_audio_dir))]\nfile_list = [file.split('.')[0] for file in file_list if file.endswith('.ogg')]\n\ndebug = False\nprint('Debug mode:', debug)\nprint('Number of test soundscapes:', len(file_list))","metadata":{"papermill":{"duration":0.105385,"end_time":"2025-03-12T14:13:00.253425","exception":false,"start_time":"2025-03-12T14:13:00.14804","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:41.000096Z","iopub.execute_input":"2025-05-25T17:07:41.000504Z","iopub.status.idle":"2025-05-25T17:07:41.010467Z","shell.execute_reply.started":"2025-05-25T17:07:41.00047Z","shell.execute_reply":"2025-05-25T17:07:41.008578Z"},"id":"FoU9WNYRDiN6"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"wav_sec = 5\nsample_rate = 32000\nmin_segment = sample_rate*wav_sec\n\nclass_labels = sorted(os.listdir('../input/birdclef-2025/train_audio/'))\n\nn_fft=1024\nwin_length=1024\nhop_length=512\nf_min=50\nf_max=16000\nn_mels=128\n\nmel_spectrogram = AT.MelSpectrogram(\n    sample_rate=sample_rate,\n    n_fft=n_fft,\n    win_length=win_length,\n    hop_length=hop_length,\n    center=True,\n    f_min=f_min,\n    f_max=f_max,\n    pad_mode=\"reflect\",\n    power=2.0,\n    norm='slaney',\n    n_mels=n_mels,\n    mel_scale=\"htk\",\n    # normalized=True\n)\n\ndef normalize_std(spec, eps=1e-6):\n    mean = torch.mean(spec)\n    std = torch.std(spec)\n    return torch.where(std == 0, spec-mean, (spec - mean) / (std+eps))\n\ndef audio_to_mel(filepath=None):\n    waveform, sample_rate = torchaudio.load(filepath,backend=\"soundfile\")\n    len_wav = waveform.shape[1]\n    waveform = waveform[0,:].reshape(1, len_wav) # stereo->mono mono->mono\n    PREDS = []\n    for i in range(12):\n        waveform2 = waveform[:,i*sample_rate*5:i*sample_rate*5+sample_rate*5]\n        melspec = mel_spectrogram(waveform2)\n        melspec = torch.log(melspec+1e-6)\n        melspec = normalize_std(melspec)\n        melspec = torch.unsqueeze(melspec, dim=0)\n\n        PREDS.append(melspec)\n    return torch.vstack(PREDS)","metadata":{"papermill":{"duration":0.144235,"end_time":"2025-03-12T14:13:00.400505","exception":false,"start_time":"2025-03-12T14:13:00.25627","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:41.012046Z","iopub.execute_input":"2025-05-25T17:07:41.012626Z","iopub.status.idle":"2025-05-25T17:07:41.0472Z","shell.execute_reply.started":"2025-05-25T17:07:41.012574Z","shell.execute_reply":"2025-05-25T17:07:41.045758Z"},"id":"DOMtEWDUDiN6"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def init_layer(layer):\n    nn.init.xavier_uniform_(layer.weight)\n    if hasattr(layer, \"bias\"):\n        if layer.bias is not None:\n            layer.bias.data.fill_(0.)\n\n\ndef init_bn(bn):\n    bn.bias.data.fill_(0.)\n    bn.weight.data.fill_(1.0)\n\n\ndef init_weights(model):\n    classname = model.__class__.__name__\n    if classname.find(\"Conv2d\") != -1:\n        nn.init.xavier_uniform_(model.weight, gain=np.sqrt(2))\n        model.bias.data.fill_(0)\n    elif classname.find(\"BatchNorm\") != -1:\n        model.weight.data.normal_(1.0, 0.02)\n        model.bias.data.fill_(0)\n    elif classname.find(\"GRU\") != -1:\n        for weight in model.parameters():\n            if len(weight.size()) > 1:\n                nn.init.orghogonal_(weight.data)\n    elif classname.find(\"Linear\") != -1:\n        model.weight.data.normal_(0, 0.01)\n        model.bias.data.zero_()\n\n\ndef interpolate(x, ratio):\n    (batch_size, time_steps, classes_num) = x.shape\n    upsampled = x[:, :, None, :].repeat(1, 1, ratio, 1)\n    upsampled = upsampled.reshape(batch_size, time_steps * ratio, classes_num)\n    return upsampled\n\n\ndef pad_framewise_output(framewise_output, frames_num):\n    output = F.interpolate(\n        framewise_output.unsqueeze(1),\n        size=(frames_num, framewise_output.size(2)),\n        align_corners=True,\n        mode=\"bilinear\").squeeze(1)\n\n    return output\n\n\nclass AttBlockV2(nn.Module):\n    def __init__(self,\n                 in_features: int,\n                 out_features: int,\n                 activation=\"linear\"):\n        super().__init__()\n\n        self.activation = activation\n        self.att = nn.Conv1d(\n            in_channels=in_features,\n            out_channels=out_features,\n            kernel_size=1,\n            stride=1,\n            padding=0,\n            bias=True)\n        self.cla = nn.Conv1d(\n            in_channels=in_features,\n            out_channels=out_features,\n            kernel_size=1,\n            stride=1,\n            padding=0,\n            bias=True)\n\n        self.init_weights()\n\n    def init_weights(self):\n        init_layer(self.att)\n        init_layer(self.cla)\n\n    def forward(self, x):\n        norm_att = torch.softmax(torch.tanh(self.att(x)), dim=-1)\n        cla = self.nonlinear_transform(self.cla(x))\n        x = torch.sum(norm_att * cla, dim=2)\n        return x, norm_att, cla\n\n    def nonlinear_transform(self, x):\n        if self.activation == 'linear':\n            return x\n        elif self.activation == 'sigmoid':\n            return torch.sigmoid(x)\n\n\nclass TimmSED(nn.Module):\n    def __init__(self, base_model_name: str, pretrained=False, num_classes=24, in_channels=1, n_mels=24):\n        super().__init__()\n\n        self.bn0 = nn.BatchNorm2d(n_mels)\n\n        base_model = timm.create_model(\n            base_model_name, pretrained=pretrained, in_chans=in_channels)\n        layers = list(base_model.children())[:-2]\n        self.encoder = nn.Sequential(*layers)\n\n        in_features = base_model.num_features\n\n        self.fc1 = nn.Linear(in_features, in_features, bias=True)\n        self.att_block2 = AttBlockV2(\n            in_features, num_classes, activation=\"sigmoid\")\n\n        self.init_weight()\n\n    def init_weight(self):\n        init_bn(self.bn0)\n        init_layer(self.fc1)\n\n\n    def forward(self, input_data):\n        x = input_data.transpose(2,3)\n        x = torch.cat((x,x,x),1)\n\n        x = x.transpose(2, 3)\n\n        x = self.encoder(x)\n\n        x = torch.mean(x, dim=2)\n\n        x1 = F.max_pool1d(x, kernel_size=3, stride=1, padding=1)\n        x2 = F.avg_pool1d(x, kernel_size=3, stride=1, padding=1)\n        x = x1 + x2\n\n        x = x.transpose(1, 2)\n        x = F.relu_(self.fc1(x))\n        x = x.transpose(1, 2)\n\n        (clipwise_output, norm_att, segmentwise_output) = self.att_block2(x)\n        logit = torch.sum(norm_att * self.att_block2.cla(x), dim=2)\n\n        output_dict = {\n            'logit': logit,\n        }\n\n        return output_dict","metadata":{"papermill":{"duration":2.175154,"end_time":"2025-03-12T14:13:02.578522","exception":false,"start_time":"2025-03-12T14:13:00.403368","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:41.051857Z","iopub.execute_input":"2025-05-25T17:07:41.052247Z","iopub.status.idle":"2025-05-25T17:07:41.07362Z","shell.execute_reply.started":"2025-05-25T17:07:41.052215Z","shell.execute_reply":"2025-05-25T17:07:41.072207Z"},"id":"1-6b1dALDiN6"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_model_name='eca_nfnet_l0'\npretrained=False\nin_channels=3\n\nMODELS = [f'/kaggle/input/birdclef-2025-sed-models-p/sed{i}.pth' for i in range(3)]\n\nMODELS","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:41.075278Z","iopub.execute_input":"2025-05-25T17:07:41.075748Z","iopub.status.idle":"2025-05-25T17:07:41.103195Z","shell.execute_reply.started":"2025-05-25T17:07:41.075701Z","shell.execute_reply":"2025-05-25T17:07:41.101994Z"},"id":"fYufxgTSDiN7"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"models = []\nfor path in MODELS:\n    model = TimmSED(base_model_name=base_model_name,\n               pretrained=pretrained,\n               num_classes=len(class_labels),\n               in_channels=in_channels,\n               n_mels=n_mels);\n    model.load_state_dict(torch.load(path, weights_only=True, map_location=torch.device('cpu')))\n    model.eval();\n    models.append(model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:41.104519Z","iopub.execute_input":"2025-05-25T17:07:41.10487Z","iopub.status.idle":"2025-05-25T17:07:44.806949Z","shell.execute_reply.started":"2025-05-25T17:07:41.104844Z","shell.execute_reply":"2025-05-25T17:07:44.805902Z"},"id":"KhuGv-3ZDiN7"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prediction(afile):\n    global pred\n    path = test_audio_dir + afile + '.ogg'\n    with torch.inference_mode():\n        sig = audio_to_mel(path)\n        outputs = None\n        for model in models:\n            model.eval()\n            p = model(sig)\n            p = torch.sigmoid(p['logit']).detach().cpu().numpy()\n            p = apply_power_to_low_ranked_cols(p, top_k=30,exponent=2)\n            if outputs is None: outputs = p\n            else: outputs += p\n\n        outputs /= len(models)\n        chunks = [[] for i in range(12)]\n        for i in range(len(chunks)):\n            chunk_end_time = (i + 1) * 5\n            row_id = afile + '_' + str(chunk_end_time)\n            pred['row_id'].append(row_id)\n            bird_no = 0\n            for bird in class_labels:\n                pred[bird].append(outputs[i,bird_no])\n                bird_no += 1\n        gc.collect()","metadata":{"papermill":{"duration":0.011209,"end_time":"2025-03-12T14:13:02.593243","exception":false,"start_time":"2025-03-12T14:13:02.582034","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:44.807805Z","iopub.execute_input":"2025-05-25T17:07:44.808094Z","iopub.status.idle":"2025-05-25T17:07:44.816203Z","shell.execute_reply.started":"2025-05-25T17:07:44.80807Z","shell.execute_reply":"2025-05-25T17:07:44.815004Z"},"id":"P8MuEBU_DiN7"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred = {'row_id': []}\nfor species_code in class_labels:\n    pred[species_code] = []\n\nstart = time.time()\nwith concurrent.futures.ThreadPoolExecutor(max_workers=5) as executor:\n    _ = list(executor.map(prediction, file_list))\nend_t = time.time()\n\nif debug == True:\n    print(700*(end_t - start)/60/debug_num)","metadata":{"papermill":{"duration":6.823541,"end_time":"2025-03-12T14:13:09.419521","exception":false,"start_time":"2025-03-12T14:13:02.59598","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:44.817789Z","iopub.execute_input":"2025-05-25T17:07:44.818677Z","iopub.status.idle":"2025-05-25T17:07:44.866157Z","shell.execute_reply.started":"2025-05-25T17:07:44.818632Z","shell.execute_reply":"2025-05-25T17:07:44.865149Z"},"id":"i-tjQTBBDiN7"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results = pd.DataFrame(pred, columns = ['row_id'] + class_labels)\ndisplay(results.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:07:44.867316Z","iopub.execute_input":"2025-05-25T17:07:44.867707Z","iopub.status.idle":"2025-05-25T17:07:44.909974Z","shell.execute_reply.started":"2025-05-25T17:07:44.867669Z","shell.execute_reply":"2025-05-25T17:07:44.90892Z"},"id":"1FUvly18DiN7"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"results.to_csv(\"submission1.csv\", index=False)\n\nsub = pd.read_csv('submission1.csv')\ncols = sub.columns[1:]\ngroups = sub['row_id'].str.rsplit('_', n=1).str[0]\ngroups = groups.values\nfor group in np.unique(groups):\n    sub_group = sub[group == groups]\n    predictions = sub_group[cols].values\n    new_predictions = predictions.copy()\n    for i in range(1, predictions.shape[0]-1):\n        new_predictions[i] = (predictions[i-1] * 0.2) + (predictions[i] * 0.6) + (predictions[i+1] * 0.2)\n    new_predictions[0] = (predictions[0] * 0.9) + (predictions[1] * 0.1)\n    new_predictions[-1] = (predictions[-1] * 0.9) + (predictions[-2] * 0.1)\n    sub_group[cols] = new_predictions\n    sub[group == groups] = sub_group\nsub.to_csv(\"submission1.csv\", index=False)\n\n\nif debug:\n    display(results)","metadata":{"papermill":{"duration":0.097214,"end_time":"2025-03-12T14:13:09.519812","exception":false,"start_time":"2025-03-12T14:13:09.422598","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:08:12.393276Z","iopub.execute_input":"2025-05-25T17:08:12.39378Z","iopub.status.idle":"2025-05-25T17:08:12.43791Z","shell.execute_reply.started":"2025-05-25T17:08:12.39374Z","shell.execute_reply":"2025-05-25T17:08:12.436518Z"},"scrolled":true,"id":"5nzDUUcwDiN8"},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<h1 style=\"color: #6cb4e4;  text-align: center;  padding: 0.25em;  border-top: solid 2.5px #6cb4e4;  border-bottom: solid 2.5px #6cb4e4;  background: -webkit-repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);  background: repeating-linear-gradient(-45deg, #f0f8ff, #f0f8ff 3px,#e9f4ff 3px, #e9f4ff 7px);height:45px;\">\n<b>\n《《《Finaly Blending》》》\n</b></h1>","metadata":{"id":"keOZX9qmDiN8"}},{"cell_type":"code","source":"# ------------------------------------------- #\n# [IMPORTANT]\n# * Blending Weight\n# ------------------------------------------- #\nsub_w = [0.25, 0.75]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:08:13.810564Z","iopub.execute_input":"2025-05-25T17:08:13.811141Z","iopub.status.idle":"2025-05-25T17:08:13.818507Z","shell.execute_reply.started":"2025-05-25T17:08:13.811101Z","shell.execute_reply":"2025-05-25T17:08:13.816457Z"},"id":"ciZfE3hJDiN8"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load target list and prepare column names\nlist_TARGETs = sorted(os.listdir('/kaggle/input/birdclef-2025/train_audio/'))\nlist_targets_0 = [f'{TARGET} 0' for TARGET in list_TARGETs]\nlist_targets_1 = [f'{TARGET} 1' for TARGET in list_TARGETs]\n\n# Load both predictions\ndf0 = pd.read_csv(\"/kaggle/working/submission0.csv\")\ndf1 = pd.read_csv(\"/kaggle/working/submission1.csv\")\n\n# Rename columns to distinguish models\ndf0 = df0.rename(columns={TARGET : f'{TARGET} 0' for TARGET in list_TARGETs})\ndf1 = df1.rename(columns={TARGET : f'{TARGET} 1' for TARGET in list_TARGETs})\n\n# Merge on row_id\ndfs = pd.merge(df0, df1, on='row_id')\n\n# Compute blended predictions all at once (avoids fragmentation)\nblended_columns = {\n    TARGET: dfs[f'{TARGET} 0'] * sub_w[0] + dfs[f'{TARGET} 1'] * sub_w[1]\n    for TARGET in list_TARGETs\n}\n\nblended_df = pd.DataFrame(blended_columns)\n\n# Final DataFrame: row_id + blended target columns\nfinal_df = pd.concat([dfs[['row_id']], blended_df], axis=1)\n\n# Save\nfinal_df.to_csv(\"submission.csv\", index=False)\nprint(\"DONE\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:08:15.035344Z","iopub.execute_input":"2025-05-25T17:08:15.035733Z","iopub.status.idle":"2025-05-25T17:08:15.215613Z","shell.execute_reply.started":"2025-05-25T17:08:15.035703Z","shell.execute_reply":"2025-05-25T17:08:15.213925Z"},"id":"Y1xgocDsDiN8"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"id":"9ZocRVzZDiN9"},"outputs":[],"execution_count":null}]}