{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.11"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11819288,"sourceType":"datasetVersion","datasetId":7423992},{"sourceId":12051777,"sourceType":"datasetVersion","datasetId":7505901}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":127.721788,"end_time":"2025-05-20T15:29:32.620693","environment_variables":{},"exception":true,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-05-20T15:27:24.898905","version":"2.6.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"0b9f241e63f2459ba21e360a462196c4":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_739c292dd99a4115842723e8c68c62f5","max":62511014,"min":0,"orientation":"horizontal","style":"IPY_MODEL_9c9d3367c79a48839e272efe51adcb71","tabbable":null,"tooltip":null,"value":62511014}},"0d130c5bbf3b4096b70526974d8babdb":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"1ef15668ae334240a3df214b6793a058":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"20a9df35a2654d7288ff839b45e22420":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"270ede3eac4740deaedc182badf15139":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"27be3b484cf748069c363df9d1745e1b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"2b83c1592af642659c1780d5bbde6192":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_44d94a6bdd034aa6b36c481943f0c7c6","placeholder":"​","style":"IPY_MODEL_20a9df35a2654d7288ff839b45e22420","tabbable":null,"tooltip":null,"value":" 62.5M/62.5M [00:00&lt;00:00, 218MB/s]"}},"44d2f1ee0e8041e9a4007acc1375f961":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"44d94a6bdd034aa6b36c481943f0c7c6":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"4b2fa3ec553e4d04973c0191a606f8c0":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_8b2f9f5f1f774807bcb1ee65736f6fde","IPY_MODEL_0b9f241e63f2459ba21e360a462196c4","IPY_MODEL_2b83c1592af642659c1780d5bbde6192"],"layout":"IPY_MODEL_8a775250b45746778a9a9350f59096d5","tabbable":null,"tooltip":null}},"4c6c001bb598476f8b990a9ec366366f":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"5dbf055f27524f60be413fd5bdf4b153":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"665078286f8d4ac5a1b4a7d615d4af10":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"739c292dd99a4115842723e8c68c62f5":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"76f3126438b540b0b0aa89d27be14ec7":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_e71489af994642e4a1950b65c56a94b0","IPY_MODEL_7fa894a90c6948c0830d358c629c2fe9","IPY_MODEL_a4dda47937d046cebe6fe463eeb7c65b"],"layout":"IPY_MODEL_270ede3eac4740deaedc182badf15139","tabbable":null,"tooltip":null}},"7c73072c2b6f45d8816ad5918872ade7":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_665078286f8d4ac5a1b4a7d615d4af10","max":102469840,"min":0,"orientation":"horizontal","style":"IPY_MODEL_7ce3e9349ab04589bd789f11f462123b","tabbable":null,"tooltip":null,"value":102469840}},"7ce3e9349ab04589bd789f11f462123b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"7fa894a90c6948c0830d358c629c2fe9":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_de3ccadddd6c44019548d23e9b7f29a9","max":21355344,"min":0,"orientation":"horizontal","style":"IPY_MODEL_5dbf055f27524f60be413fd5bdf4b153","tabbable":null,"tooltip":null,"value":21355344}},"8a775250b45746778a9a9350f59096d5":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8b2f9f5f1f774807bcb1ee65736f6fde":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_a3bb4c088b9d4a768a752abea5f1b3d6","placeholder":"​","style":"IPY_MODEL_4c6c001bb598476f8b990a9ec366366f","tabbable":null,"tooltip":null,"value":"model.safetensors: 100%"}},"9739edf8bcd9431184098a78e242f005":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"9c9d3367c79a48839e272efe51adcb71":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"a0069c0ffe934e9abd7c3a3b9872d37d":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"a3bb4c088b9d4a768a752abea5f1b3d6":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"a4dda47937d046cebe6fe463eeb7c65b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_ba7874ffa53b45d19f1215f4b1ab7fd1","placeholder":"​","style":"IPY_MODEL_a0069c0ffe934e9abd7c3a3b9872d37d","tabbable":null,"tooltip":null,"value":" 21.4M/21.4M [00:00&lt;00:00, 135MB/s]"}},"ad253f296a1b4f3eac05f47d62ca245a":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"ba7874ffa53b45d19f1215f4b1ab7fd1":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"ce0e6478084c42ac98c386d4ebc7cd33":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"de3ccadddd6c44019548d23e9b7f29a9":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"e24c7127f3df44a486da2660ab438129":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_ad253f296a1b4f3eac05f47d62ca245a","placeholder":"​","style":"IPY_MODEL_ce0e6478084c42ac98c386d4ebc7cd33","tabbable":null,"tooltip":null,"value":" 102M/102M [00:00&lt;00:00, 222MB/s]"}},"e71489af994642e4a1950b65c56a94b0":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_1ef15668ae334240a3df214b6793a058","placeholder":"​","style":"IPY_MODEL_0d130c5bbf3b4096b70526974d8babdb","tabbable":null,"tooltip":null,"value":"model.safetensors: 100%"}},"efe762a4a5ab4597b6db062e1a24797e":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_fd5ec59e66ce4ab4aca028e402077407","IPY_MODEL_7c73072c2b6f45d8816ad5918872ade7","IPY_MODEL_e24c7127f3df44a486da2660ab438129"],"layout":"IPY_MODEL_9739edf8bcd9431184098a78e242f005","tabbable":null,"tooltip":null}},"fd5ec59e66ce4ab4aca028e402077407":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_44d2f1ee0e8041e9a4007acc1375f961","placeholder":"​","style":"IPY_MODEL_27be3b484cf748069c363df9d1745e1b","tabbable":null,"tooltip":null,"value":"model.safetensors: 100%"}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":" # 🐦 BirdCLEF 2025 - Data Pipeline for Precomputed Features Extraction\n\nThis notebook implements a robust and efficient data pipeline for bird sound classification using deep learning. The workflow is designed to maximize training speed and flexibility by precomputing audio features , making it ideal for large-scale or multi-phase training experiments.\n\n---\n\n## **Key Features**\n\n- **Multiple Data Quality Splits:**  \n  Supports high-quality, medium-quality, and all-data splits, each with its own train/validation sets.\n\n- **Precomputed Feature Pipeline:**  \n  Audio files are processed **once** to extract Mel-spectrograms, YAMNet embeddings, and label vectors. These are saved as `.npz` files, drastically speeding up model training and reducing CPU bottlenecks.\n\n- **On-the-Fly Augmentation for Training:**  \n  For each training sample, multiple augmented versions (e.g., time-stretch, pitch-shift, noise) are generated and stored during precomputation. This ensures model robustness and diversity while maintaining high throughput.\n\n- **Fast DataLoader:**  \n  During training/validation, the DataLoader simply reads precomputed `.npz` files—no audio decoding or augmentation overhead at runtime.\n\n---\n","metadata":{"papermill":{"duration":0.006034,"end_time":"2025-05-20T15:27:29.095493","exception":false,"start_time":"2025-05-20T15:27:29.089459","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## 🔗 BirdCLEF 2025 - Project Notebook Links\n\nHere are the different stages of my BirdCLEF 2025 pipeline, organized by functionality:\n\n### 📊 Data Preparation\n- [BirdCLEF 2025 - Data Preparation](https://www.kaggle.com/code/sheemamasood/birdclef-2025-data-prepartion)\n\n### 🎛️ Mel Spectrogram Generation\n- [BirdCLEF 2025 - Mel Generation](https://www.kaggle.com/code/sheemamasood/birdclef2025-mel-generation)\n\n### 🏷️ Pseudo Labelling for SSL\n- [BirdCLEF 2025 - Pseudo Labelling for SSL](https://www.kaggle.com/code/sheemamasood/birdclef2025-psedolabelling-for-ssl)\n\n### 🧠 Model Training\n- [BirdCLEF 2025 - Model Training (Phase 1)](https://www.kaggle.com/code/sheemamasood/birdclef2025-model-training-phase1)\n\n### 📦 Inference & Submissions\n- [BirdCLEF 2025 - Submissions](https://www.kaggle.com/code/sheemamasood/birdclef2025-submissions)\n","metadata":{}},{"cell_type":"code","source":"# 📦 Basic Utilities\nimport os\nimport math\nimport time\nimport random\nimport logging\nimport warnings\nfrom pathlib import Path\n\n# 📊 Data Handling & Evaluation\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.metrics import roc_auc_score, classification_report, accuracy_score\nimport pickle\n\n# 🎧 Audio Processing\nimport librosa\nimport librosa.display\nimport torchaudio\nimport torchaudio.transforms as T\nimport torchaudio.functional as F\n\n# 🔥 PyTorch and Model Utilities\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader, WeightedRandomSampler\nfrom dataclasses import dataclass\nfrom typing import List\n\n# transformers\nfrom transformers import Wav2Vec2Processor, Wav2Vec2Model, Wav2Vec2ForSequenceClassification\nfrom torch.optim import AdamW\n\n# 🖼️ Visualization\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\n\n# 🔁 Progress Tracking\nfrom tqdm.notebook import tqdm  # for notebooks\nfrom tqdm import tqdm  # for scripts\nfrom tqdm.auto import tqdm\n\n# 🧠 Pretrained Models\nimport timm\n\n# ✅ Confirm librosa\nprint(f\"librosa version : {librosa.__version__}\")\nprint(f\"librosa files   : {librosa.__file__}\")\n\nprint(\"✅ All libraries successfully imported!\")\n","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2025-06-16T22:13:35.919916Z","iopub.execute_input":"2025-06-16T22:13:35.920192Z","iopub.status.idle":"2025-06-16T22:14:10.704855Z","shell.execute_reply.started":"2025-06-16T22:13:35.920166Z","shell.execute_reply":"2025-06-16T22:14:10.703988Z"},"papermill":{"duration":12.518033,"end_time":"2025-05-20T15:28:58.412890","exception":false,"start_time":"2025-05-20T15:28:45.894857","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\n\n# Check for CUDA\nif torch.cuda.is_available():\n    device = torch.device(\"cuda\")\n    print(\"✅ CUDA is available. Using GPU.\")\nelse:\n    device = torch.device(\"cpu\")\n    print(\"❌ CUDA not available. Using CPU.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T22:14:10.705817Z","iopub.execute_input":"2025-06-16T22:14:10.706515Z","iopub.status.idle":"2025-06-16T22:14:10.711749Z","shell.execute_reply.started":"2025-06-16T22:14:10.706485Z","shell.execute_reply":"2025-06-16T22:14:10.710806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    # Audio settings\n    FS = 32000  # Sampling rate (audio)\n\n    # Mel spectrogram parameters (for converting audio to image)\n    N_FFT = 1024       # FFT window size\n    HOP_LENGTH = 512   # Step size for each frame\n    FS = 32000\n    FMIN = 50          # Minimum Mel frequency\n    FMAX = 14000       # Maximum Mel frequency\n    \n    # RGB image shape (C, H, W)\n    TARGET_DURATION = 10.0\n    N_MELS = 128\n    MEL_SHAPE = (256, 256)      \n    TARGET_SHAPE = (3, 256, 256)  \n\n\n    # No limit on the number of samples during training (full dataset)\n    N_MAX = None  \n\n    # flag for training mode\n    TRAINING_MODE = True  \n    \n    # Additional training-specific configurations\n    EPOCHS = 10  \n    BATCH_SIZE = 32  \n    LEARNING_RATE = 0.001  \n\n# Create the config object\nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2025-06-16T22:14:10.713534Z","iopub.execute_input":"2025-06-16T22:14:10.713902Z","iopub.status.idle":"2025-06-16T22:14:10.759711Z","shell.execute_reply.started":"2025-06-16T22:14:10.713871Z","shell.execute_reply":"2025-06-16T22:14:10.758794Z"},"papermill":{"duration":0.03529,"end_time":"2025-05-20T15:28:58.470588","exception":false,"start_time":"2025-05-20T15:28:58.435298","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Root path where all files and folders are stored\nDATA_ROOT = '/kaggle/input/birdclef-2025'\n\n# Load CSVs\ntrain_df = pd.read_csv(os.path.join(DATA_ROOT, 'train.csv'))\ntaxonomy_df = pd.read_csv(os.path.join(DATA_ROOT, 'taxonomy.csv'))\nlocation_df = pd.read_csv(os.path.join(DATA_ROOT, 'recording_location.txt'), delimiter='\\t')\nsample_submission = pd.read_csv(os.path.join(DATA_ROOT, 'sample_submission.csv'))\n\n\n\nprint(f\"✅ Loaded train_df: {train_df.shape}\")\nprint(f\"✅ Loaded taxonomy_df: {taxonomy_df.shape}\")\nprint(f\"✅ Loaded location_df: {location_df.shape}\")\nprint(f\"✅ Loaded sample_submission: {sample_submission.shape}\")\n","metadata":{"papermill":{"duration":0.267854,"end_time":"2025-05-20T15:28:58.769484","exception":false,"start_time":"2025-05-20T15:28:58.501630","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T22:14:10.760451Z","iopub.execute_input":"2025-06-16T22:14:10.760726Z","iopub.status.idle":"2025-06-16T22:14:10.968060Z","shell.execute_reply.started":"2025-06-16T22:14:10.760706Z","shell.execute_reply":"2025-06-16T22:14:10.967322Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load your VAD-cleaned CSV file\nclean_train_df = pd.read_csv('/kaggle/input/birdcleft-clean-and-vad-filtered-data/train_audio_10sec_chunks_VAD_filtered.csv')\nchunked_train_df = pd.read_csv(\"/kaggle/input/birdcleft-clean-and-vad-filtered-data/train_audio_10sec_chunks.csv\")\nprint(f\"✅ Loaded chunked_train_df: {chunked_train_df.shape}\")\nprint(f\"✅ Loaded clean_train_df: {clean_train_df.shape}\")\n\n\nworking_df = pd.read_csv(\"/kaggle/input/melspectrogramofbirdclef-2025/working_df.csv\")\nprint(f\"✅ Loaded working_df: {working_df.shape}\")\n\nsoundscape_chunked_df = pd.read_csv(\"/kaggle/input/birdcleft-clean-and-vad-filtered-data/soundscape_10sec_chunks.csv\")\nclean_soundscape_df = pd.read_csv(\"/kaggle/input/birdcleft-clean-and-vad-filtered-data/clean_soundscapes_chunks_10sec_vad_filtered.csv\")\n\nprint(f\"✅ Loaded soundscape_chunked_df: {soundscape_chunked_df.shape}\")\nprint(f\"✅ Loaded clean_soundscape_df: {clean_soundscape_df.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T22:14:10.969015Z","iopub.execute_input":"2025-06-16T22:14:10.969940Z","iopub.status.idle":"2025-06-16T22:14:12.015865Z","shell.execute_reply.started":"2025-06-16T22:14:10.969918Z","shell.execute_reply":"2025-06-16T22:14:12.015033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the master label list and set NUM_CLASSES\nwith open(\"/kaggle/input/birdcleft-clean-and-vad-filtered-data/master_label_list.pkl\", \"rb\") as f:\n    master_labels = pickle.load(f)\nNUM_CLASSES = len(master_labels)  # should be 206\n\n\nprint(f\"total number of labels in full data : {len(master_labels)}\")  # Should print 206\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T22:14:30.595909Z","iopub.execute_input":"2025-06-16T22:14:30.596178Z","iopub.status.idle":"2025-06-16T22:14:30.602789Z","shell.execute_reply.started":"2025-06-16T22:14:30.596157Z","shell.execute_reply":"2025-06-16T22:14:30.601760Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def validate_df(df, name):\n    print(f\"\\n🔍 Validating: {name}\")\n    print(\"-\" * 50)\n    \n    # Check for NaNs\n    nan_summary = df.isna().sum()\n    if nan_summary.sum() == 0:\n        print(\"✅ No NaNs found.\")\n    else:\n        print(\"⚠️ NaNs found:\")\n        display(nan_summary[nan_summary > 0])\n    \n    # Show data types\n    print(\"\\n📊 Data Info:\")\n    display(df.info())\n    \n    # Sample rows\n    print(\"\\n🧾 Sample Rows:\")\n    display(df.sample(3, random_state=42))\n    \n    # Check essential columns (just an example set — adjust as needed)\n    expected_cols = ['chunk_id', 'filepath', 'start_sample', 'end_sample']\n    missing_cols = [col for col in expected_cols if col not in df.columns]\n    if missing_cols:\n        print(f\"❌ Missing essential columns: {missing_cols}\")\n    else:\n        print(\"✅ All essential columns are present.\")\n        \n    #unique filename\n    print(f\"Unique audio files in df: {df['filename'].nunique()}\")\n    print(f\"Average duration in df: {df['duration'].mean()}\")\n    \n    # Sanity check: duration and sample range\n    if 'duration' in df.columns and 'start_sample' in df.columns and 'end_sample' in df.columns:\n        duration_errors = df[df['end_sample'] <= df['start_sample']]\n        if len(duration_errors) > 0:\n            print(f\"❌ {len(duration_errors)} rows have invalid sample ranges!\")\n        else:\n            print(\"✅ Sample ranges are valid.\")\n\n# Run validation\nvalidate_df(working_df, \"working_df (raw/rating filtered)\")\nvalidate_df(clean_train_df, \"clean_df (VAD filtered)\")\nvalidate_df(clean_soundscape_df , \"clean_soundscape_df(VAD FILTERED)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T22:14:33.190360Z","iopub.execute_input":"2025-06-16T22:14:33.191264Z","iopub.status.idle":"2025-06-16T22:14:33.364522Z","shell.execute_reply.started":"2025-06-16T22:14:33.191236Z","shell.execute_reply":"2025-06-16T22:14:33.363899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Total rows:\", len(clean_train_df))\nprint(\"Unique samplenames:\", clean_train_df['samplename'].nunique())\nprint(clean_train_df['samplename'].value_counts().head(10))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T22:14:39.125023Z","iopub.execute_input":"2025-06-16T22:14:39.125758Z","iopub.status.idle":"2025-06-16T22:14:39.146824Z","shell.execute_reply.started":"2025-06-16T22:14:39.125725Z","shell.execute_reply":"2025-06-16T22:14:39.145860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Total rows:\", len(clean_soundscape_df))\nprint(\"Unique samplenames:\", clean_soundscape_df['samplename'].nunique())\nprint(clean_soundscape_df['samplename'].value_counts().head(10))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T22:14:39.418751Z","iopub.execute_input":"2025-06-16T22:14:39.419388Z","iopub.status.idle":"2025-06-16T22:14:39.429751Z","shell.execute_reply.started":"2025-06-16T22:14:39.419358Z","shell.execute_reply":"2025-06-16T22:14:39.428858Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🎧 MEL Spectrogram Generation\n\n- **Input**: Raw audio waveform (numpy array)\n- **Output**: Normalized mel spectrogram (numpy array with shape `[n_mels, time]`)\n\n## 🧼 Audio Preparation Function\n\nEnsures each audio file is exactly `target_len` samples:\n\n- 🔹 If too short → center **zero-padded**\n- 🔹 If too long → center **trimmed**\n","metadata":{}},{"cell_type":"code","source":"# Function jo audio ko mel spectrogram me convert karta hai\n# def audio2melspec(audio_data, config):\n#     # Agar NaN ho to usko remove karte hain\n#     if np.isnan(audio_data).any():\n#         mean_val = np.nanmean(audio_data)\n#         audio_data = np.nan_to_num(audio_data, nan=mean_val)\n\n#     # Mel spectrogram\n#     mel = librosa.feature.melspectrogram(\n#         y=audio_data,\n#         sr=config.FS,\n#         n_fft=config.N_FFT,\n#         hop_length=config.HOP_LENGTH,\n#         n_mels=config.N_MELS,\n#         fmin=config.FMIN,\n#         fmax=config.FMAX,\n#         power=2.0\n#     )\n\n#     # Usko decibels me convert karna\n#     mel_db = librosa.power_to_db(mel, ref=np.max)\n\n#     # Normalize karna\n#     mel_db = (mel_db - mel_db.min()) / (mel_db.max() - mel_db.min() + 1e-8)\n\n#     return mel_db\n\n\n# # Function to process audio dataset\n# def process_audio(df, label, config):\n#     print(f\"Processing {label} audio data...\")\n#     start_time = time.time()\n\n#     bird_data = {}\n#     errors = []\n#     target_len = int(config.TARGET_DURATION * config.FS)\n\n#     for k, row in tqdm(df.iterrows(), total=len(df)):\n#         try:\n#             audio, _ = librosa.load(row.filepath, sr=config.FS)\n#             audio = prepare_audio(audio, target_len)\n#             mel = audio2melspec(audio, config)\n\n#             if mel.shape != config.TARGET_SHAPE:\n#                 mel = cv2.resize(mel, config.TARGET_SHAPE)\n\n#             bird_data[row.samplename] = mel.astype(np.float32)\n\n#         except Exception as e:\n#             errors.append((row.filepath, str(e)))\n\n#     end_time = time.time()\n\n#     print(f\"\\nFinished processing '{label}' in {end_time - start_time:.1f} seconds\")\n#     print(f\" Successfully processed: {len(bird_data)} files of {label}\")\n#     print(f\"Failed: {len(errors)} files\")\n\n#     np.savez_compressed(f'{label}.npz', **bird_data)\n#     print(f\"Saved data as '{label}.npz'\\n\")\n\n#     return bird_data, errors\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T22:14:45.871836Z","iopub.execute_input":"2025-06-16T22:14:45.872115Z","iopub.status.idle":"2025-06-16T22:14:45.877006Z","shell.execute_reply.started":"2025-06-16T22:14:45.872097Z","shell.execute_reply":"2025-06-16T22:14:45.876271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ========== MEL FUNCTION ===================\n# config object has these:\nmel_transform = T.MelSpectrogram(\n    sample_rate=config.FS,\n    n_fft=config.N_FFT,\n    hop_length=config.HOP_LENGTH,\n    n_mels=config.N_MELS,\n    f_min=config.FMIN,\n    f_max=config.FMAX,\n    power=2.0,\n).to(device)  # 👈 GPU-par move\n\n# Your new GPU-compatible function\ndef audio2melspec_gpu(audio_data):\n    # Convert numpy to torch tensor and add batch/channel dims\n    if np.isnan(audio_data).any():\n        mean_val = np.nanmean(audio_data)\n        audio_data = np.nan_to_num(audio_data, nan=mean_val)\n\n    waveform = torch.tensor(audio_data, dtype=torch.float32).unsqueeze(0).to(device)  # shape: (1, samples)\n\n    # Apply mel spectrogram\n    mel = mel_transform(waveform)\n\n    # Convert to decibel scale\n    mel_db = F.amplitude_to_DB(mel, multiplier=10.0, amin=1e-10, db_multiplier=0.0)\n\n    # Normalize to [0, 1]\n    mel_db = (mel_db - mel_db.min()) / (mel_db.max() - mel_db.min() + 1e-8)\n\n    return mel_db.squeeze(0).cpu().numpy()  # shape: (n_mels, time)\n\ndef is_valid_5sec_audio(audio, sample_rate=32000, duration_sec=5):\n    expected_len = sample_rate * duration_sec\n    return len(audio) == expected_len\n\n    \n# ====== PROCESS FUNCTION ===================\n\n\ndef prepare_audio(audio, target_len):\n    \"\"\"\n    Ensure audio is exactly `target_len` samples.\n    - If too short: zero-pad\n    - If too long: center-trim\n    \"\"\"\n    current_len = len(audio)\n\n    if current_len < target_len:\n        # Zero pad (centered)\n        pad_left = (target_len - current_len) // 2\n        pad_right = target_len - current_len - pad_left\n        audio = np.pad(audio, (pad_left, pad_right), mode='constant')\n\n    elif current_len > target_len:\n        # Center trim\n        start = (current_len - target_len) // 2\n        audio = audio[start: start + target_len]\n\n    return audio\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T22:14:46.153764Z","iopub.execute_input":"2025-06-16T22:14:46.154054Z","iopub.status.idle":"2025-06-16T22:14:46.326455Z","shell.execute_reply.started":"2025-06-16T22:14:46.154033Z","shell.execute_reply":"2025-06-16T22:14:46.325827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(config.FS, config.TARGET_DURATION, config.TARGET_SHAPE)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T22:14:51.295324Z","iopub.execute_input":"2025-06-16T22:14:51.295644Z","iopub.status.idle":"2025-06-16T22:14:51.300178Z","shell.execute_reply.started":"2025-06-16T22:14:51.295620Z","shell.execute_reply":"2025-06-16T22:14:51.299258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_audio_with_batches(df, label, config, batch_size=1000):\n    print(f\"🔄 Processing {label} audio data with batch size {batch_size}...\\n\")\n    start_time = time.time()\n\n    bird_data = {}\n    errors = []\n    target_len = int(config.TARGET_DURATION * config.FS)\n    batch_num = 0\n    success_count = 0\n\n    for i, row in tqdm(df.iterrows(), total=len(df), leave=True, dynamic_ncols=True):\n        try:\n            # Load and resample if needed\n            audio, sr = torchaudio.load(row.filepath)\n            audio = audio.mean(dim=0).numpy()  # mono\n\n            if sr != config.FS:\n                audio = torchaudio.functional.resample(\n                    torch.tensor(audio), orig_freq=sr, new_freq=config.FS\n                ).numpy()\n\n            # Preprocess and convert to mel\n            audio = prepare_audio(audio, target_len)\n            mel = audio2melspec_gpu(audio)\n            mel = cv2.resize(mel, config.MEL_SHAPE[::-1])\n            mel = cv2.cvtColor((mel * 255).astype(np.uint8), cv2.COLOR_GRAY2RGB)\n            mel = mel.transpose(2, 0, 1)\n\n            # Save with unique key\n            bird_data[row.chunk_id] = mel.astype(np.float32)\n            success_count += 1\n\n            # Save batch\n            if (i + 1) % batch_size == 0:\n                batch_filename = f'{label}_batch_{batch_num}.npz'\n                np.savez_compressed(batch_filename, **bird_data)\n                #print(f\"💾 Batch {batch_num} saved with {len(bird_data)} samples ✅\")\n                tqdm.write(f\"💾 Batch {batch_num} saved with {len(bird_data)} samples ✅\")\n                bird_data.clear()\n                batch_num += 1\n\n        except Exception as e:\n            #print(f\"❌ Error on {row.chunk_id}: {e}\")\n            tqdm.write(f\"❌ Error on {row.chunk_id}: {e}\")\n            errors.append((row.chunk_id, row.filepath, str(e)))\n\n    # Save remaining samples\n    if bird_data:\n        batch_filename = f'{label}_batch_{batch_num}.npz'\n        np.savez_compressed(batch_filename, **bird_data)\n        print(f\"💾 Final batch {batch_num} saved with {len(bird_data)} samples ✅\")\n\n    end_time = time.time()\n    print(f\"\\n✅ Finished processing '{label}' in {end_time - start_time:.1f} seconds\")\n    print(f\"🟢 Total successful: {success_count}\")\n    print(f\"🔴 Total failed: {len(errors)}\")\n\n    return errors\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T22:14:51.539184Z","iopub.execute_input":"2025-06-16T22:14:51.539952Z","iopub.status.idle":"2025-06-16T22:14:51.549462Z","shell.execute_reply.started":"2025-06-16T22:14:51.539926Z","shell.execute_reply":"2025-06-16T22:14:51.548642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if torch.cuda.is_available():\n    torch.cuda.empty_cache()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T22:14:55.007083Z","iopub.execute_input":"2025-06-16T22:14:55.007377Z","iopub.status.idle":"2025-06-16T22:14:55.011534Z","shell.execute_reply.started":"2025-06-16T22:14:55.007357Z","shell.execute_reply":"2025-06-16T22:14:55.010589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"errors = process_audio_with_batches(clean_soundscape_df, label='clean_soundscape_mel_specs', config=config, batch_size=1000)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T22:14:55.221875Z","iopub.execute_input":"2025-06-16T22:14:55.222190Z","iopub.status.idle":"2025-06-16T23:04:25.456768Z","shell.execute_reply.started":"2025-06-16T22:14:55.222169Z","shell.execute_reply":"2025-06-16T23:04:25.455815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if torch.cuda.is_available():\n    torch.cuda.empty_cache()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T23:04:25.458245Z","iopub.execute_input":"2025-06-16T23:04:25.458527Z","iopub.status.idle":"2025-06-16T23:04:25.462553Z","shell.execute_reply.started":"2025-06-16T23:04:25.458483Z","shell.execute_reply":"2025-06-16T23:04:25.461985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ========== APPLY TO DFs ==========\n#clean_mel_specs, errors = process_audio(df=clean_train_df, label=\"clean_train_audio_mel_specs\", config=config)\n\nerrors = process_audio_with_batches(clean_train_df, label='clean_train_mel_specs', config=config, batch_size=1000)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-16T23:04:25.463402Z","iopub.execute_input":"2025-06-16T23:04:25.463897Z","execution_failed":"2025-06-17T00:11:56.910Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_full_audio_with_batches(df, label, config, batch_size=1000):\n    print(f\"🔄 Processing {label} audio data with batch size {batch_size}...\")\n    start_time = time.time()\n\n    bird_data = {}\n    errors = []\n    target_len = int(config.TARGET_DURATION * config.FS)\n    batch_num = 0\n    success_count = 0\n\n    for i, row in tqdm(df.iterrows(), total=len(df), leave=True, dynamic_ncols=True):\n        try:\n            # Load audio using torchaudio (GPU-friendly)\n            audio, sr = torchaudio.load(row.filepath)\n            audio = audio.mean(dim=0).numpy()  # convert to mono\n\n            # Resample if necessary\n            if sr != config.FS:\n                audio = torchaudio.functional.resample(\n                    torch.tensor(audio), orig_freq=sr, new_freq=config.FS\n                ).numpy()\n\n            # Pad/trim to fixed length\n            audio = prepare_audio(audio, target_len)\n\n            # Convert to mel spectrogram (GPU)\n            mel = audio2melspec_gpu(audio)\n\n            # Resize and convert to RGB\n            mel = cv2.resize(mel, config.MEL_SHAPE[::-1])\n            mel = cv2.cvtColor((mel * 255).astype(np.uint8), cv2.COLOR_GRAY2RGB)\n            mel = mel.transpose(2, 0, 1)  # (3, H, W)\n\n            # Use filename as key\n            bird_data[row.filename] = mel.astype(np.float32)\n            success_count += 1\n\n            # Save in batches\n            if (i + 1) % batch_size == 0:\n                batch_filename = f'{label}_batch_{batch_num}.npz'\n                np.savez_compressed(batch_filename, **bird_data)\n                tqdm.write(f\"💾 Batch {batch_num} saved with {len(bird_data)} samples ✅\")\n                bird_data.clear()\n                batch_num += 1\n\n        except Exception as e:\n            tqdm.write(f\"❌ Error on {row.filename}: {e}\")\n            errors.append((row.filename, row.filepath, str(e)))\n\n    # Save any remaining data\n    if bird_data:\n        batch_filename = f'{label}_batch_{batch_num}.npz'\n        np.savez_compressed(batch_filename, **bird_data)\n        print(f\"💾 Final batch {batch_num} saved with {len(bird_data)} samples ✅\")\n\n    end_time = time.time()\n    print(f\"\\n✅ Finished processing '{label}' in {end_time - start_time:.1f} seconds\")\n    print(f\"🟢 Total successful: {success_count}\")\n    print(f\"🔴 Total failed: {len(errors)}\")\n\n    return errors\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-06-17T00:11:56.911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"errors = process_full_audio_with_batches(working_df, label='working_data_mel_specs', config=config, batch_size=1000)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-06-17T00:11:56.911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}