{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.13"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":106809}],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":21784.134912,"end_time":"2026-08-27T02:54:31.754701+00:00","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-08-26T20:51:27.619789+00:00","version":"2.7.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"038bc350df07495083b00cd117fd81b3":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"063feecf6ddb4de78b7c9847f39b3af5":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"0b9882ad30df41649c42ecb70a626714":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_c3cc7968ce924a7e8595144a17809ed1","IPY_MODEL_66fb2dc5891a45e7bbbe203c4a3279bc","IPY_MODEL_3af1c4584dd540c2b0b5fcb3aaca8956"],"layout":"IPY_MODEL_e3c952726836455dad559681340ca40e","tabbable":null,"tooltip":null}},"0e8172274c5b45729edc7b66d8586deb":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"0ecd52004a1e47c1b1f3a12b9342aa55":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"17e0a20478da4f69850ed63344d311c4":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_5521e1ff9c7141478025bf35a724faf1","placeholder":"​","style":"IPY_MODEL_0ecd52004a1e47c1b1f3a12b9342aa55","tabbable":null,"tooltip":null,"value":" 2.78M/? [00:00&lt;00:00, 61.9MB/s]"}},"19728519b6be47788421d7786a3a9042":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_8cd7f9933f1e4cd997ba2942af367ccd","IPY_MODEL_2ce8fdbf38ab4d29a553e4c52cd37bc5","IPY_MODEL_ecebd255317d42a3a73c02ef08965182"],"layout":"IPY_MODEL_2dc12ff48caf4b838a645c1071b4840f","tabbable":null,"tooltip":null}},"1b6013b2a2814e409f9fd91aa42042ba":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"1f334bf4b34a4fdea2d4fc9f479e6c0e":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_acd29e93ebd54cb6acb9ae1e23640f3f","IPY_MODEL_ba7f51b151f14319be98cb04880ea1b2","IPY_MODEL_e0b388ea81a94fd49e27e3b79203ef22"],"layout":"IPY_MODEL_8602b67a20234fb48f93802926d86c11","tabbable":null,"tooltip":null}},"1fcc1e576eac40e9a579d51b0847c0bf":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"2a29c9edeea94ffcaf7ce53d56b30bec":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"2a477b85cba640d6bfcbb2cdbae52635":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_d6fb9307075c455894a29f2aa85b5194","placeholder":"​","style":"IPY_MODEL_403ac89202ec4dfdad0f2343b1333671","tabbable":null,"tooltip":null,"value":"Loading weights: 100%"}},"2ce8fdbf38ab4d29a553e4c52cd37bc5":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_49ada56f4dae41c3b9677de1cba833fb","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_905924ada3e842be920dcd4c4b23b80a","tabbable":null,"tooltip":null,"value":1}},"2dc12ff48caf4b838a645c1071b4840f":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"36c437f36d9643e8a503765011665035":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_9abc8d73304c4235aaf8eb0cdaefd5c0","IPY_MODEL_740c3649a81e4c868c64c6cdfcf9e961","IPY_MODEL_48028fbb22d049c1a2a30dea872c7356"],"layout":"IPY_MODEL_8e453879f0fc4d3d951da0060fdeafde","tabbable":null,"tooltip":null}},"3996548d65eb4f1786dbb6fb019f0ad6":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"20px"}},"39de9ba4fa834589a2890b5275399f7b":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"3af1c4584dd540c2b0b5fcb3aaca8956":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_521beb123ac64bd1976c7e7d37fb3fc4","placeholder":"​","style":"IPY_MODEL_b638453be272485db72026acc60cb1f9","tabbable":null,"tooltip":null,"value":" 15.2G/15.2G [01:03&lt;00:00, 343MB/s]"}},"3b89e01c63144db8b0594fe90ed5fffd":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_ed795f9cc3214ff4acbf51467f2ac9ea","placeholder":"​","style":"IPY_MODEL_1b6013b2a2814e409f9fd91aa42042ba","tabbable":null,"tooltip":null,"value":"Fetching 4 files: 100%"}},"3bf544e9ba8c48198e89fb8a328a719a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_ba8e6cd466594b6aa74f34e383de880c","placeholder":"​","style":"IPY_MODEL_62d94847fffe4b039685bed857908a7c","tabbable":null,"tooltip":null,"value":" 686/686 [00:00&lt;00:00, 58.1kB/s]"}},"3fa4c7d07f354607993aa9a52243d24c":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_959d863bf0484eb0a5ca45ccf1946907","IPY_MODEL_f253bf071a454160ac112cb76740336f","IPY_MODEL_e10c3d319a57451991c89ec59fb0ae2e"],"layout":"IPY_MODEL_40870bfb8fa34a849edf86c4494b5ef4","tabbable":null,"tooltip":null}},"403ac89202ec4dfdad0f2343b1333671":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"40870bfb8fa34a849edf86c4494b5ef4":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"418979b345b14597860b7c44d4c54f19":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"48028fbb22d049c1a2a30dea872c7356":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_c324e3d03641443181aa96c1af3e77e9","placeholder":"​","style":"IPY_MODEL_4f4c53a30aa646b2a074ae52933397b2","tabbable":null,"tooltip":null,"value":" 27.8k/? [00:00&lt;00:00, 2.62MB/s]"}},"49ada56f4dae41c3b9677de1cba833fb":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"20px"}},"4a6808ea812146d58ca526074c3c42f5":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"4f4c53a30aa646b2a074ae52933397b2":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"4f8c070599164ebda6a8abb454b84d04":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"4fb87c46b71b47f09d99223ca3478f2d":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"20px"}},"521beb123ac64bd1976c7e7d37fb3fc4":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"5331914f196640f6ad43eb84def9a568":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_e2e40c1e97954f2ea55f49900b98cfb1","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_d509ceaaa6974d99a49761fd97af9ac1","tabbable":null,"tooltip":null,"value":1}},"53bb2c9e1a0c4a69929e285406cc920c":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_418979b345b14597860b7c44d4c54f19","placeholder":"​","style":"IPY_MODEL_dc4e396854534910808d48b812ee5dc0","tabbable":null,"tooltip":null,"value":"config.json: 100%"}},"5521e1ff9c7141478025bf35a724faf1":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"597d5c65ebbc4266b39840f3756e420c":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_f53ed9a5ccf9465397d5603f8dbca28f","placeholder":"​","style":"IPY_MODEL_91c1c560c3184ca786a583f45e806335","tabbable":null,"tooltip":null,"value":"generation_config.json: 100%"}},"62d94847fffe4b039685bed857908a7c":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"65d76415b9a44eae8ce15238b6050542":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_53bb2c9e1a0c4a69929e285406cc920c","IPY_MODEL_89f7a86f8f05492fa35e8012635bec6a","IPY_MODEL_3bf544e9ba8c48198e89fb8a328a719a"],"layout":"IPY_MODEL_4a6808ea812146d58ca526074c3c42f5","tabbable":null,"tooltip":null}},"66dd37c354ce4436aac48ac28bc5c75a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_4f8c070599164ebda6a8abb454b84d04","max":4,"min":0,"orientation":"horizontal","style":"IPY_MODEL_2a29c9edeea94ffcaf7ce53d56b30bec","tabbable":null,"tooltip":null,"value":4}},"66fb2dc5891a45e7bbbe203c4a3279bc":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_3996548d65eb4f1786dbb6fb019f0ad6","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_b70221b38f99416da9f682100096bbaf","tabbable":null,"tooltip":null,"value":1}},"6984f14468e54fa1b7a3d13ac702773c":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"6fab4a6cbb234510a8790a8fe672013b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_ab92b8036a0440f8a66429eaad7f6720","placeholder":"​","style":"IPY_MODEL_c47a700254d14609ad00f77d4e9f97b9","tabbable":null,"tooltip":null,"value":"vocab.json: "}},"71f069844f6042118f2019d09f2c35e6":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"740c3649a81e4c868c64c6cdfcf9e961":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_9b0fff199cd844e0b096d61cbb69f72e","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_71f069844f6042118f2019d09f2c35e6","tabbable":null,"tooltip":null,"value":1}},"7608cb5099654e78a06da3ae748f8545":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"7c7185fe449d4faba39067a5d01038d5":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"7cc53fbc5f6448d7a80f98d2143eb641":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"809b2eff3490449f8d437a66c8fc7da5":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"80ac08ef3bcf45a4b67f5d294f383f97":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_6fab4a6cbb234510a8790a8fe672013b","IPY_MODEL_5331914f196640f6ad43eb84def9a568","IPY_MODEL_17e0a20478da4f69850ed63344d311c4"],"layout":"IPY_MODEL_7c7185fe449d4faba39067a5d01038d5","tabbable":null,"tooltip":null}},"8602b67a20234fb48f93802926d86c11":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"885d0479b6f84551b8cb5e63d673f551":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"89f7a86f8f05492fa35e8012635bec6a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_7608cb5099654e78a06da3ae748f8545","max":686,"min":0,"orientation":"horizontal","style":"IPY_MODEL_ae0acdc231584e03a23505e06f3b0687","tabbable":null,"tooltip":null,"value":686}},"8a6b20b801f64638b623c084434ca089":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"8cd7f9933f1e4cd997ba2942af367ccd":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_39de9ba4fa834589a2890b5275399f7b","placeholder":"​","style":"IPY_MODEL_0e8172274c5b45729edc7b66d8586deb","tabbable":null,"tooltip":null,"value":"tokenizer_config.json: "}},"8d470c2a399f4c3cafd4cc183a455580":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"8e453879f0fc4d3d951da0060fdeafde":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8e5891bafcb94cef912f4c1985aeb1a5":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8efa4ac2a29241d19150c37d25d2012a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"905924ada3e842be920dcd4c4b23b80a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"91c1c560c3184ca786a583f45e806335":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"925fe55265d34bcd88710ad030b38620":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_bfd1b07e21c940a7a7cdb8788fca2a3f","placeholder":"​","style":"IPY_MODEL_063feecf6ddb4de78b7c9847f39b3af5","tabbable":null,"tooltip":null,"value":" 4/4 [01:03&lt;00:00, 63.23s/it]"}},"959d863bf0484eb0a5ca45ccf1946907":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_6984f14468e54fa1b7a3d13ac702773c","placeholder":"​","style":"IPY_MODEL_8efa4ac2a29241d19150c37d25d2012a","tabbable":null,"tooltip":null,"value":"tokenizer.json: "}},"97d536cb993344d3bd4816e2fac227ff":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"987aa70251cc47cfbb39908bb72d56db":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_e8854cb329064a48ba79f45bf4679e7e","max":339,"min":0,"orientation":"horizontal","style":"IPY_MODEL_7cc53fbc5f6448d7a80f98d2143eb641","tabbable":null,"tooltip":null,"value":339}},"9abc8d73304c4235aaf8eb0cdaefd5c0":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_8e5891bafcb94cef912f4c1985aeb1a5","placeholder":"​","style":"IPY_MODEL_9e6f64f6bf94415589616686ffb9e9a8","tabbable":null,"tooltip":null,"value":"model.safetensors.index.json: "}},"9b0fff199cd844e0b096d61cbb69f72e":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"20px"}},"9e6f64f6bf94415589616686ffb9e9a8":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"ab92b8036a0440f8a66429eaad7f6720":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"acd29e93ebd54cb6acb9ae1e23640f3f":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_c4a1ac99460049f68e17d945dced9259","placeholder":"​","style":"IPY_MODEL_8d470c2a399f4c3cafd4cc183a455580","tabbable":null,"tooltip":null,"value":"merges.txt: "}},"ad905a5e36e3420f9ab7c671bacbbddc":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_3b89e01c63144db8b0594fe90ed5fffd","IPY_MODEL_66dd37c354ce4436aac48ac28bc5c75a","IPY_MODEL_925fe55265d34bcd88710ad030b38620"],"layout":"IPY_MODEL_97d536cb993344d3bd4816e2fac227ff","tabbable":null,"tooltip":null}},"ae0acdc231584e03a23505e06f3b0687":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"b566be24ad9e4b7fa0fea34d241101f4":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"b638453be272485db72026acc60cb1f9":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"b70221b38f99416da9f682100096bbaf":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"b843afb99d64429dbcf4a7f4cc25f5e1":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"ba7f51b151f14319be98cb04880ea1b2":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_4fb87c46b71b47f09d99223ca3478f2d","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_c8e4e14d8cbd4e0dace997aa251797d0","tabbable":null,"tooltip":null,"value":1}},"ba8e6cd466594b6aa74f34e383de880c":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"bfd1b07e21c940a7a7cdb8788fca2a3f":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"c02da610ddcc4fd3af5fefc1a40b5c9b":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"20px"}},"c324e3d03641443181aa96c1af3e77e9":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"c35c9a1f011245a789d702739e8f973c":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"c3cc7968ce924a7e8595144a17809ed1":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_e6946d7b533b41a0ad2ad709ecf5364b","placeholder":"​","style":"IPY_MODEL_c35c9a1f011245a789d702739e8f973c","tabbable":null,"tooltip":null,"value":"Download complete: 100%"}},"c47a700254d14609ad00f77d4e9f97b9":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"c4a1ac99460049f68e17d945dced9259":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"c831ad9dcf78470eb975c415761bbbe5":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"c8e4e14d8cbd4e0dace997aa251797d0":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"ca261500c5b5443f93e552dfc312fee3":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"cad54ad2f78c4c86a681d24230bc41ee":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_597d5c65ebbc4266b39840f3756e420c","IPY_MODEL_e5bd072a1dc44235bba47f6a4d787b15","IPY_MODEL_fa538382d7084317a6e09b2153a6fd58"],"layout":"IPY_MODEL_d38b42081df9430dbb84f804f41e3e56","tabbable":null,"tooltip":null}},"d38b42081df9430dbb84f804f41e3e56":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"d509ceaaa6974d99a49761fd97af9ac1":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"d6fb9307075c455894a29f2aa85b5194":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"dc4e396854534910808d48b812ee5dc0":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"df7021213b804c5cb9e2d5727e985905":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_2a477b85cba640d6bfcbb2cdbae52635","IPY_MODEL_987aa70251cc47cfbb39908bb72d56db","IPY_MODEL_dfe2000e7d0146118d5f11a4a0137cb5"],"layout":"IPY_MODEL_eba9677a5c98456ea1f6fae97f942643","tabbable":null,"tooltip":null}},"dfe2000e7d0146118d5f11a4a0137cb5":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_885d0479b6f84551b8cb5e63d673f551","placeholder":"​","style":"IPY_MODEL_8a6b20b801f64638b623c084434ca089","tabbable":null,"tooltip":null,"value":" 339/339 [01:01&lt;00:00,  6.71it/s, Materializing param=model.norm.weight]"}},"e0b388ea81a94fd49e27e3b79203ef22":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_fa69920a234f4336a3f4cc39af20a5d3","placeholder":"​","style":"IPY_MODEL_c831ad9dcf78470eb975c415761bbbe5","tabbable":null,"tooltip":null,"value":" 1.67M/? [00:00&lt;00:00, 51.0MB/s]"}},"e10c3d319a57451991c89ec59fb0ae2e":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_b566be24ad9e4b7fa0fea34d241101f4","placeholder":"​","style":"IPY_MODEL_809b2eff3490449f8d437a66c8fc7da5","tabbable":null,"tooltip":null,"value":" 7.03M/? [00:00&lt;00:00, 98.4MB/s]"}},"e2e40c1e97954f2ea55f49900b98cfb1":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"20px"}},"e30088c5ec4a4477b4db8e7f772f4156":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"e3c952726836455dad559681340ca40e":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"e5bd072a1dc44235bba47f6a4d787b15":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_038bc350df07495083b00cd117fd81b3","max":138,"min":0,"orientation":"horizontal","style":"IPY_MODEL_f827c925dd7144469894d1753a8cd3a1","tabbable":null,"tooltip":null,"value":138}},"e6946d7b533b41a0ad2ad709ecf5364b":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"e8854cb329064a48ba79f45bf4679e7e":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"eba9677a5c98456ea1f6fae97f942643":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"ecebd255317d42a3a73c02ef08965182":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_1fcc1e576eac40e9a579d51b0847c0bf","placeholder":"​","style":"IPY_MODEL_ca261500c5b5443f93e552dfc312fee3","tabbable":null,"tooltip":null,"value":" 7.23k/? [00:00&lt;00:00, 608kB/s]"}},"ed795f9cc3214ff4acbf51467f2ac9ea":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"ee059a6366ac4a90ab5959a23b6722ce":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"f253bf071a454160ac112cb76740336f":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_c02da610ddcc4fd3af5fefc1a40b5c9b","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_e30088c5ec4a4477b4db8e7f772f4156","tabbable":null,"tooltip":null,"value":1}},"f53ed9a5ccf9465397d5603f8dbca28f":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"f827c925dd7144469894d1753a8cd3a1":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"fa538382d7084317a6e09b2153a6fd58":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_ee059a6366ac4a90ab5959a23b6722ce","placeholder":"​","style":"IPY_MODEL_b843afb99d64429dbcf4a7f4cc25f5e1","tabbable":null,"tooltip":null,"value":" 138/138 [00:00&lt;00:00, 14.8kB/s]"}},"fa69920a234f4336a3f4cc39af20a5d3":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"0fe9283b","cell_type":"markdown","source":"### Block 1: Setup, Config & Utils","metadata":{"papermill":{"duration":0.018066,"end_time":"2026-08-26T20:51:30.264821+00:00","exception":false,"start_time":"2026-08-26T20:51:30.246755+00:00","status":"completed"},"tags":[]}},{"id":"8abb2318","cell_type":"code","source":"!pip install transformers accelerate bitsandbytes scipy -q\n","metadata":{"execution":{"iopub.status.busy":"2026-08-27T03:44:18.99104Z","iopub.execute_input":"2026-08-27T03:44:18.991361Z","iopub.status.idle":"2026-08-27T03:44:25.80596Z","shell.execute_reply.started":"2026-08-27T03:44:18.99133Z","shell.execute_reply":"2026-08-27T03:44:25.805118Z"},"papermill":{"duration":201.856707,"end_time":"2026-08-26T20:54:52.13747+00:00","exception":false,"start_time":"2026-08-26T20:51:30.280763+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"4e65fc84","cell_type":"code","source":"# ============================================================================\n# SETUP, IMPORTS & UTILS\n# ============================================================================\n\nimport os\nimport h5py\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.cuda.amp import autocast, GradScaler\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.nn.utils.rnn import pad_sequence, pack_padded_sequence, pad_packed_sequence\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom tqdm import tqdm\nimport random\nimport editdistance\nfrom scipy.ndimage import gaussian_filter1d\n\ntry:\n    import kagglehub\nexcept ImportError:\n    kagglehub = None\n\n\ndef resolve_competition_path(local_path, competition_slug):\n    \"\"\"Use the local Kaggle mount if present, otherwise fetch via kagglehub.\n    Removes the hard dependency on one specific mount path/environment.\"\"\"\n    if os.path.exists(local_path):\n        return local_path\n    if kagglehub is not None:\n        return os.path.join(kagglehub.competition_download(competition_slug),\n                             \"t15_copyTask_neuralData/hdf5_data_final\")\n    raise FileNotFoundError(f\"{local_path} not found and kagglehub unavailable.\")\n\n\n# 40-way phoneme inventory used by the Brain-to-Text baseline (index 0 = CTC blank,\n# last entry = word-boundary/silence). These IDs are exactly what's already stored\n# in each trial's `seq_class_ids` inside the hdf5 files, so we decode phonemes\n# directly instead of building a character vocabulary from the raw sentence text.\n# This is the single biggest accuracy lever available here: neural activity in this\n# dataset aligns to articulatory/phoneme structure far more cleanly than to letters,\n# and phoneme CTC + lexicon-constrained decoding is what the strongest solutions on\n# this benchmark use.\nPHONEME_VOCAB = [\n    'BLANK', 'AA', 'AE', 'AH', 'AO', 'AW', 'AY', 'B', 'CH', 'D', 'DH',\n    'EH', 'ER', 'EY', 'F', 'G', 'HH', 'IH', 'IY', 'JH', 'K', 'L', 'M', 'N',\n    'NG', 'OW', 'OY', 'P', 'R', 'S', 'SH', 'T', 'TH', 'UH', 'UW', 'V', 'W',\n    'Y', 'Z', 'ZH', ' | ',\n]\n\nCONFIG = {\n    'data_dir': resolve_competition_path(\n        '/kaggle/input/competitions/brain-to-text-25/t15_copyTask_neuralData/hdf5_data_final',\n        'brain-to-text-25',\n    ),\n    'device': 'cuda' if torch.cuda.is_available() else 'cpu',\n    'batch_size': 16,\n    'num_epochs': 82,\n\n    # Model architecture\n    'd_model': 384,\n    'n_heads': 6,\n    'n_layers': 4,\n    'd_ff': 1536,\n    'patch_size': 3,\n    'lstm_hidden': 256,\n    'lstm_layers': 2,\n    'n_classes': len(PHONEME_VOCAB),   # was a small character vocab built from raw text\n\n    # Regularization & Adaptation\n    'dropout': 0.4,\n    'head_dim': 256,\n    'attn_dropout': 0.5,\n    'drop_path_rate': 0.2,\n    'smooth_kernel_std': 2.0,\n    'smooth_kernel_size': 100,\n    'drift_lambda': 0.01,\n\n    # Optimizer\n    'learning_rate': 5e-4,\n    'weight_decay': 1e-4,\n    'use_augmentation': True,\n\n    # Memory / stability\n    'use_amp': True,          # NEW: mixed precision -> lower memory, room for bigger batches\n    'grad_clip': 5.0,\n\n    # Checkpointing - no hardcoded personal Kaggle model path. Leave None to train\n    # from scratch, or point it at your own checkpoint to resume.\n    'resume_from_checkpoint': None,\n    'checkpoint_dir': '/kaggle/working',\n}\n\nos.makedirs(CONFIG['checkpoint_dir'], exist_ok=True)\nprint(f\"Device: {CONFIG['device']}\")\nprint(f\"PyTorch version: {torch.__version__}\")\nprint(f\"Phoneme classes (incl. blank): {CONFIG['n_classes']}\")\n\n# --- Helper: Stochastic Depth ---\ndef drop_path(x, drop_prob: float = 0., training: bool = False, scale_by_keep: bool = True):\n    if drop_prob == 0. or not training:\n        return x\n    keep_prob = 1 - drop_prob\n    shape = (x.shape[0],) + (1,) * (x.ndim - 1)\n    random_tensor = x.new_empty(shape).bernoulli_(keep_prob)\n    if keep_prob > 0.0 and scale_by_keep:\n        random_tensor.div_(keep_prob)\n    return x * random_tensor\n\n# --- Helper: Global Session Mapper ---\ndef get_session2idx(data_dir):\n    \"\"\"Scans data directory to find all unique sessions chronologically.\"\"\"\n    from glob import glob\n    paths = glob(f'{data_dir}/**/data_*.hdf5', recursive=True)\n    sessions = sorted(list(set([Path(p).parent.name for p in paths])))\n    return {s: i for i, s in enumerate(sessions)}\n","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2026-08-27T03:44:25.807448Z","iopub.execute_input":"2026-08-27T03:44:25.807751Z","iopub.status.idle":"2026-08-27T03:44:28.600425Z","shell.execute_reply.started":"2026-08-27T03:44:25.807713Z","shell.execute_reply":"2026-08-27T03:44:28.599474Z"},"papermill":{"duration":7.207438,"end_time":"2026-08-26T20:54:59.36183+00:00","exception":false,"start_time":"2026-08-26T20:54:52.154392+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"f9980fa4","cell_type":"markdown","source":"### Block 2: Data Loading, Augmentation & Dataset","metadata":{"papermill":{"duration":0.016825,"end_time":"2026-08-26T20:54:59.395498+00:00","exception":false,"start_time":"2026-08-26T20:54:59.378673+00:00","status":"completed"},"tags":[]}},{"id":"076c0441","cell_type":"code","source":"# ============================================================================\n# DATA AUGMENTATION, LOADING & DATASET  (phoneme targets + global normalization)\n# ============================================================================\n\nclass NeuralAugmentation:\n    def __init__(self, p=0.5):\n        self.p = p\n    def __call__(self, neural):\n        if random.random() > self.p:\n            return neural\n        if random.random() < 0.3:\n            neural = self.time_warp(neural)\n        if random.random() < 0.3:\n            noise_level = random.uniform(0.01, 0.05)\n            neural = neural + torch.randn_like(neural) * noise_level\n        if random.random() < 0.2:\n            n_channels = neural.shape[1]\n            n_drop = int(n_channels * 0.1)\n            drop_indices = random.sample(range(n_channels), n_drop)\n            neural[:, drop_indices] = 0\n        return neural\n    def time_warp(self, neural):\n        seq_len = len(neural)\n        warp_factor = random.uniform(0.9, 1.1)\n        new_len = int(seq_len * warp_factor)\n        if new_len < 10: return neural\n        indices = torch.linspace(0, seq_len - 1, new_len)\n        indices_floor = indices.long()\n        indices_ceil = torch.clamp(indices_floor + 1, max=seq_len - 1)\n        alpha = (indices - indices_floor.float()).unsqueeze(1)\n        warped = (1 - alpha) * neural[indices_floor] + alpha * neural[indices_ceil]\n        final_indices = torch.linspace(0, new_len - 1, seq_len).long()\n        return warped[final_indices]\n\n\ndef load_split(data_dir, split='train'):\n    from glob import glob\n    pattern = f'{data_dir}/**/data_{split}.hdf5'\n    files = sorted(glob(pattern, recursive=True))\n\n    print(f\"\\nLoading {split} split...\")\n    all_data = {k: [] for k in ['neural', 'n_steps', 'sentence', 'phonemes',\n                                 'phoneme_len', 'session', 'block', 'trial']}\n    for filepath in tqdm(files):\n        session_name = Path(filepath).parent.name\n        with h5py.File(filepath, 'r') as f:\n            for trial_key in f.keys():\n                trial = f[trial_key]\n                all_data['neural'].append(trial['input_features'][:])\n                all_data['n_steps'].append(trial.attrs['n_time_steps'])\n                all_data['session'].append(session_name)\n                all_data['block'].append(trial.attrs['block_num'])\n                all_data['trial'].append(trial.attrs['trial_num'])\n\n                sentence = trial.attrs.get('sentence_label')\n                all_data['sentence'].append(sentence.decode('utf-8') if isinstance(sentence, bytes) else sentence)\n\n                # Phoneme targets already ship inside the hdf5 - use them directly\n                # instead of deriving a character vocabulary from the sentence text.\n                phon = trial['seq_class_ids'][:] if 'seq_class_ids' in trial else np.array([], dtype=np.int64)\n                phon_len = int(trial.attrs['seq_len']) if 'seq_len' in trial.attrs else len(phon)\n                all_data['phonemes'].append(phon)\n                all_data['phoneme_len'].append(phon_len)\n    print(f\"✓ Loaded {len(all_data['neural'])} samples\")\n    return all_data\n\n\ndef compute_channel_stats(data):\n    \"\"\"\n    Streaming (Welford) per-channel mean/std over every timestep of every TRAIN\n    trial, computed once and reused unchanged for val/test. Fixes the previous\n    setup, where every trial (train, val, and test alike) normalized itself\n    independently and test additionally clipped while train did not - a\n    train/test mismatch that especially hurts the day-adaptive layer, which\n    needs a stable input scale to learn a meaningful per-day transform.\n    \"\"\"\n    n_channels = data['neural'][0].shape[1]\n    count = 0\n    mean = np.zeros(n_channels, dtype=np.float64)\n    M2 = np.zeros(n_channels, dtype=np.float64)\n    for feat, n_steps in zip(data['neural'], data['n_steps']):\n        x = feat[:n_steps].astype(np.float64)\n        for row in x:\n            count += 1\n            delta = row - mean\n            mean += delta / count\n            M2 += delta * (row - mean)\n    std = np.sqrt(M2 / max(count - 1, 1))\n    std[std < 1e-6] = 1e-6\n    return torch.tensor(mean, dtype=torch.float32), torch.tensor(std, dtype=torch.float32)\n\n\nclass BrainToTextDataset(Dataset):\n    def __init__(self, data, session2idx, feat_mean, feat_std, augment=False, clip=5.0):\n        self.neural = data['neural']\n        self.n_steps = data['n_steps']\n        self.sentences = data['sentence']\n        self.sessions = data['session']\n        self.phonemes = data['phonemes']\n        self.phoneme_len = data['phoneme_len']\n        self.session2idx = session2idx\n        self.feat_mean = feat_mean\n        self.feat_std = feat_std\n        self.clip = clip\n        self.augment = augment\n        self.augmentation = NeuralAugmentation(p=0.5) if augment else None\n\n    def __len__(self): return len(self.neural)\n\n    def __getitem__(self, idx):\n        neural = self.neural[idx][:self.n_steps[idx]]\n        neural = torch.FloatTensor(neural)\n\n        # Global, split-independent normalization - same stats for train/val/test.\n        neural = (neural - self.feat_mean) / self.feat_std\n        neural = torch.clamp(neural, -self.clip, self.clip)\n\n        if self.augment and self.augmentation:\n            neural = self.augmentation(neural)\n\n        target = torch.LongTensor(self.phonemes[idx])\n        target_length = self.phoneme_len[idx]\n\n        return {\n            'neural': neural,\n            'target': target,\n            'length': len(neural),\n            'target_length': target_length,\n            'sentence': self.sentences[idx] if self.sentences[idx] else \"\",\n            'day_idx': self.session2idx[self.sessions[idx]]\n        }\n\n\ndef collate_fn(batch):\n    batch = sorted(batch, key=lambda x: x['length'], reverse=True)\n    neurals = pad_sequence([item['neural'] for item in batch], batch_first=True)\n    targets = pad_sequence([item['target'] for item in batch], batch_first=True)\n    return {\n        'neural': neurals,\n        'target': targets,\n        'lengths': torch.LongTensor([item['length'] for item in batch]),\n        'target_lengths': torch.LongTensor([item['target_length'] for item in batch]),\n        'sentences': [item['sentence'] for item in batch],\n        'day_idx': torch.LongTensor([item['day_idx'] for item in batch])\n    }\n","metadata":{"execution":{"iopub.status.busy":"2026-08-27T03:44:28.602872Z","iopub.execute_input":"2026-08-27T03:44:28.603808Z","iopub.status.idle":"2026-08-27T03:44:28.623555Z","shell.execute_reply.started":"2026-08-27T03:44:28.603773Z","shell.execute_reply":"2026-08-27T03:44:28.622834Z"},"papermill":{"duration":0.041962,"end_time":"2026-08-26T20:54:59.454331+00:00","exception":false,"start_time":"2026-08-26T20:54:59.412369+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"699a6f19","cell_type":"markdown","source":"### Block 3: The Enriched Hybrid Model","metadata":{"papermill":{"duration":0.016804,"end_time":"2026-08-26T20:54:59.4884+00:00","exception":false,"start_time":"2026-08-26T20:54:59.471596+00:00","status":"completed"},"tags":[]}},{"id":"7fb2df7f","cell_type":"code","source":"# ============================================================================\n# HYBRID MODEL (CNN -> BiLSTM -> Patch -> Transformer, day-adaptive input,\n# Gaussian smoothing). Architecture unchanged - output head now predicts\n# phoneme classes (vocab_size = CONFIG['n_classes'] = 40) instead of\n# characters; that swap happens at construction time, not in this class.\n# ============================================================================\n\n\nclass RoPE(nn.Module):\n    def __init__(self, head_dim, max_seq_len=2048):\n        super().__init__()\n        half_dim = head_dim // 2\n        freq = 1.0 / (10000 ** (torch.arange(0, half_dim, 2).float() / half_dim))\n        t = torch.arange(max_seq_len).float().unsqueeze(1)\n        angles = t * freq.unsqueeze(0)\n        cos = torch.cos(angles).repeat_interleave(2, dim=1)\n        sin = torch.sin(angles).repeat_interleave(2, dim=1)\n        self.register_buffer(\"cos\", cos.unsqueeze(0))\n        self.register_buffer(\"sin\", sin.unsqueeze(0))\n\n    def forward(self, x, seq_len):\n        cos = self.cos[:, :seq_len, :].to(x.device)\n        sin = self.sin[:, :seq_len, :].to(x.device)\n        x1, x2 = x.chunk(2, -1)\n        return torch.cat([x1 * cos - x2 * sin, x1 * sin + x2 * cos], -1)\n\nclass WideAttention(nn.Module):\n    def __init__(self, d_model, n_heads, head_dim, dropout):\n        super().__init__()\n        self.n_heads = n_heads\n        self.head_dim = head_dim\n        self.inner_dim = n_heads * head_dim\n        self.scale = head_dim ** -0.5\n        \n        self.qkv = nn.Linear(d_model, self.inner_dim * 3, bias=False)\n        self.out = nn.Linear(self.inner_dim, d_model)\n        self.dropout = nn.Dropout(dropout)\n        self.rope = RoPE(head_dim)\n    \n    def forward(self, x, mask=None):\n        B, T, C = x.shape\n        qkv = self.qkv(x).reshape(B, T, 3, self.n_heads, self.head_dim).permute(2, 0, 3, 1, 4)\n        q, k, v = qkv[0], qkv[1], qkv[2]\n        \n        q = self.rope(q, T)\n        k = self.rope(k, T)\n        \n        attn = (q @ k.transpose(-2, -1)) * self.scale\n        if mask is not None:\n            attn = attn.masked_fill(mask == 0, float('-inf'))\n        \n        attn = self.dropout(F.softmax(attn, -1))\n        out = (attn @ v).transpose(1, 2).reshape(B, T, -1)\n        return self.out(out)\n\nclass TransformerBlock(nn.Module):\n    def __init__(self, d_model, n_heads, d_ff, dropout, head_dim, attn_dropout):\n        super().__init__()\n        self.norm1 = nn.LayerNorm(d_model)\n        self.attn = WideAttention(d_model, n_heads, head_dim, attn_dropout)\n        self.norm2 = nn.LayerNorm(d_model)\n        self.ffn = nn.Sequential(\n            nn.Linear(d_model, d_ff),\n            nn.GELU(),\n            nn.Dropout(dropout),\n            nn.Linear(d_ff, d_model),\n            nn.Dropout(dropout)\n        )\n    \n    def forward(self, x, mask=None, drop_path_rate=0.0):\n        # Apply stochastic depth (drop path) around the residual connections\n        x = x + drop_path(self.attn(self.norm1(x), mask), drop_path_rate, self.training)\n        x = x + drop_path(self.ffn(self.norm2(x)), drop_path_rate, self.training)\n        return x\n\nclass HybridLSTMTransformerCTC(nn.Module):\n    def __init__(self, n_days, input_size=512, d_model=384, n_heads=6, n_layers=4,\n                 d_ff=1536, patch_size=3, vocab_size=50, dropout=0.4, head_dim=256, \n                 attn_dropout=0.5, lstm_hidden=256, lstm_layers=2, smooth_std=2.0, \n                 smooth_size=100, drop_path_rate=0.2):\n        super().__init__()\n        self.patch_size = patch_size\n        self.n_days = n_days\n        \n        # --- NEW: Gaussian Smoothing Initialization ---\n        inp = np.zeros(smooth_size, dtype=np.float32)\n        inp[smooth_size // 2] = 1\n        gaussKernel = gaussian_filter1d(inp, smooth_std)\n        valid_idx = np.argwhere(gaussKernel > 0.01)\n        gaussKernel = gaussKernel[valid_idx]\n        gaussKernel = np.squeeze(gaussKernel / np.sum(gaussKernel))\n        self.register_buffer(\"gauss_kernel\", torch.tensor(gaussKernel, dtype=torch.float32).view(1, 1, -1))\n        \n        # --- NEW: Day-Specific Linear Transformation ---\n        self.day_weights = nn.ParameterList([nn.Parameter(torch.eye(input_size)) for _ in range(n_days)])\n        self.day_biases = nn.ParameterList([nn.Parameter(torch.zeros(1, input_size)) for _ in range(n_days)])\n        self.day_activation = nn.Softsign()\n\n        self.cnn = nn.Sequential(\n            nn.Conv1d(input_size, 256, kernel_size=3, padding=1),\n            nn.BatchNorm1d(256), nn.ReLU(), nn.Dropout(dropout * 0.5),\n            nn.Conv1d(256, 256, kernel_size=3, padding=1),\n            nn.BatchNorm1d(256), nn.ReLU(), nn.Dropout(dropout * 0.5)\n        )\n        \n        self.lstm = nn.LSTM(\n            256, lstm_hidden, lstm_layers, batch_first=True, bidirectional=True,\n            dropout=dropout if lstm_layers > 1 else 0\n        )\n        \n        lstm_output_dim = lstm_hidden * 2\n        self.patch_embed = nn.Sequential(\n            nn.LayerNorm(lstm_output_dim * patch_size),\n            nn.Linear(lstm_output_dim * patch_size, d_model),\n            nn.LayerNorm(d_model), nn.Dropout(dropout)\n        )\n        \n        self.blocks = nn.ModuleList([\n            TransformerBlock(d_model, n_heads, d_ff, dropout, head_dim, attn_dropout)\n            for _ in range(n_layers)\n        ])\n        \n        self.drop_path_rates = [x.item() for x in torch.linspace(0, drop_path_rate, n_layers)]\n        \n        self.norm = nn.LayerNorm(d_model)\n        self.head = nn.Linear(d_model, vocab_size)\n    \n    def forward(self, x, lengths, day_idx):\n        B, T, C = x.shape\n        \n        # 1. Day Specific Weighting\n        W = torch.stack([self.day_weights[i] for i in day_idx], dim=0)\n        b = torch.cat([self.day_biases[i] for i in day_idx], dim=0).unsqueeze(1)\n        x = torch.einsum(\"btd,bdk->btk\", x, W) + b\n        x = self.day_activation(x)\n\n        # 2. Gaussian Smoothing via Conv1D\n        x = x.permute(0, 2, 1) # [B, C, T]\n        kernel = self.gauss_kernel.repeat(C, 1, 1).to(x.device)\n        x = F.conv1d(x, kernel, padding='same', groups=C)\n        \n        # 3. CNN\n        x = self.cnn(x)\n        x = x.permute(0, 2, 1) # [B, T, C]\n        \n        # 4. LSTM\n        x_packed = pack_padded_sequence(x, lengths.cpu(), batch_first=True, enforce_sorted=True)\n        lstm_out, _ = self.lstm(x_packed)\n        lstm_out, _ = pad_packed_sequence(lstm_out, batch_first=True)\n        \n        # 5. Patching\n        T_lstm = lstm_out.shape[1]\n        n_patches = T_lstm // self.patch_size\n        \n        if n_patches == 0:\n            n_patches = 1\n            x = lstm_out.mean(dim=1, keepdim=True)\n            x = self.patch_embed(x.reshape(B, 1, -1))\n            patch_lens = torch.ones(B, dtype=torch.long, device=x.device)\n        else:\n            x = lstm_out[:, :n_patches * self.patch_size].reshape(B, n_patches, -1)\n            x = self.patch_embed(x)\n            patch_lens = torch.clamp((lengths // self.patch_size).to(x.device), min=1)\n        \n        # 6. Transformer\n        mask = (torch.arange(n_patches, device=x.device)[None, :] < patch_lens[:, None])\n        mask = mask[:, None, None, :]\n        \n        for i, block in enumerate(self.blocks):\n            x = block(x, mask, drop_path_rate=self.drop_path_rates[i])\n        \n        # 7. Output Projection\n        logits = self.head(self.norm(x))\n        log_probs = torch.log_softmax(logits, dim=-1)\n        \n        return log_probs.transpose(0, 1), patch_lens","metadata":{"execution":{"iopub.status.busy":"2026-08-27T03:44:28.624557Z","iopub.execute_input":"2026-08-27T03:44:28.624996Z","iopub.status.idle":"2026-08-27T03:44:28.649781Z","shell.execute_reply.started":"2026-08-27T03:44:28.624973Z","shell.execute_reply":"2026-08-27T03:44:28.649009Z"},"papermill":{"duration":0.047213,"end_time":"2026-08-26T20:54:59.55232+00:00","exception":false,"start_time":"2026-08-26T20:54:59.505107+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"a66bd7db","cell_type":"markdown","source":"### Block 4: Training & Validation Logic","metadata":{"papermill":{"duration":0.016993,"end_time":"2026-08-26T20:54:59.587115+00:00","exception":false,"start_time":"2026-08-26T20:54:59.570122+00:00","status":"completed"},"tags":[]}},{"id":"f9bd8bfb","cell_type":"code","source":"# ============================================================================\n# TRAINING FUNCTIONS  (mixed precision + phoneme CTC + drift loss + lightweight SWA)\n# ============================================================================\n# NOTE: in the original notebook these definitions lived inside a *markdown*\n# cell, so none of this code actually ran. This is now a real code cell.\n\nimport json\nimport matplotlib.pyplot as plt\n\n\ndef greedy_decode_phonemes(log_probs, output_lengths):\n    \"\"\"CTC greedy decode -> list of phoneme-id sequences (blank/duplicates collapsed).\"\"\"\n    _, max_indices = log_probs.max(dim=-1)  # [T, B]\n    decoded = []\n    for b in range(max_indices.size(1)):\n        seq = max_indices[:output_lengths[b], b].cpu().numpy()\n        out, prev = [], None\n        for token in seq:\n            if token != 0 and token != prev:\n                out.append(int(token))\n            prev = token\n        decoded.append(out)\n    return decoded\n\n\ndef validate_model(model, val_loader, device):\n    \"\"\"\n    Phoneme Error Rate (PER) on the validation set - a cheap greedy decode with\n    no beam search / language model needed. Full word-level WER (lexicon +\n    n-gram + LLM rescoring) is only computed once at test time, so every\n    training epoch stays fast and doesn't need the LM/LLM loaded.\n    \"\"\"\n    model.eval()\n    total_edits, total_len = 0, 0\n    with torch.no_grad():\n        for batch in val_loader:\n            neural = batch['neural'].to(device)\n            lengths = batch['lengths']\n            day_idx = batch['day_idx'].to(device)\n            target = batch['target']\n            target_lengths = batch['target_lengths']\n\n            log_probs, output_lengths = model(neural, lengths, day_idx)\n            decoded = greedy_decode_phonemes(log_probs, output_lengths)\n\n            for b, hyp in enumerate(decoded):\n                ref = target[b, :target_lengths[b]].tolist()\n                total_edits += editdistance.eval(hyp, ref)\n                total_len += max(len(ref), 1)\n\n    return 100.0 * total_edits / max(total_len, 1)\n\n\ndef average_state_dicts(state_dicts):\n    \"\"\"Simple SWA-style weight averaging across a handful of checkpoints - a\n    memory-free way to get an ensembling boost without loading multiple full\n    models at inference time (which is what was causing the OOMs before).\"\"\"\n    avg = {k: torch.zeros_like(v, dtype=torch.float32) for k, v in state_dicts[0].items()}\n    for sd in state_dicts:\n        for k, v in sd.items():\n            avg[k] += v.float()\n    for k in avg:\n        avg[k] /= len(state_dicts)\n    return avg\n\n\ndef train_model(train_loader, val_loader, config, n_days):\n    model = HybridLSTMTransformerCTC(\n        n_days=n_days,\n        input_size=512, d_model=config['d_model'], n_heads=config['n_heads'],\n        n_layers=config['n_layers'], d_ff=config['d_ff'], patch_size=config['patch_size'],\n        vocab_size=config['n_classes'], dropout=config['dropout'], head_dim=config['head_dim'],\n        attn_dropout=config['attn_dropout'], lstm_hidden=config['lstm_hidden'],\n        lstm_layers=config['lstm_layers'], smooth_std=config['smooth_kernel_std'],\n        smooth_size=config['smooth_kernel_size'], drop_path_rate=config['drop_path_rate']\n    ).to(config['device'])\n\n    criterion = nn.CTCLoss(blank=0, zero_infinity=True)\n    optimizer = optim.AdamW(model.parameters(), lr=config['learning_rate'], weight_decay=config['weight_decay'])\n    scaler = GradScaler(enabled=config['use_amp'])\n\n    start_epoch = 0\n    if config.get('resume_from_checkpoint'):\n        print(f\"Resuming from {config['resume_from_checkpoint']}...\")\n        checkpoint = torch.load(config['resume_from_checkpoint'], map_location=config['device'])\n        model.load_state_dict(checkpoint['model_state_dict'])\n        optimizer.load_state_dict(checkpoint['optimizer_state_dict'])\n        start_epoch = checkpoint.get('epoch', -1) + 1\n        print(f\"Resumed at epoch {start_epoch}.\")\n\n    scheduler = optim.lr_scheduler.OneCycleLR(\n        optimizer, max_lr=config['learning_rate'], epochs=config['num_epochs'],\n        steps_per_epoch=len(train_loader), pct_start=0.1,\n        last_epoch=(start_epoch * len(train_loader) - 1) if start_epoch else -1\n    )\n\n    epoch_losses = []\n    epoch_pers = []\n    best_per = float('inf')\n    patience, no_improve = 10, 0\n    top_checkpoints = []  # (per, state_dict) kept for end-of-training SWA averaging\n\n    for epoch in range(start_epoch, config['num_epochs']):\n        model.train()\n        train_loss = 0\n\n        for batch in tqdm(train_loader, desc=f'Epoch {epoch+1}/{config[\"num_epochs\"]}'):\n            neural = batch['neural'].to(config['device'])\n            target = batch['target'].to(config['device'])\n            lengths = batch['lengths']\n            target_lengths = batch['target_lengths']\n            day_idx = batch['day_idx'].to(config['device'])\n\n            optimizer.zero_grad()\n            with autocast(enabled=config['use_amp']):\n                log_probs, output_lengths = model(neural, lengths, day_idx)\n                ctc_loss = criterion(log_probs.float(), target, output_lengths, target_lengths)\n\n                drift_loss = 0.0\n                if config['drift_lambda'] > 0 and n_days > 1:\n                    for d in range(1, n_days):\n                        w_diff = model.day_weights[d] - model.day_weights[d - 1]\n                        b_diff = model.day_biases[d] - model.day_biases[d - 1]\n                        drift_loss += (torch.sum(w_diff ** 2) + torch.sum(b_diff ** 2))\n                    drift_loss = drift_loss / (n_days - 1)\n\n                loss = ctc_loss + (config['drift_lambda'] * drift_loss)\n\n            scaler.scale(loss).backward()\n            scaler.unscale_(optimizer)\n            torch.nn.utils.clip_grad_norm_(model.parameters(), config['grad_clip'])\n            scaler.step(optimizer)\n            scaler.update()\n            scheduler.step()\n            train_loss += loss.item()\n\n        avg_loss = train_loss / len(train_loader)\n        epoch_losses.append(avg_loss)\n        val_per = validate_model(model, val_loader, config['device'])\n        epoch_pers.append(val_per)\n\n        print(f\"Epoch {epoch+1}: Loss={avg_loss:.4f}, PER={val_per:.2f}%\")\n\n        ckpt = {\n            'model_state_dict': model.state_dict(),\n            'optimizer_state_dict': optimizer.state_dict(),\n            'epoch': epoch,\n            'config': config,\n        }\n        torch.save(ckpt, os.path.join(config['checkpoint_dir'], 'latest_model.pt'))\n\n        if val_per < best_per:\n            best_per = val_per\n            no_improve = 0\n            torch.save({**ckpt, 'per': best_per}, os.path.join(config['checkpoint_dir'], 'best_model.pt'))\n        else:\n            no_improve += 1\n\n        # Keep the top-3 checkpoints (by PER) in CPU memory for SWA averaging -\n        # a single extra model's worth of RAM, no extra GPU memory.\n        top_checkpoints.append((val_per, {k: v.cpu().clone() for k, v in model.state_dict().items()}))\n        top_checkpoints = sorted(top_checkpoints, key=lambda x: x[0])[:3]\n\n        if no_improve >= patience:\n            print(f\"\\nEarly stopping after {epoch+1} epochs\")\n            break\n\n    if len(top_checkpoints) > 1:\n        swa_state = average_state_dicts([sd for _, sd in top_checkpoints])\n        torch.save({'model_state_dict': swa_state, 'config': config},\n                    os.path.join(config['checkpoint_dir'], 'swa_model.pt'))\n        print(f\"Saved SWA average of top {len(top_checkpoints)} checkpoints -> swa_model.pt \"\n              f\"(try this at inference time - it usually beats the single best checkpoint).\")\n\n    # Save raw history so the curves below can be regenerated later (e.g. for\n    # the paper) without re-running training.\n    figures_dir = os.path.join(config['checkpoint_dir'], 'figures')\n    os.makedirs(figures_dir, exist_ok=True)\n    with open(os.path.join(config['checkpoint_dir'], 'training_history.json'), 'w') as f:\n        json.dump({'loss': epoch_losses, 'per': epoch_pers}, f, indent=2)\n\n    fig, ax1 = plt.subplots(figsize=(9, 5))\n    ax1.plot(range(1, len(epoch_losses) + 1), epoch_losses, color='tab:blue', label='Training Loss')\n    ax1.set_xlabel('Epoch'); ax1.set_ylabel('CTC Loss', color='tab:blue')\n    ax1.tick_params(axis='y', labelcolor='tab:blue')\n    ax2 = ax1.twinx()\n    ax2.plot(range(1, len(epoch_pers) + 1), epoch_pers, color='tab:red', label='Val PER (%)')\n    ax2.set_ylabel('Phoneme Error Rate (%)', color='tab:red')\n    ax2.tick_params(axis='y', labelcolor='tab:red')\n    fig.suptitle('Training Loss & Validation PER over Epochs')\n    fig.tight_layout()\n    fig.savefig(os.path.join(figures_dir, 'training_curves.png'), dpi=300, bbox_inches='tight')\n    plt.show()\n\n    return model, best_per, epoch_losses, epoch_pers\n\n","metadata":{"execution":{"iopub.status.busy":"2026-08-27T03:44:28.651338Z","iopub.execute_input":"2026-08-27T03:44:28.651694Z","iopub.status.idle":"2026-08-27T03:44:28.676372Z","shell.execute_reply.started":"2026-08-27T03:44:28.651673Z","shell.execute_reply":"2026-08-27T03:44:28.675803Z"},"papermill":{"duration":0.046823,"end_time":"2026-08-26T20:54:59.650725+00:00","exception":false,"start_time":"2026-08-26T20:54:59.603902+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"689ba9a3","cell_type":"code","source":"# ============================================================================\n# BUILD DATASETS & LAUNCH TRAINING\n# ============================================================================\n# NOTE: in the original notebook this driver cell was empty, and the actual\n# train_model()/validate_model() definitions below lived inside a *markdown*\n# cell - so none of that training code ever actually ran as code. Both are\n# fixed here: this cell wires everything up, and the next cell is a proper\n# code cell.\n\nprint(\"Scanning sessions...\")\nsession2idx = get_session2idx(CONFIG['data_dir'])\nn_days = len(session2idx)\nprint(f\"Found {n_days} recording sessions/days.\")\n\ntrain_data = load_split(CONFIG['data_dir'], 'train')\nval_data = load_split(CONFIG['data_dir'], 'val')\n\nprint(\"Computing global per-channel normalization stats from TRAIN only...\")\nfeat_mean, feat_std = compute_channel_stats(train_data)\ntorch.save({'mean': feat_mean, 'std': feat_std},\n           os.path.join(CONFIG['checkpoint_dir'], 'norm_stats.pt'))\n\ntrain_dataset = BrainToTextDataset(train_data, session2idx, feat_mean, feat_std,\n                                    augment=CONFIG['use_augmentation'])\nval_dataset = BrainToTextDataset(val_data, session2idx, feat_mean, feat_std, augment=False)\n\ntrain_loader = DataLoader(train_dataset, batch_size=CONFIG['batch_size'], shuffle=True,\n                           collate_fn=collate_fn, num_workers=2, pin_memory=True)\nval_loader = DataLoader(val_dataset, batch_size=CONFIG['batch_size'], shuffle=False,\n                         collate_fn=collate_fn, num_workers=2, pin_memory=True)\n\nmodel, best_per, epoch_losses, epoch_pers = train_model(train_loader, val_loader, CONFIG, n_days)\nprint(f\"Training finished. Best validation PER: {best_per:.2f}%\")\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2026-08-27T03:44:28.677337Z","iopub.execute_input":"2026-08-27T03:44:28.67772Z"},"papermill":{"duration":21187.251857,"end_time":"2026-08-27T02:48:06.919471+00:00","exception":false,"start_time":"2026-08-26T20:54:59.667614+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"0e893cfd","cell_type":"markdown","source":"### Block 4b: Evaluation Plots (for the paper)\n\nRuns entirely on the validation set (test labels aren't available for local\nscoring). Produces, as 300dpi PNGs in `CONFIG['checkpoint_dir']/figures/`:\n- `training_curves.png` — loss & PER vs epoch (saved automatically at the end of training)\n- `phoneme_confusion_matrix.png` — alignment-based phoneme confusion matrix\n- `per_by_day.png` — validation PER broken down by recording day/session\n- `day_adaptive_drift.png` — magnitude of the learned day-specific transform vs. day index\n- `sequence_length_hist.png` — neural trial length and phoneme sequence length distributions\n","metadata":{"papermill":{"duration":2.17671,"end_time":"2026-08-27T02:48:10.975732+00:00","exception":false,"start_time":"2026-08-27T02:48:08.799022+00:00","status":"completed"},"tags":[]}},{"id":"0b83ddbb","cell_type":"code","source":"# ============================================================================\n# EVAL PLOT 1: Phoneme Confusion Matrix (validation set, alignment-based)\n# ============================================================================\n# CTC hypotheses and references have different lengths (insertions/deletions),\n# so a confusion matrix needs a Levenshtein alignment first, not just\n# position-wise comparison. Substitutions land on (true phoneme, predicted\n# phoneme); deletions land on (true phoneme, \"deletion\"); insertions land on\n# (\"insertion\", predicted phoneme).\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nfigures_dir = os.path.join(CONFIG['checkpoint_dir'], 'figures')\nos.makedirs(figures_dir, exist_ok=True)\n\n\ndef align_sequences(ref, hyp):\n    \"\"\"Standard edit-distance alignment via DP + backtrace.\n    Returns a list of (ref_token_or_None, hyp_token_or_None) pairs.\"\"\"\n    n, m = len(ref), len(hyp)\n    dp = [[0] * (m + 1) for _ in range(n + 1)]\n    for i in range(n + 1): dp[i][0] = i\n    for j in range(m + 1): dp[0][j] = j\n    for i in range(1, n + 1):\n        for j in range(1, m + 1):\n            if ref[i - 1] == hyp[j - 1]:\n                dp[i][j] = dp[i - 1][j - 1]\n            else:\n                dp[i][j] = 1 + min(dp[i - 1][j - 1], dp[i - 1][j], dp[i][j - 1])\n\n    i, j, pairs = n, m, []\n    while i > 0 or j > 0:\n        if i > 0 and j > 0 and ref[i - 1] == hyp[j - 1] and dp[i][j] == dp[i - 1][j - 1]:\n            pairs.append((ref[i - 1], hyp[j - 1])); i -= 1; j -= 1\n        elif i > 0 and j > 0 and dp[i][j] == dp[i - 1][j - 1] + 1:\n            pairs.append((ref[i - 1], hyp[j - 1])); i -= 1; j -= 1   # substitution\n        elif i > 0 and dp[i][j] == dp[i - 1][j] + 1:\n            pairs.append((ref[i - 1], None)); i -= 1                 # deletion\n        elif j > 0 and dp[i][j] == dp[i][j - 1] + 1:\n            pairs.append((None, hyp[j - 1])); j -= 1                 # insertion\n        else:\n            if i > 0: pairs.append((ref[i - 1], None)); i -= 1\n            elif j > 0: pairs.append((None, hyp[j - 1])); j -= 1\n    pairs.reverse()\n    return pairs\n\n\nphoneme_labels = PHONEME_VOCAB[1:]           # exclude BLANK (id 0)\nlabel_to_idx = {pid: i for i, pid in enumerate(range(1, len(PHONEME_VOCAB)))}\nn_labels = len(phoneme_labels)\nINS_ROW, DEL_COL = n_labels, n_labels        # extra bucket for insertions/deletions\nconf_mat = np.zeros((n_labels + 1, n_labels + 1), dtype=np.int64)\n\nmodel.eval()\nwith torch.no_grad():\n    for batch in tqdm(val_loader, desc=\"Building confusion matrix\"):\n        neural = batch['neural'].to(CONFIG['device'])\n        lengths = batch['lengths']\n        day_idx = batch['day_idx'].to(CONFIG['device'])\n        target = batch['target']\n        target_lengths = batch['target_lengths']\n\n        log_probs, output_lengths = model(neural, lengths, day_idx)\n        decoded = greedy_decode_phonemes(log_probs, output_lengths)\n\n        for b, hyp in enumerate(decoded):\n            ref = target[b, :target_lengths[b]].tolist()\n            for r, h in align_sequences(ref, hyp):\n                row = label_to_idx[r] if r is not None else INS_ROW\n                col = label_to_idx[h] if h is not None else DEL_COL\n                conf_mat[row, col] += 1\n\n# Row-normalize (recall per true phoneme) for a readable heatmap.\nrow_sums = conf_mat.sum(axis=1, keepdims=True)\nrow_sums[row_sums == 0] = 1\nconf_norm = conf_mat / row_sums\n\naxis_labels = phoneme_labels + ['(ins/del)']\nfig, ax = plt.subplots(figsize=(14, 12))\nim = ax.imshow(conf_norm, cmap='viridis', vmin=0, vmax=1)\nax.set_xticks(range(len(axis_labels))); ax.set_xticklabels(axis_labels, rotation=90, fontsize=7)\nax.set_yticks(range(len(axis_labels))); ax.set_yticklabels(axis_labels, fontsize=7)\nax.set_xlabel('Predicted phoneme'); ax.set_ylabel('True phoneme')\nax.set_title('Phoneme Confusion Matrix (row-normalized), Validation Set')\nfig.colorbar(im, ax=ax, fraction=0.046, pad=0.04, label='Fraction of true-phoneme occurrences')\nfig.tight_layout()\nfig.savefig(os.path.join(figures_dir, 'phoneme_confusion_matrix.png'), dpi=300, bbox_inches='tight')\nplt.show()\n","metadata":{"papermill":{"duration":28.474575,"end_time":"2026-08-27T02:48:41.339659+00:00","exception":false,"start_time":"2026-08-27T02:48:12.865084+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"458690e9","cell_type":"code","source":"# ============================================================================\n# EVAL PLOT 2: Validation PER by Recording Day\n# ============================================================================\n# Shows whether the day-adaptive input layer is actually closing the gap\n# across sessions, or whether some days are still much harder than others.\nidx2session = {v: k for k, v in session2idx.items()}\nday_edits = {d: 0 for d in idx2session}\nday_lens = {d: 0 for d in idx2session}\n\nmodel.eval()\nwith torch.no_grad():\n    for batch in tqdm(val_loader, desc=\"Per-day PER\"):\n        neural = batch['neural'].to(CONFIG['device'])\n        lengths = batch['lengths']\n        day_idx_batch = batch['day_idx']\n        target = batch['target']\n        target_lengths = batch['target_lengths']\n\n        log_probs, output_lengths = model(neural, lengths, day_idx_batch.to(CONFIG['device']))\n        decoded = greedy_decode_phonemes(log_probs, output_lengths)\n\n        for b, hyp in enumerate(decoded):\n            ref = target[b, :target_lengths[b]].tolist()\n            d = int(day_idx_batch[b])\n            day_edits[d] += editdistance.eval(hyp, ref)\n            day_lens[d] += max(len(ref), 1)\n\ndays_sorted = sorted(idx2session.keys())\nper_by_day = [100.0 * day_edits[d] / day_lens[d] if day_lens[d] > 0 else np.nan for d in days_sorted]\nsession_names = [idx2session[d] for d in days_sorted]\n\nfig, ax = plt.subplots(figsize=(max(10, len(days_sorted) * 0.35), 5))\nax.bar(range(len(days_sorted)), per_by_day, color='tab:orange')\nax.set_xticks(range(len(days_sorted)))\nax.set_xticklabels(session_names, rotation=90, fontsize=7)\nax.set_ylabel('Phoneme Error Rate (%)')\nax.set_title('Validation PER by Recording Session/Day')\nax.axhline(np.nanmean(per_by_day), color='black', linestyle='--', linewidth=1,\n           label=f'Mean = {np.nanmean(per_by_day):.1f}%')\nax.legend()\nfig.tight_layout()\nfig.savefig(os.path.join(figures_dir, 'per_by_day.png'), dpi=300, bbox_inches='tight')\nplt.show()\n","metadata":{"papermill":{"duration":26.959443,"end_time":"2026-08-27T02:49:10.193553+00:00","exception":false,"start_time":"2026-08-27T02:48:43.23411+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"378e1b72","cell_type":"code","source":"# ============================================================================\n# EVAL PLOT 3: Day-Adaptive Layer Drift\n# ============================================================================\n# Visualizes how far each day's learned transform has moved from identity\n# (the initialization) and from the previous day - directly illustrates what\n# the drift regularization term is constraining during training.\nwith torch.no_grad():\n    identity = torch.eye(model.day_weights[0].shape[0], device=CONFIG['device'])\n    dist_from_identity = [\n        torch.norm(model.day_weights[d] - identity).item() for d in range(n_days)\n    ]\n    consecutive_drift = [0.0] + [\n        torch.norm(model.day_weights[d] - model.day_weights[d - 1]).item()\n        for d in range(1, n_days)\n    ]\n\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(14, 5))\nax1.plot(range(n_days), dist_from_identity, marker='o', markersize=3)\nax1.set_xlabel('Day index (chronological)'); ax1.set_ylabel('‖W_day − I‖_F')\nax1.set_title('Day-Specific Transform: Distance from Identity')\nax1.grid(True, alpha=0.3)\n\nax2.plot(range(n_days), consecutive_drift, marker='o', markersize=3, color='tab:green')\nax2.set_xlabel('Day index (chronological)'); ax2.set_ylabel('‖W_day − W_(day-1)‖_F')\nax2.set_title('Day-Specific Transform: Consecutive-Day Drift')\nax2.grid(True, alpha=0.3)\n\nfig.suptitle('Effect of Day-Specific Drift Regularization (λ = %.3f)' % CONFIG['drift_lambda'])\nfig.tight_layout()\nfig.savefig(os.path.join(figures_dir, 'day_adaptive_drift.png'), dpi=300, bbox_inches='tight')\nplt.show()\n","metadata":{"papermill":{"duration":2.884469,"end_time":"2026-08-27T02:49:14.972989+00:00","exception":false,"start_time":"2026-08-27T02:49:12.08852+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"45733518","cell_type":"code","source":"# ============================================================================\n# EVAL PLOT 4: Sequence Length Distributions (dataset stats for the paper)\n# ============================================================================\ntrain_neural_lens = train_data['n_steps']\ntrain_phoneme_lens = train_data['phoneme_len']\nval_neural_lens = val_data['n_steps']\nval_phoneme_lens = val_data['phoneme_len']\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 5))\n\naxes[0].hist(train_neural_lens, bins=40, alpha=0.6, label='Train', density=True)\naxes[0].hist(val_neural_lens, bins=40, alpha=0.6, label='Val', density=True)\naxes[0].set_xlabel('Trial length (timesteps)'); axes[0].set_ylabel('Density')\naxes[0].set_title('Neural Trial Length Distribution'); axes[0].legend()\n\naxes[1].hist(train_phoneme_lens, bins=30, alpha=0.6, label='Train', density=True)\naxes[1].hist(val_phoneme_lens, bins=30, alpha=0.6, label='Val', density=True)\naxes[1].set_xlabel('Phoneme sequence length'); axes[1].set_ylabel('Density')\naxes[1].set_title('Target Phoneme Length Distribution'); axes[1].legend()\n\nfig.tight_layout()\nfig.savefig(os.path.join(figures_dir, 'sequence_length_hist.png'), dpi=300, bbox_inches='tight')\nplt.show()\n","metadata":{"papermill":{"duration":3.104859,"end_time":"2026-08-27T02:49:20.006596+00:00","exception":false,"start_time":"2026-08-27T02:49:16.901737+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"b4e53bb8","cell_type":"markdown","source":"### Block 5: Submission & Main Execution","metadata":{"papermill":{"duration":1.958945,"end_time":"2026-08-27T02:49:23.879067+00:00","exception":false,"start_time":"2026-08-27T02:49:21.920122+00:00","status":"completed"},"tags":[]}},{"id":"2bf03206-ced8-4247-a809-9668a0691895","cell_type":"markdown","source":"### [Removed] Legacy `main()` / character-vocab submission block\n\nThis cell used to hold an old `main()` (with `char2idx`, `idx2char`, a\ncharacter-level greedy decoder, and its own copy of dataset/training calls)\nleft over from before the switch to phoneme-level CTC. It referenced\n`char2idx` and a `train_model` signature that no longer exist in this\nnotebook, so if it was ever run as code it would fail immediately with a\n`NameError`. That's a likely cause of your submission run not working if you\nwere executing this cell (or \"Run All\").\n\nIts job is already done, correctly, by the cells below:\n- Training: **Block 4** (dataset build + `train_model` call)\n- Test inference + decoding + submission file: **Force Reload Model & Load\n  Datasets** → **Attach Lexicon** → **Build Decoder** → **Extract Emissions,\n  Beam Search & LLM Rescoring** (writes `submission.csv`)\n\nNothing else to do here - just don't run this cell.\n","metadata":{}},{"id":"4272ecf8","cell_type":"markdown","source":"### Force Reload Model & Load Datasets","metadata":{"papermill":{"duration":1.963398,"end_time":"2026-08-27T02:49:31.564612+00:00","exception":false,"start_time":"2026-08-27T02:49:29.601214+00:00","status":"completed"},"tags":[]}},{"id":"c89525eb","cell_type":"code","source":"# ============================================================================\n# 1. RELOAD BEST MODEL & BUILD TEST SET\n# ============================================================================\n# Reuses the exact normalization stats saved during training - no more\n# per-trial re-normalization at test time, and no more train/test mismatch\n# from clipping only on one side.\nimport os\nimport torch\nfrom torch.utils.data import DataLoader\nimport pandas as pd\nimport numpy as np\nimport h5py\nfrom glob import glob\nfrom pathlib import Path\nfrom tqdm import tqdm\n\nprint('Rebuilding session2idx / datasets / model...')\nsession2idx = get_session2idx(CONFIG['data_dir'])\nn_days = len(session2idx)\n\nnorm_stats = torch.load(os.path.join(CONFIG['checkpoint_dir'], 'norm_stats.pt'))\nfeat_mean, feat_std = norm_stats['mean'], norm_stats['std']\n\ntrain_data = load_split(CONFIG['data_dir'], 'train')  # needed for the LM corpus later\nval_data = load_split(CONFIG['data_dir'], 'val')\n\ntrain_dataset = BrainToTextDataset(train_data, session2idx, feat_mean, feat_std, augment=False)\nval_dataset = BrainToTextDataset(val_data, session2idx, feat_mean, feat_std, augment=False)\nval_loader = DataLoader(val_dataset, batch_size=CONFIG['batch_size'], shuffle=False,\n                         collate_fn=collate_fn, num_workers=2)\n\n# Prefer the SWA-averaged checkpoint if training produced one - same memory\n# footprint as a single model at inference, usually a bit better than best_model.pt.\nCKPT_CANDIDATES = [\n    os.path.join(CONFIG['checkpoint_dir'], 'swa_model.pt'),\n    os.path.join(CONFIG['checkpoint_dir'], 'best_model.pt'),\n]\nCKPT_PATH = next((p for p in CKPT_CANDIDATES if os.path.exists(p)), None)\nassert CKPT_PATH is not None, \"No checkpoint found - point CKPT_PATH at one manually.\"\nprint(f\"Loading checkpoint from: {CKPT_PATH}\")\ncheckpoint = torch.load(CKPT_PATH, map_location=CONFIG['device'])\n\nmodel = HybridLSTMTransformerCTC(\n    n_days=n_days, input_size=512, vocab_size=CONFIG['n_classes'],\n    d_model=CONFIG['d_model'], n_heads=CONFIG['n_heads'], n_layers=CONFIG['n_layers'],\n    d_ff=CONFIG['d_ff'], patch_size=CONFIG['patch_size'],\n    lstm_hidden=CONFIG['lstm_hidden'], lstm_layers=CONFIG['lstm_layers'],\n    dropout=CONFIG['dropout'], head_dim=CONFIG['head_dim'], attn_dropout=CONFIG['attn_dropout'],\n    smooth_std=CONFIG['smooth_kernel_std'], smooth_size=CONFIG['smooth_kernel_size'],\n    drop_path_rate=CONFIG['drop_path_rate'],\n).to(CONFIG['device'])\n\nmodel.load_state_dict(checkpoint['model_state_dict'])\nmodel.eval()\n\n\ndef block_load_test_data(data_dir, session2idx, feat_mean, feat_std, clip=5.0):\n    pattern = f'{data_dir}/**/data_test.hdf5'\n    files = sorted(glob(pattern, recursive=True))\n    all_samples = []\n    sample_id = 0\n    for filepath in tqdm(files, desc=\"Loading Test HDF5\"):\n        session = Path(filepath).parent.name\n        with h5py.File(filepath, 'r') as f:\n            trial_keys = [k for k in f.keys() if 'trial' in k.lower()]\n            for trial_key in trial_keys:\n                trial = f[trial_key]\n                if 'input_features' not in trial: continue\n                features = trial['input_features'][:trial.attrs['n_time_steps']]\n                features = torch.FloatTensor(features)\n                features = (features - feat_mean) / feat_std       # same stats as train/val\n                features = torch.clamp(features, -clip, clip)\n                all_samples.append({\n                    'id': sample_id, 'session': session, 'day_idx': session2idx[session],\n                    'trial_key': trial_key, 'features': features\n                })\n                sample_id += 1\n    return all_samples\n\n\ntest_samples = block_load_test_data(CONFIG['data_dir'], session2idx, feat_mean, feat_std)\nprint(f\"Setup Complete! Test samples: {len(test_samples)}\")\n","metadata":{"papermill":{"duration":295.215308,"end_time":"2026-08-27T02:54:28.734779+00:00","exception":false,"start_time":"2026-08-27T02:49:33.519471+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"ccdcf5e5","cell_type":"code","source":"# ============================================================================\n# 2. ATTACH PHONEME LEXICON + N-GRAM LM\n# ============================================================================\n# Reuses the same lexicon / tokens / KenLM binary already validated in your\n# mamba+GRU pipeline instead of compiling KenLM from source and training a new\n# *character*-level n-gram model from scratch. A phoneme-level n-gram LM is a\n# better match for phoneme CTC output, and these assets ship prebuilt - no\n# apt-get/cmake build step needed here.\nimport os\nimport kagglehub\n\nlexicon_ds = kagglehub.dataset_download('heyyousum/quality-english-dataset-for-ngram-model-v2')\nkenlm_ds = kagglehub.dataset_download('heyyousum/custom-4-gram-wiki-news-switchboard')\n\nlexicon_path = os.path.join(lexicon_ds, 'lexicon.txt')\ntokens_path = os.path.join(lexicon_ds, 'tokens.txt')\nkenlm_binary_path = os.path.join(kenlm_ds, 'custom_4gram_full.bin')\n\nprint('Lexicon:', lexicon_path)\nprint('Tokens:', tokens_path)\nprint('KenLM binary:', kenlm_binary_path)\n# NOTE: torchaudio's ctc_decoder (next cell) needs blank_token / sil_token to\n# match the exact strings used in tokens.txt - double check those two values\n# against tokens.txt if the decoder complains about an unknown token.\n","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"d3e94b8e","cell_type":"code","source":"# ============================================================================\n# 3. BUILD THE BEAM-SEARCH DECODER (phonemes -> lexicon-constrained words)\n# ============================================================================\nfrom torchaudio.models.decoder import ctc_decoder\n\nBEAM_WIDTH = 100\nNBEST = 10  # keep top-10 word hypotheses per utterance for LLM rescoring\n\ndecoder = ctc_decoder(\n    lexicon=lexicon_path,\n    tokens=tokens_path,\n    lm=kenlm_binary_path,\n    nbest=NBEST,\n    beam_size=BEAM_WIDTH,\n    lm_weight=1.5,\n    word_score=0.0,\n    blank_token='BLANK',\n    sil_token=' | ',\n)\nprint(\"Decoder ready.\")\n","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"d74b2a0c-cfe5-4d57-8224-5995ad64c369","cell_type":"code","source":"# ============================================================================\n# 4. EXTRACT EMISSIONS, BEAM SEARCH, AND 4-BIT LLM RESCORING\n# ============================================================================\nfrom transformers import AutoModelForCausalLM, AutoTokenizer, BitsAndBytesConfig\n\n# 1. Generate emissions for the test set\ndef extract_emissions(model, samples, device):\n    model.eval()\n    all_emissions, all_lengths, all_ids = [], [], []\n    with torch.no_grad():\n        for i in tqdm(range(0, len(samples), 32), desc='Extracting Test Emissions'):\n            batch = samples[i:i + 32]\n            features = [s['features'] for s in batch]\n            lengths = torch.LongTensor([len(f) for f in features])\n            day_idx = torch.LongTensor([s['day_idx'] for s in batch]).to(device)\n\n            features_padded = torch.nn.utils.rnn.pad_sequence(features, batch_first=True).to(device)\n            sorted_lengths, sorted_idx = lengths.sort(descending=True)\n\n            with autocast(enabled=CONFIG['use_amp']):\n                log_probs, output_lengths = model(features_padded[sorted_idx], sorted_lengths, day_idx[sorted_idx])\n            log_probs = log_probs.transpose(0, 1).float().cpu()  # [B, T, V] for torchaudio's decoder\n\n            unsorted = torch.empty_like(log_probs)\n            unsorted_lengths = torch.empty_like(output_lengths.cpu())\n            unsorted[sorted_idx] = log_probs\n            unsorted_lengths[sorted_idx] = output_lengths.cpu()\n\n            all_emissions.append(unsorted)\n            all_lengths.append(unsorted_lengths)\n            all_ids.extend([s['id'] for s in batch])\n    return torch.cat(all_emissions, dim=0), torch.cat(all_lengths, dim=0), all_ids\n\n\ntest_emissions, test_lengths, test_ids = extract_emissions(model, test_samples, CONFIG['device'])\n\n# 2. Batched beam search - torchaudio's ctc_decoder is already vectorized over\n#    the batch dimension, so this replaces the old per-sample python loop.\nprint(\"Running batched beam search...\")\nall_hyps = []  # list[list[str]] - nbest word strings per test sample\nDECODE_BATCH = 16\nfor i in tqdm(range(0, len(test_emissions), DECODE_BATCH), desc=\"Beam search\"):\n    emissions_chunk = test_emissions[i:i + DECODE_BATCH]\n    lengths_chunk = test_lengths[i:i + DECODE_BATCH]\n    results = decoder(emissions_chunk, lengths_chunk)\n    for sample_result in results:\n        hyps = [\" \".join(h.words) if h.words else \"\" for h in sample_result]\n        all_hyps.append(hyps if hyps else [\"\"])\n\n# 3. Load a 4-bit quantized LLM for fluency rescoring. This is the main memory\n#    fix vs. the previous fp16 7B load: roughly a 4x smaller footprint, so it\n#    comfortably shares the GPU with the acoustic model instead of OOM-ing.\nLLM_NAME = \"Qwen/Qwen2.5-7B\"\nprint(f\"Loading rescoring LLM ({LLM_NAME}) in 4-bit...\")\nbnb_config = BitsAndBytesConfig(\n    load_in_4bit=True,\n    bnb_4bit_compute_dtype=torch.float16,\n    bnb_4bit_quant_type=\"nf4\",\n    bnb_4bit_use_double_quant=True,\n)\ntokenizer = AutoTokenizer.from_pretrained(LLM_NAME)\nif tokenizer.pad_token is None:\n    tokenizer.pad_token = tokenizer.eos_token\nllm_model = AutoModelForCausalLM.from_pretrained(LLM_NAME, quantization_config=bnb_config, device_map=\"auto\")\nllm_model.eval()\n\n\ndef compute_llm_scores(sentences, batch_size=16):\n    \"\"\"Length-normalized log-likelihood (higher = more fluent) for a list of\n    candidate strings, computed in padded batches instead of one forward pass\n    per candidate - much faster with hundreds of nbest hypotheses to score.\"\"\"\n    scores = []\n    for i in range(0, len(sentences), batch_size):\n        chunk = [s if s.strip() else \"<empty>\" for s in sentences[i:i + batch_size]]\n        enc = tokenizer(chunk, return_tensors=\"pt\", padding=True, truncation=True).to(llm_model.device)\n        with torch.no_grad():\n            logits = llm_model(**enc).logits\n        shift_logits = logits[:, :-1, :]\n        shift_labels = enc[\"input_ids\"][:, 1:]\n        shift_mask = enc[\"attention_mask\"][:, 1:].float()\n        nll = F.cross_entropy(\n            shift_logits.reshape(-1, shift_logits.size(-1)),\n            shift_labels.reshape(-1),\n            reduction=\"none\",\n        ).view(shift_labels.shape)\n        token_counts = shift_mask.sum(dim=1).clamp(min=1)\n        per_seq_nll = (nll * shift_mask).sum(dim=1) / token_counts\n        scores.extend((-per_seq_nll).tolist())  # negative NLL -> higher is better\n    return scores\n\n\n# 4. Rescore each utterance's n-best list with the LLM and pick the winner\nFINAL_PREDICTIONS = []\nprint(\"Rescoring n-best hypotheses with the LLM...\")\nfor hyps in tqdm(all_hyps, desc=\"LLM rescoring\"):\n    llm_scores = compute_llm_scores(hyps, batch_size=len(hyps))\n    best_idx = int(np.argmax(llm_scores)) if llm_scores else 0\n    FINAL_PREDICTIONS.append(hyps[best_idx] if hyps else \"\")\n\n# 5. Output submission - match whatever columns the competition actually\n#    expects instead of assuming 'id'/'text'. A silent column-name mismatch\n#    against sample_submission.csv is a common reason a submission \"doesn't\n#    work\" even though the notebook itself ran fine.\nsample_sub_path = None\nfor pattern in [os.path.join(os.path.dirname(CONFIG['data_dir']), '**', 'sample_submission.csv'),\n                '/kaggle/input/**/sample_submission.csv']:\n    matches = glob(pattern, recursive=True)\n    if matches:\n        sample_sub_path = matches[0]\n        break\n\nif sample_sub_path:\n    sample_df = pd.read_csv(sample_sub_path)\n    id_col, text_col = sample_df.columns[0], sample_df.columns[1]\n    print(f\"Found sample_submission.csv at {sample_sub_path} -> using columns \"\n          f\"'{id_col}', '{text_col}'\")\nelse:\n    id_col, text_col = 'id', 'text'\n    print(\"No sample_submission.csv found automatically - defaulting to columns \"\n          \"'id', 'text'. Check the competition's Data tab and adjust id_col/text_col \"\n          \"above if it expects something else.\")\n\ndf = pd.DataFrame({id_col: test_ids, text_col: FINAL_PREDICTIONS})\ndf = df.sort_values(id_col).reset_index(drop=True)\ndf[id_col] = range(len(df))\ndf.to_csv('submission.csv', index=False)\n\nprint('\\n✓ Finished! Predictions written to submission.csv')\ndf.head(10)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"1a3eb022-6302-409f-a02f-0d09e7ea6852","cell_type":"markdown","source":"### Word Error Rate on Validation Set\n\nRuns the same lexicon + n-gram beam-search decoder used for the test\nsubmission on the validation set (which has ground-truth sentences), and\nscores it with `jiwer.wer` - the standard word-level metric. This is the\nheadline number for the paper; PER earlier is left in only as a supplementary\narchitecture diagnostic (it's what CTC training actually optimizes against,\nso it's still useful to report alongside WER, not instead of it).\n","metadata":{}},{"id":"8f355145-893a-4fce-9948-f9f64ff93a54","cell_type":"code","source":"# ============================================================================\n# WORD ERROR RATE ON VALIDATION SET + WER PLOTS\n# ============================================================================\n# No jiwer dependency - WER is just edit distance at the word level, and\n# `editdistance` is already used elsewhere in this notebook (no extra install,\n# so this part works even if the session's internet access is flaky).\n\ndef corpus_wer(refs, hyps):\n    \"\"\"Standard corpus-level WER: total word edits / total reference words.\"\"\"\n    total_edits, total_words = 0, 0\n    for r, h in zip(refs, hyps):\n        rw, hw = r.split(), h.split()\n        total_edits += editdistance.eval(rw, hw)\n        total_words += max(len(rw), 1)\n    return total_edits / max(total_words, 1)\n\n\ndef utterance_wer(ref, hyp):\n    rw, hw = ref.split(), hyp.split()\n    return editdistance.eval(rw, hw) / max(len(rw), 1)\n\n\ndef extract_emissions_from_loader(model, loader, device):\n    \"\"\"Same idea as extract_emissions() but reads directly from a DataLoader\n    that also carries reference sentences and day indices (used for val, which\n    has ground truth; the test set uses the sample-list version above).\"\"\"\n    model.eval()\n    all_emissions, all_lengths, all_refs, all_days = [], [], [], []\n    with torch.no_grad():\n        for batch in tqdm(loader, desc='Extracting Val Emissions'):\n            neural = batch['neural'].to(device)\n            lengths = batch['lengths']\n            day_idx = batch['day_idx'].to(device)\n\n            with autocast(enabled=CONFIG['use_amp']):\n                log_probs, output_lengths = model(neural, lengths, day_idx)\n            log_probs = log_probs.transpose(0, 1).float().cpu()  # [B, T, V]\n\n            all_emissions.append(log_probs)\n            all_lengths.append(output_lengths.cpu())\n            all_refs.extend(batch['sentences'])\n            all_days.extend(day_idx.cpu().tolist())\n\n    max_T = max(e.shape[1] for e in all_emissions)\n    padded = [F.pad(e, (0, 0, 0, max_T - e.shape[1])) for e in all_emissions]\n    return torch.cat(padded, dim=0), torch.cat(all_lengths, dim=0), all_refs, all_days\n\n\nval_emissions, val_lengths, val_refs, val_days = extract_emissions_from_loader(model, val_loader, CONFIG['device'])\n\nprint(\"Decoding validation set (top-1 beam, no LLM rescoring - fast WER check)...\")\nval_hyps = []\nDECODE_BATCH = 16\nfor i in tqdm(range(0, len(val_emissions), DECODE_BATCH), desc=\"Val beam search\"):\n    emissions_chunk = val_emissions[i:i + DECODE_BATCH]\n    lengths_chunk = val_lengths[i:i + DECODE_BATCH]\n    results = decoder(emissions_chunk, lengths_chunk)\n    for sample_result in results:\n        top = sample_result[0] if sample_result else None\n        val_hyps.append(\" \".join(top.words) if top and top.words else \"\")\n\n# Guard against empty references/hypotheses.\npairs = [(r, h) for r, h in zip(val_refs, val_hyps) if r and r.strip()]\nrefs_clean = [r for r, h in pairs]\nhyps_clean = [h if h.strip() else \"<empty>\" for r, h in pairs]\n\noverall_wer = corpus_wer(refs_clean, hyps_clean)\nprint(f\"\\nValidation WER (lexicon + n-gram decode, no LLM rescoring): {overall_wer * 100:.2f}%\")\nprint(\"(Run this same decode + LLM rescoring step, like the test pipeline, for the WER you'd actually submit.)\")\n\nwith open(os.path.join(CONFIG['checkpoint_dir'], 'val_wer.json'), 'w') as f:\n    json.dump({'wer': overall_wer}, f, indent=2)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"6a76673b-e4e0-4562-b11b-db3e6b4e2fde","cell_type":"code","source":"# ============================================================================\n# WER PLOTS: by-day breakdown + per-utterance distribution\n# ============================================================================\nper_utt_wer = [utterance_wer(r, h) for r, h in zip(refs_clean, hyps_clean)]\nclean_days = [d for r, d in zip(val_refs, val_days) if r and r.strip()]\n\nidx2session = {v: k for k, v in session2idx.items()}\nday_refs, day_hyps = {}, {}\nfor r, h, d in zip(refs_clean, hyps_clean, clean_days):\n    day_refs.setdefault(d, []).append(r)\n    day_hyps.setdefault(d, []).append(h)\n\ndays_sorted = sorted(day_refs.keys())\nwer_by_day = [corpus_wer(day_refs[d], day_hyps[d]) * 100 for d in days_sorted]\nsession_names = [idx2session[d] for d in days_sorted]\n\nfig, axes = plt.subplots(1, 2, figsize=(16, 5))\n\naxes[0].bar(range(len(days_sorted)), wer_by_day, color='tab:purple')\naxes[0].set_xticks(range(len(days_sorted)))\naxes[0].set_xticklabels(session_names, rotation=90, fontsize=7)\naxes[0].set_ylabel('Word Error Rate (%)')\naxes[0].set_title('Validation WER by Recording Session/Day')\naxes[0].axhline(np.mean(wer_by_day), color='black', linestyle='--', linewidth=1,\n                label=f'Mean = {np.mean(wer_by_day):.1f}%')\naxes[0].legend()\n\naxes[1].hist(np.array(per_utt_wer) * 100, bins=30, color='tab:cyan', edgecolor='black')\naxes[1].set_xlabel('Per-utterance WER (%)'); axes[1].set_ylabel('Count')\naxes[1].set_title('Distribution of Per-Utterance WER (Validation Set)')\n\nfig.tight_layout()\nfig.savefig(os.path.join(figures_dir, 'wer_by_day_and_distribution.png'), dpi=300, bbox_inches='tight')\nplt.show()\n\n# Qualitative examples table - useful for a \"sample predictions\" table in the paper\nexamples = pd.DataFrame({\n    'Reference': refs_clean[:10],\n    'Prediction': hyps_clean[:10],\n    'WER': [f'{w * 100:.1f}%' for w in per_utt_wer[:10]],\n})\nprint(\"\\nSample predictions:\")\nexamples\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"d197a2ac","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"ab106234","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"834c7632","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"80dc15dd","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"251cbd7e","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"4e8dd9ec","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"3391ec2f","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"1ed8c30d","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"87c223b4","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"a48e2698","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"3dbba267","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"33a459b8","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"3514f655","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"4b6df072","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"23c90b29","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"82af80de","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"f3304054","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"dba040db","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"c90beff3","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"7495eb41","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"fb914737","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"de3e44b0","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"fe35bdb9","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"48f795ee","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"f734f439","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"f5bbf27a","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"d17fb291","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"2f0cda5f","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"ae5ed2b0","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"e856c8a8","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"ee71e61b","cell_type":"code","source":"","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null}]}