{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.13"},"accelerator":"GPU","papermill":{"default_parameters":{},"duration":1670.465878,"end_time":"2026-07-17T03:45:08.949911+00:00","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-07-17T03:17:18.484033+00:00","version":"2.7.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"0138ce5cf4e84485ba946b185e9bf03c":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_8eea701fcf5042b7981414a0f9d98286","IPY_MODEL_b15abe24bb2941d980d7a29d0660d892","IPY_MODEL_07d2ccfad64b4464b443872b85f1843e"],"layout":"IPY_MODEL_5766909481c9421b96b8aacd76521ab5","tabbable":null,"tooltip":null}},"027c4e9418a94d49832f5fd0774ae06b":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"0725b31f08b548d6a53071adb2bcd3e8":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"07d2ccfad64b4464b443872b85f1843e":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_fcaa45fbd1ae46dd94e71e130fe41a26","placeholder":"​","style":"IPY_MODEL_84be9cbf74784f03a6749fcaa6e2731c","tabbable":null,"tooltip":null,"value":" 24/24 [04:18&lt;00:00, 10.04s/it]"}},"10fd579ad0094be1974c0250cb52be86":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"14c8b9ff69034313aef5884bbf6c1e20":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"1aa9837608684d5a99969f9c034ab34c":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_20cca6a4a2a14000a0e05d561d7dc94c","placeholder":"​","style":"IPY_MODEL_e13276c095944f9c8a887e36fe8b0f3b","tabbable":null,"tooltip":null,"value":"test: 100%"}},"1d057aa452714bf1ac2015edc7724b83":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_eca5672453734b6a8287276e5b995028","placeholder":"​","style":"IPY_MODEL_10fd579ad0094be1974c0250cb52be86","tabbable":null,"tooltip":null,"value":" 95/95 [16:59&lt;00:00, 10.21s/it]"}},"1e7092efe9124bcca6ad4cd269d00372":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_f2655e2ce8544ed19c0c479a5edb7d8b","placeholder":"​","style":"IPY_MODEL_9d1e0a2c462d4fb5abe012abd6a6f193","tabbable":null,"tooltip":null,"value":"model.safetensors: 100%"}},"20cca6a4a2a14000a0e05d561d7dc94c":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"352fe0609c0b4d44a986d7b53563af59":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"388c23010f214d1da9e101f0202865ad":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"395e03e707544b7d8c5aca367830721d":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"4247cf61a92b492ba314b27d681939ac":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_14c8b9ff69034313aef5884bbf6c1e20","max":95,"min":0,"orientation":"horizontal","style":"IPY_MODEL_a483e5069e3b4f74bebc6dabadf47f08","tabbable":null,"tooltip":null,"value":95}},"4fe79fc1e95e45d09ad187049b28ea80":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"5766909481c9421b96b8aacd76521ab5":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"5c56b72b13f84a70ba005efed0d517f8":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_b293b7e9bce049eaa2aa5e99d521eabb","placeholder":"​","style":"IPY_MODEL_f93bc3b2e39641fcb34f038076d685de","tabbable":null,"tooltip":null,"value":" 24/24 [04:06&lt;00:00,  7.33s/it]"}},"5c962edf0f28449fb145a84dfa1fe941":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_9b365e8992744fc68ced4262f65a8074","placeholder":"​","style":"IPY_MODEL_388c23010f214d1da9e101f0202865ad","tabbable":null,"tooltip":null,"value":"train: 100%"}},"6765e0d0bb8e4b6294c89379bda5ce34":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"69a2425d5b80448b9f2a01b61ef7a135":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_f69f7979fb72406bac982f90e080ec64","max":24,"min":0,"orientation":"horizontal","style":"IPY_MODEL_ec1bf13a716a4a1b951bb836f094243f","tabbable":null,"tooltip":null,"value":24}},"70637ead26404618bfd535a30a5a4514":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_027c4e9418a94d49832f5fd0774ae06b","placeholder":"​","style":"IPY_MODEL_c745af2419354be5bed571546b6ebacd","tabbable":null,"tooltip":null,"value":" 4.55G/4.55G [00:19&lt;00:00, 791MB/s]"}},"84be9cbf74784f03a6749fcaa6e2731c":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"8eea701fcf5042b7981414a0f9d98286":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_8f820d5c7c57444c8a74eed487c29a26","placeholder":"​","style":"IPY_MODEL_de5a307122624badb3f03f60dabf2635","tabbable":null,"tooltip":null,"value":"val: 100%"}},"8f820d5c7c57444c8a74eed487c29a26":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"9b365e8992744fc68ced4262f65a8074":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"9bb2e129d6a548909a8cfdf4ceae8254":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_1aa9837608684d5a99969f9c034ab34c","IPY_MODEL_69a2425d5b80448b9f2a01b61ef7a135","IPY_MODEL_5c56b72b13f84a70ba005efed0d517f8"],"layout":"IPY_MODEL_6765e0d0bb8e4b6294c89379bda5ce34","tabbable":null,"tooltip":null}},"9cf3e7453a82483bb7594b4320e30f9f":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"9d1e0a2c462d4fb5abe012abd6a6f193":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"a483e5069e3b4f74bebc6dabadf47f08":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"b15abe24bb2941d980d7a29d0660d892":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_9cf3e7453a82483bb7594b4320e30f9f","max":24,"min":0,"orientation":"horizontal","style":"IPY_MODEL_395e03e707544b7d8c5aca367830721d","tabbable":null,"tooltip":null,"value":24}},"b293b7e9bce049eaa2aa5e99d521eabb":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"bc59973cfb424006894da3f73effb47b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_1e7092efe9124bcca6ad4cd269d00372","IPY_MODEL_c674c35f9842405fa8bcc0db025b0955","IPY_MODEL_70637ead26404618bfd535a30a5a4514"],"layout":"IPY_MODEL_0725b31f08b548d6a53071adb2bcd3e8","tabbable":null,"tooltip":null}},"c674c35f9842405fa8bcc0db025b0955":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_352fe0609c0b4d44a986d7b53563af59","max":4545971222,"min":0,"orientation":"horizontal","style":"IPY_MODEL_e649220f537e44a69f22ffc6f8ff49b3","tabbable":null,"tooltip":null,"value":4545971222}},"c745af2419354be5bed571546b6ebacd":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"cfcf2def6122489bb8cebf8c7006bad1":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_5c962edf0f28449fb145a84dfa1fe941","IPY_MODEL_4247cf61a92b492ba314b27d681939ac","IPY_MODEL_1d057aa452714bf1ac2015edc7724b83"],"layout":"IPY_MODEL_4fe79fc1e95e45d09ad187049b28ea80","tabbable":null,"tooltip":null}},"de5a307122624badb3f03f60dabf2635":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"e13276c095944f9c8a887e36fe8b0f3b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"e649220f537e44a69f22ffc6f8ff49b3":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"ec1bf13a716a4a1b951bb836f094243f":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"eca5672453734b6a8287276e5b995028":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"f2655e2ce8544ed19c0c479a5edb7d8b":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"f69f7979fb72406bac982f90e080ec64":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"f93bc3b2e39641fcb34f038076d685de":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"fcaa45fbd1ae46dd94e71e130fe41a26":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 1. Setup and configuration","metadata":{}},{"cell_type":"markdown","source":"# Jaguar Re-ID · Frozen DINOv2-Giant + Triplet Head — **5-fold CV + head ensemble**\n\nThis is the 5-fold extension of the frozen DINOv2-Giant champion. The backbone stays **frozen**;\nbecause its features are extracted **once** and reused, 5-fold cross-validation only re-trains the\ntiny ~1M-parameter head five times — a few extra minutes total.\n\n**What changes vs. the single-split champion**\n- Extract frozen features for **all** training images once (not a train/val split).\n- Split with `StratifiedKFold(5)`; train a fresh head on each fold, save `triplet_best_fold{i}.pth`.\n- Report a **5-fold CV mAP** (mean ± std) instead of one split's number — removes \"split luck\".\n- At inference, **average the five heads' L2-normalised projected embeddings** into one master\n  embedding per image, then score by cosine similarity (a cheap ensemble, typically +1–2%).\n\nThe head architecture, triplet loss, batch-hard mining, PK sampling, Adagrad, and mAP computation\nare **identical** to the champion notebook — only the outer loop and inference aggregation differ.\n\n---\n\n## Kaggle usage\n1. Add the competition data (`jaguar-re-id`) as input.\n2. Set the accelerator to **GPU T4 x2** (Settings → Accelerator). Do **not** use P100 — the §1 check\n   catches the mismatch early.\n3. Leave `CONFIG[\"offline_*\"]` as `None` with internet ON (timm downloads the backbone), or attach a\n   weights dataset for offline scoring (see the champion notebook's §3 note).\n4. **Run all.** Output: `/kaggle/working/submission.csv` (137,270 rows). Runtime is well under the 9 h\n   budget — one frozen forward pass over the images dominates; the five head trainings take minutes.\n","metadata":{}},{"cell_type":"code","source":"import os, math, random, warnings\nfrom pathlib import Path\nfrom collections import Counter\n\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nfrom tqdm.auto import tqdm\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\n\nimport timm\nfrom torchvision import transforms\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics.pairwise import cosine_similarity\n\nwarnings.filterwarnings(\"ignore\")\n\nSEED = 42\nrandom.seed(SEED); np.random.seed(SEED)\ntorch.manual_seed(SEED); torch.cuda.manual_seed_all(SEED)\n\nprint(\"torch:\", torch.__version__, \"| timm:\", timm.__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-20T05:45:53.269930Z","iopub.execute_input":"2026-07-20T05:45:53.270197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Device\nif torch.cuda.is_available():\n    device = torch.device(\"cuda\")\nelif torch.backends.mps.is_available():\n    device = torch.device(\"mps\")\nelse:\n    device = torch.device(\"cpu\")\nprint(\"Device:\", device)\n\n# Fast CUDA-arch sanity check: catch GPU/torch-build mismatch in ~1s (before the 45s model download).\n# The Kaggle P100 (compute 6.0) is NOT supported by recent torch cu128 builds -> switch to GPU T4 x2.\nif device.type == \"cuda\":\n    cc = torch.cuda.get_device_capability()\n    print(f\"GPU: {torch.cuda.get_device_name(0)} | compute capability {cc[0]}.{cc[1]}\")\n    try:\n        _ = (torch.randn(64, 64, device=\"cuda\") @ torch.randn(64, 64, device=\"cuda\")).sum().item()\n        print(\"CUDA kernel sanity check: OK\")\n    except Exception as e:\n        raise RuntimeError(\n            f\"This torch build has no CUDA kernels for compute {cc[0]}.{cc[1]} \"\n            f\"({torch.cuda.get_device_name(0)}). Recent Kaggle torch (cu128) dropped older archs \"\n            \"such as the P100 (sm_60). FIX: Settings -> Accelerator -> 'GPU T4 x2' (Turing sm_75), \"\n            \"then Run All.\"\n        ) from e","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ------------------------------------------------------------------ CONFIG\nCONFIG = {\n    # --- backbone (frozen) ---\n    \"backbone_model\": \"vit_giant_patch14_dinov2.lvd142m\",  # DINOv2-Giant, 1536-d\n    \"input_size\": 518,\n    \"extract_batch_size\": 16,          # frozen forward; lower to 8 if OOM\n\n    # --- offline weights (leave None if internet is ON) ---\n    \"offline_hf_cache\": None,          # e.g. \"/kaggle/input/dinov2-giant-timm\"  (HF_HOME-style dir)\n    \"offline_weights_file\": None,      # e.g. \"/kaggle/input/.../model.safetensors\"\n\n    # --- projection head ---\n    \"embedding_dim\": 256,\n    \"hidden_dim\": 512,\n    \"dropout\": 0.3,\n\n    # --- triplet loss + PK sampling ---\n    \"triplet_margin\": 1.0,\n    \"P\": 8,                            # identities per batch\n    \"K\": 4,                            # samples per identity per batch\n\n    # --- training ---\n    \"learning_rate\": 5e-2,             # high LR is fine — only a tiny head trains\n    \"weight_decay\": 1e-4,\n    \"num_epochs\": 300,\n    \"patience\": 20,                    # early stop on val loss\n    \"val_split\": 0.2,\n    \"n_folds\": 5,                     # NEW: StratifiedKFold splits (was single 80/20)\n    \"seed\": SEED,\n\n    # --- optional enhancements (baseline = both off; see Phase P in the plan) ---\n    \"use_tta_hflip\": False,            # average frozen features of image + hflip at extraction\n\n    # --- environment ---\n    \"environment\": \"kaggle\",           # \"kaggle\" or \"local\"\n}\nCONFIG[\"batch_size\"] = CONFIG[\"P\"] * CONFIG[\"K\"]\n\nif CONFIG[\"environment\"] == \"kaggle\":\n    CONFIG[\"data_dir\"]  = Path(\"/kaggle/input/competitions/jaguar-re-id\")\n    CONFIG[\"work_dir\"]  = Path(\"/kaggle/working\")\nelse:  # local\n    CONFIG[\"data_dir\"]  = Path(\"data/jaguar-re-id\")\n    CONFIG[\"work_dir\"]  = Path(\"outputs/exp-frozen-dinov2g\")\n\nCONFIG[\"embed_dir\"] = CONFIG[\"work_dir\"] / \"embeddings\"\nCONFIG[\"work_dir\"].mkdir(parents=True, exist_ok=True)\nCONFIG[\"embed_dir\"].mkdir(parents=True, exist_ok=True)\n\nfor k, v in CONFIG.items():\n    print(f\"  {k}: {v}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def find_dir(base: Path, split: str) -> Path:\n    \"\"\"Robustly locate the image folder for a split across common Kaggle layouts.\"\"\"\n    candidates = [base / f\"{split}/{split}\", base / split, base / \"images\" / split, base]\n    for c in candidates:\n        if c.is_dir():\n            # prefer a dir that actually contains image files\n            try:\n                if any(p.suffix.lower() in {\".jpg\", \".jpeg\", \".png\"} for p in c.iterdir()):\n                    return c\n            except Exception:\n                continue\n    # fall back to the first existing candidate\n    for c in candidates:\n        if c.is_dir():\n            return c\n    raise FileNotFoundError(f\"Could not locate '{split}' images under {base}\")\n\ndef pick_col(df, *names):\n    \"\"\"Return the first column present in df from the given candidate names.\"\"\"\n    for n in names:\n        if n in df.columns:\n            return n\n    raise KeyError(f\"None of {names} in columns {list(df.columns)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Load all training data (no single split)\n\nLoad the full training table and label-encode identities. Unlike the single-split champion, we keep\n**every** training image and let `StratifiedKFold` define the folds in §5.","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(CONFIG[\"data_dir\"] / \"train.csv\")\nfn_col  = pick_col(train_df, \"filename\", \"image\", \"img\", \"file\")\nlbl_col = pick_col(train_df, \"ground_truth\", \"identity\", \"label\", \"individual\")\ntrain_df = train_df.rename(columns={fn_col: \"filename\", lbl_col: \"ground_truth\"})\n\nle = LabelEncoder()\ntrain_df[\"label\"] = le.fit_transform(train_df[\"ground_truth\"])\nnum_classes = len(le.classes_)\n\ncounts = train_df[\"ground_truth\"].value_counts()\nprint(f\"images: {len(train_df)} | identities: {num_classes}\")\nprint(f\"per-identity: min {counts.min()} ({counts.idxmin()}), \"\n      f\"max {counts.max()} ({counts.idxmax()}), mean {counts.mean():.1f}\")\n\nTRAIN_IMG_DIR = find_dir(CONFIG[\"data_dir\"], \"train\")\nprint(\"train image dir:\", TRAIN_IMG_DIR)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Load the frozen backbone","metadata":{}},{"cell_type":"code","source":"def build_backbone(cfg, device):\n    name = cfg[\"backbone_model\"]\n    if cfg.get(\"offline_hf_cache\"):\n        os.environ[\"HF_HOME\"] = str(cfg[\"offline_hf_cache\"])\n        os.environ[\"HUGGINGFACE_HUB_CACHE\"] = str(cfg[\"offline_hf_cache\"])\n        os.environ[\"HF_HUB_OFFLINE\"] = \"1\"\n        os.environ[\"TRANSFORMERS_OFFLINE\"] = \"1\"\n    try:\n        backbone = timm.create_model(name, pretrained=True, num_classes=0)\n        print(\"Loaded pretrained weights via timm.\")\n    except Exception as e:\n        ckpt = cfg.get(\"offline_weights_file\")\n        if not ckpt:\n            raise RuntimeError(\n                \"No internet and no offline weights configured. Set CONFIG['offline_hf_cache'] \"\n                \"or CONFIG['offline_weights_file'] (see the §3 markdown).\"\n            ) from e\n        print(f\"Falling back to offline checkpoint: {ckpt}\")\n        backbone = timm.create_model(name, pretrained=False, num_classes=0)\n        sd = torch.load(ckpt, map_location=\"cpu\")\n        sd = sd.get(\"state_dict\", sd)\n        missing, unexpected = backbone.load_state_dict(sd, strict=False)\n        print(f\"  loaded (missing={len(missing)}, unexpected={len(unexpected)})\")\n    backbone.eval().to(device)\n    for p in backbone.parameters():\n        p.requires_grad_(False)\n    return backbone\n\nbackbone = build_backbone(CONFIG, device)\nn_params = sum(p.numel() for p in backbone.parameters())\nwith torch.no_grad():\n    dummy = torch.randn(1, 3, CONFIG[\"input_size\"], CONFIG[\"input_size\"]).to(device)\n    backbone_dim = backbone(dummy).shape[1]\nprint(f\"backbone params: {n_params/1e6:.0f}M | feature dim: {backbone_dim}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# DINOv2 expects ImageNet-normalised RGB at 518x518\npreprocess = transforms.Compose([\n    transforms.Resize((CONFIG[\"input_size\"], CONFIG[\"input_size\"])),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n])\n\n@torch.no_grad()\ndef extract_embeddings(model, image_paths, batch_size=16, tta_hflip=False, desc=\"extract\"):\n    model.eval()\n    use_amp = (device.type == \"cuda\")\n    out = []\n    for i in tqdm(range(0, len(image_paths), batch_size), desc=desc):\n        batch = []\n        for p in image_paths[i:i + batch_size]:\n            try:\n                batch.append(preprocess(Image.open(p).convert(\"RGB\")))\n            except Exception as e:\n                print(f\"  ! error loading {p}: {e}\")\n                batch.append(torch.zeros(3, CONFIG[\"input_size\"], CONFIG[\"input_size\"]))\n        x = torch.stack(batch).to(device)\n        with torch.autocast(device_type=(\"cuda\" if use_amp else \"cpu\"),\n                            dtype=torch.float16, enabled=use_amp):\n            feat = model(x).float()\n            if tta_hflip:\n                feat = feat + model(torch.flip(x, dims=[3])).float()\n                feat = feat / 2.0\n        out.append(feat.cpu().numpy())\n    return np.vstack(out)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def cached_embeddings(tag, filenames, img_dir, model):\n    \"\"\"Extract (or load cached) frozen embeddings for a list of filenames.\"\"\"\n    suffix = \"_tta\" if CONFIG[\"use_tta_hflip\"] else \"\"\n    cache = CONFIG[\"embed_dir\"] / f\"{tag}{suffix}.npz\"\n    filenames = [str(f) for f in filenames]\n    if cache.exists():\n        z = np.load(cache, allow_pickle=True)\n        if list(z[\"filenames\"]) == filenames:\n            print(f\"[cache] {cache.name}: {z['embeddings'].shape}\")\n            return z[\"embeddings\"]\n    paths = [img_dir / f for f in filenames]\n    emb = extract_embeddings(model, paths, CONFIG[\"extract_batch_size\"],\n                             CONFIG[\"use_tta_hflip\"], desc=tag)\n    np.savez_compressed(cache, embeddings=emb, filenames=np.array(filenames, dtype=object))\n    print(f\"[saved] {cache.name}: {emb.shape}\")\n    return emb\n\ntrain_emb = cached_embeddings(\"train\", train_df[\"filename\"].tolist(), TRAIN_IMG_DIR, backbone)\n#val_emb   = cached_embeddings(\"val\",   val_data[\"filename\"].tolist(),   TRAIN_IMG_DIR, backbone)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Extract frozen features for **all** training images (once)\n\nThis is the only backbone-scale computation. Every fold in §5 reuses these cached features — the\nwhole reason 5-fold is cheap here.","metadata":{}},{"cell_type":"code","source":"all_emb    = cached_embeddings(\"train_all\", train_df[\"filename\"].tolist(), TRAIN_IMG_DIR, backbone)\nall_labels = train_df[\"label\"].values\nprint(\"all train embeddings:\", all_emb.shape, \"| labels:\", all_labels.shape)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Projection head, batch-hard mining, and PK sampler","metadata":{}},{"cell_type":"code","source":"class EmbeddingProjection(nn.Module):\n    def __init__(self, input_dim, hidden_dim=512, output_dim=256, dropout=0.3):\n        super().__init__()\n        self.net = nn.Sequential(\n            nn.Linear(input_dim, hidden_dim), nn.BatchNorm1d(hidden_dim),\n            nn.ReLU(inplace=True), nn.Dropout(dropout),\n            nn.Linear(hidden_dim, output_dim), nn.BatchNorm1d(output_dim),\n        )\n        for m in self.modules():\n            if isinstance(m, nn.Linear):\n                nn.init.kaiming_normal_(m.weight, mode=\"fan_out\", nonlinearity=\"relu\")\n                if m.bias is not None:\n                    nn.init.constant_(m.bias, 0)\n            elif isinstance(m, nn.BatchNorm1d):\n                nn.init.constant_(m.weight, 1); nn.init.constant_(m.bias, 0)\n\n    def forward(self, x):\n        return self.net(x)\n\n\nclass TripletModel(nn.Module):\n    \"\"\"Projection head returning L2-normalised embeddings.\"\"\"\n    def __init__(self, input_dim, embedding_dim=256, hidden_dim=512, dropout=0.3):\n        super().__init__()\n        self.embedding_net = EmbeddingProjection(input_dim, hidden_dim, embedding_dim, dropout)\n\n    def forward(self, x):\n        return F.normalize(self.embedding_net(x), p=2, dim=1)\n\n    def get_embeddings(self, x):\n        return self.forward(x)\n\n\ndef mine_hard_triplets(emb, labels):\n    \"\"\"For each anchor: hardest positive (farthest same-id) + hardest negative (closest diff-id).\"\"\"\n    dist = torch.cdist(emb, emb, p=2)\n    labels = labels.unsqueeze(0)\n    same = (labels == labels.T).float() - torch.eye(emb.size(0), device=emb.device)\n    diff = 1.0 - (labels == labels.T).float()\n    a_idx, p_idx, n_idx = [], [], []\n    for i in range(emb.size(0)):\n        if same[i].sum() == 0 or diff[i].sum() == 0:\n            continue\n        pos = dist[i] * same[i] + (-1e9) * (1 - same[i])\n        neg = dist[i] * diff[i] + (1e9) * (1 - diff[i])\n        a_idx.append(i); p_idx.append(pos.argmax().item()); n_idx.append(neg.argmin().item())\n    dev = emb.device\n    return (torch.tensor(a_idx, device=dev),\n            torch.tensor(p_idx, device=dev),\n            torch.tensor(n_idx, device=dev))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class EmbeddingDataset(Dataset):\n    def __init__(self, embeddings, labels):\n        self.x = torch.as_tensor(embeddings, dtype=torch.float32)\n        self.y = torch.as_tensor(labels, dtype=torch.long)\n    def __len__(self):  return len(self.y)\n    def __getitem__(self, i):  return self.x[i], self.y[i]\n\n\nclass PKBatchSampler:\n    \"\"\"Each batch = P identities x K samples (with replacement if an identity has < K).\"\"\"\n    def __init__(self, labels, P, K, num_batches=None):\n        self.labels = np.asarray(labels); self.P, self.K = P, K\n        self.by_label = {}\n        for idx, l in enumerate(self.labels):\n            self.by_label.setdefault(l, []).append(idx)\n        self.valid = [l for l, ix in self.by_label.items() if len(ix) >= 2]\n        if len(self.valid) < P:\n            print(f\"  warning: only {len(self.valid)} identities have >=2 samples; P->{len(self.valid)}\")\n            self.P = len(self.valid)\n        self.num_batches = num_batches or max(1, len(self.labels) // (self.P * self.K))\n    def __iter__(self):\n        for _ in range(self.num_batches):\n            batch = []\n            for l in np.random.choice(self.valid, self.P, replace=False):\n                ix = self.by_label[l]\n                repl = len(ix) < self.K\n                batch += list(np.random.choice(ix, self.K, replace=repl))\n            yield batch\n    def __len__(self):  return self.num_batches\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@torch.no_grad()\ndef compute_val_map(model, embeddings, labels):\n    \"\"\"Identity-balanced mAP on the validation set (closed-set proxy for the LB metric).\"\"\"\n    model.eval()\n    e = model.get_embeddings(torch.as_tensor(embeddings, dtype=torch.float32).to(device)).cpu().numpy()\n    sim = cosine_similarity(e); np.fill_diagonal(sim, -1)\n    labels = np.asarray(labels)\n    per_id = {}\n    for q in range(len(labels)):\n        match = (labels == labels[q]).astype(int); match[q] = 0\n        if match.sum() == 0:\n            continue\n        order = np.argsort(-sim[q]); m = match[order]\n        prec = np.cumsum(m) / np.arange(1, len(m) + 1)\n        ap = (prec * m).sum() / m.sum()\n        per_id.setdefault(labels[q], []).append(ap)\n    return float(np.mean([np.mean(v) for v in per_id.values()]))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. Train the head on 5 stratified folds\n\n`StratifiedKFold` keeps each fold's identity mix representative. For every fold we build a fresh head\nand train it exactly as the champion did (triplet loss + batch-hard mining + Adagrad + `ReduceLROnPlateau`,\nearly stop on validation loss). The per-fold best checkpoints and histories are kept for the ensemble\nand the telemetry plot.","metadata":{}},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=CONFIG[\"n_folds\"], shuffle=True, random_state=SEED)\n\nfold_histories, fold_best, fold_best_epoch, fold_ckpts = [], [], [], []\n\n\ndef train_fold_head(fold, tr_idx, va_idx):\n    tr_emb, tr_lab = all_emb[tr_idx], all_labels[tr_idx]\n    va_emb, va_lab = all_emb[va_idx], all_labels[va_idx]\n\n    tr_ds = EmbeddingDataset(tr_emb, tr_lab)\n    tr_sampler = PKBatchSampler(tr_lab, CONFIG[\"P\"], CONFIG[\"K\"])\n    tr_loader = DataLoader(tr_ds, batch_sampler=tr_sampler, num_workers=0)\n    va_ds = EmbeddingDataset(va_emb, va_lab)\n    va_loader = DataLoader(va_ds, batch_size=CONFIG[\"batch_size\"], shuffle=False, num_workers=0)\n\n    model = TripletModel(backbone_dim, CONFIG[\"embedding_dim\"], CONFIG[\"hidden_dim\"], CONFIG[\"dropout\"]).to(device)\n    criterion = nn.TripletMarginLoss(margin=CONFIG[\"triplet_margin\"], p=2)\n    optimizer = torch.optim.Adagrad(model.parameters(), lr=CONFIG[\"learning_rate\"],\n                                    weight_decay=CONFIG[\"weight_decay\"])\n    scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode=\"min\", factor=0.5, patience=5)\n\n    def run_epoch(loader, train=True):\n        model.train() if train else model.eval()\n        total, nb = 0.0, 0\n        for x, y in loader:\n            x, y = x.to(device), y.to(device)\n            with torch.set_grad_enabled(train):\n                emb = model(x)\n                a, p, n = mine_hard_triplets(emb, y)\n                if len(a) == 0:\n                    continue\n                loss = criterion(emb[a], emb[p], emb[n])\n                if train:\n                    optimizer.zero_grad(); loss.backward(); optimizer.step()\n            total += loss.item(); nb += 1\n        return total / max(nb, 1)\n\n    best_val_loss, best_map, best_epoch, patience_ctr = float(\"inf\"), 0.0, 0, 0\n    ckpt_path = CONFIG[\"work_dir\"] / f\"triplet_best_fold{fold}.pth\"\n    history = {\"train_loss\": [], \"val_loss\": [], \"val_map\": []}\n\n    for epoch in range(1, CONFIG[\"num_epochs\"] + 1):\n        tr = run_epoch(tr_loader, train=True)\n        vl = run_epoch(va_loader, train=False)\n        vm = compute_val_map(model, va_emb, va_lab)\n        scheduler.step(vl)\n        history[\"train_loss\"].append(tr); history[\"val_loss\"].append(vl); history[\"val_map\"].append(vm)\n\n        improved = vl < best_val_loss\n        if improved:\n            best_val_loss, best_map, best_epoch, patience_ctr = vl, vm, epoch, 0\n            torch.save({\"epoch\": epoch, \"model_state_dict\": model.state_dict(),\n                        \"val_loss\": vl, \"val_map\": vm, \"fold\": fold, \"config\": CONFIG,\n                        \"label_encoder_classes\": le.classes_.tolist()}, ckpt_path)\n        else:\n            patience_ctr += 1\n\n        if epoch % 25 == 0 or improved or epoch == 1:\n            print(f\"  [fold {fold}] epoch {epoch:3d} | train {tr:.4f} | val {vl:.4f} | \"\n                  f\"val_mAP {vm:.4f}{'  <- best' if improved else ''}\")\n        if patience_ctr >= CONFIG[\"patience\"]:\n            print(f\"  [fold {fold}] early stopping at epoch {epoch}\")\n            break\n\n    print(f\"  [fold {fold}] BEST epoch {best_epoch} | val_loss {best_val_loss:.4f} | val_mAP {best_map:.4f}\")\n    return history, best_map, best_epoch, str(ckpt_path)\n\n\nfor fold, (tr_idx, va_idx) in enumerate(skf.split(all_emb, all_labels)):\n    print(f\"\\n===== Fold {fold} / {CONFIG['n_folds']} | train {len(tr_idx)} | val {len(va_idx)} =====\")\n    hist, bmap, bep, cp = train_fold_head(fold, tr_idx, va_idx)\n    fold_histories.append(hist); fold_best.append(bmap)\n    fold_best_epoch.append(bep); fold_ckpts.append(cp)\n\ncv_mean, cv_std = float(np.mean(fold_best)), float(np.std(fold_best))\nprint(f\"\\n5-fold CV identity-balanced mAP: {cv_mean:.4f} +/- {cv_std:.4f}\")\nprint(\"per-fold best val mAP:\", [round(x, 4) for x in fold_best])\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. Per-fold training telemetry","metadata":{}},{"cell_type":"code","source":"try:\n    import matplotlib.pyplot as plt\n    fig, ax = plt.subplots(1, 3, figsize=(16, 4.6))\n    for f, h in enumerate(fold_histories):\n        ep = range(1, len(h[\"train_loss\"]) + 1)\n        ax[0].plot(ep, h[\"val_loss\"], lw=1.4, label=f\"fold {f}\")\n        ax[1].plot(ep, h[\"val_map\"],  lw=1.4, label=f\"fold {f}\")\n    ax[0].set_title(\"(a) validation triplet loss per fold\"); ax[0].set_xlabel(\"epoch\"); ax[0].legend(fontsize=8)\n    ax[1].set_title(\"(b) validation mAP per fold\");         ax[1].set_xlabel(\"epoch\"); ax[1].legend(fontsize=8)\n    bars = ax[2].bar([str(i) for i in range(len(fold_best))], fold_best, color=\"#2a78d6\")\n    for i, v in enumerate(fold_best):\n        ax[2].text(i, v + 0.005, f\"{v:.3f}\", ha=\"center\", fontsize=9, fontweight=\"bold\")\n    ax[2].axhline(cv_mean, color=\"#0ca30c\", ls=\"--\", lw=1, label=f\"CV mean {cv_mean:.4f}\")\n    ax[2].set_ylim(0, 1.0); ax[2].set_title(\"(c) best val mAP per fold\"); ax[2].set_xlabel(\"fold\"); ax[2].legend(fontsize=8)\n    fig.suptitle(\"Frozen DINOv2-Giant + triplet head — 5-fold telemetry\", fontweight=\"bold\")\n    plt.tight_layout(rect=[0, 0, 1, 0.95])\n    fig.savefig(CONFIG[\"work_dir\"] / \"telemetry_frozen_5fold_panels.png\", dpi=150, bbox_inches=\"tight\")\n    plt.show()\n\n    # also export the raw per-fold history for offline comparison\n    rows = []\n    for f, h in enumerate(fold_histories):\n        for e in range(len(h[\"train_loss\"])):\n            rows.append({\"fold\": f, \"epoch\": e + 1, \"train_loss\": h[\"train_loss\"][e],\n                         \"val_loss\": h[\"val_loss\"][e], \"val_map\": h[\"val_map\"][e]})\n    pd.DataFrame(rows).to_csv(CONFIG[\"work_dir\"] / \"telemetry_frozen_5fold_per_epoch.csv\", index=False)\n    pd.DataFrame({\"fold\": range(len(fold_best)), \"best_map\": fold_best,\n                  \"best_epoch\": fold_best_epoch}).to_csv(CONFIG[\"work_dir\"] / \"cv_fold_results.csv\", index=False)\n    print(\"saved 5-fold telemetry panels + CSVs\")\nexcept Exception as e:\n    print(\"plot skipped:\", e)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 8. Generate the submission — 5-head ensemble\n\nExtract the frozen test features once, project them through each of the five trained heads, and\n**average the L2-normalised projected embeddings** into one master embedding per image (then\nre-normalise). Score each test pair by cosine similarity of these master embeddings.","metadata":{}},{"cell_type":"code","source":"test_pairs = pd.read_csv(CONFIG[\"data_dir\"] / \"test.csv\")\nq_col  = pick_col(test_pairs, \"query_image\", \"query\", \"image_1\", \"img1\")\ng_col  = pick_col(test_pairs, \"gallery_image\", \"gallery\", \"image_2\", \"img2\")\nid_col = pick_col(test_pairs, \"row_id\", \"id\", \"ID\")\nprint(f\"test pairs: {len(test_pairs)} | cols: query='{q_col}', gallery='{g_col}', id='{id_col}'\")\n\nTEST_IMG_DIR = find_dir(CONFIG[\"data_dir\"], \"test\")\ntest_images = sorted(set(test_pairs[q_col]) | set(test_pairs[g_col]))\nprint(f\"unique test images: {len(test_images)} | dir: {TEST_IMG_DIR}\")\n\ntest_emb = cached_embeddings(\"test\", test_images, TEST_IMG_DIR, backbone)\n\n# Average each fold-head's L2-normalised projected embedding -> master embedding\nproj_sum = np.zeros((len(test_images), CONFIG[\"embedding_dim\"]), dtype=np.float64)\nfor cp in fold_ckpts:\n    m = TripletModel(backbone_dim, CONFIG[\"embedding_dim\"], CONFIG[\"hidden_dim\"], CONFIG[\"dropout\"]).to(device)\n    ck = torch.load(cp, map_location=device, weights_only=False)\n    m.load_state_dict(ck[\"model_state_dict\"]); m.eval()\n    with torch.no_grad():\n        pe = m.get_embeddings(torch.as_tensor(test_emb, dtype=torch.float32).to(device)).cpu().numpy()\n    proj_sum += pe\n    print(f\"  added fold {ck.get('fold', '?')} head (best val_mAP {ck.get('val_map', float('nan')):.4f})\")\n\nmaster = proj_sum / len(fold_ckpts)\nmaster = master / np.clip(np.linalg.norm(master, axis=1, keepdims=True), 1e-12, None)  # re-normalise\nemb_of = {fn: master[i] for i, fn in enumerate(test_images)}\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Vectorised cosine similarity for every pair (master embeddings are L2-normalised)\nq = np.stack([emb_of[f] for f in test_pairs[q_col]])\ng = np.stack([emb_of[f] for f in test_pairs[g_col]])\nsim = np.clip((q * g).sum(axis=1), 0.0, 1.0)\n\nsubmission = pd.DataFrame({\"row_id\": test_pairs[id_col].values, \"similarity\": sim})\nout_path = CONFIG[\"work_dir\"] / \"submission.csv\"\nsubmission.to_csv(out_path, index=False)\n\ntry:\n    sample = pd.read_csv(CONFIG[\"data_dir\"] / \"sample_submission.csv\")\n    assert len(submission) == len(sample), f\"row count {len(submission)} != {len(sample)}\"\n    print(\"row count matches sample_submission:\", len(submission))\nexcept FileNotFoundError:\n    print(\"sample_submission.csv not found; skipping row-count check\")\nassert submission[\"row_id\"].is_unique, \"row_id not unique\"\nassert submission[\"similarity\"].between(0, 1).all(), \"similarity out of [0,1]\"\nprint(f\"saved {out_path} | rows {len(submission)} | 5-head ensemble | \"\n      f\"CV mAP {cv_mean:.4f} +/- {cv_std:.4f}\")\nsubmission.head()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 9. Notes\n\n- **Why this can beat the single split:** the reported score no longer depends on one lucky 80/20\n  split, and averaging five heads' embeddings smooths per-head noise (typically +1–2% mAP).\n- **Cost:** only the small head is retrained five times on cached features — a few minutes on a T4.\n- **Further levers (all reuse the same cached features):** DINOv2 last-few-layer feature fusion,\n  hflip TTA at extraction (`CONFIG[\"use_tta_hflip\"]=True`), query expansion / k-reciprocal\n  re-ranking on the master embeddings, or LoRA adapters if a little backbone adaptation is worth it.\n","metadata":{}}]}