{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.13"},"papermill":{"default_parameters":{},"duration":6068.954978,"end_time":"2026-06-28T04:48:44.571838+00:00","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-06-28T03:07:35.616860+00:00","version":"2.7.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"144a89d2d8c8482189099e7a318589ae":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"187d09202b51492086724dbec212024f":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_3720377cb18b4845b94c194c8210236a","IPY_MODEL_f5e2f5580a6241ef8d2ca39ccf353ef9","IPY_MODEL_d8b76cf3f79044d79afe0716b5f2fd53"],"layout":"IPY_MODEL_32b9e97f94da4e24abcbe6c34ceee281","tabbable":null,"tooltip":null}},"1b421fa7573049c2980aba285af519ca":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_4bd03b4131a047efa42aac1636c57d4e","placeholder":"​","style":"IPY_MODEL_75c80e8231d64d0792a80bc52c857be9","tabbable":null,"tooltip":null,"value":"Caching corrected images [process x4]: 100%"}},"26336ae0b0b449ea99aa6d5ee5721d87":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"32b9e97f94da4e24abcbe6c34ceee281":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"352d64d859b346a6aad04dfee3b5d6ee":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_3c2b01410aa94cc2b2c067d50be59d95","placeholder":"​","style":"IPY_MODEL_a143883bf417484981c873db1163e34d","tabbable":null,"tooltip":null,"value":"Auditing near duplicates: 100%"}},"3720377cb18b4845b94c194c8210236a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_53483ae6c9bf4b7da8c017e39da34491","placeholder":"​","style":"IPY_MODEL_c0d13574234548bba791b3ddced17685","tabbable":null,"tooltip":null,"value":"Caching corrected images [process x4]: 100%"}},"3bf72b5918a2439d9f9bf1cd75489b80":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"3c2b01410aa94cc2b2c067d50be59d95":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"3fcd22ff15ad4af4bbe4838c4777c3e8":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"4bd03b4131a047efa42aac1636c57d4e":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"53483ae6c9bf4b7da8c017e39da34491":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"6684d26d7308461c9ead0080cd3122ea":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_3bf72b5918a2439d9f9bf1cd75489b80","placeholder":"​","style":"IPY_MODEL_69fa5e6ea1ef4a78b2ca783f22e752d4","tabbable":null,"tooltip":null,"value":" 40490/40490 [22:59&lt;00:00, 41.61it/s]"}},"69fa5e6ea1ef4a78b2ca783f22e752d4":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"73965ac63d2843868bc57ced4030563e":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"75c80e8231d64d0792a80bc52c857be9":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"7ee570e8072147718bac874011c4574b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_bd56827a530f4f7ab7675d20b95fc507","max":40109,"min":0,"orientation":"horizontal","style":"IPY_MODEL_9de78681fbe5441fbd7f740a20b571e1","tabbable":null,"tooltip":null,"value":40109}},"8ce64b62c8704c82a42a9ce4ecedaf3c":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"9de78681fbe5441fbd7f740a20b571e1":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"a143883bf417484981c873db1163e34d":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"a5c4e2c622954079869fa5a8cf1980be":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_1b421fa7573049c2980aba285af519ca","IPY_MODEL_e8822a2e3aee44f9a32af7f85d6b0690","IPY_MODEL_6684d26d7308461c9ead0080cd3122ea"],"layout":"IPY_MODEL_8ce64b62c8704c82a42a9ce4ecedaf3c","tabbable":null,"tooltip":null}},"ae7e9eba67994b739b8538572a3c5f9a":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"bd56827a530f4f7ab7675d20b95fc507":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"c0d13574234548bba791b3ddced17685":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"c225cd9f981e433981723e02b3aebe02":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_d90333059e714b08a48befa06891d653","placeholder":"​","style":"IPY_MODEL_db175a0d957a4f78b7499c9ba795b7fc","tabbable":null,"tooltip":null,"value":" 40109/40109 [02:44&lt;00:00, 605.18it/s]"}},"d81e0798d1174881aa076a43164eb291":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"d8b76cf3f79044d79afe0716b5f2fd53":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_e2467fe81824453986a06b3c57549d17","placeholder":"​","style":"IPY_MODEL_26336ae0b0b449ea99aa6d5ee5721d87","tabbable":null,"tooltip":null,"value":" 22371/22371 [20:45&lt;00:00, 25.96it/s]"}},"d90333059e714b08a48befa06891d653":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"db175a0d957a4f78b7499c9ba795b7fc":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"e2467fe81824453986a06b3c57549d17":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"e4086f71bcc54e7a9a7865fef890d0f3":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_352d64d859b346a6aad04dfee3b5d6ee","IPY_MODEL_7ee570e8072147718bac874011c4574b","IPY_MODEL_c225cd9f981e433981723e02b3aebe02"],"layout":"IPY_MODEL_73965ac63d2843868bc57ced4030563e","tabbable":null,"tooltip":null}},"e8822a2e3aee44f9a32af7f85d6b0690":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_ae7e9eba67994b739b8538572a3c5f9a","max":40490,"min":0,"orientation":"horizontal","style":"IPY_MODEL_d81e0798d1174881aa076a43164eb291","tabbable":null,"tooltip":null,"value":40490}},"f5e2f5580a6241ef8d2ca39ccf353ef9":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_144a89d2d8c8482189099e7a318589ae","max":22371,"min":0,"orientation":"horizontal","style":"IPY_MODEL_3fcd22ff15ad4af4bbe4838c4777c3e8","tabbable":null,"tooltip":null,"value":22371}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"742ef461","cell_type":"markdown","source":"# EyePACS / Diabetic Retinopathy Detection update\n\nThis notebook is adapted from `best-f1-model.ipynb` and keeps the same InceptionV3 staged training pipeline, preprocessing, calibration search, optional 480px expert models, holdout evaluation, and reporting. The data source is changed to the Kaggle Diabetic Retinopathy Detection training set only.\n\nThe notebook extracts `trainLabels.csv.zip` and multipart `train.zip.001` through `train.zip.005` into `/kaggle/temp/diabetic_retinopathy_detection`, then splits the labeled training images internally with the same policy as the best-F1 notebook:\n- train: 80%\n- model_valid: 5%\n- calibration: 5%\n- holdout/internal test: 10%\n","metadata":{"papermill":{"duration":0.011145,"end_time":"2026-06-28T03:07:38.191585+00:00","exception":false,"start_time":"2026-06-28T03:07:38.180440+00:00","status":"completed"},"tags":[]}},{"id":"9c1021dc","cell_type":"markdown","source":"# InceptionV3 Multi-Class Diabetic Retinopathy Pipeline - EyePACS Train Only\n\nThis notebook keeps the same best-F1 model pipeline, but changes the input contract to the labeled Kaggle Diabetic Retinopathy Detection training data:\n\n- dataset path: `/kaggle/input/competitions/diabetic-retinopathy-detection`\n- train archives: `/kaggle/input/competitions/diabetic-retinopathy-detection/train.zip.001` through `train.zip.005`\n- labels archive: `/kaggle/input/competitions/diabetic-retinopathy-detection/trainLabels.csv.zip`\n- extraction root: `/kaggle/temp/diabetic_retinopathy_detection`\n\nThe labels file uses `image` and `level`. The generated data module below resolves extensionless image names such as `10_left` to extracted image files such as `10_left.jpeg`, builds the same audited cache, and creates the same train/model-valid/calibration/holdout split structure as the source notebook.\n","metadata":{"papermill":{"duration":0.009125,"end_time":"2026-06-28T03:07:38.210097+00:00","exception":false,"start_time":"2026-06-28T03:07:38.200972+00:00","status":"completed"},"tags":[]}},{"id":"2ed3326c","cell_type":"markdown","source":"# What changed\n\n- Switched the notebook config to the Kaggle Diabetic Retinopathy Detection train archives and labels archive.\n- Added an extract-if-missing step for `trainLabels.csv.zip` and multipart `train.zip.001` through `train.zip.005`.\n- Kept the builder flow unchanged: `prepare()` -> `train_main_stages()` -> `train_experts()` -> `calibrate()` -> `evaluate_holdout()` -> `export_summary()`.\n- Added extension-aware manifest image resolution so EyePACS label rows map to `.jpeg` files under the extracted train image directory.\n- Kept evaluation exactly in the old style: validation/calibration/holdout are split from the labeled training images, and `evaluate_holdout()` reports the internal test performance.\n","metadata":{"papermill":{"duration":0.009044,"end_time":"2026-06-28T03:07:38.228231+00:00","exception":false,"start_time":"2026-06-28T03:07:38.219187+00:00","status":"completed"},"tags":[]}},{"id":"ea8ef465","cell_type":"markdown","source":"## Configuration","metadata":{"papermill":{"duration":0.009902,"end_time":"2026-06-28T03:07:38.247251+00:00","exception":false,"start_time":"2026-06-28T03:07:38.237349+00:00","status":"completed"},"tags":[]}},{"id":"4bcb6bba","cell_type":"code","source":"from __future__ import annotations\n\nimport json\nimport os\nimport shutil\nimport subprocess\nimport sys\nimport zipfile\nfrom pathlib import Path\n\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport torch\nfrom IPython.display import display\nfrom sklearn.metrics import classification_report, cohen_kappa_score, confusion_matrix, f1_score, accuracy_score\n\nDATASET_DIR = \"/kaggle/input/competitions/diabetic-retinopathy-detection\"\nTRAIN_ARCHIVE_FIRST = f\"{DATASET_DIR}/train.zip.001\"\nTRAIN_ARCHIVE_PARTS = [f\"{DATASET_DIR}/train.zip.{part:03d}\" for part in range(1, 6)]\nLABELS_ZIP_PATH = f\"{DATASET_DIR}/trainLabels.csv.zip\"\nEXTRACT_ROOT = \"/kaggle/temp/diabetic_retinopathy_detection\"\nLABELS_CSV_PATH = f\"{EXTRACT_ROOT}/trainLabels.csv\"\nEXTRACTED_TRAIN_IMAGES_DIR = f\"{EXTRACT_ROOT}/train\"\n\nCFG = {\n    \"dataset_root\": DATASET_DIR,\n    \"data_root\": EXTRACT_ROOT,\n    \"cleaned_dataset_root\": EXTRACT_ROOT,\n    \"images_dir\": EXTRACTED_TRAIN_IMAGES_DIR,\n    \"labels_csv_path\": LABELS_CSV_PATH,\n    \"dataset_slug\": \"diabetic-retinopathy-detection\",\n    \"dataset_variant\": \"eyepacs_train_only\",\n    \"manifest_source_name\": \"eyepacs_dr_detection\",\n    \"experiment_name\": \"dr_inceptionv3_eyepacs_train_only_512\",\n    \"output_dir\": \"/kaggle/working/dr_inceptionv3_multiclass_eyepacs_512_20260629\",\n    \"cache_dir\": \"/kaggle/temp/dr_inceptionv3_eyepacs_cache_512_v1\",\n    \"preprocessing_profile\": \"full_current\",\n    \"use_experts\": True,\n    \"run_source_holdout_evals\": True,\n    \"make_visual_qa\": True,\n    \"preprocessing_qa_per_source_class\": 2,\n    \"bootstrap_iterations\": 1000,\n    \"debug_sample_size\": 0,\n    \"build_corrected_manifest\": False,\n    \"manifest_data_root\": None,\n    \"calibration_from_val_fraction\": 0.50,\n\n    # Same preprocessing geometry as the best-F1 model.\n    \"image_size\": 512,\n    \"cache_jpeg_quality\": 95,\n    \"cache_workers\": min(16, os.cpu_count() or 4),\n    \"cache_decode_min_side\": 768,\n    \"crop_preview_size\": 512,\n    \"preprocess_blend\": 0.40,\n    \"ben_graham_sigma\": None,\n    \"clahe_clip_limit\": 2.0,\n    \"clahe_tile_grid\": 8,\n    # Full CAR/LID cache creation exceeded Kaggle's 12-hour run limit.\n    # Keep LID settings available, but use fast OpenCV caches by default.\n    \"downscale_method\": \"opencv_area\",\n    \"lid_input_size\": 2048,\n    \"lid_scale\": 4,\n    \"lid_weight_path\": \"/kaggle/input/datasets/khangcancode/car-trained-model-from-github/models/4x/kgn.pth\",\n    \"car_repo_dir\": \"/kaggle/working/CAR\",\n    \"expert_image_size\": 512,\n    \"expert_cache_dir\": \"/kaggle/temp/dr_inceptionv3_eyepacs_cache_512_experts_v1\",\n    \"expert_downscale_method\": \"opencv_area\",\n    \"expert_lid_input_size\": 2048,\n    \"expert_cache_decode_min_side\": 1024,\n    \"expert_zero_mild_zero_ratio\": 2.0,\n    \"expert_batch_size\": 16,\n    \"expert_eval_batch_size\": 24,\n    \"severe_to_zero_guard_margin\": 0.005,\n\n    # InceptionV3 backbone. A later cell resolves this to the first available timm variant.\n    \"backbone\": \"inception_v3.tf_adv_in1k\",\n    \"pretrained\": True,\n    \"require_pretrained\": True,\n    \"num_classes\": 5,\n    \"batch_size\": 32,\n    \"eval_batch_size\": 48,\n    \"num_workers\": min(4, os.cpu_count() or 4),\n    \"prefetch_factor\": 2,\n    \"drop_rate\": 0.25,\n    \"grad_clip_norm\": 1.0,\n    \"grad_accum_steps\": 1,\n    \"log_every\": 25,\n    \"aug_rotation_degrees\": 0,\n    \"use_color_jitter\": False,\n    \"use_tta\": True,\n    \"tta_views\": [\"hflip\", \"vflip\", \"hvflip\"],\n    \"seed\": 42,\n    \"split_strategy\": \"stratified_group_10fold\",\n    \"accuracy_guard\": 0.01,\n    \"target_accuracy\": 0.90,\n    \"reset_holdout_evaluation\": False,\n}\n\nOUTPUT_DIR = Path(CFG[\"output_dir\"])\nOUTPUT_DIR.mkdir(parents=True, exist_ok=True)\nprint(json.dumps(CFG, indent=2))\nprint(\"PyTorch:\", torch.__version__)\nprint(\"GPU count:\", torch.cuda.device_count())\nprint(\"GPUs:\", [torch.cuda.get_device_name(i) for i in range(torch.cuda.device_count())])\n","metadata":{"execution":{"iopub.execute_input":"2026-06-28T03:07:38.267334Z","iopub.status.busy":"2026-06-28T03:07:38.267084Z","iopub.status.idle":"2026-06-28T03:07:45.733944Z","shell.execute_reply":"2026-06-28T03:07:45.733072Z"},"papermill":{"duration":7.479288,"end_time":"2026-06-28T03:07:45.735679+00:00","exception":false,"start_time":"2026-06-28T03:07:38.256391+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"6c53c657-fc71-4d39-ab48-99e3d38f021a","cell_type":"markdown","source":"## Extract EyePACS Train Images and Labels\n\nThis cell extracts the labeled training data only. It reuses existing extracted files on rerun, then points `CFG[\"labels_csv_path\"]` and `CFG[\"images_dir\"]` at the discovered paths before the normal builder pipeline runs.\n","metadata":{}},{"id":"d894f861-33a5-42cc-82ae-e81f63176c03","cell_type":"code","source":"EXTRACT_ROOT_PATH = Path(EXTRACT_ROOT)\nEXTRACT_ROOT_PATH.mkdir(parents=True, exist_ok=True)\n\nmissing_parts = [path for path in TRAIN_ARCHIVE_PARTS if not Path(path).is_file()]\nif missing_parts:\n    raise FileNotFoundError(\"Missing multipart train archive parts:\\n  \" + \"\\n  \".join(missing_parts))\nif not Path(LABELS_ZIP_PATH).is_file():\n    raise FileNotFoundError(f\"Missing labels archive: {LABELS_ZIP_PATH}\")\nif shutil.which(\"7z\") is None:\n    raise RuntimeError(\"7z is required to extract multipart train.zip.001, but it was not found on PATH.\")\n\nlabels_csv_path = Path(LABELS_CSV_PATH)\nif not labels_csv_path.is_file():\n    print(\"Extracting labels:\", LABELS_ZIP_PATH)\n    with zipfile.ZipFile(LABELS_ZIP_PATH) as labels_zip:\n        labels_zip.extractall(EXTRACT_ROOT_PATH)\n    if not labels_csv_path.is_file():\n        candidates = sorted(EXTRACT_ROOT_PATH.rglob(\"trainLabels.csv\"))\n        if not candidates:\n            raise FileNotFoundError(\"trainLabels.csv was not found after extracting trainLabels.csv.zip.\")\n        labels_csv_path = candidates[0]\nelse:\n    print(\"Using existing labels:\", labels_csv_path)\n\n\ndef discover_train_image_dir(root: Path) -> Path | None:\n    candidates = []\n    direct_train = root / \"train\"\n    if direct_train.is_dir():\n        candidates.append(direct_train)\n    candidates.extend(path for path in root.iterdir() if path.is_dir() and path not in candidates)\n    candidates.append(root)\n    best_dir = None\n    best_count = 0\n    for candidate in candidates:\n        count = sum(1 for _ in candidate.glob(\"*.jpeg\"))\n        if count > best_count:\n            best_dir = candidate\n            best_count = count\n    return best_dir if best_count > 0 else None\n\nimages_dir = discover_train_image_dir(EXTRACT_ROOT_PATH)\nif images_dir is None:\n    print(\"Extracting train images from multipart archive:\", TRAIN_ARCHIVE_FIRST)\n    subprocess.run(\n        [\"7z\", \"x\", TRAIN_ARCHIVE_FIRST, f\"-o{EXTRACT_ROOT}\", \"-y\"],\n        check=True,\n    )\n    images_dir = discover_train_image_dir(EXTRACT_ROOT_PATH)\n\nif images_dir is None:\n    raise FileNotFoundError(f\"No extracted .jpeg train images were found under {EXTRACT_ROOT}.\")\n\nimage_count = sum(1 for _ in images_dir.glob(\"*.jpeg\"))\nif image_count <= 0:\n    raise RuntimeError(f\"Discovered image directory has no .jpeg files: {images_dir}\")\n\nCFG[\"labels_csv_path\"] = str(labels_csv_path)\nCFG[\"images_dir\"] = str(images_dir)\nCFG[\"data_root\"] = str(EXTRACT_ROOT_PATH)\nCFG[\"cleaned_dataset_root\"] = str(EXTRACT_ROOT_PATH)\n\nlabels_preview = pd.read_csv(labels_csv_path, nrows=5)\nprint(\"Labels CSV:\", CFG[\"labels_csv_path\"])\nprint(\"Train images dir:\", CFG[\"images_dir\"])\nprint(\"Train image count:\", image_count)\ndisplay(labels_preview)\n","metadata":{},"outputs":[],"execution_count":null},{"id":"16621714","cell_type":"markdown","source":"## Resolve the InceptionV3 timm Backbone","metadata":{"papermill":{"duration":0.010175,"end_time":"2026-06-28T03:07:45.756844+00:00","exception":false,"start_time":"2026-06-28T03:07:45.746669+00:00","status":"completed"},"tags":[]}},{"id":"eb325cdb","cell_type":"code","source":"import timm\n\npreferred_backbones = [\n    \"inception_v3.tf_adv_in1k\",\n    \"inception_v3.tv_in1k\",\n    \"inception_v3\",\n]\navailable = set(timm.list_models(\"inception_v3*\"))\nprint(\"Available InceptionV3 variants:\", sorted(available))\nfor name in preferred_backbones:\n    if name in available:\n        CFG[\"backbone\"] = name\n        break\nelse:\n    raise RuntimeError(\"No timm InceptionV3 backbone is available in this environment.\")\nprint(\"Selected backbone:\", CFG[\"backbone\"])","metadata":{"execution":{"iopub.execute_input":"2026-06-28T03:07:45.778908Z","iopub.status.busy":"2026-06-28T03:07:45.778305Z","iopub.status.idle":"2026-06-28T03:07:55.660869Z","shell.execute_reply":"2026-06-28T03:07:55.659985Z"},"papermill":{"duration":9.894744,"end_time":"2026-06-28T03:07:55.662576+00:00","exception":false,"start_time":"2026-06-28T03:07:45.767832+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"912401ce","cell_type":"markdown","source":"## Prepare the Optional CAR/LID Downscaler","metadata":{"papermill":{"duration":0.009425,"end_time":"2026-06-28T03:07:55.681944+00:00","exception":false,"start_time":"2026-06-28T03:07:55.672519+00:00","status":"completed"},"tags":[]}},{"id":"5a461d3b","cell_type":"code","source":"CAR_SOURCE_FILES = {'modules.py': 'import functools\\nimport numpy as np\\n\\nimport torch\\nimport torch.nn as nn\\n\\n\\nLEAKY_FACTOR = 0.2\\nMULT_FACTOR = 1\\n\\n\\n# TEST PASSED\\nclass PixelUnShuffle(nn.Module):\\n    \"\"\"\\n    Inverse process of pytorch pixel shuffle module\\n    \"\"\"\\n    def __init__(self, down_scale):\\n        \"\"\"\\n        :param down_scale: int, down scale factor\\n        \"\"\"\\n        super(PixelUnShuffle, self).__init__()\\n\\n        if not isinstance(down_scale, int):\\n            raise ValueError(\\'Down scale factor must be a integer number\\')\\n        self.down_scale = down_scale\\n\\n    def forward(self, input):\\n        \"\"\"\\n        :param input: tensor of shape (batch size, channels, height, width)\\n        :return: tensor of shape(batch size, channels * down_scale * down_scale, height / down_scale, width / down_scale)\\n        \"\"\"\\n        b, c, h, w = input.size()\\n        assert h % self.down_scale == 0\\n        assert w % self.down_scale == 0\\n\\n        oc = c * self.down_scale ** 2\\n        oh = int(h / self.down_scale)\\n        ow = int(w / self.down_scale)\\n\\n        output_reshaped = input.reshape(b, c, oh, self.down_scale, ow, self.down_scale)\\n        output = output_reshaped.permute(0, 1, 3, 5, 2, 4).reshape(b, oc, oh, ow)\\n\\n        return output\\n\\n\\nclass DownsampleBlock(nn.Module):\\n    def __init__(self, scale, input_channels, output_channels, ksize=1):\\n        super(DownsampleBlock, self).__init__()\\n        self.downsample = nn.Sequential(\\n            PixelUnShuffle(scale),\\n            nn.Conv2d(input_channels * (scale ** 2), output_channels, kernel_size=ksize, stride=1, padding=ksize//2)\\n        )\\n\\n    def forward(self, input):\\n        return self.downsample(input)\\n\\n\\nclass UpsampleBlock(nn.Module):\\n    def __init__(self, scale, input_channels, output_channels, ksize=1):\\n        super(UpsampleBlock, self).__init__()\\n        self.upsample = nn.Sequential(\\n            nn.Conv2d(input_channels, output_channels * (scale ** 2), kernel_size=1, stride=1, padding=ksize//2),\\n            nn.PixelShuffle(scale)\\n        )\\n\\n    def forward(self, input):\\n        return self.upsample(input)\\n\\n\\nclass ResidualBlock(nn.Module):\\n    def __init__(self, input_channels, channels, ksize=3,\\n                 use_instance_norm=False, affine=False):\\n        super(ResidualBlock, self).__init__()\\n        self.channels = channels\\n        self.ksize = ksize\\n        padding = self.ksize // 2\\n        if use_instance_norm:\\n            self.transform = nn.Sequential(\\n                nn.ReflectionPad2d(padding),\\n                nn.Conv2d(input_channels, channels, kernel_size=self.ksize, stride=1),\\n                nn.InstanceNorm2d(channels, affine=affine),\\n                nn.LeakyReLU(0.2),\\n                nn.ReflectionPad2d(padding),\\n                nn.Conv2d(channels, channels, kernel_size=self.ksize, stride=1),\\n                nn.InstanceNorm2d(channels)\\n            )\\n        else:\\n            self.transform = nn.Sequential(\\n                nn.ReflectionPad2d(padding),\\n                nn.Conv2d(input_channels, channels, kernel_size=self.ksize, stride=1),\\n                nn.LeakyReLU(0.2),\\n                nn.ReflectionPad2d(padding),\\n                nn.Conv2d(channels, channels, kernel_size=self.ksize, stride=1),\\n            )\\n\\n    def forward(self, input):\\n        return input + self.transform(input) * MULT_FACTOR\\n\\n\\nclass NormalizeBySum(nn.Module):\\n    def forward(self, x):\\n        return x / torch.sum(x, dim=1, keepdim=True).clamp(min=1e-7)\\n\\n\\nclass MeanShift(nn.Conv2d):\\n    def __init__(self, rgb_range, rgb_mean=(0.4488, 0.4371, 0.4040), rgb_std=(1.0, 1.0, 1.0), sign=-1):\\n        super(MeanShift, self).__init__(3, 3, kernel_size=1)\\n        std = torch.Tensor(rgb_std)\\n        self.weight.data = torch.eye(3).view(3, 3, 1, 1) / std.view(3, 1, 1, 1)\\n        self.bias.data = sign * rgb_range * torch.Tensor(rgb_mean) / std\\n        for p in self.parameters():\\n            p.requires_grad = False\\n\\n\\nclass DSN(nn.Module):\\n    def __init__(self, k_size, input_channels=3, scale=4):\\n        super(DSN, self).__init__()\\n\\n        self.k_size = k_size\\n\\n        self.sub_mean = MeanShift(1)\\n\\n        self.ds_1 = nn.Sequential(\\n            nn.ReflectionPad2d(2),\\n            nn.Conv2d(input_channels, 64, 5),\\n            nn.LeakyReLU(LEAKY_FACTOR)\\n        )\\n\\n        self.ds_2 = DownsampleBlock(2, 64, 128, ksize=1)\\n        self.ds_4 = DownsampleBlock(2, 128, 128, ksize=1)\\n\\n        res_4 = list()\\n        for idx in range(5):\\n            res_4 += [ResidualBlock(128, 128)]\\n        self.res_4 = nn.Sequential(*res_4)\\n\\n        self.ds_8 = DownsampleBlock(2, 128, 256)\\n\\n        self.kernels_trunk = nn.Sequential(\\n            nn.ReflectionPad2d(1),\\n            nn.Conv2d(256, 256, 3),\\n            nn.ReLU(),\\n            nn.ReflectionPad2d(1),\\n            nn.Conv2d(256, 256, 3),\\n            nn.ReLU(),\\n            nn.ReflectionPad2d(1),\\n            nn.Conv2d(256, 256, 3),\\n            nn.ReLU(),\\n            UpsampleBlock(8 // scale, 256, 256, ksize=1),\\n            nn.ReflectionPad2d(1),\\n            nn.Conv2d(256, 256, 3),\\n            nn.ReLU()\\n        )\\n\\n        self.kernels_weight = nn.Sequential(\\n            nn.ReflectionPad2d(1),\\n            nn.Conv2d(256, 256, 3),\\n            nn.ReLU(),\\n            nn.ReflectionPad2d(1),\\n            nn.Conv2d(256, k_size ** 2, 3)\\n        )\\n\\n        self.offsets_trunk = nn.Sequential(\\n            nn.ReflectionPad2d(1),\\n            nn.Conv2d(256, 256, 3),\\n            nn.ReLU(),\\n            nn.ReflectionPad2d(1),\\n            nn.Conv2d(256, 256, 3),\\n            nn.ReLU(),\\n            nn.ReflectionPad2d(1),\\n            nn.Conv2d(256, 256, 3),\\n            nn.ReLU(),\\n            UpsampleBlock(8 // scale, 256, 256, ksize=1),\\n            nn.ReflectionPad2d(1),\\n            nn.Conv2d(256, 256, 3),\\n            nn.ReLU()\\n        )\\n\\n        self.offsets_h_generation = nn.Sequential(\\n            nn.ReflectionPad2d(1),\\n            nn.Conv2d(256, 256, 3),\\n            nn.ReLU(),\\n            nn.ReflectionPad2d(1),\\n            nn.Conv2d(256, k_size ** 2, 3),\\n            nn.Tanh()\\n        )\\n\\n        self.offsets_v_generation = nn.Sequential(\\n            nn.ReflectionPad2d(1),\\n            nn.Conv2d(256, 256, 3),\\n            nn.ReLU(),\\n            nn.ReflectionPad2d(1),\\n            nn.Conv2d(256, k_size ** 2, 3),\\n            nn.Tanh()\\n        )\\n\\n    def forward(self, x):\\n        x = self.sub_mean(x)\\n\\n        x = self.ds_1(x)\\n        x = self.ds_2(x)\\n        x = self.ds_4(x)\\n        x = x + self.res_4(x)\\n        x = self.ds_8(x)\\n\\n        kt = self.kernels_trunk(x)\\n        k_weight = torch.clamp(self.kernels_weight(kt), min=1e-6, max=1)\\n        kernels = k_weight / torch.sum(k_weight, dim=1, keepdim=True).clamp(min=1e-6)\\n\\n        ot = self.offsets_trunk(x)\\n        offsets_h = self.offsets_h_generation(ot)\\n        offsets_v = self.offsets_v_generation(ot)\\n\\n        return kernels, offsets_h, offsets_v\\n', 'utils.py': 'import numpy as np\\nimport torch\\nfrom scipy import signal\\nfrom PIL import Image\\n\\n\\ndef matlab_style_gauss2D(shape=(3, 3), sigma=0.5):\\n    \"\"\"\\n    2D gaussian mask - should give the same result as MATLAB\\'s fspecial(\\'gaussian\\',[shape],[sigma])\\n    Acknowledgement : https://stackoverflow.com/questions/17190649/how-to-obtain-a-gaussian-filter-in-python (Author@ali_m)\\n    \"\"\"\\n    m, n = [(ss - 1.) / 2. for ss in shape]\\n    y, x = np.ogrid[-m:m + 1, -n:n + 1]\\n    h = np.exp(-(x * x + y * y) / (2. * sigma * sigma))\\n    h[h < np.finfo(h.dtype).eps * h.max()] = 0\\n    sumh = h.sum()\\n    if sumh != 0:\\n        h /= sumh\\n    return h\\n\\n\\ndef calc_ssim(X, Y, sigma=1.5, K1=0.01, K2=0.03, R=255):\\n    \\'\\'\\'\\n    X : y channel (i.e., luminance) of transformed YCbCr space of X\\n    Y : y channel (i.e., luminance) of transformed YCbCr space of Y\\n    Please follow the setting of psnr_ssim.m in EDSR (Enhanced Deep Residual Networks for Single Image Super-Resolution CVPRW2017).\\n    Official Link : https://github.com/LimBee/NTIRE2017/tree/db34606c2844e89317aac8728a2de562ef1f8aba\\n    The authors of EDSR use MATLAB\\'s ssim as the evaluation tool,\\n    thus this function is the same as ssim.m in MATLAB with C(3) == C(2)/2.\\n    \\'\\'\\'\\n    gaussian_filter = matlab_style_gauss2D((11, 11), sigma)\\n\\n    X = X.astype(np.float64)\\n    Y = Y.astype(np.float64)\\n\\n    window = gaussian_filter\\n\\n    ux = signal.convolve2d(X, window, mode=\\'same\\', boundary=\\'symm\\')\\n    uy = signal.convolve2d(Y, window, mode=\\'same\\', boundary=\\'symm\\')\\n\\n    uxx = signal.convolve2d(X * X, window, mode=\\'same\\', boundary=\\'symm\\')\\n    uyy = signal.convolve2d(Y * Y, window, mode=\\'same\\', boundary=\\'symm\\')\\n    uxy = signal.convolve2d(X * Y, window, mode=\\'same\\', boundary=\\'symm\\')\\n\\n    vx = uxx - ux * ux\\n    vy = uyy - uy * uy\\n    vxy = uxy - ux * uy\\n\\n    C1 = (K1 * R) ** 2\\n    C2 = (K2 * R) ** 2\\n\\n    A1, A2, B1, B2 = ((2 * ux * uy + C1, 2 * vxy + C2, ux ** 2 + uy ** 2 + C1, vx + vy + C2))\\n    D = B1 * B2\\n    S = (A1 * A2) / D\\n    mssim = S.mean()\\n\\n    return mssim\\n\\n\\ndef cal_psnr(img_1, img_2, benchmark=False):\\n    assert img_1.shape[0] == img_2.shape[0] and img_1.shape[1] == img_2.shape[1]\\n    img_1 = np.float64(img_1)\\n    img_2 = np.float64(img_2)\\n\\n    diff = (img_1 - img_2) / 255.0\\n    if benchmark:\\n        gray_coeff = np.array([65.738, 129.057, 25.064]).reshape(1, 1, 3) / 255.0\\n        diff = diff * gray_coeff\\n        diff = diff[:, :, 0] + diff[:, :, 1] + diff[:, :, 2]\\n\\n    mse = np.mean(diff ** 2)\\n    psnr = -10.0 * np.log10(mse)\\n\\n    return psnr\\n\\n\\ndef load_img(img_file):\\n    img = Image.open(img_file).convert(\\'RGB\\')\\n    img = np.array(img)\\n    h, w, _ = img.shape\\n    img = img[:h // 8 * 8, :w // 8 * 8, :]\\n    img = np.array(img) / 255.\\n    img = img.transpose((2, 0, 1))\\n    img = torch.from_numpy(img).float().unsqueeze(0).cuda()\\n\\n    return img\\n', 'adaptive_gridsampler/adaptive_gridsampler_cuda.cpp': '#include <ATen/ATen.h>\\n#include <torch/extension.h>\\n\\n#include \"adaptive_gridsampler_kernel.cuh\"\\n\\nint adaptive_gridsampler_cuda_forward(at::Tensor& img, at::Tensor& kernels, at::Tensor& offsets_h, at::Tensor& offsets_v, int offset_unit, int padding, at::Tensor& output)\\n{\\n    adaptive_gridsampler_kernel_forward(img, kernels, offsets_h, offsets_v, offset_unit, padding, output);\\n    return 1;\\n}\\n\\nint adaptive_gridsampler_cuda_backward(at::Tensor& img, at::Tensor& kernels, at::Tensor& offsets_h, at::Tensor& offsets_v, int offset_unit, at::Tensor& gradOutput, int padding,\\nat::Tensor& gradInput_kernels, at::Tensor& gradInput_offsets_h, at::Tensor& gradInput_offsets_v)\\n{\\n    adaptive_gridsampler_kernel_backward(img, kernels, offsets_h, offsets_v, offset_unit, gradOutput, padding, gradInput_kernels, gradInput_offsets_h, gradInput_offsets_v);\\n    return 1;\\n}\\n\\nPYBIND11_MODULE(TORCH_EXTENSION_NAME, m)\\n{\\n    m.def(\"forward\", &adaptive_gridsampler_cuda_forward, \"adaptive gridsampler forward (CUDA)\");\\n    m.def(\"backward\", &adaptive_gridsampler_cuda_backward, \"adaptive gridsampler backward (CUDA)\");\\n}', 'adaptive_gridsampler/adaptive_gridsampler_kernel.cu': '#include <stdio.h>\\n#include <torch/extension.h>\\n\\n#include \"helper_cuda.h\"\\n\\n#define BLOCK_SIZE 256\\n\\ntemplate <typename scalar_t>\\n__global__ void kernel_adaptive_gridsampler_update_output(\\nconst torch::PackedTensorAccessor32<scalar_t, 4, torch::RestrictPtrTraits> img,\\nconst torch::PackedTensorAccessor32<scalar_t, 4, torch::RestrictPtrTraits> kernels,\\nconst torch::PackedTensorAccessor32<scalar_t, 4, torch::RestrictPtrTraits> offsets_h,\\nconst torch::PackedTensorAccessor32<scalar_t, 4, torch::RestrictPtrTraits> offsets_v,\\nconst int offset_unit,\\nconst int padding,\\ntorch::PackedTensorAccessor32<scalar_t, 4, torch::RestrictPtrTraits> output,\\nconst size_t n)\\n{\\n    auto global_idx = blockDim.x * blockIdx.x + threadIdx.x;\\n    if(global_idx >= n) return;\\n\\n    auto dim_b = output.size(0);\\n    auto dim_c = output.size(1);\\n    auto dim_h = output.size(2);\\n    auto dim_w = output.size(3);\\n\\n    auto idb = (global_idx / (dim_c * dim_h * dim_w)) % dim_b;\\n    auto idc = (global_idx / (dim_h * dim_w)) % dim_c;\\n    auto idy = (global_idx / dim_w) % dim_h;\\n    auto idx = global_idx % dim_w;\\n\\n    if(idx >= dim_w || idy >= dim_h)\\n        return;\\n\\n    int k_size = sqrt(float(kernels.size(1)));\\n    float w = float(img.size(3) - 2 * padding);\\n    float h = float(img.size(2) - 2 * padding);\\n\\n    scalar_t result = 0;\\n    for(int k_y = 0; k_y < k_size; ++k_y)\\n    {\\n        for(int k_x = 0; k_x < k_size; ++k_x)\\n        {\\n            scalar_t offset_h = offsets_h[idb][k_size * k_y + k_x][idy][idx] * offset_unit;\\n            scalar_t offset_v = offsets_v[idb][k_size * k_y + k_x][idy][idx] * offset_unit;\\n\\n            scalar_t p_x = static_cast<scalar_t>(idx + 0.5) / dim_w * w + k_x + offset_h - 0.5;\\n            scalar_t p_y = static_cast<scalar_t>(idy + 0.5) / dim_h * h + k_y + offset_v - 0.5;\\n            scalar_t alpha = p_x - floor(p_x);\\n            scalar_t beta = p_y - floor(p_y);\\n\\n            int xL = max(min(int(floor(p_x)), int(w + 2 * padding - 1)), 0);\\n            int xR = max(min(xL + 1, int(w + 2 * padding - 1)), 0);\\n            int yT = max(min(int(floor(p_y)), int(h + 2 * padding - 1)), 0);\\n            int yB = max(min(yT + 1, int(h + 2 * padding - 1)), 0);\\n\\n            scalar_t val = 0;\\n            val += (1 - alpha) * (1 - beta) * img[idb][idc][yT][xL];\\n            val += alpha * (1 - beta) * img[idb][idc][yT][xR];\\n            val += (1 - alpha) * beta * img[idb][idc][yB][xL];\\n            val += alpha * beta * img[idb][idc][yB][xR];\\n\\n            result += val * kernels[idb][k_size * k_y + k_x][idy][idx];\\n        }\\n    }\\n    output[idb][idc][idy][idx] = result;\\n}\\n\\nvoid adaptive_gridsampler_kernel_forward(const torch::Tensor& img, const torch::Tensor& kernels, const torch::Tensor& offsets_h, const torch::Tensor& offsets_v, const int offset_unit, const int padding, torch::Tensor& output)\\n{\\n    kernel_adaptive_gridsampler_update_output<float><<<(output.numel() + BLOCK_SIZE - 1) / BLOCK_SIZE, BLOCK_SIZE>>>(\\n    img.packed_accessor32<float, 4, torch::RestrictPtrTraits>(), kernels.packed_accessor32<float, 4, torch::RestrictPtrTraits>(),\\n    offsets_h.packed_accessor32<float, 4, torch::RestrictPtrTraits>(), offsets_v.packed_accessor32<float, 4, torch::RestrictPtrTraits>(), offset_unit, padding,\\n    output.packed_accessor32<float, 4, torch::RestrictPtrTraits>(), output.numel());\\n\\n    checkCudaErrors(cudaGetLastError());\\n}\\n\\ntemplate <typename scalar_t>\\n__global__ void kernel_adaptive_gridsampler_backward(const torch::PackedTensorAccessor32<scalar_t, 4, torch::RestrictPtrTraits> img,\\nconst torch::PackedTensorAccessor32<scalar_t, 4, torch::RestrictPtrTraits> kernels,\\nconst torch::PackedTensorAccessor32<scalar_t, 4, torch::RestrictPtrTraits> offsets_h,\\nconst torch::PackedTensorAccessor32<scalar_t, 4, torch::RestrictPtrTraits> offsets_v,\\nconst int offset_unit,\\nconst torch::PackedTensorAccessor32<scalar_t, 4, torch::RestrictPtrTraits> gradOutput,\\nconst int padding,\\ntorch::PackedTensorAccessor32<scalar_t, 4, torch::RestrictPtrTraits> gradInput_kernels,\\ntorch::PackedTensorAccessor32<scalar_t, 4, torch::RestrictPtrTraits> gradInput_offsets_h,\\ntorch::PackedTensorAccessor32<scalar_t, 4, torch::RestrictPtrTraits> gradInput_offsets_v,\\nconst size_t n)\\n{\\n    auto global_idx = blockDim.x * blockIdx.x + threadIdx.x;\\n    if(global_idx >= n) return;\\n\\n    auto dim_b = gradInput_kernels.size(0);\\n    auto dim_c = gradInput_kernels.size(1);\\n    auto dim_h = gradInput_kernels.size(2);\\n    auto dim_w = gradInput_kernels.size(3);\\n\\n    auto idb = (global_idx / (dim_c * dim_h * dim_w)) % dim_b;\\n    auto idc = (global_idx / (dim_h * dim_w)) % dim_c;\\n    auto idy = (global_idx / dim_w) % dim_h;\\n    auto idx = global_idx % dim_w;\\n\\n    if(idx >= dim_w || idx >= dim_h)\\n        return;\\n\\n    int k_size = sqrt(float(dim_c));\\n    int k_y = idc / k_size;\\n    int k_x = idc % k_size;\\n\\n    scalar_t offset_h = offsets_h[idb][idc][idy][idx] * offset_unit;\\n    scalar_t offset_v = offsets_v[idb][idc][idy][idx] * offset_unit;\\n\\n    float w = float(img.size(3) - 2 * padding);\\n    float h = float(img.size(2) - 2 * padding);\\n\\n    scalar_t p_x = static_cast<scalar_t>(idx + 0.5) / dim_w * w + k_x + offset_h - 0.5;\\n    scalar_t p_y = static_cast<scalar_t>(idy + 0.5) / dim_h * h + k_y + offset_v - 0.5;\\n    scalar_t alpha = p_x - floor(p_x);\\n    scalar_t beta = p_y - floor(p_y);\\n\\n    int xL = max(min(int(floor(p_x)), int(w + 2 * padding - 1)), 0);\\n    int xR = max(min(xL + 1, int(w + 2 * padding - 1)), 0);\\n    int yT = max(min(int(floor(p_y)), int(h + 2 * padding - 1)), 0);\\n    int yB = max(min(yT + 1, int(h + 2 * padding - 1)), 0);\\n\\n    scalar_t grad_kernels = 0;\\n    scalar_t grad_offset_h = 0;\\n    scalar_t grad_offset_v = 0;\\n    for(int c = 0; c < img.size(1); ++c)\\n    {\\n        scalar_t c_tl = img[idb][c][yT][xL];\\n        scalar_t c_tr = img[idb][c][yT][xR];\\n        scalar_t c_bl = img[idb][c][yB][xL];\\n        scalar_t c_br = img[idb][c][yB][xR];\\n\\n        scalar_t grad = 0;\\n        grad += (1 - alpha) * (1 - beta) * c_tl;\\n        grad += alpha * (1 - beta) * c_tr;\\n        grad += (1 - alpha) * beta * c_bl;\\n        grad += alpha * beta * c_br;\\n        grad_kernels += grad * gradOutput[idb][c][idy][idx];\\n\\n        grad = (beta - 1) * c_tl + (1 - beta) * c_tr - beta * c_bl + beta * c_br;\\n        grad_offset_h += kernels[idb][idc][idy][idx] * grad * gradOutput[idb][c][idy][idx] * offset_unit;\\n\\n        grad = (alpha - 1) * c_tl - alpha * c_tr + (1 - alpha) * c_bl + alpha * c_br;\\n        grad_offset_v += kernels[idb][idc][idy][idx] * grad * gradOutput[idb][c][idy][idx] * offset_unit;\\n    }\\n\\n    gradInput_kernels[idb][idc][idy][idx] = grad_kernels;\\n\\n    gradInput_offsets_h[idb][idc][idy][idx] =  grad_offset_h;\\n    gradInput_offsets_v[idb][idc][idy][idx] = grad_offset_v;\\n}\\n\\nvoid adaptive_gridsampler_kernel_backward(const torch::Tensor& img, const torch::Tensor& kernels, const torch::Tensor& offsets_h, const torch::Tensor& offsets_v, const int offset_unit, const torch::Tensor& gradOutput, const int padding,\\ntorch::Tensor& gradInput_kernels, torch::Tensor& gradInput_offsets_h, torch::Tensor& gradInput_offsets_v)\\n{\\n    kernel_adaptive_gridsampler_backward<float><<<(gradInput_kernels.numel() + BLOCK_SIZE - 1) / BLOCK_SIZE, BLOCK_SIZE, 0>>>(\\n    img.packed_accessor32<float, 4, torch::RestrictPtrTraits>(), kernels.packed_accessor32<float, 4, torch::RestrictPtrTraits>(),\\n    offsets_h.packed_accessor32<float, 4, torch::RestrictPtrTraits>(), offsets_v.packed_accessor32<float, 4, torch::RestrictPtrTraits>(),\\n    offset_unit,\\n    gradOutput.packed_accessor32<float, 4, torch::RestrictPtrTraits>(),\\n    padding,\\n    gradInput_kernels.packed_accessor32<float, 4, torch::RestrictPtrTraits>(),\\n    gradInput_offsets_h.packed_accessor32<float, 4, torch::RestrictPtrTraits>(), gradInput_offsets_v.packed_accessor32<float, 4, torch::RestrictPtrTraits>(),\\n    gradInput_kernels.numel());\\n\\n    checkCudaErrors(cudaGetLastError());\\n}', 'adaptive_gridsampler/adaptive_gridsampler_kernel.cuh': '#ifndef ADAPTIVE_GRIDSAMPLER_KERNEL_CUH\\n#define ADAPTIVE_GRIDSAMPLER_KERNEL_CUH\\n\\n#include <torch/extension.h>\\n\\nvoid adaptive_gridsampler_kernel_forward(const torch::Tensor& img, const torch::Tensor& kernels, const torch::Tensor& offsets_h, const torch::Tensor& offsets_v, const int offset_unit, const int padding, torch::Tensor& output);\\nvoid adaptive_gridsampler_kernel_backward(const torch::Tensor& img, const torch::Tensor& kernels, const torch::Tensor& offsets_h, const torch::Tensor& offsets_v, const int offset_unit, const torch::Tensor& gradOutput, const int padding, torch::Tensor& gradInput_kernels, torch::Tensor& gradInput_offsets_h, torch::Tensor& gradInput_offsets_v);\\n\\n#endif', 'adaptive_gridsampler/gridsampler.py': 'import torch\\nimport torch.nn as nn\\nimport torch.nn.functional as F\\nfrom torch.autograd import Function, gradcheck\\n\\nfrom .adaptive_gridsampler_cuda import forward\\n\\n\\nclass GridSamplerFunction(Function):\\n    @staticmethod\\n    def forward(ctx, img, kernels, offsets_h, offsets_v, offset_unit, padding, downscale_factor):\\n        assert isinstance(downscale_factor, int)\\n        assert isinstance(padding, int)\\n\\n        ctx.padding = padding\\n        ctx.offset_unit = offset_unit\\n\\n        b, c, h, w = img.size()\\n        assert h // downscale_factor == kernels.size(2)\\n        assert w // downscale_factor == kernels.size(3)\\n\\n        img = nn.ReflectionPad2d(padding)(img)\\n        # ctx.save_for_backward(img, kernels, offsets_h, offsets_v)\\n\\n        output = img.new(b, c, h // downscale_factor, w // downscale_factor).zero_()\\n        forward(img, kernels, offsets_h, offsets_v, offset_unit, padding, output)\\n\\n        return output\\n\\n    @staticmethod\\n    def backward(ctx, grad_output):\\n        raise NotImplementedError\\n\\n\\nclass Downsampler(nn.Module):\\n    def __init__(self, ds, k_size):\\n        super(Downsampler, self).__init__()\\n        self.ds = ds\\n        self.k_size = k_size\\n\\n    def forward(self, img, kernels, offsets_h, offsets_v, offset_unit):\\n        assert self.k_size ** 2 == kernels.size(1)\\n        return GridSamplerFunction.apply(img, kernels, offsets_h, offsets_v, offset_unit, self.k_size // 2, self.ds)\\n', 'adaptive_gridsampler/helper_cuda.h': '/**\\n * Copyright 1993-2012 NVIDIA Corporation.  All rights reserved.\\n *\\n * Please refer to the NVIDIA end user license agreement (EULA) associated\\n * with this source code for terms and conditions that govern your use of\\n * this software. Any use, reproduction, disclosure, or distribution of\\n * this software and related documentation outside the terms of the EULA\\n * is strictly prohibited.\\n *\\n */\\n\\n////////////////////////////////////////////////////////////////////////////////\\n// These are CUDA Helper functions for initialization and error checking\\n\\n#ifndef HELPER_CUDA_H\\n#define HELPER_CUDA_H\\n\\n#pragma once\\n\\n#include <stdlib.h>\\n#include <stdio.h>\\n#include <string.h>\\n\\n#include \"helper_string.h\"\\n\\n//#include <string>\\n//#include <iostream>\\n//#include <sstream>\\n\\n// Note, it is required that your SDK sample to include the proper header files, please\\n// refer the CUDA examples for examples of the needed CUDA headers, which may change depending\\n// on which CUDA functions are used.\\n\\n// CUDA Runtime error messages\\n#ifdef __DRIVER_TYPES_H__\\nstatic const char *_cudaGetErrorEnum(cudaError_t error)\\n{\\n    switch (error)\\n    {\\n        case cudaSuccess:\\n            return \"cudaSuccess\";\\n\\n        case cudaErrorMissingConfiguration:\\n            return \"cudaErrorMissingConfiguration\";\\n\\n        case cudaErrorMemoryAllocation:\\n            return \"cudaErrorMemoryAllocation\";\\n\\n        case cudaErrorInitializationError:\\n            return \"cudaErrorInitializationError\";\\n\\n        case cudaErrorLaunchFailure:\\n            return \"cudaErrorLaunchFailure\";\\n\\n        case cudaErrorPriorLaunchFailure:\\n            return \"cudaErrorPriorLaunchFailure\";\\n\\n        case cudaErrorLaunchTimeout:\\n            return \"cudaErrorLaunchTimeout\";\\n\\n        case cudaErrorLaunchOutOfResources:\\n            return \"cudaErrorLaunchOutOfResources\";\\n\\n        case cudaErrorInvalidDeviceFunction:\\n            return \"cudaErrorInvalidDeviceFunction\";\\n\\n        case cudaErrorInvalidConfiguration:\\n            return \"cudaErrorInvalidConfiguration\";\\n\\n        case cudaErrorInvalidDevice:\\n            return \"cudaErrorInvalidDevice\";\\n\\n        case cudaErrorInvalidValue:\\n            return \"cudaErrorInvalidValue\";\\n\\n        case cudaErrorInvalidPitchValue:\\n            return \"cudaErrorInvalidPitchValue\";\\n\\n        case cudaErrorInvalidSymbol:\\n            return \"cudaErrorInvalidSymbol\";\\n\\n        case cudaErrorMapBufferObjectFailed:\\n            return \"cudaErrorMapBufferObjectFailed\";\\n\\n        case cudaErrorUnmapBufferObjectFailed:\\n            return \"cudaErrorUnmapBufferObjectFailed\";\\n\\n        case cudaErrorInvalidHostPointer:\\n            return \"cudaErrorInvalidHostPointer\";\\n\\n        case cudaErrorInvalidDevicePointer:\\n            return \"cudaErrorInvalidDevicePointer\";\\n\\n        case cudaErrorInvalidTexture:\\n            return \"cudaErrorInvalidTexture\";\\n\\n        case cudaErrorInvalidTextureBinding:\\n            return \"cudaErrorInvalidTextureBinding\";\\n\\n        case cudaErrorInvalidChannelDescriptor:\\n            return \"cudaErrorInvalidChannelDescriptor\";\\n\\n        case cudaErrorInvalidMemcpyDirection:\\n            return \"cudaErrorInvalidMemcpyDirection\";\\n\\n        case cudaErrorAddressOfConstant:\\n            return \"cudaErrorAddressOfConstant\";\\n\\n        case cudaErrorTextureFetchFailed:\\n            return \"cudaErrorTextureFetchFailed\";\\n\\n        case cudaErrorTextureNotBound:\\n            return \"cudaErrorTextureNotBound\";\\n\\n        case cudaErrorSynchronizationError:\\n            return \"cudaErrorSynchronizationError\";\\n\\n        case cudaErrorInvalidFilterSetting:\\n            return \"cudaErrorInvalidFilterSetting\";\\n\\n        case cudaErrorInvalidNormSetting:\\n            return \"cudaErrorInvalidNormSetting\";\\n\\n        case cudaErrorMixedDeviceExecution:\\n            return \"cudaErrorMixedDeviceExecution\";\\n\\n        case cudaErrorCudartUnloading:\\n            return \"cudaErrorCudartUnloading\";\\n\\n        case cudaErrorUnknown:\\n            return \"cudaErrorUnknown\";\\n\\n        case cudaErrorNotYetImplemented:\\n            return \"cudaErrorNotYetImplemented\";\\n\\n        case cudaErrorMemoryValueTooLarge:\\n            return \"cudaErrorMemoryValueTooLarge\";\\n\\n        case cudaErrorInvalidResourceHandle:\\n            return \"cudaErrorInvalidResourceHandle\";\\n\\n        case cudaErrorNotReady:\\n            return \"cudaErrorNotReady\";\\n\\n        case cudaErrorInsufficientDriver:\\n            return \"cudaErrorInsufficientDriver\";\\n\\n        case cudaErrorSetOnActiveProcess:\\n            return \"cudaErrorSetOnActiveProcess\";\\n\\n        case cudaErrorInvalidSurface:\\n            return \"cudaErrorInvalidSurface\";\\n\\n        case cudaErrorNoDevice:\\n            return \"cudaErrorNoDevice\";\\n\\n        case cudaErrorECCUncorrectable:\\n            return \"cudaErrorECCUncorrectable\";\\n\\n        case cudaErrorSharedObjectSymbolNotFound:\\n            return \"cudaErrorSharedObjectSymbolNotFound\";\\n\\n        case cudaErrorSharedObjectInitFailed:\\n            return \"cudaErrorSharedObjectInitFailed\";\\n\\n        case cudaErrorUnsupportedLimit:\\n            return \"cudaErrorUnsupportedLimit\";\\n\\n        case cudaErrorDuplicateVariableName:\\n            return \"cudaErrorDuplicateVariableName\";\\n\\n        case cudaErrorDuplicateTextureName:\\n            return \"cudaErrorDuplicateTextureName\";\\n\\n        case cudaErrorDuplicateSurfaceName:\\n            return \"cudaErrorDuplicateSurfaceName\";\\n\\n        case cudaErrorDevicesUnavailable:\\n            return \"cudaErrorDevicesUnavailable\";\\n\\n        case cudaErrorInvalidKernelImage:\\n            return \"cudaErrorInvalidKernelImage\";\\n\\n        case cudaErrorNoKernelImageForDevice:\\n            return \"cudaErrorNoKernelImageForDevice\";\\n\\n        case cudaErrorIncompatibleDriverContext:\\n            return \"cudaErrorIncompatibleDriverContext\";\\n\\n        case cudaErrorPeerAccessAlreadyEnabled:\\n            return \"cudaErrorPeerAccessAlreadyEnabled\";\\n\\n        case cudaErrorPeerAccessNotEnabled:\\n            return \"cudaErrorPeerAccessNotEnabled\";\\n\\n        case cudaErrorDeviceAlreadyInUse:\\n            return \"cudaErrorDeviceAlreadyInUse\";\\n\\n        case cudaErrorProfilerDisabled:\\n            return \"cudaErrorProfilerDisabled\";\\n\\n        case cudaErrorProfilerNotInitialized:\\n            return \"cudaErrorProfilerNotInitialized\";\\n\\n        case cudaErrorProfilerAlreadyStarted:\\n            return \"cudaErrorProfilerAlreadyStarted\";\\n\\n        case cudaErrorProfilerAlreadyStopped:\\n            return \"cudaErrorProfilerAlreadyStopped\";\\n\\n#if __CUDA_API_VERSION >= 0x4000\\n\\n        case cudaErrorAssert:\\n            return \"cudaErrorAssert\";\\n\\n        case cudaErrorTooManyPeers:\\n            return \"cudaErrorTooManyPeers\";\\n\\n        case cudaErrorHostMemoryAlreadyRegistered:\\n            return \"cudaErrorHostMemoryAlreadyRegistered\";\\n\\n        case cudaErrorHostMemoryNotRegistered:\\n            return \"cudaErrorHostMemoryNotRegistered\";\\n#endif\\n\\n        case cudaErrorStartupFailure:\\n            return \"cudaErrorStartupFailure\";\\n\\n        case cudaErrorApiFailureBase:\\n            return \"cudaErrorApiFailureBase\";\\n    }\\n\\n    return \"<unknown>\";\\n}\\n#endif\\n\\n#ifdef __cuda_cuda_h__\\n// CUDA Driver API errors\\nstatic const char *_cudaGetErrorEnum(CUresult error)\\n{\\n    switch (error)\\n    {\\n        case CUDA_SUCCESS:\\n            return \"CUDA_SUCCESS\";\\n\\n        case CUDA_ERROR_INVALID_VALUE:\\n            return \"CUDA_ERROR_INVALID_VALUE\";\\n\\n        case CUDA_ERROR_OUT_OF_MEMORY:\\n            return \"CUDA_ERROR_OUT_OF_MEMORY\";\\n\\n        case CUDA_ERROR_NOT_INITIALIZED:\\n            return \"CUDA_ERROR_NOT_INITIALIZED\";\\n\\n        case CUDA_ERROR_DEINITIALIZED:\\n            return \"CUDA_ERROR_DEINITIALIZED\";\\n\\n        case CUDA_ERROR_PROFILER_DISABLED:\\n            return \"CUDA_ERROR_PROFILER_DISABLED\";\\n\\n        case CUDA_ERROR_PROFILER_NOT_INITIALIZED:\\n            return \"CUDA_ERROR_PROFILER_NOT_INITIALIZED\";\\n\\n        case CUDA_ERROR_PROFILER_ALREADY_STARTED:\\n            return \"CUDA_ERROR_PROFILER_ALREADY_STARTED\";\\n\\n        case CUDA_ERROR_PROFILER_ALREADY_STOPPED:\\n            return \"CUDA_ERROR_PROFILER_ALREADY_STOPPED\";\\n\\n        case CUDA_ERROR_NO_DEVICE:\\n            return \"CUDA_ERROR_NO_DEVICE\";\\n\\n        case CUDA_ERROR_INVALID_DEVICE:\\n            return \"CUDA_ERROR_INVALID_DEVICE\";\\n\\n        case CUDA_ERROR_INVALID_IMAGE:\\n            return \"CUDA_ERROR_INVALID_IMAGE\";\\n\\n        case CUDA_ERROR_INVALID_CONTEXT:\\n            return \"CUDA_ERROR_INVALID_CONTEXT\";\\n\\n        case CUDA_ERROR_CONTEXT_ALREADY_CURRENT:\\n            return \"CUDA_ERROR_CONTEXT_ALREADY_CURRENT\";\\n\\n        case CUDA_ERROR_MAP_FAILED:\\n            return \"CUDA_ERROR_MAP_FAILED\";\\n\\n        case CUDA_ERROR_UNMAP_FAILED:\\n            return \"CUDA_ERROR_UNMAP_FAILED\";\\n\\n        case CUDA_ERROR_ARRAY_IS_MAPPED:\\n            return \"CUDA_ERROR_ARRAY_IS_MAPPED\";\\n\\n        case CUDA_ERROR_ALREADY_MAPPED:\\n            return \"CUDA_ERROR_ALREADY_MAPPED\";\\n\\n        case CUDA_ERROR_NO_BINARY_FOR_GPU:\\n            return \"CUDA_ERROR_NO_BINARY_FOR_GPU\";\\n\\n        case CUDA_ERROR_ALREADY_ACQUIRED:\\n            return \"CUDA_ERROR_ALREADY_ACQUIRED\";\\n\\n        case CUDA_ERROR_NOT_MAPPED:\\n            return \"CUDA_ERROR_NOT_MAPPED\";\\n\\n        case CUDA_ERROR_NOT_MAPPED_AS_ARRAY:\\n            return \"CUDA_ERROR_NOT_MAPPED_AS_ARRAY\";\\n\\n        case CUDA_ERROR_NOT_MAPPED_AS_POINTER:\\n            return \"CUDA_ERROR_NOT_MAPPED_AS_POINTER\";\\n\\n        case CUDA_ERROR_ECC_UNCORRECTABLE:\\n            return \"CUDA_ERROR_ECC_UNCORRECTABLE\";\\n\\n        case CUDA_ERROR_UNSUPPORTED_LIMIT:\\n            return \"CUDA_ERROR_UNSUPPORTED_LIMIT\";\\n\\n        case CUDA_ERROR_CONTEXT_ALREADY_IN_USE:\\n            return \"CUDA_ERROR_CONTEXT_ALREADY_IN_USE\";\\n\\n        case CUDA_ERROR_INVALID_SOURCE:\\n            return \"CUDA_ERROR_INVALID_SOURCE\";\\n\\n        case CUDA_ERROR_FILE_NOT_FOUND:\\n            return \"CUDA_ERROR_FILE_NOT_FOUND\";\\n\\n        case CUDA_ERROR_SHARED_OBJECT_SYMBOL_NOT_FOUND:\\n            return \"CUDA_ERROR_SHARED_OBJECT_SYMBOL_NOT_FOUND\";\\n\\n        case CUDA_ERROR_SHARED_OBJECT_INIT_FAILED:\\n            return \"CUDA_ERROR_SHARED_OBJECT_INIT_FAILED\";\\n\\n        case CUDA_ERROR_OPERATING_SYSTEM:\\n            return \"CUDA_ERROR_OPERATING_SYSTEM\";\\n\\n        case CUDA_ERROR_INVALID_HANDLE:\\n            return \"CUDA_ERROR_INVALID_HANDLE\";\\n\\n        case CUDA_ERROR_NOT_FOUND:\\n            return \"CUDA_ERROR_NOT_FOUND\";\\n\\n        case CUDA_ERROR_NOT_READY:\\n            return \"CUDA_ERROR_NOT_READY\";\\n\\n        case CUDA_ERROR_LAUNCH_FAILED:\\n            return \"CUDA_ERROR_LAUNCH_FAILED\";\\n\\n        case CUDA_ERROR_LAUNCH_OUT_OF_RESOURCES:\\n            return \"CUDA_ERROR_LAUNCH_OUT_OF_RESOURCES\";\\n\\n        case CUDA_ERROR_LAUNCH_TIMEOUT:\\n            return \"CUDA_ERROR_LAUNCH_TIMEOUT\";\\n\\n        case CUDA_ERROR_LAUNCH_INCOMPATIBLE_TEXTURING:\\n            return \"CUDA_ERROR_LAUNCH_INCOMPATIBLE_TEXTURING\";\\n\\n        case CUDA_ERROR_PEER_ACCESS_ALREADY_ENABLED:\\n            return \"CUDA_ERROR_PEER_ACCESS_ALREADY_ENABLED\";\\n\\n        case CUDA_ERROR_PEER_ACCESS_NOT_ENABLED:\\n            return \"CUDA_ERROR_PEER_ACCESS_NOT_ENABLED\";\\n\\n        case CUDA_ERROR_PRIMARY_CONTEXT_ACTIVE:\\n            return \"CUDA_ERROR_PRIMARY_CONTEXT_ACTIVE\";\\n\\n        case CUDA_ERROR_CONTEXT_IS_DESTROYED:\\n            return \"CUDA_ERROR_CONTEXT_IS_DESTROYED\";\\n\\n        case CUDA_ERROR_ASSERT:\\n            return \"CUDA_ERROR_ASSERT\";\\n\\n        case CUDA_ERROR_TOO_MANY_PEERS:\\n            return \"CUDA_ERROR_TOO_MANY_PEERS\";\\n\\n        case CUDA_ERROR_HOST_MEMORY_ALREADY_REGISTERED:\\n            return \"CUDA_ERROR_HOST_MEMORY_ALREADY_REGISTERED\";\\n\\n        case CUDA_ERROR_HOST_MEMORY_NOT_REGISTERED:\\n            return \"CUDA_ERROR_HOST_MEMORY_NOT_REGISTERED\";\\n\\n        case CUDA_ERROR_UNKNOWN:\\n            return \"CUDA_ERROR_UNKNOWN\";\\n    }\\n\\n    return \"<unknown>\";\\n}\\n#endif\\n\\n#ifdef CUBLAS_API_H_\\n// cuBLAS API errors\\nstatic const char *_cudaGetErrorEnum(cublasStatus_t error)\\n{\\n    switch (error)\\n    {\\n        case CUBLAS_STATUS_SUCCESS:\\n            return \"CUBLAS_STATUS_SUCCESS\";\\n\\n        case CUBLAS_STATUS_NOT_INITIALIZED:\\n            return \"CUBLAS_STATUS_NOT_INITIALIZED\";\\n\\n        case CUBLAS_STATUS_ALLOC_FAILED:\\n            return \"CUBLAS_STATUS_ALLOC_FAILED\";\\n\\n        case CUBLAS_STATUS_INVALID_VALUE:\\n            return \"CUBLAS_STATUS_INVALID_VALUE\";\\n\\n        case CUBLAS_STATUS_ARCH_MISMATCH:\\n            return \"CUBLAS_STATUS_ARCH_MISMATCH\";\\n\\n        case CUBLAS_STATUS_MAPPING_ERROR:\\n            return \"CUBLAS_STATUS_MAPPING_ERROR\";\\n\\n        case CUBLAS_STATUS_EXECUTION_FAILED:\\n            return \"CUBLAS_STATUS_EXECUTION_FAILED\";\\n\\n        case CUBLAS_STATUS_INTERNAL_ERROR:\\n            return \"CUBLAS_STATUS_INTERNAL_ERROR\";\\n    }\\n\\n    return \"<unknown>\";\\n}\\n#endif\\n\\n#ifdef _CUFFT_H_\\n// cuFFT API errors\\nstatic const char *_cudaGetErrorEnum(cufftResult error)\\n{\\n    switch (error)\\n    {\\n        case CUFFT_SUCCESS:\\n            return \"CUFFT_SUCCESS\";\\n\\n        case CUFFT_INVALID_PLAN:\\n            return \"CUFFT_INVALID_PLAN\";\\n\\n        case CUFFT_ALLOC_FAILED:\\n            return \"CUFFT_ALLOC_FAILED\";\\n\\n        case CUFFT_INVALID_TYPE:\\n            return \"CUFFT_INVALID_TYPE\";\\n\\n        case CUFFT_INVALID_VALUE:\\n            return \"CUFFT_INVALID_VALUE\";\\n\\n        case CUFFT_INTERNAL_ERROR:\\n            return \"CUFFT_INTERNAL_ERROR\";\\n\\n        case CUFFT_EXEC_FAILED:\\n            return \"CUFFT_EXEC_FAILED\";\\n\\n        case CUFFT_SETUP_FAILED:\\n            return \"CUFFT_SETUP_FAILED\";\\n\\n        case CUFFT_INVALID_SIZE:\\n            return \"CUFFT_INVALID_SIZE\";\\n\\n        case CUFFT_UNALIGNED_DATA:\\n            return \"CUFFT_UNALIGNED_DATA\";\\n    }\\n\\n    return \"<unknown>\";\\n}\\n#endif\\n\\n\\n#ifdef CUSPARSEAPI\\n// cuSPARSE API errors\\nstatic const char *_cudaGetErrorEnum(cusparseStatus_t error)\\n{\\n    switch (error)\\n    {\\n        case CUSPARSE_STATUS_SUCCESS:\\n            return \"CUSPARSE_STATUS_SUCCESS\";\\n\\n        case CUSPARSE_STATUS_NOT_INITIALIZED:\\n            return \"CUSPARSE_STATUS_NOT_INITIALIZED\";\\n\\n        case CUSPARSE_STATUS_ALLOC_FAILED:\\n            return \"CUSPARSE_STATUS_ALLOC_FAILED\";\\n\\n        case CUSPARSE_STATUS_INVALID_VALUE:\\n            return \"CUSPARSE_STATUS_INVALID_VALUE\";\\n\\n        case CUSPARSE_STATUS_ARCH_MISMATCH:\\n            return \"CUSPARSE_STATUS_ARCH_MISMATCH\";\\n\\n        case CUSPARSE_STATUS_MAPPING_ERROR:\\n            return \"CUSPARSE_STATUS_MAPPING_ERROR\";\\n\\n        case CUSPARSE_STATUS_EXECUTION_FAILED:\\n            return \"CUSPARSE_STATUS_EXECUTION_FAILED\";\\n\\n        case CUSPARSE_STATUS_INTERNAL_ERROR:\\n            return \"CUSPARSE_STATUS_INTERNAL_ERROR\";\\n\\n        case CUSPARSE_STATUS_MATRIX_TYPE_NOT_SUPPORTED:\\n            return \"CUSPARSE_STATUS_MATRIX_TYPE_NOT_SUPPORTED\";\\n    }\\n\\n    return \"<unknown>\";\\n}\\n#endif\\n\\n#ifdef CURAND_H_\\n// cuRAND API errors\\nstatic const char *_cudaGetErrorEnum(curandStatus_t error)\\n{\\n    switch (error)\\n    {\\n        case CURAND_STATUS_SUCCESS:\\n            return \"CURAND_STATUS_SUCCESS\";\\n\\n        case CURAND_STATUS_VERSION_MISMATCH:\\n            return \"CURAND_STATUS_VERSION_MISMATCH\";\\n\\n        case CURAND_STATUS_NOT_INITIALIZED:\\n            return \"CURAND_STATUS_NOT_INITIALIZED\";\\n\\n        case CURAND_STATUS_ALLOCATION_FAILED:\\n            return \"CURAND_STATUS_ALLOCATION_FAILED\";\\n\\n        case CURAND_STATUS_TYPE_ERROR:\\n            return \"CURAND_STATUS_TYPE_ERROR\";\\n\\n        case CURAND_STATUS_OUT_OF_RANGE:\\n            return \"CURAND_STATUS_OUT_OF_RANGE\";\\n\\n        case CURAND_STATUS_LENGTH_NOT_MULTIPLE:\\n            return \"CURAND_STATUS_LENGTH_NOT_MULTIPLE\";\\n\\n        case CURAND_STATUS_DOUBLE_PRECISION_REQUIRED:\\n            return \"CURAND_STATUS_DOUBLE_PRECISION_REQUIRED\";\\n\\n        case CURAND_STATUS_LAUNCH_FAILURE:\\n            return \"CURAND_STATUS_LAUNCH_FAILURE\";\\n\\n        case CURAND_STATUS_PREEXISTING_FAILURE:\\n            return \"CURAND_STATUS_PREEXISTING_FAILURE\";\\n\\n        case CURAND_STATUS_INITIALIZATION_FAILED:\\n            return \"CURAND_STATUS_INITIALIZATION_FAILED\";\\n\\n        case CURAND_STATUS_ARCH_MISMATCH:\\n            return \"CURAND_STATUS_ARCH_MISMATCH\";\\n\\n        case CURAND_STATUS_INTERNAL_ERROR:\\n            return \"CURAND_STATUS_INTERNAL_ERROR\";\\n    }\\n\\n    return \"<unknown>\";\\n}\\n#endif\\n\\n#ifdef NV_NPPIDEFS_H\\n// NPP API errors\\nstatic const char *_cudaGetErrorEnum(NppStatus error)\\n{\\n    switch (error)\\n    {\\n        case NPP_NOT_SUPPORTED_MODE_ERROR:\\n            return \"NPP_NOT_SUPPORTED_MODE_ERROR\";\\n\\n        case NPP_ROUND_MODE_NOT_SUPPORTED_ERROR:\\n            return \"NPP_ROUND_MODE_NOT_SUPPORTED_ERROR\";\\n\\n        case NPP_RESIZE_NO_OPERATION_ERROR:\\n            return \"NPP_RESIZE_NO_OPERATION_ERROR\";\\n\\n        case NPP_NOT_SUFFICIENT_COMPUTE_CAPABILITY:\\n            return \"NPP_NOT_SUFFICIENT_COMPUTE_CAPABILITY\";\\n\\n        case NPP_BAD_ARG_ERROR:\\n            return \"NPP_BAD_ARG_ERROR\";\\n\\n        case NPP_LUT_NUMBER_OF_LEVELS_ERROR:\\n            return \"NPP_LUT_NUMBER_OF_LEVELS_ERROR\";\\n\\n        case NPP_TEXTURE_BIND_ERROR:\\n            return \"NPP_TEXTURE_BIND_ERROR\";\\n\\n        case NPP_COEFF_ERROR:\\n            return \"NPP_COEFF_ERROR\";\\n\\n        case NPP_RECT_ERROR:\\n            return \"NPP_RECT_ERROR\";\\n\\n        case NPP_QUAD_ERROR:\\n            return \"NPP_QUAD_ERROR\";\\n\\n        case NPP_WRONG_INTERSECTION_ROI_ERROR:\\n            return \"NPP_WRONG_INTERSECTION_ROI_ERROR\";\\n\\n        case NPP_NOT_EVEN_STEP_ERROR:\\n            return \"NPP_NOT_EVEN_STEP_ERROR\";\\n\\n        case NPP_INTERPOLATION_ERROR:\\n            return \"NPP_INTERPOLATION_ERROR\";\\n\\n        case NPP_RESIZE_FACTOR_ERROR:\\n            return \"NPP_RESIZE_FACTOR_ERROR\";\\n\\n        case NPP_HAAR_CLASSIFIER_PIXEL_MATCH_ERROR:\\n            return \"NPP_HAAR_CLASSIFIER_PIXEL_MATCH_ERROR\";\\n\\n        case NPP_MEMFREE_ERR:\\n            return \"NPP_MEMFREE_ERR\";\\n\\n        case NPP_MEMSET_ERR:\\n            return \"NPP_MEMSET_ERR\";\\n\\n        case NPP_MEMCPY_ERROR:\\n            return \"NPP_MEMCPY_ERROR\";\\n\\n        case NPP_MEM_ALLOC_ERR:\\n            return \"NPP_MEM_ALLOC_ERR\";\\n\\n        case NPP_HISTO_NUMBER_OF_LEVELS_ERROR:\\n            return \"NPP_HISTO_NUMBER_OF_LEVELS_ERROR\";\\n\\n        case NPP_MIRROR_FLIP_ERR:\\n            return \"NPP_MIRROR_FLIP_ERR\";\\n\\n        case NPP_INVALID_INPUT:\\n            return \"NPP_INVALID_INPUT\";\\n\\n        case NPP_ALIGNMENT_ERROR:\\n            return \"NPP_ALIGNMENT_ERROR\";\\n\\n        case NPP_STEP_ERROR:\\n            return \"NPP_STEP_ERROR\";\\n\\n        case NPP_SIZE_ERROR:\\n            return \"NPP_SIZE_ERROR\";\\n\\n        case NPP_POINTER_ERROR:\\n            return \"NPP_POINTER_ERROR\";\\n\\n        case NPP_NULL_POINTER_ERROR:\\n            return \"NPP_NULL_POINTER_ERROR\";\\n\\n        case NPP_CUDA_KERNEL_EXECUTION_ERROR:\\n            return \"NPP_CUDA_KERNEL_EXECUTION_ERROR\";\\n\\n        case NPP_NOT_IMPLEMENTED_ERROR:\\n            return \"NPP_NOT_IMPLEMENTED_ERROR\";\\n\\n        case NPP_ERROR:\\n            return \"NPP_ERROR\";\\n\\n        case NPP_SUCCESS:\\n            return \"NPP_SUCCESS\";\\n\\n        case NPP_WARNING:\\n            return \"NPP_WARNING\";\\n\\n        case NPP_WRONG_INTERSECTION_QUAD_WARNING:\\n            return \"NPP_WRONG_INTERSECTION_QUAD_WARNING\";\\n\\n        case NPP_MISALIGNED_DST_ROI_WARNING:\\n            return \"NPP_MISALIGNED_DST_ROI_WARNING\";\\n\\n        case NPP_AFFINE_QUAD_INCORRECT_WARNING:\\n            return \"NPP_AFFINE_QUAD_INCORRECT_WARNING\";\\n\\n        case NPP_DOUBLE_SIZE_WARNING:\\n            return \"NPP_DOUBLE_SIZE_WARNING\";\\n\\n        case NPP_ODD_ROI_WARNING:\\n            return \"NPP_ODD_ROI_WARNING\";\\n\\n        case NPP_WRONG_INTERSECTION_ROI_WARNING:\\n            return \"NPP_WRONG_INTERSECTION_ROI_WARNING\";\\n    }\\n\\n    return \"<unknown>\";\\n}\\n#endif\\n\\ntemplate< typename T >\\nbool check(T result, char const *const func, const char *const file, int const line)\\n{\\n    if (result)\\n    {\\n        fprintf(stderr, \"CUDA error at %s:%d code=%d(%s) \\\\\"%s\\\\\" \\\\n\",\\n                file, line, static_cast<unsigned int>(result), _cudaGetErrorEnum(result), func);\\n        /*\\n                std::stringstream ss;\\n                std::string msg(\"CUDA error at \");\\n                msg += file;\\n                msg += \":\";\\n                ss << line;\\n                msg += ss.str();\\n                msg += \" code=\";\\n                ss << static_cast<unsigned int>(result);\\n                msg += ss.str();\\n                msg += \" (\";\\n                msg += _cudaGetErrorEnum(result);\\n                msg += \") \\\\\"\";\\n                msg += func;\\n                msg += \"\\\\\"\";\\n                //throw msg;\\n                std::cerr  << msg <<\"\\\\n\";\\n        */\\n        return true;\\n    }\\n    else\\n    {\\n        return false;\\n    }\\n}\\n\\n#ifdef __DRIVER_TYPES_H__\\n// This will output the proper CUDA error strings in the event that a CUDA host call returns an error\\n#define checkCudaErrors(val)           check ( (val), #val, __FILE__, __LINE__ )\\n\\n// This will output the proper error string when calling cudaGetLastError\\n#define getLastCudaError(msg)      __getLastCudaError (msg, __FILE__, __LINE__)\\n\\ninline void __getLastCudaError(const char *errorMessage, const char *file, const int line)\\n{\\n    cudaError_t err = cudaGetLastError();\\n\\n    if (cudaSuccess != err)\\n    {\\n        fprintf(stderr, \"%s(%i) : getLastCudaError() CUDA error : %s : (%d) %s.\\\\n\",\\n                file, line, errorMessage, (int)err, cudaGetErrorString(err));\\n        exit(EXIT_FAILURE);\\n    }\\n}\\n#endif\\n\\n#ifndef MAX\\n#define MAX(a,b) (a > b ? a : b)\\n#endif\\n\\n// Beginning of GPU Architecture definitions\\ninline int _ConvertSMVer2Cores(int major, int minor)\\n{\\n    // Defines for GPU Architecture types (using the SM version to determine the # of cores per SM\\n    typedef struct\\n    {\\n        int SM; // 0xMm (hexidecimal notation), M = SM Major version, and m = SM minor version\\n        int Cores;\\n    } sSMtoCores;\\n\\n    sSMtoCores nGpuArchCoresPerSM[] =\\n    {\\n        { 0x10,  8 }, // Tesla Generation (SM 1.0) G80 class\\n        { 0x11,  8 }, // Tesla Generation (SM 1.1) G8x class\\n        { 0x12,  8 }, // Tesla Generation (SM 1.2) G9x class\\n        { 0x13,  8 }, // Tesla Generation (SM 1.3) GT200 class\\n        { 0x20, 32 }, // Fermi Generation (SM 2.0) GF100 class\\n        { 0x21, 48 }, // Fermi Generation (SM 2.1) GF10x class\\n        { 0x30, 192}, // Kepler Generation (SM 3.0) GK10x class\\n        { 0x35, 192}, // Kepler Generation (SM 3.5) GK11x class\\n        {   -1, -1 }\\n    };\\n\\n    int index = 0;\\n\\n    while (nGpuArchCoresPerSM[index].SM != -1)\\n    {\\n        if (nGpuArchCoresPerSM[index].SM == ((major << 4) + minor))\\n        {\\n            return nGpuArchCoresPerSM[index].Cores;\\n        }\\n\\n        index++;\\n    }\\n\\n    // If we don\\'t find the values, we default use the previous one to run properly\\n    printf(\"MapSMtoCores for SM %d.%d is undefined.  Default to use %d Cores/SM\\\\n\", major, minor, nGpuArchCoresPerSM[7].Cores);\\n    return nGpuArchCoresPerSM[7].Cores;\\n}\\n// end of GPU Architecture definitions\\n\\n#ifdef __CUDA_RUNTIME_H__\\n// General GPU Device CUDA Initialization\\ninline int gpuDeviceInit(int devID)\\n{\\n    int deviceCount;\\n    checkCudaErrors(cudaGetDeviceCount(&deviceCount));\\n\\n    if (deviceCount == 0)\\n    {\\n        fprintf(stderr, \"gpuDeviceInit() CUDA error: no devices supporting CUDA.\\\\n\");\\n        exit(EXIT_FAILURE);\\n    }\\n\\n    if (devID < 0)\\n    {\\n        devID = 0;\\n    }\\n\\n    if (devID > deviceCount-1)\\n    {\\n        fprintf(stderr, \"\\\\n\");\\n        fprintf(stderr, \">> %d CUDA capable GPU device(s) detected. <<\\\\n\", deviceCount);\\n        fprintf(stderr, \">> gpuDeviceInit (-device=%d) is not a valid GPU device. <<\\\\n\", devID);\\n        fprintf(stderr, \"\\\\n\");\\n        return -devID;\\n    }\\n\\n    cudaDeviceProp deviceProp;\\n    checkCudaErrors(cudaGetDeviceProperties(&deviceProp, devID));\\n\\n    if (deviceProp.computeMode == cudaComputeModeProhibited)\\n    {\\n        fprintf(stderr, \"Error: device is running in <Compute Mode Prohibited>, no threads can use ::cudaSetDevice().\\\\n\");\\n        return -1;\\n    }\\n\\n    if (deviceProp.major < 1)\\n    {\\n        fprintf(stderr, \"gpuDeviceInit(): GPU device does not support CUDA.\\\\n\");\\n        exit(EXIT_FAILURE);\\n    }\\n\\n    checkCudaErrors(cudaSetDevice(devID));\\n    printf(\"gpuDeviceInit() CUDA Device [%d]: \\\\\"%s\\\\n\", devID, deviceProp.name);\\n\\n    return devID;\\n}\\n\\n// This function returns the best GPU (with maximum GFLOPS)\\ninline int gpuGetMaxGflopsDeviceId()\\n{\\n    int current_device     = 0, sm_per_multiproc  = 0;\\n    int max_compute_perf   = 0, max_perf_device   = 0;\\n    int device_count       = 0, best_SM_arch      = 0;\\n    cudaDeviceProp deviceProp;\\n    cudaGetDeviceCount(&device_count);\\n\\n    // Find the best major SM Architecture GPU device\\n    while (current_device < device_count)\\n    {\\n        cudaGetDeviceProperties(&deviceProp, current_device);\\n\\n        // If this GPU is not running on Compute Mode prohibited, then we can add it to the list\\n        if (deviceProp.computeMode != cudaComputeModeProhibited)\\n        {\\n            if (deviceProp.major > 0 && deviceProp.major < 9999)\\n            {\\n                best_SM_arch = MAX(best_SM_arch, deviceProp.major);\\n            }\\n        }\\n\\n        current_device++;\\n    }\\n\\n    // Find the best CUDA capable GPU device\\n    current_device = 0;\\n\\n    while (current_device < device_count)\\n    {\\n        cudaGetDeviceProperties(&deviceProp, current_device);\\n\\n        // If this GPU is not running on Compute Mode prohibited, then we can add it to the list\\n        if (deviceProp.computeMode != cudaComputeModeProhibited)\\n        {\\n            if (deviceProp.major == 9999 && deviceProp.minor == 9999)\\n            {\\n                sm_per_multiproc = 1;\\n            }\\n            else\\n            {\\n                sm_per_multiproc = _ConvertSMVer2Cores(deviceProp.major, deviceProp.minor);\\n            }\\n\\n            int compute_perf  = deviceProp.multiProcessorCount * sm_per_multiproc * deviceProp.clockRate;\\n\\n            if (compute_perf  > max_compute_perf)\\n            {\\n                // If we find GPU with SM major > 2, search only these\\n                if (best_SM_arch > 2)\\n                {\\n                    // If our device==dest_SM_arch, choose this, or else pass\\n                    if (deviceProp.major == best_SM_arch)\\n                    {\\n                        max_compute_perf  = compute_perf;\\n                        max_perf_device   = current_device;\\n                    }\\n                }\\n                else\\n                {\\n                    max_compute_perf  = compute_perf;\\n                    max_perf_device   = current_device;\\n                }\\n            }\\n        }\\n\\n        ++current_device;\\n    }\\n\\n    return max_perf_device;\\n}\\n\\n\\n// Initialization code to find the best CUDA Device\\ninline int findCudaDevice(int argc, const char **argv)\\n{\\n    cudaDeviceProp deviceProp;\\n    int devID = 0;\\n\\n    // If the command-line has a device number specified, use it\\n    if (checkCmdLineFlag(argc, argv, \"device\"))\\n    {\\n        devID = getCmdLineArgumentInt(argc, argv, \"device=\");\\n\\n        if (devID < 0)\\n        {\\n            printf(\"Invalid command line parameter\\\\n \");\\n            exit(EXIT_FAILURE);\\n        }\\n        else\\n        {\\n            devID = gpuDeviceInit(devID);\\n\\n            if (devID < 0)\\n            {\\n                printf(\"exiting...\\\\n\");\\n                exit(EXIT_FAILURE);\\n            }\\n        }\\n    }\\n    else\\n    {\\n        // Otherwise pick the device with highest Gflops/s\\n        devID = gpuGetMaxGflopsDeviceId();\\n        checkCudaErrors(cudaSetDevice(devID));\\n        checkCudaErrors(cudaGetDeviceProperties(&deviceProp, devID));\\n        printf(\"GPU Device %d: \\\\\"%s\\\\\" with compute capability %d.%d\\\\n\\\\n\", devID, deviceProp.name, deviceProp.major, deviceProp.minor);\\n    }\\n\\n    return devID;\\n}\\n\\n// General check for CUDA GPU SM Capabilities\\ninline bool checkCudaCapabilities(int major_version, int minor_version)\\n{\\n    cudaDeviceProp deviceProp;\\n    deviceProp.major = 0;\\n    deviceProp.minor = 0;\\n    int dev;\\n\\n    checkCudaErrors(cudaGetDevice(&dev));\\n    checkCudaErrors(cudaGetDeviceProperties(&deviceProp, dev));\\n\\n    if ((deviceProp.major > major_version) ||\\n        (deviceProp.major == major_version && deviceProp.minor >= minor_version))\\n    {\\n        printf(\"> Device %d: <%16s >, Compute SM %d.%d detected\\\\n\", dev, deviceProp.name, deviceProp.major, deviceProp.minor);\\n        return true;\\n    }\\n    else\\n    {\\n        printf(\"No GPU device was found that can support CUDA compute capability %d.%d.\\\\n\", major_version, minor_version);\\n        return false;\\n    }\\n}\\n#endif\\n\\n// end of CUDA Helper Functions\\n\\n\\n#endif', 'adaptive_gridsampler/helper_string.h': '/**\\n * Copyright 1993-2012 NVIDIA Corporation.  All rights reserved.\\n *\\n * Please refer to the NVIDIA end user license agreement (EULA) associated\\n * with this source code for terms and conditions that govern your use of\\n * this software. Any use, reproduction, disclosure, or distribution of\\n * this software and related documentation outside the terms of the EULA\\n * is strictly prohibited.\\n *\\n */\\n\\n// These are helper functions for the SDK samples (string parsing, timers, etc)\\n#ifndef STRING_HELPER_H\\n#define STRING_HELPER_H\\n\\n#include <stdio.h>\\n#include <stdlib.h>\\n#include <fstream>\\n#include <string>\\n\\n#ifdef _WIN32\\n#ifndef STRCASECMP\\n#define STRCASECMP  _stricmp\\n#endif\\n#ifndef STRNCASECMP\\n#define STRNCASECMP _strnicmp\\n#endif\\n#ifndef STRCPY\\n#define STRCPY(sFilePath, nLength, sPath) strcpy_s(sFilePath, nLength, sPath)\\n#endif\\n\\n#ifndef FOPEN\\n#define FOPEN(fHandle,filename,mode) fopen_s(&fHandle, filename, mode)\\n#endif\\n#ifndef FOPEN_FAIL\\n#define FOPEN_FAIL(result) (result != 0)\\n#endif\\n#ifndef SSCANF\\n#define SSCANF sscanf_s\\n#endif\\n\\n#else\\n#include <string.h>\\n#include <strings.h>\\n\\n#ifndef STRCASECMP\\n#define STRCASECMP  strcasecmp\\n#endif\\n#ifndef STRNCASECMP\\n#define STRNCASECMP strncasecmp\\n#endif\\n#ifndef STRCPY\\n#define STRCPY(sFilePath, nLength, sPath) strcpy(sFilePath, sPath)\\n#endif\\n\\n#ifndef FOPEN\\n#define FOPEN(fHandle,filename,mode) (fHandle = fopen(filename, mode))\\n#endif\\n#ifndef FOPEN_FAIL\\n#define FOPEN_FAIL(result) (result == NULL)\\n#endif\\n#ifndef SSCANF\\n#define SSCANF sscanf\\n#endif\\n#endif\\n\\n// CUDA Utility Helper Functions\\ninline int stringRemoveDelimiter(char delimiter, const char *string)\\n{\\n    int string_start = 0;\\n\\n    while (string[string_start] == delimiter)\\n    {\\n        string_start++;\\n    }\\n\\n    if (string_start >= (int)strlen(string)-1)\\n    {\\n        return 0;\\n    }\\n\\n    return string_start;\\n}\\n\\ninline int getFileExtension(char *filename, char **extension)\\n{\\n    int string_length = (int)strlen(filename);\\n\\n    while (filename[string_length--] != \\'.\\') {\\n        if (string_length == 0)\\n            break;\\n    }\\n    if (string_length > 0) string_length += 2;\\n\\n    if (string_length == 0)\\n        *extension = NULL;\\n    else\\n        *extension = &filename[string_length];\\n\\n    return string_length;\\n}\\n\\n\\ninline int checkCmdLineFlag(const int argc, const char **argv, const char *string_ref)\\n{\\n    bool bFound = false;\\n\\n    if (argc >= 1)\\n    {\\n        for (int i=1; i < argc; i++)\\n        {\\n            int string_start = stringRemoveDelimiter(\\'-\\', argv[i]);\\n            const char *string_argv = &argv[i][string_start];\\n\\n            const char *equal_pos = strchr(string_argv, \\'=\\');\\n            int argv_length = (int)(equal_pos == 0 ? strlen(string_argv) : equal_pos - string_argv);\\n\\n            int length = (int)strlen(string_ref);\\n\\n            if (length == argv_length && !STRNCASECMP(string_argv, string_ref, length))\\n            {\\n\\n                bFound = true;\\n                continue;\\n            }\\n        }\\n    }\\n\\n    return (int)bFound;\\n}\\n\\ninline int getCmdLineArgumentInt(const int argc, const char **argv, const char *string_ref)\\n{\\n    bool bFound = false;\\n    int value = -1;\\n\\n    if (argc >= 1)\\n    {\\n        for (int i=1; i < argc; i++)\\n        {\\n            int string_start = stringRemoveDelimiter(\\'-\\', argv[i]);\\n            const char *string_argv = &argv[i][string_start];\\n            int length = (int)strlen(string_ref);\\n\\n            if (!STRNCASECMP(string_argv, string_ref, length))\\n            {\\n                if (length+1 <= (int)strlen(string_argv))\\n                {\\n                    int auto_inc = (string_argv[length] == \\'=\\') ? 1 : 0;\\n                    value = atoi(&string_argv[length + auto_inc]);\\n                }\\n                else\\n                {\\n                    value = 0;\\n                }\\n\\n                bFound = true;\\n                continue;\\n            }\\n        }\\n    }\\n\\n    if (bFound)\\n    {\\n        return value;\\n    }\\n    else\\n    {\\n        return 0;\\n    }\\n}\\n\\ninline float getCmdLineArgumentFloat(const int argc, const char **argv, const char *string_ref)\\n{\\n    bool bFound = false;\\n    float value = -1;\\n\\n    if (argc >= 1)\\n    {\\n        for (int i=1; i < argc; i++)\\n        {\\n            int string_start = stringRemoveDelimiter(\\'-\\', argv[i]);\\n            const char *string_argv = &argv[i][string_start];\\n            int length = (int)strlen(string_ref);\\n\\n            if (!STRNCASECMP(string_argv, string_ref, length))\\n            {\\n                if (length+1 <= (int)strlen(string_argv))\\n                {\\n                    int auto_inc = (string_argv[length] == \\'=\\') ? 1 : 0;\\n                    value = (float)atof(&string_argv[length + auto_inc]);\\n                }\\n                else\\n                {\\n                    value = 0.f;\\n                }\\n\\n                bFound = true;\\n                continue;\\n            }\\n        }\\n    }\\n\\n    if (bFound)\\n    {\\n        return value;\\n    }\\n    else\\n    {\\n        return 0;\\n    }\\n}\\n\\ninline bool getCmdLineArgumentString(const int argc, const char **argv,\\n                                     const char *string_ref, char **string_retval)\\n{\\n    bool bFound = false;\\n\\n    if (argc >= 1)\\n    {\\n        for (int i=1; i < argc; i++)\\n        {\\n            int string_start = stringRemoveDelimiter(\\'-\\', argv[i]);\\n            char *string_argv = (char *)&argv[i][string_start];\\n            int length = (int)strlen(string_ref);\\n\\n            if (!STRNCASECMP(string_argv, string_ref, length))\\n            {\\n                *string_retval = &string_argv[length+1];\\n                bFound = true;\\n                continue;\\n            }\\n        }\\n    }\\n\\n    if (!bFound)\\n    {\\n        *string_retval = NULL;\\n    }\\n\\n    return bFound;\\n}\\n\\n//////////////////////////////////////////////////////////////////////////////\\n//! Find the path for a file assuming that\\n//! files are found in the searchPath.\\n//!\\n//! @return the path if succeeded, otherwise 0\\n//! @param filename         name of the file\\n//! @param executable_path  optional absolute path of the executable\\n//////////////////////////////////////////////////////////////////////////////\\ninline char *sdkFindFilePath(const char *filename, const char *executable_path)\\n{\\n    // <executable_name> defines a variable that is replaced with the name of the executable\\n\\n    // Typical relative search paths to locate needed companion files (e.g. sample input data, or JIT source files)\\n    // The origin for the relative search may be the .exe file, a .bat file launching an .exe, a browser .exe launching the .exe or .bat, etc\\n    const char *searchPath[] =\\n    {\\n        \"./\",                                       // same dir\\n        \"./common/\",                                // \"/common/\" subdir\\n        \"./common/data/\",                           // \"/common/data/\" subdir\\n        \"./data/\",                                  // \"/data/\" subdir\\n        \"./src/\",                                   // \"/src/\" subdir\\n        \"./src/<executable_name>/data/\",            // \"/src/<executable_name>/data/\" subdir\\n        \"./inc/\",                                   // \"/inc/\" subdir\\n        \"./0_Simple/\",                              // \"/0_Simple/\" subdir\\n        \"./1_Utilities/\",                           // \"/1_Utilities/\" subdir\\n        \"./2_Graphics/\",                            // \"/2_Graphics/\" subdir\\n        \"./3_Imaging/\",                             // \"/3_Imaging/\" subdir\\n        \"./4_Financial/\",                           // \"/4_Financial/\" subdir\\n        \"./5_Simulations/\",                         // \"/5_Simulations/\" subdir\\n        \"./6_Advanced/\",                            // \"/6_Advanced/\" subdir\\n        \"./7_CUDALibraries/\",                       // \"/7_CUDALibraries/\" subdir\\n\\n        \"../\",                                      // up 1 in tree\\n        \"../common/\",                               // up 1 in tree, \"/common/\" subdir\\n        \"../common/data/\",                          // up 1 in tree, \"/common/data/\" subdir\\n        \"../data/\",                                 // up 1 in tree, \"/data/\" subdir\\n        \"../src/\",                                  // up 1 in tree, \"/src/\" subdir\\n        \"../inc/\",                                  // up 1 in tree, \"/inc/\" subdir\\n        \"../C/src/<executable_name>/\",              // up 1 in tree, \"/C/src/<executable_name>/\" subdir\\n        \"../C/src/<executable_name>/data/\",         // up 1 in tree, \"/C/src/<executable_name>/data/\" subdir\\n        \"../C/src/<executable_name>/src/\",          // up 1 in tree, \"/C/src/<executable_name>/src/\" subdir\\n        \"../C/src/<executable_name>/inc/\",          // up 1 in tree, \"/C/src/<executable_name>/inc/\" subdir\\n        \"../C/\",                                      // up 1 in tree\\n        \"../C/common/\",                               // up 1 in tree, \"/common/\" subdir\\n        \"../C/common/data/\",                          // up 1 in tree, \"/common/data/\" subdir\\n        \"../C/data/\",                                 // up 1 in tree, \"/data/\" subdir\\n        \"../C/src/\",                                  // up 1 in tree, \"/src/\" subdir\\n        \"../C/inc/\",                                  // up 1 in tree, \"/inc/\" subdir\\n        \"../C/0_Simple/<executable_name>/data/\",         // up 1 in tree, \"/0_Simple/<executable_name>/\" subdir\\n        \"../C/1_Utilities/<executable_name>/data/\",      // up 1 in tree, \"/1_Utilities/<executable_name>/\" subdir\\n        \"../C/2_Graphics/<executable_name>/data/\",       // up 1 in tree, \"/2_Graphics/<executable_name>/\" subdir\\n        \"../C/3_Imaging/<executable_name>/data/\",        // up 1 in tree, \"/3_Imaging/<executable_name>/\" subdir\\n        \"../C/4_Financial/<executable_name>/data/\",      // up 1 in tree, \"/4_Financial/<executable_name>/\" subdir\\n        \"../C/5_Simulations/<executable_name>/data/\",    // up 1 in tree, \"/5_Simulations/<executable_name>/\" subdir\\n        \"../C/6_Advanced/<executable_name>/data/\",       // up 1 in tree, \"/6_Advanced/<executable_name>/\" subdir\\n        \"../C/7_CUDALibraries/<executable_name>/data/\",  // up 1 in tree, \"/7_CUDALibraries/<executable_name>/\" subdir\\n\\n        \"../0_Simple/<executable_name>/data/\",           // up 1 in tree, \"/0_Simple/<executable_name>/\" subdir\\n        \"../1_Utilities/<executable_name>/data/\",        // up 1 in tree, \"/1_Utilities/<executable_name>/\" subdir\\n        \"../2_Graphics/<executable_name>/data/\",         // up 1 in tree, \"/2_Graphics/<executable_name>/\" subdir\\n        \"../3_Imaging/<executable_name>/data/\",          // up 1 in tree, \"/3_Imaging/<executable_name>/\" subdir\\n        \"../4_Financial/<executable_name>/data/\",        // up 1 in tree, \"/4_Financial/<executable_name>/\" subdir\\n        \"../5_Simulations/<executable_name>/data/\",      // up 1 in tree, \"/5_Simulations/<executable_name>/\" subdir\\n        \"../6_Advanced/<executable_name>/data/\",         // up 1 in tree, \"/6_Advanced/<executable_name>/\" subdir\\n        \"../7_CUDALibraries/<executable_name>/data/\",    // up 1 in tree, \"/7_CUDALibraries/<executable_name>/\" subdir\\n        \"../../\",                                   // up 2 in tree\\n        \"../../common/\",                            // up 2 in tree, \"/common/\" subdir\\n        \"../../common/data/\",                       // up 2 in tree, \"/common/data/\" subdir\\n        \"../../data/\",                              // up 2 in tree, \"/data/\" subdir\\n        \"../../src/\",                               // up 2 in tree, \"/src/\" subdir\\n        \"../../inc/\",                               // up 2 in tree, \"/inc/\" subdir\\n        \"../../sandbox/<executable_name>/data/\",    // up 2 in tree, \"/sandbox/<executable_name>/\" subdir\\n        \"../../0_Simple/<executable_name>/data/\",        // up 2 in tree, \"/0_Simple/<executable_name>/\" subdir\\n        \"../../1_Utilities/<executable_name>/data/\",     // up 2 in tree, \"/1_Utilities/<executable_name>/\" subdir\\n        \"../../2_Graphics/<executable_name>/data/\",      // up 2 in tree, \"/2_Graphics/<executable_name>/\" subdir\\n        \"../../3_Imaging/<executable_name>/data/\",       // up 2 in tree, \"/3_Imaging/<executable_name>/\" subdir\\n        \"../../4_Financial/<executable_name>/data/\",     // up 2 in tree, \"/4_Financial/<executable_name>/\" subdir\\n        \"../../5_Simulations/<executable_name>/data/\",   // up 2 in tree, \"/5_Simulations/<executable_name>/\" subdir\\n        \"../../6_Advanced/<executable_name>/data/\",      // up 2 in tree, \"/6_Advanced/<executable_name>/\" subdir\\n        \"../../7_CUDALibraries/<executable_name>/data/\", // up 2 in tree, \"/7_CUDALibraries/<executable_name>/\" subdir\\n        \"../../../\",                                // up 3 in tree\\n        \"../../../src/<executable_name>/\",          // up 3 in tree, \"/src/<executable_name>/\" subdir\\n        \"../../../src/<executable_name>/data/\",     // up 3 in tree, \"/src/<executable_name>/data/\" subdir\\n        \"../../../src/<executable_name>/src/\",      // up 3 in tree, \"/src/<executable_name>/src/\" subdir\\n        \"../../../src/<executable_name>/inc/\",      // up 3 in tree, \"/src/<executable_name>/inc/\" subdir\\n        \"../../../sandbox/<executable_name>/\",      // up 3 in tree, \"/sandbox/<executable_name>/\" subdir\\n        \"../../../sandbox/<executable_name>/data/\", // up 3 in tree, \"/sandbox/<executable_name>/data/\" subdir\\n        \"../../../sandbox/<executable_name>/src/\",  // up 3 in tree, \"/sandbox/<executable_name>/src/\" subdir\\n        \"../../../sandbox/<executable_name>/inc/\",   // up 3 in tree, \"/sandbox/<executable_name>/inc/\" subdir\\n        \"../../../0_Simple/<executable_name>/data/\",     // up 3 in tree, \"/0_Simple/<executable_name>/\" subdir\\n        \"../../../1_Utilities/<executable_name>/data/\",  // up 3 in tree, \"/1_Utilities/<executable_name>/\" subdir\\n        \"../../../2_Graphics/<executable_name>/data/\",   // up 3 in tree, \"/2_Graphics/<executable_name>/\" subdir\\n        \"../../../3_Imaging/<executable_name>/data/\",    // up 3 in tree, \"/3_Imaging/<executable_name>/\" subdir\\n        \"../../../4_Financial/<executable_name>/data/\",  // up 3 in tree, \"/4_Financial/<executable_name>/\" subdir\\n        \"../../../5_Simulations/<executable_name>/data/\",// up 3 in tree, \"/5_Simulations/<executable_name>/\" subdir\\n        \"../../../6_Advanced/<executable_name>/data/\",   // up 3 in tree, \"/6_Advanced/<executable_name>/\" subdir\\n        \"../../../7_CUDALibraries/<executable_name>/data/\", // up 3 in tree, \"/7_CUDALibraries/<executable_name>/\" subdir\\n        \"../../../common/\",                         // up 3 in tree, \"../../../common/\" subdir\\n        \"../../../common/data/\",                    // up 3 in tree, \"../../../common/data/\" subdir\\n        \"../../../data/\",                           // up 3 in tree, \"../../../data/\" subdir\\n    };\\n\\n    // Extract the executable name\\n    std::string executable_name;\\n\\n    if (executable_path != 0)\\n    {\\n        executable_name = std::string(executable_path);\\n\\n#ifdef _WIN32\\n        // Windows path delimiter\\n        size_t delimiter_pos = executable_name.find_last_of(\\'\\\\\\\\\\');\\n        executable_name.erase(0, delimiter_pos + 1);\\n\\n        if (executable_name.rfind(\".exe\") != std::string::npos)\\n        {\\n            // we strip .exe, only if the .exe is found\\n            executable_name.resize(executable_name.size() - 4);\\n        }\\n\\n#else\\n        // Linux & OSX path delimiter\\n        size_t delimiter_pos = executable_name.find_last_of(\\'/\\');\\n        executable_name.erase(0,delimiter_pos+1);\\n#endif\\n    }\\n\\n    // Loop over all search paths and return the first hit\\n    for (unsigned int i = 0; i < sizeof(searchPath)/sizeof(char *); ++i)\\n    {\\n        std::string path(searchPath[i]);\\n        size_t executable_name_pos = path.find(\"<executable_name>\");\\n\\n        // If there is executable_name variable in the searchPath\\n        // replace it with the value\\n        if (executable_name_pos != std::string::npos)\\n        {\\n            if (executable_path != 0)\\n            {\\n                path.replace(executable_name_pos, strlen(\"<executable_name>\"), executable_name);\\n            }\\n            else\\n            {\\n                // Skip this path entry if no executable argument is given\\n                continue;\\n            }\\n        }\\n\\n#ifdef _DEBUG\\n        printf(\"sdkFindFilePath <%s> in %s\\\\n\", filename, path.c_str());\\n#endif\\n\\n        // Test if the file exists\\n        path.append(filename);\\n        FILE *fp;\\n        FOPEN(fp, path.c_str(), \"rb\");\\n\\n        if (fp != NULL)\\n        {\\n            fclose(fp);\\n            // File found\\n            // returning an allocated array here for backwards compatibility reasons\\n            char *file_path = (char *) malloc(path.length() + 1);\\n            STRCPY(file_path, path.length() + 1, path.c_str());\\n            return file_path;\\n        }\\n\\n        if (fp)\\n        {\\n            fclose(fp);\\n        }\\n    }\\n\\n    // File not found\\n    return 0;\\n}\\n\\n#endif', 'adaptive_gridsampler/setup.py': 'from setuptools import setup\\nfrom torch.utils.cpp_extension import BuildExtension, CUDAExtension\\n\\nsetup(\\n    name=\"adaptive_gridsampler_cuda\",\\n    ext_modules=[\\n        CUDAExtension(\\n            \"adaptive_gridsampler_cuda\",\\n            [\"adaptive_gridsampler_cuda.cpp\", \"adaptive_gridsampler_kernel.cu\"],\\n            extra_compile_args={\\n                \"cxx\": [\"-O3\", \"-std=c++17\"],\\n                \"nvcc\": [\"-O3\"],\\n            },\\n        )\\n    ],\\n    cmdclass={\"build_ext\": BuildExtension},\\n)', 'adaptive_gridsampler/__init__.py': ''}\n\ndef uses_lid_downscale(method: str) -> bool:\n    return str(method).lower() in {\"lid\", \"car_lid\", \"car/lid\", \"car\"}\n\nUSES_LID = uses_lid_downscale(CFG.get(\"downscale_method\")) or uses_lid_downscale(\n    CFG.get(\"expert_downscale_method\", CFG.get(\"downscale_method\"))\n)\nif USES_LID:\n    CAR_DIR = Path(CFG[\"car_repo_dir\"])\n    for relative_path, source in CAR_SOURCE_FILES.items():\n        destination = CAR_DIR / relative_path\n        destination.parent.mkdir(parents=True, exist_ok=True)\n        destination.write_text(source, encoding=\"utf-8\")\n        print(\"Wrote:\", destination)\n\n    assert torch.cuda.is_available(), \"CAR/LID downscaling requires a Kaggle GPU accelerator.\"\n    major, minor = torch.cuda.get_device_capability()\n    os.environ[\"TORCH_CUDA_ARCH_LIST\"] = f\"{major}.{minor}\"\n    os.environ[\"MAX_JOBS\"] = \"2\"\n\n    subprocess.run(\n        [sys.executable, \"setup.py\", \"build_ext\", \"--inplace\"],\n        cwd=str(CAR_DIR / \"adaptive_gridsampler\"),\n        check=True,\n    )\n    if str(CAR_DIR) not in sys.path:\n        sys.path.insert(0, str(CAR_DIR))\n\n    from adaptive_gridsampler.gridsampler import Downsampler\n    from modules import DSN\n\n    weight_path = Path(CFG[\"lid_weight_path\"])\n    assert weight_path.exists(), (\n        f\"Missing CAR/LID KGN weight: {weight_path}. \"\n        \"Attach Kaggle dataset khangcancode/car-trained-model-from-github.\"\n    )\n    print(\"CAR/LID extension ready.\")\n    print(\"CAR/LID weight:\", weight_path)\nelse:\n    print(\"Skipping CAR/LID setup; configured downscale methods use OpenCV.\")\nprint(\"Main downscale method:\", CFG[\"downscale_method\"])\nprint(\"Expert downscale method:\", CFG.get(\"expert_downscale_method\", CFG[\"downscale_method\"]))","metadata":{"execution":{"iopub.execute_input":"2026-06-28T03:07:55.707085Z","iopub.status.busy":"2026-06-28T03:07:55.706642Z","iopub.status.idle":"2026-06-28T03:07:55.738389Z","shell.execute_reply":"2026-06-28T03:07:55.737718Z"},"papermill":{"duration":0.048405,"end_time":"2026-06-28T03:07:55.739917+00:00","exception":false,"start_time":"2026-06-28T03:07:55.691512+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"6f564c79","cell_type":"markdown","source":"## Write the Shared Pipeline Modules","metadata":{"papermill":{"duration":0.009369,"end_time":"2026-06-28T03:07:55.758774+00:00","exception":false,"start_time":"2026-06-28T03:07:55.749405+00:00","status":"completed"},"tags":[]}},{"id":"aaf055e8","cell_type":"code","source":"MODULE_SOURCES = {'dr_manifest_correction.py': 'from __future__ import annotations\\n\\nimport argparse\\nimport json\\nimport re\\nfrom pathlib import Path\\n\\nimport pandas as pd\\n\\n\\nIMAGE_EXTENSIONS = {\".jpg\", \".jpeg\", \".png\", \".bmp\", \".tif\", \".tiff\", \".gif\", \".gf\"}\\nSPLIT_FOLDERS = (\"train\", \"val\", \"test\")\\nCLASS_FOLDERS = tuple(str(index) for index in range(5))\\nMESSIDOR_FILENAME_RE = re.compile(r\"^\\\\d{8}_\\\\d{5}_\\\\d{4}_PP\\\\.tiff?$\", re.IGNORECASE)\\n\\nMESSIDOR_ERRATA = {\\n    \"20051020_63045_0100_PP.tif\": 0,\\n    \"20051020_64007_0100_PP.tif\": 3,\\n    \"20051020_63936_0100_PP.tif\": 1,\\n    \"20060523_48477_0100_PP.tif\": 3,\\n}\\n\\nBASE33_DUPLICATE_PAIRS = [\\n    (\"20051202_55582_0400_PP.tif\", \"20051202_54744_0400_PP.tif\"),\\n    (\"20051202_41076_0400_PP.tif\", \"20051202_40508_0400_PP.tif\"),\\n    (\"20051202_48287_0400_PP.tif\", \"20051202_41238_0400_PP.tif\"),\\n    (\"20051202_48586_0400_PP.tif\", \"20051202_41260_0400_PP.tif\"),\\n    (\"20051202_55457_0400_PP.tif\", \"20051202_54530_0400_PP.tif\"),\\n    (\"20051202_55626_0400_PP.tif\", \"20051205_33025_0400_PP.tif\"),\\n    (\"20051202_54783_0400_PP.tif\", \"20051202_55607_0400_PP.tif\"),\\n    (\"20051202_48575_0400_PP.tif\", \"20051202_41034_0400_PP.tif\"),\\n    (\"20051205_32966_0400_PP.tif\", \"20051205_35099_0400_PP.tif\"),\\n    (\"20051202_55484_0400_PP.tif\", \"20051202_54555_0400_PP.tif\"),\\n    (\"20051205_32981_0400_PP.tif\", \"20051205_35110_0400_PP.tif\"),\\n    (\"20051202_55562_0400_PP.tif\", \"20051202_54611_0400_PP.tif\"),\\n    (\"20051202_54547_0400_PP.tif\", \"20051202_55498_0400_PP.tif\"),\\n]\\n\\nINCONSISTENT_RETINOPATHY_PAIR = frozenset(\\n    {\"20051202_55626_0400_PP.tif\", \"20051205_33025_0400_PP.tif\"}\\n)\\nINCONSISTENT_MACULAR_EDEMA_ONLY_PAIR = frozenset(\\n    {\"20051202_55562_0400_PP.tif\", \"20051202_54611_0400_PP.tif\"}\\n)\\n\\n\\ndef append_reason(current: str, reason: str) -> str:\\n    current = \"\" if pd.isna(current) else str(current)\\n    if not current:\\n        return reason\\n    if reason in current.split(\";\"):\\n        return current\\n    return f\"{current};{reason}\"\\n\\n\\ndef normalize_filename(filename: str) -> str:\\n    return Path(str(filename)).name.lower()\\n\\n\\ndef has_mixed_split_layout(root: Path) -> bool:\\n    return all((root / split).is_dir() for split in SPLIT_FOLDERS) and any(\\n        (root / split / label).is_dir()\\n        for split in SPLIT_FOLDERS\\n        for label in CLASS_FOLDERS\\n    )\\n\\n\\ndef resolve_mixed_dataset_root(data_root: str | Path, dataset_variant: str | None = None) -> Path:\\n    root = Path(data_root)\\n    candidates = [root]\\n    if dataset_variant:\\n        candidates.extend(\\n            [\\n                root / dataset_variant,\\n                root / dataset_variant / dataset_variant,\\n            ]\\n        )\\n    candidates.extend(\\n        [\\n            root / \"dr_unified_v2\",\\n            root / \"dr_unified_v2\" / \"dr_unified_v2\",\\n        ]\\n    )\\n    for candidate in candidates:\\n        if candidate.is_dir() and has_mixed_split_layout(candidate):\\n            return candidate\\n    for candidate in root.rglob(\"*\"):\\n        if candidate.is_dir() and candidate.name in {\"dr_unified_v2\", \"data\"}:\\n            if has_mixed_split_layout(candidate):\\n                return candidate\\n    searched = \"\\\\n  \".join(str(path) for path in candidates)\\n    raise FileNotFoundError(\\n        \"Could not find mixed DR folder layout {train,val,test}/{0..4}. \"\\n        f\"Searched:\\\\n  {searched}\"\\n    )\\n\\n\\ndef guess_source(filename: str) -> str:\\n    name = Path(filename).name\\n    stem = Path(filename).stem.lower()\\n    if MESSIDOR_FILENAME_RE.match(name):\\n        return \"messidor_like\"\\n    if re.search(r\"(^|[_-])(left|right|l|r)$\", stem):\\n        return \"eyepacs_like\"\\n    if re.fullmatch(r\"[0-9a-f]{8,32}\", stem):\\n        return \"aptos_like\"\\n    return \"unknown_mixed\"\\n\\n\\ndef scan_mixed_dataset(data_root: str | Path, dataset_variant: str | None = None) -> pd.DataFrame:\\n    root = resolve_mixed_dataset_root(data_root, dataset_variant)\\n    rows = []\\n    for split in SPLIT_FOLDERS:\\n        for label in CLASS_FOLDERS:\\n            folder = root / split / label\\n            if not folder.is_dir():\\n                continue\\n            for path in sorted(folder.rglob(\"*\")):\\n                if not path.is_file() or path.suffix.lower() not in IMAGE_EXTENSIONS:\\n                    continue\\n                filename = path.name\\n                rows.append(\\n                    {\\n                        \"path\": str(path),\\n                        \"filename\": filename,\\n                        \"split_folder\": split,\\n                        \"folder_label\": int(label),\\n                        \"corrected_label\": int(label),\\n                        \"source_guess\": guess_source(filename),\\n                        \"is_messidor_like_filename\": bool(MESSIDOR_FILENAME_RE.match(filename)),\\n                        \"correction_reason\": \"\",\\n                        \"duplicate_cluster_id\": \"\",\\n                        \"exclude_from_training\": False,\\n                        \"exclude_reason\": \"\",\\n                    }\\n                )\\n    manifest = pd.DataFrame(rows)\\n    if manifest.empty:\\n        raise RuntimeError(f\"No image files found below mixed dataset root: {root}\")\\n    return manifest\\n\\n\\ndef apply_messidor_errata(manifest: pd.DataFrame) -> tuple[pd.DataFrame, list[dict]]:\\n    manifest = manifest.copy()\\n    audit_rows = []\\n    normalized = manifest[\"filename\"].map(normalize_filename)\\n    for filename, corrected_label in MESSIDOR_ERRATA.items():\\n        mask = normalized == filename.lower()\\n        for index in manifest.index[mask]:\\n            old_label = int(manifest.at[index, \"corrected_label\"])\\n            manifest.at[index, \"corrected_label\"] = int(corrected_label)\\n            reason = f\"messidor_erratum:{old_label}->{int(corrected_label)}\"\\n            manifest.at[index, \"correction_reason\"] = append_reason(\\n                manifest.at[index, \"correction_reason\"],\\n                reason,\\n            )\\n            audit_rows.append(\\n                {\\n                    \"audit_type\": \"label_correction\",\\n                    \"path\": manifest.at[index, \"path\"],\\n                    \"filename\": manifest.at[index, \"filename\"],\\n                    \"split_folder\": manifest.at[index, \"split_folder\"],\\n                    \"folder_label\": int(manifest.at[index, \"folder_label\"]),\\n                    \"old_label\": old_label,\\n                    \"corrected_label\": int(corrected_label),\\n                    \"reason\": reason,\\n                }\\n            )\\n    return manifest, audit_rows\\n\\n\\ndef assign_known_duplicate_clusters(manifest: pd.DataFrame) -> pd.DataFrame:\\n    manifest = manifest.copy()\\n    normalized = manifest[\"filename\"].map(normalize_filename)\\n    for pair_index, pair in enumerate(BASE33_DUPLICATE_PAIRS, start=1):\\n        cluster_id = f\"base33_{pair_index:02d}\"\\n        pair_names = {name.lower() for name in pair}\\n        mask = normalized.isin(pair_names)\\n        manifest.loc[mask, \"duplicate_cluster_id\"] = cluster_id\\n    return manifest\\n\\n\\ndef apply_exclusion(manifest: pd.DataFrame, mask, reason: str) -> None:\\n    manifest.loc[mask, \"exclude_from_training\"] = True\\n    for index in manifest.index[mask]:\\n        manifest.at[index, \"exclude_reason\"] = append_reason(\\n            manifest.at[index, \"exclude_reason\"],\\n            reason,\\n        )\\n\\n\\ndef apply_duplicate_exclusions(manifest: pd.DataFrame) -> tuple[pd.DataFrame, list[dict]]:\\n    manifest = manifest.copy()\\n    audit_rows = []\\n    normalized = manifest[\"filename\"].map(normalize_filename)\\n\\n    retinopathy_names = {name.lower() for name in INCONSISTENT_RETINOPATHY_PAIR}\\n    retinopathy_mask = normalized.isin(retinopathy_names)\\n    apply_exclusion(\\n        manifest,\\n        retinopathy_mask,\\n        \"known_base33_duplicate_inconsistent_retinopathy_unadjudicated\",\\n    )\\n    for index in manifest.index[retinopathy_mask]:\\n        audit_rows.append(\\n            {\\n                \"audit_type\": \"exclusion\",\\n                \"path\": manifest.at[index, \"path\"],\\n                \"filename\": manifest.at[index, \"filename\"],\\n                \"split_folder\": manifest.at[index, \"split_folder\"],\\n                \"folder_label\": int(manifest.at[index, \"folder_label\"]),\\n                \"old_label\": int(manifest.at[index, \"folder_label\"]),\\n                \"corrected_label\": int(manifest.at[index, \"corrected_label\"]),\\n                \"reason\": \"known_base33_duplicate_inconsistent_retinopathy_unadjudicated\",\\n            }\\n        )\\n\\n    for cluster_id, group in manifest[manifest[\"duplicate_cluster_id\"] != \"\"].groupby(\\n        \"duplicate_cluster_id\",\\n        sort=False,\\n    ):\\n        splits = set(group[\"split_folder\"].astype(str))\\n        if len(splits) <= 1:\\n            continue\\n        if not (splits & {\"val\", \"test\"}):\\n            continue\\n        mask = manifest[\"duplicate_cluster_id\"] == cluster_id\\n        apply_exclusion(manifest, mask, \"cross_split_duplicate_cluster\")\\n        for index in manifest.index[mask]:\\n            audit_rows.append(\\n                {\\n                    \"audit_type\": \"exclusion\",\\n                    \"path\": manifest.at[index, \"path\"],\\n                    \"filename\": manifest.at[index, \"filename\"],\\n                    \"split_folder\": manifest.at[index, \"split_folder\"],\\n                    \"folder_label\": int(manifest.at[index, \"folder_label\"]),\\n                    \"old_label\": int(manifest.at[index, \"folder_label\"]),\\n                    \"corrected_label\": int(manifest.at[index, \"corrected_label\"]),\\n                    \"reason\": f\"cross_split_duplicate_cluster:{cluster_id}\",\\n                }\\n            )\\n\\n    macular_names = {name.lower() for name in INCONSISTENT_MACULAR_EDEMA_ONLY_PAIR}\\n    macular_mask = normalized.isin(macular_names)\\n    for index in manifest.index[macular_mask]:\\n        audit_rows.append(\\n            {\\n                \"audit_type\": \"duplicate_note\",\\n                \"path\": manifest.at[index, \"path\"],\\n                \"filename\": manifest.at[index, \"filename\"],\\n                \"split_folder\": manifest.at[index, \"split_folder\"],\\n                \"folder_label\": int(manifest.at[index, \"folder_label\"]),\\n                \"old_label\": int(manifest.at[index, \"folder_label\"]),\\n                \"corrected_label\": int(manifest.at[index, \"corrected_label\"]),\\n                \"reason\": \"known_base33_duplicate_inconsistent_macular_edema_only_not_excluded_for_dr\",\\n            }\\n        )\\n    return manifest, audit_rows\\n\\n\\ndef build_duplicate_audit(manifest: pd.DataFrame) -> pd.DataFrame:\\n    rows = []\\n    normalized = manifest[\"filename\"].map(normalize_filename)\\n    for pair_index, pair in enumerate(BASE33_DUPLICATE_PAIRS, start=1):\\n        cluster_id = f\"base33_{pair_index:02d}\"\\n        pair_names = {name.lower() for name in pair}\\n        group = manifest[normalized.isin(pair_names)].copy()\\n        splits = sorted(group[\"split_folder\"].astype(str).unique().tolist())\\n        folder_labels = sorted(group[\"folder_label\"].astype(int).unique().tolist())\\n        corrected_labels = sorted(group[\"corrected_label\"].astype(int).unique().tolist())\\n        rows.append(\\n            {\\n                \"duplicate_cluster_id\": cluster_id,\\n                \"left_filename\": pair[0],\\n                \"right_filename\": pair[1],\\n                \"left_present\": bool((normalized == pair[0].lower()).any()),\\n                \"right_present\": bool((normalized == pair[1].lower()).any()),\\n                \"present_count\": int(len(group)),\\n                \"split_folders\": \"|\".join(splits),\\n                \"folder_labels\": \"|\".join(map(str, folder_labels)),\\n                \"corrected_labels\": \"|\".join(map(str, corrected_labels)),\\n                \"retinopathy_inconsistent_known\": frozenset(pair)\\n                == INCONSISTENT_RETINOPATHY_PAIR,\\n                \"macular_edema_only_inconsistent_known\": frozenset(pair)\\n                == INCONSISTENT_MACULAR_EDEMA_ONLY_PAIR,\\n                \"crosses_split_folder\": len(splits) > 1,\\n                \"exclude_applied\": bool(group[\"exclude_from_training\"].any()) if len(group) else False,\\n                \"exclude_reasons\": \"|\".join(\\n                    sorted(\\n                        {\\n                            reason\\n                            for value in group[\"exclude_reason\"].astype(str)\\n                            for reason in value.split(\";\")\\n                            if reason\\n                        }\\n                    )\\n                ),\\n            }\\n        )\\n    return pd.DataFrame(rows)\\n\\n\\ndef build_summary_counts(manifest: pd.DataFrame) -> pd.DataFrame:\\n    rows = []\\n    before = (\\n        manifest.groupby([\"split_folder\", \"folder_label\"])\\n        .size()\\n        .rename(\"count\")\\n        .reset_index()\\n    )\\n    for row in before.itertuples(index=False):\\n        rows.append(\\n            {\\n                \"phase\": \"before_correction\",\\n                \"split_folder\": row.split_folder,\\n                \"label\": int(row.folder_label),\\n                \"count\": int(row.count),\\n            }\\n        )\\n    active = manifest[~manifest[\"exclude_from_training\"].astype(bool)]\\n    after = (\\n        active.groupby([\"split_folder\", \"corrected_label\"])\\n        .size()\\n        .rename(\"count\")\\n        .reset_index()\\n    )\\n    for row in after.itertuples(index=False):\\n        rows.append(\\n            {\\n                \"phase\": \"after_correction_exclusion\",\\n                \"split_folder\": row.split_folder,\\n                \"label\": int(row.corrected_label),\\n                \"count\": int(row.count),\\n            }\\n        )\\n    return pd.DataFrame(rows)\\n\\n\\ndef calculate_summary(manifest: pd.DataFrame, duplicate_audit: pd.DataFrame) -> dict:\\n    corrected_count = int(\\n        (manifest[\"corrected_label\"].astype(int) != manifest[\"folder_label\"].astype(int)).sum()\\n    )\\n    duplicate_clusters_found = int((duplicate_audit[\"present_count\"].astype(int) >= 2).sum())\\n    duplicate_clusters_crossing = int(\\n        (\\n            (duplicate_audit[\"present_count\"].astype(int) >= 2)\\n            & duplicate_audit[\"crosses_split_folder\"].astype(bool)\\n        ).sum()\\n    )\\n    return {\\n        \"total_images_scanned\": int(len(manifest)),\\n        \"messidor_like_files_found\": int(manifest[\"is_messidor_like_filename\"].astype(bool).sum()),\\n        \"corrected_labels\": corrected_count,\\n        \"excluded_images\": int(manifest[\"exclude_from_training\"].astype(bool).sum()),\\n        \"duplicate_clusters_found_current_dataset\": duplicate_clusters_found,\\n        \"duplicate_clusters_crossing_split_folders\": duplicate_clusters_crossing,\\n    }\\n\\n\\ndef print_validation_summary(\\n    manifest: pd.DataFrame,\\n    duplicate_audit: pd.DataFrame,\\n    summary_counts: pd.DataFrame,\\n) -> None:\\n    summary = calculate_summary(manifest, duplicate_audit)\\n    print(\"Dataset correction validation\")\\n    print(f\"number of total images scanned: {summary[\\'total_images_scanned\\']}\")\\n    print(f\"number of Messidor-like files found: {summary[\\'messidor_like_files_found\\']}\")\\n    print(f\"number of corrected labels: {summary[\\'corrected_labels\\']}\")\\n    print(f\"number of excluded images: {summary[\\'excluded_images\\']}\")\\n    print(\\n        \"duplicate clusters found in the current dataset: \"\\n        f\"{summary[\\'duplicate_clusters_found_current_dataset\\']}\"\\n    )\\n    print(\\n        \"duplicate clusters crossing split folders: \"\\n        f\"{summary[\\'duplicate_clusters_crossing_split_folders\\']}\"\\n    )\\n    print(\"class counts by split before correction:\")\\n    before = summary_counts[summary_counts[\"phase\"] == \"before_correction\"]\\n    print(\\n        before.pivot_table(\\n            index=\"label\",\\n            columns=\"split_folder\",\\n            values=\"count\",\\n            fill_value=0,\\n            aggfunc=\"sum\",\\n        )\\n        .reindex(index=range(5), columns=SPLIT_FOLDERS, fill_value=0)\\n        .astype(int)\\n        .to_string()\\n    )\\n    print(\"class counts by split after correction/exclusion:\")\\n    after = summary_counts[summary_counts[\"phase\"] == \"after_correction_exclusion\"]\\n    print(\\n        after.pivot_table(\\n            index=\"label\",\\n            columns=\"split_folder\",\\n            values=\"count\",\\n            fill_value=0,\\n            aggfunc=\"sum\",\\n        )\\n        .reindex(index=range(5), columns=SPLIT_FOLDERS, fill_value=0)\\n        .astype(int)\\n        .to_string()\\n    )\\n\\n\\ndef build_corrected_manifest(\\n    data_root: str | Path,\\n    output_dir: str | Path,\\n    dataset_variant: str | None = None,\\n) -> tuple[pd.DataFrame, pd.DataFrame, pd.DataFrame, dict]:\\n    output_dir = Path(output_dir)\\n    output_dir.mkdir(parents=True, exist_ok=True)\\n    manifest = scan_mixed_dataset(data_root, dataset_variant)\\n    manifest, correction_rows = apply_messidor_errata(manifest)\\n    manifest = assign_known_duplicate_clusters(manifest)\\n    manifest, exclusion_rows = apply_duplicate_exclusions(manifest)\\n    duplicate_audit = build_duplicate_audit(manifest)\\n    correction_audit = pd.DataFrame(correction_rows + exclusion_rows)\\n    summary_counts = build_summary_counts(manifest)\\n    summary = calculate_summary(manifest, duplicate_audit)\\n\\n    manifest.to_csv(output_dir / \"corrected_manifest.csv\", index=False)\\n    correction_audit.to_csv(output_dir / \"correction_audit.csv\", index=False)\\n    duplicate_audit.to_csv(output_dir / \"duplicate_audit.csv\", index=False)\\n    summary_counts.to_csv(output_dir / \"summary_counts.csv\", index=False)\\n    with open(output_dir / \"manifest_summary.json\", \"w\", encoding=\"utf-8\") as handle:\\n        json.dump(summary, handle, indent=2)\\n    print_validation_summary(manifest, duplicate_audit, summary_counts)\\n    return manifest, correction_audit, duplicate_audit, summary\\n\\n\\ndef main() -> None:\\n    parser = argparse.ArgumentParser()\\n    parser.add_argument(\"--data-root\", required=True)\\n    parser.add_argument(\"--output-dir\", required=True)\\n    parser.add_argument(\"--dataset-variant\", default=None)\\n    args = parser.parse_args()\\n    build_corrected_manifest(args.data_root, args.output_dir, args.dataset_variant)\\n\\n\\nif __name__ == \"__main__\":\\n    main()\\n', 'dr_data_corrected.py': 'from __future__ import annotations\\n\\nimport concurrent.futures as futures\\nimport hashlib\\nimport json\\nimport os\\nimport re\\nimport shutil\\nimport time\\nfrom pathlib import Path\\nfrom typing import Iterable\\n\\nimport numpy as np\\nimport pandas as pd\\nfrom PIL import Image, ImageFile, UnidentifiedImageError\\nfrom sklearn.model_selection import StratifiedGroupKFold\\nfrom tqdm.auto import tqdm\\n\\ntry:\\n    import cv2\\n\\n    cv2.setUseOptimized(True)\\n    cv2.setNumThreads(0)\\nexcept ImportError as exc:\\n    raise RuntimeError(\"OpenCV is required by the corrected data pipeline.\") from exc\\n\\nImageFile.LOAD_TRUNCATED_IMAGES = True\\n\\nIMAGE_EXTENSIONS = {\".jpg\", \".jpeg\", \".png\", \".bmp\", \".tif\", \".tiff\", \".gif\", \".gf\"}\\nCLASS_NAMES = {\\n    0: \"No DR\",\\n    1: \"Mild NPDR\",\\n    2: \"Moderate NPDR\",\\n    3: \"Severe NPDR\",\\n    4: \"Proliferative DR\",\\n}\\nCLASS_ALIASES = {\\n    \"0\": 0,\\n    \"no_dr\": 0,\\n    \"nodr\": 0,\\n    \"normal\": 0,\\n    \"1\": 1,\\n    \"mild\": 1,\\n    \"mild_dr\": 1,\\n    \"mild_npdr\": 1,\\n    \"2\": 2,\\n    \"moderate\": 2,\\n    \"moderate_dr\": 2,\\n    \"moderate_npdr\": 2,\\n    \"3\": 3,\\n    \"severe\": 3,\\n    \"severe_dr\": 3,\\n    \"severe_npdr\": 3,\\n    \"4\": 4,\\n    \"proliferative_dr\": 4,\\n    \"proliferate_dr\": 4,\\n    \"pdr\": 4,\\n}\\nEYE_SUFFIX_RE = re.compile(r\"[-_](left|right|l|r)$\", re.IGNORECASE)\\nGF_SUFFIX_RE = re.compile(r\"[-_]gf$\", re.IGNORECASE)\\nPREPROCESS_CODE_VERSION = \"corrected-2026-06-27-tiff-pil-jpeg-v1\"\\nPREPROCESSING_PROFILES = {\\n    \"crop_resize_only\": {\\n        \"ben_graham\": False,\\n        \"clahe\": False,\\n        \"mask_background\": False,\\n        \"blend\": False,\\n    },\\n    \"ben_graham_only\": {\\n        \"ben_graham\": True,\\n        \"clahe\": False,\\n        \"mask_background\": True,\\n        \"blend\": True,\\n    },\\n    \"clahe_only\": {\\n        \"ben_graham\": False,\\n        \"clahe\": True,\\n        \"mask_background\": True,\\n        \"blend\": True,\\n    },\\n    \"full_current\": {\\n        \"ben_graham\": True,\\n        \"clahe\": True,\\n        \"mask_background\": True,\\n        \"blend\": True,\\n    },\\n}\\n\\n\\ndef preprocessing_profile_config(cfg: dict) -> tuple[str, dict]:\\n    profile_name = str(cfg.get(\"preprocessing_profile\", \"full_current\")).strip() or \"full_current\"\\n    if profile_name not in PREPROCESSING_PROFILES:\\n        raise ValueError(\\n            f\"Unknown preprocessing_profile={profile_name!r}; expected one of \"\\n            f\"{sorted(PREPROCESSING_PROFILES)}.\"\\n        )\\n    return profile_name, dict(PREPROCESSING_PROFILES[profile_name])\\n\\n\\n\\nclass UnionFind:\\n    def __init__(self, values: Iterable) -> None:\\n        self.parent = {value: value for value in values}\\n        self.rank = {value: 0 for value in values}\\n\\n    def find(self, value):\\n        parent = self.parent[value]\\n        if parent != value:\\n            self.parent[value] = self.find(parent)\\n        return self.parent[value]\\n\\n    def union(self, first, second) -> None:\\n        root_a = self.find(first)\\n        root_b = self.find(second)\\n        if root_a == root_b:\\n            return\\n        if self.rank[root_a] < self.rank[root_b]:\\n            root_a, root_b = root_b, root_a\\n        self.parent[root_b] = root_a\\n        if self.rank[root_a] == self.rank[root_b]:\\n            self.rank[root_a] += 1\\n\\n\\nclass BKTree:\\n    def __init__(self) -> None:\\n        self.root = None\\n\\n    @staticmethod\\n    def distance(first: int, second: int) -> int:\\n        return int((first ^ second).bit_count())\\n\\n    def add(self, value: int) -> None:\\n        if self.root is None:\\n            self.root = [value, {}]\\n            return\\n        node = self.root\\n        while True:\\n            distance = self.distance(value, node[0])\\n            if distance == 0:\\n                return\\n            child = node[1].get(distance)\\n            if child is None:\\n                node[1][distance] = [value, {}]\\n                return\\n            node = child\\n\\n    def query(self, value: int, radius: int) -> list[int]:\\n        if self.root is None:\\n            return []\\n        matches = []\\n        stack = [self.root]\\n        while stack:\\n            node_value, children = stack.pop()\\n            distance = self.distance(value, node_value)\\n            if distance <= radius:\\n                matches.append(node_value)\\n            low = distance - radius\\n            high = distance + radius\\n            stack.extend(child for edge, child in children.items() if low <= edge <= high)\\n        return matches\\n\\n\\ndef normalize_token(value: str) -> str:\\n    return re.sub(r\"[^a-z0-9_]+\", \"\", value.lower().strip())\\n\\n\\ndef infer_label(parts: Iterable[str]) -> int | None:\\n    for part in reversed(list(parts)):\\n        token = normalize_token(part)\\n        if token in CLASS_ALIASES:\\n            return CLASS_ALIASES[token]\\n    return None\\n\\n\\ndef directory_has_labelled_images(root: Path, limit: int = 4000) -> bool:\\n    seen = 0\\n    for path in root.rglob(\"*\"):\\n        if path.suffix.lower() not in IMAGE_EXTENSIONS:\\n            continue\\n        seen += 1\\n        if infer_label(path.relative_to(root).parts) is not None:\\n            return True\\n        if seen >= limit:\\n            break\\n    return False\\n\\n\\ndef resolve_data_root(cfg: dict) -> Path:\\n    candidates = []\\n    if cfg.get(\"data_root\"):\\n        candidates.append(Path(cfg[\"data_root\"]))\\n    if cfg.get(\"dataset_slug\"):\\n        candidates.append(Path(\"/kaggle/input\") / str(cfg[\"dataset_slug\"]))\\n        candidates.append(Path(\"/kaggle/input\") / Path(str(cfg[\"dataset_slug\"])).name)\\n    if cfg.get(\"data_root\"):\\n        candidates.append(Path(\"/kaggle/input\") / Path(str(cfg[\"data_root\"])).name)\\n\\n    root = next((path for path in candidates if path.is_dir()), None)\\n    if root is None:\\n        kaggle_input = Path(\"/kaggle/input\")\\n        children = sorted(path for path in kaggle_input.iterdir() if path.is_dir())\\n        root = next((path for path in children if directory_has_labelled_images(path)), None)\\n    if root is None:\\n        raise FileNotFoundError(\"Could not locate a labelled diabetic-retinopathy dataset.\")\\n    variant = cfg.get(\"dataset_variant\")\\n    if variant and (root / str(variant)).is_dir():\\n        root = root / str(variant)\\n    return root\\n\\n\\ndef strip_gaussian_suffix(stem: str) -> str:\\n    return GF_SUFFIX_RE.sub(\"\", str(stem))\\n\\n\\ndef extract_image_id(stem: str, source: str) -> str:\\n    return f\"{source}:{strip_gaussian_suffix(stem)}\"\\n\\n\\ndef extract_patient_key(stem: str, source: str) -> str:\\n    base = EYE_SUFFIX_RE.sub(\"\", strip_gaussian_suffix(stem))\\n    return f\"{source}:{base}\"\\n\\n\\ndef scan_dataset(root: Path) -> pd.DataFrame:\\n    rows = []\\n    for image_path in tqdm(root.rglob(\"*\"), desc=\"Scanning raw dataset\"):\\n        if image_path.suffix.lower() not in IMAGE_EXTENSIONS:\\n            continue\\n        relative = image_path.relative_to(root)\\n        label = infer_label(relative.parts)\\n        if label is None:\\n            continue\\n        stat = image_path.stat()\\n        rows.append(\\n            {\\n                \"path\": str(image_path),\\n                \"label\": int(label),\\n                \"source\": relative.parts[0] if relative.parts else \"unknown\",\\n                \"stem\": image_path.stem,\\n                \"ext\": image_path.suffix.lower(),\\n                \"file_size\": int(stat.st_size),\\n                \"file_mtime_ns\": int(stat.st_mtime_ns),\\n            }\\n        )\\n    if not rows:\\n        raise RuntimeError(f\"No labelled retina images found below {root}\")\\n    frame = pd.DataFrame(rows).drop_duplicates(\"path\").reset_index(drop=True)\\n    frame[\"is_gf\"] = frame[\"stem\"].str.contains(r\"(?i)[-_]gf$\", regex=True)\\n    frame[\"image_id\"] = [\\n        extract_image_id(stem, source)\\n        for stem, source in zip(frame[\"stem\"], frame[\"source\"])\\n    ]\\n    frame[\"patient_key\"] = [\\n        extract_patient_key(stem, source)\\n        for stem, source in zip(frame[\"stem\"], frame[\"source\"])\\n    ]\\n    return frame\\n\\n\\ndef remove_filename_duplicates(frame: pd.DataFrame) -> tuple[pd.DataFrame, pd.DataFrame]:\\n    audit_rows = []\\n    keep_indices = []\\n    for image_id, group in frame.groupby(\"image_id\", sort=False):\\n        ordered = group.sort_values([\"is_gf\", \"path\"], kind=\"stable\")\\n        keep = ordered.index[0]\\n        conflicting = ordered[\"label\"].astype(int).nunique() > 1\\n        if not conflicting:\\n            keep_indices.append(keep)\\n        for index in ordered.index[1:]:\\n            audit_rows.append(\\n                {\\n                    \"duplicate_type\": \"filename_gf\",\\n                    \"left_index\": int(keep),\\n                    \"right_index\": int(index),\\n                    \"left_path\": frame.at[keep, \"path\"],\\n                    \"right_path\": frame.at[index, \"path\"],\\n                    \"left_label\": int(frame.at[keep, \"label\"]),\\n                    \"right_label\": int(frame.at[index, \"label\"]),\\n                    \"confirmed\": True,\\n                    \"action\": \"quarantine_conflicting_labels\"\\n                    if conflicting\\n                    else \"collapse\",\\n                    \"image_id\": image_id,\\n                    \"cluster_key\": f\"filename:{image_id}\",\\n                }\\n            )\\n    deduplicated = frame.loc[sorted(keep_indices)].reset_index(drop=True)\\n    return deduplicated, pd.DataFrame(audit_rows)\\n\\n\\n_REDUCED_FLAGS = {\\n    1: cv2.IMREAD_COLOR,\\n    2: cv2.IMREAD_REDUCED_COLOR_2,\\n    4: cv2.IMREAD_REDUCED_COLOR_4,\\n    8: cv2.IMREAD_REDUCED_COLOR_8,\\n}\\n\\n\\ndef read_rgb_image(path: str) -> np.ndarray:\\n    suffix = Path(path).suffix.lower()\\n    if suffix in {\".gif\", \".gf\", \".tif\", \".tiff\"}:\\n        try:\\n            with Image.open(path) as image:\\n                return np.asarray(image.convert(\"RGB\"))\\n        except (UnidentifiedImageError, OSError) as exc:\\n            raise RuntimeError(f\"Failed decoding image {path}: {exc}\") from exc\\n    if suffix not in {\".gif\", \".gf\"}:\\n        image = cv2.imread(path, cv2.IMREAD_COLOR)\\n        if image is not None:\\n            return cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\\n    try:\\n        with Image.open(path) as image:\\n            return np.asarray(image.convert(\"RGB\"))\\n    except (UnidentifiedImageError, OSError) as exc:\\n        raise RuntimeError(f\"Failed decoding image {path}: {exc}\") from exc\\n\\n\\ndef read_rgb_decimated(path: str, min_side: int) -> np.ndarray:\\n    suffix = Path(path).suffix.lower()\\n    if suffix in (\".jpg\", \".jpeg\"):\\n        short_side = 0\\n        try:\\n            with Image.open(path) as image:\\n                short_side = min(image.size)\\n        except Exception:\\n            pass\\n        reduction = 1\\n        while short_side and short_side // (reduction * 2) >= min_side and reduction < 8:\\n            reduction *= 2\\n        image = cv2.imread(path, _REDUCED_FLAGS[reduction])\\n        if image is not None:\\n            return cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\\n    return read_rgb_image(path)\\n\\n\\ndef crop_fundus_field(image: np.ndarray, preview: int) -> np.ndarray:\\n    height, width = image.shape[:2]\\n    scale = min(1.0, float(preview) / max(height, width))\\n    if scale < 1:\\n        small = cv2.resize(\\n            image,\\n            (max(1, int(width * scale)), max(1, int(height * scale))),\\n            interpolation=cv2.INTER_AREA,\\n        )\\n    else:\\n        small = image\\n        scale = 1.0\\n    gray = cv2.cvtColor(small, cv2.COLOR_RGB2GRAY)\\n    gray = cv2.GaussianBlur(gray, (5, 5), 0)\\n    mask = gray > 7\\n    if mask.mean() < 0.05:\\n        return image\\n    ys, xs = np.where(mask)\\n    if not len(ys) or not len(xs):\\n        return image\\n    y0, y1 = int(ys.min() / scale), int((ys.max() + 1) / scale)\\n    x0, x1 = int(xs.min() / scale), int((xs.max() + 1) / scale)\\n    margin = int(max(y1 - y0, x1 - x0) * 0.02)\\n    return image[\\n        max(0, y0 - margin) : min(height, y1 + margin),\\n        max(0, x0 - margin) : min(width, x1 + margin),\\n    ]\\n\\n\\n_LID_CONTEXT = None\\n\\n\\ndef use_lid_downscaling(cfg: dict) -> bool:\\n    method = str(cfg.get(\"downscale_method\", \"opencv_area\")).lower()\\n    return method in {\"lid\", \"car_lid\", \"car/lid\", \"car\"}\\n\\n\\ndef prepare_lid_downscaler(cfg: dict):\\n    global _LID_CONTEXT\\n    if _LID_CONTEXT is not None:\\n        return _LID_CONTEXT\\n\\n    try:\\n        import sys\\n\\n        import torch\\n    except ImportError as exc:\\n        raise RuntimeError(\"PyTorch is required for CAR/LID downscaling.\") from exc\\n\\n    if not torch.cuda.is_available():\\n        raise RuntimeError(\"CAR/LID downscaling requires a CUDA-enabled Kaggle GPU.\")\\n\\n    car_repo = Path(cfg.get(\"car_repo_dir\", \"/kaggle/working/CAR\"))\\n    weight_path = Path(\\n        cfg.get(\\n            \"lid_weight_path\",\\n            \"/kaggle/input/datasets/khangcancode/car-trained-model-from-github/models/4x/kgn.pth\",\\n        )\\n    )\\n    if not car_repo.exists():\\n        raise FileNotFoundError(\\n            f\"CAR repository is missing at {car_repo}. Run the notebook CAR setup cell first.\"\\n        )\\n    if not weight_path.exists():\\n        raise FileNotFoundError(\\n            f\"CAR/LID KGN weight is missing at {weight_path}. Attach the Kaggle dataset \"\\n            \"\\'khangcancode/car-trained-model-from-github\\'.\"\\n        )\\n\\n    if str(car_repo) not in sys.path:\\n        sys.path.insert(0, str(car_repo))\\n\\n    from adaptive_gridsampler.gridsampler import Downsampler\\n    from modules import DSN\\n\\n    scale = int(cfg.get(\"lid_scale\", 4))\\n    kernel_size = int(cfg.get(\"lid_kernel_size\", 3 * scale + 1))\\n    device = torch.device(\"cuda:0\")\\n    kgn = DSN(k_size=kernel_size, scale=scale).to(device).eval()\\n    try:\\n        state = torch.load(weight_path, map_location=device, weights_only=True)\\n    except TypeError:\\n        state = torch.load(weight_path, map_location=device)\\n    state = {str(key).removeprefix(\"module.\"): value for key, value in state.items()}\\n    kgn.load_state_dict(state, strict=True)\\n    sampler = Downsampler(scale, kernel_size).to(device).eval()\\n    _LID_CONTEXT = {\\n        \"torch\": torch,\\n        \"device\": device,\\n        \"scale\": scale,\\n        \"kernel_size\": kernel_size,\\n        \"kgn\": kgn,\\n        \"sampler\": sampler,\\n    }\\n    return _LID_CONTEXT\\n\\n\\ndef release_lid_downscaler() -> None:\\n    global _LID_CONTEXT\\n    context = _LID_CONTEXT\\n    _LID_CONTEXT = None\\n    if context is not None:\\n        torch = context[\"torch\"]\\n        if torch.cuda.is_available():\\n            torch.cuda.empty_cache()\\n\\n\\ndef lid_downscale_rgb(rgb_image: np.ndarray, cfg: dict) -> np.ndarray:\\n    context = prepare_lid_downscaler(cfg)\\n    torch = context[\"torch\"]\\n    scale = int(context[\"scale\"])\\n    image = np.ascontiguousarray(rgb_image)\\n    height, width = image.shape[:2]\\n    trim_unit = max(8, scale)\\n    image = image[: height // trim_unit * trim_unit, : width // trim_unit * trim_unit]\\n    with torch.inference_mode():\\n        tensor = (\\n            torch.from_numpy(image)\\n            .permute(2, 0, 1)\\n            .float()\\n            .div_(255.0)\\n            .unsqueeze(0)\\n            .to(context[\"device\"])\\n        )\\n        kernels, offset_h, offset_v = context[\"kgn\"](tensor)\\n        output = context[\"sampler\"](tensor, kernels, offset_h, offset_v, scale)\\n        output = output.clamp_(0.0, 1.0)\\n    return (\\n        output[0]\\n        .permute(1, 2, 0)\\n        .mul(255.0)\\n        .byte()\\n        .cpu()\\n        .numpy()\\n    )\\n\\n\\ndef resize_square_for_training(square: np.ndarray, cfg: dict) -> np.ndarray:\\n    size = int(cfg[\"image_size\"])\\n    if not use_lid_downscaling(cfg):\\n        return cv2.resize(square, (size, size), interpolation=cv2.INTER_AREA)\\n\\n    input_size = int(cfg.get(\"lid_input_size\", size * int(cfg.get(\"lid_scale\", 4))))\\n    scale = int(cfg.get(\"lid_scale\", 4))\\n    if input_size // scale != size:\\n        raise ValueError(\\n            f\"LID geometry mismatch: lid_input_size={input_size}, \"\\n            f\"lid_scale={scale}, image_size={size}.\"\\n        )\\n    lid_input = cv2.resize(\\n        square,\\n        (input_size, input_size),\\n        interpolation=cv2.INTER_LINEAR,\\n    )\\n    resized = lid_downscale_rgb(lid_input, cfg)\\n    if resized.shape[:2] != (size, size):\\n        raise RuntimeError(\\n            f\"CAR/LID produced shape {resized.shape}, expected {(size, size, 3)}.\"\\n        )\\n    return resized\\n\\n\\ndef preprocess_image(image: np.ndarray, cfg: dict) -> np.ndarray:\\n    profile_name, profile = preprocessing_profile_config(cfg)\\n    size = int(cfg[\"image_size\"])\\n    cropped = crop_fundus_field(image, int(cfg.get(\"crop_preview_size\", 512)))\\n    height, width = cropped.shape[:2]\\n    square_size = max(height, width)\\n    square = np.zeros((square_size, square_size, 3), dtype=np.uint8)\\n    y0 = (square_size - height) // 2\\n    x0 = (square_size - width) // 2\\n    square[y0 : y0 + height, x0 : x0 + width] = cropped\\n    resized = resize_square_for_training(square, cfg)\\n\\n    if not profile[\"ben_graham\"] and not profile[\"clahe\"]:\\n        return resized\\n\\n    gray = cv2.cvtColor(resized, cv2.COLOR_RGB2GRAY)\\n    mask = (gray > 8).astype(np.uint8)\\n    count, labels, stats, _ = cv2.connectedComponentsWithStats(mask, 8)\\n    if count > 1:\\n        largest = 1 + int(np.argmax(stats[1:, cv2.CC_STAT_AREA]))\\n        mask = (labels == largest).astype(np.uint8)\\n    mask = cv2.morphologyEx(mask, cv2.MORPH_CLOSE, np.ones((5, 5), np.uint8)).astype(bool)\\n\\n    enhanced = resized.copy()\\n    if profile[\"ben_graham\"]:\\n        sigma = cfg.get(\"ben_graham_sigma\") or max(3, size // 32)\\n        blurred = cv2.GaussianBlur(enhanced, (0, 0), int(sigma))\\n        enhanced = cv2.addWeighted(enhanced, 4.0, blurred, -4.0, 128.0)\\n        enhanced = np.clip(enhanced, 0, 255).astype(np.uint8)\\n\\n    if profile[\"mask_background\"]:\\n        enhanced[~mask] = 0\\n\\n    if profile[\"clahe\"]:\\n        lab = cv2.cvtColor(enhanced, cv2.COLOR_RGB2LAB)\\n        lightness, a, b = cv2.split(lab)\\n        tile = int(cfg.get(\"clahe_tile_grid\", 8))\\n        clahe = cv2.createCLAHE(\\n            clipLimit=float(cfg.get(\"clahe_clip_limit\", 2.0)),\\n            tileGridSize=(tile, tile),\\n        )\\n        enhanced = cv2.cvtColor(\\n            cv2.merge((clahe.apply(lightness), a, b)),\\n            cv2.COLOR_LAB2RGB,\\n        )\\n\\n    if profile[\"blend\"]:\\n        blend = float(cfg.get(\"preprocess_blend\", 0.40))\\n        output = np.clip(\\n            (1.0 - blend) * resized.astype(np.float32)\\n            + blend * enhanced.astype(np.float32),\\n            0,\\n            255,\\n        ).astype(np.uint8)\\n    else:\\n        output = enhanced.astype(np.uint8)\\n\\n    if profile[\"mask_background\"]:\\n        output[~mask] = 0\\n    return output\\n\\ndef phash64(image: np.ndarray) -> str:\\n    gray = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)\\n    small = cv2.resize(gray, (32, 32), interpolation=cv2.INTER_AREA).astype(np.float32)\\n    dct = cv2.dct(small)[:8, :8]\\n    median = float(np.median(dct.flatten()[1:]))\\n    bits = (dct > median).flatten()\\n    value = 0\\n    for bit in bits:\\n        value = (value << 1) | int(bit)\\n    return f\"{value:016x}\"\\n\\n\\ndef manifest_digest(frame: pd.DataFrame) -> str:\\n    columns = [\"path\", \"label\", \"source\", \"stem\", \"file_size\", \"file_mtime_ns\"]\\n    payload = frame[columns].sort_values(\"path\").to_csv(index=False).encode(\"utf-8\")\\n    return hashlib.sha256(payload).hexdigest()\\n\\n\\ndef preprocessing_fingerprint(frame: pd.DataFrame, cfg: dict) -> tuple[str, dict]:\\n    relevant = {\\n        key: cfg.get(key)\\n        for key in [\\n            \"image_size\",\\n            \"preprocessing_profile\",\\n            \"cache_jpeg_quality\",\\n            \"cache_decode_min_side\",\\n            \"crop_preview_size\",\\n            \"preprocess_blend\",\\n            \"ben_graham_sigma\",\\n            \"clahe_clip_limit\",\\n            \"clahe_tile_grid\",\\n            \"downscale_method\",\\n            \"lid_input_size\",\\n            \"lid_scale\",\\n            \"lid_weight_path\",\\n        ]\\n    }\\n    code_version = PREPROCESS_CODE_VERSION\\n    if use_lid_downscaling(cfg):\\n        code_version = f\"{code_version}+car-lid-2026-06-16-v1\"\\n    profile_name, profile = preprocessing_profile_config(cfg)\\n    metadata = {\\n        \"preprocess_code_version\": code_version,\\n        \"preprocessing_profile\": profile_name,\\n        \"preprocessing_profile_config\": profile,\\n        \"preprocess_config\": relevant,\\n        \"source_manifest_sha256\": manifest_digest(frame),\\n    }\\n    encoded = json.dumps(metadata, sort_keys=True).encode(\"utf-8\")\\n    return hashlib.sha256(encoded).hexdigest(), metadata\\n\\n\\ndef cache_worker(task: tuple) -> tuple:\\n    index, source_path, cache_path, cfg = task\\n    try:\\n        image = read_rgb_decimated(source_path, int(cfg.get(\"cache_decode_min_side\", 768)))\\n        processed = preprocess_image(image, cfg)\\n        pixel_sha = hashlib.sha256(np.ascontiguousarray(processed).tobytes()).hexdigest()\\n        perceptual_hash = phash64(processed)\\n        destination = Path(cache_path)\\n        destination.parent.mkdir(parents=True, exist_ok=True)\\n        cv2.imwrite(\\n            str(destination),\\n            cv2.cvtColor(processed, cv2.COLOR_RGB2BGR),\\n            [\\n                int(cv2.IMWRITE_JPEG_QUALITY),\\n                int(cfg.get(\"cache_jpeg_quality\", 95)),\\n                int(cv2.IMWRITE_JPEG_OPTIMIZE),\\n                1,\\n            ],\\n        )\\n        return index, str(destination), pixel_sha, perceptual_hash, True, \"written\"\\n    except Exception as exc:\\n        return index, str(cache_path), \"\", \"\", False, f\"{type(exc).__name__}: {exc}\"\\n\\n\\ndef build_cache(frame: pd.DataFrame, cfg: dict, output_dir: Path):\\n    fingerprint, metadata = preprocessing_fingerprint(frame, cfg)\\n    cache_root = Path(cfg[\"cache_dir\"]) / fingerprint[:16]\\n    cache_root.mkdir(parents=True, exist_ok=True)\\n    metadata = {\\n        **metadata,\\n        \"preprocessing_fingerprint\": fingerprint,\\n        \"resolved_cache_dir\": str(cache_root),\\n    }\\n    with open(cache_root / \"cache_metadata.json\", \"w\") as handle:\\n        json.dump(metadata, handle, indent=2)\\n    with open(output_dir / \"cache_metadata.json\", \"w\") as handle:\\n        json.dump(metadata, handle, indent=2)\\n\\n    cached = frame.copy().reset_index(drop=True)\\n    cached[\"cache_path\"] = [\\n        str(\\n            cache_root\\n            / str(int(row.label))\\n            / (\\n                hashlib.sha1(str(row.path).encode(\"utf-8\")).hexdigest()[:16]\\n                + \"_\"\\n                + Path(row.path).stem\\n                + \".jpg\"\\n            )\\n        )\\n        for row in cached.itertuples(index=False)\\n    ]\\n    cache_manifest_path = cache_root / \"cache_manifest.csv\"\\n    existing = pd.read_csv(cache_manifest_path) if cache_manifest_path.exists() else pd.DataFrame()\\n    existing_by_path = (\\n        existing.set_index(\"path\").to_dict(\"index\")\\n        if len(existing) and \"path\" in existing.columns\\n        else {}\\n    )\\n\\n    cached[\"processed_sha256\"] = \"\"\\n    cached[\"phash64\"] = \"\"\\n    cached[\"cache_ok\"] = False\\n    cached[\"cache_status\"] = \"pending\"\\n    tasks = []\\n    for index, row in cached.iterrows():\\n        prior = existing_by_path.get(str(row[\"path\"]))\\n        if (\\n            prior\\n            and Path(str(prior.get(\"cache_path\", \"\"))).exists()\\n            and prior.get(\"processed_sha256\")\\n            and prior.get(\"phash64\")\\n        ):\\n            cached.at[index, \"cache_path\"] = str(prior[\"cache_path\"])\\n            cached.at[index, \"processed_sha256\"] = str(prior[\"processed_sha256\"])\\n            cached.at[index, \"phash64\"] = str(prior[\"phash64\"])\\n            cached.at[index, \"cache_ok\"] = True\\n            cached.at[index, \"cache_status\"] = \"exists\"\\n        else:\\n            tasks.append((index, row[\"path\"], row[\"cache_path\"], dict(cfg)))\\n\\n    if tasks:\\n        if use_lid_downscaling(cfg):\\n            prepare_lid_downscaler(cfg)\\n            results = (\\n                cache_worker(task)\\n                for task in tqdm(tasks, total=len(tasks), desc=\"Caching CAR/LID images\")\\n            )\\n            for index, cache_path, pixel_sha, perceptual_hash, ok, status in results:\\n                cached.at[index, \"cache_path\"] = cache_path\\n                cached.at[index, \"processed_sha256\"] = pixel_sha\\n                cached.at[index, \"phash64\"] = perceptual_hash\\n                cached.at[index, \"cache_ok\"] = bool(ok)\\n                cached.at[index, \"cache_status\"] = status\\n            release_lid_downscaler()\\n        else:\\n            import multiprocessing\\n\\n            workers = max(1, min(int(cfg.get(\"cache_workers\", 4)), effective_cpu_count()))\\n            context = multiprocessing.get_context(\"fork\")\\n            chunk = max(1, len(tasks) // (workers * 8))\\n            with futures.ProcessPoolExecutor(max_workers=workers, mp_context=context) as executor:\\n                results = tqdm(\\n                    executor.map(cache_worker, tasks, chunksize=chunk),\\n                    total=len(tasks),\\n                    desc=f\"Caching corrected images [process x{workers}]\",\\n                )\\n                for index, cache_path, pixel_sha, perceptual_hash, ok, status in results:\\n                    cached.at[index, \"cache_path\"] = cache_path\\n                    cached.at[index, \"processed_sha256\"] = pixel_sha\\n                    cached.at[index, \"phash64\"] = perceptual_hash\\n                    cached.at[index, \"cache_ok\"] = bool(ok)\\n                    cached.at[index, \"cache_status\"] = status\\n\\n    cached = cached[cached[\"cache_ok\"]].reset_index(drop=True)\\n    cached.to_csv(cache_manifest_path, index=False)\\n    return cached, metadata\\n\\n\\ndef effective_cpu_count() -> int:\\n    try:\\n        return max(1, len(os.sched_getaffinity(0)))\\n    except AttributeError:\\n        return max(1, os.cpu_count() or 4)\\n\\n\\ndef comparison_thumbnail(path: str) -> np.ndarray:\\n    image = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\\n    if image is None:\\n        raise RuntimeError(f\"Cannot read cached comparison image: {path}\")\\n    image = cv2.resize(image, (64, 64), interpolation=cv2.INTER_AREA).astype(np.float32) / 255.0\\n    return image\\n\\n\\ndef confirm_near_duplicate(path_a: str, path_b: str) -> tuple[bool, float, float]:\\n    first = comparison_thumbnail(path_a)\\n    second = comparison_thumbnail(path_b)\\n    mae = float(np.mean(np.abs(first - second)))\\n    first_centered = first - first.mean()\\n    second_centered = second - second.mean()\\n    denominator = float(np.linalg.norm(first_centered) * np.linalg.norm(second_centered))\\n    correlation = (\\n        float(np.sum(first_centered * second_centered) / denominator)\\n        if denominator > 1e-12\\n        else 0.0\\n    )\\n    return correlation >= 0.998 and mae <= 0.030, correlation, mae\\n\\n\\ndef audit_duplicates(\\n    cached: pd.DataFrame,\\n    filename_audit: pd.DataFrame,\\n) -> tuple[pd.DataFrame, pd.DataFrame]:\\n    frame = cached.copy().reset_index(drop=True)\\n    image_union = UnionFind(frame.index.tolist())\\n    audit_rows = filename_audit.to_dict(\"records\") if len(filename_audit) else []\\n\\n    for pixel_sha, group in frame.groupby(\"processed_sha256\", sort=False):\\n        if not pixel_sha or len(group) < 2:\\n            continue\\n        representative = int(group.index[0])\\n        for index in group.index[1:]:\\n            index = int(index)\\n            image_union.union(representative, index)\\n            audit_rows.append(\\n                {\\n                    \"duplicate_type\": \"exact_processed_sha256\",\\n                    \"left_index\": representative,\\n                    \"right_index\": index,\\n                    \"left_path\": frame.at[representative, \"path\"],\\n                    \"right_path\": frame.at[index, \"path\"],\\n                    \"left_label\": int(frame.at[representative, \"label\"]),\\n                    \"right_label\": int(frame.at[index, \"label\"]),\\n                    \"confirmed\": True,\\n                    \"phash_distance\": 0,\\n                    \"correlation\": 1.0,\\n                    \"mae\": 0.0,\\n                }\\n            )\\n\\n    hash_groups = {\\n        int(hash_value, 16): [int(index) for index in group.index]\\n        for hash_value, group in frame.groupby(\"phash64\", sort=False)\\n        if isinstance(hash_value, str) and hash_value\\n    }\\n    tree = BKTree()\\n    for hash_value in hash_groups:\\n        tree.add(hash_value)\\n\\n    compared_pairs = set()\\n    for hash_value, indices in tqdm(hash_groups.items(), desc=\"Auditing near duplicates\"):\\n        neighbors = tree.query(hash_value, radius=4)\\n        for neighbor in neighbors:\\n            if neighbor < hash_value:\\n                continue\\n            left_indices = indices\\n            right_indices = hash_groups[neighbor]\\n            for left in left_indices:\\n                for right in right_indices:\\n                    if left >= right:\\n                        continue\\n                    pair = (left, right)\\n                    if pair in compared_pairs:\\n                        continue\\n                    compared_pairs.add(pair)\\n                    if frame.at[left, \"processed_sha256\"] == frame.at[right, \"processed_sha256\"]:\\n                        continue\\n                    confirmed, correlation, mae = confirm_near_duplicate(\\n                        frame.at[left, \"cache_path\"],\\n                        frame.at[right, \"cache_path\"],\\n                    )\\n                    if not confirmed:\\n                        continue\\n                    image_union.union(left, right)\\n                    audit_rows.append(\\n                        {\\n                            \"duplicate_type\": \"near_phash_confirmed\",\\n                            \"left_index\": left,\\n                            \"right_index\": right,\\n                            \"left_path\": frame.at[left, \"path\"],\\n                            \"right_path\": frame.at[right, \"path\"],\\n                            \"left_label\": int(frame.at[left, \"label\"]),\\n                            \"right_label\": int(frame.at[right, \"label\"]),\\n                            \"confirmed\": True,\\n                            \"phash_distance\": BKTree.distance(hash_value, neighbor),\\n                            \"correlation\": correlation,\\n                            \"mae\": mae,\\n                        }\\n                    )\\n\\n    frame[\"duplicate_cluster\"] = [image_union.find(index) for index in frame.index]\\n    patient_union = UnionFind(frame[\"patient_key\"].astype(str).unique().tolist())\\n    keep = np.ones(len(frame), dtype=bool)\\n    quarantined = np.zeros(len(frame), dtype=bool)\\n\\n    for cluster, group in frame.groupby(\"duplicate_cluster\", sort=False):\\n        labels = sorted(group[\"label\"].astype(int).unique().tolist())\\n        patients = group[\"patient_key\"].astype(str).unique().tolist()\\n        for patient in patients[1:]:\\n            patient_union.union(patients[0], patient)\\n        if len(labels) > 1:\\n            quarantined[group.index] = True\\n            keep[group.index] = False\\n            action = \"quarantine_conflicting_labels\"\\n        else:\\n            ordered = group.sort_values([\"is_gf\", \"path\"], kind=\"stable\")\\n            keep[ordered.index[1:]] = False\\n            action = \"keep_one_same_label\"\\n        for row in audit_rows:\\n            if row.get(\"duplicate_type\") == \"filename_gf\":\\n                continue\\n            if row.get(\"left_index\") in group.index or row.get(\"right_index\") in group.index:\\n                row[\"duplicate_cluster\"] = int(cluster)\\n                row[\"cluster_key\"] = f\"processed:{int(cluster)}\"\\n                row[\"cluster_labels\"] = labels\\n                row[\"action\"] = action\\n\\n    components = {}\\n    for patient in patient_union.parent:\\n        components.setdefault(patient_union.find(patient), []).append(patient)\\n    canonical = {\\n        patient: min(members)\\n        for members in components.values()\\n        for patient in members\\n    }\\n    frame[\"group_key\"] = frame[\"patient_key\"].astype(str).map(canonical)\\n    frame[\"quarantined\"] = quarantined\\n    frame[\"dedup_keep\"] = keep\\n    active = frame[keep].copy().reset_index(drop=True)\\n    return active, pd.DataFrame(audit_rows)\\n\\n\\ndef create_splits(frame: pd.DataFrame, cfg: dict) -> tuple[pd.DataFrame, dict]:\\n    strategy = str(cfg.get(\"split_strategy\", \"stratified_group_10fold\"))\\n    if strategy != \"stratified_group_10fold\":\\n        raise ValueError(\\n            f\"Unknown split_strategy={strategy!r}; expected \"\\n            \"\\'stratified_group_10fold\\'.\"\\n        )\\n    splitter = StratifiedGroupKFold(\\n        n_splits=10,\\n        shuffle=True,\\n        random_state=int(cfg[\"seed\"]),\\n    )\\n    folds = np.full(len(frame), -1, dtype=np.int64)\\n    dummy = np.zeros(len(frame), dtype=np.int8)\\n    for fold, (_, validation_indices) in enumerate(\\n        splitter.split(dummy, frame[\"label\"].to_numpy(), frame[\"group_key\"].to_numpy())\\n    ):\\n        folds[validation_indices] = fold\\n    if (folds < 0).any():\\n        raise AssertionError(\"Not every image received a stratified group fold.\")\\n\\n    result = frame.copy()\\n    result[\"fold\"] = folds\\n    result[\"partition\"] = \"train\"\\n    result.loc[result[\"fold\"] == 8, \"partition\"] = \"model_valid\"\\n    result.loc[result[\"fold\"] == 9, \"partition\"] = \"holdout\"\\n    result = split_val_partition_for_calibration(result, cfg)\\n\\n    partitions = [\"train\", \"model_valid\", \"calibration\", \"holdout\"]\\n    expected_classes = set(range(int(cfg[\"num_classes\"])))\\n    for partition in partitions:\\n        observed = set(\\n            result.loc[result[\"partition\"] == partition, \"label\"].astype(int).unique()\\n        )\\n        if observed != expected_classes:\\n            raise AssertionError(\\n                f\"Partition {partition} has classes {sorted(observed)}, \"\\n                f\"expected {sorted(expected_classes)}.\"\\n            )\\n\\n    for column in [\"group_key\", \"processed_sha256\", \"duplicate_cluster\"]:\\n        sets = {\\n            partition: set(\\n                result.loc[result[\"partition\"] == partition, column].astype(str)\\n            )\\n            for partition in partitions\\n        }\\n        for index, first in enumerate(partitions):\\n            for second in partitions[index + 1 :]:\\n                overlap = sets[first] & sets[second]\\n                if overlap:\\n                    raise AssertionError(\\n                        f\"{column} leakage between {first} and {second}: {len(overlap)}\"\\n                    )\\n\\n    result[\"orig_path\"] = result[\"path\"]\\n    result[\"path\"] = result[\"cache_path\"]\\n    split_frames = {\\n        partition: result[result[\"partition\"] == partition].reset_index(drop=True)\\n        for partition in partitions\\n    }\\n    return result, split_frames\\n\\n\\ndef write_data_audit(\\n    raw: pd.DataFrame,\\n    filename_deduped: pd.DataFrame,\\n    active: pd.DataFrame,\\n    split_manifest: pd.DataFrame,\\n    duplicate_audit: pd.DataFrame,\\n    output_dir: Path,\\n) -> None:\\n    conflicting = duplicate_audit[\\n        duplicate_audit.get(\"action\", pd.Series(index=duplicate_audit.index, dtype=str))\\n        == \"quarantine_conflicting_labels\"\\n    ]\\n    conflicting_clusters = (\\n        int(conflicting[\"cluster_key\"].dropna().astype(str).nunique())\\n        if \"cluster_key\" in conflicting.columns\\n        else 0\\n    )\\n    rows = [\\n        {\"section\": \"summary\", \"key\": \"raw_images\", \"value\": len(raw)},\\n        {\\n            \"section\": \"summary\",\\n            \"key\": \"after_filename_dedup\",\\n            \"value\": len(filename_deduped),\\n        },\\n        {\"section\": \"summary\", \"key\": \"after_duplicate_audit\", \"value\": len(active)},\\n        {\\n            \"section\": \"summary\",\\n            \"key\": \"quarantined_conflict_clusters\",\\n            \"value\": conflicting_clusters,\\n        },\\n    ]\\n    for (partition, label), count in (\\n        split_manifest.groupby([\"partition\", \"label\"]).size().items()\\n    ):\\n        rows.append(\\n            {\\n                \"section\": \"partition_class\",\\n                \"key\": f\"{partition}:class_{label}\",\\n                \"value\": int(count),\\n            }\\n        )\\n    for (source, label), count in active.groupby([\"source\", \"label\"]).size().items():\\n        rows.append(\\n            {\\n                \"section\": \"source_class\",\\n                \"key\": f\"{source}:class_{label}\",\\n                \"value\": int(count),\\n            }\\n        )\\n    pd.DataFrame(rows).to_csv(output_dir / \"data_audit.csv\", index=False)\\n\\n\\ndef _bool_series(series: pd.Series) -> pd.Series:\\n    if series.dtype == bool:\\n        return series.fillna(False)\\n    return series.fillna(False).astype(str).str.lower().isin({\"1\", \"true\", \"yes\", \"y\"})\\n\\n\\ndef corrected_manifest_configured(cfg: dict) -> bool:\\n    return bool(cfg.get(\"corrected_manifest_path\") or cfg.get(\"build_corrected_manifest\"))\\n\\n\\ndef cleaned_label_manifest_configured(cfg: dict) -> bool:\\n    return bool(\\n        cfg.get(\"labels_csv_path\")\\n        or cfg.get(\"labels_path\")\\n        or cfg.get(\"label_file\")\\n        or cfg.get(\"images_dir\")\\n        or cfg.get(\"cleaned_dataset_root\")\\n    )\\n\\n\\ndef _configured_path(cfg: dict, keys: tuple[str, ...]) -> Path | None:\\n    for key in keys:\\n        value = cfg.get(key)\\n        if value:\\n            return Path(str(value))\\n    return None\\n\\n\\ndef _dedupe_paths(paths: Iterable[Path]) -> list[Path]:\\n    result = []\\n    seen = set()\\n    for path in paths:\\n        path = Path(path)\\n        key = str(path)\\n        if key in seen:\\n            continue\\n        seen.add(key)\\n        result.append(path)\\n    return result\\n\\n\\ndef resolve_cleaned_dataset_paths(cfg: dict) -> tuple[Path, Path, Path]:\\n    configured_labels = _configured_path(\\n        cfg,\\n        (\"labels_csv_path\", \"labels_path\", \"label_file\"),\\n    )\\n    configured_images = _configured_path(cfg, (\"images_dir\", \"image_dir\"))\\n\\n    root_candidates = []\\n    for key in (\"cleaned_dataset_root\", \"data_root\", \"dataset_root\"):\\n        value = cfg.get(key)\\n        if not value:\\n            continue\\n        root = Path(str(value))\\n        root_candidates.extend([root, root / \"cleaned_dr_dataset_complete\"])\\n    if configured_labels is not None:\\n        root_candidates.append(configured_labels.parent)\\n    if configured_images is not None:\\n        root_candidates.append(configured_images.parent)\\n    root_candidates.extend([\\n        Path.cwd() / \"cleaned_dr_dataset_complete\",\\n        Path(\"cleaned_dr_dataset_complete\"),\\n    ])\\n\\n    checked = []\\n    for root in _dedupe_paths(root_candidates):\\n        label_candidates = []\\n        if configured_labels is not None and configured_labels.is_file():\\n            label_candidates.append(configured_labels)\\n        label_candidates.append(root / \"labels.csv\")\\n\\n        image_candidates = []\\n        if configured_images is not None and configured_images.is_dir():\\n            image_candidates.append(configured_images)\\n        image_candidates.append(root / \"images\")\\n\\n        for labels_path in _dedupe_paths(label_candidates):\\n            for images_dir in _dedupe_paths(image_candidates):\\n                checked.append(f\"root={root} labels={labels_path} images={images_dir}\")\\n                if labels_path.is_file() and images_dir.is_dir():\\n                    return root, labels_path, images_dir\\n\\n    preview = \"\\\\n  \".join(checked[:30])\\n    raise FileNotFoundError(\\n        \"Could not locate cleaned_dr_dataset_complete labels.csv and images/. Checked:\\\\n  \"\\n        + preview\\n    )\\n\\n\\ndef _first_column(frame: pd.DataFrame, candidates: tuple[str, ...], purpose: str) -> str:\\n    lowered = {str(column).lower(): column for column in frame.columns}\\n    for candidate in candidates:\\n        if candidate.lower() in lowered:\\n            return lowered[candidate.lower()]\\n    raise ValueError(\\n        f\"labels.csv is missing a {purpose} column. Tried: {list(candidates)}; \"\\n        f\"available columns: {list(frame.columns)}\"\\n    )\\n\\n\\ndef _optional_column(frame: pd.DataFrame, candidates: tuple[str, ...]) -> str | None:\\n    lowered = {str(column).lower(): column for column in frame.columns}\\n    for candidate in candidates:\\n        if candidate.lower() in lowered:\\n            return lowered[candidate.lower()]\\n    return None\\n\\n\\ndef _manifest_value(row: pd.Series, column: str | None, default: str) -> str:\\n    if column is None:\\n        return default\\n    value = row.get(column, default)\\n    if pd.isna(value):\\n        return default\\n    text = str(value).strip()\\n    return text if text else default\\n\\n\\ndef resolve_manifest_image_path(file_name: str, root: Path, images_dir: Path) -> Path:\\n    raw = Path(str(file_name))\\n    if raw.is_absolute():\\n        candidates = [raw]\\n    else:\\n        candidates = [root / raw, images_dir / raw, images_dir / raw.name]\\n        if raw.suffix == \"\":\\n            extension_candidates = []\\n            for candidate in candidates:\\n                extension_candidates.extend(\\n                    candidate.with_suffix(extension)\\n                    for extension in sorted(IMAGE_EXTENSIONS)\\n                )\\n            candidates.extend(extension_candidates)\\n    for candidate in _dedupe_paths(candidates):\\n        if candidate.is_file():\\n            return candidate\\n    return candidates[0]\\n\\n\\ndef load_cleaned_label_manifest(cfg: dict) -> tuple[pd.DataFrame, dict]:\\n    root, labels_path, images_dir = resolve_cleaned_dataset_paths(cfg)\\n    manifest = pd.read_csv(labels_path)\\n    if manifest.empty:\\n        raise RuntimeError(f\"labels.csv is empty: {labels_path}\")\\n\\n    file_col = _first_column(\\n        manifest,\\n        (\"file_name\", \"filename\", \"path\", \"image_path\", \"image\", \"id_code\"),\\n        \"file name/path\",\\n    )\\n    label_col = _first_column(\\n        manifest,\\n        (\"dr_grade\", \"diagnosis\", \"label\", \"level\", \"class\", \"grade\", \"corrected_label\"),\\n        \"DR grade/label\",\\n    )\\n    source_col = _optional_column(manifest, (\"source_dataset\", \"source\", \"dataset\"))\\n    split_col = _optional_column(manifest, (\"source_split\", \"split\", \"subset\"))\\n    image_id_col = _optional_column(manifest, (\"image_id\", \"id_code\", \"id\"))\\n    patient_col = _optional_column(manifest, (\"patient_id\", \"patient\", \"patient_key\"))\\n\\n    default_source = str(\\n        cfg.get(\"manifest_source_name\")\\n        or cfg.get(\"dataset_variant\")\\n        or cfg.get(\"dataset_slug\")\\n        or \"cleaned_manifest\"\\n    )\\n    rows = []\\n    missing_paths = []\\n    for _, row in manifest.iterrows():\\n        file_name = _manifest_value(row, file_col, \"\")\\n        image_path = resolve_manifest_image_path(file_name, root, images_dir)\\n        if not image_path.is_file():\\n            missing_paths.append(str(image_path))\\n            continue\\n\\n        try:\\n            label = int(row[label_col])\\n        except Exception as exc:\\n            raise ValueError(f\"Invalid label value {row[label_col]!r} for {file_name!r}\") from exc\\n\\n        source = normalize_token(_manifest_value(row, source_col, default_source)) or normalize_token(default_source) or \"cleaned_manifest\"\\n        source_split = _manifest_value(row, split_col, \"\")\\n        image_token = _manifest_value(row, image_id_col, image_path.stem)\\n        image_stem = Path(str(image_token)).stem or image_path.stem\\n        patient_token = _manifest_value(row, patient_col, \"\")\\n        patient_key = (\\n            f\"{source}:{patient_token}\"\\n            if patient_token\\n            else extract_patient_key(image_path.stem, source)\\n        )\\n        patient_key_method = \"explicit_patient_id\" if patient_token else \"filename_derived\"\\n        stat = image_path.stat()\\n        rows.append(\\n            {\\n                \"path\": str(image_path),\\n                \"label\": label,\\n                \"source\": source,\\n                \"source_split\": source_split,\\n                \"stem\": image_path.stem,\\n                \"ext\": image_path.suffix.lower(),\\n                \"file_size\": int(stat.st_size),\\n                \"file_mtime_ns\": int(stat.st_mtime_ns),\\n                \"is_gf\": bool(GF_SUFFIX_RE.search(image_path.stem)),\\n                \"image_id\": extract_image_id(image_stem, source),\\n                \"patient_key\": patient_key,\\n                \"patient_key_method\": patient_key_method,\\n                \"manifest_file_name\": file_name,\\n            }\\n        )\\n\\n    if missing_paths:\\n        preview = \"\\\\n  \".join(missing_paths[:20])\\n        raise FileNotFoundError(\\n            f\"labels.csv references {len(missing_paths)} missing image files. First missing paths:\\\\n  {preview}\"\\n        )\\n    if not rows:\\n        raise RuntimeError(f\"No usable labelled images were loaded from {labels_path}\")\\n\\n    frame = pd.DataFrame(rows).drop_duplicates(\"path\").reset_index(drop=True)\\n    frame = apply_debug_sample(frame, cfg)\\n    expected_classes = set(range(int(cfg.get(\"num_classes\", 5))))\\n    observed_classes = set(frame[\"label\"].astype(int).unique().tolist())\\n    unexpected = sorted(observed_classes - expected_classes)\\n    if unexpected:\\n        raise ValueError(\\n            f\"labels.csv contains labels outside expected classes {sorted(expected_classes)}: {unexpected}\"\\n        )\\n    print(f\"Loaded cleaned labels manifest: {len(frame)} images from {labels_path}\")\\n    print(\"Class counts:\")\\n    print(frame[\"label\"].value_counts().sort_index().to_string())\\n    if \"source\" in frame.columns:\\n        print(\"Source counts:\")\\n        print(frame[\"source\"].value_counts().sort_index().to_string())\\n\\n    metadata = {\\n        \"cleaned_dataset_root\": str(root),\\n        \"labels_csv_path\": str(labels_path),\\n        \"images_dir\": str(images_dir),\\n        \"manifest_rows\": int(len(manifest)),\\n        \"loaded_rows\": int(len(frame)),\\n        \"file_column\": str(file_col),\\n        \"label_column\": str(label_col),\\n    }\\n    return frame, metadata\\n\\n\\ndef manifest_patient_key(filename: str, source: str) -> str:\\n    stem = Path(str(filename)).stem\\n    return extract_patient_key(stem, str(source))\\n\\n\\ndef prepare_frame_from_corrected_manifest(manifest: pd.DataFrame) -> pd.DataFrame:\\n    required = {\\n        \"path\",\\n        \"filename\",\\n        \"split_folder\",\\n        \"folder_label\",\\n        \"corrected_label\",\\n        \"source_guess\",\\n        \"duplicate_cluster_id\",\\n        \"exclude_from_training\",\\n    }\\n    missing = sorted(required - set(manifest.columns))\\n    if missing:\\n        raise ValueError(f\"corrected_manifest.csv is missing required columns: {missing}\")\\n\\n    active = manifest[~_bool_series(manifest[\"exclude_from_training\"])].copy()\\n    if active.empty:\\n        raise RuntimeError(\"All rows are excluded in corrected_manifest.csv.\")\\n    active[\"label\"] = active[\"corrected_label\"].astype(int)\\n    active[\"source\"] = active[\"source_guess\"].fillna(\"unknown_mixed\").astype(str)\\n    active[\"stem\"] = active[\"filename\"].map(lambda value: Path(str(value)).stem)\\n    active[\"ext\"] = active[\"filename\"].map(lambda value: Path(str(value)).suffix.lower())\\n    active[\"is_gf\"] = active[\"stem\"].str.contains(r\"(?i)[-_]gf$\", regex=True)\\n    active[\"image_id\"] = [\\n        extract_image_id(stem, source)\\n        for stem, source in zip(active[\"stem\"], active[\"source\"])\\n    ]\\n    active[\"patient_key\"] = [\\n        manifest_patient_key(filename, source)\\n        for filename, source in zip(active[\"filename\"], active[\"source\"])\\n    ]\\n    cluster = active[\"duplicate_cluster_id\"].fillna(\"\").astype(str)\\n    active[\"group_key\"] = np.where(\\n        cluster != \"\",\\n        \"duplicate:\" + cluster,\\n        active[\"patient_key\"].astype(str),\\n    )\\n    active[\"duplicate_cluster\"] = np.where(cluster != \"\", cluster, active[\"image_id\"])\\n    active[\"file_size\"] = active[\"path\"].map(lambda value: Path(str(value)).stat().st_size)\\n    active[\"file_mtime_ns\"] = active[\"path\"].map(lambda value: Path(str(value)).stat().st_mtime_ns)\\n    return active.reset_index(drop=True)\\n\\n\\ndef _stable_hash_int(value: str, seed: int = 0) -> int:\\n    payload = f\"{int(seed)}|{value}\".encode(\"utf-8\")\\n    return int(hashlib.sha1(payload).hexdigest()[:16], 16)\\n\\n\\ndef split_val_partition_for_calibration(result: pd.DataFrame, cfg: dict) -> pd.DataFrame:\\n    result = result.copy()\\n    if \"split_folder\" in result.columns:\\n        val_mask = result[\"split_folder\"] == \"val\"\\n    else:\\n        val_mask = result[\"partition\"] == \"model_valid\"\\n    val_indices = result.index[val_mask].to_numpy()\\n    if not len(val_indices):\\n        return result\\n\\n    fraction = float(cfg.get(\"calibration_from_val_fraction\", 0.5))\\n    if not (0.0 < fraction < 1.0):\\n        raise ValueError(\"calibration_from_val_fraction must be between 0 and 1.\")\\n\\n    seed = int(cfg.get(\"calibration_split_seed\", int(cfg.get(\"seed\", 42)) + 480))\\n    labels = result.loc[val_indices, \"label\"].astype(int).to_numpy()\\n    groups = result.loc[val_indices, \"group_key\"].astype(str).to_numpy()\\n    result.loc[val_indices, \"partition\"] = \"model_valid\"\\n\\n    n_splits = max(2, int(round(1.0 / fraction)))\\n    label_counts = pd.Series(labels).value_counts()\\n    use_group_kfold = (\\n        label_counts.min() >= n_splits\\n        and len(np.unique(groups)) >= n_splits\\n    )\\n    calibration_indices: np.ndarray\\n    if use_group_kfold:\\n        splitter = StratifiedGroupKFold(\\n            n_splits=n_splits,\\n            shuffle=True,\\n            random_state=seed,\\n        )\\n        dummy = np.zeros(len(labels), dtype=np.int8)\\n        _, fold_indices = next(splitter.split(dummy, labels, groups))\\n        calibration_indices = val_indices[fold_indices]\\n    else:\\n        selected = []\\n        val_frame = result.loc[val_indices, [\"label\", \"group_key\", \"path\"]].copy()\\n        val_frame[\"_index\"] = val_indices\\n        val_frame[\"_score\"] = [\\n            _stable_hash_int(f\"{group}|{path}\", seed)\\n            for group, path in zip(val_frame[\"group_key\"], val_frame[\"path\"])\\n        ]\\n        for _, group in val_frame.groupby(\"label\", sort=False):\\n            count = int(round(len(group) * fraction))\\n            if len(group) > 1:\\n                count = min(max(1, count), len(group) - 1)\\n            else:\\n                count = 0\\n            if count:\\n                selected.extend(\\n                    group.sort_values(\"_score\", kind=\"stable\").head(count)[\"_index\"].tolist()\\n                )\\n        calibration_indices = np.asarray(selected, dtype=np.int64)\\n\\n    result.loc[calibration_indices, \"partition\"] = \"calibration\"\\n\\n    expected_classes = set(range(int(cfg[\"num_classes\"])))\\n    for partition in (\"model_valid\", \"calibration\"):\\n        observed = set(\\n            result.loc[result[\"partition\"] == partition, \"label\"].astype(int).unique()\\n        )\\n        if observed != expected_classes:\\n            raise AssertionError(\\n                f\"Logical {partition!r} split has classes {sorted(observed)}, \"\\n                f\"expected {sorted(expected_classes)}. Adjust calibration_from_val_fraction \"\\n                \"or inspect the corrected manifest exclusions.\"\\n            )\\n    return result\\n\\n\\ndef create_existing_folder_splits(cached: pd.DataFrame, cfg: dict) -> tuple[pd.DataFrame, dict]:\\n    result = cached.copy().reset_index(drop=True)\\n    split_to_partition = {\\n        \"train\": \"train\",\\n        \"val\": \"model_valid\",\\n        \"test\": \"holdout\",\\n    }\\n    result[\"partition\"] = result[\"split_folder\"].map(split_to_partition)\\n    if result[\"partition\"].isna().any():\\n        bad = sorted(result.loc[result[\"partition\"].isna(), \"split_folder\"].astype(str).unique())\\n        raise ValueError(f\"Unknown split_folder values in corrected manifest: {bad}\")\\n    result[\"fold\"] = -1\\n    result[\"orig_path\"] = result[\"path\"]\\n    result[\"path\"] = result[\"cache_path\"]\\n    result = split_val_partition_for_calibration(result, cfg)\\n\\n    expected_classes = set(range(int(cfg[\"num_classes\"])))\\n    for split_folder, partition in split_to_partition.items():\\n        observed = set(\\n            result.loc[result[\"split_folder\"] == split_folder, \"label\"].astype(int).unique()\\n        )\\n        if observed != expected_classes:\\n            raise AssertionError(\\n                f\"Split folder {split_folder!r} has classes {sorted(observed)}, \"\\n                f\"expected {sorted(expected_classes)} after correction/exclusion.\"\\n            )\\n\\n    split_frames = {\\n        \"train\": result[result[\"partition\"] == \"train\"].reset_index(drop=True),\\n        \"model_valid\": result[result[\"partition\"] == \"model_valid\"].reset_index(drop=True),\\n        \"calibration\": result[result[\"partition\"] == \"calibration\"].reset_index(drop=True),\\n        \"val\": result[result[\"split_folder\"] == \"val\"].reset_index(drop=True),\\n        \"holdout\": result[result[\"partition\"] == \"holdout\"].reset_index(drop=True),\\n        \"test\": result[result[\"partition\"] == \"holdout\"].reset_index(drop=True),\\n    }\\n    return result, split_frames\\n\\n\\n\\n\\ndef stable_row_score(*parts: object, seed: int = 0) -> int:\\n    payload = \"|\".join(str(part) for part in (seed, *parts)).encode(\"utf-8\")\\n    return int(hashlib.sha1(payload).hexdigest()[:16], 16)\\n\\n\\ndef apply_debug_sample(frame: pd.DataFrame, cfg: dict) -> pd.DataFrame:\\n    sample_size = int(cfg.get(\"debug_sample_size\", 0) or 0)\\n    if sample_size <= 0 or sample_size >= len(frame):\\n        return frame.reset_index(drop=True)\\n    expected_classes = sorted(frame[\"label\"].astype(int).unique().tolist())\\n    if sample_size < len(expected_classes):\\n        raise ValueError(\"debug_sample_size must be 0 or at least the number of observed classes.\")\\n    seed = int(cfg.get(\"seed\", 42)) + 1701\\n    scored = frame.copy().reset_index(drop=True)\\n    scored[\"_debug_score\"] = [\\n        stable_row_score(row.source, row.label, row.path, seed=seed)\\n        for row in scored.itertuples(index=False)\\n    ]\\n    per_class_floor = max(1, min(20, sample_size // max(1, len(expected_classes))))\\n    selected_indices: list[int] = []\\n    for label, group in scored.groupby(\"label\", sort=True):\\n        take = min(len(group), per_class_floor)\\n        selected_indices.extend(\\n            group.sort_values(\"_debug_score\", kind=\"stable\").head(take).index.astype(int).tolist()\\n        )\\n    remaining = sample_size - len(set(selected_indices))\\n    if remaining > 0:\\n        rest = scored.loc[~scored.index.isin(selected_indices)]\\n        selected_indices.extend(\\n            rest.sort_values(\"_debug_score\", kind=\"stable\").head(remaining).index.astype(int).tolist()\\n        )\\n    sampled = scored.loc[sorted(set(selected_indices))].drop(columns=[\"_debug_score\"])\\n    print(f\"Debug sample enabled: using {len(sampled)} of {len(frame)} manifest rows.\")\\n    return sampled.reset_index(drop=True)\\n\\n\\ndef write_group_key_audit(split_manifest: pd.DataFrame, output_dir: Path) -> None:\\n    columns = [\\n        column\\n        for column in [\\n            \"orig_path\",\\n            \"manifest_file_name\",\\n            \"source\",\\n            \"source_split\",\\n            \"label\",\\n            \"image_id\",\\n            \"patient_key\",\\n            \"patient_key_method\",\\n            \"group_key\",\\n            \"duplicate_cluster\",\\n            \"quarantined\",\\n            \"dedup_keep\",\\n            \"fold\",\\n            \"partition\",\\n        ]\\n        if column in split_manifest.columns\\n    ]\\n    split_manifest[columns].sort_values(columns[:1], kind=\"stable\").to_csv(\\n        output_dir / \"group_key_audit.csv\",\\n        index=False,\\n    )\\n\\n\\ndef write_duplicate_source_pairs(\\n    duplicate_audit: pd.DataFrame,\\n    raw_manifest: pd.DataFrame,\\n    output_dir: Path,\\n) -> None:\\n    columns = [\"left_source\", \"right_source\", \"duplicate_type\", \"action\", \"count\"]\\n    if duplicate_audit.empty:\\n        pd.DataFrame(columns=columns).to_csv(output_dir / \"duplicate_source_pairs.csv\", index=False)\\n        return\\n    source_by_path = raw_manifest.drop_duplicates(\"path\").set_index(\"path\")[\"source\"].astype(str).to_dict()\\n    audit = duplicate_audit.copy()\\n    audit[\"left_source\"] = audit.get(\"left_path\", pd.Series(dtype=str)).map(source_by_path).fillna(\"unknown\")\\n    audit[\"right_source\"] = audit.get(\"right_path\", pd.Series(dtype=str)).map(source_by_path).fillna(\"unknown\")\\n    if \"action\" not in audit.columns:\\n        audit[\"action\"] = \"\"\\n    summary = (\\n        audit.groupby([\"left_source\", \"right_source\", \"duplicate_type\", \"action\"], dropna=False)\\n        .size()\\n        .rename(\"count\")\\n        .reset_index()\\n        .sort_values([\"count\", \"left_source\", \"right_source\"], ascending=[False, True, True])\\n    )\\n    summary.to_csv(output_dir / \"duplicate_source_pairs.csv\", index=False)\\n\\n\\ndef qa_thumbnail(image: np.ndarray, size: int = 160) -> np.ndarray:\\n    height, width = image.shape[:2]\\n    scale = float(size) / max(height, width)\\n    resized = cv2.resize(\\n        image,\\n        (max(1, int(width * scale)), max(1, int(height * scale))),\\n        interpolation=cv2.INTER_AREA,\\n    )\\n    canvas = np.zeros((size, size, 3), dtype=np.uint8)\\n    y0 = (size - resized.shape[0]) // 2\\n    x0 = (size - resized.shape[1]) // 2\\n    canvas[y0 : y0 + resized.shape[0], x0 : x0 + resized.shape[1]] = resized\\n    return canvas\\n\\n\\ndef write_preprocessing_qa(split_manifest: pd.DataFrame, cfg: dict, output_dir: Path) -> None:\\n    if not bool(cfg.get(\"make_visual_qa\", True)):\\n        return\\n    qa_dir = output_dir / \"preprocessing_qa_samples\"\\n    qa_dir.mkdir(parents=True, exist_ok=True)\\n    per_group = max(1, int(cfg.get(\"preprocessing_qa_per_source_class\", 2)))\\n    seed = int(cfg.get(\"seed\", 42)) + 2606\\n    rows = []\\n    for (source, label), group in split_manifest.groupby([\"source\", \"label\"], sort=True):\\n        scored = group.copy()\\n        scored[\"_qa_score\"] = [\\n            stable_row_score(row.orig_path, row.path, seed=seed)\\n            for row in scored.itertuples(index=False)\\n        ]\\n        rows.append(scored.sort_values(\"_qa_score\", kind=\"stable\").head(per_group))\\n    if not rows:\\n        return\\n    samples = pd.concat(rows, ignore_index=True).drop(columns=[\"_qa_score\"], errors=\"ignore\")\\n    samples.to_csv(qa_dir / \"qa_sample_manifest.csv\", index=False)\\n\\n    cell_h, cell_w = 210, 340\\n    columns = 4\\n    grid_rows = int(np.ceil(len(samples) / columns))\\n    grid = np.zeros((grid_rows * cell_h, columns * cell_w, 3), dtype=np.uint8)\\n    for sample_index, row in enumerate(samples.itertuples(index=False)):\\n        r = sample_index // columns\\n        c = sample_index % columns\\n        y = r * cell_h\\n        x = c * cell_w\\n        try:\\n            original = qa_thumbnail(read_rgb_decimated(str(row.orig_path), 768), 160)\\n            processed = qa_thumbnail(read_rgb_image(str(row.path)), 160)\\n        except Exception as exc:\\n            print(f\"Skipping QA sample {getattr(row, \\'orig_path\\', \\'\\')}: {exc}\")\\n            continue\\n        pair = np.concatenate([original, processed], axis=1)\\n        grid[y + 40 : y + 200, x + 10 : x + 330] = pair\\n        title = f\"{getattr(row, \\'source\\', \\'unknown\\')} cls={int(row.label)} {getattr(row, \\'partition\\', \\'\\')}\"\\n        cv2.putText(\\n            grid,\\n            title[:42],\\n            (x + 10, y + 24),\\n            cv2.FONT_HERSHEY_SIMPLEX,\\n            0.48,\\n            (255, 255, 255),\\n            1,\\n            cv2.LINE_AA,\\n        )\\n    cv2.imwrite(str(qa_dir / \"original_vs_preprocessed.jpg\"), cv2.cvtColor(grid, cv2.COLOR_RGB2BGR))\\n\\ndef prepare_data_from_cleaned_label_manifest(cfg: dict):\\n    output_dir = Path(cfg[\"output_dir\"])\\n    output_dir.mkdir(parents=True, exist_ok=True)\\n    start = time.time()\\n\\n    raw, source_metadata = load_cleaned_label_manifest(cfg)\\n    filename_deduped, filename_audit = remove_filename_duplicates(raw)\\n    cached, cache_metadata = build_cache(filename_deduped, cfg, output_dir)\\n    active, duplicate_audit = audit_duplicates(cached, filename_audit)\\n    split_manifest, splits = create_splits(active, cfg)\\n    write_group_key_audit(split_manifest, output_dir)\\n    write_duplicate_source_pairs(duplicate_audit, raw, output_dir)\\n    write_preprocessing_qa(split_manifest, cfg, output_dir)\\n\\n    duplicate_audit.to_csv(output_dir / \"duplicate_audit.csv\", index=False)\\n    split_manifest.to_csv(output_dir / \"split_manifest.csv\", index=False)\\n    write_data_audit(\\n        raw,\\n        filename_deduped,\\n        active,\\n        split_manifest,\\n        duplicate_audit,\\n        output_dir,\\n    )\\n    split_fingerprint = hashlib.sha256(\\n        split_manifest[\\n            [\"orig_path\", \"label\", \"group_key\", \"fold\", \"partition\"]\\n        ]\\n        .sort_values(\"orig_path\")\\n        .to_csv(index=False)\\n        .encode(\"utf-8\")\\n    ).hexdigest()\\n    metadata = {\\n        **cache_metadata,\\n        **source_metadata,\\n        \"split_strategy\": str(cfg.get(\"split_strategy\", \"stratified_group_10fold\")),\\n        \"split_ratio\": \"train_80_model_valid_5_calibration_5_holdout_10\",\\n        \"debug_sample_size\": int(cfg.get(\"debug_sample_size\", 0) or 0),\\n        \"split_fingerprint\": split_fingerprint,\\n        \"elapsed_minutes\": (time.time() - start) / 60.0,\\n        \"free_cache_gb\": shutil.disk_usage(Path(cache_metadata[\"resolved_cache_dir\"])).free\\n        / (1024**3),\\n    }\\n    with open(output_dir / \"data_pipeline_metadata.json\", \"w\") as handle:\\n        json.dump(metadata, handle, indent=2)\\n    return splits, split_manifest, duplicate_audit, metadata\\n\\ndef prepare_data_from_corrected_manifest(cfg: dict):\\n    output_dir = Path(cfg[\"output_dir\"])\\n    output_dir.mkdir(parents=True, exist_ok=True)\\n    start = time.time()\\n\\n    manifest_path = Path(\\n        cfg.get(\"corrected_manifest_path\") or (output_dir / \"corrected_manifest.csv\")\\n    )\\n    if bool(cfg.get(\"build_corrected_manifest\", False)) or not manifest_path.exists():\\n        from dr_manifest_correction import build_corrected_manifest\\n\\n        build_root = cfg.get(\"manifest_data_root\") or cfg.get(\"data_root\")\\n        if not build_root:\\n            raise ValueError(\\n                \"Set cfg[\\'manifest_data_root\\'] or cfg[\\'data_root\\'] to build corrected_manifest.csv.\"\\n            )\\n        manifest, _, duplicate_audit, _ = build_corrected_manifest(\\n            build_root,\\n            manifest_path.parent,\\n            dataset_variant=cfg.get(\"dataset_variant\"),\\n        )\\n    else:\\n        manifest = pd.read_csv(manifest_path)\\n        duplicate_path = manifest_path.parent / \"duplicate_audit.csv\"\\n        duplicate_audit = (\\n            pd.read_csv(duplicate_path) if duplicate_path.exists() else pd.DataFrame()\\n        )\\n\\n    active = prepare_frame_from_corrected_manifest(manifest)\\n    cached, cache_metadata = build_cache(active, cfg, output_dir)\\n    split_manifest, splits = create_existing_folder_splits(cached, cfg)\\n    split_manifest.to_csv(output_dir / \"split_manifest.csv\", index=False)\\n    if not duplicate_audit.empty:\\n        duplicate_audit.to_csv(output_dir / \"duplicate_audit.csv\", index=False)\\n\\n    split_fingerprint = hashlib.sha256(\\n        split_manifest[\\n            [\"orig_path\", \"label\", \"group_key\", \"split_folder\", \"partition\"]\\n        ]\\n        .sort_values(\"orig_path\")\\n        .to_csv(index=False)\\n        .encode(\"utf-8\")\\n    ).hexdigest()\\n    metadata = {\\n        **cache_metadata,\\n        \"corrected_manifest_path\": str(manifest_path),\\n        \"split_strategy\": \"existing_folder_corrected_manifest\",\\n        \"split_fingerprint\": split_fingerprint,\\n        \"elapsed_minutes\": (time.time() - start) / 60.0,\\n        \"free_cache_gb\": shutil.disk_usage(Path(cache_metadata[\"resolved_cache_dir\"])).free\\n        / (1024**3),\\n    }\\n    with open(output_dir / \"data_pipeline_metadata.json\", \"w\") as handle:\\n        json.dump(metadata, handle, indent=2)\\n    return splits, split_manifest, duplicate_audit, metadata\\n\\n\\ndef prepare_data(cfg: dict):\\n    if cleaned_label_manifest_configured(cfg):\\n        return prepare_data_from_cleaned_label_manifest(cfg)\\n    if corrected_manifest_configured(cfg):\\n        return prepare_data_from_corrected_manifest(cfg)\\n\\n    output_dir = Path(cfg[\"output_dir\"])\\n    output_dir.mkdir(parents=True, exist_ok=True)\\n    start = time.time()\\n    root = resolve_data_root(cfg)\\n    raw = scan_dataset(root)\\n    filename_deduped, filename_audit = remove_filename_duplicates(raw)\\n    cached, cache_metadata = build_cache(filename_deduped, cfg, output_dir)\\n    active, duplicate_audit = audit_duplicates(cached, filename_audit)\\n    split_manifest, splits = create_splits(active, cfg)\\n\\n    duplicate_audit.to_csv(output_dir / \"duplicate_audit.csv\", index=False)\\n    split_manifest.to_csv(output_dir / \"split_manifest.csv\", index=False)\\n    write_data_audit(\\n        raw,\\n        filename_deduped,\\n        active,\\n        split_manifest,\\n        duplicate_audit,\\n        output_dir,\\n    )\\n    split_fingerprint = hashlib.sha256(\\n        split_manifest[\\n            [\"orig_path\", \"label\", \"group_key\", \"fold\", \"partition\"]\\n        ]\\n        .sort_values(\"orig_path\")\\n        .to_csv(index=False)\\n        .encode(\"utf-8\")\\n    ).hexdigest()\\n    metadata = {\\n        **cache_metadata,\\n        \"split_fingerprint\": split_fingerprint,\\n        \"elapsed_minutes\": (time.time() - start) / 60.0,\\n        \"free_cache_gb\": shutil.disk_usage(Path(cache_metadata[\"resolved_cache_dir\"])).free\\n        / (1024**3),\\n    }\\n    with open(output_dir / \"data_pipeline_metadata.json\", \"w\") as handle:\\n        json.dump(metadata, handle, indent=2)\\n    return splits, split_manifest, duplicate_audit, metadata\\n', 'dr_experts_corrected.py': 'from __future__ import annotations\\n\\nimport hashlib\\nimport json\\nimport math\\nfrom pathlib import Path\\n\\nimport numpy as np\\nimport pandas as pd\\n\\n\\nEXPERT_NAMES = (\"severity_480\", \"zero_mild_480\", \"moderate_severe_480\")\\nSEVERITY_LABEL_MAP = {1: 0, 2: 1, 3: 2, 4: 3}\\nZERO_MILD_LABEL_MAP = {0: 0, 1: 1}\\nMODERATE_SEVERE_LABEL_MAP = {2: 0, 3: 1, 4: 1}\\n\\n\\ndef normalize(probabilities: np.ndarray) -> np.ndarray:\\n    probabilities = np.clip(np.asarray(probabilities, dtype=np.float64), 1e-12, 1.0)\\n    return probabilities / probabilities.sum(axis=1, keepdims=True)\\n\\n\\ndef stable_hash_int(value: str, seed: int = 0) -> int:\\n    payload = f\"{int(seed)}|{value}\".encode(\"utf-8\")\\n    return int(hashlib.sha1(payload).hexdigest()[:16], 16)\\n\\n\\ndef deterministic_sample(frame: pd.DataFrame, count: int, seed: int) -> pd.DataFrame:\\n    if count >= len(frame):\\n        return frame.copy()\\n    if count <= 0:\\n        return frame.iloc[0:0].copy()\\n    scored = frame.copy()\\n    scored[\"_sample_score\"] = [\\n        stable_hash_int(f\"{row.group_key}|{row.orig_path}\", seed)\\n        for row in scored.itertuples(index=False)\\n    ]\\n    return (\\n        scored.sort_values(\"_sample_score\", kind=\"stable\")\\n        .head(int(count))\\n        .drop(columns=[\"_sample_score\"])\\n    )\\n\\n\\ndef make_expert_cache_cfg(cfg: dict) -> dict:\\n    expert_cfg = dict(cfg)\\n    size = int(cfg.get(\"expert_image_size\", 480))\\n    scale = int(cfg.get(\"expert_lid_scale\", cfg.get(\"lid_scale\", 4)))\\n    expert_cfg[\"image_size\"] = size\\n    expert_cfg[\"downscale_method\"] = cfg.get(\\n        \"expert_downscale_method\",\\n        cfg.get(\"downscale_method\", \"opencv_area\"),\\n    )\\n    expert_cfg[\"cache_dir\"] = cfg.get(\\n        \"expert_cache_dir\",\\n        \"/kaggle/temp/dr_inceptionv3_fundus_cache_480_experts_v1\",\\n    )\\n    expert_cfg[\"lid_scale\"] = scale\\n    expert_cfg[\"lid_input_size\"] = int(cfg.get(\"expert_lid_input_size\", size * scale))\\n    default_decode_min_side = (\\n        expert_cfg[\"lid_input_size\"]\\n        if str(expert_cfg[\"downscale_method\"]).lower() in {\"lid\", \"car_lid\", \"car/lid\", \"car\"}\\n        else max(1024, size * 2)\\n    )\\n    expert_cfg[\"cache_decode_min_side\"] = int(\\n        cfg.get(\"expert_cache_decode_min_side\", default_decode_min_side)\\n    )\\n    expert_cfg[\"cache_workers\"] = int(cfg.get(\"expert_cache_workers\", cfg.get(\"cache_workers\", 4)))\\n    return expert_cfg\\n\\n\\ndef estimate_cache_gb(row_count: int, image_size: int, bytes_per_pixel: float = 0.70) -> float:\\n    return float(row_count) * float(image_size) * float(image_size) * float(bytes_per_pixel) / (1024**3)\\n\\n\\ndef select_expert_cache_source_rows(split_manifest: pd.DataFrame, cfg: dict) -> tuple[pd.DataFrame, dict]:\\n    required = {\"orig_path\", \"path\", \"label\", \"partition\", \"group_key\"}\\n    missing = sorted(required - set(split_manifest.columns))\\n    if missing:\\n        raise ValueError(f\"split_manifest is missing required columns for 480 experts: {missing}\")\\n\\n    frame = split_manifest.copy().reset_index(drop=True)\\n    frame[\"label\"] = frame[\"label\"].astype(int)\\n    frame[\"orig_path\"] = frame[\"orig_path\"].astype(str)\\n    frame[\"main_cache_path\"] = frame[\"path\"].astype(str)\\n\\n    train_mask = frame[\"partition\"].eq(\"train\")\\n    eval_mask = frame[\"partition\"].isin([\"model_valid\", \"calibration\", \"holdout\"])\\n    train_nonzero_mask = train_mask & frame[\"label\"].isin([1, 2, 3, 4])\\n\\n    mild_train = frame[train_mask & frame[\"label\"].eq(1)]\\n    zero_train = frame[train_mask & frame[\"label\"].eq(0)]\\n    zero_ratio = float(cfg.get(\"expert_zero_mild_zero_ratio\", 2.0))\\n    zero_cap = int(math.ceil(len(mild_train) * zero_ratio))\\n    zero_sample = deterministic_sample(\\n        zero_train,\\n        zero_cap,\\n        int(cfg.get(\"expert_sampling_seed\", int(cfg.get(\"seed\", 42)) + 4800)),\\n    )\\n    zero_sample_indices = set(zero_sample.index.tolist())\\n    train_zero_sample_mask = frame.index.isin(zero_sample_indices)\\n\\n    selected = frame[eval_mask | train_nonzero_mask | train_zero_sample_mask].copy()\\n    selected[\"path\"] = selected[\"orig_path\"]\\n    if \"cache_path\" in selected.columns:\\n        selected = selected.drop(columns=[\"cache_path\"])\\n\\n    summary = {\\n        \"selected_rows\": int(len(selected)),\\n        \"eval_rows_cached\": int(eval_mask.sum()),\\n        \"train_nonzero_rows_cached\": int(train_nonzero_mask.sum()),\\n        \"train_zero_rows_available\": int(len(zero_train)),\\n        \"train_zero_rows_cached\": int(len(zero_sample)),\\n        \"zero_mild_zero_ratio\": zero_ratio,\\n    }\\n    return selected.reset_index(drop=True), summary\\n\\n\\ndef remap_expert_labels(frame: pd.DataFrame, label_map: dict[int, int]) -> pd.DataFrame:\\n    result = frame[frame[\"label\"].astype(int).isin(label_map)].copy().reset_index(drop=True)\\n    if result.empty:\\n        return result\\n    result[\"original_label\"] = result[\"label\"].astype(int)\\n    result[\"label\"] = result[\"original_label\"].map(label_map).astype(int)\\n    return result\\n\\n\\ndef expert_split_summary(expert_splits: dict[str, dict[str, pd.DataFrame]]) -> pd.DataFrame:\\n    rows = []\\n    for expert_name in EXPERT_NAMES:\\n        for split_name in (\"train\", \"model_valid\"):\\n            frame = expert_splits[expert_name][split_name]\\n            original_column = \"original_label\" if \"original_label\" in frame.columns else \"label\"\\n            counts = (\\n                frame.groupby([original_column, \"label\"])\\n                .size()\\n                .rename(\"count\")\\n                .reset_index()\\n            )\\n            for row in counts.itertuples(index=False):\\n                rows.append(\\n                    {\\n                        \"expert\": expert_name,\\n                        \"split\": split_name,\\n                        \"original_label\": int(getattr(row, original_column)),\\n                        \"expert_label\": int(row.label),\\n                        \"count\": int(row.count),\\n                    }\\n                )\\n    return pd.DataFrame(rows)\\n\\n\\ndef print_expert_split_summary(summary: pd.DataFrame) -> None:\\n    print(\"480px expert class counts:\")\\n    if summary.empty:\\n        print(\"(empty)\")\\n        return\\n    for expert_name in EXPERT_NAMES:\\n        print(f\"\\\\n{expert_name}\")\\n        view = summary[summary[\"expert\"] == expert_name]\\n        table = view.pivot_table(\\n            index=[\"original_label\", \"expert_label\"],\\n            columns=\"split\",\\n            values=\"count\",\\n            fill_value=0,\\n            aggfunc=\"sum\",\\n        ).astype(int)\\n        print(table.to_string())\\n\\n\\ndef align_cached_frame(cached_frame: pd.DataFrame, reference_frame: pd.DataFrame) -> pd.DataFrame:\\n    if \"orig_path\" not in cached_frame.columns or \"orig_path\" not in reference_frame.columns:\\n        raise ValueError(\"Both cached_frame and reference_frame must include orig_path.\")\\n    indexed = cached_frame.drop_duplicates(\"orig_path\").set_index(\"orig_path\", drop=False)\\n    order = reference_frame[\"orig_path\"].astype(str).tolist()\\n    missing = [path for path in order if path not in indexed.index]\\n    if missing:\\n        raise KeyError(f\"480 expert cache is missing {len(missing)} referenced rows. First: {missing[0]}\")\\n    return indexed.loc[order].reset_index(drop=True)\\n\\n\\ndef prepare_expert_480_data(\\n    split_manifest: pd.DataFrame,\\n    cfg: dict,\\n    output_dir: str | Path,\\n) -> tuple[dict[str, dict[str, pd.DataFrame] | pd.DataFrame], dict]:\\n    from dr_data_corrected import build_cache\\n\\n    output_dir = Path(output_dir)\\n    expert_dir = output_dir / \"expert_480_data\"\\n    expert_dir.mkdir(parents=True, exist_ok=True)\\n\\n    source_rows, source_summary = select_expert_cache_source_rows(split_manifest, cfg)\\n    expert_cfg = make_expert_cache_cfg(cfg)\\n    estimate_gb = estimate_cache_gb(len(source_rows), int(expert_cfg[\"image_size\"]))\\n    print(\\n        \"480px expert cache estimate: \"\\n        f\"{len(source_rows)} images, roughly {estimate_gb:.2f} GiB JPEG cache \"\\n        \"(actual size depends on image content and JPEG quality).\"\\n    )\\n\\n    cached, cache_metadata = build_cache(source_rows, expert_cfg, expert_dir)\\n    cached = cached.copy().reset_index(drop=True)\\n    cached[\"orig_path\"] = cached[\"path\"].astype(str)\\n    cached[\"path\"] = cached[\"cache_path\"].astype(str)\\n    cached[\"label\"] = cached[\"label\"].astype(int)\\n    cached.to_csv(expert_dir / \"expert_480_cache_manifest.csv\", index=False)\\n\\n    train = cached[cached[\"partition\"] == \"train\"].reset_index(drop=True)\\n    model_valid = cached[cached[\"partition\"] == \"model_valid\"].reset_index(drop=True)\\n    calibration = cached[cached[\"partition\"] == \"calibration\"].reset_index(drop=True)\\n    holdout = cached[cached[\"partition\"] == \"holdout\"].reset_index(drop=True)\\n\\n    expert_splits: dict[str, dict[str, pd.DataFrame] | pd.DataFrame] = {\\n        \"severity_480\": {\\n            \"train\": remap_expert_labels(train, SEVERITY_LABEL_MAP),\\n            \"model_valid\": remap_expert_labels(model_valid, SEVERITY_LABEL_MAP),\\n        },\\n        \"zero_mild_480\": {\\n            \"train\": remap_expert_labels(train, ZERO_MILD_LABEL_MAP),\\n            \"model_valid\": remap_expert_labels(model_valid, ZERO_MILD_LABEL_MAP),\\n        },\\n        \"moderate_severe_480\": {\\n            \"train\": remap_expert_labels(train, MODERATE_SEVERE_LABEL_MAP),\\n            \"model_valid\": remap_expert_labels(model_valid, MODERATE_SEVERE_LABEL_MAP),\\n        },\\n        \"calibration\": calibration,\\n        \"holdout\": holdout,\\n        \"all_cached\": cached,\\n    }\\n    for expert_name in EXPERT_NAMES:\\n        for split_name in (\"train\", \"model_valid\"):\\n            if expert_splits[expert_name][split_name].empty:\\n                raise RuntimeError(f\"{expert_name} {split_name} split is empty after label remapping.\")\\n\\n    summary = expert_split_summary(expert_splits)\\n    summary.to_csv(expert_dir / \"expert_480_split_summary.csv\", index=False)\\n    print_expert_split_summary(summary)\\n\\n    metadata = {\\n        \"expert_cache_cfg\": {\\n            key: expert_cfg.get(key)\\n            for key in [\\n                \"image_size\",\\n                \"cache_dir\",\\n                \"cache_decode_min_side\",\\n                \"downscale_method\",\\n                \"lid_input_size\",\\n                \"lid_scale\",\\n                \"lid_weight_path\",\\n            ]\\n        },\\n        \"source_summary\": source_summary,\\n        \"estimated_cache_gb\": estimate_gb,\\n        \"cache_metadata\": cache_metadata,\\n    }\\n    with open(expert_dir / \"expert_480_metadata.json\", \"w\", encoding=\"utf-8\") as handle:\\n        json.dump(metadata, handle, indent=2)\\n    return expert_splits, metadata\\n\\n\\ndef apply_bias(probs: np.ndarray, bias: np.ndarray) -> np.ndarray:\\n    logits = np.log(np.clip(np.asarray(probs, dtype=np.float64), 1e-12, 1.0))\\n    logits += np.asarray(bias, dtype=np.float64).reshape(1, -1)\\n    logits -= logits.max(axis=1, keepdims=True)\\n    exp = np.exp(logits)\\n    return normalize(exp)\\n\\n\\ndef weighted_main_probs(component_probs: dict[str, np.ndarray], weights: dict[str, float]) -> np.ndarray:\\n    output = None\\n    for name, weight in weights.items():\\n        contribution = float(weight) * np.asarray(component_probs[name], dtype=np.float64)\\n        output = contribution if output is None else output + contribution\\n    return normalize(output)\\n\\n\\ndef calibrate_binary_probability(probability: np.ndarray, threshold: float) -> np.ndarray:\\n    probability = np.clip(np.asarray(probability, dtype=np.float64), 1e-6, 1.0 - 1e-6)\\n    threshold = float(np.clip(threshold, 1e-6, 1.0 - 1e-6))\\n    odds = probability / (1.0 - probability)\\n    threshold_odds = threshold / (1.0 - threshold)\\n    shifted = odds / threshold_odds\\n    return shifted / (1.0 + shifted)\\n\\n\\ndef apply_severity_expert(probs: np.ndarray, severity_probs: np.ndarray, weight: float) -> np.ndarray:\\n    weight = float(np.clip(weight, 0.0, 1.0))\\n    if weight <= 0:\\n        return normalize(probs)\\n    output = normalize(probs).copy()\\n    disease_mass = output[:, 1:].sum(axis=1, keepdims=True)\\n    main_relative = normalize(output[:, 1:])\\n    expert_relative = normalize(severity_probs)\\n    output[:, 1:] = disease_mass * normalize(\\n        (1.0 - weight) * main_relative + weight * expert_relative\\n    )\\n    return normalize(output)\\n\\n\\ndef apply_zero_mild_expert(\\n    probs: np.ndarray,\\n    zero_mild_probs: np.ndarray,\\n    weight: float,\\n    threshold: float,\\n) -> np.ndarray:\\n    weight = float(np.clip(weight, 0.0, 1.0))\\n    if weight <= 0:\\n        return normalize(probs)\\n    output = normalize(probs).copy()\\n    pre_pred = output.argmax(axis=1)\\n    eligible = pre_pred <= 1\\n    if not eligible.any():\\n        return output\\n    mass01 = output[eligible, 0:2].sum(axis=1)\\n    current_mild_share = output[eligible, 1] / np.maximum(mass01, 1e-12)\\n    expert_mild_share = calibrate_binary_probability(zero_mild_probs[eligible, 1], threshold)\\n    mild_share = np.clip(\\n        (1.0 - weight) * current_mild_share + weight * expert_mild_share,\\n        0.0,\\n        1.0,\\n    )\\n    output[eligible, 0] = mass01 * (1.0 - mild_share)\\n    output[eligible, 1] = mass01 * mild_share\\n    return normalize(output)\\n\\n\\ndef apply_moderate_severe_expert(\\n    probs: np.ndarray,\\n    moderate_severe_probs: np.ndarray,\\n    weight: float,\\n    threshold: float,\\n) -> np.ndarray:\\n    weight = float(np.clip(weight, 0.0, 1.0))\\n    if weight <= 0:\\n        return normalize(probs)\\n    output = normalize(probs).copy()\\n    mass234 = output[:, 2:5].sum(axis=1)\\n    active = mass234 > 1e-12\\n    if not active.any():\\n        return output\\n    current_severe_share = (output[active, 3] + output[active, 4]) / mass234[active]\\n    expert_severe_share = calibrate_binary_probability(moderate_severe_probs[active, 1], threshold)\\n    severe_share = np.clip(\\n        (1.0 - weight) * current_severe_share + weight * expert_severe_share,\\n        0.0,\\n        1.0,\\n    )\\n    severe_mass = mass234[active] * severe_share\\n    moderate_mass = mass234[active] - severe_mass\\n    severe_ratio = output[active, 4] / np.maximum(output[active, 3] + output[active, 4], 1e-12)\\n    output[active, 2] = moderate_mass\\n    output[active, 3] = severe_mass * (1.0 - severe_ratio)\\n    output[active, 4] = severe_mass * severe_ratio\\n    return normalize(output)\\n\\n\\ndef predict_with_expert_candidate(\\n    component_probs: dict[str, np.ndarray],\\n    candidate: dict,\\n) -> tuple[np.ndarray, np.ndarray]:\\n    probs = weighted_main_probs(component_probs, candidate[\"weights\"])\\n    probs = apply_bias(probs, np.asarray(candidate.get(\"bias\", np.zeros(5)), dtype=np.float64))\\n    params = candidate.get(\"expert_params\", {})\\n    if \"severity_480\" in component_probs:\\n        probs = apply_severity_expert(\\n            probs,\\n            component_probs[\"severity_480\"],\\n            float(params.get(\"severity_weight\", 0.0)),\\n        )\\n    if \"zero_mild_480\" in component_probs:\\n        probs = apply_zero_mild_expert(\\n            probs,\\n            component_probs[\"zero_mild_480\"],\\n            float(params.get(\"zero_mild_weight\", 0.0)),\\n            float(params.get(\"zero_mild_threshold\", 0.5)),\\n        )\\n    if \"moderate_severe_480\" in component_probs:\\n        probs = apply_moderate_severe_expert(\\n            probs,\\n            component_probs[\"moderate_severe_480\"],\\n            float(params.get(\"moderate_severe_weight\", 0.0)),\\n            float(params.get(\"moderate_severe_threshold\", 0.5)),\\n        )\\n    return normalize(probs), probs.argmax(axis=1).astype(int)\\n\\n\\ndef safety_metrics(labels: np.ndarray, predictions: np.ndarray) -> dict:\\n    labels = np.asarray(labels, dtype=np.int64)\\n    predictions = np.asarray(predictions, dtype=np.int64)\\n    dr_mask = labels > 0\\n    severe_mask = labels >= 3\\n    dr_to_zero = dr_mask & (predictions == 0)\\n    severe_to_zero = severe_mask & (predictions == 0)\\n    return {\\n        \"dr_to_0_count\": int(dr_to_zero.sum()),\\n        \"dr_to_0_rate\": float(dr_to_zero.sum() / max(1, dr_mask.sum())),\\n        \"severe_to_0_count\": int(severe_to_zero.sum()),\\n        \"severe_to_0_rate\": float(severe_to_zero.sum() / max(1, severe_mask.sum())),\\n    }\\n\\n\\ndef score_with_safety(labels: np.ndarray, predictions: np.ndarray, metric_fn) -> dict:\\n    metrics = dict(metric_fn(labels, predictions))\\n    metrics.update(safety_metrics(labels, predictions))\\n    return metrics\\n', 'dr_runtime_corrected.py': 'from __future__ import annotations\\n\\nimport hashlib\\nimport os\\nimport random\\nfrom contextlib import nullcontext\\nfrom pathlib import Path\\n\\nimport numpy as np\\nimport pandas as pd\\nimport torch\\nimport torchvision.transforms.functional as TF\\nfrom PIL import Image, ImageFile, UnidentifiedImageError\\nfrom sklearn.metrics import accuracy_score, cohen_kappa_score, f1_score\\nfrom torch import nn\\nfrom torch.utils.data import Dataset\\n\\ntry:\\n    import cv2\\n\\n    cv2.setNumThreads(0)\\nexcept ImportError:\\n    cv2 = None\\n\\nImageFile.LOAD_TRUNCATED_IMAGES = True\\n\\ntry:\\n    from torch.amp import GradScaler as _GradScaler\\n    from torch.amp import autocast as _autocast\\n\\n    _AMP_NEW = True\\nexcept ImportError:\\n    from torch.cuda.amp import GradScaler as _GradScaler\\n    from torch.cuda.amp import autocast as _autocast\\n\\n    _AMP_NEW = False\\n\\n_MEAN = torch.tensor([0.485, 0.456, 0.406]).view(3, 1, 1)\\n_STD = torch.tensor([0.229, 0.224, 0.225]).view(3, 1, 1)\\n\\n\\ndef amp_autocast(device: torch.device):\\n    if device.type != \"cuda\":\\n        return nullcontext()\\n    return _autocast(device_type=\"cuda\") if _AMP_NEW else _autocast()\\n\\n\\ndef make_scaler(device: torch.device):\\n    enabled = device.type == \"cuda\"\\n    if _AMP_NEW:\\n        try:\\n            return _GradScaler(device=\"cuda\", enabled=enabled)\\n        except TypeError:\\n            return _GradScaler(enabled=enabled)\\n    return _GradScaler(enabled=enabled)\\n\\n\\ndef effective_cpu_count() -> int:\\n    try:\\n        return max(1, len(os.sched_getaffinity(0)))\\n    except AttributeError:\\n        return max(1, os.cpu_count() or 4)\\n\\n\\ndef worker_init(worker_id: int) -> None:\\n    os.environ[\"OMP_NUM_THREADS\"] = \"1\"\\n    os.environ[\"MKL_NUM_THREADS\"] = \"1\"\\n    random.seed(torch.initial_seed() % (2**32))\\n    np.random.seed(torch.initial_seed() % (2**32))\\n    torch.set_num_threads(1)\\n    if cv2 is not None:\\n        cv2.setNumThreads(0)\\n\\n\\ndef read_rgb_image(path: str) -> np.ndarray:\\n    suffix = Path(path).suffix.lower()\\n    if cv2 is not None and suffix not in {\".gif\", \".gf\"}:\\n        image = cv2.imread(path, cv2.IMREAD_COLOR)\\n        if image is not None:\\n            return cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\\n    try:\\n        with Image.open(path) as image:\\n            return np.asarray(image.convert(\"RGB\"))\\n    except (UnidentifiedImageError, OSError) as exc:\\n        raise RuntimeError(f\"Failed decoding image {path}: {exc}\") from exc\\n\\n\\nclass FastRetinaTransform:\\n    def __init__(self, cfg: dict, train: bool) -> None:\\n        self.size = int(cfg[\"image_size\"])\\n        self.train = bool(train)\\n        self.rotation = float(cfg.get(\"aug_rotation_degrees\", 0.0)) if train else 0.0\\n        self.color_jitter = bool(cfg.get(\"use_color_jitter\", False)) and train\\n\\n    def __call__(self, image: np.ndarray) -> torch.Tensor:\\n        tensor = torch.from_numpy(np.ascontiguousarray(image)).permute(2, 0, 1).float().div_(255.0)\\n        if tensor.shape[1:] != (self.size, self.size):\\n            tensor = TF.resize(tensor, [self.size, self.size], antialias=True)\\n        if self.train:\\n            if random.random() < 0.5:\\n                tensor = TF.hflip(tensor)\\n            if random.random() < 0.5:\\n                tensor = TF.vflip(tensor)\\n            if self.rotation > 0 and random.random() < 0.5:\\n                tensor = TF.rotate(tensor, random.uniform(-self.rotation, self.rotation))\\n            if self.color_jitter:\\n                tensor = TF.adjust_brightness(tensor, 1.0 + random.uniform(-0.05, 0.05))\\n                tensor = TF.adjust_contrast(tensor, 1.0 + random.uniform(-0.05, 0.05))\\n                tensor = tensor.clamp_(0.0, 1.0)\\n        return (tensor - _MEAN) / _STD\\n\\n\\nclass CachedRetinaDataset(Dataset):\\n    def __init__(self, frame: pd.DataFrame, cfg: dict, train: bool) -> None:\\n        self.paths = frame[\"path\"].astype(str).tolist()\\n        self.labels = frame[\"label\"].astype(\"int64\").to_numpy()\\n        self.transform = FastRetinaTransform(cfg, train=train)\\n\\n    def __len__(self) -> int:\\n        return len(self.paths)\\n\\n    def __getitem__(self, index: int):\\n        return {\\n            \"image\": self.transform(read_rgb_image(self.paths[index])),\\n            \"label\": int(self.labels[index]),\\n        }\\n\\n\\nclass UnlabeledCachedRetinaDataset(Dataset):\\n    def __init__(self, paths, cfg: dict) -> None:\\n        self.paths = [str(path) for path in paths]\\n        self.transform = FastRetinaTransform(cfg, train=False)\\n\\n    def __len__(self) -> int:\\n        return len(self.paths)\\n\\n    def __getitem__(self, index: int):\\n        return {\"image\": self.transform(read_rgb_image(self.paths[index]))}\\n\\n\\nclass DistributedWeightedSampler:\\n    def __init__(self, weights, total: int, world: int, rank: int, seed: int) -> None:\\n        self.weights = torch.as_tensor(weights, dtype=torch.double)\\n        self.total = int(total)\\n        self.world = int(world)\\n        self.rank = int(rank)\\n        self.seed = int(seed)\\n        self.epoch = 0\\n        self.num_samples = self.total // self.world\\n\\n    def set_epoch(self, epoch: int) -> None:\\n        self.epoch = int(epoch)\\n\\n    def __len__(self) -> int:\\n        return self.num_samples\\n\\n    def __iter__(self):\\n        generator = torch.Generator()\\n        generator.manual_seed(self.seed + self.epoch)\\n        indices = torch.multinomial(\\n            self.weights,\\n            self.total,\\n            replacement=True,\\n            generator=generator,\\n        )\\n        shard = indices[self.rank : self.world * self.num_samples : self.world]\\n        return iter(shard.tolist())\\n\\n\\ndef build_model(cfg: dict, pretrained: bool | None = None) -> nn.Module:\\n    import timm\\n\\n    requested = bool(cfg.get(\"pretrained\", False)) if pretrained is None else bool(pretrained)\\n    out_dim = int(cfg[\"num_classes\"])\\n    try:\\n        return timm.create_model(\\n            cfg[\"backbone\"],\\n            pretrained=requested,\\n            num_classes=out_dim,\\n            drop_rate=float(cfg.get(\"drop_rate\", 0.0)),\\n        )\\n    except Exception as exc:\\n        if requested and bool(cfg.get(\"require_pretrained\", False)):\\n            raise RuntimeError(\\n                \"Stage 1 requires pretrained weights, but they could not be loaded. \"\\n                \"Enable Kaggle internet or attach the pretrained weights before retrying.\"\\n            ) from exc\\n        raise\\n\\n\\ndef load_parent_weights(model: nn.Module, checkpoint: Path, mode: str) -> None:\\n    saved = torch.load(checkpoint, map_location=\"cpu\", weights_only=False)\\n    source = saved[\"model_state\"]\\n    if mode == \"strict\":\\n        model.load_state_dict(source, strict=True)\\n        return\\n    if mode != \"compatible\":\\n        raise ValueError(f\"Unknown init_mode={mode!r}; expected \\'strict\\' or \\'compatible\\'.\")\\n    destination = model.state_dict()\\n    compatible = {\\n        key: value\\n        for key, value in source.items()\\n        if key in destination and tuple(destination[key].shape) == tuple(value.shape)\\n    }\\n    model.load_state_dict(compatible, strict=False)\\n\\n\\ndef class_weights(frame: pd.DataFrame, cfg: dict, device: torch.device):\\n    if cfg.get(\"imbalance_mode\", \"none\") not in (\"class_weight\", \"both\"):\\n        return None\\n    power = float(cfg.get(\"class_weight_power\", 0.0))\\n    if power <= 0:\\n        return None\\n    counts = np.bincount(\\n        frame[\"label\"].to_numpy(),\\n        minlength=int(cfg[\"num_classes\"]),\\n    ).astype(np.float32)\\n    counts[counts == 0] = 1\\n    weights = (counts.sum() / (int(cfg[\"num_classes\"]) * counts)) ** power\\n    weights /= weights.mean()\\n    return torch.tensor(weights, dtype=torch.float32, device=device)\\n\\n\\ndef build_criterion(cfg: dict, frame: pd.DataFrame, device: torch.device) -> nn.Module:\\n    return nn.CrossEntropyLoss(\\n        weight=class_weights(frame, cfg, device),\\n        label_smoothing=float(cfg.get(\"label_smoothing\", 0.0)),\\n    )\\n\\n\\ndef calculate_metrics(labels: np.ndarray, predictions: np.ndarray) -> dict:\\n    return {\\n        \"accuracy\": float(accuracy_score(labels, predictions)),\\n        \"macro_f1\": float(f1_score(labels, predictions, average=\"macro\", zero_division=0)),\\n        \"weighted_f1\": float(f1_score(labels, predictions, average=\"weighted\", zero_division=0)),\\n        \"quadratic_kappa\": float(\\n            cohen_kappa_score(labels, predictions, weights=\"quadratic\")\\n        ),\\n    }\\n\\n\\ndef select_guarded_epoch(history: list[dict], accuracy_guard: float = 0.01) -> dict:\\n    if not history:\\n        raise RuntimeError(\"No validation history is available for checkpoint selection.\")\\n    best_accuracy = max(float(row[\"val_accuracy\"]) for row in history)\\n    floor = best_accuracy - float(accuracy_guard)\\n    eligible = [row for row in history if float(row[\"val_accuracy\"]) >= floor]\\n    return max(\\n        eligible,\\n        key=lambda row: (\\n            float(row[\"val_macro_f1\"]),\\n            float(row[\"val_quadratic_kappa\"]),\\n            float(row[\"val_accuracy\"]),\\n        ),\\n    )\\n\\n\\ndef sha256_file(path: str | Path) -> str:\\n    digest = hashlib.sha256()\\n    with open(path, \"rb\") as handle:\\n        for chunk in iter(lambda: handle.read(1024 * 1024), b\"\"):\\n            digest.update(chunk)\\n    return digest.hexdigest()\\n', 'train_stage_corrected.py': 'from __future__ import annotations\\n\\nimport argparse\\nimport json\\nimport math\\nimport os\\nimport random\\nimport shutil\\nimport time\\nfrom pathlib import Path\\n\\nimport numpy as np\\nimport pandas as pd\\nimport torch\\nimport torch.distributed as dist\\nfrom torch.nn.parallel import DistributedDataParallel as DDP\\nfrom torch.utils.data import DataLoader\\nfrom torch.utils.data.distributed import DistributedSampler\\n\\nimport dr_runtime_corrected as rt\\n\\n\\ndef build_sampler(frame, cfg, world, rank, seed):\\n    use_weighted = (\\n        cfg.get(\"imbalance_mode\", \"none\") in (\"sampler\", \"both\")\\n        and float(cfg.get(\"sampler_power\", 0.0)) > 0\\n    )\\n    if not use_weighted:\\n        return DistributedSampler(\\n            range(len(frame)),\\n            num_replicas=world,\\n            rank=rank,\\n            shuffle=True,\\n            drop_last=True,\\n        )\\n    labels = frame[\"label\"].to_numpy()\\n    counts = np.bincount(labels, minlength=int(cfg[\"num_classes\"])).astype(np.float64)\\n    counts[counts == 0] = 1\\n    class_weights = (1.0 / counts) ** float(cfg[\"sampler_power\"])\\n    sample_weights = class_weights[labels]\\n    sample_weights /= sample_weights.mean()\\n    return rt.DistributedWeightedSampler(\\n        sample_weights,\\n        len(frame),\\n        world,\\n        rank,\\n        seed,\\n    )\\n\\n\\n@torch.inference_mode()\\ndef evaluate(model, loader, device):\\n    model.eval()\\n    probabilities = []\\n    labels = []\\n    for batch in loader:\\n        images = batch[\"image\"].to(device, non_blocking=True).contiguous(\\n            memory_format=torch.channels_last\\n        )\\n        with rt.amp_autocast(device):\\n            logits = model(images)\\n        probabilities.append(torch.softmax(logits.float(), dim=1).cpu().numpy())\\n        labels.append(batch[\"label\"].numpy())\\n    probs = np.concatenate(probabilities)\\n    truth = np.concatenate(labels)\\n    return rt.calculate_metrics(truth, probs.argmax(axis=1))\\n\\n\\ndef main() -> None:\\n    parser = argparse.ArgumentParser()\\n    parser.add_argument(\"--workdir\", required=True)\\n    parser.add_argument(\"--init-ckpt\")\\n    args = parser.parse_args()\\n\\n    work = Path(args.workdir)\\n    cfg = json.load(open(work / \"run_config.json\"))\\n    stage_name = str(cfg[\"stage_name\"])\\n\\n    rank = int(os.environ.get(\"RANK\", \"0\"))\\n    world = int(os.environ.get(\"WORLD_SIZE\", \"1\"))\\n    local_rank = int(os.environ.get(\"LOCAL_RANK\", \"0\"))\\n    torch.cuda.set_device(local_rank)\\n    device = torch.device(\"cuda\", local_rank)\\n    dist.init_process_group(backend=\"nccl\", init_method=\"env://\", device_id=device)\\n    is_main = rank == 0\\n\\n    seed = int(cfg.get(\"seed\", 42)) + int(cfg.get(\"stage_seed_offset\", 0))\\n    random.seed(seed + rank)\\n    np.random.seed(seed + rank)\\n    torch.manual_seed(seed + rank)\\n    torch.cuda.manual_seed_all(seed + rank)\\n    torch.backends.cudnn.benchmark = False\\n\\n    train_frame = pd.read_csv(work / \"split_train.csv\")\\n    valid_frame = pd.read_csv(work / \"split_model_valid.csv\")\\n\\n    global_batch = int(cfg[\"batch_size\"])\\n    per_rank_batch = max(1, global_batch // world)\\n    accumulation = max(1, int(cfg.get(\"grad_accum_steps\", 1)))\\n    total_workers = min(\\n        max(1, int(cfg.get(\"num_workers\", 4))),\\n        rt.effective_cpu_count(),\\n    )\\n    workers = max(1, total_workers // world)\\n    sampler = build_sampler(train_frame, cfg, world, rank, seed)\\n\\n    loader_kwargs = {\\n        \"num_workers\": workers,\\n        \"pin_memory\": True,\\n        \"persistent_workers\": workers > 0,\\n        \"worker_init_fn\": rt.worker_init if workers > 0 else None,\\n    }\\n    if workers > 0:\\n        loader_kwargs[\"prefetch_factor\"] = 2\\n\\n    train_loader = DataLoader(\\n        rt.CachedRetinaDataset(train_frame, cfg, train=True),\\n        batch_size=per_rank_batch,\\n        sampler=sampler,\\n        drop_last=True,\\n        **loader_kwargs,\\n    )\\n\\n    if is_main:\\n        model = rt.build_model(cfg, pretrained=bool(cfg.get(\"pretrained\", False)))\\n        dist.barrier()\\n    else:\\n        dist.barrier()\\n        model = rt.build_model(cfg, pretrained=False)\\n\\n    if args.init_ckpt:\\n        rt.load_parent_weights(\\n            model,\\n            Path(args.init_ckpt),\\n            str(cfg.get(\"init_mode\", \"strict\")),\\n        )\\n        if is_main:\\n            print(f\"[{stage_name}] initialized from {args.init_ckpt}\", flush=True)\\n\\n    model = model.to(device, memory_format=torch.channels_last)\\n    model = DDP(model, device_ids=[local_rank], output_device=local_rank)\\n    criterion = rt.build_criterion(cfg, train_frame, device)\\n\\n    try:\\n        optimizer = torch.optim.AdamW(\\n            model.parameters(),\\n            lr=float(cfg[\"lr\"]),\\n            weight_decay=float(cfg[\"weight_decay\"]),\\n            fused=True,\\n        )\\n    except (TypeError, RuntimeError):\\n        optimizer = torch.optim.AdamW(\\n            model.parameters(),\\n            lr=float(cfg[\"lr\"]),\\n            weight_decay=float(cfg[\"weight_decay\"]),\\n        )\\n\\n    steps_per_epoch = max(1, math.ceil(len(train_loader) / accumulation))\\n    total_steps = int(cfg[\"epochs\"]) * steps_per_epoch\\n    scheduler = torch.optim.lr_scheduler.OneCycleLR(\\n        optimizer,\\n        max_lr=float(cfg[\"lr\"]),\\n        epochs=int(cfg[\"epochs\"]),\\n        steps_per_epoch=steps_per_epoch,\\n    )\\n    scaler = rt.make_scaler(device)\\n    clip_norm = float(cfg.get(\"grad_clip_norm\", 1.0))\\n    schedule_steps = 0\\n    history = []\\n    start_epoch = 1\\n    best_macro_f1 = -1.0\\n    no_improvement = 0\\n    patience = int(cfg.get(\"early_stopping_patience\", 0))\\n    last_checkpoint = work / \"last_checkpoint.pt\"\\n\\n    if bool(cfg.get(\"resume_from_checkpoint\", True)) and last_checkpoint.exists():\\n        saved = torch.load(last_checkpoint, map_location=device, weights_only=False)\\n        model.module.load_state_dict(saved[\"model_state\"], strict=True)\\n        optimizer.load_state_dict(saved[\"optimizer_state\"])\\n        scheduler.load_state_dict(saved[\"scheduler_state\"])\\n        scaler.load_state_dict(saved[\"scaler_state\"])\\n        history = list(saved.get(\"history\", []))\\n        schedule_steps = int(saved.get(\"schedule_steps\", 0))\\n        best_macro_f1 = float(saved.get(\"best_macro_f1\", -1.0))\\n        no_improvement = int(saved.get(\"no_improvement\", 0))\\n        start_epoch = int(saved.get(\"epoch\", 0)) + 1\\n        if is_main:\\n            print(\\n                f\"[{stage_name}] resumed at epoch {start_epoch}/{cfg[\\'epochs\\']}\",\\n                flush=True,\\n            )\\n    dist.barrier()\\n\\n    valid_loader = None\\n    if is_main:\\n        valid_kwargs = dict(loader_kwargs)\\n        valid_kwargs[\"persistent_workers\"] = False\\n        valid_loader = DataLoader(\\n            rt.CachedRetinaDataset(valid_frame, cfg, train=False),\\n            batch_size=min(int(cfg.get(\"eval_batch_size\", global_batch)), global_batch),\\n            **valid_kwargs,\\n        )\\n        print(\\n            f\"[{stage_name}] world={world} global_batch={per_rank_batch * world} \"\\n            f\"per_rank_batch={per_rank_batch} workers/rank={workers}\",\\n            flush=True,\\n        )\\n\\n    for epoch in range(start_epoch, int(cfg[\"epochs\"]) + 1):\\n        model.train()\\n        sampler.set_epoch(epoch)\\n        optimizer.zero_grad(set_to_none=True)\\n        running_loss = torch.zeros((), device=device)\\n        batch_count = 0\\n        start_time = time.time()\\n\\n        for step, batch in enumerate(train_loader, start=1):\\n            images = batch[\"image\"].to(device, non_blocking=True).contiguous(\\n                memory_format=torch.channels_last\\n            )\\n            targets = batch[\"label\"].to(device, non_blocking=True)\\n            with rt.amp_autocast(device):\\n                raw_loss = criterion(model(images), targets)\\n                loss = raw_loss / accumulation\\n            scaler.scale(loss).backward()\\n            running_loss += raw_loss.detach()\\n            batch_count += 1\\n\\n            if step % accumulation == 0 or step == len(train_loader):\\n                scaler.unscale_(optimizer)\\n                torch.nn.utils.clip_grad_norm_(model.parameters(), clip_norm)\\n                old_scale = scaler.get_scale()\\n                scaler.step(optimizer)\\n                scaler.update()\\n                optimizer.zero_grad(set_to_none=True)\\n                if scaler.get_scale() >= old_scale and schedule_steps < total_steps:\\n                    scheduler.step()\\n                    schedule_steps += 1\\n\\n            if is_main and step % int(cfg.get(\"log_every\", 25)) == 0:\\n                print(\\n                    f\"  {stage_name} epoch {epoch} {step}/{len(train_loader)} \"\\n                    f\"local_loss={(running_loss / max(1, batch_count)).item():.4f}\",\\n                    flush=True,\\n                )\\n\\n        loss_packet = torch.tensor(\\n            [running_loss.item(), float(batch_count)],\\n            dtype=torch.float64,\\n            device=device,\\n        )\\n        dist.all_reduce(loss_packet, op=dist.ReduceOp.SUM)\\n        global_train_loss = float(\\n            loss_packet[0].item() / max(1.0, loss_packet[1].item())\\n        )\\n        dist.barrier()\\n\\n        stop = torch.zeros(1, device=device)\\n        if is_main:\\n            metrics = evaluate(model.module, valid_loader, device)\\n            record = {\\n                \"epoch\": int(epoch),\\n                \"train_loss\": global_train_loss,\\n                \"elapsed_minutes\": (time.time() - start_time) / 60.0,\\n                **{f\"val_{key}\": value for key, value in metrics.items()},\\n            }\\n            history.append(record)\\n            epoch_checkpoint = work / f\"epoch_{epoch:03d}.pt\"\\n            torch.save(\\n                {\\n                    \"model_state\": model.module.state_dict(),\\n                    \"config\": cfg,\\n                    \"epoch\": epoch,\\n                    \"validation_metrics\": metrics,\\n                },\\n                epoch_checkpoint,\\n            )\\n            print(\\n                f\"[{stage_name}] epoch {epoch} train_loss={global_train_loss:.4f} \"\\n                f\"model_valid={metrics}\",\\n                flush=True,\\n            )\\n\\n            current_macro_f1 = float(metrics[\"macro_f1\"])\\n            if current_macro_f1 > best_macro_f1 + 1e-6:\\n                best_macro_f1 = current_macro_f1\\n                no_improvement = 0\\n            else:\\n                no_improvement += 1\\n\\n            torch.save(\\n                {\\n                    \"epoch\": int(epoch),\\n                    \"model_state\": model.module.state_dict(),\\n                    \"optimizer_state\": optimizer.state_dict(),\\n                    \"scheduler_state\": scheduler.state_dict(),\\n                    \"scaler_state\": scaler.state_dict(),\\n                    \"schedule_steps\": schedule_steps,\\n                    \"history\": history,\\n                    \"best_macro_f1\": best_macro_f1,\\n                    \"no_improvement\": no_improvement,\\n                    \"config\": cfg,\\n                },\\n                last_checkpoint,\\n            )\\n            pd.DataFrame(history).to_csv(work / \"training_history.csv\", index=False)\\n\\n            if patience > 0 and no_improvement >= patience:\\n                stop.fill_(1)\\n                print(\\n                    f\"[{stage_name}] early stopping after {patience} \"\\n                    \"model_valid macro-F1 misses.\",\\n                    flush=True,\\n                )\\n\\n        dist.broadcast(stop, src=0)\\n        if stop.item() >= 1:\\n            break\\n\\n    if is_main:\\n        selected = rt.select_guarded_epoch(\\n            history,\\n            accuracy_guard=float(cfg.get(\"accuracy_guard\", 0.01)),\\n        )\\n        selected_epoch = int(selected[\"epoch\"])\\n        selected_path = work / f\"epoch_{selected_epoch:03d}.pt\"\\n        shutil.copy2(selected_path, work / \"best_model_weights.pt\")\\n        with open(work / \"selection.json\", \"w\") as handle:\\n            json.dump(selected, handle, indent=2)\\n        print(\\n            f\"[{stage_name}] selected epoch {selected_epoch} by guarded macro-F1: \"\\n            f\"{selected}\",\\n            flush=True,\\n        )\\n\\n    dist.barrier()\\n    dist.destroy_process_group()\\n\\n\\nif __name__ == \"__main__\":\\n    main()\\n', 'dr_inference_corrected.py': 'from __future__ import annotations\\n\\nimport hashlib\\nimport json\\nfrom pathlib import Path\\n\\nimport numpy as np\\nimport pandas as pd\\nimport torch\\n\\nimport dr_runtime_corrected as rt\\n\\ntry:\\n    import cv2\\nexcept ImportError as exc:\\n    raise RuntimeError(\"OpenCV is required by the corrected preprocessing pipeline.\") from exc\\n\\nfrom PIL import Image\\n\\n\\n_REDUCED_FLAGS = {\\n    1: cv2.IMREAD_COLOR,\\n    2: cv2.IMREAD_REDUCED_COLOR_2,\\n    4: cv2.IMREAD_REDUCED_COLOR_4,\\n    8: cv2.IMREAD_REDUCED_COLOR_8,\\n}\\n\\n\\ndef read_rgb_decimated(path: str, min_side: int = 768) -> np.ndarray:\\n    suffix = Path(path).suffix.lower()\\n    if suffix in (\".jpg\", \".jpeg\"):\\n        short_side = 0\\n        try:\\n            with Image.open(path) as image:\\n                short_side = min(image.size)\\n        except Exception:\\n            pass\\n        reduction = 1\\n        while (\\n            short_side\\n            and short_side // (reduction * 2) >= int(min_side)\\n            and reduction < 8\\n        ):\\n            reduction *= 2\\n        image = cv2.imread(path, _REDUCED_FLAGS[reduction])\\n        if image is not None:\\n            return cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\\n    return rt.read_rgb_image(path)\\n\\n\\ndef crop_fundus_field(\\n    image: np.ndarray,\\n    threshold: int = 7,\\n    margin_ratio: float = 0.02,\\n    preview: int = 512,\\n) -> np.ndarray:\\n    height, width = image.shape[:2]\\n    scale = min(1.0, float(preview) / max(height, width))\\n    if scale < 1.0:\\n        small = cv2.resize(\\n            image,\\n            (max(1, int(width * scale)), max(1, int(height * scale))),\\n            interpolation=cv2.INTER_AREA,\\n        )\\n    else:\\n        small = image\\n        scale = 1.0\\n    gray = cv2.cvtColor(small, cv2.COLOR_RGB2GRAY)\\n    gray = cv2.GaussianBlur(gray, (5, 5), 0)\\n    mask = gray > threshold\\n    if mask.mean() < 0.05:\\n        return image\\n    ys, xs = np.where(mask)\\n    if not len(ys) or not len(xs):\\n        return image\\n    y0, y1 = int(ys.min() / scale), int((ys.max() + 1) / scale)\\n    x0, x1 = int(xs.min() / scale), int((xs.max() + 1) / scale)\\n    margin = int(max(y1 - y0, x1 - x0) * margin_ratio)\\n    return image[\\n        max(0, y0 - margin) : min(height, y1 + margin),\\n        max(0, x0 - margin) : min(width, x1 + margin),\\n    ]\\n\\n\\ndef pad_to_square(image: np.ndarray) -> np.ndarray:\\n    height, width = image.shape[:2]\\n    size = max(height, width)\\n    canvas = np.zeros((size, size, 3), dtype=np.uint8)\\n    y0 = (size - height) // 2\\n    x0 = (size - width) // 2\\n    canvas[y0 : y0 + height, x0 : x0 + width] = image\\n    return canvas\\n\\n\\ndef fundus_mask(image: np.ndarray) -> np.ndarray:\\n    gray = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)\\n    mask = (gray > 8).astype(np.uint8)\\n    count, labels, stats, _ = cv2.connectedComponentsWithStats(mask, 8)\\n    if count > 1:\\n        largest = 1 + int(np.argmax(stats[1:, cv2.CC_STAT_AREA]))\\n        mask = (labels == largest).astype(np.uint8)\\n    mask = cv2.morphologyEx(mask, cv2.MORPH_CLOSE, np.ones((5, 5), np.uint8))\\n    return mask.astype(bool)\\n\\n\\ndef preprocess_raw_image(path: str, cfg: dict) -> np.ndarray:\\n    size = int(cfg[\"image_size\"])\\n    image = read_rgb_decimated(path, int(cfg.get(\"cache_decode_min_side\", 768)))\\n    image = crop_fundus_field(\\n        image,\\n        preview=int(cfg.get(\"crop_preview_size\", 512)),\\n    )\\n    image = pad_to_square(image)\\n    image = cv2.resize(image, (size, size), interpolation=cv2.INTER_AREA)\\n    mask = fundus_mask(image)\\n    sigma = cfg.get(\"ben_graham_sigma\") or max(3, size // 32)\\n    blurred = cv2.GaussianBlur(image, (0, 0), max(1, int(sigma)))\\n    enhanced = cv2.addWeighted(image, 4.0, blurred, -4.0, 128.0)\\n    enhanced = np.clip(enhanced, 0, 255).astype(np.uint8)\\n    enhanced[~mask] = 0\\n    lab = cv2.cvtColor(enhanced, cv2.COLOR_RGB2LAB)\\n    lightness, a, b = cv2.split(lab)\\n    clahe = cv2.createCLAHE(\\n        clipLimit=float(cfg.get(\"clahe_clip_limit\", 2.0)),\\n        tileGridSize=(\\n            int(cfg.get(\"clahe_tile_grid\", 8)),\\n            int(cfg.get(\"clahe_tile_grid\", 8)),\\n        ),\\n    )\\n    enhanced = cv2.cvtColor(\\n        cv2.merge((clahe.apply(lightness), a, b)),\\n        cv2.COLOR_LAB2RGB,\\n    )\\n    blend = float(cfg.get(\"preprocess_blend\", 0.40))\\n    output = np.clip(\\n        (1.0 - blend) * image.astype(np.float32)\\n        + blend * enhanced.astype(np.float32),\\n        0,\\n        255,\\n    ).astype(np.uint8)\\n    output[~mask] = 0\\n    success, encoded = cv2.imencode(\\n        \".jpg\",\\n        cv2.cvtColor(output, cv2.COLOR_RGB2BGR),\\n        [\\n            int(cv2.IMWRITE_JPEG_QUALITY),\\n            int(cfg.get(\"cache_jpeg_quality\", 95)),\\n            int(cv2.IMWRITE_JPEG_OPTIMIZE),\\n            1,\\n        ],\\n    )\\n    if not success:\\n        raise RuntimeError(f\"Failed to reproduce cache JPEG encoding for {path}\")\\n    decoded = cv2.imdecode(encoded, cv2.IMREAD_COLOR)\\n    if decoded is None:\\n        raise RuntimeError(f\"Failed to reproduce cache JPEG decoding for {path}\")\\n    return cv2.cvtColor(decoded, cv2.COLOR_BGR2RGB)\\n\\n\\ndef _sha256_file(path: Path) -> str:\\n    digest = hashlib.sha256()\\n    with open(path, \"rb\") as handle:\\n        for chunk in iter(lambda: handle.read(1024 * 1024), b\"\"):\\n            digest.update(chunk)\\n    return digest.hexdigest()\\n\\n\\ndef _tta_inputs(images: torch.Tensor, views: list[str]) -> list[torch.Tensor]:\\n    batches = [images]\\n    if \"hflip\" in views:\\n        batches.append(torch.flip(images, dims=[3]))\\n    if \"vflip\" in views:\\n        batches.append(torch.flip(images, dims=[2]))\\n    if \"hvflip\" in views:\\n        batches.append(torch.flip(images, dims=[2, 3]))\\n    return batches\\n\\n\\ndef _load_model(checkpoint: Path, cfg: dict, device: torch.device):\\n    model_cfg = dict(cfg)\\n    model_cfg[\"pretrained\"] = False\\n    model_cfg[\"require_pretrained\"] = False\\n    model = rt.build_model(model_cfg, pretrained=False).to(device)\\n    saved = torch.load(checkpoint, map_location=device, weights_only=False)\\n    model.load_state_dict(saved[\"model_state\"], strict=True)\\n    model.eval()\\n    return model\\n\\n\\n@torch.inference_mode()\\ndef _predict_component(\\n    paths: list[str],\\n    checkpoint: Path,\\n    cfg: dict,\\n    device: torch.device,\\n) -> np.ndarray:\\n    model = _load_model(checkpoint, cfg, device)\\n    transform = rt.FastRetinaTransform(cfg, train=False)\\n    batch_size = int(cfg.get(\"inference_batch_size\", 32))\\n    views = list(cfg.get(\"tta_views\", [])) if cfg.get(\"use_tta\", True) else []\\n    chunks = []\\n    for start in range(0, len(paths), batch_size):\\n        images = [\\n            transform(preprocess_raw_image(path, cfg))\\n            for path in paths[start : start + batch_size]\\n        ]\\n        tensor = torch.stack(images).to(device, non_blocking=True)\\n        accumulated = None\\n        for augmented in _tta_inputs(tensor, views):\\n            with rt.amp_autocast(device):\\n                logits = model(augmented)\\n            probabilities = torch.softmax(logits.float(), dim=1)\\n            accumulated = (\\n                probabilities\\n                if accumulated is None\\n                else accumulated + probabilities\\n            )\\n        chunks.append((accumulated / (1 + len(views))).cpu().numpy())\\n    del model\\n    if device.type == \"cuda\":\\n        torch.cuda.empty_cache()\\n    return np.concatenate(chunks)\\n\\n\\ndef _normalize(probabilities: np.ndarray) -> np.ndarray:\\n    probabilities = np.clip(np.asarray(probabilities, dtype=np.float64), 1e-12, 1.0)\\n    return probabilities / probabilities.sum(axis=1, keepdims=True)\\n\\n\\ndef _main_probabilities(component_probs: dict[str, np.ndarray], source: dict) -> np.ndarray:\\n    output = None\\n    for name, weight in source.items():\\n        contribution = float(weight) * component_probs[name]\\n        output = contribution if output is None else output + contribution\\n    return _normalize(output)\\n\\n\\ndef apply_locked_fusion(\\n    main_probs: np.ndarray,\\n    severity_probs: np.ndarray,\\n    gate_probs: np.ndarray | None,\\n    disease_source: str,\\n    params: dict,\\n) -> tuple[np.ndarray, np.ndarray]:\\n    main = _normalize(main_probs)\\n    severity = _normalize(severity_probs)\\n    main_relative = _normalize(main[:, 1:])\\n    relative = _normalize(\\n        (1.0 - float(params[\"expert_weight\"])) * main_relative\\n        + float(params[\"expert_weight\"]) * severity\\n    )\\n    logits = np.log(relative) / float(params[\"temp\"])\\n    logits += np.asarray(params[\"class_bias\"], dtype=np.float64).reshape(1, 4)\\n    relative = _normalize(np.exp(logits - logits.max(axis=1, keepdims=True)))\\n\\n    main_disease = 1.0 - main[:, 0]\\n    if disease_source == \"main\":\\n        disease_confidence = main_disease\\n    elif disease_source == \"gate\":\\n        if gate_probs is None:\\n            raise RuntimeError(\"The locked candidate requires the binary gate.\")\\n        disease_confidence = gate_probs[:, 1]\\n    elif disease_source == \"main_gate_50\":\\n        if gate_probs is None:\\n            raise RuntimeError(\"The locked candidate requires the binary gate.\")\\n        disease_confidence = 0.5 * main_disease + 0.5 * gate_probs[:, 1]\\n    else:\\n        raise ValueError(f\"Unknown disease source: {disease_source}\")\\n\\n    disease_confidence = np.clip(\\n        disease_confidence + float(params[\"disease_bias\"]),\\n        0.0,\\n        1.0,\\n    )\\n    zero_confidence = np.clip(\\n        main[:, 0] + float(params[\"zero_bias\"]),\\n        0.0,\\n        1.0,\\n    )\\n    output = np.zeros((len(main), 5), dtype=np.float64)\\n    output[:, 0] = 1.0 - disease_confidence\\n    output[:, 1:] = disease_confidence.reshape(-1, 1) * relative\\n\\n    grades = np.where(\\n        disease_confidence >= float(params[\"disease_threshold\"]),\\n        relative.argmax(axis=1) + 1,\\n        0,\\n    )\\n    zero_override = zero_confidence >= float(params[\"zero_override\"])\\n    grades[zero_override] = 0\\n    output[zero_override] = 0.0\\n    output[zero_override, 0] = 1.0\\n\\n    severity_confidence = relative.max(axis=1)\\n    severity_grade = relative.argmax(axis=1) + 1\\n    rescue = (\\n        (severity_confidence >= float(params[\"rescue_threshold\"]))\\n        & (\\n            disease_confidence\\n            >= float(params[\"rescue_disease_threshold\"])\\n        )\\n        & ~zero_override\\n    )\\n    grades[rescue] = severity_grade[rescue]\\n    return _normalize(output), grades.astype(int)\\n\\n\\ndef predict_paths(paths, bundle_dir, device: str | None = None) -> pd.DataFrame:\\n    paths = [str(path) for path in paths]\\n    if not paths:\\n        return pd.DataFrame(\\n            columns=[\"path\", \"prob_0\", \"prob_1\", \"prob_2\", \"prob_3\", \"prob_4\", \"grade\"]\\n        )\\n    bundle_dir = Path(bundle_dir)\\n    metadata = json.load(open(bundle_dir / \"bundle.json\"))\\n    target = torch.device(\\n        device or (\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\\n    )\\n\\n    component_probs = {}\\n    for name, component in metadata[\"components\"].items():\\n        checkpoint = bundle_dir / component[\"checkpoint\"]\\n        if _sha256_file(checkpoint) != component[\"sha256\"]:\\n            raise RuntimeError(f\"Inference bundle checkpoint hash mismatch for {name!r}.\")\\n        component_probs[name] = _predict_component(\\n            paths,\\n            checkpoint,\\n            component[\"config\"],\\n            target,\\n        )\\n\\n    main = _main_probabilities(component_probs, metadata[\"candidate\"][\"main_source\"])\\n    severity = component_probs[\"severity\"]\\n    gate = component_probs.get(\"gate\")\\n    probabilities, grades = apply_locked_fusion(\\n        main,\\n        severity,\\n        gate,\\n        metadata[\"candidate\"][\"disease_source\"],\\n        metadata[\"candidate\"][\"params\"],\\n    )\\n    frame = pd.DataFrame({\"path\": paths})\\n    for class_index in range(5):\\n        frame[f\"prob_{class_index}\"] = probabilities[:, class_index]\\n    frame[\"grade\"] = grades\\n    return frame\\n', 'dr_selection_corrected.py': 'from __future__ import annotations\\n\\nimport hashlib\\nimport json\\nimport shutil\\nfrom copy import deepcopy\\nfrom pathlib import Path\\n\\nimport numpy as np\\nimport pandas as pd\\nimport torch\\nfrom sklearn.metrics import classification_report, confusion_matrix\\nfrom sklearn.model_selection import StratifiedGroupKFold\\nfrom torch.utils.data import DataLoader\\n\\nimport dr_inference_corrected as inference\\nimport dr_runtime_corrected as rt\\n\\n\\ndef normalize(probabilities: np.ndarray) -> np.ndarray:\\n    probabilities = np.clip(np.asarray(probabilities, dtype=np.float64), 1e-12, 1.0)\\n    return probabilities / probabilities.sum(axis=1, keepdims=True)\\n\\n\\ndef tta_inputs(images: torch.Tensor, views: list[str]) -> list[torch.Tensor]:\\n    batches = [images]\\n    if \"hflip\" in views:\\n        batches.append(torch.flip(images, dims=[3]))\\n    if \"vflip\" in views:\\n        batches.append(torch.flip(images, dims=[2]))\\n    if \"hvflip\" in views:\\n        batches.append(torch.flip(images, dims=[2, 3]))\\n    return batches\\n\\n\\n@torch.inference_mode()\\ndef collect_probabilities(\\n    checkpoint: str | Path,\\n    cfg: dict,\\n    paths,\\n    device: torch.device | None = None,\\n) -> np.ndarray:\\n    device = device or torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\\n    model_cfg = dict(cfg)\\n    model_cfg[\"pretrained\"] = False\\n    model_cfg[\"require_pretrained\"] = False\\n    model = rt.build_model(model_cfg, pretrained=False).to(device)\\n    saved = torch.load(checkpoint, map_location=device, weights_only=False)\\n    model.load_state_dict(saved[\"model_state\"], strict=True)\\n    model.eval()\\n\\n    dataset = rt.UnlabeledCachedRetinaDataset(paths, model_cfg)\\n    loader = DataLoader(\\n        dataset,\\n        batch_size=int(model_cfg.get(\"eval_batch_size\", 96)),\\n        shuffle=False,\\n        num_workers=max(0, int(model_cfg.get(\"num_workers\", 4))),\\n        pin_memory=device.type == \"cuda\",\\n        persistent_workers=False,\\n        worker_init_fn=rt.worker_init if int(model_cfg.get(\"num_workers\", 4)) > 0 else None,\\n    )\\n    views = list(model_cfg.get(\"tta_views\", [])) if model_cfg.get(\"use_tta\", True) else []\\n    chunks = []\\n    for batch in loader:\\n        images = batch[\"image\"].to(device, non_blocking=True)\\n        accumulated = None\\n        for augmented in tta_inputs(images, views):\\n            with rt.amp_autocast(device):\\n                logits = model(augmented)\\n            probabilities = torch.softmax(logits.float(), dim=1)\\n            accumulated = probabilities if accumulated is None else accumulated + probabilities\\n        chunks.append((accumulated / (1 + len(views))).cpu().numpy())\\n    del model, loader\\n    if device.type == \"cuda\":\\n        torch.cuda.empty_cache()\\n    return np.concatenate(chunks)\\n\\n\\ndef compose_main(component_probs: dict[str, np.ndarray], weights: dict[str, float]):\\n    result = None\\n    for name, weight in weights.items():\\n        contribution = float(weight) * component_probs[name]\\n        result = contribution if result is None else result + contribution\\n    return normalize(result)\\n\\n\\ndef predict_candidate(component_probs: dict[str, np.ndarray], candidate: dict):\\n    main = compose_main(component_probs, candidate[\"main_source\"])\\n    gate = component_probs.get(\"gate\")\\n    return inference.apply_locked_fusion(\\n        main,\\n        component_probs[\"severity\"],\\n        gate,\\n        candidate[\"disease_source\"],\\n        candidate[\"params\"],\\n    )\\n\\n\\ndef calibration_fold_ids(labels, groups, seed: int) -> np.ndarray:\\n    splitter = StratifiedGroupKFold(\\n        n_splits=5,\\n        shuffle=True,\\n        random_state=int(seed),\\n    )\\n    folds = np.full(len(labels), -1, dtype=np.int64)\\n    dummy = np.zeros(len(labels), dtype=np.int8)\\n    for fold, (_, validation_indices) in enumerate(\\n        splitter.split(dummy, labels, groups)\\n    ):\\n        folds[validation_indices] = fold\\n    if (folds < 0).any():\\n        raise AssertionError(\"Calibration fold assignment is incomplete.\")\\n    expected_classes = set(np.asarray(labels, dtype=np.int64).tolist())\\n    for fold in range(5):\\n        observed_classes = set(\\n            np.asarray(labels, dtype=np.int64)[folds == fold].tolist()\\n        )\\n        if observed_classes != expected_classes:\\n            raise AssertionError(\\n                f\"Calibration fold {fold} has classes {sorted(observed_classes)}, \"\\n                f\"expected {sorted(expected_classes)}.\"\\n            )\\n    return folds\\n\\n\\ndef evaluate_candidate(\\n    component_probs: dict[str, np.ndarray],\\n    labels: np.ndarray,\\n    folds: np.ndarray,\\n    candidate: dict,\\n) -> dict:\\n    _, predictions = predict_candidate(component_probs, candidate)\\n    metrics = rt.calculate_metrics(labels, predictions)\\n    result = {**metrics}\\n    for fold in range(5):\\n        mask = folds == fold\\n        fold_metrics = rt.calculate_metrics(labels[mask], predictions[mask])\\n        for name, value in fold_metrics.items():\\n            result[f\"fold_{fold}_{name}\"] = value\\n    return result\\n\\n\\ndef candidate_params(rng: np.random.Generator, count: int) -> list[dict]:\\n    bias_profiles = [\\n        [0.00, 0.00, 0.00, 0.00],\\n        [0.20, -0.05, 0.35, 0.25],\\n        [0.35, -0.10, 0.55, 0.40],\\n        [0.50, -0.15, 0.80, 0.55],\\n    ]\\n    manual = [\\n        {\\n            \"expert_weight\": 0.75,\\n            \"disease_threshold\": 0.50,\\n            \"disease_bias\": 0.10,\\n            \"zero_bias\": -0.10,\\n            \"zero_override\": 0.85,\\n            \"temp\": 1.00,\\n            \"rescue_threshold\": 0.70,\\n            \"rescue_disease_threshold\": 0.20,\\n            \"class_bias\": [0.20, -0.05, 0.35, 0.25],\\n        },\\n        {\\n            \"expert_weight\": 1.00,\\n            \"disease_threshold\": 0.35,\\n            \"disease_bias\": 0.00,\\n            \"zero_bias\": 0.00,\\n            \"zero_override\": 0.90,\\n            \"temp\": 1.00,\\n            \"rescue_threshold\": 0.70,\\n            \"rescue_disease_threshold\": 0.20,\\n            \"class_bias\": [0.20, -0.05, 0.35, 0.25],\\n        },\\n    ]\\n    generated = []\\n    for _ in range(int(count)):\\n        profile = np.asarray(bias_profiles[rng.integers(0, len(bias_profiles))])\\n        generated.append(\\n            {\\n                \"expert_weight\": float(rng.uniform(0.0, 1.0)),\\n                \"disease_threshold\": float(rng.uniform(0.20, 0.70)),\\n                \"disease_bias\": float(rng.uniform(-0.15, 0.20)),\\n                \"zero_bias\": float(rng.uniform(-0.30, 0.20)),\\n                \"zero_override\": float(rng.uniform(0.75, 0.98)),\\n                \"temp\": float(rng.uniform(0.70, 1.40)),\\n                \"rescue_threshold\": float(rng.uniform(0.55, 0.90)),\\n                \"rescue_disease_threshold\": float(rng.uniform(0.10, 0.45)),\\n                \"class_bias\": list(profile + rng.normal(0.0, 0.08, 4)),\\n            }\\n        )\\n    return manual + generated\\n\\n\\ndef guarded_mask(table: pd.DataFrame, accuracy_guard: float) -> pd.Series:\\n    overall_floor = float(table[\"accuracy\"].max()) - float(accuracy_guard)\\n    eligible = table[\"accuracy\"] >= overall_floor\\n    for fold in range(5):\\n        column = f\"fold_{fold}_accuracy\"\\n        fold_floor = float(table[column].max()) - float(accuracy_guard)\\n        eligible &= table[column] >= fold_floor\\n    return eligible\\n\\n\\ndef candidate_sort_key(metrics: dict):\\n    return (\\n        float(metrics[\"macro_f1\"]),\\n        float(metrics[\"quadratic_kappa\"]),\\n        float(metrics[\"accuracy\"]),\\n    )\\n\\n\\ndef refine_candidate(\\n    component_probs: dict[str, np.ndarray],\\n    labels: np.ndarray,\\n    folds: np.ndarray,\\n    candidate: dict,\\n    fold_floors: dict,\\n    overall_floor: float,\\n) -> tuple[dict, dict]:\\n    current = deepcopy(candidate)\\n    current_metrics = evaluate_candidate(component_probs, labels, folds, current)\\n\\n    def eligible(metrics):\\n        if float(metrics[\"accuracy\"]) < overall_floor:\\n            return False\\n        return all(\\n            float(metrics[f\"fold_{fold}_accuracy\"]) >= fold_floors[fold]\\n            for fold in range(5)\\n        )\\n\\n    for step in (0.08, 0.04, 0.02, 0.01):\\n        improved = True\\n        loops = 0\\n        while improved and loops < 20:\\n            improved = False\\n            loops += 1\\n            for key, low, high in [\\n                (\"expert_weight\", 0.0, 1.0),\\n                (\"disease_threshold\", 0.10, 0.80),\\n                (\"disease_bias\", -0.30, 0.35),\\n                (\"zero_bias\", -0.45, 0.35),\\n                (\"zero_override\", 0.65, 0.99),\\n                (\"temp\", 0.55, 1.60),\\n                (\"rescue_threshold\", 0.45, 0.95),\\n                (\"rescue_disease_threshold\", 0.05, 0.55),\\n            ]:\\n                original = float(current[\"params\"][key])\\n                for direction in (-1.0, 1.0):\\n                    trial = deepcopy(current)\\n                    trial[\"params\"][key] = float(\\n                        np.clip(original + direction * step, low, high)\\n                    )\\n                    metrics = evaluate_candidate(component_probs, labels, folds, trial)\\n                    if eligible(metrics) and candidate_sort_key(metrics) > candidate_sort_key(\\n                        current_metrics\\n                    ):\\n                        current, current_metrics, improved = trial, metrics, True\\n            for class_index in range(4):\\n                original_bias = np.asarray(current[\"params\"][\"class_bias\"], dtype=np.float64)\\n                for direction in (-1.0, 1.0):\\n                    trial = deepcopy(current)\\n                    bias = original_bias.copy()\\n                    bias[class_index] = np.clip(\\n                        bias[class_index] + direction * step * 2.0,\\n                        -1.0,\\n                        1.5,\\n                    )\\n                    trial[\"params\"][\"class_bias\"] = bias.tolist()\\n                    metrics = evaluate_candidate(component_probs, labels, folds, trial)\\n                    if eligible(metrics) and candidate_sort_key(metrics) > candidate_sort_key(\\n                        current_metrics\\n                    ):\\n                        current, current_metrics, improved = trial, metrics, True\\n    return current, current_metrics\\n\\n\\ndef run_guarded_search(\\n    component_probs: dict[str, np.ndarray],\\n    labels: np.ndarray,\\n    groups: np.ndarray,\\n    seed: int,\\n    accuracy_guard: float = 0.01,\\n    random_per_combination: int = 250,\\n):\\n    folds = calibration_fold_ids(labels, groups, seed + 5000)\\n    main_sources = {\\n        \"stage1\": {\"stage1\": 1.0},\\n        \"stage2\": {\"stage2\": 1.0},\\n        \"stage3\": {\"stage3\": 1.0},\\n        \"stage2_40_stage3_60\": {\"stage2\": 0.40, \"stage3\": 0.60},\\n        \"stage1_20_stage2_40_stage3_40\": {\\n            \"stage1\": 0.20,\\n            \"stage2\": 0.40,\\n            \"stage3\": 0.40,\\n        },\\n    }\\n    disease_sources = [\"main\", \"gate\", \"main_gate_50\"]\\n    rng = np.random.default_rng(int(seed) + 60614)\\n    candidates = []\\n    rows = []\\n\\n    for source_name, source_weights in main_sources.items():\\n        for disease_source in disease_sources:\\n            for params in candidate_params(rng, random_per_combination):\\n                candidate = {\\n                    \"main_source_name\": source_name,\\n                    \"main_source\": source_weights,\\n                    \"disease_source\": disease_source,\\n                    \"params\": params,\\n                }\\n                metrics = evaluate_candidate(component_probs, labels, folds, candidate)\\n                candidate_id = len(candidates)\\n                candidates.append(candidate)\\n                rows.append(\\n                    {\\n                        \"candidate_id\": candidate_id,\\n                        \"main_source\": source_name,\\n                        \"disease_source\": disease_source,\\n                        **metrics,\\n                    }\\n                )\\n\\n    broad = pd.DataFrame(rows)\\n    eligible_broad = broad[guarded_mask(broad, accuracy_guard)].copy()\\n    if eligible_broad.empty:\\n        raise RuntimeError(\"No broad-search candidate satisfied the accuracy guard.\")\\n    eligible_broad = eligible_broad.sort_values(\\n        [\"macro_f1\", \"quadratic_kappa\", \"accuracy\"],\\n        ascending=False,\\n    )\\n    fold_floors = {\\n        fold: float(broad[f\"fold_{fold}_accuracy\"].max()) - float(accuracy_guard)\\n        for fold in range(5)\\n    }\\n    overall_floor = float(broad[\"accuracy\"].max()) - float(accuracy_guard)\\n\\n    refined_candidates = []\\n    refined_rows = []\\n    for candidate_id in eligible_broad.head(10)[\"candidate_id\"].astype(int):\\n        candidate, metrics = refine_candidate(\\n            component_probs,\\n            labels,\\n            folds,\\n            candidates[candidate_id],\\n            fold_floors,\\n            overall_floor,\\n        )\\n        refined_id = len(refined_candidates)\\n        refined_candidates.append(candidate)\\n        refined_rows.append(\\n            {\\n                \"candidate_id\": len(candidates) + refined_id,\\n                \"main_source\": candidate[\"main_source_name\"],\\n                \"disease_source\": candidate[\"disease_source\"],\\n                **metrics,\\n            }\\n        )\\n\\n    all_candidates = candidates + refined_candidates\\n    table = pd.concat([broad, pd.DataFrame(refined_rows)], ignore_index=True)\\n    eligible = table[guarded_mask(table, accuracy_guard)].copy()\\n    if eligible.empty:\\n        raise RuntimeError(\\n            \"No calibration candidate remained within the accuracy guard globally \"\\n            \"and on every patient-group fold.\"\\n        )\\n    eligible = eligible.sort_values(\\n        [\"macro_f1\", \"quadratic_kappa\", \"accuracy\"],\\n        ascending=False,\\n    )\\n    selected_row = eligible.iloc[0]\\n    selected = deepcopy(all_candidates[int(selected_row[\"candidate_id\"])])\\n    selected[\"calibration_metrics\"] = {\\n        key: float(selected_row[key])\\n        for key in [\"accuracy\", \"macro_f1\", \"weighted_f1\", \"quadratic_kappa\"]\\n    }\\n    selected[\"calibration_fold_metrics\"] = {\\n        f\"fold_{fold}\": {\\n            metric: float(selected_row[f\"fold_{fold}_{metric}\"])\\n            for metric in [\"accuracy\", \"macro_f1\", \"weighted_f1\", \"quadratic_kappa\"]\\n        }\\n        for fold in range(5)\\n    }\\n    selected[\"selection_policy\"] = {\\n        \"primary\": \"macro_f1\",\\n        \"accuracy_guard\": float(accuracy_guard),\\n        \"tie_breakers\": [\"quadratic_kappa\", \"accuracy\"],\\n        \"calibration_group_folds\": 5,\\n    }\\n    return table, selected\\n\\n\\ndef json_hash(payload: dict) -> str:\\n    encoded = json.dumps(payload, sort_keys=True, separators=(\",\", \":\")).encode(\"utf-8\")\\n    return hashlib.sha256(encoded).hexdigest()\\n\\n\\ndef lock_candidate(\\n    selected: dict,\\n    component_paths: dict[str, Path],\\n    metadata: dict,\\n    output_dir: Path,\\n    reset: bool = False,\\n) -> dict:\\n    payload = {\\n        \"candidate\": selected,\\n        \"component_sha256\": {\\n            name: rt.sha256_file(path) for name, path in component_paths.items()\\n        },\\n        \"preprocessing_fingerprint\": metadata[\"preprocessing_fingerprint\"],\\n        \"split_fingerprint\": metadata[\"split_fingerprint\"],\\n    }\\n    payload[\"lock_sha256\"] = json_hash(payload)\\n    path = output_dir / \"candidate_lock.json\"\\n    if path.exists():\\n        existing = json.load(open(path))\\n        if existing.get(\"lock_sha256\") != payload[\"lock_sha256\"] and not reset:\\n            raise RuntimeError(\\n                \"candidate_lock.json already contains a different candidate. \"\\n                \"Set CFG[\\'reset_candidate_lock\\']=True only for an intentional new experiment.\"\\n            )\\n        if existing.get(\"lock_sha256\") == payload[\"lock_sha256\"]:\\n            return existing\\n    with open(path, \"w\") as handle:\\n        json.dump(payload, handle, indent=2)\\n    return payload\\n\\n\\ndef export_inference_bundle(\\n    lock: dict,\\n    component_paths: dict[str, Path],\\n    component_configs: dict[str, dict],\\n    output_dir: Path,\\n    runtime_source: Path,\\n    inference_source: Path,\\n) -> Path:\\n    bundle_dir = output_dir / \"inference_bundle\"\\n    models_dir = bundle_dir / \"models\"\\n    models_dir.mkdir(parents=True, exist_ok=True)\\n    components = {}\\n    for name, checkpoint in component_paths.items():\\n        relative = Path(\"models\") / f\"{name}.pt\"\\n        checkpoint_sha256 = rt.sha256_file(checkpoint)\\n        locked_sha256 = lock[\"component_sha256\"][name]\\n        if checkpoint_sha256 != locked_sha256:\\n            raise RuntimeError(\\n                f\"Checkpoint {name!r} changed after the candidate was locked.\"\\n            )\\n        shutil.copy2(checkpoint, bundle_dir / relative)\\n        if rt.sha256_file(bundle_dir / relative) != checkpoint_sha256:\\n            raise RuntimeError(f\"Bundled checkpoint hash mismatch for {name!r}.\")\\n        config = dict(component_configs[name])\\n        config[\"pretrained\"] = False\\n        config[\"require_pretrained\"] = False\\n        components[name] = {\\n            \"checkpoint\": str(relative),\\n            \"sha256\": checkpoint_sha256,\\n            \"config\": config,\\n        }\\n    shutil.copy2(runtime_source, bundle_dir / \"dr_runtime_corrected.py\")\\n    shutil.copy2(inference_source, bundle_dir / \"dr_inference_corrected.py\")\\n    bundle = {\\n        \"bundle_version\": \"corrected-2026-06-14-v1\",\\n        \"candidate_lock_sha256\": lock[\"lock_sha256\"],\\n        \"preprocessing_fingerprint\": lock[\"preprocessing_fingerprint\"],\\n        \"split_fingerprint\": lock[\"split_fingerprint\"],\\n        \"candidate\": {\\n            \"main_source\": lock[\"candidate\"][\"main_source\"],\\n            \"disease_source\": lock[\"candidate\"][\"disease_source\"],\\n            \"params\": lock[\"candidate\"][\"params\"],\\n        },\\n        \"components\": components,\\n        \"class_names\": {\\n            \"0\": \"No DR\",\\n            \"1\": \"Mild NPDR\",\\n            \"2\": \"Moderate NPDR\",\\n            \"3\": \"Severe NPDR\",\\n            \"4\": \"Proliferative DR\",\\n        },\\n    }\\n    with open(bundle_dir / \"bundle.json\", \"w\") as handle:\\n        json.dump(bundle, handle, indent=2)\\n    return bundle_dir\\n\\n\\ndef bootstrap_intervals(\\n    labels: np.ndarray,\\n    predictions: np.ndarray,\\n    groups: np.ndarray,\\n    iterations: int,\\n    seed: int,\\n) -> dict:\\n    rng = np.random.default_rng(seed)\\n    unique_groups = np.unique(groups)\\n    indices_by_group = {\\n        group: np.flatnonzero(groups == group) for group in unique_groups\\n    }\\n    samples = {name: [] for name in rt.calculate_metrics(labels, predictions)}\\n    for _ in range(int(iterations)):\\n        sampled_groups = rng.choice(unique_groups, size=len(unique_groups), replace=True)\\n        sampled_indices = np.concatenate([indices_by_group[group] for group in sampled_groups])\\n        metrics = rt.calculate_metrics(labels[sampled_indices], predictions[sampled_indices])\\n        for name, value in metrics.items():\\n            samples[name].append(value)\\n    return {\\n        name: {\\n            \"lower_95\": float(np.percentile(values, 2.5)),\\n            \"upper_95\": float(np.percentile(values, 97.5)),\\n        }\\n        for name, values in samples.items()\\n    }\\n\\n\\ndef evaluate_locked_holdout_once(\\n    holdout_frame: pd.DataFrame,\\n    lock_path: Path,\\n    component_paths: dict[str, Path],\\n    component_configs: dict[str, dict],\\n    output_dir: Path,\\n    reset: bool = False,\\n    bootstrap_iterations: int = 1000,\\n) -> dict:\\n    lock = json.load(open(lock_path))\\n    result_path = output_dir / \"holdout_evaluation.json\"\\n    if result_path.exists():\\n        existing = json.load(open(result_path))\\n        if existing.get(\"candidate_lock_sha256\") == lock[\"lock_sha256\"] and not reset:\\n            return existing\\n        if not reset:\\n            raise RuntimeError(\\n                \"A holdout evaluation already exists for another candidate. \"\\n                \"Set CFG[\\'reset_holdout_evaluation\\']=True only for an explicitly audited reset.\"\\n            )\\n\\n    paths = holdout_frame[\"path\"].astype(str).tolist()\\n    component_probs = {\\n        name: collect_probabilities(component_paths[name], component_configs[name], paths)\\n        for name in [\"stage1\", \"stage2\", \"stage3\", \"gate\", \"severity\"]\\n    }\\n    _, predictions = predict_candidate(component_probs, lock[\"candidate\"])\\n\\n    labels = holdout_frame[\"label\"].astype(int).to_numpy()\\n    groups = holdout_frame[\"group_key\"].astype(str).to_numpy()\\n    metrics = rt.calculate_metrics(labels, predictions)\\n    report = classification_report(\\n        labels,\\n        predictions,\\n        labels=[0, 1, 2, 3, 4],\\n        output_dict=True,\\n        zero_division=0,\\n    )\\n    matrix = confusion_matrix(labels, predictions, labels=[0, 1, 2, 3, 4]).tolist()\\n    intervals = bootstrap_intervals(\\n        labels,\\n        predictions,\\n        groups,\\n        iterations=bootstrap_iterations,\\n        seed=20260614,\\n    )\\n    result = {\\n        \"candidate_lock_sha256\": lock[\"lock_sha256\"],\\n        \"metrics\": metrics,\\n        \"patient_bootstrap_95_ci\": intervals,\\n        \"classification_report\": report,\\n        \"confusion_matrix\": matrix,\\n        \"historical_non_leaked_reference\": {\\n            \"accuracy\": 0.8620837348752544,\\n            \"macro_f1\": 0.5520043656720068,\\n            \"quadratic_kappa\": 0.7670249601098079,\\n        },\\n    }\\n    with open(result_path, \"w\") as handle:\\n        json.dump(result, handle, indent=2)\\n    pd.DataFrame(\\n        {\\n            \"orig_path\": holdout_frame[\"orig_path\"].astype(str).to_numpy(),\\n            \"true\": labels,\\n            \"pred\": predictions,\\n            \"group_key\": groups,\\n        }\\n    ).to_csv(output_dir / \"holdout_predictions.csv\", index=False)\\n    return result\\n', 'dr_training_builder.py': 'from __future__ import annotations\\n\\nimport json\\nimport os\\nimport subprocess\\nimport sys\\nfrom pathlib import Path\\n\\nimport numpy as np\\nimport pandas as pd\\nimport torch\\nfrom sklearn.metrics import classification_report, confusion_matrix\\n\\nimport dr_runtime_corrected as rt\\nfrom dr_data_corrected import CLASS_NAMES, prepare_data\\nfrom dr_experts_corrected import (\\n    align_cached_frame,\\n    apply_bias,\\n    predict_with_expert_candidate,\\n    prepare_expert_480_data,\\n    safety_metrics,\\n    score_with_safety,\\n)\\nfrom dr_selection_corrected import collect_probabilities\\n\\n\\ndef normalize_probs(probs: np.ndarray) -> np.ndarray:\\n    probs = np.clip(np.asarray(probs, dtype=np.float64), 1e-12, 1.0)\\n    return probs / probs.sum(axis=1, keepdims=True)\\n\\n\\ndef weighted_probs(component_probs: dict[str, np.ndarray], weights: dict[str, float]) -> np.ndarray:\\n    out = None\\n    for name, weight in weights.items():\\n        contribution = float(weight) * component_probs[name]\\n        out = contribution if out is None else out + contribution\\n    return normalize_probs(out)\\n\\n\\ndef weighted_objective(metrics: dict, weights: dict) -> float:\\n    return sum(float(weights.get(key, 0.0)) * float(metrics[key]) for key in weights)\\n\\n\\ndef coordinate_bias_search(\\n    probs: np.ndarray,\\n    labels: np.ndarray,\\n    objective_weights: dict,\\n    min_accuracy: float | None = None,\\n) -> tuple[np.ndarray, dict]:\\n    classes = probs.shape[1]\\n    bias = np.zeros(classes, dtype=np.float64)\\n    if min_accuracy is None:\\n        min_accuracy = 0.0\\n\\n    def candidate_score(candidate_bias: np.ndarray) -> tuple[float, dict]:\\n        pred = apply_bias(probs, candidate_bias).argmax(axis=1)\\n        metrics = rt.calculate_metrics(labels, pred)\\n        score = weighted_objective(metrics, objective_weights)\\n        if metrics[\"accuracy\"] < min_accuracy:\\n            score -= 20.0 * (min_accuracy - metrics[\"accuracy\"])\\n        return score, metrics\\n\\n    best_score, best_metrics = candidate_score(bias)\\n    for step in (1.5, 1.0, 0.5, 0.25, 0.10, 0.05, 0.02, 0.01):\\n        improved = True\\n        loops = 0\\n        while improved and loops < 30:\\n            improved = False\\n            loops += 1\\n            for cls_idx in range(classes):\\n                for direction in (-1.0, 1.0):\\n                    trial = bias.copy()\\n                    trial[cls_idx] += direction * step\\n                    trial -= trial.mean()\\n                    score, metrics = candidate_score(trial)\\n                    if score > best_score + 1e-10:\\n                        bias, best_score, best_metrics, improved = trial, score, metrics, True\\n    return bias, best_metrics\\n\\n\\ndef generate_simplex_weights(step: float = 0.05) -> list[dict[str, float]]:\\n    values = np.round(np.arange(0.0, 1.0 + 1e-9, step), 4)\\n    weights = []\\n    for w1 in values:\\n        for w2 in values:\\n            w3 = round(1.0 - float(w1) - float(w2), 4)\\n            if w3 < -1e-9:\\n                continue\\n            weights.append({\"stage1\": float(w1), \"stage2\": float(w2), \"stage3\": float(w3)})\\n    return weights\\n\\n\\ndef expert_param_grid() -> list[dict]:\\n    severity_weights = [0.0, 0.25, 0.50, 0.75, 1.0]\\n    zero_options = [(0.0, 0.5)]\\n    for weight in [0.35, 0.70]:\\n        for threshold in [0.45, 0.50, 0.55, 0.60]:\\n            zero_options.append((weight, threshold))\\n    moderate_options = [(0.0, 0.5)]\\n    for weight in [0.35, 0.70]:\\n        for threshold in [0.45, 0.50, 0.55]:\\n            moderate_options.append((weight, threshold))\\n    params = []\\n    for severity_weight in severity_weights:\\n        for zero_weight, zero_threshold in zero_options:\\n            for moderate_weight, moderate_threshold in moderate_options:\\n                params.append({\\n                    \"severity_weight\": float(severity_weight),\\n                    \"zero_mild_weight\": float(zero_weight),\\n                    \"zero_mild_threshold\": float(zero_threshold),\\n                    \"moderate_severe_weight\": float(moderate_weight),\\n                    \"moderate_severe_threshold\": float(moderate_threshold),\\n                })\\n    return params\\n\\n\\nDISABLED_EXPERT_PARAMS = {\\n    \"severity_weight\": 0.0,\\n    \"zero_mild_weight\": 0.0,\\n    \"zero_mild_threshold\": 0.5,\\n    \"moderate_severe_weight\": 0.0,\\n    \"moderate_severe_threshold\": 0.5,\\n}\\n\\n\\nclass DRTrainingBuilder:\\n    def __init__(self, cfg: dict) -> None:\\n        self.cfg = dict(cfg)\\n        self.output_dir = Path(self.cfg[\"output_dir\"])\\n        self.output_dir.mkdir(parents=True, exist_ok=True)\\n        self.sealed_dir = self.output_dir / \"sealed\"\\n        self.trainer_path = Path(\"/kaggle/working/train_stage_corrected.py\")\\n        self.cached_splits: dict[str, pd.DataFrame] = {}\\n        self.split_manifest: pd.DataFrame | None = None\\n        self.duplicate_audit: pd.DataFrame | None = None\\n        self.data_metadata: dict = {}\\n        self.expert_splits: dict = {}\\n        self.expert_metadata: dict = {}\\n        self.holdout_path: Path | None = None\\n        self.expert_holdout_path: Path | None = None\\n        self.component_paths: dict[str, Path] = {}\\n        self.component_configs: dict[str, dict] = {}\\n        self.selected_candidate: dict | None = None\\n        self.calibration_search_df: pd.DataFrame | None = None\\n        self.base_calibration_search_df: pd.DataFrame | None = None\\n        self.holdout_result: dict | None = None\\n\\n    @property\\n    def use_experts(self) -> bool:\\n        return bool(self.cfg.get(\"use_experts\", True))\\n\\n    def write_experiment_config(self) -> Path:\\n        path = self.output_dir / \"experiment_config.json\"\\n        with open(path, \"w\", encoding=\"utf-8\") as handle:\\n            json.dump(self.cfg, handle, indent=2)\\n        return path\\n\\n    def prepare(self) -> dict:\\n        self.write_experiment_config()\\n        all_splits, split_manifest, duplicate_audit, data_metadata = prepare_data(self.cfg)\\n        self.split_manifest = split_manifest\\n        self.duplicate_audit = duplicate_audit\\n        self.data_metadata = data_metadata\\n\\n        self.sealed_dir.mkdir(parents=True, exist_ok=True)\\n        holdout_key = \"holdout\" if \"holdout\" in all_splits else \"test\"\\n        if holdout_key not in all_splits:\\n            raise KeyError(\"prepare_data(CFG) did not return a holdout/test split.\")\\n        self.holdout_path = self.sealed_dir / \"split_holdout.csv\"\\n        all_splits[holdout_key].to_csv(self.holdout_path, index=False)\\n        for name in (\"train\", \"model_valid\", \"calibration\"):\\n            if name not in all_splits:\\n                raise KeyError(f\"prepare_data(CFG) did not return required split {name!r}.\")\\n        self.cached_splits = {name: all_splits[name] for name in (\"train\", \"model_valid\", \"calibration\")}\\n\\n        if self.use_experts:\\n            self.expert_splits, self.expert_metadata = prepare_expert_480_data(\\n                split_manifest,\\n                self.cfg,\\n                self.output_dir,\\n            )\\n            self.expert_holdout_path = self.sealed_dir / \"split_holdout_480_experts.csv\"\\n            self.expert_splits[\"holdout\"].to_csv(self.expert_holdout_path, index=False)\\n        else:\\n            self.expert_splits, self.expert_metadata = {}, {\"skipped\": True, \"reason\": \"use_experts=False\"}\\n\\n        summary = (\\n            split_manifest.groupby([\"partition\", \"label\"])\\n            .size()\\n            .unstack(fill_value=0)\\n            .rename(columns=CLASS_NAMES)\\n        )\\n        summary.to_csv(self.output_dir / \"split_class_summary.csv\")\\n        print(\"Preprocessing fingerprint:\", data_metadata[\"preprocessing_fingerprint\"])\\n        print(\"Split fingerprint:\", data_metadata[\"split_fingerprint\"])\\n        print(\"Sealed holdout path:\", self.holdout_path)\\n        if self.expert_holdout_path is not None:\\n            print(\"480 expert holdout path:\", self.expert_holdout_path)\\n        return {\\n            \"cached_splits\": self.cached_splits,\\n            \"split_manifest\": self.split_manifest,\\n            \"duplicate_audit\": self.duplicate_audit,\\n            \"data_metadata\": self.data_metadata,\\n            \"expert_splits\": self.expert_splits,\\n            \"expert_metadata\": self.expert_metadata,\\n            \"holdout_path\": self.holdout_path,\\n            \"expert_holdout_path\": self.expert_holdout_path,\\n        }\\n\\n    def stage_config(self, stage_name: str, **updates) -> dict:\\n        cfg = {\\n            **self.cfg,\\n            \"stage_name\": stage_name,\\n            \"output_dir\": str(self.output_dir / stage_name),\\n            \"pretrained\": False,\\n            \"require_pretrained\": False,\\n            \"init_mode\": \"strict\",\\n            \"resume_from_checkpoint\": True,\\n            \"accuracy_guard\": self.cfg[\"accuracy_guard\"],\\n        }\\n        cfg.update(updates)\\n        return cfg\\n\\n    def run_ddp_stage(\\n        self,\\n        stage_cfg: dict,\\n        init_checkpoint: Path | None = None,\\n        train_frame: pd.DataFrame | None = None,\\n        valid_frame: pd.DataFrame | None = None,\\n    ) -> Path:\\n        if torch.cuda.device_count() == 0:\\n            raise RuntimeError(\"This notebook is configured for Kaggle GPU/DDP training.\")\\n        if train_frame is None:\\n            train_frame = self.cached_splits[\"train\"]\\n        if valid_frame is None:\\n            valid_frame = self.cached_splits[\"model_valid\"]\\n        work = Path(stage_cfg[\"output_dir\"])\\n        work.mkdir(parents=True, exist_ok=True)\\n        train_frame[[\"path\", \"label\"]].to_csv(work / \"split_train.csv\", index=False)\\n        valid_frame[[\"path\", \"label\"]].to_csv(work / \"split_model_valid.csv\", index=False)\\n        with open(work / \"run_config.json\", \"w\", encoding=\"utf-8\") as handle:\\n            json.dump(stage_cfg, handle, indent=2)\\n        command = [\\n            sys.executable,\\n            \"-m\",\\n            \"torch.distributed.run\",\\n            \"--standalone\",\\n            f\"--nproc_per_node={torch.cuda.device_count()}\",\\n            str(self.trainer_path),\\n            \"--workdir\",\\n            str(work),\\n        ]\\n        if init_checkpoint is not None:\\n            command.extend([\"--init-ckpt\", str(init_checkpoint)])\\n        env = {**os.environ, \"OMP_NUM_THREADS\": \"1\", \"TOKENIZERS_PARALLELISM\": \"false\"}\\n        print(\"Launching:\", \" \".join(command), flush=True)\\n        process = subprocess.Popen(\\n            command,\\n            env=env,\\n            stdout=subprocess.PIPE,\\n            stderr=subprocess.STDOUT,\\n            text=True,\\n            bufsize=1,\\n        )\\n        assert process.stdout is not None\\n        for line in process.stdout:\\n            print(line, end=\"\")\\n        process.wait()\\n        if process.returncode != 0:\\n            raise RuntimeError(f\"{stage_cfg[\\'stage_name\\']} failed with exit code {process.returncode}.\")\\n        checkpoint = work / \"best_model_weights.pt\"\\n        if not checkpoint.exists():\\n            raise FileNotFoundError(f\"Missing selected checkpoint: {checkpoint}\")\\n        return checkpoint\\n\\n    def train_main_stages(self) -> dict[str, Path]:\\n        if not self.cached_splits:\\n            self.prepare()\\n        stage1_cfg = self.stage_config(\\n            \"stage1_inceptionv3\",\\n            pretrained=True,\\n            require_pretrained=True,\\n            num_classes=5,\\n            epochs=8,\\n            lr=1.8e-4,\\n            weight_decay=1.0e-4,\\n            label_smoothing=0.05,\\n            imbalance_mode=\"sampler\",\\n            sampler_power=0.40,\\n            class_weight_power=0.0,\\n            early_stopping_patience=4,\\n            stage_seed_offset=0,\\n        )\\n        stage1_ckpt = self.run_ddp_stage(stage1_cfg)\\n        stage2_cfg = self.stage_config(\\n            \"stage2_inceptionv3_accqwk\",\\n            num_classes=5,\\n            epochs=4,\\n            lr=4.5e-5,\\n            weight_decay=8.0e-5,\\n            label_smoothing=0.02,\\n            imbalance_mode=\"class_weight\",\\n            sampler_power=0.0,\\n            class_weight_power=0.10,\\n            early_stopping_patience=4,\\n            stage_seed_offset=1000,\\n        )\\n        stage2_ckpt = self.run_ddp_stage(stage2_cfg, stage1_ckpt)\\n        stage3_cfg = self.stage_config(\\n            \"stage3_inceptionv3_macro_f1\",\\n            num_classes=5,\\n            epochs=3,\\n            lr=2.8e-5,\\n            weight_decay=1.0e-4,\\n            label_smoothing=0.0,\\n            imbalance_mode=\"both\",\\n            sampler_power=0.12,\\n            class_weight_power=0.18,\\n            early_stopping_patience=3,\\n            stage_seed_offset=2000,\\n        )\\n        stage3_ckpt = self.run_ddp_stage(stage3_cfg, stage2_ckpt)\\n        self.component_paths.update({\"stage1\": stage1_ckpt, \"stage2\": stage2_ckpt, \"stage3\": stage3_ckpt})\\n        self.component_configs.update({\"stage1\": stage1_cfg, \"stage2\": stage2_cfg, \"stage3\": stage3_cfg})\\n        return {\"stage1\": stage1_ckpt, \"stage2\": stage2_ckpt, \"stage3\": stage3_ckpt}\\n\\n    def train_experts(self) -> dict[str, Path]:\\n        if not self.use_experts:\\n            print(\"Skipping 480px experts because CFG[\\'use_experts\\'] is false.\")\\n            return {}\\n        if not self.expert_splits:\\n            self.prepare()\\n        if \"stage3\" not in self.component_paths:\\n            raise RuntimeError(\"Train main stages before training experts.\")\\n        stage3_ckpt = self.component_paths[\"stage3\"]\\n        severity_cfg = self.stage_config(\\n            \"expert_severity_480\",\\n            num_classes=4,\\n            image_size=self.cfg[\"expert_image_size\"],\\n            batch_size=self.cfg[\"expert_batch_size\"],\\n            eval_batch_size=self.cfg[\"expert_eval_batch_size\"],\\n            epochs=3,\\n            lr=3.0e-5,\\n            weight_decay=1.0e-4,\\n            label_smoothing=0.02,\\n            imbalance_mode=\"both\",\\n            sampler_power=0.15,\\n            class_weight_power=0.20,\\n            early_stopping_patience=3,\\n            init_mode=\"compatible\",\\n            stage_seed_offset=3000,\\n        )\\n        severity_ckpt = self.run_ddp_stage(\\n            severity_cfg,\\n            stage3_ckpt,\\n            self.expert_splits[\"severity_480\"][\"train\"],\\n            self.expert_splits[\"severity_480\"][\"model_valid\"],\\n        )\\n        zero_cfg = self.stage_config(\\n            \"expert_zero_mild_480\",\\n            num_classes=2,\\n            image_size=self.cfg[\"expert_image_size\"],\\n            batch_size=self.cfg[\"expert_batch_size\"],\\n            eval_batch_size=self.cfg[\"expert_eval_batch_size\"],\\n            epochs=3,\\n            lr=3.2e-5,\\n            weight_decay=1.0e-4,\\n            label_smoothing=0.01,\\n            imbalance_mode=\"sampler\",\\n            sampler_power=0.25,\\n            class_weight_power=0.0,\\n            early_stopping_patience=3,\\n            init_mode=\"compatible\",\\n            stage_seed_offset=4000,\\n        )\\n        zero_ckpt = self.run_ddp_stage(\\n            zero_cfg,\\n            stage3_ckpt,\\n            self.expert_splits[\"zero_mild_480\"][\"train\"],\\n            self.expert_splits[\"zero_mild_480\"][\"model_valid\"],\\n        )\\n        moderate_cfg = self.stage_config(\\n            \"expert_moderate_severe_480\",\\n            num_classes=2,\\n            image_size=self.cfg[\"expert_image_size\"],\\n            batch_size=self.cfg[\"expert_batch_size\"],\\n            eval_batch_size=self.cfg[\"expert_eval_batch_size\"],\\n            epochs=3,\\n            lr=3.2e-5,\\n            weight_decay=1.0e-4,\\n            label_smoothing=0.01,\\n            imbalance_mode=\"both\",\\n            sampler_power=0.20,\\n            class_weight_power=0.15,\\n            early_stopping_patience=3,\\n            init_mode=\"compatible\",\\n            stage_seed_offset=5000,\\n        )\\n        moderate_ckpt = self.run_ddp_stage(\\n            moderate_cfg,\\n            stage3_ckpt,\\n            self.expert_splits[\"moderate_severe_480\"][\"train\"],\\n            self.expert_splits[\"moderate_severe_480\"][\"model_valid\"],\\n        )\\n        self.component_paths.update({\\n            \"severity_480\": severity_ckpt,\\n            \"zero_mild_480\": zero_ckpt,\\n            \"moderate_severe_480\": moderate_ckpt,\\n        })\\n        self.component_configs.update({\\n            \"severity_480\": severity_cfg,\\n            \"zero_mild_480\": zero_cfg,\\n            \"moderate_severe_480\": moderate_cfg,\\n        })\\n        return {\"severity_480\": severity_ckpt, \"zero_mild_480\": zero_ckpt, \"moderate_severe_480\": moderate_ckpt}\\n\\n    def _collect_calibration_probs(self) -> tuple[dict[str, np.ndarray], np.ndarray]:\\n        calibration_frame = self.cached_splits[\"calibration\"]\\n        calibration_paths = calibration_frame[\"path\"].astype(str).tolist()\\n        labels = calibration_frame[\"label\"].astype(int).to_numpy()\\n        probs = {\\n            name: collect_probabilities(self.component_paths[name], self.component_configs[name], calibration_paths)\\n            for name in [\"stage1\", \"stage2\", \"stage3\"]\\n        }\\n        if self.use_experts and all(name in self.component_paths for name in [\"severity_480\", \"zero_mild_480\", \"moderate_severe_480\"]):\\n            expert_frame = align_cached_frame(self.expert_splits[\"calibration\"], calibration_frame)\\n            expert_paths = expert_frame[\"path\"].astype(str).tolist()\\n            probs.update({\\n                name: collect_probabilities(self.component_paths[name], self.component_configs[name], expert_paths)\\n                for name in [\"severity_480\", \"zero_mild_480\", \"moderate_severe_480\"]\\n            })\\n        return probs, labels\\n\\n    def calibrate(self) -> dict:\\n        if not all(name in self.component_paths for name in [\"stage1\", \"stage2\", \"stage3\"]):\\n            raise RuntimeError(\"Train main stages before calibration.\")\\n        calibration_probs, labels = self._collect_calibration_probs()\\n        rows = []\\n        candidates = []\\n        for weights in generate_simplex_weights(step=0.10):\\n            probs = weighted_probs(calibration_probs, weights)\\n            raw_pred = probs.argmax(axis=1)\\n            raw_metrics = score_with_safety(labels, raw_pred, rt.calculate_metrics)\\n            candidate_id = len(candidates)\\n            candidates.append({\\n                \"weights\": weights,\\n                \"bias\": np.zeros(5).tolist(),\\n                \"variant\": \"raw\",\\n                \"expert_params\": dict(DISABLED_EXPERT_PARAMS),\\n            })\\n            rows.append({\"candidate_id\": candidate_id, \"variant\": \"raw\", **weights, **raw_metrics})\\n            accuracy_floor = max(0.0, raw_metrics[\"accuracy\"] - self.cfg[\"accuracy_guard\"])\\n            searches = {\\n                \"cal_accuracy\": {\"accuracy\": 1.0, \"quadratic_kappa\": 0.0, \"macro_f1\": 0.0, \"weighted_f1\": 0.0},\\n                \"cal_balanced\": {\"accuracy\": 0.45, \"quadratic_kappa\": 0.20, \"macro_f1\": 0.30, \"weighted_f1\": 0.05},\\n                \"cal_macro_guard\": {\"accuracy\": 0.25, \"quadratic_kappa\": 0.15, \"macro_f1\": 0.55, \"weighted_f1\": 0.05},\\n            }\\n            for variant, objective_weights in searches.items():\\n                bias, _ = coordinate_bias_search(probs, labels, objective_weights, min_accuracy=accuracy_floor)\\n                pred = apply_bias(probs, bias).argmax(axis=1)\\n                metrics = score_with_safety(labels, pred, rt.calculate_metrics)\\n                candidate_id = len(candidates)\\n                candidates.append({\\n                    \"weights\": weights,\\n                    \"bias\": bias.tolist(),\\n                    \"variant\": variant,\\n                    \"expert_params\": dict(DISABLED_EXPERT_PARAMS),\\n                })\\n                rows.append({\"candidate_id\": candidate_id, \"variant\": variant, **weights, **metrics})\\n\\n        base_search_df = pd.DataFrame(rows)\\n        self.base_calibration_search_df = base_search_df\\n        base_search_df.to_csv(self.output_dir / \"calibration_base_candidate_search.csv\", index=False)\\n        base_accuracy_floor = float(base_search_df[\"accuracy\"].max()) - float(self.cfg[\"accuracy_guard\"])\\n        base_pool = base_search_df[base_search_df[\"accuracy\"] >= base_accuracy_floor].copy()\\n        base_pool = base_pool.sort_values(\\n            [\"macro_f1\", \"quadratic_kappa\", \"accuracy\", \"severe_to_0_rate\", \"dr_to_0_rate\"],\\n            ascending=[False, False, False, True, True],\\n        ).head(8)\\n\\n        has_expert_probs = self.use_experts and all(\\n            name in calibration_probs for name in [\"severity_480\", \"zero_mild_480\", \"moderate_severe_480\"]\\n        )\\n        if has_expert_probs:\\n            expert_rows = []\\n            expert_candidates = []\\n            for base_row in base_pool.itertuples(index=False):\\n                base_candidate = candidates[int(base_row.candidate_id)]\\n                for params in expert_param_grid():\\n                    candidate = {\\n                        \"weights\": dict(base_candidate[\"weights\"]),\\n                        \"bias\": list(base_candidate[\"bias\"]),\\n                        \"variant\": f\"{base_candidate[\\'variant\\']}+experts\",\\n                        \"expert_params\": params,\\n                    }\\n                    _, pred = predict_with_expert_candidate(calibration_probs, candidate)\\n                    metrics = score_with_safety(labels, pred, rt.calculate_metrics)\\n                    candidate_id = len(expert_candidates)\\n                    expert_candidates.append(candidate)\\n                    expert_rows.append({\\n                        \"candidate_id\": candidate_id,\\n                        \"base_candidate_id\": int(base_row.candidate_id),\\n                        \"variant\": candidate[\"variant\"],\\n                        **candidate[\"weights\"],\\n                        **params,\\n                        **metrics,\\n                    })\\n            search_df = pd.DataFrame(expert_rows)\\n            accuracy_floor = float(search_df[\"accuracy\"].max()) - float(self.cfg[\"accuracy_guard\"])\\n            eligible = search_df[search_df[\"accuracy\"] >= accuracy_floor].copy()\\n            best_severe_to_zero = float(eligible[\"severe_to_0_rate\"].min())\\n            guard_margin = float(self.cfg.get(\"severe_to_zero_guard_margin\", 0.005))\\n            guarded = eligible[eligible[\"severe_to_0_rate\"] <= best_severe_to_zero + guard_margin].copy()\\n            if guarded.empty:\\n                guarded = eligible\\n            selected_row = guarded.sort_values(\\n                [\"macro_f1\", \"quadratic_kappa\", \"accuracy\", \"severe_to_0_rate\", \"dr_to_0_rate\"],\\n                ascending=[False, False, False, True, True],\\n            ).iloc[0]\\n            selected = expert_candidates[int(selected_row[\"candidate_id\"])]\\n        else:\\n            search_df = base_search_df.copy()\\n            selected_row = base_pool.iloc[0]\\n            selected = candidates[int(selected_row[\"candidate_id\"])]\\n\\n        selected[\"calibration_metrics\"] = {\\n            key: float(selected_row[key])\\n            for key in [\\n                \"accuracy\",\\n                \"macro_f1\",\\n                \"weighted_f1\",\\n                \"quadratic_kappa\",\\n                \"dr_to_0_count\",\\n                \"dr_to_0_rate\",\\n                \"severe_to_0_count\",\\n                \"severe_to_0_rate\",\\n            ]\\n        }\\n        search_df.to_csv(self.output_dir / \"calibration_candidate_search.csv\", index=False)\\n        with open(self.output_dir / \"selected_multiclass_candidate.json\", \"w\", encoding=\"utf-8\") as handle:\\n            json.dump(selected, handle, indent=2)\\n        self.calibration_search_df = search_df\\n        self.selected_candidate = selected\\n        return selected\\n\\n    @staticmethod\\n    def binary_metrics(labels: np.ndarray, predictions: np.ndarray, threshold: int, name: str) -> dict:\\n        truth = np.asarray(labels) >= int(threshold)\\n        pred = np.asarray(predictions) >= int(threshold)\\n        tp = int((truth & pred).sum())\\n        tn = int((~truth & ~pred).sum())\\n        fp = int((~truth & pred).sum())\\n        fn = int((truth & ~pred).sum())\\n        sensitivity = tp / max(1, tp + fn)\\n        specificity = tn / max(1, tn + fp)\\n        precision = tp / max(1, tp + fp)\\n        f1 = 2 * precision * sensitivity / max(1e-12, precision + sensitivity)\\n        return {\\n            \"task\": name,\\n            \"threshold\": int(threshold),\\n            \"tp\": tp,\\n            \"tn\": tn,\\n            \"fp\": fp,\\n            \"fn\": fn,\\n            \"sensitivity\": float(sensitivity),\\n            \"specificity\": float(specificity),\\n            \"precision\": float(precision),\\n            \"f1\": float(f1),\\n            \"false_negative_rate\": float(fn / max(1, tp + fn)),\\n        }\\n\\n    def clinical_binary_metrics_table(self, frame: pd.DataFrame, predictions: np.ndarray) -> pd.DataFrame:\\n        labels = frame[\"label\"].astype(int).to_numpy()\\n        rows = [\\n            self.binary_metrics(labels, predictions, 1, \"any_dr_ge_1\"),\\n            self.binary_metrics(labels, predictions, 2, \"referable_dr_ge_2\"),\\n            self.binary_metrics(labels, predictions, 3, \"severe_or_pdr_ge_3\"),\\n        ]\\n        return pd.DataFrame(rows)\\n\\n    def metrics_by_source_table(self, frame: pd.DataFrame, predictions: np.ndarray) -> pd.DataFrame:\\n        rows = []\\n        for source, group in frame.assign(_pred=predictions).groupby(\"source\", sort=True):\\n            group_labels = group[\"label\"].astype(int).to_numpy()\\n            group_pred = group[\"_pred\"].astype(int).to_numpy()\\n            rows.append({\"source\": source, \"rows\": int(len(group)), **score_with_safety(group_labels, group_pred, rt.calculate_metrics)})\\n        return pd.DataFrame(rows)\\n\\n    def bootstrap_metric_intervals(\\n        self,\\n        labels: np.ndarray,\\n        predictions: np.ndarray,\\n        groups: np.ndarray,\\n        iterations: int,\\n    ) -> dict:\\n        if iterations <= 0:\\n            return {}\\n        rng = np.random.default_rng(int(self.cfg.get(\"seed\", 42)) + 9090)\\n        group_values = groups.astype(str)\\n        unique_groups = np.unique(group_values)\\n        indices_by_group = {group: np.flatnonzero(group_values == group) for group in unique_groups}\\n        samples = {name: [] for name in rt.calculate_metrics(labels, predictions)}\\n        for _ in range(int(iterations)):\\n            sampled_groups = rng.choice(unique_groups, size=len(unique_groups), replace=True)\\n            sampled_indices = np.concatenate([indices_by_group[group] for group in sampled_groups])\\n            metrics = rt.calculate_metrics(labels[sampled_indices], predictions[sampled_indices])\\n            for name, value in metrics.items():\\n                samples[name].append(value)\\n        return {\\n            name: {\\n                \"lower_95\": float(np.percentile(values, 2.5)),\\n                \"upper_95\": float(np.percentile(values, 97.5)),\\n            }\\n            for name, values in samples.items()\\n        }\\n\\n    def evaluate_holdout(self) -> dict:\\n        if self.selected_candidate is None:\\n            self.calibrate()\\n        if self.holdout_path is None:\\n            raise RuntimeError(\"Call prepare() before evaluate_holdout().\")\\n        holdout_frame = pd.read_csv(self.holdout_path)\\n        holdout_paths = holdout_frame[\"path\"].astype(str).tolist()\\n        labels = holdout_frame[\"label\"].astype(int).to_numpy()\\n        probs = {\\n            name: collect_probabilities(self.component_paths[name], self.component_configs[name], holdout_paths)\\n            for name in [\"stage1\", \"stage2\", \"stage3\"]\\n        }\\n        if self.use_experts and all(name in self.component_paths for name in [\"severity_480\", \"zero_mild_480\", \"moderate_severe_480\"]):\\n            expert_holdout_frame = align_cached_frame(self.expert_splits[\"holdout\"], holdout_frame)\\n            expert_paths = expert_holdout_frame[\"path\"].astype(str).tolist()\\n            probs.update({\\n                name: collect_probabilities(self.component_paths[name], self.component_configs[name], expert_paths)\\n                for name in [\"severity_480\", \"zero_mild_480\", \"moderate_severe_480\"]\\n            })\\n        base_candidate = {**self.selected_candidate, \"expert_params\": dict(DISABLED_EXPERT_PARAMS)}\\n        base_probs, base_predictions = predict_with_expert_candidate(probs, base_candidate)\\n        final_probs, predictions = predict_with_expert_candidate(probs, self.selected_candidate)\\n        metrics = score_with_safety(labels, predictions, rt.calculate_metrics)\\n        report = classification_report(\\n            labels,\\n            predictions,\\n            labels=[0, 1, 2, 3, 4],\\n            target_names=[CLASS_NAMES[i] for i in range(5)],\\n            output_dict=True,\\n            zero_division=0,\\n        )\\n        cm = confusion_matrix(labels, predictions, labels=[0, 1, 2, 3, 4])\\n        metrics_by_source = self.metrics_by_source_table(holdout_frame, predictions)\\n        binary_metrics = self.clinical_binary_metrics_table(holdout_frame, predictions)\\n        intervals = self.bootstrap_metric_intervals(\\n            labels,\\n            predictions,\\n            holdout_frame[\"group_key\"].astype(str).to_numpy(),\\n            int(self.cfg.get(\"bootstrap_iterations\", 1000) or 0),\\n        )\\n        metrics_by_source.to_csv(self.output_dir / \"metrics_by_source.csv\", index=False)\\n        binary_metrics.to_csv(self.output_dir / \"binary_clinical_metrics.csv\", index=False)\\n        prediction_frame = pd.DataFrame({\\n            \"orig_path\": holdout_frame[\"orig_path\"].astype(str).to_numpy(),\\n            \"path\": holdout_frame[\"path\"].astype(str).to_numpy(),\\n            \"source\": holdout_frame.get(\"source\", pd.Series([\"unknown\"] * len(holdout_frame))).astype(str).to_numpy(),\\n            \"source_split\": holdout_frame.get(\"source_split\", pd.Series([\"\"] * len(holdout_frame))).astype(str).to_numpy(),\\n            \"group_key\": holdout_frame[\"group_key\"].astype(str).to_numpy(),\\n            \"true\": labels,\\n            \"base_pred\": base_predictions,\\n            \"pred\": predictions,\\n            **{f\"base_prob_{i}\": base_probs[:, i] for i in range(5)},\\n            **{f\"prob_{i}\": final_probs[:, i] for i in range(5)},\\n        })\\n        prediction_frame.to_csv(self.output_dir / \"holdout_predictions.csv\", index=False)\\n        result = {\\n            \"selected_candidate\": self.selected_candidate,\\n            \"metrics\": metrics,\\n            \"metrics_by_source\": metrics_by_source.to_dict(\"records\"),\\n            \"binary_clinical_metrics\": binary_metrics.to_dict(\"records\"),\\n            \"patient_bootstrap_95_ci\": intervals,\\n            \"dangerous_false_negatives_before_experts\": safety_metrics(labels, base_predictions),\\n            \"dangerous_false_negatives_after_experts\": safety_metrics(labels, predictions),\\n            \"classification_report\": report,\\n            \"confusion_matrix\": cm.tolist(),\\n            \"target_accuracy\": float(self.cfg[\"target_accuracy\"]),\\n        }\\n        with open(self.output_dir / \"holdout_evaluation.json\", \"w\", encoding=\"utf-8\") as handle:\\n            json.dump(result, handle, indent=2)\\n        self.holdout_result = result\\n        return result\\n\\n    def export_summary(self) -> dict:\\n        summary = {\\n            \"experiment_name\": self.cfg.get(\"experiment_name\"),\\n            \"output_dir\": str(self.output_dir),\\n            \"preprocessing_profile\": self.cfg.get(\"preprocessing_profile\", \"full_current\"),\\n            \"use_experts\": self.use_experts,\\n            \"data_metadata\": self.data_metadata,\\n            \"expert_metadata\": self.expert_metadata,\\n            \"component_paths\": {name: str(path) for name, path in self.component_paths.items()},\\n            \"selected_candidate\": self.selected_candidate,\\n            \"holdout_metrics\": (self.holdout_result or {}).get(\"metrics\"),\\n            \"artifacts\": {\\n                \"experiment_config\": str(self.output_dir / \"experiment_config.json\"),\\n                \"split_manifest\": str(self.output_dir / \"split_manifest.csv\"),\\n                \"data_pipeline_metadata\": str(self.output_dir / \"data_pipeline_metadata.json\"),\\n                \"group_key_audit\": str(self.output_dir / \"group_key_audit.csv\"),\\n                \"duplicate_source_pairs\": str(self.output_dir / \"duplicate_source_pairs.csv\"),\\n                \"preprocessing_qa_samples\": str(self.output_dir / \"preprocessing_qa_samples\"),\\n                \"metrics_by_source\": str(self.output_dir / \"metrics_by_source.csv\"),\\n                \"binary_clinical_metrics\": str(self.output_dir / \"binary_clinical_metrics.csv\"),\\n                \"holdout_predictions\": str(self.output_dir / \"holdout_predictions.csv\"),\\n            },\\n        }\\n        with open(self.output_dir / \"experiment_summary.json\", \"w\", encoding=\"utf-8\") as handle:\\n            json.dump(summary, handle, indent=2)\\n        return summary\\n'}\nMODULE_ROOT = Path(\"/kaggle/working\")\nfor module_name, module_source in MODULE_SOURCES.items():\n    destination = MODULE_ROOT / module_name\n    destination.write_text(module_source, encoding=\"utf-8\")\n    subprocess.run([sys.executable, \"-m\", \"py_compile\", str(destination)], check=True)\n    print(\"Wrote and compiled:\", destination)\n\nif str(MODULE_ROOT) not in sys.path:\n    sys.path.insert(0, str(MODULE_ROOT))\n","metadata":{"execution":{"iopub.execute_input":"2026-06-28T03:07:55.781776Z","iopub.status.busy":"2026-06-28T03:07:55.781552Z","iopub.status.idle":"2026-06-28T03:07:56.591066Z","shell.execute_reply":"2026-06-28T03:07:56.590222Z"},"papermill":{"duration":0.824708,"end_time":"2026-06-28T03:07:56.592625+00:00","exception":false,"start_time":"2026-06-28T03:07:55.767917+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"17d5914b","cell_type":"markdown","source":"## Build the Audited Fundus Cache and Splits","metadata":{"papermill":{"duration":0.009321,"end_time":"2026-06-28T03:07:56.612046+00:00","exception":false,"start_time":"2026-06-28T03:07:56.602725+00:00","status":"completed"},"tags":[]}},{"id":"60bbc6fa","cell_type":"code","source":"from dr_data_corrected import CLASS_NAMES\nfrom dr_training_builder import DRTrainingBuilder\n\nbuilder = DRTrainingBuilder(CFG)\nARTIFACTS = builder.prepare()\nCACHED_SPLITS = builder.cached_splits\nsplit_manifest = builder.split_manifest\nDUPLICATE_AUDIT = builder.duplicate_audit\nDATA_METADATA = builder.data_metadata\nEXPERT_SPLITS = builder.expert_splits\nEXPERT_METADATA = builder.expert_metadata\nSEALED_HOLDOUT_PATH = builder.holdout_path\nEXPERT_HOLDOUT_PATH = builder.expert_holdout_path\n\nsummary = (\n    split_manifest.groupby([\"partition\", \"label\"])\n    .size()\n    .unstack(fill_value=0)\n    .rename(columns=CLASS_NAMES)\n)\ndisplay(summary)\nprint(\"Preprocessing fingerprint:\", DATA_METADATA[\"preprocessing_fingerprint\"])\nprint(\"Split fingerprint:\", DATA_METADATA[\"split_fingerprint\"])\nprint(\"Sealed holdout path:\", SEALED_HOLDOUT_PATH)\nif EXPERT_HOLDOUT_PATH is not None:\n    print(\"480 expert holdout path:\", EXPERT_HOLDOUT_PATH)\n","metadata":{"execution":{"iopub.execute_input":"2026-06-28T03:07:56.631831Z","iopub.status.busy":"2026-06-28T03:07:56.631592Z","iopub.status.idle":"2026-06-28T03:56:59.341419Z","shell.execute_reply":"2026-06-28T03:56:59.340379Z"},"papermill":{"duration":2942.723767,"end_time":"2026-06-28T03:56:59.345159+00:00","exception":false,"start_time":"2026-06-28T03:07:56.621392+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"708930e1","cell_type":"markdown","source":"## Training Builder Stage Orchestration\n","metadata":{"papermill":{"duration":0.016363,"end_time":"2026-06-28T03:56:59.372521+00:00","exception":false,"start_time":"2026-06-28T03:56:59.356158+00:00","status":"completed"},"tags":[]}},{"id":"98e9c60b","cell_type":"code","source":"# DDP stage launching is now owned by DRTrainingBuilder.\n# The builder still writes each stage's run_config.json, split CSVs, checkpoints,\n# selection.json, and training_history.csv under OUTPUT_DIR.\n","metadata":{"execution":{"iopub.execute_input":"2026-06-28T03:56:59.397035Z","iopub.status.busy":"2026-06-28T03:56:59.396218Z","iopub.status.idle":"2026-06-28T03:56:59.400205Z","shell.execute_reply":"2026-06-28T03:56:59.399423Z"},"papermill":{"duration":0.01777,"end_time":"2026-06-28T03:56:59.401743+00:00","exception":false,"start_time":"2026-06-28T03:56:59.383973+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"98eaf2ed","cell_type":"markdown","source":"## Main InceptionV3 Stages\n","metadata":{"papermill":{"duration":0.010085,"end_time":"2026-06-28T03:56:59.422212+00:00","exception":false,"start_time":"2026-06-28T03:56:59.412127+00:00","status":"completed"},"tags":[]}},{"id":"2f4876ab","cell_type":"code","source":"MAIN_CKPTS = builder.train_main_stages()\nSTAGE1_CKPT = MAIN_CKPTS[\"stage1\"]\nSTAGE2_CKPT = MAIN_CKPTS[\"stage2\"]\nSTAGE3_CKPT = MAIN_CKPTS[\"stage3\"]\nSTAGE1_CFG = builder.component_configs[\"stage1\"]\nSTAGE2_CFG = builder.component_configs[\"stage2\"]\nSTAGE3_CFG = builder.component_configs[\"stage3\"]\n\ndisplay(pd.DataFrame([\n    {\"component\": name, \"checkpoint\": str(path), \"stage\": builder.component_configs[name][\"stage_name\"]}\n    for name, path in MAIN_CKPTS.items()\n]))\n","metadata":{"execution":{"iopub.execute_input":"2026-06-28T03:56:59.442964Z","iopub.status.busy":"2026-06-28T03:56:59.442645Z","iopub.status.idle":"2026-06-28T04:25:13.244438Z","shell.execute_reply":"2026-06-28T04:25:13.243745Z"},"papermill":{"duration":1693.814042,"end_time":"2026-06-28T04:25:13.246080+00:00","exception":false,"start_time":"2026-06-28T03:56:59.432038+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"d0d38d4d","cell_type":"markdown","source":"## Stage 2 Summary\n","metadata":{"papermill":{"duration":0.019461,"end_time":"2026-06-28T04:25:13.285615+00:00","exception":false,"start_time":"2026-06-28T04:25:13.266154+00:00","status":"completed"},"tags":[]}},{"id":"c25683b6","cell_type":"code","source":"print(\"Stage 2 checkpoint:\", STAGE2_CKPT)\nprint(\"Stage 2 config:\")\nprint(json.dumps({key: STAGE2_CFG[key] for key in [\"stage_name\", \"epochs\", \"lr\", \"imbalance_mode\", \"class_weight_power\"]}, indent=2))\n","metadata":{"execution":{"iopub.execute_input":"2026-06-28T04:25:13.327153Z","iopub.status.busy":"2026-06-28T04:25:13.326520Z","iopub.status.idle":"2026-06-28T04:25:13.331763Z","shell.execute_reply":"2026-06-28T04:25:13.330873Z"},"papermill":{"duration":0.027717,"end_time":"2026-06-28T04:25:13.333243+00:00","exception":false,"start_time":"2026-06-28T04:25:13.305526+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"2c9056b8","cell_type":"markdown","source":"## Stage 3 Summary\n","metadata":{"papermill":{"duration":0.01951,"end_time":"2026-06-28T04:25:13.372820+00:00","exception":false,"start_time":"2026-06-28T04:25:13.353310+00:00","status":"completed"},"tags":[]}},{"id":"eb8d77db","cell_type":"code","source":"print(\"Stage 3 checkpoint:\", STAGE3_CKPT)\nprint(\"Stage 3 config:\")\nprint(json.dumps({key: STAGE3_CFG[key] for key in [\"stage_name\", \"epochs\", \"lr\", \"imbalance_mode\", \"sampler_power\", \"class_weight_power\"]}, indent=2))\n","metadata":{"execution":{"iopub.execute_input":"2026-06-28T04:25:13.413408Z","iopub.status.busy":"2026-06-28T04:25:13.413051Z","iopub.status.idle":"2026-06-28T04:25:13.417954Z","shell.execute_reply":"2026-06-28T04:25:13.417065Z"},"papermill":{"duration":0.026687,"end_time":"2026-06-28T04:25:13.419319+00:00","exception":false,"start_time":"2026-06-28T04:25:13.392632+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"80050eae","cell_type":"markdown","source":"## Optional 480px Expert Stages\n","metadata":{"papermill":{"duration":0.019435,"end_time":"2026-06-28T04:25:13.458009+00:00","exception":false,"start_time":"2026-06-28T04:25:13.438574+00:00","status":"completed"},"tags":[]}},{"id":"0f0d7760","cell_type":"code","source":"EXPERT_CKPTS = builder.train_experts()\nSEVERITY_480_CKPT = EXPERT_CKPTS.get(\"severity_480\")\nZERO_MILD_480_CKPT = EXPERT_CKPTS.get(\"zero_mild_480\")\nMODERATE_SEVERE_480_CKPT = EXPERT_CKPTS.get(\"moderate_severe_480\")\nSEVERITY_480_CFG = builder.component_configs.get(\"severity_480\")\nZERO_MILD_480_CFG = builder.component_configs.get(\"zero_mild_480\")\nMODERATE_SEVERE_480_CFG = builder.component_configs.get(\"moderate_severe_480\")\n\ndisplay(pd.DataFrame([\n    {\"component\": name, \"checkpoint\": str(path), \"stage\": builder.component_configs[name][\"stage_name\"]}\n    for name, path in EXPERT_CKPTS.items()\n]))\n","metadata":{"execution":{"iopub.execute_input":"2026-06-28T04:25:13.498847Z","iopub.status.busy":"2026-06-28T04:25:13.498295Z","iopub.status.idle":"2026-06-28T04:35:47.268867Z","shell.execute_reply":"2026-06-28T04:35:47.268088Z"},"papermill":{"duration":633.792568,"end_time":"2026-06-28T04:35:47.270493+00:00","exception":false,"start_time":"2026-06-28T04:25:13.477925+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"9bc2b611","cell_type":"markdown","source":"## Optional Expert: No-DR vs Mild\n","metadata":{"papermill":{"duration":0.023264,"end_time":"2026-06-28T04:35:47.318328+00:00","exception":false,"start_time":"2026-06-28T04:35:47.295064+00:00","status":"completed"},"tags":[]}},{"id":"aabb5dae","cell_type":"code","source":"if ZERO_MILD_480_CKPT is None:\n    print(\"Zero-vs-mild expert skipped because CFG['use_experts'] is false.\")\nelse:\n    print(\"Zero-vs-mild expert checkpoint:\", ZERO_MILD_480_CKPT)\n","metadata":{"execution":{"iopub.execute_input":"2026-06-28T04:35:47.366771Z","iopub.status.busy":"2026-06-28T04:35:47.366351Z","iopub.status.idle":"2026-06-28T04:35:47.370940Z","shell.execute_reply":"2026-06-28T04:35:47.370076Z"},"papermill":{"duration":0.03054,"end_time":"2026-06-28T04:35:47.372518+00:00","exception":false,"start_time":"2026-06-28T04:35:47.341978+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"9feb93f1","cell_type":"markdown","source":"## Optional Expert: Moderate vs Severe+\n","metadata":{"papermill":{"duration":0.023363,"end_time":"2026-06-28T04:35:47.419531+00:00","exception":false,"start_time":"2026-06-28T04:35:47.396168+00:00","status":"completed"},"tags":[]}},{"id":"8f189ed3","cell_type":"code","source":"if MODERATE_SEVERE_480_CKPT is None:\n    print(\"Moderate-vs-severe expert skipped because CFG['use_experts'] is false.\")\nelse:\n    print(\"Moderate-vs-severe expert checkpoint:\", MODERATE_SEVERE_480_CKPT)\n","metadata":{"execution":{"iopub.execute_input":"2026-06-28T04:35:47.467385Z","iopub.status.busy":"2026-06-28T04:35:47.467156Z","iopub.status.idle":"2026-06-28T04:35:47.471455Z","shell.execute_reply":"2026-06-28T04:35:47.470790Z"},"papermill":{"duration":0.030098,"end_time":"2026-06-28T04:35:47.472743+00:00","exception":false,"start_time":"2026-06-28T04:35:47.442645+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"aea3161e","cell_type":"markdown","source":"## Training Curves","metadata":{"papermill":{"duration":0.023542,"end_time":"2026-06-28T04:35:47.519202+00:00","exception":false,"start_time":"2026-06-28T04:35:47.495660+00:00","status":"completed"},"tags":[]}},{"id":"f9c26ef3","cell_type":"code","source":"def plot_training_histories(stage_cfgs: list[dict]) -> None:\n    histories = []\n    for cfg in stage_cfgs:\n        path = Path(cfg[\"output_dir\"]) / \"training_history.csv\"\n        if not path.exists():\n            print(\"Missing history:\", path)\n            continue\n        history = pd.read_csv(path)\n        history[\"stage\"] = cfg[\"stage_name\"]\n        histories.append(history)\n    if not histories:\n        return\n\n    table = pd.concat(histories, ignore_index=True)\n    display(table)\n\n    fig, axes = plt.subplots(1, 4, figsize=(18, 4))\n    metrics = [\n        (\"train_loss\", \"Training Loss\"),\n        (\"val_accuracy\", \"Validation Accuracy\"),\n        (\"val_macro_f1\", \"Validation Macro-F1\"),\n        (\"val_quadratic_kappa\", \"Validation QWK\"),\n    ]\n    for ax, (column, title) in zip(axes, metrics):\n        for stage, group in table.groupby(\"stage\"):\n            ax.plot(group[\"epoch\"], group[column], marker=\"o\", label=stage)\n        ax.set_title(title)\n        ax.set_xlabel(\"Epoch\")\n        ax.grid(alpha=0.25)\n    axes[-1].legend(loc=\"best\", fontsize=8)\n    plt.tight_layout()\n    plt.show()\n\nplot_training_histories([\n    STAGE1_CFG,\n    STAGE2_CFG,\n    STAGE3_CFG,\n    SEVERITY_480_CFG,\n    ZERO_MILD_480_CFG,\n    MODERATE_SEVERE_480_CFG,\n])","metadata":{"execution":{"iopub.execute_input":"2026-06-28T04:35:47.568233Z","iopub.status.busy":"2026-06-28T04:35:47.567526Z","iopub.status.idle":"2026-06-28T04:35:48.208366Z","shell.execute_reply":"2026-06-28T04:35:48.207615Z"},"papermill":{"duration":0.66814,"end_time":"2026-06-28T04:35:48.211017+00:00","exception":false,"start_time":"2026-06-28T04:35:47.542877+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"4891b7dc","cell_type":"markdown","source":"## Builder Calibration Search\n","metadata":{"papermill":{"duration":0.026702,"end_time":"2026-06-28T04:35:48.263899+00:00","exception":false,"start_time":"2026-06-28T04:35:48.237197+00:00","status":"completed"},"tags":[]}},{"id":"ec44e848","cell_type":"code","source":"SELECTED_CANDIDATE = builder.calibrate()\nCOMPONENT_PATHS = builder.component_paths\nCOMPONENT_CONFIGS = builder.component_configs\nCALIBRATION_SEARCH_DF = builder.calibration_search_df\nBASE_CALIBRATION_SEARCH_DF = builder.base_calibration_search_df\n\ndisplay(CALIBRATION_SEARCH_DF.sort_values(\n    [\"accuracy\", \"macro_f1\", \"quadratic_kappa\"],\n    ascending=False,\n).head(20))\nprint(json.dumps(SELECTED_CANDIDATE, indent=2))\n","metadata":{"execution":{"iopub.execute_input":"2026-06-28T04:35:48.316063Z","iopub.status.busy":"2026-06-28T04:35:48.315336Z","iopub.status.idle":"2026-06-28T04:41:55.694976Z","shell.execute_reply":"2026-06-28T04:41:55.693851Z"},"papermill":{"duration":367.43552,"end_time":"2026-06-28T04:41:55.724366+00:00","exception":false,"start_time":"2026-06-28T04:35:48.288846+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"108d1311","cell_type":"markdown","source":"## Builder Holdout Evaluation And Artifact Export\n","metadata":{"papermill":{"duration":0.02613,"end_time":"2026-06-28T04:41:55.776448+00:00","exception":false,"start_time":"2026-06-28T04:41:55.750318+00:00","status":"completed"},"tags":[]}},{"id":"8f2bedb2","cell_type":"code","source":"HOLDOUT_RESULT = builder.evaluate_holdout()\nEXPERIMENT_SUMMARY = builder.export_summary()\n\nprint(json.dumps(HOLDOUT_RESULT[\"metrics\"], indent=2))\nprint(\"Holdout dangerous false negatives before expert fusion:\")\nprint(json.dumps(HOLDOUT_RESULT.get(\"dangerous_false_negatives_before_experts\", {}), indent=2))\nprint(\"Holdout dangerous false negatives after expert fusion:\")\nprint(json.dumps(HOLDOUT_RESULT.get(\"dangerous_false_negatives_after_experts\", {}), indent=2))\ndisplay(pd.DataFrame(HOLDOUT_RESULT[\"classification_report\"]).T)\ndisplay(pd.DataFrame(HOLDOUT_RESULT.get(\"metrics_by_source\", [])))\ndisplay(pd.DataFrame(HOLDOUT_RESULT.get(\"binary_clinical_metrics\", [])))\nprint(\"Experiment summary:\", OUTPUT_DIR / \"experiment_summary.json\")\n","metadata":{"execution":{"iopub.execute_input":"2026-06-28T04:41:55.830802Z","iopub.status.busy":"2026-06-28T04:41:55.830519Z","iopub.status.idle":"2026-06-28T04:48:41.437199Z","shell.execute_reply":"2026-06-28T04:48:41.436227Z"},"papermill":{"duration":405.650398,"end_time":"2026-06-28T04:48:41.452702+00:00","exception":false,"start_time":"2026-06-28T04:41:55.802304+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"aeba7caf","cell_type":"markdown","source":"## Confusion Matrix and Target Check","metadata":{"papermill":{"duration":0.02594,"end_time":"2026-06-28T04:48:41.504666+00:00","exception":false,"start_time":"2026-06-28T04:48:41.478726+00:00","status":"completed"},"tags":[]}},{"id":"8ebc5c45","cell_type":"code","source":"def plot_confusion_matrix(result: dict) -> None:\n    cm = np.asarray(result[\"confusion_matrix\"], dtype=np.int64)\n    cm_norm = cm / np.maximum(cm.sum(axis=1, keepdims=True), 1)\n    labels = list(range(5))\n\n    fig, ax = plt.subplots(figsize=(7, 6))\n    image = ax.imshow(cm_norm, cmap=\"Blues\", vmin=0, vmax=1)\n    ax.set_title(\"Holdout Confusion Matrix\")\n    ax.set_xlabel(\"Predicted label\")\n    ax.set_ylabel(\"True label\")\n    ax.set_xticks(labels)\n    ax.set_yticks(labels)\n    ax.set_xticklabels([str(i) for i in labels])\n    ax.set_yticklabels([f\"{i}: {CLASS_NAMES[i]}\" for i in labels])\n    for i in labels:\n        for j in labels:\n            ax.text(\n                j,\n                i,\n                f\"{cm_norm[i, j]:.2f}\\n({cm[i, j]})\",\n                ha=\"center\",\n                va=\"center\",\n                color=\"white\" if cm_norm[i, j] > 0.60 else \"black\",\n                fontsize=8,\n            )\n    fig.colorbar(image, ax=ax, fraction=0.046, pad=0.04)\n    plt.tight_layout()\n    plt.show()\n\nplot_confusion_matrix(HOLDOUT_RESULT)\n\nmetrics = HOLDOUT_RESULT[\"metrics\"]\ntarget = float(CFG[\"target_accuracy\"])\nif metrics[\"accuracy\"] >= target:\n    print(f\"Target met: holdout accuracy {metrics['accuracy']:.4f} >= {target:.4f}\")\nelse:\n    print(f\"Target not met yet: holdout accuracy {metrics['accuracy']:.4f} < {target:.4f}\")\nprint(f\"Macro-F1: {metrics['macro_f1']:.4f}\")\nprint(f\"Weighted-F1: {metrics['weighted_f1']:.4f}\")\nprint(f\"Quadratic weighted kappa: {metrics['quadratic_kappa']:.4f}\")","metadata":{"execution":{"iopub.execute_input":"2026-06-28T04:48:41.560825Z","iopub.status.busy":"2026-06-28T04:48:41.560068Z","iopub.status.idle":"2026-06-28T04:48:41.785877Z","shell.execute_reply":"2026-06-28T04:48:41.785025Z"},"papermill":{"duration":0.254962,"end_time":"2026-06-28T04:48:41.787568+00:00","exception":false,"start_time":"2026-06-28T04:48:41.532606+00:00","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}