{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":117682,"databundleVersionId":15062069,"sourceType":"competition"},{"sourceId":14330461,"sourceType":"datasetVersion","datasetId":8944423},{"sourceId":289294474,"sourceType":"kernelVersion"},{"sourceId":290917305,"sourceType":"kernelVersion"},{"sourceId":723836,"sourceType":"modelInstanceVersion","modelInstanceId":550808,"modelId":563434},{"sourceId":738660,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":563333,"modelId":563810}],"dockerImageVersionId":31260,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":295.830892,"end_time":"2026-01-21T08:58:33.678693","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-01-21T08:53:37.847801","version":"2.6.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"018098a00b554631beb2cc28688186d1":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"08f2d67691a040e393894b8131860188":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_6393b6f8fd154183940c2cd92f78ddfe","IPY_MODEL_21e81d3ce03d4e0d9e6d28a53a7afcda","IPY_MODEL_e35e09d52de941d5b819c82844fc7a41"],"layout":"IPY_MODEL_8d732f73f082470598b2ca222d65ad75","tabbable":null,"tooltip":null}},"10db870600784f739e8cafbcaf0023ff":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_3d1239a2b2234df7a7af2b8d712a4da0","placeholder":"​","style":"IPY_MODEL_018098a00b554631beb2cc28688186d1","tabbable":null,"tooltip":null,"value":" 1/1 [00:02&lt;00:00,  2.16s/it]"}},"21e81d3ce03d4e0d9e6d28a53a7afcda":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_b62ba08784fa40c196204a62fb87f08b","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_f9f4d04f1b6e419abfc64fe848459d35","tabbable":null,"tooltip":null,"value":1}},"26e25d48187c471e8dfc3410f7bdc9f9":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"27265f8643634247922a5bce6515f54b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"277cf49b08924a8b933d35e13f0ea6d1":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"3b000264ff2143d69a28656bdb159007":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"3b6b6fa613c54d358bd5b9e42607058d":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"3cbfbc94777249159bb22af6f7ee92fd":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"3d1239a2b2234df7a7af2b8d712a4da0":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"418d23f12eb749abbc2d40487aad7055":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_3b6b6fa613c54d358bd5b9e42607058d","max":786,"min":0,"orientation":"horizontal","style":"IPY_MODEL_b009de37a6144fdea80b44dbc68870cf","tabbable":null,"tooltip":null,"value":786}},"500e788e3a204ac8ab73886258912706":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"519adb2b31624603a6615c58c46bb038":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_9bf3f40e228e4f90a813695b9527596c","IPY_MODEL_bc3b9d6a32ed46cd9c419cad68fe3f2e","IPY_MODEL_10db870600784f739e8cafbcaf0023ff"],"layout":"IPY_MODEL_e9b919333f9743ce9a2495f26be59d2f","tabbable":null,"tooltip":null}},"6393b6f8fd154183940c2cd92f78ddfe":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_687e0470870744ac91da09502c3e7d33","placeholder":"​","style":"IPY_MODEL_3b000264ff2143d69a28656bdb159007","tabbable":null,"tooltip":null,"value":"Preparing test data: 100%"}},"687e0470870744ac91da09502c3e7d33":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8af51986079545bd8b03ebd21957b123":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"8d732f73f082470598b2ca222d65ad75":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"9bf3f40e228e4f90a813695b9527596c":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_fc14042873314b72a442c0c8e7bcf3fc","placeholder":"​","style":"IPY_MODEL_27265f8643634247922a5bce6515f54b","tabbable":null,"tooltip":null,"value":"Converting to TIFF: 100%"}},"9c4a9e015ac84d448643d1d0e1563bd4":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_b33798f0806d4775a4625ed3f9b5f646","IPY_MODEL_418d23f12eb749abbc2d40487aad7055","IPY_MODEL_d6cb0c7ebd4447abb8235a5c7dfd289f"],"layout":"IPY_MODEL_3cbfbc94777249159bb22af6f7ee92fd","tabbable":null,"tooltip":null}},"b009de37a6144fdea80b44dbc68870cf":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"b120817c4def442ca617b2f5ee4cd319":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"b33798f0806d4775a4625ed3f9b5f646":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_277cf49b08924a8b933d35e13f0ea6d1","placeholder":"​","style":"IPY_MODEL_8af51986079545bd8b03ebd21957b123","tabbable":null,"tooltip":null,"value":"Preparing dataset: 100%"}},"b62ba08784fa40c196204a62fb87f08b":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"b746eced39e54064ab0a9f7d4f85db8b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"bc3b9d6a32ed46cd9c419cad68fe3f2e":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_edaab4572eb049448dc7a042aa30e752","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_b746eced39e54064ab0a9f7d4f85db8b","tabbable":null,"tooltip":null,"value":1}},"d6cb0c7ebd4447abb8235a5c7dfd289f":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_500e788e3a204ac8ab73886258912706","placeholder":"​","style":"IPY_MODEL_26e25d48187c471e8dfc3410f7bdc9f9","tabbable":null,"tooltip":null,"value":" 786/786 [00:58&lt;00:00, 16.25it/s]"}},"e1bc539a70e3481598802de9b712f501":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"e35e09d52de941d5b819c82844fc7a41":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_e1bc539a70e3481598802de9b712f501","placeholder":"​","style":"IPY_MODEL_b120817c4def442ca617b2f5ee4cd319","tabbable":null,"tooltip":null,"value":" 1/1 [00:00&lt;00:00,  6.47it/s]"}},"e9b919333f9743ce9a2495f26be59d2f":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"edaab4572eb049448dc7a042aa30e752":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"f9f4d04f1b6e419abfc64fe848459d35":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"fc14042873314b72a442c0c8e7bcf3fc":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# DONT CARE THIS","metadata":{"papermill":{"duration":0.010958,"end_time":"2026-01-21T08:53:40.313262","exception":false,"start_time":"2026-01-21T08:53:40.302304","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"```\n# Summary\nInput: .tif\nOutput: binary segmentation mask\nFramework: nnUnetv2\n\n# Autoconfig by nnUnet\n- Patch size\n- Batch size\n- Depth, Norm, Data aug, lr scheduler\n\n# Manualconfig\n- epochs (chooses from list, cannot choose outside the predefined list)\n- panner\n- fold\n- model config\n- gpu\n```","metadata":{"papermill":{"duration":0.008969,"end_time":"2026-01-21T08:53:40.331338","exception":false,"start_time":"2026-01-21T08:53:40.322369","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import os\nimport json\nimport shutil\nimport subprocess\nfrom functools import partial\nfrom multiprocessing import Pool\nfrom pathlib import Path\nfrom typing import Optional, Tuple, List, Literal, Union\n\nEpochs = Literal[1, 5, 10, 20, 50, 100, 250, 500, 750, 1000, 2000, 4000, 8000]\nDATA_DIR = Path(\"/kaggle/input/vesuvius-challenge-surface-detection\")\nPREPARED_DATA_DIR = Path(\"/kaggle/input/vesuvius-surface-nnunet-preprocessed\")\nWORKING_DIR = Path(\"/kaggle/temp\") # large file storage\nOUTPUT_DIR = Path(\"/kaggle/working\") # persisted output\n\nNNUNET_BASE = WORKING_DIR / \"nnUNet_data\"\nNNUNET_RAW = NNUNET_BASE / \"nnUNet_raw\"\nNNUNET_PREPROCESSED = NNUNET_BASE / \"nnUNet_preprocessed\"\nNNUNET_RESULTS = OUTPUT_DIR / \"nnUNet_results\"\n\nDS_ID = 100\nDS_NAME = f\"Dataset{DS_ID:03d}_VesuviusSurface\" # DatasetXXXX_{CustomName}\n\nFOLD: Union[int, str] = \"all\"\nCONFIGURATION = \"3d_fullres\"\nPLANNER = \"nnUNetPlannerResEncM\"\nPLANS_NAME = \"nnUNetResEncUNetMPlans\"\n\nNUM_WORKERS = os.cpu_count() or 4\nEPOCHS: Optional[Epochs] = 50\nCOMMAND_TIMEOUT: Optional[int] = 21600 # 8 hours + 1 hour buffer\n\nimport torch\nNUM_GPUS: int = torch.cuda.device_count() ","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:40.353025Z","iopub.status.busy":"2026-01-21T08:53:40.352797Z","iopub.status.idle":"2026-01-21T08:53:43.923675Z","shell.execute_reply":"2026-01-21T08:53:43.923043Z"},"papermill":{"duration":3.582378,"end_time":"2026-01-21T08:53:43.925348","exception":false,"start_time":"2026-01-21T08:53:40.342970","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def _get_trainer_name_simple(epochs: Optional[int]) -> str:\n    \"\"\"Get trainer class name based on epochs (simple version for path construction).\"\"\"\n    if epochs is None or epochs == 1000:\n        return \"nnUNetTrainer\"\n    elif epochs == 1:\n        return \"nnUNetTrainer_1epoch\"  # Special case: singular form\n    else:\n        return f\"nnUNetTrainer_{epochs}epochs\"\n    \ndef get_training_output_dir(\n    epochs: Optional[Epochs] = None,\n    plans: str = PLANS_NAME,\n    config: str = CONFIGURATION,\n    fold: Union[int, str] = FOLD\n) -> Path:\n    \"\"\"\n    Get the training output directory path based on configuration.\n    \n    nnUNet creates this folder structure:\n    NNUNET_RESULTS/Dataset100_VesuviusSurface/nnUNetTrainer__nnUNetResEncUNetMPlans__3d_lowres/fold_all/\n    \n    Use this to find checkpoints, logs, and progress.png\n    \"\"\"\n    _epochs = epochs if epochs is not None else EPOCHS\n    trainer = _get_trainer_name_simple(_epochs)\n    return NNUNET_RESULTS / DS_NAME / f\"{trainer}__{plans}__{config}\" / f\"fold_{fold}\"\n\ndef get_progress_image_path(\n    epochs: Optional[Epochs] = None,\n    plans: str = PLANS_NAME,\n    config: str = CONFIGURATION,\n    fold: Union[int, str] = FOLD\n) -> Path:\n    \"\"\"Get path to training progress image (loss curves, metrics over epochs).\"\"\"\n    return get_training_output_dir(epochs, plans, config, fold) / \"progress.png\"","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:43.945904Z","iopub.status.busy":"2026-01-21T08:53:43.945565Z","iopub.status.idle":"2026-01-21T08:53:43.951720Z","shell.execute_reply":"2026-01-21T08:53:43.951157Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.018184,"end_time":"2026-01-21T08:53:43.952991","exception":false,"start_time":"2026-01-21T08:53:43.934807","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Environment\n\n- nnUNet_raw: Where nnUNet looks for raw dataset\n- nnUNet_preprocessed: Where preprocessed data is stored\n- nnUNet_results: Where trained models are saved\n- nnUNet_compile: Disable torch.compile (can cause issues)","metadata":{"papermill":{"duration":0.009463,"end_time":"2026-01-21T08:53:43.971431","exception":false,"start_time":"2026-01-21T08:53:43.961968","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def setup_environment():\n    \"\"\"Set up nnUNet environment variables and directories.\"\"\"\n    for d in [NNUNET_RAW, NNUNET_PREPROCESSED, NNUNET_RESULTS, OUTPUT_DIR]:\n        d.mkdir(parents=True, exist_ok=True)\n    \n    os.environ[\"nnUNet_raw\"] = str(NNUNET_RAW)\n    os.environ[\"nnUNet_preprocessed\"] = str(NNUNET_PREPROCESSED)\n    os.environ[\"nnUNet_results\"] = str(NNUNET_RESULTS)\n    os.environ[\"nnUNet_compile\"] = \"true\"\n    print(f\"nnUNet_raw: {NNUNET_RAW}\")\n    print(f\"nnUNet_preprocessed: {NNUNET_PREPROCESSED}\")\n    print(f\"nnUNet_results: {NNUNET_RESULTS}\")\n    print(f\"nnUNet_USE_BLOSC2: {os.environ.get('nnUNet_USE_BLOSC2', 'not set')} (0=NPZ, 1=blosc2)\")\n    print(f\"NUM_WORKERS: {NUM_WORKERS}\")\n\n\nsetup_environment()","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:43.990756Z","iopub.status.busy":"2026-01-21T08:53:43.990487Z","iopub.status.idle":"2026-01-21T08:53:43.996404Z","shell.execute_reply":"2026-01-21T08:53:43.995682Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.017172,"end_time":"2026-01-21T08:53:43.997648","exception":false,"start_time":"2026-01-21T08:53:43.980476","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def _link_prepared_preprocessed() -> bool:\n    \"\"\"\n    Link/copy pre-prepared preprocessed data if available.\n    \n    Copies the folder structure and metadata files (which nnUNet may need to modify),\n    but symlinks the heavy .npz/.b2nd data files to save space.\n    \n    Handles two possible structures:\n    1. PREPARED_DATA_DIR points directly to Dataset100_* folder\n    2. PREPARED_DATA_DIR contains Dataset100_* as subfolder\n    \n    Returns True if linked/copied successfully.\n    \"\"\"\n    if not PREPARED_DATA_DIR.exists():\n        return False\n    \n    source_dir = PREPARED_DATA_DIR\n    if not (source_dir / \"dataset.json\").exists():\n        dataset_folders = list(PREPARED_DATA_DIR.glob(f\"Dataset*_{DS_NAME.split('_')[1]}*\"))\n        if not dataset_folders:\n            dataset_folders = list(PREPARED_DATA_DIR.glob(\"Dataset*\"))\n        if dataset_folders:\n            source_dir = dataset_folders[0]\n        else:\n            print(f\"No dataset folder found in {PREPARED_DATA_DIR}\")\n            return False\n    \n    dest_dir = NNUNET_PREPROCESSED / DS_NAME\n    if dest_dir.exists():\n        print(f\"Preprocessed data already exists at {dest_dir}\")\n        return True\n    \n    dest_dir.mkdir(parents=True, exist_ok=True)\n    # copy_patterns = [\"*.json\", \"*.pkl\", \"*.txt\"]\n    symlink_patterns = [\"*.npz\", \"*.npy\", \"*.b2nd\"]\n\n    cnt_copy = 0   \n    cnt_symlink = 0\n    for src_path in source_dir.rglob(\"*\"): # recursive\n        if src_path.is_dir():\n            continue\n        # Compute relative path and create target path\n        rel_path = src_path.relative_to(source_dir)\n        dst_path = dest_dir / rel_path\n        dst_path.parent.mkdir(parents=True, exist_ok=True)\n        \n        # Check if this is a heavy data file (symlink) or metadata (copy)\n        is_data_file = any(src_path.match(pat) for pat in symlink_patterns)\n        \n        if is_data_file:\n            if not dst_path.exists():\n                dst_path.symlink_to(src_path.resolve())\n                cnt_symlink += 1\n        else:\n            if not dst_path.exists():\n                shutil.copy2(src_path, dst_path)\n                cnt_copy += 1\n    print(f\"Prepared preprocessed data: {cnt_copy} files copied, {cnt_symlink} files symlinked\")\n    print(f\"Location: {dest_dir}\")\n    return True\n","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:44.017522Z","iopub.status.busy":"2026-01-21T08:53:44.017029Z","iopub.status.idle":"2026-01-21T08:53:44.024080Z","shell.execute_reply":"2026-01-21T08:53:44.023459Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.018497,"end_time":"2026-01-21T08:53:44.025440","exception":false,"start_time":"2026-01-21T08:53:44.006943","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Install nnUNetv2, tifffile","metadata":{"papermill":{"duration":0.009269,"end_time":"2026-01-21T08:53:44.043780","exception":false,"start_time":"2026-01-21T08:53:44.034511","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!mkdir -p /kaggle/temp\n!mkdir predictions_tiff\n# !pip install nnunetv2\n# !pip install nnunetv2 nibabel imagecodecs tifffile tqdm --no-index --find-links \"/kaggle/input/surface-package-scraper\"\n# !pip install nnunetv2 --no-index --find-links \"/kaggle/input/surface-package-scraper\"\n# !pip install nibabel tifffile tqdm -q --no-index -f  \"/kaggle/input/vsdetection-packages-offline-installer-only\"\n# !pip install nnunetv2 -q --no-index -f \"/kaggle/input/surface-package-scraper\"\n# !pip install nnunetv2 nibabel imagecodecs tifffile tqdm lightning\n\n# IMPORTANT: Set this BEFORE importing nnunetv2\n# Blosc2 is a newer compression format but can cause compatibility issues\nos.environ[\"nnUNet_USE_BLOSC2\"] = \"1\"  # Use blosc2 format (faster, smaller files)\n\nfrom lightning import LightningModule\nimport torch.nn as nn\nimport nibabel as nib\nimport numpy as np\nimport pandas as pd\nimport tifffile\nimport matplotlib.pyplot as plt\nfrom tqdm.auto import tqdm\n\n# Show GPU configuration\nprint(f\"Available GPUs: {torch.cuda.device_count()}\")\nprint(f\"Using NUM_GPUS={NUM_GPUS}\")","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:44.063172Z","iopub.status.busy":"2026-01-21T08:53:44.062933Z","iopub.status.idle":"2026-01-21T08:53:56.915957Z","shell.execute_reply":"2026-01-21T08:53:56.915020Z"},"papermill":{"duration":12.864589,"end_time":"2026-01-21T08:53:56.917479","exception":false,"start_time":"2026-01-21T08:53:44.052890","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## DATA","metadata":{"papermill":{"duration":0.009404,"end_time":"2026-01-21T08:53:56.936539","exception":false,"start_time":"2026-01-21T08:53:56.927135","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### Utilities","metadata":{"papermill":{"duration":0.009333,"end_time":"2026-01-21T08:53:56.955196","exception":false,"start_time":"2026-01-21T08:53:56.945863","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def create_spacing_json(output_path: Path, shape: tuple, spacing: tuple = (1.0, 1.0, 1.0)):\n    \"\"\"Create JSON sidecar with spacing info for TIFF files.\"\"\"\n    json_data = {\"spacing\": list(spacing)}\n    output_path.parent.mkdir(parents=True, exist_ok=True)  # Ensure directory exists\n    with open(output_path, \"w\") as f:\n        json.dump(json_data, f, indent=2)\n        f.flush()  # Ensure data is written to disk\n    # Verify file was created\n    if not output_path.exists():\n        raise IOError(f\"Failed to create JSON file: {output_path}\")\n\ndef load_nifti(path: Path) -> np.ndarray:\n    \"\"\"Load NIfTI file (used for loading nnUNet predictions).\"\"\"\n    return nib.load(str(path)).get_fdata()\n\ndef create_dataset_json(output_dir: Path, num_training: int, file_ending: str = \".tif\") -> dict:\n    \"\"\"Create dataset.json with ignore label support and 3D TIFF reader.\"\"\"\n    \n    dataset_json = {\n        \"channel_names\": {\"0\": \"CT\"},\n        \"labels\": {\"background\": 0, \"surface\": 1, \"ignore\": 2},\n        \"numTraining\": num_training,\n        \"file_ending\": file_ending,\n        \"overwrite_image_reader_writer\": \"SimpleTiffIO\" # custom reader \n    }\n    \n    json_path = output_dir / \"dataset.json\"\n    with open(json_path, \"w\") as f:\n        json.dump(dataset_json, f, indent=4)\n    \n    print(f\"Created {json_path}\")\n    print(f\"  - {num_training} training cases\")\n    print(f\"  - Labels: background(0), surface(1), ignore(2)\")\n    print(f\"  - Reader: SimpleTiffIO (3D TIFF)\")\n    \n    return dataset_json","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:56.975825Z","iopub.status.busy":"2026-01-21T08:53:56.975337Z","iopub.status.idle":"2026-01-21T08:53:56.982029Z","shell.execute_reply":"2026-01-21T08:53:56.981415Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.018462,"end_time":"2026-01-21T08:53:56.983355","exception":false,"start_time":"2026-01-21T08:53:56.964893","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Prepare","metadata":{"papermill":{"duration":0.009052,"end_time":"2026-01-21T08:53:57.001584","exception":false,"start_time":"2026-01-21T08:53:56.992532","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"nnUNet Dataset Structure:\n```\nnnUNet_raw/Dataset100_VesuviusSurface/\n├── imagesTr/           # Training images\n│   ├── case001_0000.tif  # _0000 suffix = channel 0 (only one for CT)\n│   └── case001_0000.json # Spacing information\n├── labelsTr/           # Training labels\n│   ├── case001.tif\n│   └── case001.json\n└── dataset.json        # Dataset configuration\n```","metadata":{"papermill":{"duration":0.008975,"end_time":"2026-01-21T08:53:57.019724","exception":false,"start_time":"2026-01-21T08:53:57.010749","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def prepare_single_case(\n    src_path: Path, \n    dest_path: Path, \n    json_path: Path, \n    use_symlinks: bool = True\n) -> bool:\n    \"\"\"\n    Prepare a single TIFF file for nnUNet: create symlink/copy and JSON sidecar.\n    Returns True on success, False on failure.\n    \"\"\"\n    try: \n        # Ensure destination directory exists\n        dest_path.parent.mkdir(parents=True, exist_ok=True)\n        \n        # Read TIFF shape\n        with tifffile.TiffFile(src_path) as tif:\n            shape = tif.pages[0].shape if len(tif.pages) == 1 \\\n                    else (len(tif.pages), *tif.pages[0].shape)\n        \n        # Link or copy TIFF file\n        if use_symlinks:\n            if not dest_path.exists():\n                dest_path.symlink_to(src_path.resolve())\n        else:\n            if not dest_path.exists():\n                shutil.copy2(src_path, dest_path)\n        \n        # Create JSON sidecar\n        create_spacing_json(json_path, shape)\n        \n        # Verify both files exist\n        if not dest_path.exists():\n            raise IOError(f\"TIFF file not created: {dest_path}\")\n        if not json_path.exists():\n            raise IOError(f\"JSON file not created: {json_path}\")\n            \n        return True\n\n    except Exception as e:\n        print(f\"Error processing {src_path.name}: {e}\")\n        return False\n    \ndef prepare_training_case(\n    img_path: Path,\n    train_labels_dir: Path,\n    images_dir: Path,\n    labels_dir: Path,\n    use_symlinks: bool\n) -> bool:\n    \"\"\"Worker function for parallel dataset preparation.\"\"\"\n    case_id = img_path.stem\n    label_path = train_labels_dir / img_path.name\n    \n    if not label_path.exists():\n        return False\n    \n    img_ok = prepare_single_case(\n        src_path=img_path,\n        dest_path=images_dir / f\"{case_id}_0000.tif\",\n        json_path=images_dir / f\"{case_id}_0000.json\",\n        use_symlinks=use_symlinks\n    )\n\n    label_ok = prepare_single_case(\n        src_path=label_path,\n        dest_path=labels_dir / f\"{case_id}.tif\",\n        json_path=labels_dir / f\"{case_id}.json\",\n        use_symlinks=use_symlinks\n    )\n    return img_ok and label_ok\n\ndef prepare_dataset(input_dir: Path, max_cases: Optional[int] = None, use_symlinks: bool = True):\n    \"\"\"\n    Convert competition data to nnUNet format using TIFF directly (no NIfTI).\n    Uses multiprocessing for faster preparation.\n    \n    Competition structure:\n    - train_images/*.tif  (3D volumes)\n    - train_labels/*.tif  (3D labels: 0=bg, 1=surface, 2=ignore)\n    \"\"\"\n    dataset_dir = NNUNET_RAW / DS_NAME\n    images_dir = dataset_dir / \"imagesTr\"\n    labels_dir = dataset_dir / \"labelsTr\"\n\n    images_dir.mkdir(parents=True, exist_ok=True)\n    labels_dir.mkdir(parents=True, exist_ok=True)\n\n    train_images_dir = input_dir / \"train_images\"\n    train_labels_dir = input_dir / \"train_labels\"\n    if not train_images_dir.exists():\n        print(f\"ERROR: {train_images_dir} not found!\")\n        return None\n    \n    image_files = sorted(train_images_dir.glob(\"*.tif\"))\n    if max_cases:\n        image_files = image_files[:max_cases]\n    \n    print(f\"Found {len(image_files)} training cases\")\n    print(f\"Using {'symlinks' if use_symlinks else 'copy'}\")\n    print(f\"Processing with {NUM_WORKERS} workers...\")\n    \n    # Create worker function with fixed arguments\n    worker = partial(\n        prepare_training_case,\n        train_labels_dir=train_labels_dir,\n        images_dir=images_dir,\n        labels_dir=labels_dir,\n        use_symlinks=use_symlinks\n    )\n    # Process in parallel with progress bar\n    with Pool(NUM_WORKERS) as pool:\n        results = list(tqdm(\n            pool.imap(worker, image_files),\n            total=len(image_files),\n            desc=\"Preparing dataset\"\n        ))\n    \n    num_converted = sum(results)\n    create_dataset_json(dataset_dir, num_converted, file_ending=\".tif\")\n    \n    # Verify JSON files were created\n    json_count = len(list(images_dir.glob(\"*.json\")))\n    print(f\"\\nDataset prepared: {num_converted} cases\")\n    print(f\"JSON files created: {json_count} (expected: {num_converted * 2})\")\n    print(f\"Location: {dataset_dir}\")\n    \n    if json_count < num_converted:\n        print(f\"WARNING: Missing JSON files! Expected {num_converted}, found {json_count}\")\n    \n    return dataset_dir","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:57.039184Z","iopub.status.busy":"2026-01-21T08:53:57.038934Z","iopub.status.idle":"2026-01-21T08:53:57.049829Z","shell.execute_reply":"2026-01-21T08:53:57.049118Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.022212,"end_time":"2026-01-21T08:53:57.051103","exception":false,"start_time":"2026-01-21T08:53:57.028891","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CMD\n1. run_preprocessing() → nnUNetv2_plan_and_preprocess\n2. run_training() → nnUNetv2_train\n3. run_inference() → nnUNetv2_predict\n```\n# Preprocessing\nnnUNetv2_plan_and_preprocess -d 100 -np 4 -pl nnUNetPlannerResEncM\n\n# Training (uses ignore label automatically)\nnnUNetv2_train 100 3d_fullres 0 -p nnUNetResEncUNetMPlans\n\n# Training with fewer epochs (faster)\nnnUNetv2_train 100 3d_fullres 0 -p nnUNetResEncUNetMPlans -tr nnUNetTrainer_250epochs\n\n# Training all folds in parallel (different GPUs)\nCUDA_VISIBLE_DEVICES=0 nnUNetv2_train 100 3d_fullres 0 ... &\nCUDA_VISIBLE_DEVICES=1 nnUNetv2_train 100 3d_fullres 1 ... &\n\n# Multi-GPU DDP training (single fold, multiple GPUs)\nnnUNetv2_train 100 3d_fullres 0 ... -num_gpus 2\n\n# Inference\nnnUNetv2_predict -d 100 -c 3d_fullres -f 0 -i INPUT -o OUTPUT -p nnUNetResEncUNetMPlans\n```","metadata":{"papermill":{"duration":0.008988,"end_time":"2026-01-21T08:53:57.069130","exception":false,"start_time":"2026-01-21T08:53:57.060142","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def _run_command(\n    cmd: str,\n    name: str = \"Command\",\n    tail_lines: int = 20,\n    timeout: Optional[int] = COMMAND_TIMEOUT\n) -> bool:\n    \"\"\"\n    Execute shell command and handle output parsing.\n    \n    Args:\n        cmd: Shell command to execute\n        name: Display name for logging\n        tail_lines: Number of stdout lines to show on success\n        timeout: Timeout in seconds (None for no timeout)\n    \n    Returns:\n        True if command succeeded, False otherwise\n    \"\"\"\n    print(f\"Running: {cmd}\")\n    if timeout:\n        print(f\"Timeout: {timeout}s ({timeout/3600:.1f}h)\")\n    \n    try:\n        result = subprocess.run(\n            cmd, shell=True,\n            capture_output=True,\n            text=True,\n            timeout=timeout\n        )\n    except subprocess.TimeoutExpired:\n        print(f\"{name} timed out after {timeout} seconds.\")\n        return False\n    if result.returncode != 0:\n        print(f\"{name} failed\")\n        print(f\"STDERR:\\n{result.stderr[-3000:]}\")\n        return False\n    print(f\"{name} succeeded\")\n    if result.stdout.strip():\n        lines = result.stdout.strip().split('\\n')\n        print('\\n'.join(lines[-tail_lines:]))\n    \n    return True","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:57.088582Z","iopub.status.busy":"2026-01-21T08:53:57.087969Z","iopub.status.idle":"2026-01-21T08:53:57.093223Z","shell.execute_reply":"2026-01-21T08:53:57.092727Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.016362,"end_time":"2026-01-21T08:53:57.094474","exception":false,"start_time":"2026-01-21T08:53:57.078112","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Run preprocessing","metadata":{"papermill":{"duration":0.009153,"end_time":"2026-01-21T08:53:57.112763","exception":false,"start_time":"2026-01-21T08:53:57.103610","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def run_preprocessing(\n    dataset_id: int = DS_ID, \n    planner: str = PLANNER,\n    num_workers: int = NUM_WORKERS,\n    configurations: Optional[List[str]] = None,\n    timeout: Optional[int] = COMMAND_TIMEOUT\n) -> bool:\n    \"\"\"\n    Run nnUNet preprocessing.\n    \n    Args:\n        dataset_id: nnUNet dataset ID\n        planner: Planner class name\n        num_workers: Number of CPU workers for parallel processing\n        configurations: List of configs to preprocess (e.g., [\"3d_fullres\"])\n        timeout: Timeout in seconds (None for no timeout)\n    \n    Returns:\n        True if preprocessing succeeded\n    \"\"\"\n    if configurations is None:\n        configurations = [CONFIGURATION]\n\n    cmd = f\"nnUNetv2_plan_and_preprocess -d {dataset_id:03d} \" \\\n          f\"-np {num_workers} \" \\\n          f\"-pl {planner} \" \\\n          f\"-c {' '.join(configurations)}\"\n    return _run_command(\n        cmd,\n        name=\"Preprocessing\",\n        timeout=timeout\n    )","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:57.132052Z","iopub.status.busy":"2026-01-21T08:53:57.131822Z","iopub.status.idle":"2026-01-21T08:53:57.136322Z","shell.execute_reply":"2026-01-21T08:53:57.135620Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.015681,"end_time":"2026-01-21T08:53:57.137579","exception":false,"start_time":"2026-01-21T08:53:57.121898","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Run trainer","metadata":{"papermill":{"duration":0.009164,"end_time":"2026-01-21T08:53:57.156038","exception":false,"start_time":"2026-01-21T08:53:57.146874","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def _get_trainer_name(epochs: Optional[Epochs]) -> str:\n    \"\"\"Get trainer class name based on epochs.\"\"\"\n    if epochs is None or epochs == 1000:\n        return \"nnUNetTrainer\"\n    elif epochs == 1:\n        return \"nnUNetTrainer_1epoch\"\n    else:\n        return f\"nnUNetTrainer_{epochs}epochs\"\n\n\ndef run_training(\n    dataset_id: int = DS_ID,\n    config: str = CONFIGURATION,\n    fold: Union[int, str] = FOLD,\n    plans: str = PLANS_NAME,\n    epochs: Optional[Epochs] = EPOCHS,\n    pretrained_weights: Optional[Path] = None,\n    continue_training: bool = False,\n    only_run_validation: bool = False,\n    disable_checkpointing: bool = False,\n    npz: bool = False,\n    num_gpus: int = NUM_GPUS,\n    timeout: Optional[int] = COMMAND_TIMEOUT\n) -> bool:\n    \"\"\"\n    Run nnUNet training.\n    \n    Args:\n        dataset_id: nnUNet dataset ID\n        config: Configuration name (3d_fullres, 2d, etc.)\n        fold: Fold number (0-4) or \"all\" for training on all data\n        plans: Plans name matching the planner used\n        epochs: Number of epochs. Available: 1, 5, 10, 20, 50, 100, 150, 200, 250, \n                300, 400, 500, 750, 1000, 2000, 4000, 8000. None = 1000 (default)\n        pretrained_weights: Optional path to pretrained checkpoint for fine-tuning\n        continue_training: Continue from last checkpoint (add -c flag)\n        only_run_validation: Only run validation, skip training\n        disable_checkpointing: Disable saving checkpoints (saves disk space)\n        npz: Save softmax outputs during validation (needed for ensembling)\n        num_gpus: Number of GPUs for DDP training (default: auto-detected)\n        timeout: Timeout in seconds (None for no timeout)\n    \n    Returns:\n        True if training succeeded\n    \n    Note:\n        For multi-GPU (DDP) training, batch size should be divisible by num_gpus.\n        The first run extracts preprocessed data - wait for GPU usage before \n        starting additional folds on other GPUs.\n        \n        When using fold=\"all\", nnUNet will run validation on ALL training data\n        after training completes, which can be slow. This is normal behavior.\n    \"\"\"\n    trainer = _get_trainer_name(epochs)\n    cmd = f\"nnUNetv2_train {dataset_id:03d} {config} {fold} -p {plans} -tr {trainer}\"\n    \n    if pretrained_weights:\n        cmd += f\" -pretrained_weights {pretrained_weights}\"\n    if continue_training:\n        # Find checkpoint to resume from: prefer checkpoint_final.pth, fallback to checkpoint_best.pth\n        model_dir = get_training_output_dir(epochs=epochs, plans=plans, config=config, fold=fold)\n        checkpoint_final = model_dir / \"checkpoint_final.pth\"\n        checkpoint_best = model_dir / \"checkpoint_best.pth\"\n        if checkpoint_final.exists():\n            print(f\"Resuming from: {checkpoint_final}\")\n            cmd += \" --c\"\n        elif checkpoint_best.exists():\n            print(f\"Resuming from: {checkpoint_best}\")\n            cmd += \" --c\"\n        else:\n            print(f\"WARNING: No checkpoint found in {model_dir}, starting fresh\")\n    if only_run_validation:\n        cmd += \" --val\"\n    if disable_checkpointing:\n        cmd += \" --disable_checkpointing\"\n    if npz:\n        cmd += \" --npz\"\n    if num_gpus > 1:\n        cmd += f\" -num_gpus {num_gpus}\"\n    \n    epochs_str = epochs if epochs else 1000\n    gpu_str = f\", {num_gpus} GPUs\" if num_gpus > 1 else \"\"\n    return _run_command(cmd, f\"Training ({epochs_str} epochs{gpu_str})\", tail_lines=30, timeout=timeout)\n","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:57.175524Z","iopub.status.busy":"2026-01-21T08:53:57.175279Z","iopub.status.idle":"2026-01-21T08:53:57.182943Z","shell.execute_reply":"2026-01-21T08:53:57.182351Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.019049,"end_time":"2026-01-21T08:53:57.184256","exception":false,"start_time":"2026-01-21T08:53:57.165207","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Run inference","metadata":{"papermill":{"duration":0.009105,"end_time":"2026-01-21T08:53:57.202534","exception":false,"start_time":"2026-01-21T08:53:57.193429","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def run_inference(\n    input_dir: Path,\n    output_dir: Path,\n    dataset_id: int = DS_ID,\n    config: str = CONFIGURATION,\n    fold: Union[int, str] = FOLD,\n    plans: str = PLANS_NAME,\n    epochs: Optional[Epochs] = EPOCHS,\n    save_probabilities: bool = True,\n    num_processes_preprocessing: int = 2,\n    num_processes_segmentation: int = 2,\n    timeout: Optional[int] = COMMAND_TIMEOUT\n) -> bool:\n    \"\"\"\n    Run inference with trained model.\n    \n    Args:\n        input_dir: Directory with test images (must have _0000 suffix)\n        output_dir: Directory to save predictions\n        dataset_id: nnUNet dataset ID\n        config: Configuration name\n        fold: Fold number used for training (or \"all\" or tuple like \"0,1,2\")\n        plans: Plans name\n        epochs: Epochs used during training (must match trained model)\n        save_probabilities: Whether to save probability maps (.npz files)\n        num_processes_preprocessing: Parallel processes for preprocessing\n        num_processes_segmentation: Parallel processes for segmentation\n        timeout: Timeout in seconds (None for no timeout)\n    \n    Returns:\n        True if inference succeeded\n    \"\"\"\n    output_dir.mkdir(parents=True, exist_ok=True)\n    \n    trainer = _get_trainer_name(epochs)\n    \n    cmd = f\"nnUNetv2_predict -d {dataset_id:03d} \" \\\n          f\"-c {config} -f {fold} \" \\\n          f\"-i {input_dir} -o {output_dir} \" \\\n          f\"-p {plans} -tr {trainer} \" \\\n          f\"-npp {num_processes_preprocessing} -nps {num_processes_segmentation} \" \\\n          \"--verbose\"\n    \n    if save_probabilities:\n        cmd += \" --save_probabilities\"\n    \n    return _run_command(cmd, \"Inference\", timeout=timeout)","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:57.222084Z","iopub.status.busy":"2026-01-21T08:53:57.221849Z","iopub.status.idle":"2026-01-21T08:53:57.226991Z","shell.execute_reply":"2026-01-21T08:53:57.226440Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.016297,"end_time":"2026-01-21T08:53:57.228263","exception":false,"start_time":"2026-01-21T08:53:57.211966","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prepare_test_data(input_dir: Path, output_dir: Path, use_symlinks: bool = True) -> Path:\n    \"\"\"Prepare test TIFF images for nnUNet inference.\"\"\"\n    \n    output_dir.mkdir(parents=True, exist_ok=True)\n    \n    test_images_dir = input_dir / \"test_images\"\n    \n    if not test_images_dir.exists():\n        print(f\"ERROR: {test_images_dir} not found!\")\n        return output_dir\n    \n    test_files = sorted(test_images_dir.glob(\"*.tif\"))\n    print(f\"Found {len(test_files)} test cases\")\n    print(f\"Using {'symlinks' if use_symlinks else 'copy'}\")\n    \n    for img_path in tqdm(test_files, desc=\"Preparing test data\"):\n        case_id = img_path.stem\n        prepare_single_case(\n            img_path,\n            output_dir / f\"{case_id}_0000.tif\",\n            output_dir / f\"{case_id}_0000.json\",\n            use_symlinks\n        )\n    \n    return output_dir","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:57.247583Z","iopub.status.busy":"2026-01-21T08:53:57.247349Z","iopub.status.idle":"2026-01-21T08:53:57.251987Z","shell.execute_reply":"2026-01-21T08:53:57.251423Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.015887,"end_time":"2026-01-21T08:53:57.253293","exception":false,"start_time":"2026-01-21T08:53:57.237406","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Inference utilities","metadata":{"papermill":{"duration":0.00924,"end_time":"2026-01-21T08:53:57.271730","exception":false,"start_time":"2026-01-21T08:53:57.262490","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def load_probabilities(npz_path: Path) -> np.ndarray:\n    \"\"\"\n    Load probability maps from nnUNet inference.\n    \n    Only available if inference was run with save_probabilities=True.\n    Shape: (num_classes, D, H, W) with float32 values in [0, 1].\n    \"\"\"\n    data = np.load(npz_path)\n    return data['probabilities']\n\n\ndef predictions_to_tiff(pred_dir: Path, output_dir: Path):\n    \"\"\"\n    Convert nnUNet predictions to 3D TIFF files.\n    \n    nnUNet outputs:\n    - .npz files with probability maps (if save_probabilities=True)\n    - .tif files with predictions (our SimpleTiffIO format)\n    - .pkl files with metadata\n    \n    This function:\n    1. First tries to load .npz files and convert probabilities to binary predictions\n    2. Falls back to .tif files if .npz not found\n    3. Saves as uint8 TIFF (0=background, 1=surface)\n    \"\"\"\n    output_dir.mkdir(parents=True, exist_ok=True)\n    \n    # Try NPZ files first (probability maps)\n    npz_files = list(pred_dir.glob(\"*.npz\"))\n    tif_files = list(pred_dir.glob(\"*.tif\"))\n    nii_files = list(pred_dir.glob(\"*.nii.gz\"))\n    \n    if npz_files:\n        print(f\"Converting {len(npz_files)} NPZ probability files to TIFF...\")\n        for npz_path in tqdm(npz_files, desc=\"Converting to TIFF\"):\n            case_id = npz_path.stem\n            # Load probabilities and take argmax to get class predictions\n            probs = load_probabilities(npz_path)\n            pred = np.argmax(probs, axis=0).astype(np.uint8)\n            tifffile.imwrite(output_dir / f\"{case_id}.tif\", pred)\n    elif tif_files:\n        print(f\"Copying {len(tif_files)} TIFF prediction files...\")\n        for tif_path in tqdm(tif_files, desc=\"Copying TIFF\"):\n            case_id = tif_path.stem\n            # Load and ensure uint8\n            pred = tifffile.imread(str(tif_path)).astype(np.uint8)\n            tifffile.imwrite(output_dir / f\"{case_id}.tif\", pred)\n    # Try NIfTI as last resort (legacy format)\n    elif nii_files:\n        print(f\"Converting {len(nii_files)} NIfTI files to TIFF...\")\n        for nii_path in tqdm(nii_files, desc=\"Converting to TIFF\"):\n            case_id = nii_path.stem.replace(\".nii\", \"\")\n            pred = load_nifti(nii_path).astype(np.uint8)\n            tifffile.imwrite(output_dir / f\"{case_id}.tif\", pred)\n    else:\n        print(f\"WARNING: No prediction files found in {pred_dir}\")\n        print(f\"  Checked for: *.npz, *.tif, *.nii.gz\")","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:57.291145Z","iopub.status.busy":"2026-01-21T08:53:57.290914Z","iopub.status.idle":"2026-01-21T08:53:57.297788Z","shell.execute_reply":"2026-01-21T08:53:57.297112Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.018264,"end_time":"2026-01-21T08:53:57.299084","exception":false,"start_time":"2026-01-21T08:53:57.280820","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Pipeline","metadata":{"papermill":{"duration":0.008968,"end_time":"2026-01-21T08:53:57.317200","exception":false,"start_time":"2026-01-21T08:53:57.308232","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"```\n# Full pipeline with defaults\nfull_pipeline()\n\n# Quick experiment\nfull_pipeline(epochs=50, config=\"2d\")\n\n# Resume training\nfull_pipeline(continue_training=True, epochs=250)\n\n# Inference only (trained model exists)\nfull_pipeline(do_preprocess=False, do_train=False)\n```","metadata":{"papermill":{"duration":0.008999,"end_time":"2026-01-21T08:53:57.335226","exception":false,"start_time":"2026-01-21T08:53:57.326227","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def _parse_model_path(model_path: Union[str, Path, None]) -> Tuple[Optional[Path], Optional[int], Optional[str], Optional[str], Optional[str]]:\n    \"\"\"\n    Parse model path to extract configuration parameters.\n    \n    Model path format: .../DatasetXXX_Name/TrainerName__PlansName__Config/fold_X/checkpoint.pth\n    \n    Returns:\n        (model_dir, epochs, plans, config, fold) - extracted from path, None if not found\n    \n    Note: Parses path structure even if file doesn't exist (for path validation).\n    \"\"\"\n    if model_path is None:\n        return None, None, None, None, None\n    \n    model_path = Path(model_path)\n    \n    # Determine model directory from path (even if doesn't exist)\n    # If path ends with .pth, use parent directory\n    if model_path.suffix == \".pth\" or (model_path.exists() and model_path.is_file()):\n        model_dir = model_path.parent\n    else:\n        model_dir = model_path\n    \n    # Try to parse folder structure\n    # Expected: .../fold_X or .../TrainerName__PlansName__Config/fold_X\n    try:\n        parts = model_dir.parts\n        \n        # Find fold\n        fold = None\n        for part in reversed(parts):\n            if part.startswith(\"fold_\"):\n                fold = part.replace(\"fold_\", \"\")\n                break\n        \n        # Find trainer__plans__config\n        epochs = None\n        plans = None\n        config = None\n        for part in parts:\n            if \"__\" not in part:\n                continue\n            segments = part.split(\"__\")\n            if len(segments) >= 3:\n                trainer_name = segments[0]\n                plans = segments[1]\n                config = segments[2]\n                # Extract epochs from trainer name\n                if \"epochs\" in trainer_name:\n                    import re\n                    match = re.search(r'(\\d+)epochs?', trainer_name)\n                    if match:\n                        epochs = int(match.group(1))\n                elif trainer_name == \"nnUNetTrainer\":\n                    epochs = 1000  # Default\n                break\n        \n        return model_dir, epochs, plans, config, fold\n    except Exception:\n        return model_dir, None, None, None, None","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:57.354452Z","iopub.status.busy":"2026-01-21T08:53:57.354227Z","iopub.status.idle":"2026-01-21T08:53:57.360853Z","shell.execute_reply":"2026-01-21T08:53:57.360325Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.01792,"end_time":"2026-01-21T08:53:57.362165","exception":false,"start_time":"2026-01-21T08:53:57.344245","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def full_pipeline_1(\n    # Data options\n    max_cases: Optional[int] = None,\n    # Stage control\n    do_preprocess: bool = True,\n    do_train: bool = True,\n    do_inference: bool = True,\n    # Training options (all tunable parameters)\n    config: str = CONFIGURATION,\n    fold: Union[int, str] = FOLD,\n    planner: str = PLANNER,\n    plans: str = PLANS_NAME,\n    epochs: Optional[Epochs] = EPOCHS,\n    pretrained_weights: Optional[Path] = None,\n    continue_training: bool = False,\n    num_gpus: int = NUM_GPUS,\n    # Inference options\n    save_probabilities: bool = True,\n    # External model (for inference without training)\n    model_path: Optional[Union[str, Path]] = None,\n    # Timeout\n    timeout: Optional[int] = COMMAND_TIMEOUT,\n):\n    \"\"\"\n    Run complete pipeline: setup -> data prep -> preprocess -> train -> predict.\n    \n    Args:\n        max_cases: Limit number of training cases (None = use all)\n        do_preprocess: Run preprocessing step\n        do_train: Run training step\n        do_inference: Run inference step\n        \n        config: nnUNet configuration (3d_fullres, 2d, 3d_lowres, 3d_cascade_fullres)\n        fold: Fold number (0-4) or \"all\" for training on all data\n        planner: Planner class name\n        plans: Plans name matching the planner\n        epochs: Number of training epochs (1, 5, 10, 20, 50, 100, 150, 200, 250, \n                300, 400, 500, 750, 1000, 2000, 4000, 8000). None = 1000\n        pretrained_weights: Path to pretrained weights for fine-tuning\n        continue_training: Continue from last checkpoint\n        num_gpus: Number of GPUs for DDP training\n        \n        save_probabilities: Save probability maps during inference\n        \n        model_path: Path to trained model checkpoint or directory (str or Path).\n                    When provided with do_train=False, parameters (epochs, plans, config, fold)\n                    are auto-extracted from the path if possible.\n                    Example: \"/path/to/nnUNetTrainer_5epochs__nnUNetResEncUNetMPlans__3d_fullres/fold_all/checkpoint_best.pth\"\n        timeout: Command timeout in seconds (None for no timeout)\n    \n    Returns:\n        True if pipeline completed successfully\n    \"\"\"\n    \n    print(\"=\" * 60)\n    print(\"Vesuvius Surface Detection - nnUNet Pipeline\")\n    print(\"=\" * 60)\n    print(f\"Stages: preprocess={do_preprocess}, train={do_train}, inference={do_inference}\")\n    print(f\"Config: {config}, Fold: {fold}, Epochs: {epochs or 1000}, GPUs: {num_gpus}\")\n    \n    # 1. Setup\n    print(\"\\n[1/5] Environment setup...\")\n    setup_environment()\n    \n    # 2. Prepare raw data (always - uses symlinks, fast)\n    print(\"\\n[2/5] Preparing raw dataset (symlinks)...\")\n    raw_dataset_dir = NNUNET_RAW / DS_NAME\n    if not raw_dataset_dir.exists():\n        prepare_dataset(DATA_DIR, max_cases=max_cases)\n    else:\n        print(f\"Raw dataset already exists: {raw_dataset_dir}\")\n    \n    # 3. Preprocessing\n    if do_preprocess:\n        # Check if pre-prepared preprocessed data exists\n        if _link_prepared_preprocessed():\n            print(\"\\n[3/5] Using pre-prepared preprocessed data...\")\n        else:\n            print(\"\\n[3/5] Preprocessing...\")\n            success = run_preprocessing(planner=planner, configurations=[config], timeout=timeout)\n            if not success:\n                print(\"Preprocessing failed!\")\n                return False\n    else:\n        print(\"\\n[3/5] Skipping preprocessing...\")\n        _link_prepared_preprocessed()  # Still link if available\n    \n    \n    return True\n\ndef full_pipeline_2(\n    # Data options\n    max_cases: Optional[int] = None,\n    # Stage control\n    do_preprocess: bool = True,\n    do_train: bool = True,\n    do_inference: bool = True,\n    # Training options (all tunable parameters)\n    config: str = CONFIGURATION,\n    fold: Union[int, str] = FOLD,\n    planner: str = PLANNER,\n    plans: str = PLANS_NAME,\n    epochs: Optional[Epochs] = EPOCHS,\n    pretrained_weights: Optional[Path] = None,\n    continue_training: bool = False,\n    num_gpus: int = NUM_GPUS,\n    # Inference options\n    save_probabilities: bool = True,\n    # External model (for inference without training)\n    model_path: Optional[Union[str, Path]] = None,\n    # Timeout\n    timeout: Optional[int] = COMMAND_TIMEOUT,\n):\n    \n    # 4. Training\n    if do_train:\n        print(\"\\n[4/5] Training...\")\n        success = run_training(\n            config=config,\n            fold=fold,\n            plans=plans,\n            epochs=epochs,\n            pretrained_weights=pretrained_weights,\n            continue_training=continue_training,\n            num_gpus=num_gpus,\n            timeout=timeout\n        )\n        if not success:\n            print(\"Training failed!\")\n            return False\n    else:\n        print(\"\\n[4/5] Skipping training...\")\n        \n        # Parse model_path to extract configuration if provided\n        if model_path is not None:\n            model_path = Path(model_path) if isinstance(model_path, str) else model_path\n            _, parsed_epochs, parsed_plans, parsed_config, parsed_fold = _parse_model_path(model_path)\n            \n            # Override parameters with parsed values\n            if parsed_epochs is not None:\n                epochs = parsed_epochs\n                print(f\"  Detected epochs={epochs} from model path\")\n            if parsed_plans is not None:\n                plans = parsed_plans\n                print(f\"  Detected plans={plans} from model path\")\n            if parsed_config is not None:\n                config = parsed_config\n                print(f\"  Detected config={config} from model path\")\n            if parsed_fold is not None:\n                fold = parsed_fold\n                print(f\"  Detected fold={fold} from model path\")\n            \n            if model_path.exists():\n                print(f\"Using model: {model_path}\")\n            else:\n                print(f\"WARNING: model_path does not exist: {model_path}\")\n        \n        # Verify model exists\n        expected_model_dir = get_training_output_dir(epochs=epochs, plans=plans, config=config, fold=fold)\n        expected_model_dir.mkdir(exist_ok=True, parents=True)\n        checkpoint_final = expected_model_dir / \"checkpoint_final.pth\"\n        checkpoint_best = expected_model_dir / \"checkpoint_best.pth\"\n\n        if model_path and Path(model_path).exists():\n            # Symlink external -> final so nnUNet finds it\n            if not checkpoint_final.exists():\n                checkpoint_final.symlink_to(model_path)\n                print(f\"Map model: {checkpoint_final}\")\n            else:\n                print(f\"Model already exists: {checkpoint_final}\")\n        elif checkpoint_final.exists():\n            print(f\"Found model: {checkpoint_final}\")\n        elif checkpoint_best.exists():\n            # Symlink best -> final so nnUNet finds it\n            if not checkpoint_final.exists():\n                checkpoint_final.symlink_to(checkpoint_best)\n                print(f\"Found model: {checkpoint_best} (symlinked to checkpoint_final.pth)\")\n            else:\n                print(f\"Model already exists: {checkpoint_final}\")\n        elif do_inference:\n            print(f\"WARNING: No model found at {expected_model_dir}\")\n            print(\"  Provide model_path to a valid nnUNet checkpoint file\")\n    \n    # 5. Inference\n    if do_inference:\n        print(\"\\n[5/5] Running inference...\")\n        \n        # Prepare test data in temp location\n        test_input_dir = WORKING_DIR / \"test_input\"\n        prepare_test_data(DATA_DIR, test_input_dir)\n        \n        # Run inference\n        predictions_dir = WORKING_DIR / \"predictions\"\n        success = run_inference(\n            test_input_dir, \n            predictions_dir,\n            config=config,\n            fold=fold,\n            plans=plans,\n            epochs=epochs,\n            save_probabilities=save_probabilities,\n            timeout=timeout\n        )\n        if not success:\n            print(\"Inference failed!\")\n            return False\n        \n        # Convert predictions to TIFF\n        print(\"\\nConverting predictions to TIFF...\")\n        tiff_output_dir = OUTPUT_DIR / \"predictions_tiff\"\n        predictions_to_tiff(predictions_dir, tiff_output_dir)\n        \n        print(f\"\\nPredictions saved to: {tiff_output_dir}\")\n    else:\n        print(\"\\n[5/5] Skipping inference...\")\n    \n    print(\"\\n\" + \"=\" * 60)\n    print(\"Pipeline complete!\")\n    print(\"=\" * 60)\n    \n    # Show training progress if training was done\n    if do_train:\n        print(\"\\n[Visualization] Training progress:\")\n        show_progress(epochs=epochs, plans=plans, config=config, fold=fold)\n    \n    # Visualize predictions if inference was done\n    if do_inference:\n        print(\"\\n[Visualization] Sample prediction:\")\n        visualize_predictions(num_samples=1)\n    return True","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:57.381654Z","iopub.status.busy":"2026-01-21T08:53:57.381420Z","iopub.status.idle":"2026-01-21T08:53:57.397306Z","shell.execute_reply":"2026-01-21T08:53:57.396782Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.027234,"end_time":"2026-01-21T08:53:57.398520","exception":false,"start_time":"2026-01-21T08:53:57.371286","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def show_progress(\n    epochs: Optional[Epochs] = None,\n    plans: str = PLANS_NAME,\n    config: str = CONFIGURATION,\n    fold: Union[int, str] = FOLD\n):\n    \"\"\"\n    Display training progress image from nnUNet.\n    \n    Args:\n        epochs: Number of epochs used for training (to find correct folder)\n        plans: Plans name\n        config: Configuration name\n        fold: Fold number or \"all\"\n    \"\"\"\n    progress_path = get_progress_image_path(epochs, plans, config, fold)\n    \n    if not progress_path.exists():\n        print(f\"Progress image not found: {progress_path}\")\n        print(\"Training may not have started or completed yet.\")\n        return\n    \n    from IPython.display import Image, display\n    print(f\"Training progress: {progress_path}\")\n    display(Image(filename=str(progress_path)))\n\n\ndef plot_three_axis_cuts(\n    image_vol_path: Path,\n    mask_vol_path: Path,\n    figsize: tuple = (12, 15)\n):\n    \"\"\"\n    Plot middle slices of XY, XZ, and YZ planes for image volume and predicted mask.\n    \n    Args:\n        image_vol_path: Path to input image TIFF\n        mask_vol_path: Path to prediction mask TIFF\n        figsize: Figure size (width, height)\n    \"\"\"\n    print(f\"Visualizing: {image_vol_path.name}\")\n    \n    # Load volumes\n    image_vol = tifffile.imread(str(image_vol_path))\n    mask_vol = tifffile.imread(str(mask_vol_path)).astype(np.uint8)\n    \n    # Get dimensions\n    d, h, w = image_vol.shape\n    z_mid, y_mid, x_mid = d // 2, h // 2, w // 2\n    \n    # Extract slices\n    slices = {\n        'XY Plane (Z-axis)': (image_vol[z_mid, :, :], mask_vol[z_mid, :, :]),\n        'XZ Plane (Y-axis)': (image_vol[:, y_mid, :], mask_vol[:, y_mid, :]),\n        'YZ Plane (X-axis)': (image_vol[:, :, x_mid], mask_vol[:, :, x_mid])\n    }\n    \n    fig, axes = plt.subplots(3, 2, figsize=figsize)\n    for i, (plane_name, (img_slice, mask_slice)) in enumerate(slices.items()):\n        # Image Volume\n        axes[i, 0].imshow(img_slice, cmap='gray')\n        axes[i, 0].set_title(f\"{plane_name} - Image Volume\")\n        axes[i, 0].axis('off')\n        \n        # Mask\n        axes[i, 1].imshow(mask_slice, cmap='gray')\n        axes[i, 1].set_title(f\"{plane_name} - Predicted Mask\")\n        axes[i, 1].axis('off')\n    \n    plt.tight_layout()\n    plt.show()\n\n\ndef visualize_predictions(\n    predictions_dir: Path = OUTPUT_DIR / \"predictions_tiff\",\n    test_images_dir: Path = DATA_DIR / \"test_images\",\n    num_samples: int = 1\n):\n    \"\"\"\n    Visualize prediction results by showing image/mask pairs.\n    \n    Args:\n        predictions_dir: Directory containing prediction TIFFs\n        test_images_dir: Directory containing input test images\n        num_samples: Number of samples to visualize\n    \"\"\"\n    if not predictions_dir.exists():\n        print(f\"Predictions directory not found: {predictions_dir}\")\n        return\n    \n    predictions = sorted(predictions_dir.glob(\"*.tif\"))\n    if not predictions:\n        print(f\"No TIFF predictions found in {predictions_dir}\")\n        return\n    \n    print(f\"Found {len(predictions)} predictions\")\n    \n    for pred_path in predictions[:num_samples]:\n        image_path = test_images_dir / pred_path.name\n        if image_path.exists():\n            plot_three_axis_cuts(image_path, pred_path)\n        else:\n            print(f\"Warning: Input image not found: {image_path}\")","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:57.418031Z","iopub.status.busy":"2026-01-21T08:53:57.417833Z","iopub.status.idle":"2026-01-21T08:53:57.426951Z","shell.execute_reply":"2026-01-21T08:53:57.426387Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.020507,"end_time":"2026-01-21T08:53:57.428215","exception":false,"start_time":"2026-01-21T08:53:57.407708","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission","metadata":{"papermill":{"duration":0.009284,"end_time":"2026-01-21T08:53:57.446573","exception":false,"start_time":"2026-01-21T08:53:57.437289","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def generate_submission(\n    predictions_tiff_dir: Path = OUTPUT_DIR / \"predictions_tiff\",\n    output_zip: Path = OUTPUT_DIR / \"submission.zip\",\n    delete_after_zip: bool = True  # Default True to save space on Kaggle\n) -> Optional[Path]:\n    \"\"\"\n    Create submission ZIP from TIFF predictions.\n    \n    Args:\n        predictions_tiff_dir: Directory containing predicted TIFF files\n        output_zip: Output ZIP file path\n        delete_after_zip: Delete TIFF files after adding to ZIP (saves space)\n    \n    Returns:\n        Path to submission ZIP if successful, None otherwise\n    \"\"\"\n    import zipfile\n    \n    if not predictions_tiff_dir.exists():\n        print(f\"ERROR: Predictions directory not found: {predictions_tiff_dir}\")\n        print(\"Run inference first!\")\n        return None\n    \n    tiff_files = sorted(predictions_tiff_dir.glob(\"*.tif\"))\n    \n    if not tiff_files:\n        print(f\"No TIFF files found in {predictions_tiff_dir}\")\n        return None\n    \n    print(f\"Creating submission ZIP with {len(tiff_files)} files...\")\n    \n    with zipfile.ZipFile(output_zip, 'w', zipfile.ZIP_DEFLATED) as zipf:\n        for tiff_path in tqdm(tiff_files, desc=\"Zipping predictions\"):\n            # Add file with just the filename (no directory structure)\n            zipf.write(tiff_path, tiff_path.name)\n            \n            if delete_after_zip:\n                tiff_path.unlink()\n    \n    zip_size_mb = output_zip.stat().st_size / (1024 * 1024)\n    print(f\"Submission saved: {output_zip} ({zip_size_mb:.1f} MB)\")\n    \n    return output_zip\n\n# generate_submission()\n# generate_submission(delete_after_zip=True)  # Delete TIFFs after zipping to save space","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:57.465894Z","iopub.status.busy":"2026-01-21T08:53:57.465630Z","iopub.status.idle":"2026-01-21T08:53:57.470975Z","shell.execute_reply":"2026-01-21T08:53:57.470466Z"},"jupyter":{"source_hidden":true},"papermill":{"duration":0.01652,"end_time":"2026-01-21T08:53:57.472235","exception":false,"start_time":"2026-01-21T08:53:57.455715","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# RUN THIS","metadata":{"papermill":{"duration":0.009235,"end_time":"2026-01-21T08:53:57.492014","exception":false,"start_time":"2026-01-21T08:53:57.482779","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Default Configuration:\n\n- fold=\"all\" - Train on all data (no cross-validation)\n- config=\"3d_fullres\" - Best quality single-stage\n- epochs=1000 - Full training (can reduce to 250-500)\n- num_gpus=auto - Uses all available GPUs\n\nAfter Running:\n- Training progress displayed (progress.png)\n- Sample prediction visualized\n- Submission ZIP created at /kaggle/working/submission.zip","metadata":{"papermill":{"duration":0.009208,"end_time":"2026-01-21T08:53:57.510303","exception":false,"start_time":"2026-01-21T08:53:57.501095","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Sample pipeline with 50 epochs and time limit\n# full_pipeline(epochs=100)\n\n# Then resume when training finishes and run inference with the best model\nEPOCHS = 1000\n\nfull_pipeline_1(\n    do_preprocess=False, epochs=EPOCHS\n)\n\n# Change configuration using baseline_plan\n# Plans.json is in the preprocessed directory, not the results directory\nPLAN_PATH = NNUNET_PREPROCESSED / DS_NAME / \"plans.json\"\nprint(f\"Updating plans at: {PLAN_PATH}\")\n\nif not PLAN_PATH.exists():\n    print(f\"ERROR: Plans file not found at {PLAN_PATH}\")\n    print(\"Make sure preprocessing has been completed or pre-prepared data is linked\")\nelse:\n    default_plan = {}\n    with open(PLAN_PATH, \"r\") as f:\n        default_plan = json.load(f)\n    \n    default_plan[\"configurations\"] = baseline_plan[\"configurations\"]\n    with open(PLAN_PATH, \"w\") as f:\n        json.dump(default_plan, f, indent=4)\n    print(\"Plans updated with baseline configuration\")\n\n# full_pipeline_2(\n#     do_preprocess=False, epochs=EPOCHS\n# )\n\n\nKAGGLE_INPUT_MODEL = Path(\"/kaggle/input/vesuvius-baseline-1000e/pytorch/2000e/1/nnUNet_results\")\nif KAGGLE_INPUT_MODEL.exists():\n    print(f\"\\nCopying pretrained model from {KAGGLE_INPUT_MODEL}\")\n    print(f\"Destination: {NNUNET_RESULTS}\")\n    \n    # Remove existing results if present\n    if NNUNET_RESULTS.exists():\n        print(f\"Removing existing nnUNet_results at {NNUNET_RESULTS}\")\n        shutil.rmtree(NNUNET_RESULTS)\n    \n    # Copy the entire nnUNet_results folder\n    shutil.copytree(KAGGLE_INPUT_MODEL, NNUNET_RESULTS)\n    print(f\"Model copied successfully!\")\n    \n    # Verify checkpoint exists\n    expected_model_dir = get_training_output_dir(epochs=EPOCHS)\n    checkpoint_files = list(expected_model_dir.glob(\"*.pth\"))\n    if checkpoint_files:\n        print(f\"Found checkpoints: {[f.name for f in checkpoint_files]}\")\n    else:\n        print(f\"WARNING: No checkpoint found in {expected_model_dir}\")\nelse:\n    print(f\"\\nKaggle input model not found at {KAGGLE_INPUT_MODEL}\")\n    print(\"Starting training from scratch or using existing local checkpoint\")\n\n# full_pipeline_2(\n#     do_preprocess=False, epochs=EPOCHS,\n#     continue_training=True\n# )\nfull_pipeline_2(\n    do_preprocess=False, do_train=False,\n    model_path=\"/kaggle/input/vesuvius-baseline-1000e/pytorch/2000e/1/nnUNet_results/Dataset100_VesuviusSurface/nnUNetTrainer__nnUNetResEncUNetMPlans__3d_fullres/fold_all/checkpoint_best.pth\"\n)","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:53:57.530159Z","iopub.status.busy":"2026-01-21T08:53:57.529940Z","iopub.status.idle":"2026-01-21T08:58:16.655097Z","shell.execute_reply":"2026-01-21T08:58:16.654281Z"},"papermill":{"duration":259.147344,"end_time":"2026-01-21T08:58:16.667026","exception":false,"start_time":"2026-01-21T08:53:57.519682","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !nnUNetv2_find_best_configuration 100 -p nnUNetResEncUNetMPlans -c 3d_fullres -f 0 --disable_ensembling","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:58:16.704368Z","iopub.status.busy":"2026-01-21T08:58:16.703883Z","iopub.status.idle":"2026-01-21T08:58:16.707448Z","shell.execute_reply":"2026-01-21T08:58:16.706852Z"},"papermill":{"duration":0.023408,"end_time":"2026-01-21T08:58:16.708789","exception":false,"start_time":"2026-01-21T08:58:16.685381","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Postprocess","metadata":{"papermill":{"duration":0.016258,"end_time":"2026-01-21T08:58:16.741523","exception":false,"start_time":"2026-01-21T08:58:16.725265","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import cc3d\nfrom glob import glob\ndef remove_small_fragments(prediction_mask, min_volume=50):\n    \"\"\"\n    prediction_mask: Binary boolean array (D, H, W)\n    min_volume: Blobs smaller than this many voxels get removed\n    \"\"\"\n    \n    # 1. Label the blobs\n    # connectivity=26 means diagonals count as connections\n    labels_out = cc3d.connected_components(prediction_mask, connectivity=26)\n\n    # 2. Calculate statistics for each blob\n    # This is much faster than looping through Python lists\n    stats = cc3d.statistics(labels_out)\n    \n    # stats['voxel_counts'] gives volume of each label. Index 0 is background.\n    voxel_counts = stats['voxel_counts']\n    \n    # 3. Filter\n    # Create a mask of which labels are big enough\n    # Note: voxel_counts[0] is background, usually massive, we ignore it.\n    valid_labels = np.where(voxel_counts > min_volume)[0]\n    \n    # 4. Reconstruct the clean mask\n    # This magic line keeps only the voxels that belong to valid labels\n    clean_mask = np.isin(labels_out, valid_labels)\n    \n    # Ensure background (label 0) is not treated as a valid object unless logic demands\n    clean_mask[labels_out == 0] = False \n    \n    return clean_mask\n\nall_files = glob(\"/kaggle/working/predictions_tiff/*.tif\")\nfor filename in all_files:\n    pred_tif = tifffile.imread(filename)\n    cleaned_mask = remove_small_fragments(pred_tif.astype(bool), min_volume=100)\n    tifffile.imwrite(filename, cleaned_mask.astype('uint8'))","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:58:16.776015Z","iopub.status.busy":"2026-01-21T08:58:16.775389Z","iopub.status.idle":"2026-01-21T08:58:17.605042Z","shell.execute_reply":"2026-01-21T08:58:17.604304Z"},"papermill":{"duration":0.849013,"end_time":"2026-01-21T08:58:17.606867","exception":false,"start_time":"2026-01-21T08:58:16.757854","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Fix","metadata":{"papermill":{"duration":0.016584,"end_time":"2026-01-21T08:58:17.640725","exception":false,"start_time":"2026-01-21T08:58:17.624141","status":"completed"},"tags":[]}},{"cell_type":"code","source":"class SheetCurveLoss(nn.Module):\n    def __init__(self, w_rec=1.0, w_grad=1.0):\n        super().__init__()\n        self.w_rec = w_rec\n        self.w_grad = w_grad\n        self.l1 = nn.L1Loss(reduction='none')\n\n    def gradient_loss(self, pred, target, mask):\n        # Calculate gradients (sobel-like or simple diff) in H and W directions\n        # This forces the \"slope\" of the stitch to match the sheet\n        dy_pred = torch.abs(pred[:, :, 1:, :] - pred[:, :, :-1, :])\n        dy_gt = torch.abs(target[:, :, 1:, :] - target[:, :, :-1, :])\n        dx_pred = torch.abs(pred[:, :, :, 1:] - pred[:, :, :, :-1])\n        dx_gt = torch.abs(target[:, :, :, 1:] - target[:, :, :, :-1])\n\n        # We only care about gradient matching inside/near the hole\n        # We resize mask to match the diff dimensions\n        mask_y = mask[:, :, 1:, :]\n        mask_x = mask[:, :, :, 1:]\n\n        loss_y = torch.mean(torch.abs(dy_pred - dy_gt) * mask_y)\n        loss_x = torch.mean(torch.abs(dx_pred - dx_gt) * mask_x)\n        return loss_y + loss_x\n\n    def forward(self, pred, target, mask):\n        \"\"\"\n        pred: (B, 1, H, W) -> The filled sheet\n        target: (B, 1, H, W) -> The Ground Truth sheet\n        mask: (B, 1, H, W) -> 1.0 where the hole is, 0.0 where valid data is\n        \"\"\"\n        # 1. Reconstruction Loss (L1)\n        # We weigh the hole region higher (e.g., 5x) to force the model to fix it\n        pixel_weights = 1.0 + (mask * 4.0) \n        rec_loss = torch.mean(self.l1(pred, target) * pixel_weights)\n\n        # 2. Gradient Loss (Smoothness)\n        grad_loss = self.gradient_loss(pred, target, mask)\n\n        total = (self.w_rec * rec_loss) + (self.w_grad * grad_loss)\n        return total, {\"rec_loss\": rec_loss, \"grad_loss\": grad_loss}\n\nclass AttentionBlock(nn.Module):\n    def __init__(self, channels):\n        super().__init__()\n        self.query = nn.Conv2d(channels, channels // 8, 1)\n        self.key = nn.Conv2d(channels, channels // 8, 1)\n        self.value = nn.Conv2d(channels, channels, 1)\n        self.gamma = nn.Parameter(torch.zeros(1))\n    \n    def forward(self, x):\n        B, C, H, W = x.shape\n        q = self.query(x).view(B, -1, H * W).permute(0, 2, 1)\n        k = self.key(x).view(B, -1, H * W)\n        v = self.value(x).view(B, -1, H * W)\n        attention = F.softmax(torch.bmm(q, k), dim=-1)\n        out = torch.bmm(v, attention.permute(0, 2, 1))\n        out = out.view(B, C, H, W)\n        return self.gamma * out + x\n\nclass ConvBlock2D(nn.Module):\n    def __init__(self, in_ch, out_ch, use_attention=False):\n        super().__init__()\n        self.conv1 = nn.Conv2d(in_ch, out_ch, 3, padding=1, bias=False)\n        self.bn1 = nn.BatchNorm2d(out_ch)\n        self.conv2 = nn.Conv2d(out_ch, out_ch, 3, padding=1, bias=False)\n        self.bn2 = nn.BatchNorm2d(out_ch)\n        self.relu = nn.ReLU(inplace=True)\n        self.attention = AttentionBlock(out_ch) if use_attention else None\n        self.residual = nn.Conv2d(in_ch, out_ch, 1) if in_ch != out_ch else nn.Identity()\n    \n    def forward(self, x):\n        residual = self.residual(x)\n        out = self.relu(self.bn1(self.conv1(x)))\n        out = self.bn2(self.conv2(out))\n        if self.attention is not None:\n            out = self.attention(out)\n        out += residual\n        return self.relu(out)\n\nclass UNet25(nn.Module):\n    def __init__(self, in_channels=7, num_classes=1, base_features=24):\n        super().__init__()\n        \n        # NOTE: We add +1 to in_channels to account for the mask input\n        self.enc1 = ConvBlock2D(in_channels + 1, base_features)\n        \n        self.pool1 = nn.MaxPool2d(2)\n        self.enc2 = ConvBlock2D(base_features, base_features * 2)\n        self.pool2 = nn.MaxPool2d(2)\n        self.enc3 = ConvBlock2D(base_features * 2, base_features * 4)\n        self.pool3 = nn.MaxPool2d(2)\n        self.enc4 = ConvBlock2D(base_features * 4, base_features * 8, use_attention=True)\n        self.pool4 = nn.MaxPool2d(2)\n        self.bottleneck = ConvBlock2D(base_features * 8, base_features * 16, use_attention=True)\n        self.up4 = nn.ConvTranspose2d(base_features * 16, base_features * 8, 2, stride=2)\n        self.dec4 = ConvBlock2D(base_features * 16, base_features * 8)\n        self.up3 = nn.ConvTranspose2d(base_features * 8, base_features * 4, 2, stride=2)\n        self.dec3 = ConvBlock2D(base_features * 8, base_features * 4)\n        self.up2 = nn.ConvTranspose2d(base_features * 4, base_features * 2, 2, stride=2)\n        self.dec2 = ConvBlock2D(base_features * 4, base_features * 2)\n        self.up1 = nn.ConvTranspose2d(base_features * 2, base_features, 2, stride=2)\n        self.dec1 = ConvBlock2D(base_features * 2, base_features)\n        self.out = nn.Conv2d(base_features, num_classes, 1)\n    \n    def forward(self, x, mask):\n        # x: [B, C, H, W]\n        # mask: [B, 1, H, W]\n        \n        # 1. Early Fusion: Concatenate the image and the mask\n        x = torch.cat([x, mask], dim=1) \n        \n        enc1 = self.enc1(x)\n        enc2 = self.enc2(self.pool1(enc1))\n        enc3 = self.enc3(self.pool2(enc2))\n        enc4 = self.enc4(self.pool3(enc3))\n        bottleneck = self.bottleneck(self.pool4(enc4))\n        dec4 = self.up4(bottleneck)\n        dec4 = torch.cat([dec4, enc4], dim=1)\n        dec4 = self.dec4(dec4)\n        dec3 = self.up3(dec4)\n        dec3 = torch.cat([dec3, enc3], dim=1)\n        dec3 = self.dec3(dec3)\n        dec2 = self.up2(dec3)\n        dec2 = torch.cat([dec2, enc2], dim=1)\n        dec2 = self.dec2(dec2)\n        dec1 = self.up1(dec2)\n        dec1 = torch.cat([dec1, enc1], dim=1)\n        dec1 = self.dec1(dec1)\n        return self.out(dec1)\n\n# ==============================================================\n# ==============================================================\n# ==============================================================\n\nclass UNet25Lightning(LightningModule):\n    def __init__(self, in_channels=7, num_classes=1, base_features=24):\n        super().__init__()\n        self.save_hyperparameters()\n        \n        self.model = UNet25(\n            in_channels=in_channels,\n            num_classes=num_classes, # Should be 1 for Regression/Stitching\n            base_features=base_features\n        )\n        \n        # Initialize our custom \"Stitching\" Loss\n        self.criterion = SheetCurveLoss(w_rec=1.0, w_grad=1.0)\n\n    def calc_loss(self, outputs, labels, input_masks):\n        # Clip outputs to avoid exploding gradients during early training\n        outputs = torch.clamp(outputs, min=-50, max=50)\n        \n        # Calculate loss passing the mask so we can focus on the hole\n        total_loss, logs = self.criterion(outputs, labels, input_masks)\n        \n        # Return dict for logging\n        return total_loss, logs\n\n    def forward(self, x, mask):\n        return self.model(x, mask)\n        \n    def training_step(self, batch, batch_idx):\n        # Unpack triplet: Original Stack, Corrupted Mask (Hole), Ground Truth\n        imgs, corrupted_masks, gt_masks = batch \n        hole_mask = gt_masks - corrupted_masks \n        \n        # Forward pass with TWO inputs\n        pred_logits = self.model(imgs, corrupted_masks) \n        \n        # Calculate loss\n        loss, loss_logs = self.calc_loss(pred_logits, gt_masks, hole_mask)\n        \n        # Logging\n        self.log(\"train_loss\", loss, on_step=True, on_epoch=True, prog_bar=True, sync_dist=True)\n        self.log(\"train_rec_loss\", loss_logs[\"rec_loss\"], on_step=False, on_epoch=True, sync_dist=True)\n        self.log(\"train_grad_loss\", loss_logs[\"grad_loss\"], on_step=False, on_epoch=True, sync_dist=True)\n        \n        return loss\n\n    def validation_step(self, batch, batch_idx):\n        imgs, corrupted_masks, gt_masks = batch \n        hole_mask = gt_masks - corrupted_masks \n        \n        pred_logits = self.model(imgs, corrupted_masks)\n        loss, loss_logs = self.calc_loss(pred_logits, gt_masks, hole_mask)\n\n        self.log(\"val_loss\", loss, on_epoch=True, sync_dist=True)\n        return loss\n    \n    def predict_step(self, batch, batch_idx, dataloader_idx=0):\n        # Handle prediction where we might not have labels\n        if len(batch) == 3:\n            imgs, corrupted_masks, _ = batch\n        else:\n            imgs, corrupted_masks = batch\n            \n        pred_logits = self.model(imgs, corrupted_masks)\n        return pred_logits\n\n    def configure_optimizers(self):\n        optimizer = torch.optim.AdamW(self.parameters(), lr=config.lr, weight_decay=config.weight_decay, eps=config.eps)\n        def lr_lambda(epoch):\n            if epoch < 25:\n                return 1.0 \n            elif epoch < 40:\n                return 0.2\n            else:\n                return 0.1\n        scheduler = torch.optim.lr_scheduler.LambdaLR(optimizer, lr_lambda)\n        return {\n            'optimizer': optimizer,\n            'lr_scheduler': {\n                'scheduler': scheduler,\n                'interval': 'epoch',\n                'frequency': 1\n            }\n        }","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:58:17.674956Z","iopub.status.busy":"2026-01-21T08:58:17.674459Z","iopub.status.idle":"2026-01-21T08:58:17.700513Z","shell.execute_reply":"2026-01-21T08:58:17.699816Z"},"papermill":{"duration":0.045119,"end_time":"2026-01-21T08:58:17.701999","exception":false,"start_time":"2026-01-21T08:58:17.656880","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport numpy as np\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nfrom tqdm import tqdm\n\nclass InferenceDataset(Dataset):\n    def __init__(self, image_path, mask_path, patch_size=128, num_slices=7, overlap=0.25):\n        super().__init__()\n        self.patch_size = patch_size\n        self.num_slices = num_slices\n        \n        print(f\"Loading Inference Volumes...\")\n        self.image_vol = tifffile.imread(image_path)\n        self.mask_vol = tifffile.imread(mask_path)\n        \n        # 1. Capture Original Dimensions BEFORE padding\n        self.orig_D, self.orig_H, self.orig_W = self.image_vol.shape\n        \n        # 2. Pad volumes\n        self.image_vol = self._pad_vol(self.image_vol)\n        self.mask_vol = self._pad_vol(self.mask_vol)\n        \n        # 3. Normalize\n        if self.image_vol.max() > 255:\n            self.image_vol = self.image_vol / 65535.0\n        else:\n            self.image_vol = self.image_vol / 255.0\n            \n        # Get new padded dimensions\n        self.pad_D, self.pad_H, self.pad_W = self.image_vol.shape\n        \n        # 4. Generate Sliding Window Coordinates\n        self.coords = []\n        stride = int(patch_size * (1 - overlap))\n        \n        print(\"Generating patch coordinates...\")\n        # Iterate over the PADDED dimensions\n        for d in range(0, self.pad_D - num_slices + 1, 1): \n            for h in range(0, self.pad_H - patch_size + 1, stride):\n                for w in range(0, self.pad_W - patch_size + 1, stride):\n                    # Check mask to skip empty air\n                    mask_patch = self.mask_vol[d + num_slices//2, h:h+patch_size, w:w+patch_size]\n                    if np.any(mask_patch > 0):\n                        self.coords.append((d, h, w))\n                        \n        print(f\"Inference Queue: {len(self.coords)} patches.\")\n\n    def _pad_vol(self, vol):\n        D, H, W = vol.shape\n        pad_d = self.num_slices//2\n        pad_h = (self.patch_size - (H % self.patch_size)) % self.patch_size\n        pad_w = (self.patch_size - (W % self.patch_size)) % self.patch_size\n        vol = np.pad(\n            vol,\n            ((pad_d, pad_d), (0, 0), (0, 0)),\n            mode='constant',\n            constant_values=0\n        )\n        # Only pad H and W (Right and Bottom)\n        if pad_h > 0 or pad_w > 0:\n            vol = np.pad(\n                vol, \n                ((0, 0), (0, pad_h), (0, pad_w)), \n                mode='edge'\n            )\n        return vol\n\n    def __len__(self):\n        return len(self.coords)\n\n    def __getitem__(self, idx):\n        d, h, w = self.coords[idx]\n        \n        img_patch = self.image_vol[d : d + self.num_slices, h : h + self.patch_size, w : w + self.patch_size]\n        \n        mid_z = d + self.num_slices // 2\n        mask_patch = self.mask_vol[mid_z, h : h + self.patch_size, w : w + self.patch_size]\n        mask_patch = np.expand_dims(mask_patch, axis=0)\n        \n        img_tensor = torch.from_numpy(img_patch).float()\n        mask_tensor = torch.from_numpy(mask_patch).float()\n        \n        return img_tensor, mask_tensor, (d, h, w)","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:58:17.736509Z","iopub.status.busy":"2026-01-21T08:58:17.736094Z","iopub.status.idle":"2026-01-21T08:58:17.747052Z","shell.execute_reply":"2026-01-21T08:58:17.746288Z"},"papermill":{"duration":0.029661,"end_time":"2026-01-21T08:58:17.748406","exception":false,"start_time":"2026-01-21T08:58:17.718745","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import scipy.signal\n\ndef get_gaussian_kernel(patch_size, sigma_scale=1/8):\n    \"\"\"\n    Creates a 2D Gaussian Bump.\n    Center = 1.0, Edges ~= 0.0\n    \"\"\"\n    # Create a 1D Gaussian\n    k = scipy.signal.windows.gaussian(patch_size, std=patch_size * sigma_scale)\n    # Create 2D from 1D (Outer Product)\n    kernel_2d = np.outer(k, k)\n    return kernel_2d\n\ndef run_inference_smooth(model, image_path, mask_path, output_path, device='cuda'):\n    model.eval()\n    model.to(device)\n    \n    # 1. Setup\n    ds = InferenceDataset(image_path, mask_path, patch_size=128, overlap=0.50) \n    dl = DataLoader(ds, batch_size=32, shuffle=False, num_workers=4, pin_memory=True)\n    \n    # 2. Pre-calculate Gaussian Weight Map\n    # (Since we predict 1 slice at a time, we use a 2D kernel for the H,W plane)\n    weight_map = get_gaussian_kernel(ds.patch_size, sigma_scale=1/4)\n    weight_map = torch.from_numpy(weight_map).float().to(device)\n    \n    # Prepare Canvas (Padded size)\n    full_pred = torch.zeros((ds.mask_vol.shape), device=device, dtype=torch.float32)\n    count_map = torch.zeros((ds.mask_vol.shape), device=device, dtype=torch.float32)\n    \n    \n    with torch.no_grad():\n        for batch in tqdm(dl):\n            imgs, input_masks, coords = batch\n            imgs = imgs.to(device)\n            input_masks = input_masks.to(device)\n            \n            # Predict\n            preds = model(imgs, input_masks) # (B, 1, H, W)\n            preds = preds.squeeze(1) # (B, H, W)\n            \n            d_indices, h_indices, w_indices = coords\n            \n            for i in range(len(preds)):\n                d = d_indices[i].item()\n                h = h_indices[i].item()\n                w = w_indices[i].item()\n                target_z = d + ds.num_slices // 2\n                \n                # Apply Gaussian Weighting\n                # The prediction is multiplied by the \"Bump\" (Bright center, dark edges)\n                weighted_pred = preds[i] * weight_map\n                \n                # Add to Canvas\n                full_pred[target_z, h:h+ds.patch_size, w:w+ds.patch_size] += weighted_pred\n                count_map[target_z, h:h+ds.patch_size, w:w+ds.patch_size] += weight_map\n\n    # 3. Average & Crop\n    # Avoid div by zero\n    mask = count_map > 0\n    full_pred[mask] /= count_map[mask]\n    \n    full_pred = full_pred.cpu().numpy()\n    \n    final_output = (full_pred > 0.5).astype(np.uint8)\n    final_output = final_output[ds.num_slices//2 : ds.orig_D+ds.num_slices//2, :ds.orig_H, :ds.orig_W]\n    \n    tifffile.imwrite(output_path, final_output)","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:58:17.782608Z","iopub.status.busy":"2026-01-21T08:58:17.782249Z","iopub.status.idle":"2026-01-21T08:58:17.790647Z","shell.execute_reply":"2026-01-21T08:58:17.790087Z"},"papermill":{"duration":0.027429,"end_time":"2026-01-21T08:58:17.792083","exception":false,"start_time":"2026-01-21T08:58:17.764654","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Example Usage\nfrom glob import glob\n# DATA_DIR\n# PREPARED_DATA_DIR\nmodel = UNet25Lightning.load_from_checkpoint(\"/kaggle/input/vesuvius-ae/pytorch/default/1/logs/my_experiment/version_0/checkpoints/best_loss.ckpt\").model\nfor pred_path in glob(\"/kaggle/working/predictions_tiff/*.tif\"):\n    pred_filename = os.path.splitext(pred_path.split(\"/\")[-1])[0]\n    img_path = f\"{DATA_DIR}/test_images/{pred_filename}.tif\"\n    run_inference_smooth(model, image_path=img_path, \n                  mask_path=pred_path, \n                  output_path=pred_path)\n\nall_files = glob(\"/kaggle/working/predictions_tiff/*.tif\")\nfor filename in all_files:\n    pred_tif = tifffile.imread(filename)\n    cleaned_mask = remove_small_fragments(pred_tif.astype(bool), min_volume=100)\n    tifffile.imwrite(filename, cleaned_mask.astype('uint8'))","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:58:17.827414Z","iopub.status.busy":"2026-01-21T08:58:17.826830Z","iopub.status.idle":"2026-01-21T08:58:31.652543Z","shell.execute_reply":"2026-01-21T08:58:31.651637Z"},"papermill":{"duration":13.845397,"end_time":"2026-01-21T08:58:31.654368","exception":false,"start_time":"2026-01-21T08:58:17.808971","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generate submission ZIP\ngenerate_submission()","metadata":{"execution":{"iopub.execute_input":"2026-01-21T08:58:31.696767Z","iopub.status.busy":"2026-01-21T08:58:31.695969Z","iopub.status.idle":"2026-01-21T08:58:32.035491Z","shell.execute_reply":"2026-01-21T08:58:32.034724Z"},"papermill":{"duration":0.362057,"end_time":"2026-01-21T08:58:32.036919","exception":false,"start_time":"2026-01-21T08:58:31.674862","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}