{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"accelerator":"GPU","colab":{"gpuType":"T4","provenance":[]},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":49349,"databundleVersionId":5447706}],"dockerImageVersionId":31259,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":2055.441981,"end_time":"2026-02-01T01:21:01.983445","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-02-01T00:46:46.541464","version":"2.6.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"1dfb9843e57b4511abf2a3c8e05de871":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"23c011b8be3d45e7a3ddcfccd2444dd5":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"28e538d49e0d4b88a3f807fd06e8e2a1":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"2c513e3019594ed4ba385b0732999d53":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_1dfb9843e57b4511abf2a3c8e05de871","placeholder":"​","style":"IPY_MODEL_bd171fdfac884621822833731fcea10a","tabbable":null,"tooltip":null,"value":"model.safetensors: 100%"}},"2e822e4525d044bbb72f4eb012ed3c01":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"5a3da78c877748e4a1b1f80f5842309a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"693c8227a1704baab6c896af63ed797b":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"702488738add459a98368c26bc23a988":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"73c23fd0959e4b3bbcbf541aa3ec9775":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8490df42da89495eab85e0ca891859fb":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_2c513e3019594ed4ba385b0732999d53","IPY_MODEL_c1a82d235e194cd4b315f02510df94b1","IPY_MODEL_f6b797fe88e14f4ca5c6137722821e4e"],"layout":"IPY_MODEL_a9a5cb6fdb4b45fe81ee0ec6028ef4f2","tabbable":null,"tooltip":null}},"8c21db462b014a6e92e33feb19c13378":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_a457899b54cd44408b6eb4155b6b5bab","max":548,"min":0,"orientation":"horizontal","style":"IPY_MODEL_f7fcd500dd324ea7aa4a4ca32e530fef","tabbable":null,"tooltip":null,"value":548}},"a12b2f9c205849f1bf302f45f1f53e35":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"a457899b54cd44408b6eb4155b6b5bab":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"a9a5cb6fdb4b45fe81ee0ec6028ef4f2":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"b5627ec19194447b860bba67832fca20":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_c8a884bed820417cad1766c88c5ec954","IPY_MODEL_8c21db462b014a6e92e33feb19c13378","IPY_MODEL_c937239837ef48119f74962f382c9807"],"layout":"IPY_MODEL_2e822e4525d044bbb72f4eb012ed3c01","tabbable":null,"tooltip":null}},"b6627c01ebdc45b38cccbf88247349aa":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"bd171fdfac884621822833731fcea10a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"bdc483ecb0c04145a1c7e138a7138d62":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_e4296d06779c4a1484668b71879f45d0","max":436,"min":0,"orientation":"horizontal","style":"IPY_MODEL_23c011b8be3d45e7a3ddcfccd2444dd5","tabbable":null,"tooltip":null,"value":436}},"c1a82d235e194cd4b315f02510df94b1":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_73c23fd0959e4b3bbcbf541aa3ec9775","max":346345912,"min":0,"orientation":"horizontal","style":"IPY_MODEL_c7cf1487f95d43019a9dc12d414dec47","tabbable":null,"tooltip":null,"value":346345912}},"c7cf1487f95d43019a9dc12d414dec47":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"c8a884bed820417cad1766c88c5ec954":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_d841dd56666143a7bd2220326480519a","placeholder":"​","style":"IPY_MODEL_f5086665a4244ed28ec7784cde1313ca","tabbable":null,"tooltip":null,"value":"config.json: 100%"}},"c937239837ef48119f74962f382c9807":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_b6627c01ebdc45b38cccbf88247349aa","placeholder":"​","style":"IPY_MODEL_cede455a888a43cd97f026261571ebdb","tabbable":null,"tooltip":null,"value":" 548/548 [00:00&lt;00:00, 48.2kB/s]"}},"cede455a888a43cd97f026261571ebdb":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"d3347085be044a3aa04ffc85a1d93282":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_a12b2f9c205849f1bf302f45f1f53e35","placeholder":"​","style":"IPY_MODEL_5a3da78c877748e4a1b1f80f5842309a","tabbable":null,"tooltip":null,"value":" 436/436 [00:00&lt;00:00, 34.3kB/s]"}},"d841dd56666143a7bd2220326480519a":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"e4296d06779c4a1484668b71879f45d0":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"e98845cb58f0456990c81bd206a107a3":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"ea616fdef5b24d41bc31a1bdd36c8535":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"f5086665a4244ed28ec7784cde1313ca":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"f6b797fe88e14f4ca5c6137722821e4e":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_693c8227a1704baab6c896af63ed797b","placeholder":"​","style":"IPY_MODEL_e98845cb58f0456990c81bd206a107a3","tabbable":null,"tooltip":null,"value":" 346M/346M [00:04&lt;00:00, 67.6MB/s]"}},"f7fcd500dd324ea7aa4a4ca32e530fef":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"f859836e662f40199756c88d30f3f80a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_702488738add459a98368c26bc23a988","placeholder":"​","style":"IPY_MODEL_ea616fdef5b24d41bc31a1bdd36c8535","tabbable":null,"tooltip":null,"value":"preprocessor_config.json: 100%"}},"ffda0bc665e54fcc941d9785f46bce55":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_f859836e662f40199756c88d30f3f80a","IPY_MODEL_bdc483ecb0c04145a1c7e138a7138d62","IPY_MODEL_d3347085be044a3aa04ffc85a1d93282"],"layout":"IPY_MODEL_28e538d49e0d4b88a3f807fd06e8e2a1","tabbable":null,"tooltip":null}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.008642,"end_time":"2026-02-01T00:46:50.778963","exception":false,"start_time":"2026-02-01T00:46:50.770321","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **3D Reconstruction MASt3R wo/biplet** \n\n","metadata":{"id":"qDQLX3PArmh8","papermill":{"duration":0.011,"end_time":"2026-02-01T00:46:50.724903","exception":false,"start_time":"2026-02-01T00:46:50.713903","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Setup","metadata":{"papermill":{"duration":0.008587,"end_time":"2026-02-01T00:46:50.796117","exception":false,"start_time":"2026-02-01T00:46:50.78753","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import os\nimport sys\nimport gc\nimport h5py\nimport numpy as np\nimport torch\nimport torch.nn.functional as F\nfrom tqdm import tqdm\nfrom pathlib import Path\nimport subprocess\nfrom PIL import Image, ImageFilter\nimport struct\n\n# Transformers for DINO\nfrom transformers import AutoImageProcessor, AutoModel","metadata":{"execution":{"iopub.execute_input":"2026-02-01T00:46:50.815888Z","iopub.status.busy":"2026-02-01T00:46:50.815104Z","iopub.status.idle":"2026-02-01T00:47:41.546891Z","shell.execute_reply":"2026-02-01T00:47:41.545647Z"},"papermill":{"duration":50.744687,"end_time":"2026-02-01T00:47:41.549558","exception":false,"start_time":"2026-02-01T00:46:50.804871","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    # Feature extraction\n    N_KEYPOINTS = 4096\n    IMAGE_SIZE = 1024\n\n    # Pair selection - CRITICAL for memory\n    GLOBAL_TOPK = 20\n    MIN_MATCHES = 10\n    RATIO_THR = 1.2\n\n    # Paths\n    DINO_MODEL = \"facebook/dinov2-base\"\n\n    MAST3R_MODEL = \"/kaggle/working/mast3r/checkpoints/MASt3R_ViTLarge_BaseDecoder_512_catmlpdpt_metric.pth\"\n    MAST3R_IMAGE_SIZE = 256  # should be %16=0\n\n    # Device\n    DEVICE = torch.device('cpu')","metadata":{"execution":{"iopub.execute_input":"2026-02-01T00:47:41.570546Z","iopub.status.busy":"2026-02-01T00:47:41.569816Z","iopub.status.idle":"2026-02-01T00:47:41.576176Z","shell.execute_reply":"2026-02-01T00:47:41.57504Z"},"papermill":{"duration":0.01973,"end_time":"2026-02-01T00:47:41.578526","exception":false,"start_time":"2026-02-01T00:47:41.558796","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================================\n# Memory Management Utilities\n# ============================================================================\n\ndef clear_memory():\n    \"\"\"Aggressively clear GPU and CPU memory\"\"\"\n    gc.collect()\n    if torch.cuda.is_available():\n        torch.cuda.empty_cache()\n        torch.cuda.synchronize()\n\ndef get_memory_info():\n    \"\"\"Get current memory usage\"\"\"\n    if torch.cuda.is_available():\n        allocated = torch.cuda.memory_allocated() / 1024**3\n        reserved = torch.cuda.memory_reserved() / 1024**3\n        print(f\"GPU Memory - Allocated: {allocated:.2f}GB, Reserved: {reserved:.2f}GB\")\n    \n    import psutil\n    cpu_mem = psutil.virtual_memory().percent\n    print(f\"CPU Memory Usage: {cpu_mem:.1f}%\")\n\n# ============================================================================\n# Environment Setup\n# ============================================================================\n\ndef run_cmd(cmd, check=True, capture=False):\n    \"\"\"Run command with better error handling\"\"\"\n    print(f\"Running: {' '.join(cmd)}\")\n    result = subprocess.run(\n        cmd,\n        capture_output=capture,\n        text=True,\n        check=False\n    )\n    if check and result.returncode != 0:\n        print(f\"❌ Command failed with code {result.returncode}\")\n        if capture:\n            print(f\"STDOUT: {result.stdout}\")\n            print(f\"STDERR: {result.stderr}\")\n    return result\n\n\ndef setup_base_environment():\n    \"\"\"Setup base Python environment\"\"\"\n    print(\"\\n=== Setting up Base Environment ===\")\n    \n    # NumPy fix for Python 3.12\n    print(\"\\n📦 Fixing NumPy...\")\n    run_cmd([sys.executable, \"-m\", \"pip\", \"uninstall\", \"-y\", \"numpy\"])\n    run_cmd([sys.executable, \"-m\", \"pip\", \"install\", \"numpy==1.26.4\"])\n    \n    # PyTorch\n    print(\"\\n📦 Installing PyTorch...\")\n    run_cmd([\n        sys.executable, \"-m\", \"pip\", \"install\",\n        \"torch\", \"torchvision\", \"torchaudio\"\n    ])\n    \n    # Core utilities\n    print(\"\\n📦 Installing core utilities...\")\n    run_cmd([\n        sys.executable, \"-m\", \"pip\", \"install\",\n        \"opencv-python\",\n        \"pillow\",\n        \"imageio\",\n        \"imageio-ffmpeg\",\n        \"plyfile\",\n        \"tqdm\",\n        \"tensorboard\",\n        \"scipy\",  # for rotation conversions and image resizing\n        \"psutil\"  # for memory monitoring\n    ])\n    \n    # Transformers for DINO\n    print(\"\\n📦 Installing transformers...\")\n    run_cmd([\n        sys.executable, \"-m\", \"pip\", \"install\",\n        \"transformers==4.40.0\"\n    ])\n    \n    # pycolmap for COLMAP format\n    print(\"\\n📦 Installing pycolmap...\")\n    run_cmd([sys.executable, \"-m\", \"pip\", \"install\", \"pycolmap\"])\n    \n    print(\"✓ Base environment setup complete!\")\n\n\ndef setup_mast3r():\n    \"\"\"Install and setup MASt3R\"\"\"\n    print(\"\\n=== Setting up MASt3R ===\")\n    \n    os.chdir('/kaggle/working')\n    \n    # Remove existing installation\n    if os.path.exists('mast3r'):\n        print(\"Removing existing MASt3R installation...\")\n        os.system('rm -rf mast3r')\n    \n    # Clone repository\n    print(\"Cloning MASt3R repository...\")\n    os.system('git clone --recursive https://github.com/naver/mast3r')\n    os.chdir('/kaggle/working/mast3r')\n    \n    # Check dust3r directory\n    print(\"Checking dust3r structure...\")\n    os.system('ls -la dust3r/')\n    \n    # Install dust3r\n    print(\"Installing dust3r...\")\n    os.system('cd dust3r && python -m pip install -e .')\n    \n    # Install croco\n    print(\"Installing croco...\")\n    os.system('cd dust3r/croco && python -m pip install -e .')\n    \n    # Install requirements\n    print(\"Installing MASt3R requirements...\")\n    os.system('pip install -r requirements.txt')\n    \n    # Download model weights\n    print(\"Downloading model weights...\")\n    os.system('mkdir -p checkpoints')\n    os.system('wget -P checkpoints/ https://download.europe.naverlabs.com/ComputerVision/MASt3R/MASt3R_ViTLarge_BaseDecoder_512_catmlpdpt_metric.pth')\n    \n    # Install additional dependencies\n    print(\"Installing additional dependencies...\")\n    os.system('pip install trimesh matplotlib roma')\n    \n    # Add to path\n    sys.path.insert(0, '/kaggle/working/mast3r')\n    sys.path.insert(0, '/kaggle/working/mast3r/dust3r')\n    \n    # Verification\n    print(\"\\n🔍 Verifying MASt3R installation...\")\n    try:\n        from mast3r.model import AsymmetricMASt3R\n        print(\"  ✓ MASt3R import: OK\")\n    except Exception as e:\n        print(f\"  ❌ MASt3R import failed: {e}\")\n        raise\n    \n    print(\"✓ MASt3R setup complete!\")\n\ndef setup_gaussian_splatting():\n    \"\"\"Setup Gaussian Splatting\"\"\"\n    print(\"\\n=== Setting up Gaussian Splatting ===\")\n    \n    os.chdir('/kaggle/working')\n    \n    WORK_DIR = \"gaussian-splatting\"\n    \n    if not os.path.exists(WORK_DIR):\n        print(\"Cloning Gaussian Splatting repository...\")\n        run_cmd([\n            \"git\", \"clone\", \"--recursive\",\n            \"https://github.com/graphdeco-inria/gaussian-splatting.git\",\n            WORK_DIR\n        ])\n    else:\n        print(\"✓ Repository already exists\")\n    \n    os.chdir(WORK_DIR)\n    \n    # Install requirements\n    print(\"Installing Gaussian Splatting requirements...\")\n    run_cmd([sys.executable, \"-m\", \"pip\", \"install\", \"-r\", \"requirements.txt\"])\n    \n    # Build submodules\n    print(\"\\n📦 Building Gaussian Splatting submodules...\")\n    \n    submodules = {\n        \"diff-gaussian-rasterization\":\n            \"https://github.com/graphdeco-inria/diff-gaussian-rasterization.git\",\n        \"simple-knn\":\n            \"https://github.com/camenduru/simple-knn.git\"\n    }\n    \n    for name, repo in submodules.items():\n        print(f\"\\n📦 Installing {name}...\")\n        path = os.path.join(\"submodules\", name)\n        if not os.path.exists(path):\n            run_cmd([\"git\", \"clone\", repo, path])\n        run_cmd([sys.executable, \"-m\", \"pip\", \"install\", path])\n    \n    print(\"✓ Gaussian Splatting setup complete!\")","metadata":{"execution":{"iopub.execute_input":"2026-02-01T00:47:41.599811Z","iopub.status.busy":"2026-02-01T00:47:41.598971Z","iopub.status.idle":"2026-02-01T00:47:41.622402Z","shell.execute_reply":"2026-02-01T00:47:41.62118Z"},"papermill":{"duration":0.036598,"end_time":"2026-02-01T00:47:41.624522","exception":false,"start_time":"2026-02-01T00:47:41.587924","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"setup_base_environment()\nclear_memory()\n\nsetup_mast3r()\nclear_memory()","metadata":{"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-02-01T00:47:41.645704Z","iopub.status.busy":"2026-02-01T00:47:41.644719Z","iopub.status.idle":"2026-02-01T00:50:15.907838Z","shell.execute_reply":"2026-02-01T00:50:15.906444Z"},"papermill":{"duration":154.276646,"end_time":"2026-02-01T00:50:15.910191","exception":false,"start_time":"2026-02-01T00:47:41.633545","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"    # ============================================================================\n    # Step 0: Biplet-Square Normalization (PRESERVED FROM ORIGINAL)\n    # ============================================================================\n    \n    def normalize_image_sizes_biplet(input_dir, output_dir=None, size=1024, max_images=None):\n        \"\"\"\n        Generates two square crops (Left & Right or Top & Bottom)\n        from each image in a directory and returns the output directory\n        and the list of generated file paths.\n    \n        Args:\n            input_dir: Input directory containing source images\n            output_dir: Output directory for processed images\n            size: Target square size (default: 1024)\n            max_images: Maximum number of SOURCE images to process (default: None = all images)\n        \"\"\"\n        if output_dir is None:\n            output_dir = 'output/images_biplet'\n        os.makedirs(output_dir, exist_ok=True)\n    \n        print(f\"--- Step 1: Biplet-Square Normalization ---\")\n        print(f\"Generating 2 cropped squares (Left/Right or Top/Bottom) for each image...\")\n        print()\n    \n        generated_paths = []\n        converted_count = 0\n        size_stats = {}\n    \n        # Sort for consistent processing order\n        image_files = sorted([f for f in os.listdir(input_dir)\n                             if f.lower().endswith(('.jpg', '.jpeg', '.png'))])\n    \n        if max_images is not None:\n            image_files = image_files[:max_images]\n            print(f\"Processing limited to {max_images} source images (will generate {max_images * 2} cropped images)\")\n    \n        for img_file in image_files:\n            input_path = os.path.join(input_dir, img_file)\n            try:\n                img = Image.open(input_path)\n                original_size = img.size\n    \n                # Tracking original aspect ratios\n                size_key = f\"{original_size[0]}x{original_size[1]}\"\n                size_stats[size_key] = size_stats.get(size_key, 0) + 1\n    \n                # Generate 2 crops using the helper function\n                crops = generate_two_crops(img, size)\n                base_name, ext = os.path.splitext(img_file)\n    \n                for mode, cropped_img in crops.items():\n                    output_path = os.path.join(output_dir, f\"{base_name}_{mode}{ext}\")\n                    cropped_img.save(output_path, quality=95)\n                    generated_paths.append(output_path)\n    \n                converted_count += 1\n                print(f\"  ✓ {img_file}: {original_size} → 2 square images generated\")\n    \n            except Exception as e:\n                print(f\"  ✗ Error processing {img_file}: {e}\")\n    \n        print(f\"\\nProcessing complete: {converted_count} source images processed\")\n        print(f\"Total output images: {len(generated_paths)}\")\n        print(f\"Original size distribution: {size_stats}\")\n    \n        return output_dir, generated_paths\n    \n    \n    def generate_two_crops(img, size):\n        \"\"\"\n        Generates two square crops from an image.\n        \"\"\"\n        # If size is a tuple or list, extract the first value\n        if isinstance(size, (tuple, list)):\n            size = size[0]\n        \n        width, height = img.size\n        crops = {}\n        \n        if width >= height:\n            # Landscape: Split into left and right squares\n            box_left = (0, 0, height, height)\n            box_right = (width - height, 0, width, height)\n            crops['left'] = img.crop(box_left).resize((size, size), Image.LANCZOS)\n            crops['right'] = img.crop(box_right).resize((size, size), Image.LANCZOS)\n        else:\n            # Portrait: Split into top and bottom squares\n            box_top = (0, 0, width, width)\n            box_bottom = (0, height - width, width, height)\n            crops['top'] = img.crop(box_top).resize((size, size), Image.LANCZOS)\n            crops['bottom'] = img.crop(box_bottom).resize((size, size), Image.LANCZOS)\n        \n        return crops","metadata":{"execution":{"iopub.execute_input":"2026-02-01T00:50:16.167059Z","iopub.status.busy":"2026-02-01T00:50:16.166116Z","iopub.status.idle":"2026-02-01T00:50:16.178481Z","shell.execute_reply":"2026-02-01T00:50:16.177349Z"},"papermill":{"duration":0.145414,"end_time":"2026-02-01T00:50:16.180835","exception":false,"start_time":"2026-02-01T00:50:16.035421","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# ============================================================================\n# Step 1: DINO-based Pair Selection (PRESERVED FROM ORIGINAL)\n# ============================================================================\n\ndef load_torch_image(fname, device):\n    \"\"\"Load image as torch tensor\"\"\"\n    import torchvision.transforms as T\n\n    img = Image.open(fname).convert('RGB')\n    transform = T.Compose([\n        T.ToTensor(),\n    ])\n    return transform(img).unsqueeze(0).to(device)\n\ndef extract_dino_global(image_paths, model_path, device):\n    \"\"\"Extract DINO global descriptors with memory management\"\"\"\n    print(\"\\n=== Extracting DINO Global Features ===\")\n    print(\"Initial memory state:\")\n    get_memory_info()\n\n    processor = AutoImageProcessor.from_pretrained(model_path)\n    model = AutoModel.from_pretrained(model_path).eval().to(device)\n\n    global_descs = []\n    batch_size = 4  # Small batch to save memory\n    \n    for i in tqdm(range(0, len(image_paths), batch_size)):\n        batch_paths = image_paths[i:i+batch_size]\n        batch_imgs = []\n        \n        for img_path in batch_paths:\n            img = load_torch_image(img_path, device)\n            batch_imgs.append(img)\n        \n        batch_tensor = torch.cat(batch_imgs, dim=0)\n        \n        with torch.no_grad():\n            inputs = processor(images=batch_tensor, return_tensors=\"pt\", do_rescale=False).to(device)\n            outputs = model(**inputs)\n            desc = F.normalize(outputs.last_hidden_state[:, 1:].max(dim=1)[0], dim=1, p=2)\n            global_descs.append(desc.cpu())\n        \n        # Clear batch memory\n        del batch_tensor, inputs, outputs, desc\n        clear_memory()\n\n    global_descs = torch.cat(global_descs, dim=0)\n\n    del model, processor\n    clear_memory()\n    \n    print(\"After DINO extraction:\")\n    get_memory_info()\n\n    return global_descs\n\ndef build_topk_pairs(global_feats, k, device):\n    \"\"\"Build top-k similar pairs from global features\"\"\"\n    g = global_feats.to(device)\n    sim = g @ g.T\n    sim.fill_diagonal_(-1)\n\n    N = sim.size(0)\n    k = min(k, N - 1)\n\n    topk_indices = torch.topk(sim, k, dim=1).indices.cpu()\n\n    pairs = []\n    for i in range(N):\n        for j in topk_indices[i]:\n            j = j.item()\n            if i < j:\n                pairs.append((i, j))\n\n    # Remove duplicates\n    pairs = list(set(pairs))\n    \n    return pairs\n\ndef select_diverse_pairs(pairs, max_pairs, num_images):\n    \"\"\"\n    Select diverse pairs to ensure good image coverage\n    Strategy: Select pairs that maximize image coverage\n    \"\"\"\n    import random\n    random.seed(42)\n    \n    if len(pairs) <= max_pairs:\n        return pairs\n    \n    print(f\"Selecting {max_pairs} diverse pairs from {len(pairs)} candidates...\")\n    \n    # Count how many times each image appears in pairs\n    image_counts = {i: 0 for i in range(num_images)}\n    for i, j in pairs:\n        image_counts[i] += 1\n        image_counts[j] += 1\n    \n    # Sort pairs by: prefer pairs with less-connected images\n    def pair_score(pair):\n        i, j = pair\n        # Lower score = images appear in fewer pairs = more diverse\n        return image_counts[i] + image_counts[j]\n    \n    pairs_scored = [(pair, pair_score(pair)) for pair in pairs]\n    pairs_scored.sort(key=lambda x: x[1])\n    \n    # Select pairs greedily to maximize coverage\n    selected = []\n    selected_images = set()\n    \n    # Phase 1: Select pairs that add new images (greedy coverage)\n    for pair, score in pairs_scored:\n        if len(selected) >= max_pairs:\n            break\n        i, j = pair\n        # Prefer pairs that include new images\n        if i not in selected_images or j not in selected_images:\n            selected.append(pair)\n            selected_images.add(i)\n            selected_images.add(j)\n    \n    # Phase 2: Fill remaining slots with high-similarity pairs\n    if len(selected) < max_pairs:\n        remaining = [p for p, s in pairs_scored if p not in selected]\n        random.shuffle(remaining)\n        selected.extend(remaining[:max_pairs - len(selected)])\n    \n    print(f\"Selected pairs cover {len(selected_images)} / {num_images} images ({100*len(selected_images)/num_images:.1f}%)\")\n    \n    return selected\n\ndef get_image_pairs_dino(image_paths, max_pairs=None):\n    \"\"\"DINO-based pair selection with intelligent limiting\"\"\"\n    device = Config.DEVICE\n\n    # DINO global features\n    global_feats = extract_dino_global(image_paths, Config.DINO_MODEL, device)\n    pairs = build_topk_pairs(global_feats, Config.GLOBAL_TOPK, device)\n\n    print(f\"Initial pairs from DINO: {len(pairs)}\")\n    \n    # Apply intelligent pair selection if limit specified\n    if max_pairs and len(pairs) > max_pairs:\n        pairs = select_diverse_pairs(pairs, max_pairs, len(image_paths))\n    \n    return pairs\n\n# ============================================================================\n# Step 2: MASt3R Reconstruction (REPLACES ALIKED/LIGHTGLUE/COLMAP)\n# ============================================================================\n\ndef load_mast3r_model(device='cuda'):\n    \"\"\"Load MASt3R model\"\"\"\n    from mast3r.model import AsymmetricMASt3R\n    \n    model = AsymmetricMASt3R.from_pretrained(Config.MAST3R_MODEL).to(device)\n    model.eval()\n    \n    print(f\"✓ MASt3R model loaded on {device}\")\n    return model\n\ndef load_images_for_mast3r(image_paths, size=224):\n    \"\"\"Load images using DUSt3R's format with reduced size\"\"\"\n    print(f\"\\n=== Loading images for MASt3R (size={size}) ===\")\n    \n    from dust3r.utils.image import load_images\n    \n    # Load images using DUSt3R's loader with reduced size\n    images = load_images(image_paths, size=size, verbose=True)\n    \n    return images\n\ndef run_mast3r_pairs(model, image_paths, pairs, device='cuda', batch_size=1, max_pairs=None):\n    \"\"\"Run MASt3R on selected pairs with memory management\"\"\"\n    print(\"\\n=== Running MASt3R Reconstruction ===\")\n    print(\"Initial memory state:\")\n    get_memory_info()\n    \n    from dust3r.inference import inference\n    from dust3r.cloud_opt import global_aligner, GlobalAlignerMode\n    \n    # Limit number of pairs if specified\n    if max_pairs and len(pairs) > max_pairs:\n        print(f\"Limiting pairs from {len(pairs)} to {max_pairs}\")\n        # Select pairs more evenly distributed\n        step = max(1, len(pairs) // max_pairs)\n        pairs = pairs[::step][:max_pairs]\n    \n    print(f\"Processing {len(pairs)} pairs...\")\n    \n    # Load images in smaller size\n    print(f\"Loading {len(image_paths)} images at {Config.MAST3R_IMAGE_SIZE}x{Config.MAST3R_IMAGE_SIZE}...\")\n    images = load_images_for_mast3r(image_paths, size=Config.MAST3R_IMAGE_SIZE)\n    \n    print(f\"Loaded {len(images)} images\")\n    print(\"After loading images:\")\n    get_memory_info()\n    \n    # Create all image pairs at once\n    print(f\"Creating {len(pairs)} image pairs...\")\n    mast3r_pairs = []\n    for idx1, idx2 in tqdm(pairs, desc=\"Preparing pairs\"):\n        mast3r_pairs.append((images[idx1], images[idx2]))\n    \n    print(f\"Running MASt3R inference on {len(mast3r_pairs)} pairs...\")\n    \n    # Run inference (this returns the dict format we need)\n    output = inference(mast3r_pairs, model, device, batch_size=batch_size, verbose=True)\n    \n    # Clear pairs from memory\n    del mast3r_pairs\n    clear_memory()\n    \n    print(\"✓ MASt3R inference complete\")\n    print(\"After inference:\")\n    get_memory_info()\n    \n    # Global alignment\n    print(\"Running global alignment...\")\n    scene = global_aligner(\n        output, \n        device=device, \n        mode=GlobalAlignerMode.PointCloudOptimizer\n    )\n    \n    # Clear output after creating scene\n    del output\n    clear_memory()\n    \n    print(\"Computing global alignment...\")\n    loss = scene.compute_global_alignment(\n        init=\"mst\", \n        niter=150,  # Reduced from 300\n        schedule='cosine', \n        lr=0.01\n    )\n    \n    print(f\"✓ Global alignment complete (final loss: {loss:.6f})\")\n    print(\"Final memory state:\")\n    get_memory_info()\n    \n    return scene, images","metadata":{"execution":{"iopub.execute_input":"2026-02-01T00:50:16.706101Z","iopub.status.busy":"2026-02-01T00:50:16.705658Z","iopub.status.idle":"2026-02-01T00:50:16.733234Z","shell.execute_reply":"2026-02-01T00:50:16.731879Z"},"papermill":{"duration":0.16189,"end_time":"2026-02-01T00:50:16.735871","exception":false,"start_time":"2026-02-01T00:50:16.573981","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Process3","metadata":{"papermill":{"duration":0.130559,"end_time":"2026-02-01T00:50:16.992418","exception":false,"start_time":"2026-02-01T00:50:16.861859","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### **The MASt3R scene is converted into COLMAP-compatible outputs.**","metadata":{"papermill":{"duration":0.128,"end_time":"2026-02-01T00:50:17.249655","exception":false,"start_time":"2026-02-01T00:50:17.121655","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# ============================================================================\n# COLMAP Conversion (process3_10.py) - COMPLETE FIXED VERSION\n# ============================================================================\n\nimport numpy as np\nimport cv2\nfrom pathlib import Path\nimport struct\nfrom scipy.spatial.transform import Rotation\nimport torch\nfrom PIL import Image\n\n\ndef write_next_bytes(fid, data, format_str):\n    \"\"\"Helper function to write bytes to file\"\"\"\n    if isinstance(data, (list, tuple, np.ndarray)):\n        fid.write(struct.pack(\"<\" + format_str, *data))\n    else:\n        fid.write(struct.pack(\"<\" + format_str, data))\n\n\ndef matrix_to_quaternion_translation(matrix: np.ndarray):\n    \"\"\"Robust conversion of 4x4 transformation matrix to quaternion and translation.\"\"\"\n    R = matrix[:3, :3]\n    t = matrix[:3, 3]\n\n    # Use scipy for robust quaternion conversion\n    rot = Rotation.from_matrix(R)\n    quat = rot.as_quat()  # Returns [x, y, z, w]\n    \n    # COLMAP format is [w, x, y, z]\n    qvec = np.array([quat[3], quat[0], quat[1], quat[2]])\n\n    return qvec, t\n\n\ndef write_cameras_binary(cameras, path_to_model_file):\n    \"\"\"Write COLMAP cameras.bin file\"\"\"\n    with open(path_to_model_file, \"wb\") as fid:\n        write_next_bytes(fid, len(cameras), \"Q\")\n        for camera_id, cam in cameras.items():\n            model_id = 1  # PINHOLE\n            write_next_bytes(fid, camera_id, \"I\")\n            write_next_bytes(fid, model_id, \"I\")\n            write_next_bytes(fid, cam['width'], \"Q\")\n            write_next_bytes(fid, cam['height'], \"Q\")\n            for p in cam['params']:\n                write_next_bytes(fid, float(p), \"d\")\n\n\ndef write_images_binary(images, path_to_model_file):\n    \"\"\"Write COLMAP images.bin file\"\"\"\n    with open(path_to_model_file, \"wb\") as fid:\n        write_next_bytes(fid, len(images), \"Q\")\n        for image_id, img in images.items():\n            write_next_bytes(fid, image_id, \"I\")\n            write_next_bytes(fid, img['qvec'], \"dddd\")\n            write_next_bytes(fid, img['tvec'], \"ddd\")\n            write_next_bytes(fid, img['camera_id'], \"I\")\n            \n            # Write image name\n            for char in img['name']:\n                write_next_bytes(fid, char.encode(\"utf-8\"), \"c\")\n            write_next_bytes(fid, b\"\\x00\", \"c\")\n            \n            # Write 2D points\n            write_next_bytes(fid, len(img['xys']), \"Q\")\n            for xy, point3D_id in zip(img['xys'], img['point3D_ids']):\n                write_next_bytes(fid, xy, \"dd\")\n                write_next_bytes(fid, point3D_id, \"Q\")\n\n\ndef write_points3d_binary(points3D, path_to_model_file):\n    \"\"\"\n    Write COLMAP points3D.bin file\n    \n    Args:\n        points3D: list or dict of 3D point data\n        path_to_model_file: path to points3D.bin\n    \"\"\"\n    with open(path_to_model_file, \"wb\") as fid:\n        # Write number of points\n        if isinstance(points3D, dict):\n            write_next_bytes(fid, len(points3D), \"Q\")\n            points_iter = points3D.values()\n        else:\n            write_next_bytes(fid, len(points3D), \"Q\")\n            points_iter = points3D\n        \n        # Write each point\n        for point_id, point in enumerate(points_iter):\n            # Handle both dict with 'id' key and list with index\n            if isinstance(point, dict) and 'id' in point:\n                pid = point['id']\n            else:\n                pid = point_id\n            \n            write_next_bytes(fid, pid, \"Q\")\n            write_next_bytes(fid, point['xyz'], \"ddd\")\n            write_next_bytes(fid, point['rgb'], \"BBB\")\n            write_next_bytes(fid, point['error'], \"d\")\n            \n            # Write track\n            track_length = len(point['image_ids'])\n            write_next_bytes(fid, track_length, \"Q\")\n            for image_id, point2D_idx in zip(point['image_ids'], point['point2D_idxs']):\n                write_next_bytes(fid, int(image_id), \"I\")\n                write_next_bytes(fid, int(point2D_idx), \"I\")\n\n\ndef save_image_data(scene, images_dir, depth_dir, normal_dir, mask_dir, min_conf_thr, verbose, processed_image_paths=None):\n    \"\"\"Save RGB images, depth maps, normal maps, and masks\"\"\"\n    if verbose:\n        print(\"\\nSaving image data...\")\n    \n    # Ensure directories exist\n    images_dir.mkdir(parents=True, exist_ok=True)\n    depth_dir.mkdir(parents=True, exist_ok=True)\n    normal_dir.mkdir(parents=True, exist_ok=True)\n    mask_dir.mkdir(parents=True, exist_ok=True)\n    \n    # Get the number of views\n    if hasattr(scene, 'imgs'):\n        num_views = len(scene.imgs)\n        imgs = scene.imgs\n    elif hasattr(scene, 'views'):\n        num_views = len(scene.views)\n        imgs = scene.views\n    else:\n        if verbose:\n            print(\"  Warning: Cannot access views\")\n        return\n    \n    # Use processed images if provided\n    if processed_image_paths is not None and len(processed_image_paths) > 0:\n        if verbose:\n            print(f\"  Using {len(processed_image_paths)} processed images\")\n        \n        import shutil\n        for idx, src_path in enumerate(processed_image_paths):\n            if idx >= num_views:\n                break\n            \n            try:\n                # Copy processed images\n                dst_path = images_dir / f'image_{idx:04d}.jpg'\n                shutil.copy2(src_path, dst_path)\n                \n                if verbose and idx < 3:\n                    print(f\"  Copied image {idx}: {Path(src_path).name}\")\n            except Exception as e:\n                if verbose:\n                    print(f\"  Error copying image {idx}: {e}\")\n    else:\n        # If no processed images, extract images from the scene\n        if verbose:\n            print(\"  No processed images provided, extracting from scene...\")\n        \n        for idx in range(num_views):\n            try:\n                # Save RGB images\n                img_path = images_dir / f'image_{idx:04d}.jpg'\n                \n                # Retrieve image data\n                if hasattr(imgs[idx], 'img'):\n                    img = imgs[idx].img\n                elif hasattr(imgs[idx], 'image'):\n                    img = imgs[idx].image\n                else:\n                    img = imgs[idx]\n                \n                # Convert tensor to numpy array\n                if isinstance(img, torch.Tensor):\n                    img = img.detach().cpu().numpy()\n                \n                # Convert image to correct format\n                if isinstance(img, np.ndarray):\n                    # Convert (C, H, W) -> (H, W, C)\n                    if img.ndim == 3 and img.shape[0] in [1, 3, 4]:\n                        img = np.transpose(img, (1, 2, 0))\n                    \n                    # Normalize values to [0, 255] range\n                    if img.max() <= 1.0:\n                        img = (img * 255).astype(np.uint8)\n                    else:\n                        img = img.astype(np.uint8)\n                    \n                    # Convert grayscale to RGB\n                    if img.ndim == 2:\n                        img = np.stack([img, img, img], axis=-1)\n                    elif img.shape[-1] == 1:\n                        img = np.repeat(img, 3, axis=-1)\n                    \n                    # Save the image\n                    Image.fromarray(img).save(img_path)\n                    \n                    if verbose and idx < 3:\n                        print(f\"  Saved image {idx}: {img_path}\")\n            except Exception as e:\n                if verbose:\n                    print(f\"  Error saving image {idx}: {e}\")\n    \n    # Save depth maps\n    try:\n        if hasattr(scene, 'get_depthmaps'):\n            depthmaps = scene.get_depthmaps()\n            if depthmaps is not None:\n                for idx in range(min(num_views, len(depthmaps))):\n                    depth = depthmaps[idx]\n                    if isinstance(depth, torch.Tensor):\n                        depth = depth.detach().cpu().numpy()\n                    \n                    if isinstance(depth, np.ndarray):\n                        depth_path = depth_dir / f'depth_{idx:04d}.npy'\n                        np.save(depth_path, depth)\n                        \n                        if verbose and idx < 3:\n                            print(f\"  Saved depth {idx}: {depth_path}\")\n    except Exception as e:\n        if verbose:\n            print(f\"  Note: Could not save depth maps: {e}\")\n    \n    # Save masks\n    try:\n        if hasattr(scene, 'get_masks'):\n            masks = scene.get_masks()\n            if masks is not None:\n                for idx in range(min(num_views, len(masks))):\n                    mask = masks[idx]\n                    if isinstance(mask, torch.Tensor):\n                        mask = mask.detach().cpu().numpy()\n                    \n                    if isinstance(mask, np.ndarray):\n                        mask_path = mask_dir / f'mask_{idx:04d}.png'\n                        mask_img = (mask * 255).astype(np.uint8)\n                        Image.fromarray(mask_img).save(mask_path)\n                        \n                        if verbose and idx < 3:\n                            print(f\"  Saved mask {idx}: {mask_path}\")\n    except Exception as e:\n        if verbose:\n            print(f\"  Note: Could not save masks: {e}\")\n    \n    if verbose:\n        print(f\"  Completed saving {num_views} images\")\n\n\ndef extract_scene_data(scene, min_conf_thr, verbose):\n    \"\"\"Extract cameras, images, and 3D points from MASt3R scene\"\"\"\n    cameras = {}\n    images_data = {}\n    points3D = []\n    \n    if verbose:\n        print(\"\\nExtracting scene data...\")\n    \n    # Check scene structure\n    if hasattr(scene, 'imgs'):\n        num_views = len(scene.imgs)\n        imgs = scene.imgs\n    elif hasattr(scene, 'views'):\n        num_views = len(scene.views)\n        imgs = scene.views\n    else:\n        num_views = 0\n        imgs = []\n    \n    if verbose:\n        print(f\"Number of views: {num_views}\")\n    \n    # Extract camera parameters and poses\n    for idx in range(num_views):\n        # Get image size\n        if hasattr(scene, 'imshapes') and idx < len(scene.imshapes):\n            height, width = scene.imshapes[idx]\n        else:\n            height, width = 192, 256\n        \n        # Get intrinsics\n        fx = fy = 260.0\n        cx = width / 2.0\n        cy = height / 2.0\n        \n        try:\n            if hasattr(scene, 'get_intrinsics'):\n                K = scene.get_intrinsics()\n                if K is not None:\n                    if isinstance(K, torch.Tensor):\n                        K = K.detach().cpu().numpy()\n                    if K.ndim >= 2:\n                        K_view = K[idx] if K.ndim == 3 else K\n                        if K_view.shape[0] >= 3 and K_view.shape[1] >= 3:\n                            fx = float(K_view[0, 0])\n                            fy = float(K_view[1, 1])\n                            cx = float(K_view[0, 2])\n                            cy = float(K_view[1, 2])\n        except:\n            pass\n        \n        cameras[idx] = {\n            'model': 'PINHOLE',\n            'width': int(width),\n            'height': int(height),\n            'params': [fx, fy, cx, cy]\n        }\n        \n        # Get pose\n        qvec = np.array([1.0, 0.0, 0.0, 0.0])\n        tvec = np.array([0.0, 0.0, 0.0])\n        \n        try:\n            if hasattr(scene, 'get_im_poses'):\n                poses = scene.get_im_poses()\n                if poses is not None and idx < len(poses):\n                    pose = poses[idx]\n                    if isinstance(pose, torch.Tensor):\n                        pose = pose.detach().cpu().numpy()\n                    \n                    if isinstance(pose, np.ndarray) and pose.ndim == 2 and pose.shape == (4, 4):\n                        det = np.linalg.det(pose)\n                        if abs(det) > 1e-10:\n                            pose_inv = np.linalg.inv(pose)\n                            qvec, tvec = matrix_to_quaternion_translation(pose_inv)\n        except:\n            pass\n        \n        images_data[idx + 1] = {\n            'qvec': qvec,\n            'tvec': tvec,\n            'camera_id': idx,\n            'name': f'image_{idx:04d}.jpg',\n            'xys': np.array([]),\n            'point3D_ids': np.array([])\n        }\n    \n    # Extract 3D points WITH COLORS\n    if verbose:\n        print(\"\\nExtracting 3D points with colors...\")\n    \n    try:\n        if hasattr(scene, 'get_pts3d'):\n            pts3d = scene.get_pts3d()\n            \n            if pts3d is not None:\n                # Handle list of arrays\n                if isinstance(pts3d, list):\n                    all_points = []\n                    all_colors = []\n                    \n                    for view_idx, pts in enumerate(pts3d):\n                        if isinstance(pts, torch.Tensor):\n                            pts = pts.detach().cpu().numpy()\n                        if isinstance(pts, np.ndarray):\n                            all_points.append(pts.reshape(-1, 3))\n                            \n                            # Extract colors from corresponding image\n                            if view_idx < len(imgs):\n                                img = imgs[view_idx]\n                                if isinstance(img, torch.Tensor):\n                                    img = img.detach().cpu().numpy()\n                                \n                                # Convert image format\n                                if img.ndim == 3:\n                                    # (C, H, W) -> (H, W, C)\n                                    if img.shape[0] in [1, 3, 4]:\n                                        img = np.transpose(img, (1, 2, 0))\n                                \n                                # Normalize to 0-255\n                                if img.max() <= 1.0:\n                                    img = (img * 255).astype(np.uint8)\n                                else:\n                                    img = img.astype(np.uint8)\n                                \n                                # Handle grayscale\n                                if img.ndim == 2 or img.shape[-1] == 1:\n                                    img = np.stack([img.squeeze()] * 3, axis=-1)\n                                \n                                # Reshape to match points\n                                img_flat = img.reshape(-1, 3)\n                                all_colors.append(img_flat)\n                            else:\n                                # Default gray if no image available\n                                n_pts = pts.reshape(-1, 3).shape[0]\n                                all_colors.append(np.full((n_pts, 3), 128, dtype=np.uint8))\n                    \n                    pts3d_combined = np.vstack(all_points) if all_points else None\n                    colors_combined = np.vstack(all_colors) if all_colors else None\n                        \n                elif isinstance(pts3d, torch.Tensor):\n                    pts3d_combined = pts3d.detach().cpu().numpy().reshape(-1, 3)\n                    \n                    # Extract colors from first image\n                    if len(imgs) > 0:\n                        img = imgs[0]\n                        if isinstance(img, torch.Tensor):\n                            img = img.detach().cpu().numpy()\n                        \n                        if img.ndim == 3 and img.shape[0] in [1, 3, 4]:\n                            img = np.transpose(img, (1, 2, 0))\n                        \n                        if img.max() <= 1.0:\n                            img = (img * 255).astype(np.uint8)\n                        else:\n                            img = img.astype(np.uint8)\n                        \n                        if img.ndim == 2 or img.shape[-1] == 1:\n                            img = np.stack([img.squeeze()] * 3, axis=-1)\n                        \n                        colors_combined = img.reshape(-1, 3)\n                    else:\n                        colors_combined = None\n                        \n                elif isinstance(pts3d, np.ndarray):\n                    pts3d_combined = pts3d.reshape(-1, 3)\n                    \n                    # Extract colors from first image\n                    if len(imgs) > 0:\n                        img = imgs[0]\n                        if isinstance(img, torch.Tensor):\n                            img = img.detach().cpu().numpy()\n                        \n                        if img.ndim == 3 and img.shape[0] in [1, 3, 4]:\n                            img = np.transpose(img, (1, 2, 0))\n                        \n                        if img.max() <= 1.0:\n                            img = (img * 255).astype(np.uint8)\n                        else:\n                            img = img.astype(np.uint8)\n                        \n                        if img.ndim == 2 or img.shape[-1] == 1:\n                            img = np.stack([img.squeeze()] * 3, axis=-1)\n                        \n                        colors_combined = img.reshape(-1, 3)\n                    else:\n                        colors_combined = None\n                else:\n                    pts3d_combined = None\n                    colors_combined = None\n                \n                if pts3d_combined is not None and len(pts3d_combined) > 0:\n                    # Get confidence\n                    conf_combined = None\n                    if hasattr(scene, 'get_conf'):\n                        conf = scene.get_conf()\n                        if conf is not None:\n                            if isinstance(conf, list):\n                                all_conf = []\n                                for c in conf:\n                                    if isinstance(c, torch.Tensor):\n                                        c = c.detach().cpu().numpy()\n                                    all_conf.append(c.flatten())\n                                conf_combined = np.concatenate(all_conf) if all_conf else None\n                            elif isinstance(conf, torch.Tensor):\n                                conf_combined = conf.detach().cpu().numpy().flatten()\n                            elif isinstance(conf, np.ndarray):\n                                conf_combined = conf.flatten()\n                    \n                    # Ensure all arrays have the same size\n                    min_size = len(pts3d_combined)\n                    if colors_combined is not None:\n                        min_size = min(min_size, len(colors_combined))\n                    if conf_combined is not None:\n                        min_size = min(min_size, len(conf_combined))\n                    \n                    pts3d_combined = pts3d_combined[:min_size]\n                    if colors_combined is not None:\n                        colors_combined = colors_combined[:min_size]\n                    else:\n                        colors_combined = np.full((min_size, 3), 128, dtype=np.uint8)\n                    \n                    # Filter by confidence\n                    if conf_combined is not None and len(conf_combined) > 0:\n                        conf_combined = conf_combined[:min_size]\n                        mask = conf_combined >= min_conf_thr\n                        pts3d_filtered = pts3d_combined[mask]\n                        colors_filtered = colors_combined[mask]\n                    else:\n                        pts3d_filtered = pts3d_combined\n                        colors_filtered = colors_combined\n                    \n                    # Create point cloud with colors\n                    for pt, color in zip(pts3d_filtered, colors_filtered):\n                        if np.all(np.isfinite(pt)):\n                            points3D.append({\n                                'xyz': pt,\n                                'rgb': color.astype(np.uint8),  #use actual color\n                                'error': 0.0,\n                                'image_ids': np.array([]),\n                                'point2D_idxs': np.array([])\n                            })\n                    \n                    if verbose:\n                        print(f\"  Extracted {len(points3D)} 3D points with colors\")\n                        print(f\"  Sample colors: {[p['rgb'].tolist() for p in points3D[:3]]}\")\n    except Exception as e:\n        if verbose:\n            print(f\"  Error extracting 3D points: {e}\")\n        import traceback\n        traceback.print_exc()\n    \n    if verbose:\n        print(f\"\\nTotal: {len(cameras)} cameras, {len(images_data)} images, {len(points3D)} points\")\n    \n    return cameras, images_data, points3D\n\n\ndef convert_mast3r_to_colmap(scene, output_dir, min_conf_thr=1.5, clean_depth=True, \n                            mask_images=True, verbose=True, processed_image_paths=None):\n    \"\"\"\n    Convert MASt3R scene to COLMAP format\n    \n    Args:\n        scene: MASt3R optimized scene\n        output_dir: Output directory path\n        min_conf_thr: Minimum confidence threshold for 3D points\n        clean_depth: Whether to clean depth maps\n        mask_images: Whether to apply masks\n        verbose: Print verbose output\n        processed_image_paths: List of paths to processed (square) images\n    \"\"\"\n    output_dir = Path(output_dir)\n    sparse_dir = output_dir / \"sparse\" / \"0\"\n    images_dir = output_dir / \"images\"\n    depth_dir = output_dir / \"depth\"\n    normal_dir = output_dir / \"normal\"\n    mask_dir = output_dir / \"mask\"\n    \n    # Create directories\n    sparse_dir.mkdir(parents=True, exist_ok=True)\n    images_dir.mkdir(parents=True, exist_ok=True)\n    depth_dir.mkdir(parents=True, exist_ok=True)\n    normal_dir.mkdir(parents=True, exist_ok=True)\n    mask_dir.mkdir(parents=True, exist_ok=True)\n    \n    if verbose:\n        print(\"\\n\" + \"=\"*70)\n        print(\"Converting MASt3R scene to COLMAP format\")\n        print(\"=\"*70)\n        print(f\"Output directory: {output_dir}\")\n    \n    cameras, images_data, points3D = extract_scene_data(scene, min_conf_thr, verbose)\n    \n    save_image_data(scene, images_dir, depth_dir, normal_dir, mask_dir, \n                    min_conf_thr, verbose, processed_image_paths=processed_image_paths)\n    \n    if verbose:\n        print(\"\\nWriting COLMAP binary files...\")\n    \n    write_cameras_binary(cameras, sparse_dir / \"cameras.bin\")\n    if verbose:\n        print(f\"  ✓ cameras.bin ({len(cameras)} cameras)\")\n    \n    write_images_binary(images_data, sparse_dir / \"images.bin\")\n    if verbose:\n        print(f\"  ✓ images.bin ({len(images_data)} images)\")\n    \n    write_points3d_binary(points3D, sparse_dir / \"points3D.bin\")\n    if verbose:\n        print(f\"  ✓ points3D.bin ({len(points3D)} points)\")\n    \n    if verbose:\n        print(\"\\n\" + \"=\"*70)\n        print(\"✓ COLMAP conversion complete!\")\n        print(\"=\"*70)\n    \n    return output_dir","metadata":{"execution":{"iopub.execute_input":"2026-02-01T00:50:17.513124Z","iopub.status.busy":"2026-02-01T00:50:17.51265Z","iopub.status.idle":"2026-02-01T00:50:17.57919Z","shell.execute_reply":"2026-02-01T00:50:17.578197Z"},"papermill":{"duration":0.201318,"end_time":"2026-02-01T00:50:17.581712","exception":false,"start_time":"2026-02-01T00:50:17.380394","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# bin to ply","metadata":{"papermill":{"duration":0.128287,"end_time":"2026-02-01T00:50:17.840736","exception":false,"start_time":"2026-02-01T00:50:17.712449","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def convert_colmap_bin_to_ply(sparse_dir, output_ply_path):\n    \"\"\"\n    Generate a PLY file from COLMAP binary files using pycolmap.\n    \n    Args:\n        sparse_dir: Path to the sparse/0 directory.\n        output_ply_path: Path where the output PLY file will be saved.\n    \"\"\"\n    import pycolmap\n    from plyfile import PlyData, PlyElement\n    import numpy as np\n    \n    print(f\"\\n=== Converting COLMAP bin to PLY ===\")\n    \n    # Load the reconstruction using pycolmap\n    reconstruction = pycolmap.Reconstruction(str(sparse_dir))\n    \n    print(f\"Loaded reconstruction:\")\n    print(f\"  - {len(reconstruction.cameras)} cameras\")\n    print(f\"  - {len(reconstruction.images)} images\")\n    print(f\"  - {len(reconstruction.points3D)} points\")\n    \n    if len(reconstruction.points3D) == 0:\n        print(\"❌ No 3D points found in reconstruction!\")\n        return 0\n    \n    # Extract 3D points and colors\n    points = []\n    colors = []\n    \n    for point3D_id, point3D in reconstruction.points3D.items():\n        points.append(point3D.xyz)\n        colors.append(point3D.color)\n    \n    points = np.array(points)\n    colors = np.array(colors)\n    \n    print(f\"\\nPoint cloud statistics:\")\n    print(f\"  Total points: {len(points)}\")\n    print(f\"  X range: [{points[:, 0].min():.3f}, {points[:, 0].max():.3f}]\")\n    print(f\"  Y range: [{points[:, 1].min():.3f}, {points[:, 1].max():.3f}]\")\n    print(f\"  Z range: [{points[:, 2].min():.3f}, {points[:, 2].max():.3f}]\")\n    \n    # Save as a PLY file\n    vertices = np.array(\n        [(p[0], p[1], p[2], c[0], c[1], c[2]) \n         for p, c in zip(points, colors)],\n        dtype=[('x', 'f4'), ('y', 'f4'), ('z', 'f4'),\n               ('red', 'u1'), ('green', 'u1'), ('blue', 'u1')]\n    )\n    \n    el = PlyElement.describe(vertices, 'vertex')\n    PlyData([el]).write(output_ply_path)\n    \n    print(f\"✓ Saved PLY file to {output_ply_path}\")\n    \n    return len(points)\n\n\n'''\n# Example usage:\ncolmap_output_dir = '/kaggle/working/output/colmap'\nsparse_dir = os.path.join(colmap_output_dir, 'sparse', '0')\nply_path = os.path.join(colmap_output_dir, 'point_cloud.ply')\nnum_points = convert_colmap_bin_to_ply(sparse_dir, ply_path)\n'''","metadata":{"execution":{"iopub.execute_input":"2026-02-01T00:50:18.132937Z","iopub.status.busy":"2026-02-01T00:50:18.132455Z","iopub.status.idle":"2026-02-01T00:50:18.147358Z","shell.execute_reply":"2026-02-01T00:50:18.146182Z"},"papermill":{"duration":0.158482,"end_time":"2026-02-01T00:50:18.149426","exception":false,"start_time":"2026-02-01T00:50:17.990944","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.13107,"end_time":"2026-02-01T00:50:18.408412","exception":false,"start_time":"2026-02-01T00:50:18.277342","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# main_pipeline","metadata":{"papermill":{"duration":0.13012,"end_time":"2026-02-01T00:50:18.671527","exception":false,"start_time":"2026-02-01T00:50:18.541407","status":"completed"},"tags":[]}},{"cell_type":"code","source":"image_dir = \"/kaggle/input/competitions/image-matching-challenge-2023/train/heritage/cyprus/images\"\noutput_dir = \"/kaggle/working/output\"\n\n#square_size=1024 \n#iterations=3000   \nmax_images=40\nmax_pairs=1000    \nmax_points=3000000    \n\nos.makedirs(output_dir, exist_ok=True)\nprocessed_image_dir = image_dir #os.path.join(output_dir, \"processed_images\")\n\n# Get original images first\noriginal_image_paths = sorted([\n    os.path.join(image_dir, f)\n    for f in os.listdir(image_dir)\n    if f.lower().endswith(('.jpg', '.jpeg', '.png', 'JPG'))\n])\n\n'''\n# Process the temp directory\nnormalize_image_sizes_biplet(\n    input_dir=image_dir,\n    output_dir=processed_image_dir,\n    size=square_size,\n    max_images=max_images\n)\n'''\n\n\n# Get processed image paths\nimage_paths = sorted([\n    os.path.join(processed_image_dir, f)\n    for f in os.listdir(processed_image_dir)\n    if f.lower().endswith(('.jpg', '.jpeg', '.png', 'JPG'))\n])\n\nprint(f\"\\n📸 Processing {len(image_paths)} images (after biplet-square)\")\nprint(f\"⚠️  Will use maximum {max_pairs} pairs to save memory\")\n\n# Step 2: DINO-based pair selection\nprint(\"\\n\" + \"=\"*70)\nprint(\"Step 2: DINO Pair Selection\")\nprint(\"=\"*70)\n\npairs = get_image_pairs_dino(image_paths, max_pairs=max_pairs)\nclear_memory()\n\nprint(f\"✓ Using {len(pairs)} pairs for reconstruction\")\n\n#processed_image_dir = image_dir ##'/kaggle/working/output/processed_images'\nprocessed_image_paths = sorted([\n    os.path.join(processed_image_dir, f) \n    for f in os.listdir(processed_image_dir) \n    if f.endswith(('.jpg', '.jpeg', '.png', 'JPG'))\n])\n\nprint(f\"Found {len(processed_image_paths)} processed images\")\n\ndevice = Config.DEVICE\nmodel = load_mast3r_model(device)\n\nscene, mast3r_images = run_mast3r_pairs(\n    model, processed_image_paths, pairs, device,\n    max_pairs=None  # Already limited in get_image_pairs_dino\n)\n\n# Clear model from memory\ndel model\nclear_memory()\n\ncolmap_dir = convert_mast3r_to_colmap(\n    scene=scene,\n    output_dir='/kaggle/working/output/colmap',\n    min_conf_thr=2.0,\n    clean_depth=False,\n    mask_images=False,\n    verbose=True,\n    processed_image_paths=processed_image_paths \n)\n","metadata":{"execution":{"iopub.execute_input":"2026-02-01T00:50:18.934464Z","iopub.status.busy":"2026-02-01T00:50:18.933917Z","iopub.status.idle":"2026-02-01T01:20:22.055534Z","shell.execute_reply":"2026-02-01T01:20:22.054041Z"},"papermill":{"duration":1803.253247,"end_time":"2026-02-01T01:20:22.058201","exception":false,"start_time":"2026-02-01T00:50:18.804954","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"colmap_output_dir='/kaggle/working/output/colmap'\nsparse_dir = os.path.join(colmap_output_dir, 'sparse', '0')\nply_path = os.path.join(colmap_output_dir, 'point_cloud.ply')\nnum_points = convert_colmap_bin_to_ply(sparse_dir, ply_path)","metadata":{"execution":{"iopub.execute_input":"2026-02-01T01:20:22.38155Z","iopub.status.busy":"2026-02-01T01:20:22.379656Z","iopub.status.idle":"2026-02-01T01:20:26.897531Z","shell.execute_reply":"2026-02-01T01:20:26.896404Z"},"papermill":{"duration":4.672089,"end_time":"2026-02-01T01:20:26.899566","exception":false,"start_time":"2026-02-01T01:20:22.227477","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Ply Viewer","metadata":{"papermill":{"duration":0.14993,"end_time":"2026-02-01T01:20:27.198278","exception":false,"start_time":"2026-02-01T01:20:27.048348","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!pip install open3d","metadata":{"_kg_hide-output":true,"execution":{"iopub.execute_input":"2026-02-01T01:20:27.502887Z","iopub.status.busy":"2026-02-01T01:20:27.501924Z","iopub.status.idle":"2026-02-01T01:20:53.432808Z","shell.execute_reply":"2026-02-01T01:20:53.431463Z"},"papermill":{"duration":26.083362,"end_time":"2026-02-01T01:20:53.435489","exception":false,"start_time":"2026-02-01T01:20:27.352127","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\nPLY Viewer for Kaggle Notebook\nDisplay PLY files from /kaggle/working directory\n\"\"\"\n\nfrom IPython.display import HTML, display\nimport base64\n\ndef display_ply_viewer(ply_file_path):\n    \"\"\"\n    Display PLY file in Kaggle notebook\n    \n    Args:\n        ply_file_path: Path to PLY file (e.g., '/kaggle/working/model.ply')\n    \"\"\"\n    \n    # Read PLY file and encode to Base64\n    with open(ply_file_path, 'rb') as f:\n        ply_data = f.read()\n        ply_base64 = base64.b64encode(ply_data).decode('utf-8')\n    \n    # Create HTML viewer\n    html_content = f\"\"\"\n    <!DOCTYPE html>\n    <html>\n    <head>\n        <meta charset=\"UTF-8\">\n        <title>PLY Viewer</title>\n        <style>\n            body {{ margin: 0; font-family: Arial, sans-serif; }}\n            #container {{ width: 100%; height: 600px; position: relative; }}\n            #info {{\n                position: absolute;\n                top: 10px;\n                left: 10px;\n                background: rgba(0,0,0,0.7);\n                color: white;\n                padding: 10px;\n                border-radius: 5px;\n                font-size: 12px;\n                z-index: 100;\n            }}\n            .controls {{\n                position: absolute;\n                top: 10px;\n                right: 10px;\n                background: rgba(0,0,0,0.7);\n                padding: 10px;\n                border-radius: 5px;\n                z-index: 100;\n            }}\n            .button {{\n                background: #4CAF50;\n                color: white;\n                border: none;\n                padding: 8px 12px;\n                margin: 2px;\n                border-radius: 3px;\n                cursor: pointer;\n                font-size: 13px;\n            }}\n            .button:hover {{\n                background: #45a049;\n            }}\n        </style>\n    </head>\n    <body>\n        <div id=\"container\">\n            <div id=\"info\">Loading...</div>\n            <div class=\"controls\">\n                <button id=\"reset-view\" class=\"button\">Reset View</button>\n            </div>\n        </div>\n\n        <script type=\"importmap\">\n        {{\n          \"imports\": {{\n            \"three\": \"https://unpkg.com/three@0.160.0/build/three.module.js\",\n            \"three/examples/jsm/loaders/PLYLoader.js\": \"https://unpkg.com/three@0.160.0/examples/jsm/loaders/PLYLoader.js\",\n            \"three/examples/jsm/controls/OrbitControls.js\": \"https://unpkg.com/three@0.160.0/examples/jsm/controls/OrbitControls.js\"\n          }}\n        }}\n        </script>\n        <script type=\"module\">\n            import * as THREE from 'three';\n            import {{ PLYLoader }} from 'three/examples/jsm/loaders/PLYLoader.js';\n            import {{ OrbitControls }} from 'three/examples/jsm/controls/OrbitControls.js';\n\n            let scene, camera, renderer, controls;\n            let currentPointCloud = null;\n\n            function init() {{\n                const container = document.getElementById('container');\n                \n                scene = new THREE.Scene();\n                scene.background = new THREE.Color(0x1a1a1a);\n\n                camera = new THREE.PerspectiveCamera(60, container.clientWidth / container.clientHeight, 0.1, 10000);\n                camera.position.set(5, 5, 10);\n\n                renderer = new THREE.WebGLRenderer({{ antialias: true }});\n                renderer.setSize(container.clientWidth, container.clientHeight);\n                container.appendChild(renderer.domElement);\n\n                controls = new OrbitControls(camera, renderer.domElement);\n                controls.enableDamping = true;\n                controls.dampingFactor = 0.05;\n\n                const ambientLight = new THREE.AmbientLight(0xffffff, 0.6);\n                scene.add(ambientLight);\n\n                const directionalLight = new THREE.DirectionalLight(0xffffff, 0.8);\n                directionalLight.position.set(10, 20, 15);\n                scene.add(directionalLight);\n\n                const gridHelper = new THREE.GridHelper(100, 50, 0x444444, 0x222222);\n                scene.add(gridHelper);\n\n                document.getElementById('reset-view').addEventListener('click', resetView);\n\n                loadPLY();\n                animate();\n            }}\n\n            function loadPLY() {{\n                const plyBase64 = '{ply_base64}';\n                const binaryString = atob(plyBase64);\n                const bytes = new Uint8Array(binaryString.length);\n                for (let i = 0; i < binaryString.length; i++) {{\n                    bytes[i] = binaryString.charCodeAt(i);\n                }}\n\n                const loader = new PLYLoader();\n                try {{\n                    const geometry = loader.parse(bytes.buffer);\n                    displayGeometry(geometry);\n                }} catch (error) {{\n                    document.getElementById('info').innerHTML = 'Error: ' + error.message;\n                }}\n            }}\n\n            function displayGeometry(geometry) {{\n                const hasColors = geometry.attributes.color !== undefined;\n                \n                // Fixed point size: 0.002\n                const material = new THREE.PointsMaterial({{\n                    size: 0.002,\n                    vertexColors: hasColors,\n                    sizeAttenuation: true\n                }});\n\n                if (!hasColors) {{\n                    material.color = new THREE.Color(0x00aaff);\n                }}\n\n                const pointCloud = new THREE.Points(geometry, material);\n                scene.add(pointCloud);\n                currentPointCloud = pointCloud;\n\n                geometry.computeBoundingBox();\n                const bbox = geometry.boundingBox;\n                const center = new THREE.Vector3();\n                bbox.getCenter(center);\n                const size = bbox.getSize(new THREE.Vector3());\n                const maxDim = Math.max(size.x, size.y, size.z);\n\n                const fov = camera.fov * (Math.PI / 180);\n                let cameraDistance = Math.abs(maxDim / Math.tan(fov / 2)) * 1.5;\n                \n                camera.position.set(cameraDistance, cameraDistance * 0.7, cameraDistance);\n                controls.target.copy(center);\n                controls.update();\n\n                const pointCount = geometry.attributes.position.count;\n                document.getElementById('info').innerHTML = \n                    `<strong>Point Cloud</strong><br>\n                     Points: ${{pointCount.toLocaleString()}}<br>\n                     Size: (${{size.x.toFixed(2)}}, ${{size.y.toFixed(2)}}, ${{size.z.toFixed(2)}})`;\n            }}\n\n            function resetView() {{\n                if (currentPointCloud) {{\n                    currentPointCloud.geometry.computeBoundingBox();\n                    const bbox = currentPointCloud.geometry.boundingBox;\n                    const center = new THREE.Vector3();\n                    bbox.getCenter(center);\n                    const size = bbox.getSize(new THREE.Vector3());\n                    const maxDim = Math.max(size.x, size.y, size.z);\n                    \n                    const fov = camera.fov * (Math.PI / 180);\n                    let cameraDistance = Math.abs(maxDim / Math.tan(fov / 2)) * 1.5;\n                    \n                    camera.position.set(cameraDistance, cameraDistance * 0.7, cameraDistance);\n                    controls.target.copy(center);\n                    controls.update();\n                }}\n            }}\n\n            function animate() {{\n                requestAnimationFrame(animate);\n                controls.update();\n                renderer.render(scene, camera);\n            }}\n\n            init();\n        </script>\n    </body>\n    </html>\n    \"\"\"\n    \n    display(HTML(html_content))\n\n","metadata":{"execution":{"iopub.execute_input":"2026-02-01T01:20:54.00497Z","iopub.status.busy":"2026-02-01T01:20:54.004537Z","iopub.status.idle":"2026-02-01T01:20:54.018719Z","shell.execute_reply":"2026-02-01T01:20:54.017594Z"},"papermill":{"duration":0.421856,"end_time":"2026-02-01T01:20:54.021114","exception":false,"start_time":"2026-02-01T01:20:53.599258","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# you can change position and zoom ratio.\ndisplay_ply_viewer('/kaggle/working/output/colmap/point_cloud.ply')","metadata":{"execution":{"iopub.execute_input":"2026-02-01T01:20:54.462194Z","iopub.status.busy":"2026-02-01T01:20:54.459959Z","iopub.status.idle":"2026-02-01T01:20:54.700463Z","shell.execute_reply":"2026-02-01T01:20:54.69717Z"},"papermill":{"duration":0.685216,"end_time":"2026-02-01T01:20:54.932014","exception":false,"start_time":"2026-02-01T01:20:54.246798","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.366776,"end_time":"2026-02-01T01:20:55.676721","exception":false,"start_time":"2026-02-01T01:20:55.309945","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}