{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":16880},{"sourceType":"datasetVersion","sourceId":20014943},{"sourceType":"datasetVersion","sourceId":20085498},{"sourceType":"datasetVersion","sourceId":20093096},{"sourceType":"kernelVersion","sourceId":353512671}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":5.826741,"end_time":"2026-09-26T08:51:11.036313+00:00","environment_variables":{},"exception":true,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2026-09-26T08:51:05.209572+00:00","version":"2.7.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"02a92056912e46cf8abec2a7fa8ae29a":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"02c978f0e673480d81b149ed981244d1":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_4c05c7d86bd44756821cfaa82f348432","placeholder":"​","style":"IPY_MODEL_04ae3c75fe6447679cebf805bea9a99b","tabbable":null,"tooltip":null,"value":"Trích đặc trưng test/Real: "}},"049147b859924c94ba53aaa6ced8bad5":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"04ae3c75fe6447679cebf805bea9a99b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"09100fe72ac944719c32e95b57028620":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"095f1971f2e148708e512bea1f9ad907":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"0aad6a9a36744117b2a5dc622032053e":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_204875d58ed84740a1ff615714046908","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_a125f1a98a694494b177e3b8d266ae2b","tabbable":null,"tooltip":null,"value":0}},"0ec35a89741840e4ae55eb070cb20a0e":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"101ca5d7008242a0bf45525161314620":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"20px"}},"137a4ce2aa664500ab1df55366c01e11":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"20px"}},"15a776ba1e9640bb9935d3c76d9693ad":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"18fc117a38424a909998f5c848c47c90":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"1ce85f91f05e4d4487d5ee6b3e531908":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"20px"}},"1d7757d159ed45abb5573836b068c27f":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_96595e9f76eb4db1a55db40e9a51e040","placeholder":"​","style":"IPY_MODEL_18fc117a38424a909998f5c848c47c90","tabbable":null,"tooltip":null,"value":"Làm sạch Real: "}},"1f2f7a06af2b4808a50946cec5376750":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_137a4ce2aa664500ab1df55366c01e11","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_e3b868f8b17a412d94dbcebaa80c5f2e","tabbable":null,"tooltip":null,"value":0}},"204875d58ed84740a1ff615714046908":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"20px"}},"21b743fcfa6847d7b50e0ffccc7401e7":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"20px"}},"2b3b971600a449e7840283ea3137c6f4":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"2e39ad88003749e8b522a8f42a489dd7":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_02c978f0e673480d81b149ed981244d1","IPY_MODEL_1f2f7a06af2b4808a50946cec5376750","IPY_MODEL_4766ff5fc9d244b4b3439137a0f77e22"],"layout":"IPY_MODEL_7e74659307d44ab2858ee36969a487c8","tabbable":null,"tooltip":null}},"2f4e0d8d6cb54e26a1bca0566bdd3893":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"320e955c0663455487abc75f455c3333":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"3637fa08b6fd48688c0033334e4839cc":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_5245eb269e7449ec923d9ff96fc5c39b","IPY_MODEL_bba4ecd6b34a4046bf7903264eb16690","IPY_MODEL_967ad87c830d47968196be4d307c6b03"],"layout":"IPY_MODEL_15a776ba1e9640bb9935d3c76d9693ad","tabbable":null,"tooltip":null}},"3d93e5d69a10459f9c79fa060c47a6e6":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"20px"}},"3fa94fa49fa24762bccacc2dfa49f2d2":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"425e7a36c0d646fd98fbcf284ff7aaf4":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"4766ff5fc9d244b4b3439137a0f77e22":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_09100fe72ac944719c32e95b57028620","placeholder":"​","style":"IPY_MODEL_dc4065cb550f470aa8b767e3a72f82da","tabbable":null,"tooltip":null,"value":" 0/0 [00:00&lt;?, ?it/s]"}},"4c05c7d86bd44756821cfaa82f348432":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"5245eb269e7449ec923d9ff96fc5c39b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_92373a0be8de42a6924c373897285959","placeholder":"​","style":"IPY_MODEL_948174e4335b48998fd12f03dc1c2a21","tabbable":null,"tooltip":null,"value":"Trích đặc trưng test/Fake: "}},"6132d911f795444c865aeda6168eb7eb":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_101ca5d7008242a0bf45525161314620","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_320e955c0663455487abc75f455c3333","tabbable":null,"tooltip":null,"value":0}},"6b03e2a983db4f52a4c34c565c2fb62a":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_9e63ec2fc4b940a08ffbe072ab1164d5","IPY_MODEL_6132d911f795444c865aeda6168eb7eb","IPY_MODEL_cfc2feaa8dce45bc82d8586c43133d63"],"layout":"IPY_MODEL_713d86a4518a4b948aafd360f60ce6e8","tabbable":null,"tooltip":null}},"713d86a4518a4b948aafd360f60ce6e8":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"74d91f7da935467bab864940d74b6ead":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"764638d599834e6b8e2a02ec1e956076":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_f8ac50cdc83d4ac8bf7e5c5043dd45ce","placeholder":"​","style":"IPY_MODEL_425e7a36c0d646fd98fbcf284ff7aaf4","tabbable":null,"tooltip":null,"value":"Trích đặc trưng train/Fake: "}},"7e74659307d44ab2858ee36969a487c8":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8fc53900e8544296a07243ce2f2b7959":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"92373a0be8de42a6924c373897285959":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"948174e4335b48998fd12f03dc1c2a21":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"96595e9f76eb4db1a55db40e9a51e040":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"967ad87c830d47968196be4d307c6b03":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_2f4e0d8d6cb54e26a1bca0566bdd3893","placeholder":"​","style":"IPY_MODEL_b1ad0f50ff9f43ca8b12ee189c4ebe34","tabbable":null,"tooltip":null,"value":" 0/0 [00:00&lt;?, ?it/s]"}},"99b46a0c0d114430b81799ee0d269da6":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"9e63ec2fc4b940a08ffbe072ab1164d5":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_9f982b20e209443f950de35967e73b0a","placeholder":"​","style":"IPY_MODEL_cd13736ee7b6475b93c625d9ed73d1f1","tabbable":null,"tooltip":null,"value":"Làm sạch Fake: "}},"9f782b27fa8f43828f1102d4ecc38532":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_21b743fcfa6847d7b50e0ffccc7401e7","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_a1d16a6d2af840a2bdf89da9c6a2811d","tabbable":null,"tooltip":null,"value":0}},"9f982b20e209443f950de35967e73b0a":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"a125f1a98a694494b177e3b8d266ae2b":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"a1d16a6d2af840a2bdf89da9c6a2811d":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"ac0bc04c604847d9aff932c0c3feb9bf":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_1ce85f91f05e4d4487d5ee6b3e531908","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_f6666fc371f74b888c726c79c33eaa97","tabbable":null,"tooltip":null,"value":0}},"ae05af7c122c4fc8812e4405ca2d8e1f":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"b01301fddf494c88b6da931d74cd6f66":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_764638d599834e6b8e2a02ec1e956076","IPY_MODEL_0aad6a9a36744117b2a5dc622032053e","IPY_MODEL_fdbf52c22079426697993ad7cdd5daa2"],"layout":"IPY_MODEL_3fa94fa49fa24762bccacc2dfa49f2d2","tabbable":null,"tooltip":null}},"b1ad0f50ff9f43ca8b12ee189c4ebe34":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"b301f60f1978417995e959efce91115e":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"ba41a097abab4cb2ad3b8595e7cc63b8":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"bba4ecd6b34a4046bf7903264eb16690":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"ProgressView","bar_style":"success","description":"","description_allow_html":false,"layout":"IPY_MODEL_3d93e5d69a10459f9c79fa060c47a6e6","max":1,"min":0,"orientation":"horizontal","style":"IPY_MODEL_095f1971f2e148708e512bea1f9ad907","tabbable":null,"tooltip":null,"value":0}},"bbd00fb7bfbd4d6089d34e44fde85f1c":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_cc9b03e5182a41dcabed71133b7d399e","IPY_MODEL_ac0bc04c604847d9aff932c0c3feb9bf","IPY_MODEL_f37b3d107a13447590b402492d0747fd"],"layout":"IPY_MODEL_02a92056912e46cf8abec2a7fa8ae29a","tabbable":null,"tooltip":null}},"c0ae946f3eca498887a5e13f46117e42":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_1d7757d159ed45abb5573836b068c27f","IPY_MODEL_9f782b27fa8f43828f1102d4ecc38532","IPY_MODEL_e40bd691dc3e42ea8aadf3a0ea9457a1"],"layout":"IPY_MODEL_2b3b971600a449e7840283ea3137c6f4","tabbable":null,"tooltip":null}},"cc9b03e5182a41dcabed71133b7d399e":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_ba41a097abab4cb2ad3b8595e7cc63b8","placeholder":"​","style":"IPY_MODEL_0ec35a89741840e4ae55eb070cb20a0e","tabbable":null,"tooltip":null,"value":"Trích đặc trưng train/Real: "}},"cd13736ee7b6475b93c625d9ed73d1f1":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"cfc2feaa8dce45bc82d8586c43133d63":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_fc9985b8bcf44c04bd145084d95c61a2","placeholder":"​","style":"IPY_MODEL_ae05af7c122c4fc8812e4405ca2d8e1f","tabbable":null,"tooltip":null,"value":" 0/0 [00:00&lt;?, ?it/s]"}},"d0cfc9f0b2ee4dd191f70d77f42e4309":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"dc4065cb550f470aa8b767e3a72f82da":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","background":null,"description_width":"","font_size":null,"text_color":null}},"e3b868f8b17a412d94dbcebaa80c5f2e":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"e40bd691dc3e42ea8aadf3a0ea9457a1":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_d0cfc9f0b2ee4dd191f70d77f42e4309","placeholder":"​","style":"IPY_MODEL_8fc53900e8544296a07243ce2f2b7959","tabbable":null,"tooltip":null,"value":" 0/0 [00:00&lt;?, ?it/s]"}},"f37b3d107a13447590b402492d0747fd":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_049147b859924c94ba53aaa6ced8bad5","placeholder":"​","style":"IPY_MODEL_b301f60f1978417995e959efce91115e","tabbable":null,"tooltip":null,"value":" 0/0 [00:00&lt;?, ?it/s]"}},"f6666fc371f74b888c726c79c33eaa97":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"f8ac50cdc83d4ac8bf7e5c5043dd45ce":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"fc9985b8bcf44c04bd145084d95c61a2":{"model_module":"@jupyter-widgets/base","model_module_version":"2.0.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"2.0.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"2.0.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border_bottom":null,"border_left":null,"border_right":null,"border_top":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"fdbf52c22079426697993ad7cdd5daa2":{"model_module":"@jupyter-widgets/controls","model_module_version":"2.0.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"2.0.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"2.0.0","_view_name":"HTMLView","description":"","description_allow_html":false,"layout":"IPY_MODEL_99b46a0c0d114430b81799ee0d269da6","placeholder":"​","style":"IPY_MODEL_74d91f7da935467bab864940d74b6ead","tabbable":null,"tooltip":null,"value":" 0/0 [00:00&lt;?, ?it/s]"}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nprint(os.listdir(\"/kaggle/input\"))\nprint(os.listdir(\"/kaggle/input/competitions/deepfake-detection-challenge\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:04:39.826686Z","iopub.execute_input":"2026-09-28T12:04:39.827044Z","iopub.status.idle":"2026-09-28T12:04:39.845565Z","shell.execute_reply.started":"2026-09-28T12:04:39.826986Z","shell.execute_reply":"2026-09-28T12:04:39.844609Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nprint(\"input:\", os.listdir(\"/kaggle/input\"))\nfor root, dirs, files in os.walk(\"/kaggle/input\"):\n    depth = root.count(\"/\") - 2\n    if depth <= 4:\n        print(\"  \" * depth + root, f\"({len(files)} file)\")\n    if depth >= 4:\n        dirs.clear()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:04:43.110279Z","iopub.execute_input":"2026-09-28T12:04:43.110604Z","iopub.status.idle":"2026-09-28T12:04:44.723696Z","shell.execute_reply.started":"2026-09-28T12:04:43.110555Z","shell.execute_reply":"2026-09-28T12:04:44.722459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfor root, dirs, files in os.walk(\"/kaggle/input\"):\n    if root.endswith(\"raw\") or root.endswith(\"Real\") or root.endswith(\"Fake\"):\n        print(root, \"->\", len(os.listdir(root)), \"mục\") #Tìm mục data ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:04:50.469611Z","iopub.execute_input":"2026-09-28T12:04:50.469941Z","iopub.status.idle":"2026-09-28T12:04:51.029312Z","shell.execute_reply.started":"2026-09-28T12:04:50.46989Z","shell.execute_reply":"2026-09-28T12:04:51.028134Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Phát hiện Deepfake Ảnh — Hướng Data Mining\n\nĐồ án phát hiện ảnh deepfake bằng cách trích đặc trưng thủ công (LBP, HOG, FFT,\nColor Histogram) và so sánh nhiều thuật toán Machine Learning truyền thống\n(Logistic Regression, KNN, Naive Bayes, Decision Tree, Random Forest, SVM,\nGradient Boosting).\n\nNotebook này gộp lại toàn bộ pipeline gốc (`src/preprocess_data.py` →\n`src/feature_extraction.py` → `src/eda.py` → `src/train_models.py`) thành\nmột luồng chạy tuần tự, tiện để trình bày / báo cáo và xem kết quả trực\ntiếp (hình vẽ hiện ngay dưới cell thay vì chỉ lưu file).\n\n**Cấu trúc notebook:**\n\n0. Cài đặt & cấu hình đường dẫn\n1. Tiền xử lý ảnh thô (face crop, lọc mờ/trùng lặp, cân bằng lớp)\n2. Trích xuất đặc trưng (LBP, HOG, FFT, Color Histogram)\n3. Phân tích khám phá dữ liệu (EDA)\n4. Huấn luyện & so sánh các thuật toán ML\n5. Demo dự đoán trên 1 ảnh bất kỳ (thay cho `streamlit run src/demo_app.py`)\n\n> **Chuẩn bị dữ liệu trước khi chạy:** đặt ảnh vào `data/raw/{Real,Fake}`\n> nếu ảnh còn thô/chưa crop mặt, hoặc thẳng vào `data/processed/{Real,Fake}`\n> nếu ảnh đã được crop mặt sẵn (khi đó có thể bỏ qua Bước 1).","metadata":{"papermill":{"duration":0.004977,"end_time":"2026-09-26T08:51:07.877965+00:00","exception":false,"start_time":"2026-09-26T08:51:07.872988+00:00","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## 0. Cài đặt thư viện & cấu hình đường dẫn\n\nChạy cell bên dưới 1 lần để cài các thư viện cần thiết (bỏ qua nếu môi\ntrường đã có sẵn).","metadata":{"papermill":{"duration":0.003999,"end_time":"2026-09-26T08:51:07.886721+00:00","exception":false,"start_time":"2026-09-26T08:51:07.882722+00:00","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Bỏ dấu # ở dòng dưới nếu môi trường chưa cài các thư viện cần thiết\n#!pip install opencv-python numpy pandas scikit-learn scikit-image matplotlib joblib tqdm Pillow","metadata":{"papermill":{"duration":0.009626,"end_time":"2026-09-26T08:51:07.900266+00:00","exception":false,"start_time":"2026-09-26T08:51:07.89064+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -U scikit-learn #Phiên bản scikit-learn mới nhất","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:07:17.053187Z","iopub.execute_input":"2026-09-28T12:07:17.053519Z","iopub.status.idle":"2026-09-28T12:07:24.031462Z","shell.execute_reply.started":"2026-09-28T12:07:17.05347Z","shell.execute_reply":"2026-09-28T12:07:24.030258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\nimport shutil\nimport random\nimport warnings\nfrom dataclasses import dataclass, field\n\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport joblib\nfrom tqdm.notebook import tqdm\n\nwarnings.filterwarnings(\"ignore\")\n%matplotlib inline\n","metadata":{"execution":{"iopub.status.busy":"2026-09-28T12:08:04.812879Z","iopub.execute_input":"2026-09-28T12:08:04.813243Z","iopub.status.idle":"2026-09-28T12:08:04.822041Z","shell.execute_reply.started":"2026-09-28T12:08:04.813193Z","shell.execute_reply":"2026-09-28T12:08:04.820926Z"},"papermill":{"duration":1.701929,"end_time":"2026-09-26T08:51:09.606493+00:00","exception":false,"start_time":"2026-09-26T08:51:07.904564+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\n\nsource = \"/kaggle/input/datasets/phngtrqun/data-raw1/raw\"\ndestination = \"/kaggle/working/data/raw\"\n\nif os.path.exists(destination):\n    shutil.rmtree(destination)\n\nshutil.copytree(source, destination)\n\nprint(\"Đã copy thành công!\")\nprint(\"Path:\", destination)\nprint(\"Files:\", os.listdir(destination))","metadata":{"execution":{"iopub.status.busy":"2026-09-28T12:08:08.444569Z","iopub.execute_input":"2026-09-28T12:08:08.444915Z","iopub.status.idle":"2026-09-28T12:08:09.271714Z","shell.execute_reply.started":"2026-09-28T12:08:08.444859Z","shell.execute_reply":"2026-09-28T12:08:09.270131Z"},"papermill":{"duration":0.026356,"end_time":"2026-09-26T08:51:09.67907+00:00","exception":false,"start_time":"2026-09-26T08:51:09.652714+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===================== CẤU HÌNH ĐƯỜNG DẪN & THAM SỐ =====================\n# Chỉnh lại các đường dẫn/tham số này cho phù hợp với môi trường của bạn.\n\nRAW_DIR = \"data/raw\"                 # ảnh thô: RAW_DIR/{Real,Fake}\nPROCESSED_DIR = \"data/processed\"     # sau bước 1 sẽ có PROCESSED_DIR/{train,test}/{Real,Fake}\nOUTPUT_DIR = \"outputs\"\nEDA_DIR = os.path.join(OUTPUT_DIR, \"eda\")\nFEATURES_CSV = os.path.join(OUTPUT_DIR, \"features.csv\")\n\nIMG_SIZE = 128  # phải khớp giữa bước tiền xử lý và bước trích đặc trưng\n\nfor d in (PROCESSED_DIR, OUTPUT_DIR, EDA_DIR):\n    os.makedirs(d, exist_ok=True)\n\nprint(\"RAW_DIR       :\", os.path.abspath(RAW_DIR))\nprint(\"PROCESSED_DIR :\", os.path.abspath(PROCESSED_DIR))\nprint(\"OUTPUT_DIR    :\", os.path.abspath(OUTPUT_DIR))\n","metadata":{"execution":{"iopub.status.busy":"2026-09-28T12:08:17.814567Z","iopub.execute_input":"2026-09-28T12:08:17.814931Z","iopub.status.idle":"2026-09-28T12:08:17.824088Z","shell.execute_reply.started":"2026-09-28T12:08:17.814861Z","shell.execute_reply":"2026-09-28T12:08:17.823153Z"},"papermill":{"duration":0.011861,"end_time":"2026-09-26T08:51:09.695033+00:00","exception":false,"start_time":"2026-09-26T08:51:09.683172+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1. Tiền xử lý ảnh thô (tương đương `src/preprocess_data.py`)\n\nBước này đọc ảnh từ `data/raw/{Real,Fake}` (frame trích từ video, hoặc ảnh\ncòn dính nền/nhiều người) và thực hiện lần lượt:\n\n1. **Face detection + crop** — chỉ giữ vùng khuôn mặt (bỏ nền/nhiễu)\n2. **Lọc ảnh mờ** — dựa trên phương sai Laplacian\n3. **Lọc ảnh trùng lặp** — average hash (aHash) + Hamming distance\n4. **Chia train/test theo ảnh gốc đã lọc sạch** — làm **trước** augmentation\n5. **Cân bằng lớp (class balancing)** — augmentation (xoay nhẹ, lật, đổi độ\n   sáng) chỉ áp dụng cho tập **train** của lớp thiểu số\n\n> **Vì sao chia train/test trước rồi mới augment?** Nếu augment trước, ảnh\n> gốc và bản augment của nó (gần như giống hệt) có thể rơi vào cả train\n> lẫn test → model \"nhìn thấy\" ảnh gần giống test ngay lúc train → điểm số\n> bị thổi phồng ảo (data leakage). Thứ tự trong notebook này (split trước,\n> augment sau, augment chỉ trong train) tránh được rủi ro đó.\n\nKết quả được ghi ra `data/processed/train/{Real,Fake}` và\n`data/processed/test/{Real,Fake}`.\n\n> **Bỏ qua bước này nếu** ảnh của bạn đã được crop mặt sẵn và đã nằm trong\n> `data/processed/{Real,Fake}` (không cần cấu trúc train/test) — nhảy thẳng\n> tới **Bước 2**.","metadata":{"papermill":{"duration":0.003914,"end_time":"2026-09-26T08:51:09.703428+00:00","exception":false,"start_time":"2026-09-26T08:51:09.699514+00:00","status":"completed"},"tags":[]}},{"cell_type":"code","source":"VALID_EXTS = (\".jpg\", \".jpeg\", \".png\")\nCLASSES = (\"Real\", \"Fake\")\n\n# --------------------------------------------------------------------- #\n# 1) FACE DETECTION + CROP\n# --------------------------------------------------------------------- #\n_FACE_CASCADE = None\n\n\ndef get_face_detector() -> cv2.CascadeClassifier:\n    \"\"\"Load Haar Cascade face detector 1 lần duy nhất (cache).\"\"\"\n    global _FACE_CASCADE\n    if _FACE_CASCADE is None:\n        cascade_path = cv2.data.haarcascades + \"haarcascade_frontalface_default.xml\"\n        _FACE_CASCADE = cv2.CascadeClassifier(cascade_path)\n    return _FACE_CASCADE\n\n\ndef detect_and_crop_face(bgr_img: np.ndarray, margin: float = 0.3):\n    \"\"\"\n    Phát hiện khuôn mặt lớn nhất trong ảnh và crop kèm margin xung quanh.\n    Trả về (cropped_img, found_face: bool). Nếu không tìm thấy mặt, trả về\n    ảnh gốc (giả định đã crop sẵn) kèm found_face=False.\n    \"\"\"\n    detector = get_face_detector()\n    gray = cv2.cvtColor(bgr_img, cv2.COLOR_BGR2GRAY)\n    faces = detector.detectMultiScale(\n        gray, scaleFactor=1.1, minNeighbors=5, minSize=(40, 40)\n    )\n\n    if len(faces) == 0:\n        return bgr_img, False\n\n    x, y, w, h = max(faces, key=lambda f: f[2] * f[3])\n\n    mx, my = int(w * margin), int(h * margin)\n    h_img, w_img = bgr_img.shape[:2]\n    x0 = max(0, x - mx)\n    y0 = max(0, y - my)\n    x1 = min(w_img, x + w + mx)\n    y1 = min(h_img, y + h + my)\n\n    cropped = bgr_img[y0:y1, x0:x1]\n    if cropped.size == 0:\n        return bgr_img, False\n    return cropped, True\n","metadata":{"execution":{"iopub.status.busy":"2026-09-28T12:08:23.141101Z","iopub.execute_input":"2026-09-28T12:08:23.141456Z","iopub.status.idle":"2026-09-28T12:08:23.156283Z","shell.execute_reply.started":"2026-09-28T12:08:23.141382Z","shell.execute_reply":"2026-09-28T12:08:23.154934Z"},"papermill":{"duration":0.012858,"end_time":"2026-09-26T08:51:09.720304+00:00","exception":false,"start_time":"2026-09-26T08:51:09.707446+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --------------------------------------------------------------------- #\n# 2) BLUR / QUALITY FILTER\n# --------------------------------------------------------------------- #\n\ndef is_blurry(bgr_img: np.ndarray, threshold: float = 80.0) -> bool:\n    \"\"\"\n    Ảnh càng nét thì phương sai Laplacian càng cao. threshold nên tinh\n    chỉnh theo dataset thực tế (in thử phân phối giá trị nếu cần).\n    \"\"\"\n    gray = cv2.cvtColor(bgr_img, cv2.COLOR_BGR2GRAY)\n    variance = cv2.Laplacian(gray, cv2.CV_64F).var()\n    return variance < threshold\n\n\n# --------------------------------------------------------------------- #\n# 3) DUPLICATE DETECTION (average hash)\n# --------------------------------------------------------------------- #\n\ndef average_hash(bgr_img: np.ndarray, hash_size: int = 8) -> np.ndarray:\n    \"\"\"aHash: resize về (hash_size x hash_size), so mỗi pixel với giá trị\n    trung bình -> chuỗi bit. Ảnh gần giống nhau sẽ có hash gần giống nhau.\"\"\"\n    gray = cv2.cvtColor(bgr_img, cv2.COLOR_BGR2GRAY)\n    small = cv2.resize(gray, (hash_size, hash_size), interpolation=cv2.INTER_AREA)\n    avg = small.mean()\n    return (small > avg).flatten()\n\n\ndef hamming_distance(hash_a: np.ndarray, hash_b: np.ndarray) -> int:\n    return int(np.count_nonzero(hash_a != hash_b))\n\n\nclass DuplicateFilter:\n    \"\"\"Theo dõi hash đã thấy trong từng lớp (Real/Fake riêng biệt) ở giai\n    đoạn làm sạch, trước khi chia train/test.\"\"\"\n\n    def __init__(self, max_distance: int = 5):\n        self.max_distance = max_distance\n        self.seen: dict = {}\n\n    def is_duplicate(self, label: str, img_hash: np.ndarray) -> bool:\n        for existing in self.seen.get(label, []):\n            if hamming_distance(existing, img_hash) <= self.max_distance:\n                return True\n        return False\n\n    def add(self, label: str, img_hash: np.ndarray):\n        self.seen.setdefault(label, []).append(img_hash)\n","metadata":{"execution":{"iopub.status.busy":"2026-09-28T12:08:27.446234Z","iopub.execute_input":"2026-09-28T12:08:27.446578Z","iopub.status.idle":"2026-09-28T12:08:27.461325Z","shell.execute_reply.started":"2026-09-28T12:08:27.446528Z","shell.execute_reply":"2026-09-28T12:08:27.459934Z"},"papermill":{"duration":0.013366,"end_time":"2026-09-26T08:51:09.738098+00:00","exception":false,"start_time":"2026-09-26T08:51:09.724732+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --------------------------------------------------------------------- #\n# 4) DATA AUGMENTATION (chỉ dùng cho tập TRAIN, để cân bằng lớp)\n# --------------------------------------------------------------------- #\n\ndef augment_rotate(img: np.ndarray, max_angle: float = 10.0) -> np.ndarray:\n    angle = random.uniform(-max_angle, max_angle)\n    h, w = img.shape[:2]\n    matrix = cv2.getRotationMatrix2D((w / 2, h / 2), angle, 1.0)\n    return cv2.warpAffine(img, matrix, (w, h), borderMode=cv2.BORDER_REFLECT)\n\n\ndef augment_flip(img: np.ndarray) -> np.ndarray:\n    return cv2.flip(img, 1)\n\n\ndef augment_brightness(img: np.ndarray, delta_range: int = 30) -> np.ndarray:\n    delta = random.randint(-delta_range, delta_range)\n    return cv2.convertScaleAbs(img, alpha=1.0, beta=delta)\n\n\ndef random_augment(img: np.ndarray) -> np.ndarray:\n    \"\"\"Áp ngẫu nhiên 1-2 phép biến đổi để tạo ảnh mới đa dạng hơn.\"\"\"\n    ops = [augment_rotate, augment_flip, augment_brightness]\n    chosen = random.sample(ops, k=random.choice([1, 2]))\n    out = img.copy()\n    for op in chosen:\n        out = op(out)\n    return out\n\n\n# --------------------------------------------------------------------- #\n# STATS\n# --------------------------------------------------------------------- #\n\n@dataclass\nclass Stats:\n    scanned: int = 0\n    no_face: int = 0\n    dropped_blurry: int = 0\n    dropped_duplicate: int = 0\n    dropped_unreadable: int = 0\n    cleaned: dict = field(default_factory=lambda: {\"Real\": 0, \"Fake\": 0})\n    train_kept: dict = field(default_factory=lambda: {\"Real\": 0, \"Fake\": 0})\n    test_kept: dict = field(default_factory=lambda: {\"Real\": 0, \"Fake\": 0})\n    train_augmented: dict = field(default_factory=lambda: {\"Real\": 0, \"Fake\": 0})\n","metadata":{"execution":{"iopub.status.busy":"2026-09-28T12:08:33.041104Z","iopub.execute_input":"2026-09-28T12:08:33.041478Z","iopub.status.idle":"2026-09-28T12:08:33.056725Z","shell.execute_reply.started":"2026-09-28T12:08:33.041416Z","shell.execute_reply":"2026-09-28T12:08:33.055453Z"},"papermill":{"duration":0.013594,"end_time":"2026-09-26T08:51:09.755603+00:00","exception":false,"start_time":"2026-09-26T08:51:09.742009+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --------------------------------------------------------------------- #\n# STAGE A — làm sạch (chưa chia train/test): face crop, blur filter, dedup\n# --------------------------------------------------------------------- #\n\ndef _clean_raw_images(raw_dir, scratch_dir, stats, blur_threshold, dup_max_distance, face_margin):\n    \"\"\"Trả về dict label -> list đường dẫn ảnh đã làm sạch (trong scratch_dir).\"\"\"\n    dup_filter = DuplicateFilter(max_distance=dup_max_distance)\n    cleaned_paths = {label: [] for label in CLASSES}\n\n    for label in CLASSES:\n        src_folder = os.path.join(raw_dir, label)\n        scratch_label_dir = os.path.join(scratch_dir, label)\n        os.makedirs(scratch_label_dir, exist_ok=True)\n\n        img_paths = [\n            p for p in glob.glob(os.path.join(src_folder, \"*\"))\n            if p.lower().endswith(VALID_EXTS)\n        ]\n        print(f\"[INFO] {label}: tìm thấy {len(img_paths)} ảnh thô trong {src_folder}\")\n\n        for idx, p in enumerate(tqdm(img_paths, desc=f\"Làm sạch {label}\")):\n            stats.scanned += 1\n            img = cv2.imread(p)\n            if img is None:\n                stats.dropped_unreadable += 1\n                continue\n\n            cropped, found = detect_and_crop_face(img, margin=face_margin)\n            if not found:\n                stats.no_face += 1\n\n            resized = cv2.resize(cropped, (IMG_SIZE, IMG_SIZE))\n\n            if is_blurry(resized, threshold=blur_threshold):\n                stats.dropped_blurry += 1\n                continue\n\n            img_hash = average_hash(resized)\n            if dup_filter.is_duplicate(label, img_hash):\n                stats.dropped_duplicate += 1\n                continue\n            dup_filter.add(label, img_hash)\n\n            out_path = os.path.join(scratch_label_dir, f\"{label.lower()}_{idx:06d}.jpg\")\n            cv2.imwrite(out_path, resized)\n            cleaned_paths[label].append(out_path)\n            stats.cleaned[label] += 1\n\n    return cleaned_paths\n\n\n# --------------------------------------------------------------------- #\n# STAGE B — chia train/test theo ảnh gốc ĐÃ LÀM SẠCH (chưa augment)\n# --------------------------------------------------------------------- #\n\ndef _split_and_move(cleaned_paths, processed_dir, test_size, seed, stats):\n    rng = random.Random(seed)\n    final_paths = {\"train\": {}, \"test\": {}}\n\n    for label in CLASSES:\n        paths = list(cleaned_paths[label])\n        rng.shuffle(paths)\n        n_test = int(round(len(paths) * test_size))\n        test_subset = paths[:n_test]\n        train_subset = paths[n_test:]\n\n        for split_name, subset in ((\"train\", train_subset), (\"test\", test_subset)):\n            dst_dir = os.path.join(processed_dir, split_name, label)\n            os.makedirs(dst_dir, exist_ok=True)\n            moved = []\n            for i, src in enumerate(subset):\n                dst = os.path.join(dst_dir, f\"{label.lower()}_{i:06d}.jpg\")\n                shutil.move(src, dst)\n                moved.append(dst)\n            final_paths[split_name][label] = moved\n\n        stats.train_kept[label] = len(train_subset)\n        stats.test_kept[label] = len(test_subset)\n\n    return final_paths\n\n\n# --------------------------------------------------------------------- #\n# STAGE C — cân bằng lớp bằng augmentation, CHỈ trong tập train\n# --------------------------------------------------------------------- #\n\ndef _balance_train_classes(train_paths, processed_dir, stats, max_aug_multiplier):\n    n_real = len(train_paths[\"Real\"])\n    n_fake = len(train_paths[\"Fake\"])\n    if n_real == 0 or n_fake == 0:\n        print(\"[WARN] Một trong hai lớp không có ảnh train nào sau lọc -> bỏ qua cân bằng lớp.\")\n        return\n\n    majority_label, minority_label = (\"Real\", \"Fake\") if n_real >= n_fake else (\"Fake\", \"Real\")\n    n_majority = max(n_real, n_fake)\n    n_minority = min(n_real, n_fake)\n    target = min(n_majority, int(n_minority * max_aug_multiplier))\n    n_to_generate = max(0, target - n_minority)\n\n    if n_to_generate == 0:\n        print(f\"[INFO] Tập train đã tương đối cân bằng (Real={n_real}, Fake={n_fake}), bỏ qua augmentation.\")\n        return\n\n    print(\n        f\"[INFO] Tập TRAIN mất cân bằng: {majority_label}={n_majority}, {minority_label}={n_minority}. \"\n        f\"Sinh thêm {n_to_generate} ảnh augment cho lớp '{minority_label}' (chỉ trong train, \"\n        f\"giới hạn tối đa {max_aug_multiplier}x số ảnh gốc để tránh trùng lặp quá nhiều).\"\n    )\n\n    minority_paths = train_paths[minority_label]\n    dst_folder = os.path.join(processed_dir, \"train\", minority_label)\n\n    for i in tqdm(range(n_to_generate), desc=f\"Augment train/{minority_label}\"):\n        src_path = random.choice(minority_paths)\n        img = cv2.imread(src_path)\n        if img is None:\n            continue\n        aug_img = random_augment(img)\n        out_name = f\"{minority_label.lower()}_aug_{i:06d}.jpg\"\n        out_path = os.path.join(dst_folder, out_name)\n        cv2.imwrite(out_path, aug_img)\n        stats.train_augmented[minority_label] += 1\n","metadata":{"execution":{"iopub.status.busy":"2026-09-28T12:08:35.789578Z","iopub.execute_input":"2026-09-28T12:08:35.789897Z","iopub.status.idle":"2026-09-28T12:08:35.816581Z","shell.execute_reply.started":"2026-09-28T12:08:35.789844Z","shell.execute_reply":"2026-09-28T12:08:35.815389Z"},"papermill":{"duration":0.018044,"end_time":"2026-09-26T08:51:09.778365+00:00","exception":false,"start_time":"2026-09-26T08:51:09.760321+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --------------------------------------------------------------------- #\n# MAIN PIPELINE (tiền xử lý)\n# --------------------------------------------------------------------- #\n\ndef process_dataset(\n    raw_dir, processed_dir,\n    blur_threshold=80.0, dup_max_distance=5, face_margin=0.3,\n    balance=True, max_aug_multiplier=3.0, test_size=0.2, seed=42,\n) -> Stats:\n    stats = Stats()\n    scratch_dir = os.path.join(processed_dir, \"_scratch_clean\")\n\n    cleaned_paths = _clean_raw_images(\n        raw_dir, scratch_dir, stats, blur_threshold, dup_max_distance, face_margin\n    )\n    final_paths = _split_and_move(cleaned_paths, processed_dir, test_size, seed, stats)\n    shutil.rmtree(scratch_dir, ignore_errors=True)\n\n    if balance:\n        _balance_train_classes(final_paths[\"train\"], processed_dir, stats, max_aug_multiplier)\n\n    _print_report(stats, test_size)\n    return stats\n\n\ndef _print_report(stats: Stats, test_size: float):\n    print(\"\\n===== BÁO CÁO TIỀN XỬ LÝ DỮ LIỆU =====\")\n    print(f\"Tổng số ảnh thô đã quét      : {stats.scanned}\")\n    print(f\"Không đọc được (lỗi file)     : {stats.dropped_unreadable}\")\n    print(f\"Không phát hiện được mặt      : {stats.no_face} (vẫn giữ, dùng ảnh gốc)\")\n    print(f\"Loại vì quá mờ                : {stats.dropped_blurry}\")\n    print(f\"Loại vì trùng lặp             : {stats.dropped_duplicate}\")\n    print(f\"Sạch sau lọc (trước khi chia) -> Real: {stats.cleaned['Real']:5d} | Fake: {stats.cleaned['Fake']:5d}\")\n    print(f\"\\nChia train/test (test_size={test_size}, split theo ẢNH GỐC trước augment):\")\n    print(f\"  TRAIN (trước augment) -> Real: {stats.train_kept['Real']:5d} | Fake: {stats.train_kept['Fake']:5d}\")\n    print(f\"  TEST  (giữ nguyên, KHÔNG augment) -> Real: {stats.test_kept['Real']:5d} | Fake: {stats.test_kept['Fake']:5d}\")\n    print(f\"  Augment thêm vào TRAIN -> Real: {stats.train_augmented['Real']:5d} | Fake: {stats.train_augmented['Fake']:5d}\")\n    train_real_final = stats.train_kept[\"Real\"] + stats.train_augmented[\"Real\"]\n    train_fake_final = stats.train_kept[\"Fake\"] + stats.train_augmented[\"Fake\"]\n    print(f\"  TRAIN TỔNG cuối cùng -> Real: {train_real_final:5d} | Fake: {train_fake_final:5d}\")\n    print(\n        \"\\n(Test set không augment và không cân bằng lớp — giữ đúng phân bố tự nhiên \"\n        \"để đánh giá model khách quan, tránh data leakage giữa train/test.)\"\n    )\n","metadata":{"execution":{"iopub.status.busy":"2026-09-28T12:08:40.533679Z","iopub.execute_input":"2026-09-28T12:08:40.53405Z","iopub.status.idle":"2026-09-28T12:08:40.547005Z","shell.execute_reply.started":"2026-09-28T12:08:40.533995Z","shell.execute_reply":"2026-09-28T12:08:40.545761Z"},"papermill":{"duration":0.013331,"end_time":"2026-09-26T08:51:09.795847+00:00","exception":false,"start_time":"2026-09-26T08:51:09.782516+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Chạy tiền xử lý\n\nChỉnh các tham số bên dưới nếu cần rồi chạy cell (bỏ qua cell này nếu dữ\nliệu của bạn đã nằm sẵn trong `data/processed/{Real,Fake}` dạng đã crop\nmặt).","metadata":{"papermill":{"duration":0.00385,"end_time":"2026-09-26T08:51:09.804067+00:00","exception":false,"start_time":"2026-09-26T08:51:09.800217+00:00","status":"completed"},"tags":[]}},{"cell_type":"code","source":"RUN_PREPROCESS = True  # đặt False nếu ảnh đã crop mặt sẵn trong data/processed/{Real,Fake}\n\nif RUN_PREPROCESS:\n    stats = process_dataset(\n        raw_dir=RAW_DIR,\n        processed_dir=PROCESSED_DIR,\n        blur_threshold=80.0,\n        dup_max_distance=5,\n        face_margin=0.3,\n        balance=True,\n        max_aug_multiplier=3.0,\n        test_size=0.2,\n        seed=42,\n    )\nelse:\n    print(\"[SKIP] Bỏ qua bước tiền xử lý ảnh thô.\")\n","metadata":{"execution":{"iopub.status.busy":"2026-09-28T12:08:43.856445Z","iopub.execute_input":"2026-09-28T12:08:43.856793Z","iopub.status.idle":"2026-09-28T12:08:50.768671Z","shell.execute_reply.started":"2026-09-28T12:08:43.856735Z","shell.execute_reply":"2026-09-28T12:08:50.767276Z"},"papermill":{"duration":0.034966,"end_time":"2026-09-26T08:51:09.842979+00:00","exception":false,"start_time":"2026-09-26T08:51:09.808013+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Trích xuất đặc trưng (tương đương `src/feature_extraction.py`)\n\nTrích xuất đặc trưng thủ công (hand-crafted features) từ ảnh khuôn mặt:\n\n1. **LBP** (Local Binary Pattern) — bắt texture bất thường trên da / vùng blend\n2. **HOG** (Histogram of Oriented Gradients) — bắt cấu trúc cạnh khuôn mặt\n3. **FFT** (Fast Fourier Transform) — bắt artifact ở miền tần số cao (dấu vết GAN)\n4. **Color Histogram** (HSV) — bắt lệch màu da do ghép mặt\n\nQuét thư mục `data/processed/{Real,Fake}` (hoặc `data/processed/{train,test}/{Real,Fake}`\nnếu đã chạy Bước 1) và xuất ra `outputs/features.csv`.","metadata":{"papermill":{"duration":0.004804,"end_time":"2026-09-26T08:51:09.852772+00:00","exception":false,"start_time":"2026-09-26T08:51:09.847968+00:00","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from skimage.feature import local_binary_pattern, hog\n\nLBP_RADIUS = 2\nLBP_POINTS = 8 * LBP_RADIUS\n\n\ndef extract_lbp(gray_img: np.ndarray) -> np.ndarray:\n    \"\"\"Trả về histogram LBP đã chuẩn hóa (uniform pattern).\"\"\"\n    lbp = local_binary_pattern(gray_img, LBP_POINTS, LBP_RADIUS, method=\"uniform\")\n    n_bins = LBP_POINTS + 2\n    hist, _ = np.histogram(lbp.ravel(), bins=n_bins, range=(0, n_bins))\n    hist = hist.astype(\"float32\")\n    hist /= (hist.sum() + 1e-7)\n    return hist\n\n\ndef extract_hog(gray_img: np.ndarray) -> np.ndarray:\n    \"\"\"Trả về vector đặc trưng HOG.\"\"\"\n    features = hog(\n        gray_img, orientations=9, pixels_per_cell=(16, 16),\n        cells_per_block=(2, 2), block_norm=\"L2-Hys\", feature_vector=True,\n    )\n    return features.astype(\"float32\")\n\n\ndef extract_fft_features(gray_img: np.ndarray) -> np.ndarray:\n    \"\"\"\n    Trích đặc trưng miền tần số bằng FFT 2D. Deepfake (đặc biệt ảnh GAN)\n    thường để lại năng lượng bất thường ở dải tần số cao. Chia phổ tần số\n    thành các dải vòng (radial bins) và lấy năng lượng trung bình mỗi dải.\n    \"\"\"\n    f = np.fft.fft2(gray_img)\n    fshift = np.fft.fftshift(f)\n    magnitude = np.log1p(np.abs(fshift))\n\n    h, w = magnitude.shape\n    cy, cx = h // 2, w // 2\n    y, x = np.indices((h, w))\n    r = np.sqrt((x - cx) ** 2 + (y - cy) ** 2).astype(int)\n\n    n_bins = 16\n    max_r = r.max()\n    bin_edges = np.linspace(0, max_r, n_bins + 1)\n    radial_features = []\n    for i in range(n_bins):\n        mask = (r >= bin_edges[i]) & (r < bin_edges[i + 1])\n        if mask.sum() > 0:\n            radial_features.append(magnitude[mask].mean())\n        else:\n            radial_features.append(0.0)\n\n    radial_features = np.array(radial_features, dtype=\"float32\")\n    low = radial_features[: n_bins // 3].sum() + 1e-7\n    high = radial_features[2 * n_bins // 3 :].sum()\n    ratio = np.array([high / low], dtype=\"float32\")\n\n    return np.concatenate([radial_features, ratio])\n\n\ndef extract_color_histogram(bgr_img: np.ndarray) -> np.ndarray:\n    \"\"\"Histogram màu trong không gian HSV (bắt lệch tông da khi ghép mặt).\"\"\"\n    hsv = cv2.cvtColor(bgr_img, cv2.COLOR_BGR2HSV)\n    hist = cv2.calcHist([hsv], [0, 1, 2], None, [8, 8, 8], [0, 180, 0, 256, 0, 256])\n    hist = cv2.normalize(hist, hist).flatten()\n    return hist.astype(\"float32\")\n\n\ndef extract_features(bgr_img: np.ndarray) -> np.ndarray:\n    \"\"\"Ghép toàn bộ đặc trưng thành 1 vector duy nhất cho 1 ảnh.\"\"\"\n    bgr_img = cv2.resize(bgr_img, (IMG_SIZE, IMG_SIZE))\n    gray = cv2.cvtColor(bgr_img, cv2.COLOR_BGR2GRAY)\n\n    lbp_feat = extract_lbp(gray)\n    hog_feat = extract_hog(gray)\n    fft_feat = extract_fft_features(gray)\n    color_feat = extract_color_histogram(bgr_img)\n\n    return np.concatenate([lbp_feat, hog_feat, fft_feat, color_feat])\n","metadata":{"execution":{"iopub.status.busy":"2026-09-28T12:08:56.768076Z","iopub.execute_input":"2026-09-28T12:08:56.768435Z","iopub.status.idle":"2026-09-28T12:08:56.789577Z","shell.execute_reply.started":"2026-09-28T12:08:56.76837Z","shell.execute_reply":"2026-09-28T12:08:56.788145Z"},"papermill":{"duration":0.562492,"end_time":"2026-09-26T08:51:10.420051+00:00","exception":false,"start_time":"2026-09-26T08:51:09.857559+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_dataset(data_dir: str, output_csv: str) -> pd.DataFrame:\n    \"\"\"\n    Trích đặc trưng cho toàn bộ ảnh và lưu ra 1 file CSV: cột 'label'\n    (0=Fake, 1=Real) + các cột đặc trưng.\n\n    Tự nhận diện 2 kiểu cấu trúc thư mục:\n      a) data_dir/{train,test}/{Real,Fake} (do Bước 1 tạo ra)\n         -> thêm cột \"split\" (train/test) vào CSV\n      b) data_dir/{Real,Fake} (không có train/test)\n         -> không có cột \"split\", Bước 4 sẽ tự chia ngẫu nhiên\n    \"\"\"\n    train_dir = os.path.join(data_dir, \"train\")\n    test_dir = os.path.join(data_dir, \"test\")\n    use_split_mode = os.path.isdir(train_dir) and os.path.isdir(test_dir)\n\n    if use_split_mode:\n        scan_targets = [(\"train\", train_dir), (\"test\", test_dir)]\n        print(\"[INFO] Phát hiện cấu trúc train/test -> sẽ gắn cột 'split' vào features.csv.\")\n    else:\n        scan_targets = [(None, data_dir)]\n\n    rows, labels, splits = [], [], []\n\n    for split_name, base_dir in scan_targets:\n        for label_name, label_val in [(\"Real\", 1), (\"Fake\", 0)]:\n            folder = os.path.join(base_dir, label_name)\n            img_paths = (\n                glob.glob(os.path.join(folder, \"*.jpg\"))\n                + glob.glob(os.path.join(folder, \"*.jpeg\"))\n                + glob.glob(os.path.join(folder, \"*.png\"))\n            )\n            tag = f\"{split_name}/{label_name}\" if split_name else label_name\n            print(f\"[INFO] {tag}: tìm thấy {len(img_paths)} ảnh\")\n\n            for p in tqdm(img_paths, desc=f\"Trích đặc trưng {tag}\"):\n                img = cv2.imread(p)\n                if img is None:\n                    continue\n                feat = extract_features(img)\n                rows.append(feat)\n                labels.append(label_val)\n                if use_split_mode:\n                    splits.append(split_name)\n\n    if len(rows) == 0:\n        raise RuntimeError(\n            f\"Không tìm thấy ảnh nào trong {data_dir}. Kiểm tra lại cấu trúc thư mục \"\n            \"(data/processed/Real, data/processed/Fake hoặc \"\n            \"data/processed/train|test/Real, .../Fake nếu đã chạy Bước 1).\"\n        )\n\n    X = np.vstack(rows)\n    col_names = [f\"f_{i}\" for i in range(X.shape[1])]\n    df = pd.DataFrame(X, columns=col_names)\n    df[\"label\"] = labels\n    if use_split_mode:\n        df[\"split\"] = splits\n\n    os.makedirs(os.path.dirname(output_csv), exist_ok=True)\n    df.to_csv(output_csv, index=False)\n    print(f\"[OK] Đã lưu {len(df)} mẫu, {X.shape[1]} đặc trưng -> {output_csv}\")\n    return df\n","metadata":{"execution":{"iopub.status.busy":"2026-09-28T12:09:13.663325Z","iopub.execute_input":"2026-09-28T12:09:13.66367Z","iopub.status.idle":"2026-09-28T12:09:13.679481Z","shell.execute_reply.started":"2026-09-28T12:09:13.663611Z","shell.execute_reply":"2026-09-28T12:09:13.678511Z"},"papermill":{"duration":0.015996,"end_time":"2026-09-26T08:51:10.443718+00:00","exception":false,"start_time":"2026-09-26T08:51:10.427722+00:00","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Chạy trích xuất đặc trưng","metadata":{"papermill":{"duration":0.005114,"end_time":"2026-09-26T08:51:10.453454+00:00","exception":false,"start_time":"2026-09-26T08:51:10.44834+00:00","status":"completed"},"tags":[]}},{"cell_type":"code","source":"features_df = build_dataset(PROCESSED_DIR, FEATURES_CSV)\nfeatures_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2026-09-28T12:09:39.196715Z","iopub.execute_input":"2026-09-28T12:09:39.197069Z","iopub.status.idle":"2026-09-28T12:09:44.869479Z","shell.execute_reply.started":"2026-09-28T12:09:39.197009Z","shell.execute_reply":"2026-09-28T12:09:44.868551Z"},"papermill":{"duration":0.054754,"end_time":"2026-09-26T08:51:10.512778+00:00","exception":true,"start_time":"2026-09-26T08:51:10.458024+00:00","status":"failed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Phân tích khám phá dữ liệu — EDA (tương đương `src/eda.py`)\n\nTrực quan hoá tập đặc trưng vừa trích xuất: phân bố lớp, khả năng phân\ntách của dữ liệu (PCA/t-SNE), và tương quan giữa các đặc trưng. Các biểu\nđồ vừa hiện trực tiếp trong notebook, vừa được lưu vào `outputs/eda/`.","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]}},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nfrom sklearn.manifold import TSNE\nfrom sklearn.preprocessing import StandardScaler\n\n\ndef plot_class_distribution(df: pd.DataFrame, out_dir: str):\n    counts = df[\"label\"].value_counts().sort_index()\n    labels = [\"Fake (0)\", \"Real (1)\"]\n    plt.figure(figsize=(5, 4))\n    plt.bar(labels, [counts.get(0, 0), counts.get(1, 0)], color=[\"#e74c3c\", \"#2ecc71\"])\n    plt.title(\"Phân bố số lượng mẫu Real / Fake\")\n    plt.ylabel(\"Số lượng ảnh\")\n    for i, v in enumerate([counts.get(0, 0), counts.get(1, 0)]):\n        plt.text(i, v, str(v), ha=\"center\", va=\"bottom\")\n    plt.tight_layout()\n    plt.savefig(os.path.join(out_dir, \"class_distribution.png\"), dpi=150)\n    plt.show()\n\n\ndef plot_feature_projection(X: np.ndarray, y: np.ndarray, out_dir: str, method: str = \"pca\"):\n    \"\"\"Chiếu vector đặc trưng nhiều chiều xuống 2D để quan sát khả năng phân tách.\"\"\"\n    scaler = StandardScaler()\n    X_scaled = scaler.fit_transform(X)\n\n    if method == \"pca\":\n        reducer = PCA(n_components=2, random_state=42)\n        title = \"Chiếu đặc trưng xuống 2D bằng PCA\"\n        fname = \"pca_projection.png\"\n    else:\n        reducer = TSNE(n_components=2, random_state=42, init=\"pca\", perplexity=30)\n        title = \"Chiếu đặc trưng xuống 2D bằng t-SNE\"\n        fname = \"tsne_projection.png\"\n\n    X_2d = reducer.fit_transform(X_scaled)\n\n    plt.figure(figsize=(6, 5))\n    for label_val, label_name, color in [(0, \"Fake\", \"#e74c3c\"), (1, \"Real\", \"#2ecc71\")]:\n        mask = y == label_val\n        plt.scatter(X_2d[mask, 0], X_2d[mask, 1], s=10, alpha=0.6, label=label_name, color=color)\n    plt.legend()\n    plt.title(title)\n    plt.tight_layout()\n    plt.savefig(os.path.join(out_dir, fname), dpi=150)\n    plt.show()\n\n\ndef plot_feature_correlation_sample(df: pd.DataFrame, out_dir: str, n_features: int = 30):\n    \"\"\"Ma trận tương quan của N đặc trưng đầu tiên (không lấy hết vì quá nhiều cột).\"\"\"\n    exclude = {\"label\", \"split\"}\n    feat_cols = [c for c in df.columns if c not in exclude][:n_features]\n    corr = df[feat_cols].corr()\n    plt.figure(figsize=(8, 7))\n    plt.imshow(corr, cmap=\"coolwarm\", vmin=-1, vmax=1)\n    plt.colorbar(label=\"Hệ số tương quan\")\n    plt.title(f\"Tương quan giữa {n_features} đặc trưng đầu tiên\")\n    plt.tight_layout()\n    plt.savefig(os.path.join(out_dir, \"feature_correlation.png\"), dpi=150)\n    plt.show()\n\n\ndef run_eda(df: pd.DataFrame, out_dir: str):\n    os.makedirs(out_dir, exist_ok=True)\n    y = df[\"label\"].values\n    X = df.drop(columns=[c for c in (\"label\", \"split\") if c in df.columns]).values\n\n    print(f\"[INFO] Tổng số mẫu: {len(df)} | Số đặc trưng: {X.shape[1]}\")\n\n    plot_class_distribution(df, out_dir)\n    plot_feature_projection(X, y, out_dir, method=\"pca\")\n    plot_feature_projection(X, y, out_dir, method=\"tsne\")\n    plot_feature_correlation_sample(df, out_dir)\n\n    print(f\"[OK] Đã lưu các biểu đồ EDA vào: {out_dir}\")\n","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:09:53.341237Z","iopub.execute_input":"2026-09-28T12:09:53.341596Z","iopub.status.idle":"2026-09-28T12:09:53.367363Z","shell.execute_reply.started":"2026-09-28T12:09:53.341544Z","shell.execute_reply":"2026-09-28T12:09:53.366302Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Chạy EDA","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]}},{"cell_type":"code","source":"run_eda(features_df, EDA_DIR)\n","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:10:01.348932Z","iopub.execute_input":"2026-09-28T12:10:01.349268Z","iopub.status.idle":"2026-09-28T12:10:05.012724Z","shell.execute_reply.started":"2026-09-28T12:10:01.349218Z","shell.execute_reply":"2026-09-28T12:10:05.011741Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Huấn luyện & so sánh các thuật toán ML (tương đương `src/train_models.py`)\n\nHuấn luyện 7 thuật toán ML truyền thống trên tập đặc trưng:\n\n- Logistic Regression\n- K-Nearest Neighbors (KNN)\n- Naive Bayes\n- Decision Tree\n- Random Forest\n- SVM (RBF kernel)\n- Gradient Boosting\n\nSau khi train xong: in bảng so sánh Accuracy/Precision/Recall/F1/AUC, vẽ\nconfusion matrix từng model, vẽ biểu đồ so sánh, và lưu model tốt nhất\n(theo F1-score) ra `outputs/best_model.pkl`.","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import (\n    accuracy_score, precision_score, recall_score, f1_score,\n    roc_auc_score, confusion_matrix, classification_report,\n)\n\n\ndef get_models() -> dict:\n    # class_weight=\"balanced\" giúp model không thiên vị lớp đa số, đóng vai\n    # trò lớp bảo hiểm thứ 2 bên cạnh bước cân bằng lớp bằng augmentation ở\n    # Bước 1 (phòng khi dữ liệu vẫn còn lệch nhẹ sau bước đó). KNN,\n    # NaiveBayes, GradientBoosting (bản sklearn) không hỗ trợ tham số này.\n    return {\n        \"LogisticRegression\": LogisticRegression(max_iter=1000, class_weight=\"balanced\"),\n        \"KNN\": KNeighborsClassifier(n_neighbors=7),\n        \"NaiveBayes\": GaussianNB(),\n        \"DecisionTree\": DecisionTreeClassifier(max_depth=12, random_state=42, class_weight=\"balanced\"),\n        \"RandomForest\": RandomForestClassifier(\n            n_estimators=300, random_state=42, n_jobs=-1, class_weight=\"balanced\"\n        ),\n        \"SVM_RBF\": SVC(kernel=\"rbf\", C=2.0, probability=True, random_state=42, class_weight=\"balanced\"),\n        \"GradientBoosting\": GradientBoostingClassifier(random_state=42),\n    }\n\n\ndef evaluate_model(model, X_test, y_test) -> dict:\n    y_pred = model.predict(X_test)\n    if hasattr(model, \"predict_proba\"):\n        y_score = model.predict_proba(X_test)[:, 1]\n    else:\n        y_score = y_pred\n\n    return {\n        \"accuracy\": accuracy_score(y_test, y_pred),\n        \"precision\": precision_score(y_test, y_pred, zero_division=0),\n        \"recall\": recall_score(y_test, y_pred, zero_division=0),\n        \"f1\": f1_score(y_test, y_pred, zero_division=0),\n        \"auc\": roc_auc_score(y_test, y_score),\n        \"y_pred\": y_pred,\n    }\n\n\ndef plot_confusion_matrix(y_test, y_pred, model_name: str, out_dir: str):\n    cm = confusion_matrix(y_test, y_pred)\n    plt.figure(figsize=(4, 4))\n    plt.imshow(cm, cmap=\"Blues\")\n    plt.title(f\"Confusion Matrix - {model_name}\")\n    plt.xticks([0, 1], [\"Fake\", \"Real\"])\n    plt.yticks([0, 1], [\"Fake\", \"Real\"])\n    plt.xlabel(\"Dự đoán\")\n    plt.ylabel(\"Thực tế\")\n    for i in range(2):\n        for j in range(2):\n            plt.text(j, i, str(cm[i, j]), ha=\"center\", va=\"center\",\n                      color=\"white\" if cm[i, j] > cm.max() / 2 else \"black\")\n    plt.tight_layout()\n    plt.savefig(os.path.join(out_dir, f\"cm_{model_name}.png\"), dpi=150)\n    plt.show()\n\n\ndef plot_comparison_chart(results_df: pd.DataFrame, out_dir: str):\n    metrics = [\"accuracy\", \"precision\", \"recall\", \"f1\", \"auc\"]\n    x = np.arange(len(results_df))\n    width = 0.15\n\n    plt.figure(figsize=(11, 6))\n    for i, m in enumerate(metrics):\n        plt.bar(x + i * width, results_df[m], width, label=m)\n\n    plt.xticks(x + width * 2, results_df[\"model\"], rotation=20)\n    plt.ylabel(\"Điểm số\")\n    plt.title(\"So sánh các thuật toán Machine Learning\")\n    plt.legend()\n    plt.tight_layout()\n    plt.savefig(os.path.join(out_dir, \"model_comparison.png\"), dpi=150)\n    plt.show()\n","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:10:14.758382Z","iopub.execute_input":"2026-09-28T12:10:14.758739Z","iopub.status.idle":"2026-09-28T12:10:14.910679Z","shell.execute_reply.started":"2026-09-28T12:10:14.758688Z","shell.execute_reply":"2026-09-28T12:10:14.909307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def run_training(features_csv: str, out_dir: str, test_size: float = 0.2):\n    os.makedirs(out_dir, exist_ok=True)\n    df = pd.read_csv(features_csv)\n\n    if \"split\" in df.columns:\n        # features.csv được sinh từ data/processed/train|test/... (Bước 1)\n        # -> dùng đúng phần chia train/test đã làm ở bước tiền xử lý (train\n        # được augment để cân bằng lớp, test giữ nguyên phân bố tự nhiên).\n        # Không chia ngẫu nhiên lại ở đây để tránh phá vỡ việc chống data leakage.\n        print(\"[INFO] Tìm thấy cột 'split' -> dùng train/test đã chia sẵn từ Bước 1 \"\n              \"(chống data leakage do augmentation).\")\n        train_df = df[df[\"split\"] == \"train\"]\n        test_df = df[df[\"split\"] == \"test\"]\n        y_train = train_df[\"label\"].values\n        X_train = train_df.drop(columns=[\"label\", \"split\"]).values\n        y_test = test_df[\"label\"].values\n        X_test = test_df.drop(columns=[\"label\", \"split\"]).values\n    else:\n        # Cấu trúc cũ (không qua Bước 1) -> tự chia ngẫu nhiên, giữ tương thích ngược.\n        print(\"[INFO] Không có cột 'split' -> tự chia train/test ngẫu nhiên \"\n              f\"(test_size={test_size}). Khuyến nghị chạy Bước 1 trước nếu dữ liệu \"\n              \"có dùng augmentation để tránh data leakage.\")\n        y = df[\"label\"].values\n        X = df.drop(columns=[\"label\"]).values\n        X_train, X_test, y_train, y_test = train_test_split(\n            X, y, test_size=test_size, stratify=y, random_state=42\n        )\n\n    scaler = StandardScaler()\n    X_train_s = scaler.fit_transform(X_train)\n    X_test_s = scaler.transform(X_test)\n\n    models = get_models()\n    results = []\n    trained_models = {}\n\n    for name, model in models.items():\n        print(f\"[TRAIN] {name} ...\")\n        model.fit(X_train_s, y_train)\n        metrics = evaluate_model(model, X_test_s, y_test)\n\n        print(classification_report(y_test, metrics[\"y_pred\"], target_names=[\"Fake\", \"Real\"]))\n        plot_confusion_matrix(y_test, metrics[\"y_pred\"], name, out_dir)\n\n        results.append({\n            \"model\": name,\n            \"accuracy\": metrics[\"accuracy\"],\n            \"precision\": metrics[\"precision\"],\n            \"recall\": metrics[\"recall\"],\n            \"f1\": metrics[\"f1\"],\n            \"auc\": metrics[\"auc\"],\n        })\n        trained_models[name] = model\n\n    results_df = pd.DataFrame(results).sort_values(\"f1\", ascending=False).reset_index(drop=True)\n    results_df.to_csv(os.path.join(out_dir, \"model_comparison.csv\"), index=False)\n    plot_comparison_chart(results_df, out_dir)\n\n    print(\"\\n===== BẢNG SO SÁNH KẾT QUẢ (sắp xếp theo F1-score) =====\")\n    print(results_df.to_string(index=False))\n\n    best_name = results_df.iloc[0][\"model\"]\n    best_model = trained_models[best_name]\n    print(f\"\\n[BEST MODEL] {best_name} (F1 = {results_df.iloc[0]['f1']:.4f})\")\n\n    joblib.dump(\n        {\"model\": best_model, \"scaler\": scaler, \"model_name\": best_name},\n        os.path.join(out_dir, \"best_model.pkl\"),\n    )\n    print(f\"[OK] Đã lưu model tốt nhất -> {os.path.join(out_dir, 'best_model.pkl')}\")\n\n    return results_df\n","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:10:19.501222Z","iopub.execute_input":"2026-09-28T12:10:19.501579Z","iopub.status.idle":"2026-09-28T12:10:19.52002Z","shell.execute_reply.started":"2026-09-28T12:10:19.50151Z","shell.execute_reply":"2026-09-28T12:10:19.518727Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Chạy huấn luyện & so sánh","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]}},{"cell_type":"code","source":"results_df = run_training(FEATURES_CSV, OUTPUT_DIR, test_size=0.2)\nresults_df\n","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:10:23.50888Z","iopub.execute_input":"2026-09-28T12:10:23.509218Z","iopub.status.idle":"2026-09-28T12:10:39.491985Z","shell.execute_reply.started":"2026-09-28T12:10:23.509167Z","shell.execute_reply":"2026-09-28T12:10:39.491028Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Nạp model ảnh đã train (dùng cho phần video ở Bước 6)\n\nÔ bên dưới định nghĩa `VIDEO_EXTS` và hàm `_load_best_model()` — đọc\n`outputs/best_model.pkl` (model tốt nhất ở Bước 4) để Bước 6 dùng dự đoán từng\nkhung hình của video.","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]}},{"cell_type":"code","source":"MODEL_PATH = os.path.join(OUTPUT_DIR, \"best_model.pkl\")\nVIDEO_EXTS = (\".mp4\", \".avi\", \".mov\", \".mkv\", \".webm\")\n\n\ndef _load_best_model():\n    if not os.path.exists(MODEL_PATH):\n        raise FileNotFoundError(\n            \"Chưa tìm thấy model đã train. Hãy chạy Bước 2 và Bước 4 trước.\"\n        )\n    bundle = joblib.load(MODEL_PATH)\n    return bundle[\"model\"], bundle[\"scaler\"], bundle[\"model_name\"]\n","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:11:01.020302Z","iopub.execute_input":"2026-09-28T12:11:01.020657Z","iopub.status.idle":"2026-09-28T12:11:01.026735Z","shell.execute_reply.started":"2026-09-28T12:11:01.020598Z","shell.execute_reply":"2026-09-28T12:11:01.02575Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Ghi chú báo cáo\n\n- Dự án tập trung hướng **Data Mining**: trích đặc trưng thủ công thay vì\n  để mạng CNN tự học đặc trưng (Deep Learning).\n- Có thể mở rộng thêm hướng Deep Learning (CNN/Transfer Learning) để so\n  sánh hai cách tiếp cận trong phần đánh giá của báo cáo.\n- Bước tiền xử lý (Bước 1) là phần có thể trình bày kỹ trong báo cáo ở\n  mục \"Data Cleaning / Preprocessing\" của quy trình Data Mining\n  (CRISP-DM): face detection & crop, loại nhiễu (ảnh mờ, trùng lặp), và\n  cân bằng lớp bằng augmentation — đều là các bước chuẩn hóa dữ liệu kinh\n  điển trước khi trích đặc trưng và train model.\n\n**Kết quả sau khi chạy notebook:**\n\n- `outputs/features.csv` — bảng đặc trưng đã trích xuất\n- `outputs/model_comparison.csv` / `.png` — so sánh các thuật toán\n- `outputs/cm_<model>.png` — confusion matrix từng thuật toán\n- `outputs/best_model.pkl` — model tốt nhất (dùng cho demo)\n- `outputs/eda/` — các biểu đồ phân tích dữ liệu","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]}},{"cell_type":"markdown","source":"## 6. Nhận diện video nâng cao — xác suất theo từng khung hình & phân loại cấp video\n\nPhần này dùng model ảnh ở Bước 4 để nhận diện **video**, gồm:\n\n1. **Xác suất Fake theo từng khung hình** + biểu đồ để thấy đoạn nào của video bị nghi ngờ nhất.\n2. **Hai cách tổng hợp ra kết luận cho cả video:**\n   - **Cách A:** lấy *trung bình xác suất Fake* (hoặc bỏ phiếu) của các khung hình, dùng lại model ảnh ở Bước 4.\n   - **Cách B:** gộp đặc trưng của các khung hình thành **1 vector cho cả video** (mean / std / min / max) + đặc trưng \"thời gian\" (độ chênh lệch giữa các khung), rồi **train model phân loại ở mức video**.\n3. **Đánh giá ở mức video**, chia train/test **theo video gốc** (tránh data leakage).\n4. **Demo** trên bộ video mẫu (ví dụ 5 video).\n\n**Chuẩn bị dữ liệu video** (chọn 1 trong 2 kiểu):\n\n- **Kiểu DFDC:** đặt các file `.mp4` cùng `metadata.json` (nhãn `REAL`/`FAKE` + trường `original`) vào `VIDEO_DIR`.\n- **Kiểu thư mục:** `VIDEO_DIR/Real/*.mp4` và `VIDEO_DIR/Fake/*.mp4`.\n\n> Model ảnh ở Bước 4 phải đã được train (có `outputs/best_model.pkl`).","metadata":{}},{"cell_type":"code","source":"import os, json, random, shutil\nfrom collections import defaultdict\n\nSRC = \"/kaggle/input/competitions/deepfake-detection-challenge/train_sample_videos\"\nDST = \"/kaggle/working/data/videos\"\nN_ORIG = 20        # số video Real gốc (chỉ lấy video có bản Fake)\nFAKE_PER = 2       # số bản Fake lấy cho mỗi video gốc (đặt 1 nếu muốn Real:Fake = 1:1)\nrandom.seed(42)\n\nwith open(os.path.join(SRC, \"metadata.json\"), \"r\", encoding=\"utf-8\") as f:\n    meta = json.load(f)\n\n# Gom các bản Fake theo video Real gốc\nchildren = defaultdict(list)\nfor n, m in meta.items():\n    if m[\"label\"] == \"FAKE\" and m.get(\"original\") in meta:\n        children[m[\"original\"]].append(n)\n\norigs = list(children.keys())\nrandom.shuffle(origs)\n\npick_real, pick_fake = [], []\nfor o in origs[:N_ORIG]:\n    pick_real.append(o)\n    pick_fake += random.sample(children[o], min(FAKE_PER, len(children[o])))\n\nfor d in (\"Real\", \"Fake\"):\n    shutil.rmtree(os.path.join(DST, d), ignore_errors=True)\n    os.makedirs(os.path.join(DST, d), exist_ok=True)\n\ndef link(src, dst):\n    if not os.path.exists(src):\n        return False\n    try:\n        os.symlink(src, dst)\n    except OSError:\n        shutil.copy2(src, dst)\n    return True\n\nn_r = sum(link(os.path.join(SRC, n), os.path.join(DST, \"Real\", n)) for n in pick_real)\n\nn_f = 0\nfor n in pick_fake:\n    orig_stem = os.path.splitext(meta[n][\"original\"])[0]\n    n_f += link(os.path.join(SRC, n), os.path.join(DST, \"Fake\", f\"{orig_stem}__{n}\"))\n\nprint(f\"Có {len(origs)} video Real gốc có bản Fake trong sample\")\nprint(f\"Real: {n_r} video | Fake: {n_f} video\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:11:05.637013Z","iopub.execute_input":"2026-09-28T12:11:05.637342Z","iopub.status.idle":"2026-09-28T12:11:05.749283Z","shell.execute_reply.started":"2026-09-28T12:11:05.637295Z","shell.execute_reply":"2026-09-28T12:11:05.747993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nfrom sklearn.model_selection import GroupShuffleSplit\nfrom sklearn.preprocessing import StandardScaler\n\n# ===================== 6.1 CẤU HÌNH PHẦN VIDEO =====================\nVIDEO_DIR = \"/kaggle/working/data/videos\"       # thư mục Real/ + Fake/ vừa chia\nDEMO_VIDEO_DIR = \"/kaggle/working/data/demo_videos\"\nN_FRAMES = 16\nTHRESHOLD = 0.5\nMAX_VIDEOS_PER_CLASS = 24        # đủ để lấy hết bộ demo (20 Real, 24 Fake)\n\nVIDEO_MODEL_PATH = os.path.join(OUTPUT_DIR, \"best_video_model.pkl\")\nVIDEO_FEATURES_CSV = os.path.join(OUTPUT_DIR, \"video_features.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:11:11.556511Z","iopub.execute_input":"2026-09-28T12:11:11.556813Z","iopub.status.idle":"2026-09-28T12:11:11.563765Z","shell.execute_reply.started":"2026-09-28T12:11:11.556766Z","shell.execute_reply":"2026-09-28T12:11:11.562152Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 6.2 Phân tích từng khung hình của 1 video\n\n`analyze_video_frames` lấy mẫu khung hình rải đều → detect + crop mặt (khung nào không thấy mặt thì bỏ qua) →\ntrích đặc trưng → dùng model ảnh để ra **P(Fake) cho từng khung**.","metadata":{}},{"cell_type":"code","source":"def sample_video_frames(video_path: str, n_frames: int = N_FRAMES):\n    \"\"\"Lấy tối đa n_frames khung hình rải đều (không lấy hết vì các khung liên tiếp gần như giống nhau).\"\"\"\n    cap = cv2.VideoCapture(video_path)\n    total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    if total <= 0:\n        cap.release()\n        raise RuntimeError(f\"Không đọc được video (thiếu codec hoặc file lỗi): {video_path}\")\n    fps = cap.get(cv2.CAP_PROP_FPS) or 25.0\n    ids = np.linspace(0, total - 1, min(n_frames, total), dtype=int)\n\n    frames, kept_ids = [], []\n    for fid in ids:\n        cap.set(cv2.CAP_PROP_POS_FRAMES, int(fid))\n        ok, frame = cap.read()\n        if ok:\n            frames.append(frame)\n            kept_ids.append(int(fid))\n    cap.release()\n    if not frames:\n        raise RuntimeError(f\"Không trích được khung hình nào từ video: {video_path}\")\n    return frames, kept_ids, fps\n\n\ndef analyze_video_frames(video_path: str, model, scaler, n_frames: int = N_FRAMES, skip_no_face: bool = True) -> dict:\n    \"\"\"Trả về dict gồm: p_fake (mỗi khung), crops, feats (đặc trưng thô, chưa scale), frame_ids, fps, n_no_face.\"\"\"\n    frames, frame_ids, fps = sample_video_frames(video_path, n_frames)\n\n    crops, feats, used_ids = [], [], []\n    for fr, fid in zip(frames, frame_ids):\n        crop, found = detect_and_crop_face(fr, margin=0.3)\n        if skip_no_face and not found:\n            continue\n        crops.append(crop)\n        feats.append(extract_features(crop))\n        used_ids.append(fid)\n    n_no_face = len(frames) - len(crops)\n\n    if not feats:\n        # Không khung nào detect được mặt (video đã crop sẵn / mặt quá nhỏ) -> dùng toàn bộ khung hình\n        for fr, fid in zip(frames, frame_ids):\n            crops.append(fr)\n            feats.append(extract_features(fr))\n            used_ids.append(fid)\n        n_no_face = len(frames)\n\n    feats = np.vstack(feats)\n    X = scaler.transform(feats)\n    if hasattr(model, \"predict_proba\"):\n        p_fake = model.predict_proba(X)[:, list(model.classes_).index(0)]  # nhãn 0 = Fake\n    else:\n        p_fake = (model.predict(X) == 0).astype(float)\n\n    return {\"p_fake\": p_fake, \"crops\": crops, \"feats\": feats,\n            \"frame_ids\": used_ids, \"fps\": fps, \"n_no_face\": n_no_face}\n\n\ndef aggregate_video_a(p_fake: np.ndarray, thr: float = THRESHOLD) -> dict:\n    \"\"\"Cách A: tổng hợp xác suất từng khung thành kết luận cho cả video.\"\"\"\n    mean_p = float(np.mean(p_fake))\n    frac_fake = float(np.mean(p_fake >= thr))\n    return {\n        \"mean_p_fake\": mean_p,\n        \"frac_fake_frames\": frac_fake,\n        \"pred_mean\": \"Fake\" if mean_p >= thr else \"Real\",\n        \"pred_vote\": \"Fake\" if frac_fake > 0.5 else \"Real\",\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:11:14.144797Z","iopub.execute_input":"2026-09-28T12:11:14.145121Z","iopub.status.idle":"2026-09-28T12:11:14.163839Z","shell.execute_reply.started":"2026-09-28T12:11:14.145072Z","shell.execute_reply":"2026-09-28T12:11:14.1626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict_video_detailed(video_path: str, n_frames: int = N_FRAMES, show: bool = True) -> dict:\n    \"\"\"Dự đoán 1 video theo Cách A + vẽ biểu đồ P(Fake) theo từng khung hình.\"\"\"\n    model, scaler, model_name = _load_best_model()\n    info = analyze_video_frames(video_path, model, scaler, n_frames)\n    agg = aggregate_video_a(info[\"p_fake\"])\n\n    p = info[\"p_fake\"]\n    n_fake_frames = int((p >= THRESHOLD).sum())\n    label = agg[\"pred_mean\"]\n    conf = max(agg[\"mean_p_fake\"], 1 - agg[\"mean_p_fake\"])\n\n    if show:\n        n_show = min(len(info[\"crops\"]), 8)\n        pick = np.linspace(0, len(info[\"crops\"]) - 1, n_show, dtype=int)\n\n        fig = plt.figure(figsize=(12, 5.2))\n        gs = fig.add_gridspec(2, n_show, height_ratios=[1, 1.4])\n        for k, idx in enumerate(pick):\n            ax = fig.add_subplot(gs[0, k])\n            ax.imshow(cv2.cvtColor(info[\"crops\"][idx], cv2.COLOR_BGR2RGB))\n            ax.set_title(f\"P(Fake)={p[idx]:.2f}\", fontsize=8,\n                         color=\"#e74c3c\" if p[idx] >= THRESHOLD else \"#2ecc71\")\n            ax.axis(\"off\")\n\n        ax = fig.add_subplot(gs[1, :])\n        t = np.array(info[\"frame_ids\"]) / info[\"fps\"]\n        colors = [\"#e74c3c\" if v >= THRESHOLD else \"#2ecc71\" for v in p]\n        ax.plot(t, p, color=\"#7f8c8d\", linewidth=1, zorder=1)\n        ax.scatter(t, p, c=colors, s=45, zorder=2)\n        ax.axhline(THRESHOLD, color=\"black\", linestyle=\"--\", linewidth=0.8, label=f\"ngưỡng {THRESHOLD}\")\n        ax.set_ylim(-0.02, 1.02)\n        ax.set_xlabel(\"Thời điểm trong video (giây)\")\n        ax.set_ylabel(\"Xác suất Fake\")\n        ax.legend(loc=\"upper right\", fontsize=8)\n\n        verdict = \"FAKE (nghi vấn deepfake)\" if label == \"Fake\" else \"REAL (video thật)\"\n        fig.suptitle(\n            f\"[{model_name}] {os.path.basename(video_path)}: {verdict} — độ tin cậy {conf * 100:.1f}%\\n\"\n            f\"{n_fake_frames}/{len(p)} khung hình bị đánh giá Fake  |  P(Fake) trung bình = {agg['mean_p_fake']:.2f}\",\n            fontsize=11,\n        )\n        plt.tight_layout()\n        plt.show()\n\n    return {\n        \"file\": os.path.basename(video_path), \"type\": \"video\", \"model\": model_name,\n        \"pred\": label, \"confidence\": conf, \"mean_p_fake\": agg[\"mean_p_fake\"],\n        \"fake_frames\": f\"{n_fake_frames}/{len(p)}\", \"pred_vote\": agg[\"pred_vote\"],\n        \"n_no_face\": info[\"n_no_face\"], \"info\": info,\n    }\n\n\n# Ví dụ sử dụng — thay đường dẫn video của bạn vào đây:\n# predict_video_detailed(\"data/demo_videos/fake_01.mp4\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:11:19.989506Z","iopub.execute_input":"2026-09-28T12:11:19.989858Z","iopub.status.idle":"2026-09-28T12:11:20.008881Z","shell.execute_reply.started":"2026-09-28T12:11:19.989797Z","shell.execute_reply":"2026-09-28T12:11:20.007128Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 6.3 Cách B — đặc trưng cấp video (gộp các khung hình + đặc trưng thời gian)\n\nMỗi video được biến thành **1 vector duy nhất** gồm:\n\n- **Thống kê các đặc trưng tần số/texture** (LBP + FFT — nhóm nhẹ, ít chiều) qua các khung: mean, std, min, max\n- **Thống kê xác suất Fake** của model ảnh qua các khung: mean, std, min, max, median, tỉ lệ khung Fake\n- **Đặc trưng thời gian:** độ chênh lệch pixel giữa các khung liên tiếp (frame differencing), độ biến động của vector đặc trưng giữa các khung, độ dao động của P(Fake)\n\n> HOG và Color Histogram rất nhiều chiều (hàng nghìn) nên **không** đem thống kê từng chiều lên cấp video — với số video ít sẽ bị overfit. Thông tin của chúng đã được \"nén\" vào P(Fake) của model ảnh.","metadata":{}},{"cell_type":"code","source":"def _lowdim_feature_idx() -> np.ndarray:\n    \"\"\"Chỉ số các cột LBP + FFT trong vector đặc trưng của extract_features (bỏ HOG & Color Histogram).\"\"\"\n    g = np.zeros((IMG_SIZE, IMG_SIZE), np.uint8)\n    n_lbp, n_hog, n_fft = len(extract_lbp(g)), len(extract_hog(g)), len(extract_fft_features(g))\n    return np.concatenate([np.arange(0, n_lbp), np.arange(n_lbp + n_hog, n_lbp + n_hog + n_fft)])\n\nLOWDIM_IDX = _lowdim_feature_idx()\n\n\ndef video_feature_vector(info: dict) -> np.ndarray:\n    \"\"\"Biến kết quả phân tích khung hình của 1 video thành 1 vector đặc trưng cấp video.\"\"\"\n    F = info[\"feats\"][:, LOWDIM_IDX]\n    p = info[\"p_fake\"]\n\n    agg = np.concatenate([F.mean(0), F.std(0), F.min(0), F.max(0)])\n    p_stats = np.array([p.mean(), p.std(), p.min(), p.max(), np.median(p), (p >= THRESHOLD).mean()])\n\n    if len(p) > 1:\n        gray = [cv2.resize(cv2.cvtColor(c, cv2.COLOR_BGR2GRAY), (64, 64)).astype(np.float32) for c in info[\"crops\"]]\n        pix_diff = np.array([np.abs(gray[i + 1] - gray[i]).mean() for i in range(len(gray) - 1)])\n        feat_diff = np.linalg.norm(np.diff(F, axis=0), axis=1)\n        temporal = np.array([pix_diff.mean(), pix_diff.std(), feat_diff.mean(), feat_diff.std(),\n                             np.abs(np.diff(p)).mean()])\n    else:\n        temporal = np.zeros(5)\n\n    return np.concatenate([agg, p_stats, temporal]).astype(\"float32\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:11:24.048246Z","iopub.execute_input":"2026-09-28T12:11:24.048575Z","iopub.status.idle":"2026-09-28T12:11:24.079467Z","shell.execute_reply.started":"2026-09-28T12:11:24.048526Z","shell.execute_reply":"2026-09-28T12:11:24.078629Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 6.4 Xây dựng dataset cấp video (nhãn + nhóm theo video gốc)\n\nVới DFDC, nhiều video Fake được tạo từ cùng 1 video Real gốc. Trường `original` trong `metadata.json`\nđược dùng làm **`group`**: video gốc và các bản fake của nó luôn nằm **cùng** train hoặc **cùng** test.","metadata":{}},{"cell_type":"code","source":"def list_labeled_videos(video_dir: str) -> pd.DataFrame:\n    \"\"\"Trả về DataFrame: path, file, label (1=Real, 0=Fake), group (id video gốc, dùng để chia train/test).\"\"\"\n    rows = []\n    meta_path = os.path.join(video_dir, \"metadata.json\")\n\n    if os.path.exists(meta_path):                       # kiểu DFDC\n        with open(meta_path, \"r\", encoding=\"utf-8\") as f:\n            meta = json.load(f)\n        for name, m in meta.items():\n            p = os.path.join(video_dir, name)\n            if not os.path.exists(p):                   # chỉ giữ video thực sự có mặt trong thư mục\n                continue\n            is_real = str(m.get(\"label\", \"\")).upper() == \"REAL\"\n            group = name if is_real else (m.get(\"original\") or name)\n            rows.append({\"path\": p, \"file\": name, \"label\": int(is_real), \"group\": group})\n    else:                                               # kiểu thư mục Real/ + Fake/\n        for label_name, label_val in ((\"Real\", 1), (\"Fake\", 0)):\n            for p in glob.glob(os.path.join(video_dir, label_name, \"*\")):\n                if p.lower().endswith(VIDEO_EXTS):\n                    rows.append({\"path\": p, \"file\": os.path.basename(p), \"label\": label_val,\n                                 \"group\": os.path.splitext(os.path.basename(p).split(\"__\")[0])[0]})\n\n    if not rows:\n        raise RuntimeError(\n            f\"Không tìm thấy video nào trong {video_dir}. Cần metadata.json (DFDC) hoặc thư mục con Real/ và Fake/.\"\n        )\n    return pd.DataFrame(rows)\n\n\ndef build_video_dataset(video_df: pd.DataFrame, max_per_class=MAX_VIDEOS_PER_CLASS, seed: int = 42,\n                        n_frames: int = N_FRAMES, out_csv: str = VIDEO_FEATURES_CSV) -> pd.DataFrame:\n    \"\"\"Trích vector đặc trưng cấp video + kết quả Cách A cho từng video, lưu ra CSV.\"\"\"\n    model, scaler, _ = _load_best_model()\n\n    parts = []\n    for label_val in (1, 0):\n        sub = video_df[video_df[\"label\"] == label_val]\n        if max_per_class is not None and len(sub) > max_per_class:\n            sub = sub.sample(max_per_class, random_state=seed)\n        parts.append(sub)\n    chosen = pd.concat(parts).reset_index(drop=True)\n    print(f\"[INFO] Dùng {len(chosen)} video (Real={int((chosen.label == 1).sum())}, Fake={int((chosen.label == 0).sum())})\")\n\n    rows = []\n    for _, r in tqdm(chosen.iterrows(), total=len(chosen), desc=\"Trích đặc trưng video\"):\n        try:\n            info = analyze_video_frames(r[\"path\"], model, scaler, n_frames)\n            vec = video_feature_vector(info)\n            agg = aggregate_video_a(info[\"p_fake\"])\n        except Exception as e:\n            print(f\"[WARN] Bỏ qua {r['file']}: {e}\")\n            continue\n        row = {\"file\": r[\"file\"], \"label\": r[\"label\"], \"group\": r[\"group\"],\n               \"mean_p_fake\": agg[\"mean_p_fake\"], \"frac_fake_frames\": agg[\"frac_fake_frames\"]}\n        row.update({f\"vf_{i}\": v for i, v in enumerate(vec)})\n        rows.append(row)\n\n    if not rows:\n        raise RuntimeError(\"Không trích được đặc trưng của video nào.\")\n    df = pd.DataFrame(rows)\n    os.makedirs(os.path.dirname(out_csv), exist_ok=True)\n    df.to_csv(out_csv, index=False)\n    print(f\"[OK] Đã lưu {len(df)} video -> {out_csv}\")\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:11:29.180698Z","iopub.execute_input":"2026-09-28T12:11:29.181065Z","iopub.status.idle":"2026-09-28T12:11:29.202989Z","shell.execute_reply.started":"2026-09-28T12:11:29.181002Z","shell.execute_reply":"2026-09-28T12:11:29.201832Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Chạy: build dataset cấp video\n\nBước này chậm (mỗi video phải đọc + detect mặt + trích đặc trưng ~16 khung hình). Có thể giảm `MAX_VIDEOS_PER_CLASS` ở cell cấu hình nếu chạy lâu.","metadata":{}},{"cell_type":"code","source":"video_list_df = list_labeled_videos(VIDEO_DIR)\nprint(video_list_df[\"label\"].map({1: \"Real\", 0: \"Fake\"}).value_counts())\n\nvideo_feat_df = build_video_dataset(video_list_df)\nvideo_feat_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:11:33.067849Z","iopub.execute_input":"2026-09-28T12:11:33.068205Z","iopub.status.idle":"2026-09-28T12:23:18.159156Z","shell.execute_reply.started":"2026-09-28T12:11:33.068147Z","shell.execute_reply":"2026-09-28T12:23:18.157876Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 6.5 Huấn luyện & so sánh ở mức video (chia train/test theo video gốc)\n\nSo sánh trên **cùng tập test gồm các video**:\n\n - **Cách B** với từng thuật toán trong `get_models()` (train trên vector đặc trưng cấp video)\n\nChia bằng `GroupShuffleSplit` theo cột `group` → video gốc và các bản fake của nó không bị tách qua 2 tập.","metadata":{}},{"cell_type":"code","source":"def _video_metrics(y_true, y_pred, y_score) -> dict:\n    try:\n        auc = roc_auc_score(y_true, y_score)\n    except ValueError:      # tập test chỉ có 1 lớp\n        auc = float(\"nan\")\n    return {\n        \"accuracy\": accuracy_score(y_true, y_pred),\n        \"precision\": precision_score(y_true, y_pred, zero_division=0),\n        \"recall\": recall_score(y_true, y_pred, zero_division=0),\n        \"f1\": f1_score(y_true, y_pred, zero_division=0),\n        \"auc\": auc,\n    }\n\n\ndef run_video_training(df: pd.DataFrame, out_dir: str = OUTPUT_DIR, test_size: float = 0.25, seed: int = 42):\n    feat_cols = [c for c in df.columns if c.startswith(\"vf_\")]\n\n    gss = GroupShuffleSplit(n_splits=1, test_size=test_size, random_state=seed)\n    tr_idx, te_idx = next(gss.split(df, df[\"label\"], groups=df[\"group\"]))\n    train_df, test_df = df.iloc[tr_idx], df.iloc[te_idx]\n    y_train, y_test = train_df[\"label\"].values, test_df[\"label\"].values\n    print(f\"[INFO] Train: {len(train_df)} video (Real={int((y_train == 1).sum())}, Fake={int((y_train == 0).sum())}) | \"\n          f\"Test: {len(test_df)} video (Real={int((y_test == 1).sum())}, Fake={int((y_test == 0).sum())})\")\n    if len(set(y_train)) < 2:\n        raise RuntimeError(\"Tập train chỉ có 1 lớp — cần thêm video hoặc đổi seed/test_size.\")\n\n    results, preds = [], {}\n\n    # ---- Cách A: dùng thẳng model ảnh, không train thêm (quy ước nhãn: 1=Real, 0=Fake) ----\n    a_mean = (test_df[\"mean_p_fake\"].values < THRESHOLD).astype(int)\n    a_vote = (test_df[\"frac_fake_frames\"].values <= 0.5).astype(int)\n    for name, y_pred, y_score in (\n        (\"A: trung bình xác suất\", a_mean, 1 - test_df[\"mean_p_fake\"].values),\n        (\"A: bỏ phiếu đa số\", a_vote, 1 - test_df[\"frac_fake_frames\"].values),\n    ):\n        results.append({\"method\": name, **_video_metrics(y_test, y_pred, y_score)})\n        preds[name] = y_pred\n\n    # ---- Cách B: train model trên vector đặc trưng cấp video ----\n    scaler_v = StandardScaler()\n    X_train = scaler_v.fit_transform(train_df[feat_cols].values)\n    X_test = scaler_v.transform(test_df[feat_cols].values)\n\n    trained = {}\n    for name, model in get_models().items():\n        try:\n            model.fit(X_train, y_train)\n        except Exception as e:\n            print(f\"[WARN] Bỏ qua {name}: {e}\")\n            continue\n        y_pred = model.predict(X_test)\n        y_score = model.predict_proba(X_test)[:, 1] if hasattr(model, \"predict_proba\") else y_pred\n        label = f\"B: {name}\"\n        results.append({\"method\": label, **_video_metrics(y_test, y_pred, y_score)})\n        preds[label] = y_pred\n        trained[label] = model\n\n    res_df = pd.DataFrame(results).sort_values(\"f1\", ascending=False).reset_index(drop=True)\n    res_df.to_csv(os.path.join(out_dir, \"video_model_comparison.csv\"), index=False)\n\n    print(\"\\n===== SO SÁNH Ở MỨC VIDEO (sắp xếp theo F1-score) =====\")\n    print(res_df.to_string(index=False))\n\n    # Biểu đồ so sánh\n    metrics = [\"accuracy\", \"precision\", \"recall\", \"f1\", \"auc\"]\n    x, width = np.arange(len(res_df)), 0.16\n    plt.figure(figsize=(12, 5.5))\n    for i, m in enumerate(metrics):\n        plt.bar(x + i * width, res_df[m], width, label=m)\n    plt.xticks(x + width * 2, res_df[\"method\"], rotation=25, ha=\"right\")\n    plt.ylabel(\"Điểm số\")\n    plt.title(\"So sánh các cách nhận diện ở mức video (test chia theo video gốc)\")\n    plt.legend()\n    plt.tight_layout()\n    plt.savefig(os.path.join(out_dir, \"video_comparison.png\"), dpi=150)\n    plt.show()\n\n    # Confusion matrix của phương pháp tốt nhất\n    best_method = res_df.iloc[0][\"method\"]\n    plot_confusion_matrix(y_test, preds[best_method], \"video_\" + best_method.replace(\":\", \"\").replace(\" \", \"_\"), out_dir)\n\n    # Lưu model Cách B tốt nhất (nếu Cách B có mặt trong bảng) để dùng cho demo\n    best_b = res_df[res_df[\"method\"].isin(trained.keys())]\n    if len(best_b):\n        b_name = best_b.iloc[0][\"method\"]\n        joblib.dump({\"model\": trained[b_name], \"scaler\": scaler_v, \"model_name\": b_name, \"feat_cols\": feat_cols},\n                    VIDEO_MODEL_PATH)\n        print(f\"\\n[OK] Đã lưu model video (Cách B) tốt nhất: {b_name} -> {VIDEO_MODEL_PATH}\")\n\n    return res_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:26:34.358914Z","iopub.execute_input":"2026-09-28T12:26:34.359423Z","iopub.status.idle":"2026-09-28T12:26:34.387463Z","shell.execute_reply.started":"2026-09-28T12:26:34.359307Z","shell.execute_reply":"2026-09-28T12:26:34.386413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"video_results_df = run_video_training(video_feat_df, OUTPUT_DIR, test_size=0.25)\nvideo_results_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:26:38.900387Z","iopub.execute_input":"2026-09-28T12:26:38.9007Z","iopub.status.idle":"2026-09-28T12:26:41.444609Z","shell.execute_reply.started":"2026-09-28T12:26:38.90065Z","shell.execute_reply":"2026-09-28T12:26:41.443475Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 6.6 Demo: nhận diện video Real/Fake\n\n`predict_video_full` chạy cả **Cách A** (kèm biểu đồ xác suất theo khung hình) và **Cách B** (nếu đã train xong\nmodel video ở 6.5). `demo_videos` chạy cho cả 1 thư mục (ví dụ 5 video demo) và tổng hợp thành bảng.\n\nNếu tên file bắt đầu bằng `real` / `fake` (ví dụ `real_01.mp4`, `fake_02.mp4`) hoặc thư mục có `metadata.json`,\nbảng kết quả sẽ có thêm cột đáp án để đối chiếu.","metadata":{}},{"cell_type":"code","source":"import os, json, random, shutil\nfrom collections import defaultdict\n\nSRC = \"/kaggle/input/competitions/deepfake-detection-challenge/train_sample_videos\"\nTRAIN_DIR = \"/kaggle/working/data/videos\"\nDEMO_VIDEO_DIR = \"/kaggle/working/data/demo_videos\"\nrandom.seed(42)\n\nwith open(os.path.join(SRC, \"metadata.json\"), \"r\", encoding=\"utf-8\") as f:\n    meta = json.load(f)\n\n# Các video gốc đã dùng để train (tên file trong Real/)\nused = {os.path.splitext(f)[0] for f in os.listdir(os.path.join(TRAIN_DIR, \"Real\"))}\n\n# Gom bản Fake theo video Real gốc\nchildren = defaultdict(list)\nfor n, m in meta.items():\n    if m[\"label\"] == \"FAKE\" and m.get(\"original\") in meta:\n        children[m[\"original\"]].append(n)\n\n# Chỉ lấy video gốc CHƯA dùng để train\nunused = [o for o in children if os.path.splitext(o)[0] not in used]\nrandom.shuffle(unused)\ndemo_real = unused[:3]\ndemo_fake = [random.choice(children[o]) for o in unused[3:6]]\n\nshutil.rmtree(DEMO_VIDEO_DIR, ignore_errors=True)\nos.makedirs(DEMO_VIDEO_DIR, exist_ok=True)\n\ndef link(src, dst):\n    try:\n        os.symlink(src, dst)\n    except OSError:\n        shutil.copy2(src, dst)\n\nfor i, n in enumerate(demo_real, 1):\n    link(os.path.join(SRC, n), os.path.join(DEMO_VIDEO_DIR, f\"real_{i:02d}.mp4\"))\nfor i, n in enumerate(demo_fake, 1):\n    link(os.path.join(SRC, n), os.path.join(DEMO_VIDEO_DIR, f\"fake_{i:02d}.mp4\"))\n\nprint(sorted(os.listdir(DEMO_VIDEO_DIR)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:26:45.382849Z","iopub.execute_input":"2026-09-28T12:26:45.383175Z","iopub.status.idle":"2026-09-28T12:26:45.407305Z","shell.execute_reply.started":"2026-09-28T12:26:45.383128Z","shell.execute_reply":"2026-09-28T12:26:45.406015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def _load_video_model():\n    return joblib.load(VIDEO_MODEL_PATH) if os.path.exists(VIDEO_MODEL_PATH) else None\n\n\ndef _guess_truth(path: str):\n    \"\"\"Đáp án (nếu biết): từ metadata.json cùng thư mục, hoặc từ tiền tố tên file real*/fake*.\"\"\"\n    name = os.path.basename(path)\n    meta_path = os.path.join(os.path.dirname(path), \"metadata.json\")\n    if os.path.exists(meta_path):\n        try:\n            with open(meta_path, \"r\", encoding=\"utf-8\") as f:\n                m = json.load(f).get(name)\n            if m:\n                return \"Real\" if str(m.get(\"label\", \"\")).upper() == \"REAL\" else \"Fake\"\n        except Exception:\n            pass\n    low = name.lower()\n    if low.startswith(\"real\"):\n        return \"Real\"\n    if low.startswith(\"fake\"):\n        return \"Fake\"\n    return None\n\n\ndef predict_video_full(video_path: str, n_frames: int = N_FRAMES, show: bool = True) -> dict:\n    \"\"\"Cách A (luôn có) + Cách B (nếu đã train model video).\"\"\"\n    res = predict_video_detailed(video_path, n_frames=n_frames, show=show)\n    info = res.pop(\"info\")\n\n    out = {\n        \"file\": res[\"file\"], \"truth\": _guess_truth(video_path),\n        \"A_pred\": res[\"pred\"], \"A_conf\": res[\"confidence\"],\n        \"fake_frames\": res[\"fake_frames\"], \"A_vote\": res[\"pred_vote\"],\n        \"B_pred\": None, \"B_conf\": None,\n    }\n\n    bundle = _load_video_model()\n    if bundle is not None:\n        vec = video_feature_vector(info).reshape(1, -1)\n        Xv = bundle[\"scaler\"].transform(vec)\n        pred = int(bundle[\"model\"].predict(Xv)[0])\n        out[\"B_pred\"] = \"Real\" if pred == 1 else \"Fake\"\n        if hasattr(bundle[\"model\"], \"predict_proba\"):\n            out[\"B_conf\"] = float(bundle[\"model\"].predict_proba(Xv)[0][pred])\n    return out\n\n\ndef demo_videos(folder_path: str, n_frames: int = N_FRAMES, show: bool = True) -> pd.DataFrame:\n    \"\"\"Chạy nhận diện cho mọi video trong 1 thư mục (không đệ quy) và trả về bảng tổng hợp.\"\"\"\n    paths = sorted(p for p in glob.glob(os.path.join(folder_path, \"*\")) if p.lower().endswith(VIDEO_EXTS))\n    if not paths:\n        raise RuntimeError(f\"Không tìm thấy video nào trong: {folder_path}\")\n\n    rows = []\n    for p in tqdm(paths, desc=\"Nhận diện video\"):\n        try:\n            rows.append(predict_video_full(p, n_frames=n_frames, show=show))\n        except Exception as e:\n            print(f\"[WARN] Lỗi khi xử lý {p}: {e}\")\n            rows.append({\"file\": os.path.basename(p), \"truth\": _guess_truth(p), \"A_pred\": \"ERROR\"})\n\n    df = pd.DataFrame(rows)\n    if df[\"truth\"].notna().any():\n        df[\"A_đúng\"] = df.apply(lambda r: (r[\"A_pred\"] == r[\"truth\"]) if isinstance(r[\"truth\"], str) else None, axis=1)\n        if \"B_pred\" in df.columns and df[\"B_pred\"].notna().any():\n            df[\"B_đúng\"] = df.apply(\n                lambda r: (r[\"B_pred\"] == r[\"truth\"]) if isinstance(r[\"truth\"], str) and isinstance(r[\"B_pred\"], str) else None,\n                axis=1,\n            )\n    return df\n\n\n# Chạy demo cho thư mục video mẫu:\ndemo_df = demo_videos(DEMO_VIDEO_DIR)\ndemo_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:26:50.182568Z","iopub.execute_input":"2026-09-28T12:26:50.182906Z","iopub.status.idle":"2026-09-28T12:29:06.782592Z","shell.execute_reply.started":"2026-09-28T12:26:50.182856Z","shell.execute_reply":"2026-09-28T12:29:06.781428Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Lưu ý khi đọc kết quả phần video\n\n- **Domain gap:** model ảnh (Bước 4) được train trên ảnh tĩnh, còn video DFDC bị nén và sinh bằng nhiều phương pháp khác nhau\n  → điểm Cách A trên video có thể thấp hơn nhiều so với điểm trên ảnh. Muốn tốt hơn, train lại model ảnh trên **khung hình trích từ chính video**\n  (vẫn chia train/test theo video gốc).\n- **Số video ít** (vài chục) → điểm ở 6.5 dao động mạnh giữa các lần chia; nên thử vài `seed` khác nhau, hoặc dùng `GroupKFold`.\n- **Đặc trưng thời gian** ở đây tính trên các khung hình cách xa nhau nên chỉ phản ánh độ biến động thô. Muốn bắt hiện tượng \"giật\" thật sự\n  cần lấy các khung **liên tiếp** (hoặc thêm optical flow / landmark).\n- Nên trình bày trong báo cáo: bảng so sánh Cách A vs Cách B ở 6.5, biểu đồ P(Fake) theo khung hình của 1 video Real và 1 video Fake.","metadata":{}},{"cell_type":"code","source":"import shutil\nimport matplotlib.pyplot as plt\n\nDETECT_OUT = \"/kaggle/working/detections\"\n\ndef detect_and_save(video_path, out_dir=DETECT_OUT, n_frames=N_FRAMES, top_k=3):\n    model, scaler, _ = _load_best_model()\n    info = analyze_video_frames(video_path, model, scaler, n_frames)\n    p = info[\"p_fake\"]\n    agg = aggregate_video_a(p)\n    label = agg[\"pred_mean\"]                       # \"Real\" hoặc \"Fake\"\n\n    stem = os.path.splitext(os.path.basename(video_path))[0]\n    dest = os.path.join(out_dir, f\"Predicted_{label}\", stem)\n    os.makedirs(dest, exist_ok=True)\n\n    # 1) Copy video gốc\n    shutil.copy2(video_path, os.path.join(dest, os.path.basename(video_path)))\n\n    # 2) Biểu đồ P(Fake) theo thời gian\n    t = np.array(info[\"frame_ids\"]) / info[\"fps\"]\n    plt.figure(figsize=(7, 3))\n    plt.plot(t, p, color=\"#7f8c8d\")\n    plt.scatter(t, p, c=[\"#e74c3c\" if v >= THRESHOLD else \"#2ecc71\" for v in p], s=40)\n    plt.axhline(THRESHOLD, color=\"black\", linestyle=\"--\", linewidth=0.8)\n    plt.ylim(-0.02, 1.02); plt.xlabel(\"Thời gian (s)\"); plt.ylabel(\"P(Fake)\")\n    plt.title(f\"{stem} -> {label} (mean P(Fake)={agg['mean_p_fake']:.2f})\")\n    plt.tight_layout()\n    plt.savefig(os.path.join(dest, \"p_fake_chart.png\"), dpi=120)\n    plt.close()\n\n    # 3) Lưu top_k khung mặt nghi ngờ nhất\n    for rank, idx in enumerate(np.argsort(-p)[:top_k], 1):\n        cv2.imwrite(os.path.join(dest, f\"suspect_{rank}_p{p[idx]:.2f}.jpg\"), info[\"crops\"][idx])\n\n    # 4) Lưu đặc trưng\n    np.save(os.path.join(dest, \"frame_features.npy\"), info[\"feats\"])\n    np.save(os.path.join(dest, \"video_feature_vector.npy\"), video_feature_vector(info))\n    np.save(os.path.join(dest, \"p_fake.npy\"), p)\n\n    row = {\"file\": os.path.basename(video_path), \"truth\": _guess_truth(video_path),\n           \"pred\": label, \"mean_p_fake\": round(agg[\"mean_p_fake\"], 3),\n           \"fake_frames\": int((p >= THRESHOLD).sum()), \"n_frames\": len(p),\n           \"folder\": dest}\n\n    # 5) Cách B (nếu đã train model video ở 6.5)\n    bundle = _load_video_model()\n    if bundle is not None:\n        vec = video_feature_vector(info).reshape(1, -1)\n        pb = int(bundle[\"model\"].predict(bundle[\"scaler\"].transform(vec))[0])\n        row[\"B_pred\"] = \"Real\" if pb == 1 else \"Fake\"\n    return row\n\n\ndef detect_folder_and_save(folder_path, out_dir=DETECT_OUT):\n    paths = sorted(p for p in glob.glob(os.path.join(folder_path, \"*\")) if p.lower().endswith(VIDEO_EXTS))\n    if not paths:\n        raise RuntimeError(f\"Không tìm thấy video nào trong: {folder_path}\")\n    rows = []\n    for p in tqdm(paths, desc=\"Nhận diện + lưu\"):\n        try:\n            rows.append(detect_and_save(p, out_dir))\n        except Exception as e:\n            print(f\"[WARN] {p}: {e}\")\n    df = pd.DataFrame(rows)\n    os.makedirs(out_dir, exist_ok=True)\n    df.to_csv(os.path.join(out_dir, \"detection_results.csv\"), index=False)\n    return df\n\n\ndetect_df = detect_folder_and_save(DEMO_VIDEO_DIR)\ndetect_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-28T12:29:49.190137Z","iopub.execute_input":"2026-09-28T12:29:49.190456Z","iopub.status.idle":"2026-09-28T12:32:00.477203Z","shell.execute_reply.started":"2026-09-28T12:29:49.190409Z","shell.execute_reply":"2026-09-28T12:32:00.476412Z"}},"outputs":[],"execution_count":null}]}