{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":1157383,"sourceType":"datasetVersion","datasetId":548681},{"sourceId":1538322,"sourceType":"datasetVersion","datasetId":907200},{"sourceId":3341486,"sourceType":"datasetVersion","datasetId":2017372},{"sourceId":6717213,"sourceType":"datasetVersion","datasetId":1317048},{"sourceId":14470267,"sourceType":"datasetVersion","datasetId":9242266},{"sourceId":14472750,"sourceType":"datasetVersion","datasetId":9243989},{"sourceId":14477023,"sourceType":"datasetVersion","datasetId":9246794},{"sourceId":14477662,"sourceType":"datasetVersion","datasetId":9247207},{"sourceId":14480316,"sourceType":"datasetVersion","datasetId":9248475}],"dockerImageVersionId":31234,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import os\n# import shutil\n# import pandas as pd\n# from tqdm import tqdm\n\n# # 1. Định nghĩa đường dẫn nguồn trên Kaggle (Thay đổi tên cho khớp với tên folder bạn add)\n# mapping = {\n#     \"cohen\": \"/kaggle/input/covid-chest-xray-cohen/images\",\n#     \"rsna\": \"/kaggle/input/rsna-png/rsna_png\",\n#     \"sirm\": \"/kaggle/input/covid19-radiography-database/COVID-19_Radiography_Dataset/COVID/images\",\n#     \"fig1\": \"/kaggle/input/figure1-covid-chestxray-dataset/Figure1-COVID-chestxray-dataset/images\",\n#     \"actmed\": \"/kaggle/input/actualmedcovidchestxraydataset/images\",\n#     \"ds4c\": \"/kaggle/input/covid19-radiography-database/COVID-19_Radiography_Dataset/COVID/images\" # Thêm ds4c nếu có\n# }\n\n# # 2. Tạo thư mục đích trong /kaggle/working\n# os.makedirs(\"/kaggle/working/data/train\", exist_ok=True)\n# os.makedirs(\"/kaggle/working/data/test\", exist_ok=True)\n\n# def process_split(split_file, output_dir):\n#     with open(split_file, 'r') as f:\n#         lines = f.readlines()\n        \n#     for line in tqdm(lines):\n#         parts = line.strip().split()\n#         # Cấu trúc file thường là: patient_id file_name label source\n#         if len(parts) < 4: continue\n        \n#         file_name = parts[1]\n#         label = parts[2]\n#         source = parts[3]\n        \n#         # Tìm đường dẫn gốc dựa trên source\n#         if source in mapping:\n#             src_path = os.path.join(mapping[source], file_name)\n#             dst_path = os.path.join(output_dir, file_name)\n            \n#             # Kiểm tra nếu file tồn tại thì mới copy\n#             if os.path.exists(src_path):\n#                 shutil.copy(src_path, dst_path)\n\n# # 3. Chạy lệnh gộp\n# process_split('/kaggle/input/txt-traing-and-test/train_split_fixed.txt', '/kaggle/working/data/train')\n# process_split('/kaggle/input/txt-traing-and-test/test_COVIDx4.txt', '/kaggle/working/data/test')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-12T05:52:11.156122Z","iopub.execute_input":"2026-01-12T05:52:11.157265Z","iopub.status.idle":"2026-01-12T05:52:11.165430Z","shell.execute_reply.started":"2026-01-12T05:52:11.157223Z","shell.execute_reply":"2026-01-12T05:52:11.164135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\nimport re\nfrom tqdm import tqdm\n\n# 1. Cấu hình đường dẫn nguồn (Hãy đảm bảo các tên folder Input này là chính xác nhất)\nmapping = {\n    \"cohen\": \"/kaggle/input/covid-chest-xray-cohen/images\",\n    \"rsna\": \"/kaggle/input/rsna-png-version/rsna_png\",\n    \"sirm\": \"/kaggle/input/covid19-radiography-database/COVID-19_Radiography_Dataset/COVID/images\",\n    \"fig1\": \"/kaggle/input/figure1-covid-chestxray-dataset/Figure1-COVID-chestxray-dataset/images\",\n    \"actmed\": \"/kaggle/input/actualmedcovidchestxraydataset/images\",\n    \"ds4c\": \"/kaggle/input/covid19-radiography-database/COVID-19_Radiography_Dataset/COVID/images\"\n}\n\n# 2. Tạo thư mục đích\nos.makedirs(\"/kaggle/working/data/train\", exist_ok=True)\nos.makedirs(\"/kaggle/working/data/test\", exist_ok=True)\n\ndef smart_process_split(split_file, output_dir):\n    print(f\"\\n🚀 Đang xử lý: {os.path.basename(split_file)}\")\n    \n    with open(split_file, 'r') as f:\n        lines = f.readlines()\n        \n    stats = {\"found\": 0, \"missing\": 0, \"fixed_name\": 0}\n    \n    for line in tqdm(lines):\n        parts = line.strip().split()\n        if len(parts) < 3: continue\n        \n        file_name = parts[1]\n        # Nếu chỉ có 3 cột, mặc định nguồn là rsna\n        source = parts[3] if len(parts) >= 4 else \"rsna\"\n        \n        # --- BƯỚC A: Sửa tên file SIRM nếu cần ---\n        if source == 'sirm' and 'COVID-19(' in file_name:\n            match = re.search(r'\\((\\d+)\\)', file_name)\n            if match:\n                file_name = f\"COVID-{match.group(1)}.png\"\n                stats[\"fixed_name\"] += 1\n\n        # --- BƯỚC B: Tìm file (Smart Search) ---\n        src_path = None\n        \n        # 1. Thử nguồn ưu tiên trước\n        primary_dir = mapping.get(source)\n        if primary_dir:\n            temp_path = os.path.join(primary_dir, file_name)\n            if os.path.exists(temp_path):\n                src_path = temp_path\n        \n        # 2. Nếu không thấy, quét tất cả các nguồn khác (Xử lý vụ file SARS... nằm nhầm nguồn)\n        if not src_path:\n            for s_name, s_dir in mapping.items():\n                temp_path = os.path.join(s_dir, file_name)\n                if os.path.exists(temp_path):\n                    src_path = temp_path\n                    break\n                    \n        # 3. Thử đổi đuôi file (jpg <-> jpeg) cho nguồn Cohen\n        if not src_path and ('.jpg' in file_name or '.jpeg' in file_name):\n            alt_name = file_name.replace('.jpg', '.jpeg') if '.jpg' in file_name else file_name.replace('.jpeg', '.jpg')\n            for s_dir in mapping.values():\n                temp_path = os.path.join(s_dir, alt_name)\n                if os.path.exists(temp_path):\n                    src_path = temp_path\n                    file_name = alt_name # Cập nhật tên mới để copy\n                    break\n\n        # --- BƯỚC C: Copy file ---\n        if src_path:\n            dst_path = os.path.join(output_dir, file_name)\n            shutil.copy(src_path, dst_path)\n            stats[\"found\"] += 1\n        else:\n            stats[\"missing\"] += 1\n\n    # In tổng kết cho tập đang xử lý\n    print(f\"✅ Thành công: {stats['found']}\")\n    print(f\"🛠️ Đã sửa tên: {stats['fixed_name']}\")\n    print(f\"❌ Thất bại: {stats['missing']}\")\n    return stats\n\n# 3. Thực thi\ntrain_stats = smart_process_split('/kaggle/input/training-file/train_split.txt', '/kaggle/working/data/train')\ntest_stats = smart_process_split('/kaggle/input/training-file/test_split_fixed.txt', '/kaggle/working/data/test')\n\nprint(\"\\n\" + \"=\"*40)\nprint(f\"🏆 TỔNG KẾT CUỐI CÙNG:\")\nprint(f\"📁 TRAIN: {train_stats['found']} ảnh\")\nprint(f\"📁 TEST : {test_stats['found']} ảnh\")\nprint(f\"📍 Toàn bộ dữ liệu nằm tại: /kaggle/working/data\")\nprint(\"=\"*40)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n\n# # 1. Cấu hình lại mapping (Bỏ base_input thừa và kiểm tra kỹ tên folder trên Kaggle)\n# # Lưu ý: Hãy chắc chắn tên dataset (slug) khớp với phần 'Add Data' của bạn\n# mapping = {\n#     \"cohen\": \"/kaggle/input/covid-chest-xray-cohen/images\",\n#     \"rsna\": \"/kaggle/input/rsna-png-version/rsna_png\",\n#     \"sirm\": \"/kaggle/input/covid19-radiography-database/COVID-19_Radiography_Dataset/COVID/images\",\n#     \"fig1\": \"/kaggle/input/figure1-covid-chestxray-dataset/Figure1-COVID-chestxray-dataset/images\",\n#     \"actmed\": \"/kaggle/input/actualmedcovidchestxraydataset/images\",\n#     \"ds4c\": \"/kaggle/input/covid19-radiography-database/COVID-19_Radiography_Dataset/COVID/images\" # Thêm ds4c nếu có\n# }\n\n# missing_count = 0\n# found_count = 0\n# source_stats = {k: {\"found\": 0, \"missing\": 0} for k in mapping.keys()}\n\n# # Đường dẫn tới file txt của bạn\n# txt_path = '/kaggle/input/training-file/train_split.txt'\n\n# print(\"--- Đang kiểm tra dữ liệu ---\")\n\n# if not os.path.exists(txt_path):\n#     print(f\"❌ Lỗi: Không tìm thấy file {txt_path}\")\n# else:\n#     with open(txt_path, 'r') as f:\n#         for line in f:\n#             parts = line.strip().split()\n#             if len(parts) < 4:\n#                 continue\n            \n#             filename = parts[1] # Tên file ảnh\n#             source = parts[3]   # Nguồn (cohen, rsna, sirm...)\n\n#             if source in mapping:\n#                 # Lấy đường dẫn thư mục tương ứng với source này\n#                 dir_path = mapping[source]\n#                 full_path = os.path.join(dir_path, filename)\n                \n#                 if os.path.exists(full_path):\n#                     found_count += 1\n#                     source_stats[source][\"found\"] += 1\n#                 else:\n#                     missing_count += 1\n#                     source_stats[source][\"missing\"] += 1\n#                     # In ra 5 file đầu tiên bị thiếu của mỗi loại để debug\n#                     if source_stats[source][\"missing\"] <= 1:\n#                         print(f\"⚠️ Thiếu file: {filename} tại nguồn {source}\")\n#             else:\n#                 missing_count += 1\n#                 if found_count < 1: # Chỉ in cảnh báo source lạ một lần\n#                     print(f\"❓ Source lạ không có trong mapping: {source}\")\n\n# print(\"\\n\" + \"=\"*30)\n# print(f\"📊 TỔNG KẾT KIỂM TRA:\")\n# print(f\"✅ Tổng file tìm thấy: {found_count}\")\n# print(f\"❌ Tổng file bị thiếu: {missing_count}\")\n# print(\"-\" * 30)\n# for src, stats in source_stats.items():\n#     total = stats['found'] + stats['missing']\n#     if total > 0:\n#         print(f\"📍 {src.upper()}: Tìm thấy {stats['found']}/{total} file\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-13T03:21:04.727995Z","iopub.execute_input":"2026-01-13T03:21:04.728453Z","iopub.status.idle":"2026-01-13T03:21:27.322470Z","shell.execute_reply.started":"2026-01-13T03:21:04.728420Z","shell.execute_reply":"2026-01-13T03:21:27.321294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n\n# def locate_missing_file(filename):\n#     search_paths = [\n#         \"/kaggle/input\",\n#         \"/kaggle/working\"\n#     ]\n#     for path in search_paths:\n#         for root, dirs, files in os.walk(path):\n#             if filename in files:\n#                 return os.path.join(root, filename)\n#     return \"Không tìm thấy trong toàn bộ hệ thống\"\n\n# # Thử tìm file đầu tiên bị thiếu trong log của bạn\n# print(f\"🔍 Vị trí thực tế của file thiếu: {locate_missing_file('47c78742-4998-4878-aec4-37b11b1354ac.png')}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-13T01:52:39.349667Z","iopub.execute_input":"2026-01-13T01:52:39.350011Z","iopub.status.idle":"2026-01-13T01:55:29.107426Z","shell.execute_reply.started":"2026-01-13T01:52:39.349984Z","shell.execute_reply":"2026-01-13T01:55:29.106521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# import shutil\n# from tqdm import tqdm\n\n# # Mapping hiện tại của bạn\n# mapping = {\n#     \"cohen\": \"/kaggle/input/covid-chest-xray-cohen/images\",\n#     \"rsna\": \"/kaggle/input/rsna-png-version/data/train\", # Đường dẫn chuẩn bạn đã tìm\n#     \"sirm\": \"/kaggle/input/covid19-radiography-database/COVID-19_Radiography_Dataset/COVID/images\",\n#     \"fig1\": \"/kaggle/input/figure1-covid-chestxray-dataset/Figure1-COVID-chestxray-dataset/images\",\n#     \"actmed\": \"/kaggle/input/actualmedcovidchestxraydataset/images\"\n# }\n\n# def smart_merge(txt_file, output_dir):\n#     os.makedirs(output_dir, exist_ok=True)\n#     found, missing = 0, 0\n    \n#     with open(txt_file, 'r') as f:\n#         lines = f.readlines()\n        \n#     for line in tqdm(lines):\n#         parts = line.strip().split()\n#         if len(parts) < 3: continue\n        \n#         filename = parts[1]\n#         primary_source = parts[3] if len(parts) >= 4 else \"rsna\"\n        \n#         # 1. Thử tìm ở nguồn chính (Primary Source)\n#         src_path = os.path.join(mapping.get(primary_source, \"\"), filename)\n        \n#         # 2. Nếu không thấy, duyệt qua TẤT CẢ các nguồn khác để tìm (Smart Search)\n#         if not os.path.exists(src_path):\n#             for src_name, src_dir in mapping.items():\n#                 temp_path = os.path.join(src_dir, filename)\n#                 if os.path.exists(temp_path):\n#                     src_path = temp_path\n#                     break\n        \n#         # 3. Copy nếu tìm thấy\n#         if os.path.exists(src_path):\n#             shutil.copy(src_path, os.path.join(output_dir, filename))\n#             found += 1\n#         else:\n#             missing += 1\n#             print(f\"❌ Vẫn không tìm thấy: {filename}\")\n\n#     print(f\"\\n✅ Hoàn tất! Tìm thấy: {found} | Thiếu: {missing}\")\n\n# # Chạy cho tập TEST\n# smart_merge('/kaggle/input/test-file/test_split_fixed.txt', '/kaggle/working/data/test')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-13T01:50:55.375226Z","iopub.execute_input":"2026-01-13T01:50:55.376303Z","iopub.status.idle":"2026-01-13T01:51:00.171635Z","shell.execute_reply.started":"2026-01-13T01:50:55.376241Z","shell.execute_reply":"2026-01-13T01:51:00.170642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import os\n# import shutil\n# from tqdm import tqdm\n\n# # 1. Cấu hình Mapping (Giữ nguyên cấu hình bạn đã cung cấp)\n# mapping = {\n#     \"cohen\": \"/kaggle/input/covid-chest-xray-cohen/images\",\n#     \"rsna\": \"/kaggle/input/rsna-png/rsna_png\",\n#     \"sirm\": \"/kaggle/input/covid19-radiography-database/COVID-19_Radiography_Dataset/COVID/images\",\n#     \"fig1\": \"/kaggle/input/figure1-covid-chestxray-dataset/Figure1-COVID-chestxray-dataset/images\",\n#     \"actmed\": \"/kaggle/input/actualmedcovidchestxraydataset/images\",\n#     \"ds4c\": \"/kaggle/input/covid19-radiography-database/COVID-19_Radiography_Dataset/COVID/images\"\n# }\n\n# # 2. Cấu hình danh sách file split và thư mục đích\n# # Format: (đường dẫn file txt, tên thư mục con trong working)\n# splits = [\n#     ('/kaggle/input/txt-traing-and-test/train_split_fixed.txt', 'train'),\n#     ('/kaggle/input/txt-traing-and-test/test_COVIDx4.txt', 'test') # Hãy kiểm tra lại đường dẫn này\n# ]\n\n# output_base = \"/kaggle/working/data\"\n\n# def process_merge():\n#     overall_found = 0\n#     overall_missing = 0\n    \n#     for txt_path, folder_name in splits:\n#         if not os.path.exists(txt_path):\n#             print(f\"❌ Lỗi: Không tìm thấy file {txt_path}\")\n#             continue\n            \n#         dest_dir = os.path.join(output_base, folder_name)\n#         os.makedirs(dest_dir, exist_ok=True)\n        \n#         print(f\"\\n--- Đang xử lý tập: {folder_name.upper()} ({txt_path}) ---\")\n        \n#         found_count = 0\n#         missing_count = 0\n#         source_stats = {k: {\"found\": 0, \"missing\": 0} for k in mapping.keys()}\n\n#         with open(txt_path, 'r') as f:\n#             lines = f.readlines()\n            \n#         for line in tqdm(lines, desc=f\"Copying {folder_name}\"):\n#             parts = line.strip().split()\n#             if len(parts) < 4:\n#                 continue\n                \n#             filename = parts[1]\n#             source = parts[3]\n            \n#             if source in mapping:\n#                 src_path = os.path.join(mapping[source], filename)\n#                 dst_path = os.path.join(dest_dir, filename)\n                \n#                 if os.path.exists(src_path):\n#                     # Thực hiện copy file\n#                     shutil.copy(src_path, dst_path)\n#                     found_count += 1\n#                     source_stats[source][\"found\"] += 1\n#                 else:\n#                     missing_count += 1\n#                     source_stats[source][\"missing\"] += 1\n#             else:\n#                 missing_count += 1\n        \n#         # In tổng kết cho từng tập\n#         print(f\"✅ {folder_name.upper()}: Tìm thấy & Copy {found_count} file\")\n#         print(f\"❌ {folder_name.upper()}: Thiếu {missing_count} file\")\n#         overall_found += found_count\n#         overall_missing += missing_count\n\n#     print(\"\\n\" + \"=\"*40)\n#     print(f\"🚀 TỔNG KẾT TOÀN BỘ DATASET:\")\n#     print(f\"📁 Tổng file đã copy thành công: {overall_found}\")\n#     print(f\"⚠️ Tổng file bị lỗi/thiếu: {overall_missing}\")\n#     print(f\"📍 Dữ liệu đã sẵn sàng tại: {output_base}\")\n#     print(\"=\"*40)\n\n# # Chạy script\n# process_merge()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !find /kaggle/input/covidx-cxr-dataset/rsna_png -maxdepth 1 -name \"*.png\" | wc -l","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-12T10:02:23.286509Z","iopub.execute_input":"2026-01-12T10:02:23.286979Z","iopub.status.idle":"2026-01-12T10:03:09.769333Z","shell.execute_reply.started":"2026-01-12T10:02:23.286946Z","shell.execute_reply":"2026-01-12T10:03:09.768368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Liệt kê các thư mục để tìm xem rsna_png nằm ở đâu\n# !find /kaggle/input/covidx-cxr-dataset -maxdepth 4 -type d","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-12T10:09:12.435144Z","iopub.execute_input":"2026-01-12T10:09:12.435669Z","iopub.status.idle":"2026-01-12T10:09:27.420139Z","shell.execute_reply.started":"2026-01-12T10:09:12.435630Z","shell.execute_reply":"2026-01-12T10:09:27.418880Z"}},"outputs":[],"execution_count":null}]}