{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install ultralytics timm torchxrayvision pydicom\n\nimport os\nimport shutil\nimport glob\n\nprint(\"\\n--- Searching for uploaded code ---\")\nsource_dir = None\nfor path in glob.glob('/kaggle/input/**/data_prep.py', recursive=True):\n    source_dir = os.path.dirname(path)\n    break\n\nif source_dir:\n    print(f\"✅ Found code at: {source_dir}\")\n    dest_dir = '/kaggle/working/pipeline-code'\n    \n    if os.path.exists(dest_dir):\n        shutil.rmtree(dest_dir)\n        \n    shutil.copytree(source_dir, dest_dir)\n    print(\"✅ Workspace copied to /kaggle/working/pipeline-code and ready!\")\n    \n    # --- FIXES FOR EVALUATION ---\n    eval_path = '/kaggle/working/pipeline-code/evaluate.py'\n    with open(eval_path, 'r') as f:\n        eval_code = f.read()\n    eval_code = eval_code.replace('torch.load(ckpt_path, map_location=device)', \n                                  'torch.load(ckpt_path, map_location=device, weights_only=False)')\n    eval_code = eval_code.replace('model.load_state_dict(ckpt[\"state_dict\"])', \n                                  'model.load_state_dict(ckpt[\"state_dict\"], strict=False)')\n    with open(eval_path, 'w') as f:\n        f.write(eval_code)\n    print(\"✅ evaluate.py patched successfully!\")\nelse:\n    print(\"❌ FATAL: Could not find data_prep.py. Attach your dataset zip.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-22T07:26:51.782994Z","iopub.execute_input":"2026-07-22T07:26:51.783783Z","iopub.status.idle":"2026-07-22T07:27:43.448132Z","shell.execute_reply.started":"2026-07-22T07:26:51.783744Z","shell.execute_reply":"2026-07-22T07:27:43.447125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%cd /kaggle/working/pipeline-code\n!python data_prep.py","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%cd /kaggle/working/pipeline-code\n!python train_classifier.py --backbone xrv_densenet121 --seed 42\n!python train_classifier.py --backbone densenet121 --seed 42","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CELL 3.5 (NEW): AGGRESSIVE DISK CLEANUP\nimport pandas as pd\nimport os\n\nprint(\"\\n--- Freeing up Disk Space for YOLO ---\")\nsplits_path = '/kaggle/working/pneumonia_pipeline/splits.csv'\n\ntry:\n    df = pd.read_csv(splits_path)\n    \n    # Identify image path column\n    col_name = 'image_path' if 'image_path' in df.columns else 'path'\n    \n    # Filter out all images that are NOT in the test set\n    train_val_df = df[df['split'] != 'test']\n    \n    deleted = 0\n    for img_path in train_val_df[col_name]:\n        if os.path.exists(img_path):\n            os.remove(img_path)\n            deleted += 1\n            \n    print(f\"✅ Deleted {deleted} redundant PNGs from the classifier directory!\")\n    print(\"✅ ~8+ GB of disk space freed. YOLO is ready for 60 epochs!\")\nexcept Exception as e:\n    print(f\"⚠️ Could not perform targeted cleanup: {e}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%cd /kaggle/working/pipeline-code\n!python train_yolo.py --epochs 60","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%cd /kaggle/working/pipeline-code\n!python evaluate.py --clf_ckpt /kaggle/working/pneumonia_pipeline/checkpoints/clf_xrv_densenet121_seed42_best.pth --yolo_ckpt /kaggle/working/pneumonia_pipeline/results/yolo_runs/yolov8n_e60_seed42/weights/best.pt\n\nprint(\"\\n--- Final Cleanup Before Zipping ---\")\nimport shutil\nimport os\n\n# Delete the YOLO image dataset (we don't need it in the final zip, saving ~4GB+)\nyolo_data_path = '/kaggle/working/pneumonia_pipeline/yolo_dataset'\nif os.path.exists(yolo_data_path):\n    shutil.rmtree(yolo_data_path)\n    print(\"✅ Deleted YOLO dataset images to free space for the zip file.\")\n\n%cd /kaggle/working\nprint(\"Zipping results...\")\nshutil.make_archive('/kaggle/working/final_paper_results', 'zip', '/kaggle/working/pneumonia_pipeline')\nprint(\"✅ Full training complete! Download your final_paper_results.zip\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}