{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042},{"sourceType":"datasetVersion","sourceId":2822650,"datasetId":1715304,"databundleVersionId":2869088},{"sourceType":"datasetVersion","sourceId":9350898,"datasetId":5668240,"databundleVersionId":9548994},{"sourceType":"datasetVersion","sourceId":4700752,"datasetId":2719985,"databundleVersionId":4763192},{"sourceType":"datasetVersion","sourceId":104884,"datasetId":54339,"databundleVersionId":111874},{"sourceType":"datasetVersion","sourceId":6785866,"datasetId":3904493,"databundleVersionId":6870842}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-10T05:38:36.022132Z","iopub.execute_input":"2026-05-10T05:38:36.022930Z","iopub.status.idle":"2026-05-10T05:38:37.277982Z","shell.execute_reply.started":"2026-05-10T05:38:36.022825Z","shell.execute_reply":"2026-05-10T05:38:37.276936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# MedImageGuard Kaggle Full Runner v2\n# Research only. Not for clinical diagnostic use.\n#\n# This version:\n# - Uses ONE repo folder\n# - Finds HAM10000 images more aggressively\n# - Builds/seals manifests only when images exist\n# - Runs reports/figures\n# - Prints exactly what worked and what is missing\n# ============================================================\n\nimport os\nimport sys\nimport json\nimport shutil\nimport subprocess\nfrom pathlib import Path\n\nGITHUB_REPO_URL = \"https://github.com/aaditmehtacoder/medimageguard-agent.git\"\n\nREPO = Path(\"/kaggle/working/medimageguard_agent_repo\")\nDUPLICATE_REPO = Path(\"/kaggle/working/medimageguard-agent\")\nKAGGLE_INPUT = Path(\"/kaggle/input\")\n\ndef run(cmd, check=False):\n    print(\"\\n>>>\", cmd)\n    result = subprocess.run(cmd, shell=True)\n    if check and result.returncode != 0:\n        raise RuntimeError(f\"Command failed: {cmd}\")\n    return result.returncode\n\ndef safe_remove(path):\n    path = Path(path)\n    if path.exists():\n        print(\"Removing:\", path)\n        shutil.rmtree(path)\n\ndef safe_link(src, dst):\n    if src is None:\n        return False\n    src = Path(src)\n    dst = Path(dst)\n    dst.parent.mkdir(parents=True, exist_ok=True)\n\n    if dst.exists() or dst.is_symlink():\n        try:\n            if dst.is_symlink() or dst.is_file():\n                dst.unlink()\n            elif dst.is_dir():\n                shutil.rmtree(dst)\n        except Exception:\n            pass\n\n    try:\n        os.symlink(src, dst, target_is_directory=src.is_dir())\n        print(\"Linked:\", dst, \"->\", src)\n    except Exception:\n        if src.is_dir():\n            shutil.copytree(src, dst, dirs_exist_ok=True)\n        else:\n            shutil.copy2(src, dst)\n        print(\"Copied:\", src, \"->\", dst)\n\n    return True\n\ndef find_files_by_name(name):\n    return list(KAGGLE_INPUT.rglob(name))\n\ndef find_first_file(name):\n    matches = find_files_by_name(name)\n    return matches[0] if matches else None\n\ndef find_dirs_by_name(name):\n    return [p for p in KAGGLE_INPUT.rglob(name) if p.is_dir()]\n\ndef find_first_dir(name):\n    matches = find_dirs_by_name(name)\n    return matches[0] if matches else None\n\ndef count_images(path):\n    path = Path(path)\n    if not path.exists():\n        return 0\n    return sum(1 for p in path.rglob(\"*\") if p.suffix.lower() in [\".jpg\", \".jpeg\", \".png\", \".tif\", \".tiff\"])\n\n# ----------------------------\n# 1. Clean duplicate repo and clone/update main repo\n# ----------------------------\nsafe_remove(DUPLICATE_REPO)\n\nif not REPO.exists():\n    run(f\"git clone {GITHUB_REPO_URL} {REPO}\", check=True)\nelse:\n    print(\"Repo already exists:\", REPO)\n\nos.chdir(REPO)\nprint(\"Working directory:\", Path.cwd())\n\n# ----------------------------\n# 2. Install requirements\n# ----------------------------\nrun(f\"{sys.executable} -m pip install -q -r requirements.txt\", check=False)\nrun(f\"{sys.executable} -m pip install -q torchvision pydicom kaggle openpyxl matplotlib\", check=False)\n\n# ----------------------------\n# 3. Show Kaggle input overview\n# ----------------------------\nprint(\"\\nKaggle input folders:\")\nrun(\"find /kaggle/input -maxdepth 3 -type d | sort | head -200\", check=False)\n\nprint(\"\\nSample Kaggle input files:\")\nrun(\"find /kaggle/input -maxdepth 5 -type f | sort | head -200\", check=False)\n\n# ----------------------------\n# 4. Reset local raw/artifact folders for clean rerun\n# ----------------------------\nsafe_remove(REPO / \"data\" / \"raw\")\nsafe_remove(REPO / \"data\" / \"processed\")\nsafe_remove(REPO / \"artifacts\")\n\nRAW = REPO / \"data\" / \"raw\"\nRAW.mkdir(parents=True, exist_ok=True)\n\n# ----------------------------\n# 5. Link HAM10000 robustly\n# ----------------------------\nham_root = RAW / \"ham10000\"\nham_root.mkdir(parents=True, exist_ok=True)\n\nham_metadata = find_first_file(\"HAM10000_metadata.csv\")\nsafe_link(ham_metadata, ham_root / \"HAM10000_metadata.csv\")\n\n# Common folder names\nham_image_dirs = []\nfor name in [\n    \"HAM10000_images_part_1\",\n    \"HAM10000_images_part_2\",\n    \"ham10000_images_part_1\",\n    \"ham10000_images_part_2\",\n]:\n    ham_image_dirs.extend(find_dirs_by_name(name))\n\n# If folder names are different, infer folders containing many ISIC jpgs\nfor possible in KAGGLE_INPUT.rglob(\"*\"):\n    if possible.is_dir():\n        try:\n            sample_count = len(list(possible.glob(\"ISIC_*.jpg\"))) + len(list(possible.glob(\"ISIC_*.png\")))\n            if sample_count > 20:\n                ham_image_dirs.append(possible)\n        except Exception:\n            pass\n\n# De-duplicate\nseen = set()\nunique_ham_dirs = []\nfor d in ham_image_dirs:\n    if str(d) not in seen:\n        seen.add(str(d))\n        unique_ham_dirs.append(d)\n\nfor i, src in enumerate(unique_ham_dirs, start=1):\n    safe_link(src, ham_root / src.name)\n\nprint(\"\\nHAM10000 metadata:\", ham_metadata)\nprint(\"HAM10000 image dirs found:\", [str(p) for p in unique_ham_dirs])\nprint(\"HAM10000 linked image count:\", count_images(ham_root))\n\n# ----------------------------\n# 6. Link APTOS 2019 if attached\n# ----------------------------\naptos_root = RAW / \"aptos2019\"\naptos_root.mkdir(parents=True, exist_ok=True)\n\naptos_train_csv_candidates = [\n    p for p in KAGGLE_INPUT.rglob(\"train.csv\")\n    if \"aptos\" in str(p).lower() or \"blindness\" in str(p).lower()\n]\naptos_train_csv = aptos_train_csv_candidates[0] if aptos_train_csv_candidates else find_first_file(\"train.csv\")\n\naptos_train_images = None\nfor d in find_dirs_by_name(\"train_images\"):\n    if \"aptos\" in str(d).lower() or \"blindness\" in str(d).lower():\n        aptos_train_images = d\n        break\nif aptos_train_images is None:\n    aptos_train_images = find_first_dir(\"train_images\")\n\nsafe_link(aptos_train_csv, aptos_root / \"train.csv\")\nsafe_link(aptos_train_images, aptos_root / \"train_images\")\n\nprint(\"\\nAPTOS train csv:\", aptos_train_csv)\nprint(\"APTOS train_images:\", aptos_train_images)\nprint(\"APTOS linked image count:\", count_images(aptos_root))\n\n# ----------------------------\n# 7. Link NIH ChestX-ray14 if attached\n# ----------------------------\nnih_root = RAW / \"nih_chestxray14\"\nnih_root.mkdir(parents=True, exist_ok=True)\n\nnih_metadata = find_first_file(\"Data_Entry_2017.csv\")\nsafe_link(nih_metadata, nih_root / \"Data_Entry_2017.csv\")\n\nnih_image_dirs = []\nfor p in KAGGLE_INPUT.rglob(\"*\"):\n    if p.is_dir() and (\"nih\" in str(p).lower() or \"chest\" in str(p).lower() or p.name.startswith(\"images\")):\n        img_count = count_images(p)\n        if img_count > 20:\n            nih_image_dirs.append(p)\n\nseen = set()\nfor src in nih_image_dirs:\n    if str(src) in seen:\n        continue\n    seen.add(str(src))\n    safe_link(src, nih_root / src.name)\n\nprint(\"\\nNIH metadata:\", nih_metadata)\nprint(\"NIH image dirs:\", [str(p) for p in nih_image_dirs[:10]])\nprint(\"NIH linked image count:\", count_images(nih_root))\n\n# ----------------------------\n# 8. Link RSNA Pneumonia if attached\n# ----------------------------\nrsna_root = RAW / \"rsna_pneumonia\"\nrsna_root.mkdir(parents=True, exist_ok=True)\n\nrsna_labels = find_first_file(\"stage_2_train_labels.csv\")\nrsna_info = find_first_file(\"stage_2_detailed_class_info.csv\")\nrsna_images = find_first_dir(\"stage_2_train_images\")\n\nsafe_link(rsna_labels, rsna_root / \"stage_2_train_labels.csv\")\nsafe_link(rsna_info, rsna_root / \"stage_2_detailed_class_info.csv\")\nsafe_link(rsna_images, rsna_root / \"stage_2_train_images\")\n\nprint(\"\\nRSNA labels:\", rsna_labels)\nprint(\"RSNA images:\", rsna_images)\nprint(\"RSNA DICOM count:\", len(list((rsna_root / 'stage_2_train_images').rglob('*.dcm'))) if (rsna_root / 'stage_2_train_images').exists() else 0)\n\n# ----------------------------\n# 9. Prepare manifests\n# ----------------------------\nenv = \"PYTHONPATH=src\"\n\n# HAM10000\nrun(\n    f\"{env} {sys.executable} scripts/prepare_ham10000.py \"\n    \"--metadata-csv data/raw/ham10000/HAM10000_metadata.csv \"\n    \"--images-root data/raw/ham10000 \"\n    \"--output-csv artifacts/manifests/ham10000.csv\",\n    check=False,\n)\n\n# RSNA can be slow; only run if DICOM folder exists\nif rsna_labels and rsna_images:\n    run(\n        f\"{env} {sys.executable} scripts/prepare_rsna_pneumonia.py \"\n        \"--labels-csv data/raw/rsna_pneumonia/stage_2_train_labels.csv \"\n        \"--images-root data/raw/rsna_pneumonia/stage_2_train_images \"\n        \"--png-output-dir data/processed/rsna_pneumonia_png \"\n        \"--output-csv artifacts/manifests/rsna_pneumonia.csv\",\n        check=False,\n    )\nelse:\n    print(\"Skipping RSNA conversion: labels/images not found.\")\n\n# APTOS / NIH\nrun(f\"{env} {sys.executable} scripts/prepare_kaggle_available_datasets.py --audit\", check=False)\n\n# Recommended optional public datasets\nrun(f\"{env} {sys.executable} scripts/prepare_recommended_public_datasets.py --audit\", check=False)\n\n# ----------------------------\n# 10. Validate and seal non-empty manifests\n# ----------------------------\nmanifest_dir = Path(\"artifacts/manifests\")\nmanifest_dir.mkdir(parents=True, exist_ok=True)\n\nprint(\"\\nManifest row counts:\")\nfor manifest in sorted(manifest_dir.glob(\"*.csv\")):\n    try:\n        import pandas as pd\n        rows = len(pd.read_csv(manifest))\n    except Exception:\n        rows = -1\n\n    print(manifest, \"rows:\", rows)\n\n    if rows > 0:\n        seal = manifest.with_suffix(\".seal.json\")\n        run(f\"{env} {sys.executable} -m medimageguard.cli validate-manifest {manifest}\", check=False)\n        run(f\"{env} {sys.executable} -m medimageguard.cli seal-manifest {manifest} {seal}\", check=False)\n    else:\n        print(\"Skipping seal for empty manifest:\", manifest)\n\n# ----------------------------\n# 11. Run tests\n# ----------------------------\nrun(f\"{env} {sys.executable} -m pytest\", check=False)\n\n# ----------------------------\n# 12. Generate figure reports\n# ----------------------------\nrun(f\"{env} {sys.executable} scripts/generate_manuscript_figures.py --output-dir artifacts/figures\", check=False)\n\n# ----------------------------\n# 13. Print final summary\n# ----------------------------\nprint(\"\\nFinal artifacts:\")\nrun(\"find artifacts -maxdepth 4 -type f | sort\", check=False)\n\nsummary = Path(\"artifacts/figures/manuscript_figures_summary.json\")\nprint(\"\\nFigure summary:\")\nprint(summary.read_text() if summary.exists() else \"NO SUMMARY FOUND\")\n\nprep = Path(\"artifacts/reports/kaggle_available_dataset_preparation.json\")\nprint(\"\\nKaggle dataset preparation report:\")\nprint(prep.read_text()[:8000] if prep.exists() else \"NO DATASET PREP REPORT FOUND\")\n\n# ----------------------------\n# 14. Zip final outputs\n# ----------------------------\nrun(\n    \"zip -r medimageguard_results.zip \"\n    \"artifacts docs \"\n    \"-x '*.pt' '*.dcm' '*.zip' || true\",\n    check=False,\n)\n\nprint(\"\\nDONE.\")\nprint(\"Download:\")\nprint(\"/kaggle/working/medimageguard_agent_repo/medimageguard_results.zip\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T05:39:44.507112Z","iopub.execute_input":"2026-05-10T05:39:44.507396Z"}},"outputs":[],"execution_count":null}]}