{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042},{"sourceType":"datasetVersion","sourceId":23812,"datasetId":17810,"databundleVersionId":23851},{"sourceType":"datasetVersion","sourceId":1493513,"datasetId":876960,"databundleVersionId":1527499},{"sourceType":"datasetVersion","sourceId":6717213,"datasetId":1317048,"databundleVersionId":6801677},{"sourceType":"datasetVersion","sourceId":18613,"datasetId":5839,"databundleVersionId":18613},{"sourceType":"datasetVersion","sourceId":15374689,"datasetId":9834806,"databundleVersionId":16286847}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# --- Imports ---\nimport os\nimport time\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport pydicom\nfrom pathlib import Path\nfrom PIL import Image\n\nimport torch\nimport torchvision.transforms as T\nfrom torch.utils.data import Dataset, DataLoader\n\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:09:27.284255Z","iopub.execute_input":"2026-03-26T14:09:27.284545Z","iopub.status.idle":"2026-03-26T14:09:40.199065Z","shell.execute_reply.started":"2026-03-26T14:09:27.284523Z","shell.execute_reply":"2026-03-26T14:09:40.198497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Environment Checks & Config ---\ncuda_ok = torch.cuda.is_available()\nDEVICE = torch.device(\"cuda\" if cuda_ok else \"cpu\")\n\nprint(f\"PyTorch : {torch.__version__}\")\nprint(f\"Device  : {DEVICE}\")\nif cuda_ok:\n    print(f\"GPU     : {torch.cuda.get_device_name(0)}\")\n\n# Constants\nPARQUET_PATH = \"/kaggle/input/datasets/rohanpesuecbtech2023/train-parquet\"\nOUTPUT_DIR = Path(\"/kaggle/working/outputs\")\nDCM_OUT_DIR = Path(\"/kaggle/working/converted_pngs\")\n\nSAMPLES_PER_CLASS = 1000\nRANDOM_STATE = 42\nBATCH_SIZE = 64\nNUM_WORKERS = 2\n\n# Ensure writable directories exist\nOUTPUT_DIR.mkdir(parents=True, exist_ok=True)\nDCM_OUT_DIR.mkdir(parents=True, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:09:40.200328Z","iopub.execute_input":"2026-03-26T14:09:40.200666Z","iopub.status.idle":"2026-03-26T14:09:40.500211Z","shell.execute_reply.started":"2026-03-26T14:09:40.200645Z","shell.execute_reply":"2026-03-26T14:09:40.499499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Path Checks & Mapping Logic ---\nDATASET_ROOTS = {\n    \"CheXpert\": \"/kaggle/input/datasets/willarevalo/chexpert-v10-small/CheXpert-v1.0-small\",\n    \"Pediatric\": \"/kaggle/input/datasets/paultimothymooney/chest-xray-pneumonia/chest_xray\",\n    \"NIH\": \"/kaggle/input/datasets/organizations/nih-chest-xrays/data\",\n    \"COVIDx\": \"/kaggle/input/datasets/andyczhao/covidx-cxr2\",\n    \"RSNA\": \"/kaggle/input/competitions/rsna-pneumonia-detection-challenge/stage_2_train_images\"\n}\n\ndef resolve_path(row):\n    \"\"\"Maps relative parquet paths to absolute Kaggle dataset paths.\"\"\"\n    root = DATASET_ROOTS.get(row[\"dataset\"])\n    if root is None:\n        return row[\"path\"]\n    \n    p = row[\"path\"]\n    if row[\"dataset\"] == \"RSNA\":\n        filename = Path(p).stem + \".dcm\"\n        return str(Path(root) / filename)\n        \n    return str(Path(root) / p)\n\nprint(\"Dataset root mapping defined. Ready for sampling.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:12:23.980981Z","iopub.execute_input":"2026-03-26T14:12:23.981630Z","iopub.status.idle":"2026-03-26T14:12:23.986969Z","shell.execute_reply.started":"2026-03-26T14:12:23.981599Z","shell.execute_reply":"2026-03-26T14:12:23.986225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_raw = pd.read_parquet(PARQUET_PATH, columns=[\"path\", \"disease\", \"dataset\"])\n\nprint(f\"[ INSPECT ]  Raw parquet\")\nprint(f\"  Total rows : {len(df_raw):,}\")\nprint(f\"\\n  Disease counts:\\n{df_raw['disease'].value_counts().to_string()}\")\nprint(f\"\\n  Dataset counts:\\n{df_raw['dataset'].value_counts().to_string()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:22:09.199128Z","iopub.execute_input":"2026-03-26T14:22:09.199460Z","iopub.status.idle":"2026-03-26T14:22:09.261681Z","shell.execute_reply.started":"2026-03-26T14:22:09.199433Z","shell.execute_reply":"2026-03-26T14:22:09.260902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_raw[\"path\"] = df_raw.apply(resolve_path, axis=1)\n\nprint(\"[ PATH CHECK ]  5 random samples per dataset\\n\")\nall_ok = True\nfor ds, grp in df_raw.groupby(\"dataset\"):\n    sample  = grp.sample(n=min(5, len(grp)), random_state=RANDOM_STATE)\n    missing = [p for p in sample[\"path\"] if not Path(p).exists()]\n    status  = \"OK\" if not missing else f\"FAIL  ({len(missing)} missing)\"\n    print(f\"  {ds:<12}  {status}\")\n    for p in missing:\n        print(f\"               {p}\")\n    all_ok = all_ok and not missing\n\nprint()\nprint(f\"  Result : {'all paths resolved' if all_ok else 'fix DATASET_ROOTS in Cell 3 then re-run'}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:23:36.716665Z","iopub.execute_input":"2026-03-26T14:23:36.717221Z","iopub.status.idle":"2026-03-26T14:23:38.663162Z","shell.execute_reply.started":"2026-03-26T14:23:36.717192Z","shell.execute_reply":"2026-03-26T14:23:38.662549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- PHASE 1: Data Sampling ---\nprint(\"[ PHASE 1 ] Data Sampling\")\n\ndf_raw = pd.read_parquet(PARQUET_PATH, columns=[\"path\", \"disease\", \"dataset\"])\ndf_raw[\"path\"] = df_raw.apply(resolve_path, axis=1)\n\ndf_sample = (\n    df_raw\n    .groupby(\"disease\", group_keys=False)\n    .apply(lambda g: g.sample(n=min(len(g), SAMPLES_PER_CLASS), random_state=RANDOM_STATE))\n    .reset_index(drop=True)\n)\n\nprint(f\"Total sampled rows : {len(df_sample)}\")\nprint(f\"Class distribution :\\n{df_sample['disease'].value_counts().to_string()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:12:28.666841Z","iopub.execute_input":"2026-03-26T14:12:28.667398Z","iopub.status.idle":"2026-03-26T14:12:30.513348Z","shell.execute_reply.started":"2026-03-26T14:12:28.667366Z","shell.execute_reply":"2026-03-26T14:12:30.512593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- PHASE 1B: Path Verification ---\nprint(\"[ PHASE 1B ] Verifying file paths on disk\")\n\ndef check_paths(df):\n    missing_paths = []\n    valid_indices = []\n    \n    for idx, row in df.iterrows():\n        if Path(row[\"path\"]).exists():\n            valid_indices.append(idx)\n        else:\n            missing_paths.append(row[\"path\"])\n            \n    # Filter the dataframe to only include valid paths\n    df_clean = df.loc[valid_indices].reset_index(drop=True)\n    \n    return df_clean, missing_paths\n\n# Run the check on the sampled dataframe\ndf_sample, missing = check_paths(df_sample)\n\nprint(f\"Total paths checked : {len(df_sample) + len(missing)}\")\nprint(f\"Valid paths found   : {len(df_sample)}\")\n\nif missing:\n    print(f\"WARNING: Found {len(missing)} missing files. They have been removed from the sample.\")\n    print(\"Sample of missing paths:\")\n    for p in missing[:5]:\n        print(f\"  - {p}\")\nelse:\n    print(\"All sampled paths exist on disk.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:12:37.296469Z","iopub.execute_input":"2026-03-26T14:12:37.297200Z","iopub.status.idle":"2026-03-26T14:12:52.475606Z","shell.execute_reply.started":"2026-03-26T14:12:37.297144Z","shell.execute_reply":"2026-03-26T14:12:52.474779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- PHASE 3: Image Transformation Pipeline ---\nprint(\"[ PHASE 3 ] Defining Transformations\")\n\n# Strict constraints applied: Grayscale -> 3-channel -> Resize -> Tensor -> Normalize\ntransform = T.Compose([\n    T.Grayscale(num_output_channels=3),\n    T.Resize((224, 224)),\n    T.ToTensor(),\n    T.Normalize(\n        mean=[0.485, 0.456, 0.406],\n        std =[0.229, 0.224, 0.225],\n    ),\n])\n\nprint(\"Transform pipeline ready.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:13:18.991398Z","iopub.execute_input":"2026-03-26T14:13:18.991915Z","iopub.status.idle":"2026-03-26T14:13:18.996916Z","shell.execute_reply.started":"2026-03-26T14:13:18.991886Z","shell.execute_reply":"2026-03-26T14:13:18.996188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- PHASE 2: Dataset + Loading Logic ---\nprint(\"[ PHASE 2 ] Initializing Dataset and DataLoader\")\n\ndef safe_dcm_to_png(dcm_path):\n    \"\"\"Safely converts DICOM to PNG in a writable directory to prevent Kaggle I/O errors.\"\"\"\n    filename = Path(dcm_path).stem + \".png\"\n    out_path = DCM_OUT_DIR / filename\n    \n    if out_path.exists():\n        return Image.open(out_path)\n        \n    ds = pydicom.dcmread(dcm_path)\n    arr = ds.pixel_array.astype(np.float32)\n    arr = (arr - arr.min()) / (arr.max() - arr.min() + 1e-8) * 255.0\n    \n    img = Image.fromarray(arr.astype(np.uint8))\n    img.save(out_path)\n    return img\n\nclass ChestXrayDataset(Dataset):\n    def __init__(self, df, transform):\n        self.paths = df[\"path\"].tolist()\n        self.diseases = df[\"disease\"].tolist()\n        self.datasets = df[\"dataset\"].tolist()\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.paths)\n\n    def __getitem__(self, idx):\n        path = self.paths[idx]\n        try:\n            if path.endswith(\".dcm\"):\n                img = safe_dcm_to_png(path)\n            else:\n                img = Image.open(path)\n                \n            # Force standard PIL format before transforms\n            img = img.convert(\"L\")\n            tensor_img = self.transform(img)\n            \n        except Exception as e:\n            print(f\"Failed to load {path} - Error: {e}\")\n            return None\n            \n        return {\n            \"image\": tensor_img,\n            \"disease\": self.diseases[idx],\n            \"dataset\": self.datasets[idx],\n            \"path\": path,\n        }\n\ndef safe_collate(batch):\n    batch = [b for b in batch if b is not None]\n    if not batch:\n        return None\n    return {\n        \"image\": torch.stack([b[\"image\"] for b in batch]),\n        \"disease\": [b[\"disease\"] for b in batch],\n        \"dataset\": [b[\"dataset\"] for b in batch],\n        \"path\": [b[\"path\"] for b in batch],\n    }\n\ndataset = ChestXrayDataset(df_sample, transform)\ndataloader = DataLoader(\n    dataset,\n    batch_size=BATCH_SIZE,\n    shuffle=False,\n    num_workers=NUM_WORKERS,\n    collate_fn=safe_collate,\n    pin_memory=True\n)\n\nprint(f\"Dataset size: {len(dataset)}\")\nprint(f\"Batches: {len(dataloader)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:13:22.985510Z","iopub.execute_input":"2026-03-26T14:13:22.985771Z","iopub.status.idle":"2026-03-26T14:13:22.997216Z","shell.execute_reply.started":"2026-03-26T14:13:22.985751Z","shell.execute_reply":"2026-03-26T14:13:22.996653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- PHASE 4: Model Setup (DINOv2) ---\nprint(\"[ PHASE 4 ] Loading DINOv2 ViT-S/14\")\n\nmodel = torch.hub.load(\"facebookresearch/dinov2\", \"dinov2_vits14\", pretrained=True, verbose=False)\nmodel.eval()\n\n# Disable gradients for feature extraction\nfor param in model.parameters():\n    param.requires_grad = False\n    \nmodel.to(DEVICE)\n\nwith torch.no_grad():\n    dummy = torch.zeros(1, 3, 224, 224, device=DEVICE)\n    emb_dim = model(dummy).shape[-1]\n\nprint(f\"Model loaded in eval mode on {DEVICE}.\")\nprint(f\"Expected embedding dimension: {emb_dim}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:13:28.199404Z","iopub.execute_input":"2026-03-26T14:13:28.199782Z","iopub.status.idle":"2026-03-26T14:13:28.720831Z","shell.execute_reply.started":"2026-03-26T14:13:28.199754Z","shell.execute_reply":"2026-03-26T14:13:28.720211Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- PHASE 5: Feature Extraction ---\nprint(\"[ PHASE 5 ] Extracting Embeddings\")\n\nall_features = []\nall_diseases = []\nall_datasets = []\nall_paths = []\n\ntotal_batches = len(dataloader)\nt0 = time.time()\n\nwith torch.no_grad():\n    for i, batch in enumerate(dataloader, 1):\n        if batch is None:\n            continue\n            \n        images = batch[\"image\"].to(DEVICE, non_blocking=True)\n        embeddings = model(images)\n        \n        all_features.append(embeddings.cpu().numpy())\n        all_diseases.extend(batch[\"disease\"])\n        all_datasets.extend(batch[\"dataset\"])\n        all_paths.extend(batch[\"path\"])\n        \n        if i % 5 == 0 or i == total_batches:\n            elapsed = time.time() - t0\n            n_done = sum(len(f) for f in all_features)\n            print(f\"Processed Batch {i:>3}/{total_batches} | Samples: {n_done:>4} | Elapsed: {elapsed/60:.1f} min\", end=\"\\r\")\n\nprint(\"\\n\\nExtraction complete.\")\nfeatures = np.concatenate(all_features, axis=0)\nlabels_arr = np.array(all_diseases)\ndatasets_arr = np.array(all_datasets)\npaths_arr = np.array(all_paths)\n\nprint(f\"Final Features Shape: {features.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:13:34.924635Z","iopub.execute_input":"2026-03-26T14:13:34.925243Z","iopub.status.idle":"2026-03-26T14:14:39.445717Z","shell.execute_reply.started":"2026-03-26T14:13:34.925209Z","shell.execute_reply":"2026-03-26T14:14:39.444979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- PHASE 6: Feature Storage ---\nprint(\"[ PHASE 6 ] Saving Features to Disk\")\n\nnp.save(OUTPUT_DIR / \"features.npy\", features)\nnp.save(OUTPUT_DIR / \"labels.npy\", labels_arr)\nnp.save(OUTPUT_DIR / \"datasets.npy\", datasets_arr)\nnp.save(OUTPUT_DIR / \"paths.npy\", paths_arr)\n\ntotal_mb = sum(f.stat().st_size for f in OUTPUT_DIR.glob(\"*.npy\")) / 1e6\nprint(f\"Saved arrays to {OUTPUT_DIR}\")\nprint(f\"Total storage size: {total_mb:.1f} MB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:14:59.199731Z","iopub.execute_input":"2026-03-26T14:14:59.200058Z","iopub.status.idle":"2026-03-26T14:14:59.213414Z","shell.execute_reply.started":"2026-03-26T14:14:59.200018Z","shell.execute_reply":"2026-03-26T14:14:59.212786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- PHASE 7: Validation Checks ---\nprint(\"[ PHASE 7 ] Validating Outputs\")\n\nN, D = features.shape\n\nchecks = [\n    (\"Features dimension matches DINOv2 output\", D == 384),\n    (\"Labels align with features\", labels_arr.shape[0] == N),\n    (\"Datasets align with features\", datasets_arr.shape[0] == N),\n    (\"No NaN values in features\", not np.isnan(features).any()),\n    (\"No Inf values in features\", not np.isinf(features).any()),\n    (\"Samples extracted > 0\", N > 0)\n]\n\nall_passed = True\nfor name, passed in checks:\n    status = \"PASS\" if passed else \"FAIL\"\n    print(f\"  [{status}] {name}\")\n    if not passed:\n        all_passed = False\n\nif all_passed:\n    print(\"\\nAll validation checks passed. Ready for dimensionality reduction (Phase 8).\")\nelse:\n    print(\"\\nWARNING: One or more validation checks failed.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:15:03.212848Z","iopub.execute_input":"2026-03-26T14:15:03.213658Z","iopub.status.idle":"2026-03-26T14:15:03.220819Z","shell.execute_reply.started":"2026-03-26T14:15:03.213625Z","shell.execute_reply":"2026-03-26T14:15:03.220066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- FULL DATASET EXTRACTION ---\nprint(\"[ FULL SCALE ] Verifying all paths in the raw dataset\")\n\n# 1. Clean the entire raw dataframe (using the check_paths function from Phase 1B)\ndf_full_clean, missing_full = check_paths(df_raw)\nprint(f\"Total original rows : {len(df_raw)}\")\nprint(f\"Valid paths found   : {len(df_full_clean)}\")\nif missing_full:\n    print(f\"Ignored {len(missing_full)} missing files.\")\n\n# 2. Initialize the Full Dataset and DataLoader\nprint(\"\\n[ FULL SCALE ] Initializing Full DataLoader\")\nfull_dataset = ChestXrayDataset(df_full_clean, transform)\nfull_dataloader = DataLoader(\n    full_dataset,\n    batch_size=BATCH_SIZE,\n    shuffle=False,\n    num_workers=NUM_WORKERS,\n    collate_fn=safe_collate,\n    pin_memory=True\n)\nprint(f\"Total Batches to process: {len(full_dataloader)}\")\n\n# 3. Extract Embeddings\nprint(\"\\n[ FULL SCALE ] Extracting Embeddings\")\nall_features_full = []\nall_diseases_full = []\nall_datasets_full = []\nall_paths_full = []\n\ntotal_batches = len(full_dataloader)\nt0 = time.time()\n\nwith torch.no_grad():\n    for i, batch in enumerate(full_dataloader, 1):\n        if batch is None:\n            continue\n            \n        images = batch[\"image\"].to(DEVICE, non_blocking=True)\n        embeddings = model(images)\n        \n        all_features_full.append(embeddings.cpu().numpy())\n        all_diseases_full.extend(batch[\"disease\"])\n        all_datasets_full.extend(batch[\"dataset\"])\n        all_paths_full.extend(batch[\"path\"])\n        \n        # Print update every 50 batches for a cleaner output on long runs\n        if i % 50 == 0 or i == total_batches:\n            elapsed = time.time() - t0\n            n_done = sum(len(f) for f in all_features_full)\n            eta = (elapsed / i) * (total_batches - i)\n            print(f\"Processed Batch {i:>4}/{total_batches} | Samples: {n_done:>6} | Elapsed: {elapsed/60:.1f}m | ETA: {eta/60:.1f}m\", end=\"\\r\")\n\nprint(\"\\n\\n[ FULL SCALE ] Extraction complete.\")\nfeatures_full = np.concatenate(all_features_full, axis=0)\n\n# 4. Save the massive feature banks\nprint(\"[ FULL SCALE ] Saving massive arrays to disk\")\nnp.save(OUTPUT_DIR / \"features_full.npy\", features_full)\nnp.save(OUTPUT_DIR / \"labels_full.npy\", np.array(all_diseases_full))\nnp.save(OUTPUT_DIR / \"datasets_full.npy\", np.array(all_datasets_full))\nnp.save(OUTPUT_DIR / \"paths_full.npy\", np.array(all_paths_full))\n\ntotal_mb = sum(f.stat().st_size for f in OUTPUT_DIR.glob(\"*_full.npy\")) / 1e6\nprint(f\"Saved full arrays to {OUTPUT_DIR}\")\nprint(f\"Total storage size: {total_mb:.1f} MB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-26T14:26:35.555418Z","iopub.execute_input":"2026-03-26T14:26:35.555743Z","iopub.status.idle":"2026-03-26T15:11:28.090360Z","shell.execute_reply.started":"2026-03-26T14:26:35.555718Z","shell.execute_reply":"2026-03-26T15:11:28.089479Z"}},"outputs":[],"execution_count":null}]}