{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":10338,"databundleVersionId":862042},{"sourceType":"datasetVersion","sourceId":15406443,"datasetId":9855746,"databundleVersionId":16322125},{"sourceType":"datasetVersion","sourceId":15400563,"datasetId":9852074,"databundleVersionId":16315483}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport sys\nimport pandas as pd\nimport torch\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader\nfrom sklearn.model_selection import train_test_split\n\n# ==========================================\n# 1. BULLETPROOF DYNAMIC PATH FINDER\n# ==========================================\nprint(\"🔍 Scanning Kaggle directories for required files...\")\n\nscript_dir = None\ncsv_path = None\nimage_dir = None\n\n# Search the entire /kaggle/input directory tree\nfor root, dirs, files in os.walk('/kaggle/input'):\n    # Find your custom code folder\n    if 'config.py' in files and script_dir is None:\n        script_dir = root\n        print(f\"✅ Found custom scripts at: {script_dir}\")\n    \n    # Find the RSNA dataset folder\n    if 'stage_2_train_labels.csv' in files and csv_path is None:\n        csv_path = os.path.join(root, 'stage_2_train_labels.csv')\n        image_dir = os.path.join(root, 'stage_2_train_images')\n        print(f\"✅ Found dataset CSV at: {csv_path}\")\n        print(f\"✅ Assuming images are at: {image_dir}\")\n\n# ==========================================\n# 2. SAFETY CHECKS & PATH INJECTION\n# ==========================================\nif not script_dir:\n    raise FileNotFoundError(\"❌ ERROR: Could not find 'config.py'. Did you upload your VS Code folder?\")\nif not csv_path:\n    raise FileNotFoundError(\"❌ ERROR: Could not find 'stage_2_train_labels.csv'. Did you click 'Add Data' and attach the RSNA Pneumonia dataset to this notebook?\")\n\n# Add scripts to system path so we can import them\nsys.path.insert(0, script_dir)\n\n# Import your custom modules\nimport config\nfrom data_pipeline.dataset import PneumoniaDataset\nfrom models.attention_unet import build_model\nfrom training.loss import PneumoniaLoss\nfrom training.trainer import train_one_epoch, validate_one_epoch\n\nprint(\"✅ All custom modules imported successfully!\")\n\n# ==========================================\n# 3. OVERRIDE CONFIG PATHS DYNAMICALLY\n# ==========================================\n# Overwrite config.py paths with the exact paths Kaggle just found\nconfig.TRAIN_CSV = csv_path\nconfig.IMAGE_DIR = image_dir\nconfig.OUTPUT_DIR = '/kaggle/working' # Kaggle requires saving to /kaggle/working\n\n# ==========================================\n# 4. DATA PREPARATION\n# ==========================================\nprint(\"\\n🚀 Loading and splitting data...\")\ndf = pd.read_csv(config.TRAIN_CSV)\n\n# 🛑 THE FIX: Fill NaNs for ALL bounding box columns so math doesn't crash on healthy lungs\ndf[['x', 'y', 'width', 'height']] = df[['x', 'y', 'width', 'height']].fillna(0)\n\ntrain_df, val_df = train_test_split(df, test_size=0.2, random_state=42)\n\ntrain_dataset = PneumoniaDataset(train_df, config.IMAGE_DIR, is_train=True)\nval_dataset = PneumoniaDataset(val_df, config.IMAGE_DIR, is_train=False)\n\ntrain_loader = DataLoader(train_dataset, batch_size=config.BATCH_SIZE, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=config.BATCH_SIZE, shuffle=False)\n\n# ==========================================\n# 5. MODEL & OPTIMIZER INITIALIZATION\n# ==========================================\nprint(\"⚙️ Initializing Attention U-Net and Optimizer...\")\nmodel = build_model(encoder_name=config.ENCODER, weights=config.WEIGHTS).to(config.DEVICE)\ncriterion = PneumoniaLoss()\noptimizer = optim.AdamW(model.parameters(), lr=config.LR)\n\n# Updated to the newer PyTorch syntax to remove that red warning you saw!\nscaler = torch.amp.GradScaler('cuda')\n# ==========================================\n# 6. EXECUTE TRAINING LOOP\n# ==========================================\nbest_val_loss = float('inf')\n\nprint(\"\\n🔥 Starting Training Pipeline...\")\nfor epoch in range(config.EPOCHS):\n    print(f\"\\n--- Epoch {epoch+1}/{config.EPOCHS} ---\")\n    \n    train_loss = train_one_epoch(model, train_loader, optimizer, criterion, config.DEVICE, scaler)\n    val_loss = validate_one_epoch(model, val_loader, criterion, config.DEVICE)\n    \n    print(f\"Train Loss: {train_loss:.4f} | Val Loss: {val_loss:.4f}\")\n    \n    # Save the best model locally on Kaggle\n    if val_loss < best_val_loss:\n        best_val_loss = val_loss\n        save_path = f\"{config.OUTPUT_DIR}/best_pneumonia_model.pth\"\n        torch.save(model.state_dict(), save_path)\n        print(f\"⭐ Model Improved! Saved to: {save_path}\")\n\nprint(\"\\n🎉 Training Complete! You can now download your model weights from the 'Output' folder on the right.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-27T17:39:28.509135Z","iopub.execute_input":"2026-03-27T17:39:28.509735Z","iopub.status.idle":"2026-03-27T20:18:31.4828Z","shell.execute_reply.started":"2026-03-27T17:39:28.509703Z","shell.execute_reply":"2026-03-27T20:18:31.48189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport sys\nimport torch\nimport pandas as pd\nfrom torch.utils.data import DataLoader\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\n\nprint(\"🔍 Scanning Kaggle directories...\")\n\n# ==========================================\n# 1. DYNAMIC PATH FINDERS\n# ==========================================\nscript_dir = None\nmodel_path = None\n\nfor root, dirs, files in os.walk('/kaggle/input'):\n    # Find the scripts folder\n    if 'config.py' in files and script_dir is None:\n        script_dir = root\n        print(f\"✅ Found scripts at: {script_dir}\")\n        \n    # Find the trained model weights\n    if 'best_pneumonia_model.pth' in files and model_path is None:\n        model_path = os.path.join(root, 'best_pneumonia_model.pth')\n        print(f\"✅ Found model weights at: {model_path}\")\n\n# Strict Error Checking\nif not script_dir:\n    raise FileNotFoundError(\"❌ ERROR: Could not find 'config.py'. Make sure 'pneumonia-scripts' is added to the notebook.\")\nif not model_path:\n    raise FileNotFoundError(\"❌ ERROR: Could not find 'best_pneumonia_model.pth'. Make sure 'newsddfs' is added to the notebook.\")\n\n# Add scripts to path\nsys.path.insert(0, script_dir)\n\n# ==========================================\n# 2. SAFE IMPORTS & METRICS\n# ==========================================\nimport config\nfrom data_pipeline.dataset import PneumoniaDataset\nfrom models.attention_unet import build_model\n\ndef calculate_segmentation_metrics(pred_mask, true_mask, threshold=0.5):\n    \"\"\"Calculates Dice Score and IoU for medical segmentation.\"\"\"\n    pred_binary = (pred_mask > threshold).float()\n    true_binary = true_mask.float()\n    \n    pred_flat = pred_binary.view(-1)\n    true_flat = true_binary.view(-1)\n    \n    intersection = (pred_flat * true_flat).sum()\n    \n    dice_score = (2. * intersection + 1e-6) / (pred_flat.sum() + true_flat.sum() + 1e-6)\n    union = pred_flat.sum() + true_flat.sum() - intersection\n    iou_score = (intersection + 1e-6) / (union + 1e-6)\n    \n    return dice_score.item(), iou_score.item()\n\n# ==========================================\n# 3. LOAD MODEL & EVALUATE\n# ==========================================\nCSV_PATH = \"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv\"\nIMAGE_DIR = \"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images\"\n\nprint(f\"\\n⏳ Loading your trained model from {model_path}...\")\nmodel = build_model(encoder_name=config.ENCODER, weights=None).to(config.DEVICE)\nmodel.load_state_dict(torch.load(model_path))\nmodel.eval()\nprint(\"✅ Model Loaded Successfully!\")\n\n# Extract exactly 5% unseen test set\nprint(\"📊 Preparing test data...\")\ndf = pd.read_csv(CSV_PATH)\ndf[['x', 'y', 'width', 'height']] = df[['x', 'y', 'width', 'height']].fillna(0)\n_, test_df = train_test_split(df, test_size=0.05, random_state=42)\n\ntest_dataset = PneumoniaDataset(test_df, IMAGE_DIR, is_train=False)\ntest_loader = DataLoader(test_dataset, batch_size=config.BATCH_SIZE, shuffle=False)\n\nprint(f\"🚀 Running Evaluation on {len(test_df)} Unseen Patients...\")\ntotal_dice = 0.0\ntotal_iou = 0.0\nbatches = 0\n\nwith torch.no_grad():\n    # Use torch.amp.autocast for faster inference\n    with torch.amp.autocast('cuda'):\n        for images, masks in tqdm(test_loader, desc=\"Evaluating\"):\n            images = images.to(config.DEVICE)\n            masks = masks.to(config.DEVICE)\n            \n            predictions = model(images)\n            predictions = torch.sigmoid(predictions)\n            \n            dice, iou = calculate_segmentation_metrics(predictions, masks)\n            total_dice += dice\n            total_iou += iou\n            batches += 1\n\nprint(\"\\n🏆 --- FINAL RESUME METRICS --- 🏆\")\nprint(f\"Average Dice Score: {(total_dice/batches):.4f}\")\nprint(f\"Average IoU: {(total_iou/batches):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-28T06:01:26.597368Z","iopub.execute_input":"2026-03-28T06:01:26.597627Z","iopub.status.idle":"2026-03-28T06:01:37.320952Z","shell.execute_reply.started":"2026-03-28T06:01:26.597604Z","shell.execute_reply":"2026-03-28T06:01:37.319957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install segmentation-models-pytorch pydicom\nimport os\nimport sys\nimport torch\nimport pandas as pd\nfrom torch.utils.data import DataLoader\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\n\nprint(\"🔍 Scanning Kaggle directories for ALL files...\")\n\n# ==========================================\n# 1. ULTIMATE DYNAMIC PATH FINDER\n# ==========================================\nscript_dir = None\nmodel_path = None\ncsv_path = None\nimage_dir = None\n\nfor root, dirs, files in os.walk('/kaggle/input'):\n    # Find scripts\n    if 'config.py' in files and script_dir is None:\n        script_dir = root\n    # Find model weights\n    if 'best_pneumonia_model.pth' in files and model_path is None:\n        model_path = os.path.join(root, 'best_pneumonia_model.pth')\n    # Find RSNA CSV and Images\n    if 'stage_2_train_labels.csv' in files and csv_path is None:\n        csv_path = os.path.join(root, 'stage_2_train_labels.csv')\n        image_dir = os.path.join(root, 'stage_2_train_images')\n\nsys.path.insert(0, script_dir)\n\n# ==========================================\n# 2. SAFE IMPORTS & METRICS\n# ==========================================\nimport config\nfrom data_pipeline.dataset import PneumoniaDataset\nfrom models.attention_unet import build_model\n\ndef calculate_segmentation_metrics(pred_mask, true_mask, threshold=0.5):\n    \"\"\"Calculates Dice Score and IoU for medical segmentation.\"\"\"\n    pred_binary = (pred_mask > threshold).float()\n    true_binary = true_mask.float()\n    \n    pred_flat = pred_binary.view(-1)\n    true_flat = true_binary.view(-1)\n    \n    intersection = (pred_flat * true_flat).sum()\n    \n    dice_score = (2. * intersection + 1e-6) / (pred_flat.sum() + true_flat.sum() + 1e-6)\n    union = pred_flat.sum() + true_flat.sum() - intersection\n    iou_score = (intersection + 1e-6) / (union + 1e-6)\n    \n    return dice_score.item(), iou_score.item()\n\n# ==========================================\n# 3. LOAD MODEL (DYNAMIC DEVICE: GPU OR CPU)\n# ==========================================\n# 🔥 YAHAN CHANGE KIYA HAI: Automatically pick GPU if available\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nprint(f\"\\n⏳ Loading your trained model onto {DEVICE.type.upper()}...\")\nmodel = build_model(encoder_name=config.ENCODER, weights=None).to(DEVICE)\nmodel.load_state_dict(torch.load(model_path, map_location=DEVICE))\nmodel.eval()\nprint(f\"✅ Model Loaded Successfully on {DEVICE.type.upper()}!\")\n\n# ==========================================\n# 4. PREPARE STRICT UNSEEN TEST DATA (FIXED LEAKAGE)\n# ==========================================\nprint(\"\\n📊 Preparing STRICT unseen test data...\")\ndf = pd.read_csv(csv_path)\ndf[['x', 'y', 'width', 'height']] = df[['x', 'y', 'width', 'height']].fillna(0)\n\n# Exact wahi split laga rahe hain jo training mein lagaya tha\ntrain_df_kaggle, _ = train_test_split(df, test_size=0.2, random_state=42)\n\n# Un patients ki list nikal rahe hain jo model dekh chuka hai\nseen_patients = set(train_df_kaggle['patientId'].unique())\n\n# Original data mein se seen patients ko HATA rahe hain (~ operator)\nstrict_test_df = df[~df['patientId'].isin(seen_patients)].reset_index(drop=True)\n\nprint(f\"✅ Total Patients in dataset: {df['patientId'].nunique()}\")\nprint(f\"✅ Patients seen during training: {len(seen_patients)}\")\nprint(f\"🚀 STRICT Unseen Patients for evaluation: {strict_test_df['patientId'].nunique()}\")\nprint(f\"🚀 Total X-rays for testing: {len(strict_test_df)}\")\n\n# Naya strict test loader banaya\ntest_dataset = PneumoniaDataset(strict_test_df, image_dir, is_train=False)\ntest_loader = DataLoader(test_dataset, batch_size=config.BATCH_SIZE, shuffle=False)\n\n# ==========================================\n# 5. RUN EVALUATION\n# ==========================================\nprint(f\"\\n🚀 Running Evaluation on {strict_test_df['patientId'].nunique()} Unseen Patients...\")\ntotal_dice = 0.0\ntotal_iou = 0.0\nbatches = 0\n\nwith torch.no_grad():\n    # Progress bar ka description bhi update kar diya\n    for images, masks in tqdm(test_loader, desc=f\"Evaluating ({DEVICE.type.upper()})\"):\n        images = images.to(DEVICE)\n        masks = masks.to(DEVICE)\n        \n        # Mixed Precision (AMP) use kar rahe hain thoda aur fast karne ke liye\n        with torch.amp.autocast('cuda' if DEVICE.type == 'cuda' else 'cpu'):\n            predictions = model(images)\n            predictions = torch.sigmoid(predictions)\n        \n        dice, iou = calculate_segmentation_metrics(predictions, masks)\n        total_dice += dice\n        total_iou += iou\n        batches += 1\n\nprint(\"\\n🏆 --- FINAL HONEST METRICS --- 🏆\")\nprint(f\"Average Dice Score: {(total_dice/batches):.4f}\")\nprint(f\"Average IoU: {(total_iou/batches):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-06T17:53:37.61309Z","iopub.execute_input":"2026-04-06T17:53:37.61386Z","iopub.status.idle":"2026-04-06T17:55:56.514394Z","shell.execute_reply.started":"2026-04-06T17:53:37.613825Z","shell.execute_reply":"2026-04-06T17:55:56.513321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nimport pandas as pd\nfrom torch.utils.data import DataLoader\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\n\n# ... (Tera baaki imports aur setup code same rahega) ...\n\nprint(\"📊 Preparing STRICT unseen test data...\")\n\n# 1. Poora data load kar\ndf = pd.read_csv(csv_path)\ndf[['x', 'y', 'width', 'height']] = df[['x', 'y', 'width', 'height']].fillna(0)\n\n# 2. EXACT wahi split laga jo tune Kaggle training ke time lagaya tha\n# Isse humein pata chal jayega ki model ne actually kaunse rows dekhe the\ntrain_df_kaggle, _ = train_test_split(df, test_size=0.2, random_state=42)\n\n# 3. Un saare patients ke IDs ki ek list (set) bana le jo training mein the\nseen_patients = set(train_df_kaggle['patientId'].unique())\n\n# 4. MAGIC STEP: Ab original dataframe (df) mein se wo saare patients HATA de \n# jo model ne train hote time dekhe the. (~ ka matlab 'NOT IN' hota hai)\nstrict_test_df = df[~df['patientId'].isin(seen_patients)].reset_index(drop=True)\n\nprint(f\"✅ Total Patients in original data: {df['patientId'].nunique()}\")\nprint(f\"✅ Patients seen during training: {len(seen_patients)}\")\nprint(f\"🚀 STRICT Unseen Patients for evaluation: {strict_test_df['patientId'].nunique()}\")\nprint(f\"🚀 Total images/rows for testing: {len(strict_test_df)}\")\n\n# 5. Ab is ekdum saaf 'strict_test_df' ko DataLoader mein daal\ntest_dataset = PneumoniaDataset(strict_test_df, image_dir, is_train=False)\ntest_loader = DataLoader(test_dataset, batch_size=config.BATCH_SIZE, shuffle=False)\n\n# ==========================================\n# RUN EVALUATION LOOP (Pura same rahega)\n# ==========================================\nprint(f\"🚀 Running Evaluation on Pure Unseen Patients...\")\ntotal_dice = 0.0\ntotal_iou = 0.0\nbatches = 0\n\nwith torch.no_grad():\n    for images, masks in tqdm(test_loader, desc=\"Evaluating (CPU)\"):\n        images = images.to(DEVICE)\n        masks = masks.to(DEVICE)\n        \n        predictions = model(images)\n        predictions = torch.sigmoid(predictions)\n        \n        dice, iou = calculate_segmentation_metrics(predictions, masks)\n        total_dice += dice\n        total_iou += iou\n        batches += 1\n\nprint(\"\\n🏆 --- FINAL HONEST METRICS --- 🏆\")\nprint(f\"Average Dice Score: {(total_dice/batches):.4f}\")\nprint(f\"Average IoU: {(total_iou/batches):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-06T17:53:28.027491Z","iopub.status.idle":"2026-04-06T17:53:28.027782Z","shell.execute_reply.started":"2026-04-06T17:53:28.027653Z","shell.execute_reply":"2026-04-06T17:53:28.02767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install segmentation-models-pytorch pydicom","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-06T17:49:15.584278Z","iopub.execute_input":"2026-04-06T17:49:15.585128Z","iopub.status.idle":"2026-04-06T17:49:20.732413Z","shell.execute_reply.started":"2026-04-06T17:49:15.585093Z","shell.execute_reply":"2026-04-06T17:49:20.731521Z"}},"outputs":[],"execution_count":null}]}