{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"dockerImageVersionId":31240,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"CELL 1 — Imports & Seeds (Optimized)","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\n\nfrom torchvision import datasets, models, transforms\nfrom torchvision.models import MobileNet_V3_Large_Weights\nfrom torch.utils.data import DataLoader, Dataset, WeightedRandomSampler\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\nfrom PIL import Image\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# Reproducibility\ndef seed_everything(seed=42):\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = False\n    torch.backends.cudnn.benchmark = True\n\nseed_everything()\n\nprint(\"✅ Imports & seeds ready\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T07:05:51.508080Z","iopub.execute_input":"2026-01-04T07:05:51.508378Z","iopub.status.idle":"2026-01-04T07:05:51.516318Z","shell.execute_reply.started":"2026-01-04T07:05:51.508359Z","shell.execute_reply":"2026-01-04T07:05:51.515545Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"CELL 2 — Config (Multi-GPU + Speed)","metadata":{}},{"cell_type":"code","source":"IMG_SIZE = 384\nBATCH_SIZE = 32            # effective = 32 × 2 GPUs = 64\nEPOCHS = 30\nLR = 3e-4\nPATIENCE = 5               # Early stopping patience\n\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nNUM_GPUS = torch.cuda.device_count()\n\nprint(f\"🚀 Device: {DEVICE}\")\nprint(f\"🔥 GPUs available: {NUM_GPUS}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T07:05:51.517799Z","iopub.execute_input":"2026-01-04T07:05:51.518197Z","iopub.status.idle":"2026-01-04T07:05:51.531203Z","shell.execute_reply.started":"2026-01-04T07:05:51.518175Z","shell.execute_reply":"2026-01-04T07:05:51.530429Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"CELL 3 — Dataset Paths & CSV","metadata":{}},{"cell_type":"code","source":"TRAIN_CSV = \"/kaggle/input/siim-isic-melanoma-classification/train.csv\"\nTRAIN_IMG_DIR = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/train\"\n\ndf = pd.read_csv(TRAIN_CSV)\nprint(f\"📊 Samples: {len(df)}\")\nprint(df['target'].value_counts())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T07:05:51.531996Z","iopub.execute_input":"2026-01-04T07:05:51.532282Z","iopub.status.idle":"2026-01-04T07:05:51.588184Z","shell.execute_reply.started":"2026-01-04T07:05:51.532257Z","shell.execute_reply":"2026-01-04T07:05:51.587365Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"CELL 4 — Dataset Class (Safe + Fast)","metadata":{}},{"cell_type":"code","source":"class MelanomaDataset(Dataset):\n    def __init__(self, df, img_dir, transform=None):\n        self.df = df.reset_index(drop=True)\n        self.img_dir = img_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = os.path.join(self.img_dir, f\"{row.image_name}.jpg\")\n\n        try:\n            img = Image.open(img_path).convert(\"RGB\")\n        except:\n            img = Image.new(\"RGB\", (IMG_SIZE, IMG_SIZE), (128,128,128))\n\n        if self.transform:\n            img = self.transform(img)\n\n        label = torch.tensor(row.target, dtype=torch.long)\n        return img, label\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T07:05:51.589922Z","iopub.execute_input":"2026-01-04T07:05:51.590483Z","iopub.status.idle":"2026-01-04T07:05:51.597203Z","shell.execute_reply.started":"2026-01-04T07:05:51.590460Z","shell.execute_reply":"2026-01-04T07:05:51.596286Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"CELL 5 — Transforms","metadata":{}},{"cell_type":"code","source":"train_tfms = transforms.Compose([\n    transforms.Resize((IMG_SIZE, IMG_SIZE)),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomVerticalFlip(),\n    transforms.RandomRotation(15),\n    transforms.ColorJitter(0.1,0.1,0.1),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485,0.456,0.406],[0.229,0.224,0.225])\n])\n\nval_tfms = transforms.Compose([\n    transforms.Resize((IMG_SIZE, IMG_SIZE)),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485,0.456,0.406],[0.229,0.224,0.225])\n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T07:05:51.599109Z","iopub.execute_input":"2026-01-04T07:05:51.599507Z","iopub.status.idle":"2026-01-04T07:05:51.613063Z","shell.execute_reply.started":"2026-01-04T07:05:51.599485Z","shell.execute_reply":"2026-01-04T07:05:51.612338Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"CELL 6 — Train/Val Split + Sampler","metadata":{}},{"cell_type":"code","source":"# ===============================\n# TRAIN / VALIDATION SPLIT\n# ===============================\n\nfrom sklearn.model_selection import train_test_split\n\n# Load CSV if not already loaded\nif 'df' not in globals():\n    df = pd.read_csv(TRAIN_CSV)\n\n# Make sure target column exists\nassert 'target' in df.columns, \"❌ 'target' column not found in CSV\"\n\ntrain_df, val_df = train_test_split(\n    df,\n    test_size=0.2,\n    stratify=df['target'],\n    random_state=42\n)\n\ntrain_df = train_df.reset_index(drop=True)\nval_df   = val_df.reset_index(drop=True)\n\nprint(\"✅ train_df and val_df created\")\nprint(\"Train size:\", len(train_df))\nprint(\"Val size:\", len(val_df))\nprint(train_df['target'].value_counts())\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T07:05:51.613917Z","iopub.execute_input":"2026-01-04T07:05:51.614173Z","iopub.status.idle":"2026-01-04T07:05:51.652371Z","shell.execute_reply.started":"2026-01-04T07:05:51.614151Z","shell.execute_reply":"2026-01-04T07:05:51.651452Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"CELL 7 — Model (Multi-GPU Ready)","metadata":{}},{"cell_type":"code","source":"class MelanomaNet(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.model = models.mobilenet_v3_large(\n            weights=MobileNet_V3_Large_Weights.IMAGENET1K_V2\n        )\n        in_features = self.model.classifier[0].in_features\n        self.model.classifier = nn.Sequential(\n            nn.Dropout(0.3),\n            nn.Linear(in_features, 256),\n            nn.BatchNorm1d(256),\n            nn.Hardswish(),\n            nn.Linear(256, 2)\n        )\n\n    def forward(self, x):\n        return self.model(x)\n\nmodel = MelanomaNet().to(DEVICE)\n\n# 🔥 MULTI-GPU\nif NUM_GPUS > 1:\n    model = nn.DataParallel(model)\n\nprint(\"✅ Model ready (Multi-GPU enabled)\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T07:05:51.693640Z","iopub.execute_input":"2026-01-04T07:05:51.693902Z","iopub.status.idle":"2026-01-04T07:05:51.848248Z","shell.execute_reply.started":"2026-01-04T07:05:51.693882Z","shell.execute_reply":"2026-01-04T07:05:51.847442Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"CELL 8 — Loss, Optimizer, AMP","metadata":{}},{"cell_type":"code","source":"criterion = nn.CrossEntropyLoss()\n\noptimizer = optim.AdamW(model.parameters(), lr=LR, weight_decay=1e-4)\n\nscaler = torch.cuda.amp.GradScaler()\n\nscheduler = optim.lr_scheduler.CosineAnnealingWarmRestarts(\n    optimizer, T_0=5, T_mult=2\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T07:05:51.849886Z","iopub.execute_input":"2026-01-04T07:05:51.850099Z","iopub.status.idle":"2026-01-04T07:05:51.855951Z","shell.execute_reply.started":"2026-01-04T07:05:51.850082Z","shell.execute_reply":"2026-01-04T07:05:51.855303Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"CELL 9 — Training + Validation (FAST)","metadata":{}},{"cell_type":"code","source":"def train_epoch(loader):\n    model.train()\n    total, correct, loss_sum = 0, 0, 0\n\n    for x,y in tqdm(loader, leave=False):\n        x,y = x.to(DEVICE), y.to(DEVICE)\n        optimizer.zero_grad()\n\n        with torch.cuda.amp.autocast():\n            out = model(x)\n            loss = criterion(out, y)\n\n        scaler.scale(loss).backward()\n        scaler.step(optimizer)\n        scaler.update()\n\n        loss_sum += loss.item()*y.size(0)\n        correct += (out.argmax(1)==y).sum().item()\n        total += y.size(0)\n\n    return loss_sum/total, 100*correct/total\n\n\ndef validate_epoch(loader):\n    model.eval()\n    total, correct, loss_sum = 0, 0, 0\n\n    with torch.no_grad():\n        for x,y in loader:\n            x,y = x.to(DEVICE), y.to(DEVICE)\n            out = model(x)\n            loss = criterion(out,y)\n\n            loss_sum += loss.item()*y.size(0)\n            correct += (out.argmax(1)==y).sum().item()\n            total += y.size(0)\n\n    return loss_sum/total, 100*correct/total\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T07:05:51.856823Z","iopub.execute_input":"2026-01-04T07:05:51.857082Z","iopub.status.idle":"2026-01-04T07:05:51.869206Z","shell.execute_reply.started":"2026-01-04T07:05:51.857059Z","shell.execute_reply":"2026-01-04T07:05:51.868530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===============================\n# FIX CELL — ENSURE DATALOADERS EXIST\n# ===============================\n\nassert 'train_df' in globals(), \"❌ train_df missing (run split cell)\"\nassert 'val_df' in globals(), \"❌ val_df missing (run split cell)\"\nassert 'train_tfms' in globals(), \"❌ train_tfms missing\"\nassert 'val_tfms' in globals(), \"❌ val_tfms missing\"\n\ntrain_ds = MelanomaDataset(train_df, TRAIN_IMG_DIR, train_tfms)\nval_ds   = MelanomaDataset(val_df, TRAIN_IMG_DIR, val_tfms)\n\ncounts = train_df['target'].value_counts()\nweights = train_df['target'].map({0:1.0, 1:counts[0]/counts[1]}).values\n\nsampler = WeightedRandomSampler(weights, len(weights), replacement=True)\n\ntrain_loader = DataLoader(\n    train_ds,\n    batch_size=BATCH_SIZE,\n    sampler=sampler,\n    num_workers=4,\n    pin_memory=True,\n    persistent_workers=True\n)\n\nval_loader = DataLoader(\n    val_ds,\n    batch_size=BATCH_SIZE,\n    shuffle=False,\n    num_workers=4,\n    pin_memory=True\n)\n\nprint(\"✅ train_loader and val_loader are now defined\")\nprint(f\"   Train batches: {len(train_loader)}\")\nprint(f\"   Val batches:   {len(val_loader)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T07:05:51.869923Z","iopub.execute_input":"2026-01-04T07:05:51.870221Z","iopub.status.idle":"2026-01-04T07:05:51.888677Z","shell.execute_reply.started":"2026-01-04T07:05:51.870192Z","shell.execute_reply":"2026-01-04T07:05:51.887986Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"CELL 10 — Early Stopping Training Loop","metadata":{}},{"cell_type":"code","source":"# ===============================\n# CELL 10 — TRAINING WITH EARLY STOPPING\n# ===============================\n\nbest_acc = 0.0\npatience_counter = 0\n\ntrain_history = []\nval_history = []\n\nprint(\"\\n🚀 Starting Training...\\n\")\n\nfor epoch in range(EPOCHS):\n    print(f\"🔁 Epoch {epoch+1}/{EPOCHS}\")\n\n    # --- TRAIN ---\n    train_loss, train_acc = train_epoch(train_loader)\n\n    # --- VALIDATE ---\n    val_loss, val_acc = validate_epoch(val_loader)\n\n    # Scheduler step AFTER validation\n    scheduler.step()\n\n    # Save history\n    train_history.append((train_loss, train_acc))\n    val_history.append((val_loss, val_acc))\n\n    print(\n        f\"Train Loss: {train_loss:.4f} | Train Acc: {train_acc:.2f}% || \"\n        f\"Val Loss: {val_loss:.4f} | Val Acc: {val_acc:.2f}%\"\n    )\n\n    # --- CHECKPOINT ---\n    if val_acc > best_acc:\n        best_acc = val_acc\n        patience_counter = 0\n\n        torch.save(\n            model.module.state_dict() if isinstance(model, nn.DataParallel)\n            else model.state_dict(),\n            \"best_model.pth\"\n        )\n\n        print(f\"💾 Best model saved (Val Acc = {best_acc:.2f}%)\")\n\n    else:\n        patience_counter += 1\n        print(f\"⏳ EarlyStopping Counter: {patience_counter}/{PATIENCE}\")\n\n    # --- EARLY STOPPING ---\n    if patience_counter >= PATIENCE:\n        print(\"⛔ Early stopping triggered\")\n        break\n\nprint(\"\\n🎯 Training Finished\")\nprint(f\"🔥 Best Validation Accuracy: {best_acc:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T07:05:51.890850Z","iopub.execute_input":"2026-01-04T07:05:51.891155Z","iopub.status.idle":"2026-01-04T13:57:33.894712Z","shell.execute_reply.started":"2026-01-04T07:05:51.891106Z","shell.execute_reply":"2026-01-04T13:57:33.893852Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"CELL — Load Best Model","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import (\n    accuracy_score, precision_score, recall_score,\n    f1_score, roc_auc_score, confusion_matrix,\n    classification_report, f1_score\n)\n\n# Load best model\nif isinstance(model, nn.DataParallel):\n    model.module.load_state_dict(torch.load(\"best_model.pth\"))\nelse:\n    model.load_state_dict(torch.load(\"best_model.pth\"))\n\nmodel.eval()\nprint(\"✅ Best model loaded\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T13:57:33.895927Z","iopub.execute_input":"2026-01-04T13:57:33.896284Z","iopub.status.idle":"2026-01-04T13:57:34.080399Z","shell.execute_reply.started":"2026-01-04T13:57:33.896259Z","shell.execute_reply":"2026-01-04T13:57:34.079799Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"CELL 2 — Run Evaluation (Get Predictions & Probabilities)","metadata":{}},{"cell_type":"code","source":"all_targets = []\nall_preds = []\nall_probs = []\n\nwith torch.no_grad():\n    for images, targets in tqdm(val_loader, desc=\"🔍 Evaluating\"):\n        images = images.to(DEVICE)\n        targets = targets.to(DEVICE)\n\n        outputs = model(images)\n        probs = F.softmax(outputs, dim=1)[:, 1]  # malignant prob\n        preds = torch.argmax(outputs, dim=1)\n\n        all_targets.extend(targets.cpu().numpy())\n        all_preds.extend(preds.cpu().numpy())\n        all_probs.extend(probs.cpu().numpy())\n\nprint(\"✅ Evaluation forward pass completed\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T13:57:34.081150Z","iopub.execute_input":"2026-01-04T13:57:34.081862Z","iopub.status.idle":"2026-01-04T14:05:57.595550Z","shell.execute_reply.started":"2026-01-04T13:57:34.081834Z","shell.execute_reply":"2026-01-04T14:05:57.594656Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"CELL 3 — Final Metrics (MAIN RESULTS)","metadata":{}},{"cell_type":"code","source":"acc = accuracy_score(all_targets, all_preds)\nprecision = precision_score(all_targets, all_preds, zero_division=0)\nrecall = recall_score(all_targets, all_preds, zero_division=0)\nf1 = f1_score(all_targets, all_preds, zero_division=0)\nauc = roc_auc_score(all_targets, all_probs)\ncm = confusion_matrix(all_targets, all_preds)\n\nprint(\"\\n🎯 FINAL EVALUATION SUMMARY\")\nprint(\"=\"*70)\nprint(f\"Accuracy  : {acc*100:.2f}%\")\nprint(f\"Precision : {precision:.4f}\")\nprint(f\"Recall    : {recall:.4f}\")\nprint(f\"F1-score  : {f1:.4f}\")\nprint(f\"AUC-ROC   : {auc:.4f}\")\n\nprint(\"\\n📊 Confusion Matrix\")\nprint(cm)\n\nprint(\"\\n📋 Classification Report\")\nprint(\n    classification_report(\n        all_targets,\n        all_preds,\n        target_names=[\"Benign\", \"Malignant\"],\n        digits=4\n    )\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T14:05:57.596551Z","iopub.execute_input":"2026-01-04T14:05:57.596787Z","iopub.status.idle":"2026-01-04T14:05:57.664899Z","shell.execute_reply.started":"2026-01-04T14:05:57.596767Z","shell.execute_reply":"2026-01-04T14:05:57.664368Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"CELL 4 — F1-Score Curve (Threshold vs F1)","metadata":{}},{"cell_type":"code","source":"import numpy as np\n\nthresholds = np.linspace(0.05, 0.95, 50)\nf1_scores = []\n\nfor t in thresholds:\n    preds_t = (np.array(all_probs) >= t).astype(int)\n    f1_scores.append(\n        f1_score(all_targets, preds_t, zero_division=0)\n    )\n\nbest_idx = np.argmax(f1_scores)\nbest_threshold = thresholds[best_idx]\nbest_f1 = f1_scores[best_idx]\n\nprint(f\"⭐ Best F1-score: {best_f1:.4f} at threshold = {best_threshold:.2f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T14:05:57.665613Z","iopub.execute_input":"2026-01-04T14:05:57.665848Z","iopub.status.idle":"2026-01-04T14:05:57.970594Z","shell.execute_reply.started":"2026-01-04T14:05:57.665831Z","shell.execute_reply":"2026-01-04T14:05:57.969822Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"CELL 5 — Plot F1-Score Graph","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8,5))\nplt.plot(thresholds, f1_scores, marker='o', linewidth=2)\nplt.axvline(best_threshold, color='red', linestyle='--', label=f'Best Threshold = {best_threshold:.2f}')\nplt.scatter(best_threshold, best_f1, color='red', s=80)\n\nplt.xlabel(\"Decision Threshold\")\nplt.ylabel(\"F1-score\")\nplt.title(\"F1-score vs Classification Threshold\")\nplt.grid(alpha=0.3)\nplt.legend()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-04T14:05:57.971326Z","iopub.execute_input":"2026-01-04T14:05:57.971528Z","iopub.status.idle":"2026-01-04T14:05:58.275418Z","shell.execute_reply.started":"2026-01-04T14:05:57.971499Z","shell.execute_reply":"2026-01-04T14:05:58.274678Z"}},"outputs":[],"execution_count":null}]}