{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-01-26T18:47:33.755465Z","iopub.execute_input":"2026-01-26T18:47:33.755673Z","iopub.status.idle":"2026-01-26T18:49:35.870296Z","shell.execute_reply.started":"2026-01-26T18:47:33.755652Z","shell.execute_reply":"2026-01-26T18:49:35.868688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nos.listdir('/kaggle/input')\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\n\nfrom torchvision import datasets, models, transforms\nfrom torchvision.models import MobileNet_V3_Large_Weights\nfrom torch.utils.data import DataLoader, Dataset, WeightedRandomSampler\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\nfrom PIL import Image\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# Reproducibility\ndef seed_everything(seed=42):\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = False\n    torch.backends.cudnn.benchmark = True\n\nseed_everything()\n\nprint(\"✅ Imports & seeds ready\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T11:22:23.696996Z","iopub.execute_input":"2026-01-27T11:22:23.697692Z","iopub.status.idle":"2026-01-27T11:22:31.762719Z","shell.execute_reply.started":"2026-01-27T11:22:23.697665Z","shell.execute_reply":"2026-01-27T11:22:31.762058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE = 384\nBATCH_SIZE = 32            # effective = 32 × 2 GPUs = 64\nEPOCHS = 30\nLR = 3e-4\nPATIENCE = 5               # Early stopping patience\n\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nNUM_GPUS = torch.cuda.device_count()\n\nprint(f\"🚀 Device: {DEVICE}\")\nprint(f\"🔥 GPUs available: {NUM_GPUS}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T11:22:31.764119Z","iopub.execute_input":"2026-01-27T11:22:31.764430Z","iopub.status.idle":"2026-01-27T11:22:31.842563Z","shell.execute_reply.started":"2026-01-27T11:22:31.764410Z","shell.execute_reply":"2026-01-27T11:22:31.841920Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\nTRAIN_CSV = \"/kaggle/input/siim-isic-melanoma-classification/train.csv\"\nTRAIN_IMG_DIR = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/train\"\n\ndf = pd.read_csv(TRAIN_CSV)\nprint(f\"📊 Samples: {len(df)}\")\nprint(df['target'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T11:22:31.843492Z","iopub.execute_input":"2026-01-27T11:22:31.843785Z","iopub.status.idle":"2026-01-27T11:22:31.940028Z","shell.execute_reply.started":"2026-01-27T11:22:31.843732Z","shell.execute_reply":"2026-01-27T11:22:31.939195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class MelanomaDataset(Dataset):\n    def __init__(self, df, img_dir, transform=None):\n        self.df = df.reset_index(drop=True)\n        self.img_dir = img_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = os.path.join(self.img_dir, f\"{row.image_name}.jpg\")\n\n        try:\n            img = Image.open(img_path).convert(\"RGB\")\n        except:\n            img = Image.new(\"RGB\", (IMG_SIZE, IMG_SIZE), (128,128,128))\n\n        if self.transform:\n            img = self.transform(img)\n\n        label = torch.tensor(row.target, dtype=torch.long)\n        return img, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T11:22:31.941474Z","iopub.execute_input":"2026-01-27T11:22:31.942202Z","iopub.status.idle":"2026-01-27T11:22:31.947700Z","shell.execute_reply.started":"2026-01-27T11:22:31.942177Z","shell.execute_reply":"2026-01-27T11:22:31.947088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_tfms = transforms.Compose([\n    transforms.Resize((IMG_SIZE, IMG_SIZE)),\n    transforms.RandomHorizontalFlip(),\n    transforms.RandomVerticalFlip(),\n    transforms.RandomRotation(15),\n    transforms.ColorJitter(0.1,0.1,0.1),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485,0.456,0.406],[0.229,0.224,0.225])\n])\n\nval_tfms = transforms.Compose([\n    transforms.Resize((IMG_SIZE, IMG_SIZE)),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485,0.456,0.406],[0.229,0.224,0.225])\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T11:22:34.456537Z","iopub.execute_input":"2026-01-27T11:22:34.457442Z","iopub.status.idle":"2026-01-27T11:22:34.462956Z","shell.execute_reply.started":"2026-01-27T11:22:34.457411Z","shell.execute_reply":"2026-01-27T11:22:34.462113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===============================\n# TRAIN / VALIDATION SPLIT\n# ===============================\n\nfrom sklearn.model_selection import train_test_split\n\n# Load CSV if not already loaded\nif 'df' not in globals():\n    df = pd.read_csv(TRAIN_CSV)\n\n# Make sure target column exists\nassert 'target' in df.columns, \"❌ 'target' column not found in CSV\"\n\ntrain_df, val_df = train_test_split(\n    df,\n    test_size=0.2,\n    stratify=df['target'],\n    random_state=42\n)\n\ntrain_df = train_df.reset_index(drop=True)\nval_df   = val_df.reset_index(drop=True)\n\nprint(\"✅ train_df and val_df created\")\nprint(\"Train size:\", len(train_df))\nprint(\"Val size:\", len(val_df))\nprint(train_df['target'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T11:22:38.062436Z","iopub.execute_input":"2026-01-27T11:22:38.063078Z","iopub.status.idle":"2026-01-27T11:22:38.098807Z","shell.execute_reply.started":"2026-01-27T11:22:38.063051Z","shell.execute_reply":"2026-01-27T11:22:38.098161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class MelanomaNet(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.model = models.mobilenet_v3_large(\n            weights=MobileNet_V3_Large_Weights.IMAGENET1K_V2\n        )\n        in_features = self.model.classifier[0].in_features\n        self.model.classifier = nn.Sequential(\n            nn.Dropout(0.3),\n            nn.Linear(in_features, 256),\n            nn.BatchNorm1d(256),\n            nn.Hardswish(),\n            nn.Linear(256, 2)\n        )\n\n    def forward(self, x):\n        return self.model(x)\n\nmodel = MelanomaNet().to(DEVICE)\n\n# 🔥 MULTI-GPU\nif NUM_GPUS > 1:\n    model = nn.DataParallel(model)\n\nprint(\"✅ Model ready (Multi-GPU enabled)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T11:22:41.733126Z","iopub.execute_input":"2026-01-27T11:22:41.733432Z","iopub.status.idle":"2026-01-27T11:22:42.240386Z","shell.execute_reply.started":"2026-01-27T11:22:41.733410Z","shell.execute_reply":"2026-01-27T11:22:42.239547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"criterion = nn.CrossEntropyLoss()\n\noptimizer = optim.AdamW(model.parameters(), lr=LR, weight_decay=1e-4)\n\nscaler = torch.cuda.amp.GradScaler()\n\nscheduler = optim.lr_scheduler.CosineAnnealingWarmRestarts(\n    optimizer, T_0=5, T_mult=2\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T11:22:45.485199Z","iopub.execute_input":"2026-01-27T11:22:45.485491Z","iopub.status.idle":"2026-01-27T11:22:45.491334Z","shell.execute_reply.started":"2026-01-27T11:22:45.485467Z","shell.execute_reply":"2026-01-27T11:22:45.490288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_epoch(loader):\n    model.train()\n    total, correct, loss_sum = 0, 0, 0\n\n    for x,y in tqdm(loader, leave=False):\n        x,y = x.to(DEVICE), y.to(DEVICE)\n        optimizer.zero_grad()\n\n        with torch.cuda.amp.autocast():\n            out = model(x)\n            loss = criterion(out, y)\n\n        scaler.scale(loss).backward()\n        scaler.step(optimizer)\n        scaler.update()\n\n        loss_sum += loss.item()*y.size(0)\n        correct += (out.argmax(1)==y).sum().item()\n        total += y.size(0)\n\n    return loss_sum/total, 100*correct/total\n\n\ndef validate_epoch(loader):\n    model.eval()\n    total, correct, loss_sum = 0, 0, 0\n\n    with torch.no_grad():\n        for x,y in loader:\n            x,y = x.to(DEVICE), y.to(DEVICE)\n            out = model(x)\n            loss = criterion(out,y)\n\n            loss_sum += loss.item()*y.size(0)\n            correct += (out.argmax(1)==y).sum().item()\n            total += y.size(0)\n\n    return loss_sum/total, 100*correct/total","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T11:22:48.946190Z","iopub.execute_input":"2026-01-27T11:22:48.946696Z","iopub.status.idle":"2026-01-27T11:22:48.953922Z","shell.execute_reply.started":"2026-01-27T11:22:48.946671Z","shell.execute_reply":"2026-01-27T11:22:48.953071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===============================\n# FIX CELL — ENSURE DATALOADERS EXIST\n# ===============================\n\nassert 'train_df' in globals(), \"❌ train_df missing (run split cell)\"\nassert 'val_df' in globals(), \"❌ val_df missing (run split cell)\"\nassert 'train_tfms' in globals(), \"❌ train_tfms missing\"\nassert 'val_tfms' in globals(), \"❌ val_tfms missing\"\n\ntrain_ds = MelanomaDataset(train_df, TRAIN_IMG_DIR, train_tfms)\nval_ds   = MelanomaDataset(val_df, TRAIN_IMG_DIR, val_tfms)\n\ncounts = train_df['target'].value_counts()\nweights = train_df['target'].map({0:1.0, 1:counts[0]/counts[1]}).values\n\nsampler = WeightedRandomSampler(weights, len(weights), replacement=True)\n\ntrain_loader = DataLoader(\n    train_ds,\n    batch_size=BATCH_SIZE,\n    sampler=sampler,\n    num_workers=4,\n    pin_memory=True,\n    persistent_workers=True\n)\n\nval_loader = DataLoader(\n    val_ds,\n    batch_size=BATCH_SIZE,\n    shuffle=False,\n    num_workers=4,\n    pin_memory=True\n)\n\nprint(\"✅ train_loader and val_loader are now defined\")\nprint(f\"   Train batches: {len(train_loader)}\")\nprint(f\"   Val batches:   {len(val_loader)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-27T11:22:55.704653Z","iopub.execute_input":"2026-01-27T11:22:55.705374Z","iopub.status.idle":"2026-01-27T11:22:55.717467Z","shell.execute_reply.started":"2026-01-27T11:22:55.705347Z","shell.execute_reply":"2026-01-27T11:22:55.716649Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_acc = 0.0\npatience_counter = 0\n\ntrain_history = []\nval_history = []\n\nprint(\"\\n🚀 Starting Training...\\n\")\n\nfor epoch in range(EPOCHS):\n    print(f\"🔁 Epoch {epoch+1}/{EPOCHS}\")\n\n    # --- TRAIN ---\n    train_loss, train_acc = train_epoch(train_loader)\n\n    # --- VALIDATE ---\n    val_loss, val_acc = validate_epoch(val_loader)\n\n    # Scheduler step AFTER validation\n    scheduler.step()\n\n    # Save history\n    train_history.append((train_loss, train_acc))\n    val_history.append((val_loss, val_acc))\n\n    print(\n        f\"Train Loss: {train_loss:.4f} | Train Acc: {train_acc:.2f}% || \"\n        f\"Val Loss: {val_loss:.4f} | Val Acc: {val_acc:.2f}%\"\n    )\n\n    # --- CHECKPOINT ---\n    if val_acc > best_acc:\n        best_acc = val_acc\n        patience_counter = 0\n\n        torch.save(\n            model.module.state_dict() if isinstance(model, nn.DataParallel)\n            else model.state_dict(),\n            \"best_model.pth\"\n        )\n\n        print(f\"💾 Best model saved (Val Acc = {best_acc:.2f}%)\")\n\n    else:\n        patience_counter += 1\n        print(f\"⏳ EarlyStopping Counter: {patience_counter}/{PATIENCE}\")\n\n    # --- EARLY STOPPING ---\n    if patience_counter >= PATIENCE:\n        print(\"⛔ Early stopping triggered\")\n        break\n\nprint(\"\\n🎯 Training Finished\")\nprint(f\"🔥 Best Validation Accuracy: {best_acc:.2f}%\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":" ","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}