{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"e3ca33e2","cell_type":"markdown","source":"# APTOS 2019 — Organized Pipeline (Preprocess → DataLoaders → CNN Features → ML + (Optional) Fine-Tuning)\n\n\n","metadata":{}},{"id":"6cbfaa7f","cell_type":"markdown","source":"## 0) Environment (Kaggle) — optional installs\n","metadata":{}},{"id":"dca1ca92","cell_type":"code","source":"# If you're on Kaggle, you usually DON'T need this.\n# Use only if you got version conflicts.\n\n# !pip uninstall -y scikit-learn imbalanced-learn\n# !pip install -q scikit-learn==1.1.3 imbalanced-learn==0.10.1\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:28:40.870369Z","iopub.execute_input":"2026-02-09T19:28:40.870678Z","iopub.status.idle":"2026-02-09T19:28:40.874687Z","shell.execute_reply.started":"2026-02-09T19:28:40.87065Z","shell.execute_reply":"2026-02-09T19:28:40.873885Z"}},"outputs":[],"execution_count":null},{"id":"0c60b52d","cell_type":"markdown","source":"## 1) Imports + Reproducibility\n","metadata":{}},{"id":"3be51e7f","cell_type":"code","source":"import os, random, math, gc\nimport numpy as np\nimport pandas as pd\n\nSEED = 42\nrandom.seed(SEED)\nnp.random.seed(SEED)\nos.environ[\"PYTHONHASHSEED\"] = str(SEED)\n\nimport matplotlib.pyplot as plt\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:28:40.876375Z","iopub.execute_input":"2026-02-09T19:28:40.876659Z","iopub.status.idle":"2026-02-09T19:28:40.890088Z","shell.execute_reply.started":"2026-02-09T19:28:40.876627Z","shell.execute_reply":"2026-02-09T19:28:40.889452Z"}},"outputs":[],"execution_count":null},{"id":"0b0a2f2d","cell_type":"markdown","source":"## 2) Paths + Load Metadata\n","metadata":{}},{"id":"bb822fc3","cell_type":"code","source":"DATA_DIR = \"/kaggle/input/aptos2019-blindness-detection\"\nTRAIN_CSV = os.path.join(DATA_DIR, \"train.csv\")\nTRAIN_IMG_DIR = os.path.join(DATA_DIR, \"train_images\")\n\n# Where we will write processed images (so transforms stay consistent)\nPROCESSED_DIR = \"/kaggle/working/processed_images\"\nos.makedirs(PROCESSED_DIR, exist_ok=True)\n\ndf = pd.read_csv(TRAIN_CSV)\ndf.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:28:40.891072Z","iopub.execute_input":"2026-02-09T19:28:40.89135Z","iopub.status.idle":"2026-02-09T19:28:40.920887Z","shell.execute_reply.started":"2026-02-09T19:28:40.891298Z","shell.execute_reply":"2026-02-09T19:28:40.920362Z"}},"outputs":[],"execution_count":null},{"id":"12457f3e","cell_type":"markdown","source":"## 3) Quick EDA (Class distribution)\n","metadata":{}},{"id":"945a8374","cell_type":"code","source":"class_counts = df[\"diagnosis\"].value_counts().sort_index()\ndisplay(class_counts)\n\nplt.figure(figsize=(6,4))\nplt.bar(class_counts.index.astype(str), class_counts.values)\nplt.title(\"Class distribution (original)\")\nplt.xlabel(\"diagnosis\")\nplt.ylabel(\"count\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:28:40.921685Z","iopub.execute_input":"2026-02-09T19:28:40.921946Z","iopub.status.idle":"2026-02-09T19:28:41.031027Z","shell.execute_reply.started":"2026-02-09T19:28:40.921925Z","shell.execute_reply":"2026-02-09T19:28:41.030236Z"}},"outputs":[],"execution_count":null},{"id":"7d33dbc6","cell_type":"markdown","source":"## 4) Preprocessing Functions (crop borders + circular crop + Ben Graham)\n","metadata":{}},{"id":"00868160","cell_type":"code","source":"import cv2\n\ndef smart_resize(img_bgr, target_size=512):\n    # keep aspect ratio then pad if needed\n    h, w = img_bgr.shape[:2]\n    scale = target_size / max(h, w)\n    nh, nw = int(h*scale), int(w*scale)\n    img = cv2.resize(img_bgr, (nw, nh), interpolation=cv2.INTER_AREA)\n    # pad to square\n    top = (target_size - nh)//2\n    bottom = target_size - nh - top\n    left = (target_size - nw)//2\n    right = target_size - nw - left\n    img = cv2.copyMakeBorder(img, top, bottom, left, right, cv2.BORDER_CONSTANT, value=(0,0,0))\n    return img\n\ndef crop_circular(img_bgr):\n    # create circular mask around image center\n    h, w = img_bgr.shape[:2]\n    r = min(h, w)//2\n    mask = np.zeros((h, w), np.uint8)\n    cv2.circle(mask, (w//2, h//2), r, 255, -1)\n    out = cv2.bitwise_and(img_bgr, img_bgr, mask=mask)\n    return out\n\ndef ben_graham_preprocess(img_bgr, sigmaX=10):\n    # standard retinal enhancement trick\n    blur = cv2.GaussianBlur(img_bgr, (0,0), sigmaX)\n    out = cv2.addWeighted(img_bgr, 4, blur, -4, 128)\n    return out\n\ndef preprocess_one_image(img_path, target_size=512):\n    img = cv2.imread(img_path)\n    if img is None:\n        raise FileNotFoundError(f\"Could not read: {img_path}\")\n    img = smart_resize(img, target_size=target_size)\n    img = crop_circular(img)\n    img = ben_graham_preprocess(img, sigmaX=10)\n    return img\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:28:41.033218Z","iopub.execute_input":"2026-02-09T19:28:41.033663Z","iopub.status.idle":"2026-02-09T19:28:41.044017Z","shell.execute_reply.started":"2026-02-09T19:28:41.033629Z","shell.execute_reply":"2026-02-09T19:28:41.043383Z"}},"outputs":[],"execution_count":null},{"id":"df7a0953","cell_type":"markdown","source":"## 5) Sanity Check: Before/After on one image\n","metadata":{}},{"id":"dd5307ca","cell_type":"code","source":"sample_id = df.iloc[0][\"id_code\"]\nimg_path = os.path.join(TRAIN_IMG_DIR, f\"{sample_id}.png\")\n\nraw = cv2.imread(img_path)\nproc = preprocess_one_image(img_path)\n\nraw_rgb = cv2.cvtColor(raw, cv2.COLOR_BGR2RGB)\nproc_rgb = cv2.cvtColor(proc, cv2.COLOR_BGR2RGB)\n\nplt.figure(figsize=(10,4))\nplt.subplot(1,2,1); plt.imshow(raw_rgb); plt.title(\"Raw\"); plt.axis(\"off\")\nplt.subplot(1,2,2); plt.imshow(proc_rgb); plt.title(\"Preprocessed\"); plt.axis(\"off\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:28:41.044939Z","iopub.execute_input":"2026-02-09T19:28:41.045152Z","iopub.status.idle":"2026-02-09T19:28:41.97173Z","shell.execute_reply.started":"2026-02-09T19:28:41.045134Z","shell.execute_reply":"2026-02-09T19:28:41.97098Z"}},"outputs":[],"execution_count":null},{"id":"c4efc298","cell_type":"markdown","source":"## 6) Batch Preprocess + Save (one-time)\n","metadata":{}},{"id":"00cdec0c","cell_type":"code","source":"from tqdm.auto import tqdm\n\ndef preprocess_and_save_all(df, src_dir, dst_dir, target_size=512):\n    missing = 0\n    for img_id in tqdm(df[\"id_code\"].values, desc=\"Preprocessing\"):\n        src = os.path.join(src_dir, f\"{img_id}.png\")\n        dst = os.path.join(dst_dir, f\"{img_id}.png\")\n        if os.path.exists(dst):\n            continue  # already done\n        try:\n            img = preprocess_one_image(src, target_size=target_size)\n            cv2.imwrite(dst, img)\n        except Exception:\n            missing += 1\n    print(\"Missing/failed:\", missing)\n\n# Run once (re-run is safe; it skips existing files)\npreprocess_and_save_all(df, TRAIN_IMG_DIR, PROCESSED_DIR, target_size=512)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:28:41.972663Z","iopub.execute_input":"2026-02-09T19:28:41.972944Z","iopub.status.idle":"2026-02-09T19:28:42.012375Z","shell.execute_reply.started":"2026-02-09T19:28:41.972922Z","shell.execute_reply":"2026-02-09T19:28:42.01162Z"}},"outputs":[],"execution_count":null},{"id":"cd53bb97","cell_type":"markdown","source":"## 7) Train/Val/Test Split (stratified — avoid leakage)\n","metadata":{}},{"id":"e5090038","cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_val_df, test_df = train_test_split(\n    df, test_size=0.20, random_state=SEED, stratify=df[\"diagnosis\"]\n)\n\ntrain_df, val_df = train_test_split(\n    train_val_df, test_size=0.20, random_state=SEED, stratify=train_val_df[\"diagnosis\"]\n)\n\nprint(\"train:\", len(train_df), \"val:\", len(val_df), \"test:\", len(test_df))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:28:42.013269Z","iopub.execute_input":"2026-02-09T19:28:42.013569Z","iopub.status.idle":"2026-02-09T19:28:42.027232Z","shell.execute_reply.started":"2026-02-09T19:28:42.013538Z","shell.execute_reply":"2026-02-09T19:28:42.026458Z"}},"outputs":[],"execution_count":null},{"id":"8a010d57","cell_type":"markdown","source":"## 8) DataLoaders (Albumentations)\n","metadata":{}},{"id":"74fc94fa","cell_type":"code","source":"import torch\nfrom torch.utils.data import Dataset, DataLoader\nfrom PIL import Image\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\nMEAN = (0.485, 0.456, 0.406)\nSTD  = (0.229, 0.224, 0.225)\n\nclass APTOSDataset(Dataset):\n    def __init__(self, df, processed_dir, transform=None):\n        self.df = df.reset_index(drop=True)\n        self.processed_dir = processed_dir\n        self.transform = transform\n\n    def __len__(self): \n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = os.path.join(self.processed_dir, f\"{row['id_code']}.png\")\n        img = Image.open(img_path).convert(\"RGB\")\n        img = np.array(img)\n        y = int(row[\"diagnosis\"])\n        if self.transform:\n            img = self.transform(image=img)[\"image\"]\n        return img, torch.tensor(y, dtype=torch.long)\n\ntrain_tfms = A.Compose([\n    A.Resize(256,256),\n    A.CenterCrop(224,224),\n    A.HorizontalFlip(p=0.5),\n    A.RandomBrightnessContrast(0.1,0.1, p=0.3),\n    A.Rotate(limit=5, border_mode=cv2.BORDER_CONSTANT, p=0.3),\n    A.Normalize(mean=MEAN, std=STD),\n    ToTensorV2(),\n])\n\ntest_tfms = A.Compose([\n    A.Resize(256,256),\n    A.CenterCrop(224,224),\n    A.Normalize(mean=MEAN, std=STD),\n    ToTensorV2(),\n])\n\ntrain_ds = APTOSDataset(train_df, PROCESSED_DIR, transform=train_tfms)\nval_ds   = APTOSDataset(val_df,   PROCESSED_DIR, transform=test_tfms)\ntest_ds  = APTOSDataset(test_df,  PROCESSED_DIR, transform=test_tfms)\n\nBATCH_SIZE = 32\nnum_workers = min(4, os.cpu_count() or 2)\n\ntrain_loader = DataLoader(train_ds, batch_size=BATCH_SIZE, shuffle=True,\n                          num_workers=num_workers, pin_memory=True)\nval_loader   = DataLoader(val_ds, batch_size=BATCH_SIZE, shuffle=False,\n                          num_workers=num_workers, pin_memory=True)\ntest_loader  = DataLoader(test_ds, batch_size=BATCH_SIZE, shuffle=False,\n                          num_workers=num_workers, pin_memory=True)\n\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\ndevice\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:28:42.028134Z","iopub.execute_input":"2026-02-09T19:28:42.028384Z","iopub.status.idle":"2026-02-09T19:28:42.049468Z","shell.execute_reply.started":"2026-02-09T19:28:42.028363Z","shell.execute_reply":"2026-02-09T19:28:42.048895Z"}},"outputs":[],"execution_count":null},{"id":"3f695943","cell_type":"markdown","source":"## 9) Feature Extraction (EfficientNet-B3 pretrained)\n","metadata":{}},{"id":"7a00e7c1","cell_type":"code","source":"import torch.nn as nn\nfrom torchvision import models\nfrom tqdm.auto import tqdm\n\nFEATURE_DIR = \"/kaggle/working/efficientnet_b3_features\"\nos.makedirs(FEATURE_DIR, exist_ok=True)\n\ndef extract_efficientnet_b3_features(loader, split_name, save_dir, device):\n    model = models.efficientnet_b3(weights=models.EfficientNet_B3_Weights.IMAGENET1K_V1).to(device)\n    model.eval()\n\n    feature_extractor = nn.Sequential(\n        model.features,\n        nn.AdaptiveAvgPool2d((1,1))\n    ).to(device)\n    feature_extractor.eval()\n\n    feats_list, y_list = [], []\n    with torch.no_grad():\n        for x, y in tqdm(loader, desc=f\"Extract {split_name}\"):\n            x = x.to(device)\n            feats = feature_extractor(x).flatten(1)  # [B, 1536]\n            feats_list.append(feats.cpu().numpy())\n            y_list.append(y.numpy())\n\n    X = np.vstack(feats_list)\n    y = np.hstack(y_list)\n\n    np.save(os.path.join(save_dir, f\"{split_name}_X.npy\"), X)\n    np.save(os.path.join(save_dir, f\"{split_name}_y.npy\"), y)\n    print(split_name, X.shape, y.shape)\n    return X, y\n\nX_train, y_train = extract_efficientnet_b3_features(train_loader, \"train\", FEATURE_DIR, device)\nX_val,   y_val   = extract_efficientnet_b3_features(val_loader,   \"val\",   FEATURE_DIR, device)\nX_test,  y_test  = extract_efficientnet_b3_features(test_loader,  \"test\",  FEATURE_DIR, device)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:28:42.051449Z","iopub.execute_input":"2026-02-09T19:28:42.051714Z","iopub.status.idle":"2026-02-09T19:29:57.627678Z","shell.execute_reply.started":"2026-02-09T19:28:42.051695Z","shell.execute_reply":"2026-02-09T19:29:57.626838Z"}},"outputs":[],"execution_count":null},{"id":"a0bc8276-b026-43a5-a99e-0a047dab5e59","cell_type":"markdown","source":"## 9B) Feature Extraction (DenseNet121 pretrained)\n\nThis section extracts **DenseNet121** pooled features (ImageNet-pretrained) using the **same** loaders/splits/preprocessing, and saves them to a separate directory for fair comparison.\n","metadata":{}},{"id":"a04d3a7a-3b64-4c9f-bd47-21fa93d52bbc","cell_type":"code","source":"import torch.nn as nn\nfrom torchvision import models\nfrom tqdm.auto import tqdm\n\nDENSE_FEATURE_DIR = \"/kaggle/working/densenet121_features\"\nos.makedirs(DENSE_FEATURE_DIR, exist_ok=True)\n\ndef extract_densenet121_features(loader, split_name, save_dir, device):\n    model = models.densenet121(weights=models.DenseNet121_Weights.IMAGENET1K_V1).to(device)\n    model.eval()\n\n    feature_extractor = nn.Sequential(\n        model.features,                    # [B, 1024, H/32, W/32]\n        nn.ReLU(inplace=True),\n        nn.AdaptiveAvgPool2d((1,1))\n    ).to(device)\n    feature_extractor.eval()\n\n    feats_list, y_list = [], []\n    with torch.no_grad():\n        for x, y in tqdm(loader, desc=f\"DenseNet feat {split_name}\"):\n            x = x.to(device)\n            feats = feature_extractor(x).flatten(1)  # [B, 1024]\n            feats_list.append(feats.cpu().numpy())\n            y_list.append(y.numpy())\n\n    X = np.vstack(feats_list).astype(np.float32)\n    y = np.hstack(y_list).astype(np.int64)\n\n    np.save(os.path.join(save_dir, f\"{split_name}_X.npy\"), X)\n    np.save(os.path.join(save_dir, f\"{split_name}_y.npy\"), y)\n    print(f\"[DenseNet121] {split_name}: X={X.shape} y={y.shape}\")\n    return X, y\n\nX_train_dn, y_train_dn = extract_densenet121_features(train_loader, \"train\", DENSE_FEATURE_DIR, device)\nX_val_dn,   y_val_dn   = extract_densenet121_features(val_loader,   \"val\",   DENSE_FEATURE_DIR, device)\nX_test_dn,  y_test_dn  = extract_densenet121_features(test_loader,  \"test\",  DENSE_FEATURE_DIR, device)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:29:57.629369Z","iopub.execute_input":"2026-02-09T19:29:57.629645Z","iopub.status.idle":"2026-02-09T19:31:17.694536Z","shell.execute_reply.started":"2026-02-09T19:29:57.629606Z","shell.execute_reply":"2026-02-09T19:31:17.693594Z"}},"outputs":[],"execution_count":null},{"id":"95429e4c","cell_type":"markdown","source":"## 10) Metrics (Accuracy / F1 / QWK)\n","metadata":{}},{"id":"d346163e","cell_type":"code","source":"from sklearn.metrics import accuracy_score, f1_score, classification_report, confusion_matrix\nfrom sklearn.metrics import cohen_kappa_score\n\ndef eval_metrics(y_true, y_pred):\n    acc = accuracy_score(y_true, y_pred)\n    f1  = f1_score(y_true, y_pred, average=\"macro\")\n    qwk = cohen_kappa_score(y_true, y_pred, weights=\"quadratic\")\n    return {\"acc\": acc, \"f1_macro\": f1, \"qwk\": qwk}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:31:17.696489Z","iopub.execute_input":"2026-02-09T19:31:17.697221Z","iopub.status.idle":"2026-02-09T19:31:17.703335Z","shell.execute_reply.started":"2026-02-09T19:31:17.697181Z","shell.execute_reply":"2026-02-09T19:31:17.702615Z"}},"outputs":[],"execution_count":null},{"id":"f29a235a","cell_type":"markdown","source":"## 11) ML Pipeline (scale → optional PCA → models)\n","metadata":{}},{"id":"4550f2f8","cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\n\n# 1) scale\nscaler = StandardScaler()\nX_train_s = scaler.fit_transform(X_train)\nX_val_s   = scaler.transform(X_val)\nX_test_s  = scaler.transform(X_test)\n\n# 2) optional PCA (keep 95% variance)\npca = PCA(n_components=0.95, random_state=SEED)\nX_train_p = pca.fit_transform(X_train_s)\nX_val_p   = pca.transform(X_val_s)\nX_test_p  = pca.transform(X_test_s)\n\nX_train_p.shape, X_val_p.shape\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:31:17.704351Z","iopub.execute_input":"2026-02-09T19:31:17.704711Z","iopub.status.idle":"2026-02-09T19:31:19.009446Z","shell.execute_reply.started":"2026-02-09T19:31:17.704687Z","shell.execute_reply":"2026-02-09T19:31:19.00615Z"}},"outputs":[],"execution_count":null},{"id":"cd3dd85e-8eba-4d47-b6bc-90efc7c9f76d","cell_type":"markdown","source":"## 12B) Baseline classical models on DenseNet121 features\n\nSame scaler/PCA + same models, trained on **DenseNet121** pooled features for an apples-to-apples comparison.\n","metadata":{}},{"id":"82eb6021-359c-4488-bb49-d4ff17013d68","cell_type":"code","source":"# --- scale (and optional PCA) for DenseNet features ---\nscaler_dn = StandardScaler()\nX_train_dn_s = scaler_dn.fit_transform(X_train_dn)\nX_val_dn_s   = scaler_dn.transform(X_val_dn)\nX_test_dn_s  = scaler_dn.transform(X_test_dn)\n\n# Optional PCA (set N_COMPONENTS to None to disable)\nN_COMPONENTS = 256\nif N_COMPONENTS is not None:\n    pca_dn = PCA(n_components=N_COMPONENTS, random_state=SEED)\n    X_train_dn_p = pca_dn.fit_transform(X_train_dn_s)\n    X_val_dn_p   = pca_dn.transform(X_val_dn_s)\n    X_test_dn_p  = pca_dn.transform(X_test_dn_s)\nelse:\n    X_train_dn_p, X_val_dn_p, X_test_dn_p = X_train_dn_s, X_val_dn_s, X_test_dn_s\n\nresults_dn = {}\nfor name, m in models_dict.items():\n    mm = clone(m)\n    mm.fit(X_train_dn_p, y_train_dn)\n    pred = mm.predict(X_val_dn_p)\n    results_dn[name] = eval_metrics(y_val_dn, pred)\n\nprint(\"DenseNet121 — VAL\")\ndisplay(pd.DataFrame(results_dn).T.sort_values(\"qwk\", ascending=False))\n\n# quick test check with best by qwk\nbest_name_dn = max(results_dn, key=lambda k: results_dn[k][\"qwk\"])\nbest_model_dn = clone(models_dict[best_name_dn]).fit(X_train_dn_p, y_train_dn)\npred_test_dn = best_model_dn.predict(X_test_dn_p)\nprint(\"DenseNet121 — TEST (best model:\", best_name_dn, \")\")\nprint(eval_metrics(y_test_dn, pred_test_dn))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:31:19.010523Z","iopub.execute_input":"2026-02-09T19:31:19.011062Z","iopub.status.idle":"2026-02-09T19:32:28.814878Z","shell.execute_reply.started":"2026-02-09T19:31:19.011035Z","shell.execute_reply":"2026-02-09T19:32:28.814166Z"}},"outputs":[],"execution_count":null},{"id":"7ece16e2","cell_type":"markdown","source":"## 12) Train/Tune a few classical models (baseline)\n","metadata":{}},{"id":"0254390f","cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.ensemble import RandomForestClassifier, ExtraTreesClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom xgboost import XGBClassifier\n\nmodels_dict = {\n    \"logreg\": LogisticRegression(max_iter=2000, n_jobs=-1),\n    \"svm_rbf\": SVC(kernel=\"rbf\", probability=True),\n    \"rf\": RandomForestClassifier(n_estimators=300, random_state=SEED, n_jobs=-1),\n    \"extratrees\": ExtraTreesClassifier(n_estimators=600, random_state=SEED, n_jobs=-1),\n    \"mlp\": MLPClassifier(hidden_layer_sizes=(256,128), max_iter=300, random_state=SEED),\n    \"xgb\": XGBClassifier(\n        n_estimators=600, learning_rate=0.05, max_depth=5,\n        subsample=0.9, colsample_bytree=0.9,\n        reg_lambda=1.0, random_state=SEED, n_jobs=-1,\n        objective=\"multi:softprob\", num_class=5\n    ),\n}\n\nresults = {}\nfor name, m in models_dict.items():\n    m.fit(X_train_p, y_train)\n    pred = m.predict(X_val_p)\n    results[name] = eval_metrics(y_val, pred)\n\npd.DataFrame(results).T.sort_values(\"qwk\", ascending=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:32:28.81593Z","iopub.execute_input":"2026-02-09T19:32:28.816161Z","iopub.status.idle":"2026-02-09T19:34:03.476008Z","shell.execute_reply.started":"2026-02-09T19:32:28.81614Z","shell.execute_reply":"2026-02-09T19:34:03.475281Z"}},"outputs":[],"execution_count":null},{"id":"57ffb4a9","cell_type":"markdown","source":"## 13) Stacking (OOF → Meta-Classifier)\n","metadata":{}},{"id":"cb95685d","cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nfrom sklearn.base import clone\n\nbase_names = [\"xgb\", \"extratrees\", \"mlp\"]\nbase_models = [models_dict[n] for n in base_names]\n\nskf = StratifiedKFold(n_splits=5, shuffle=True, random_state=SEED)\n\ndef oof_predict_proba(models, X, y):\n    oof = np.zeros((len(X), len(np.unique(y))*len(models)))\n    for mi, model in enumerate(models):\n        oof_part = np.zeros((len(X), len(np.unique(y))))\n        for tr_idx, va_idx in skf.split(X, y):\n            m = clone(model)\n            m.fit(X[tr_idx], y[tr_idx])\n            oof_part[va_idx] = m.predict_proba(X[va_idx])\n        oof[:, mi*oof_part.shape[1]:(mi+1)*oof_part.shape[1]] = oof_part\n    return oof\n\nXtr_oof = oof_predict_proba(base_models, X_train_p, y_train)\n\n# fit base on full train for val/test meta-features\ndef concat_proba(models, X_fit, y_fit, X_eval):\n    parts=[]\n    for m in models:\n        mm = clone(m)\n        mm.fit(X_fit, y_fit)\n        parts.append(mm.predict_proba(X_eval))\n    return np.hstack(parts)\n\nXva_meta = concat_proba(base_models, X_train_p, y_train, X_val_p)\nXte_meta = concat_proba(base_models, X_train_p, y_train, X_test_p)\n\nmeta = LogisticRegression(max_iter=3000, n_jobs=-1)\nmeta.fit(Xtr_oof, y_train)\n\nval_pred = meta.predict(Xva_meta)\ntest_pred = meta.predict(Xte_meta)\n\nprint(\"VAL:\", eval_metrics(y_val, val_pred))\nprint(\"TEST:\", eval_metrics(y_test, test_pred))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:34:03.47694Z","iopub.execute_input":"2026-02-09T19:34:03.477569Z","iopub.status.idle":"2026-02-09T19:42:14.495426Z","shell.execute_reply.started":"2026-02-09T19:34:03.477544Z","shell.execute_reply":"2026-02-09T19:42:14.494684Z"}},"outputs":[],"execution_count":null},{"id":"4cf4b1af","cell_type":"markdown","source":"## 14) Reports + Confusion Matrix\n","metadata":{}},{"id":"c31c2726","cell_type":"code","source":"print(classification_report(y_test, test_pred, digits=4))\n\ncm = confusion_matrix(y_test, test_pred)\nplt.figure(figsize=(6,5))\nplt.imshow(cm)\nplt.title(\"Confusion matrix (test)\")\nplt.xlabel(\"Pred\")\nplt.ylabel(\"True\")\nplt.colorbar()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:42:14.496554Z","iopub.execute_input":"2026-02-09T19:42:14.496843Z","iopub.status.idle":"2026-02-09T19:42:14.663772Z","shell.execute_reply.started":"2026-02-09T19:42:14.496804Z","shell.execute_reply":"2026-02-09T19:42:14.663165Z"}},"outputs":[],"execution_count":null},{"id":"a1fdcf50","cell_type":"markdown","source":"## 15) Optional (Recommended): End-to-End Fine-Tuning CNN\n\n- Replace the classifier head to 5 classes\n- Freeze backbone for 1–2 epochs, then unfreeze last blocks\n- Track Accuracy/F1/QWK\n","metadata":{}},{"id":"902a7603-4348-4639-a159-82e939dc713f","cell_type":"code","source":"import torch.nn.functional as F\nfrom sklearn.utils.class_weight import compute_class_weight\n\n# =========================\n# Fine-Tuning (Stronger)\n# - supports EfficientNet-B3 OR DenseNet121\n# - uses class-weights to help minority grades\n# - trains short warmup (head only) then unfreezes backbone\n# =========================\n\nBACKBONE = \"densenet121\"   # \"efficientnet_b3\" or \"densenet121\"\nNUM_CLASSES = 5\nEPOCHS_WARMUP = 2\nEPOCHS_FT = 5\n\n# ---- class weights from TRAIN split ----\nclasses = np.arange(NUM_CLASSES)\ncw = compute_class_weight(class_weight=\"balanced\", classes=classes, y=y_train)\nclass_weights = torch.tensor(cw, dtype=torch.float32, device=device)\nprint(\"Class weights:\", cw)\n\ndef build_model(backbone: str, num_classes: int = 5):\n    if backbone == \"efficientnet_b3\":\n        m = models.efficientnet_b3(weights=models.EfficientNet_B3_Weights.IMAGENET1K_V1)\n        in_f = m.classifier[1].in_features\n        m.classifier[1] = nn.Linear(in_f, num_classes)\n        backbone_params = m.features.parameters()\n        return m, backbone_params\n\n    if backbone == \"densenet121\":\n        m = models.densenet121(weights=models.DenseNet121_Weights.IMAGENET1K_V1)\n        in_f = m.classifier.in_features\n        m.classifier = nn.Sequential(\n            nn.Dropout(0.2),\n            nn.Linear(in_f, num_classes)\n        )\n        backbone_params = m.features.parameters()\n        return m, backbone_params\n\n    raise ValueError(\"Unknown backbone: \" + backbone)\n\ndef loss_fn(logits, y):\n    # weighted CE (+ optional label smoothing)\n    return F.cross_entropy(logits, y, weight=class_weights, label_smoothing=0.05)\n\ndef train_one_epoch(model, loader, opt):\n    model.train()\n    total_loss = 0.0\n    ys, ps = [], []\n    for x, y in loader:\n        x, y = x.to(device), y.to(device)\n        opt.zero_grad(set_to_none=True)\n        logits = model(x)\n        loss = loss_fn(logits, y)\n        loss.backward()\n        torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0)\n        opt.step()\n\n        total_loss += loss.item() * len(x)\n        ys.append(y.detach().cpu().numpy())\n        ps.append(logits.argmax(1).detach().cpu().numpy())\n\n    y_true = np.hstack(ys); y_pred = np.hstack(ps)\n    return total_loss / len(loader.dataset), eval_metrics(y_true, y_pred)\n\n@torch.no_grad()\ndef eval_epoch(model, loader):\n    model.eval()\n    total_loss = 0.0\n    ys, ps = [], []\n    for x, y in loader:\n        x, y = x.to(device), y.to(device)\n        logits = model(x)\n        loss = loss_fn(logits, y)\n        total_loss += loss.item() * len(x)\n        ys.append(y.detach().cpu().numpy())\n        ps.append(logits.argmax(1).detach().cpu().numpy())\n    y_true = np.hstack(ys); y_pred = np.hstack(ps)\n    return total_loss / len(loader.dataset), eval_metrics(y_true, y_pred)\n\n# ---- build + warmup head only ----\nmodel_ft, backbone_params = build_model(BACKBONE, NUM_CLASSES)\nmodel_ft = model_ft.to(device)\n\nfor p in backbone_params:\n    p.requires_grad = False\n\nopt = torch.optim.AdamW(\n    filter(lambda p: p.requires_grad, model_ft.parameters()),\n    lr=2e-4, weight_decay=1e-2\n)\n\nbest_qwk = -1.0\nbest_path = f\"/kaggle/working/best_{BACKBONE}_ft.pt\"\n\nfor epoch in range(EPOCHS_WARMUP):\n    tr_loss, tr_m = train_one_epoch(model_ft, train_loader, opt)\n    va_loss, va_m = eval_epoch(model_ft, val_loader)\n    print(f\"[WARMUP {epoch}] train loss={tr_loss:.4f} {tr_m} | val loss={va_loss:.4f} {va_m}\")\n\n# ---- fine-tune: unfreeze all (simple + effective) ----\nfor p in model_ft.parameters():\n    p.requires_grad = True\n\nopt = torch.optim.AdamW(model_ft.parameters(), lr=5e-5, weight_decay=1e-2)\n\n# ✅ FIX: remove verbose (some torch versions don't support it)\nscheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(\n    opt, mode=\"max\", factor=0.5, patience=1\n)\n\nfor epoch in range(EPOCHS_FT):\n    tr_loss, tr_m = train_one_epoch(model_ft, train_loader, opt)\n    va_loss, va_m = eval_epoch(model_ft, val_loader)\n\n    scheduler.step(va_m[\"qwk\"])\n\n    print(f\"[FT {epoch}] train loss={tr_loss:.4f} {tr_m} | val loss={va_loss:.4f} {va_m}\")\n\n    if float(va_m[\"qwk\"]) > best_qwk:\n        best_qwk = float(va_m[\"qwk\"])\n        torch.save(model_ft.state_dict(), best_path)\n        print(\"Saved best:\", best_path, \"qwk=\", best_qwk)\n\n# ---- evaluate best on test ----\nmodel_ft.load_state_dict(torch.load(best_path, map_location=device))\ntest_loss, test_m = eval_epoch(model_ft, test_loader)\nprint(\"BEST VAL QWK:\", best_qwk)\nprint(\"TEST:\", test_m)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:48:41.434419Z","iopub.execute_input":"2026-02-09T19:48:41.435359Z","iopub.status.idle":"2026-02-09T19:50:25.83573Z","shell.execute_reply.started":"2026-02-09T19:48:41.435293Z","shell.execute_reply":"2026-02-09T19:50:25.834824Z"}},"outputs":[],"execution_count":null},{"id":"4b39f60d-0fe8-40db-9311-d7592751f384","cell_type":"code","source":"# =========================\n# Classification Report + Confusion Matrix (TEST)\n# =========================\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport torch\n\n@torch.no_grad()\ndef predict_loader(model, loader):\n    model.eval()\n    ys, ps = [], []\n    for x, y in loader:\n        x = x.to(device)\n        logits = model(x)\n        pred = logits.argmax(1).detach().cpu().numpy()\n        ys.append(y.numpy())\n        ps.append(pred)\n    return np.hstack(ys), np.hstack(ps)\n\n# ---- predictions on TEST ----\ny_true_test, y_pred_test = predict_loader(model_ft, test_loader)\n\nprint(\"\\n=== TEST Classification Report ===\")\nprint(classification_report(y_true_test, y_pred_test, digits=4))\n\n# ---- confusion matrix ----\ncm = confusion_matrix(y_true_test, y_pred_test)\nprint(\"\\n=== TEST Confusion Matrix ===\")\nprint(cm)\n\n# ---- heatmap (اختياري بس مفيد) ----\nplt.figure(figsize=(6,5))\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\",\n            xticklabels=[0,1,2,3,4],\n            yticklabels=[0,1,2,3,4])\nplt.xlabel(\"Predicted\")\nplt.ylabel(\"True\")\nplt.title(\"Confusion Matrix - TEST\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-09T19:57:42.131968Z","iopub.execute_input":"2026-02-09T19:57:42.13229Z","iopub.status.idle":"2026-02-09T19:57:45.067242Z","shell.execute_reply.started":"2026-02-09T19:57:42.132259Z","shell.execute_reply":"2026-02-09T19:57:45.066546Z"}},"outputs":[],"execution_count":null},{"id":"2e642a8d-ec35-4fe6-b907-ea8a0f0fc1e8","cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}