{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"accelerator":"TPU","colab":{"gpuType":"V5E1","machine_shape":"hm","runtime_attributes":{"runtime_version":"2026.04"},"provenance":[]}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"NBnqsJJXkIFY","cell_type":"markdown","source":"### Step 1.1 — Mount Google Drive (Checkpoints + Logs Only)","metadata":{"id":"NBnqsJJXkIFY"}},{"id":"d1effdd0-ffd2-44d2-885b-2747faaa11c6","cell_type":"code","source":"import os\nprint(os.listdir('/kaggle/input/competitions/intel-mobileodt-cervical-cancer-screening'))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"eYCzJN83kIFZ","cell_type":"code","source":"# Kaggle Notebooks: no Drive mount needed — dataset is pre-mounted\nprint(\"✅ Running on Kaggle Notebooks\")\nprint(\"   Dataset : /kaggle/input/competitions/intel-mobileodt-cervical-cancer-screening/\")\nprint(\"   Outputs : /kaggle/working/\")","metadata":{"id":"eYCzJN83kIFZ","outputId":"634439d3-9957-479c-9ebf-4eef09752e32","trusted":true,"execution":{"iopub.status.busy":"2026-07-06T07:27:51.832898Z","iopub.execute_input":"2026-07-06T07:27:51.833406Z","iopub.status.idle":"2026-07-06T07:27:51.842684Z","shell.execute_reply.started":"2026-07-06T07:27:51.83337Z","shell.execute_reply":"2026-07-06T07:27:51.84196Z"}},"outputs":[],"execution_count":null},{"id":"v2AhCF9-kIFa","cell_type":"markdown","source":"### Step 1.2 — GPU Memory Growth","metadata":{"id":"v2AhCF9-kIFa"}},{"id":"7dPmAFvUkIFa","cell_type":"code","source":"import tensorflow as tf\n\ngpus = tf.config.list_physical_devices('GPU')\nif gpus:\n    for gpu in gpus:\n        tf.config.experimental.set_memory_growth(gpu, True)\n    print(f\"✅ GPU memory growth enabled on {len(gpus)} GPU(s)\")\nelse:\n    print(\"⚠️ No GPU — check runtime settings!\")","metadata":{"id":"7dPmAFvUkIFa","outputId":"1c8e9c9c-b959-423e-cffd-c98fa25a03c6","trusted":true,"execution":{"iopub.status.busy":"2026-07-06T07:27:57.789784Z","iopub.execute_input":"2026-07-06T07:27:57.790529Z","iopub.status.idle":"2026-07-06T07:28:15.024648Z","shell.execute_reply.started":"2026-07-06T07:27:57.790497Z","shell.execute_reply":"2026-07-06T07:28:15.023578Z"}},"outputs":[],"execution_count":null},{"id":"2vX0fF1IkIFa","cell_type":"markdown","source":"### Step 1.3 — Import All Libraries","metadata":{"id":"2vX0fF1IkIFa"}},{"id":"BQY3N12AkIFb","cell_type":"code","source":"import os, glob, shutil, stat, random, warnings, cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.gridspec as gridspec\nimport seaborn as sns\nfrom PIL import Image, ImageFile\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.cluster import KMeans                     # 🏆 For cluster-based validation split\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.preprocessing import label_binarize\nfrom sklearn.metrics import (\n    classification_report, confusion_matrix,\n    roc_curve, auc\n)\nfrom tensorflow import keras\nfrom tensorflow.keras import layers, Model\nfrom tensorflow.keras.applications import EfficientNetB3, ResNet50, InceptionV3\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.callbacks import (\n    EarlyStopping, ModelCheckpoint,\n    ReduceLROnPlateau, TensorBoard\n)\nfrom tensorflow.keras.optimizers import Adam\n\nwarnings.filterwarnings('ignore')\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '2'\nprint(f\"✅ Libraries loaded | TF: {tf.__version__}\")","metadata":{"id":"BQY3N12AkIFb","outputId":"d0ac35f2-9e24-4da3-823b-26e4a18e411c","trusted":true,"execution":{"iopub.status.busy":"2026-07-06T07:29:02.890291Z","iopub.execute_input":"2026-07-06T07:29:02.891463Z","iopub.status.idle":"2026-07-06T07:29:02.898865Z","shell.execute_reply.started":"2026-07-06T07:29:02.891428Z","shell.execute_reply":"2026-07-06T07:29:02.89793Z"}},"outputs":[],"execution_count":null},{"id":"l1-UYi_qkIFc","cell_type":"markdown","source":"### Step 1.4 — Global Constants","metadata":{"id":"l1-UYi_qkIFc"}},{"id":"EJV14lzZkIFc","cell_type":"code","source":"# ── Kaggle Notebooks paths ────────────────────────────────────────\nKAGGLE_INPUT   = '/kaggle/input/competitions/intel-mobileodt-cervical-cancer-screening'\nKAGGLE_WORKING = '/kaggle/working'\n\nBASE_DIR        = os.path.join(KAGGLE_WORKING, 'cervix_project')\nCHECKPOINT_DIR  = os.path.join(BASE_DIR, 'checkpoints')\nLOG_DIR         = os.path.join(BASE_DIR, 'logs')\n\n# ── Dataset paths (pre-mounted, no download needed) ───────────────\nDATASET_DIR     = KAGGLE_INPUT\nTRAIN_DIR       = os.path.join(DATASET_DIR, 'train', 'train')\n\n# ── Model Hyperparameters ─────────────────────────────────────────\nIMAGE_SIZE      = (300, 300)\nCROP_SIZE       = (224, 224)\nBATCH_SIZE      = 32\nNUM_CLASSES     = 3\nN_FOLDS         = 5\nSEED            = 42\nCLASS_NAMES     = ['Type_1', 'Type_2', 'Type_3']\n\nfor d in [CHECKPOINT_DIR, LOG_DIR]:\n    os.makedirs(d, exist_ok=True)\n\nprint(\"✅ Constants defined\")\nprint(f\"   Dataset  → {DATASET_DIR}  (pre-mounted)\")\nprint(f\"   Checkpts → {CHECKPOINT_DIR}\")","metadata":{"id":"EJV14lzZkIFc","outputId":"320f1e12-b7e6-4a96-e5fb-a18b31b8f7eb","trusted":true,"execution":{"iopub.status.busy":"2026-07-06T07:29:03.680676Z","iopub.execute_input":"2026-07-06T07:29:03.681256Z","iopub.status.idle":"2026-07-06T07:29:03.68907Z","shell.execute_reply.started":"2026-07-06T07:29:03.681228Z","shell.execute_reply":"2026-07-06T07:29:03.688038Z"}},"outputs":[],"execution_count":null},{"id":"7k8oLqS1kIFc","cell_type":"markdown","source":"### Step 2.1 — Install kagglehub","metadata":{"id":"7k8oLqS1kIFc"}},{"id":"Baky90XlkIFd","cell_type":"code","source":"# Kaggle Notebooks: kaggle package is pre-installed, nothing to install\nprint(\"✅ Kaggle environment ready\")","metadata":{"id":"Baky90XlkIFd","trusted":true,"execution":{"iopub.status.busy":"2026-07-06T07:29:04.420505Z","iopub.execute_input":"2026-07-06T07:29:04.420985Z","iopub.status.idle":"2026-07-06T07:29:04.425336Z","shell.execute_reply.started":"2026-07-06T07:29:04.420955Z","shell.execute_reply":"2026-07-06T07:29:04.424665Z"}},"outputs":[],"execution_count":null},{"id":"uzkJMB0lkIFd","cell_type":"markdown","source":"### Step 2.2 — Configure Kaggle Credentials","metadata":{"id":"uzkJMB0lkIFd"}},{"id":"_yD3931wkIFd","cell_type":"code","source":"# Kaggle Notebooks: credentials are pre-configured — no upload needed\nprint(\"✅ Kaggle credentials pre-configured\")","metadata":{"id":"_yD3931wkIFd","outputId":"b27231a4-9f66-4376-fcd2-23ea4d83b541","trusted":true,"execution":{"iopub.status.busy":"2026-06-23T05:49:55.929125Z","iopub.execute_input":"2026-06-23T05:49:55.929392Z","iopub.status.idle":"2026-06-23T05:49:55.940553Z","shell.execute_reply.started":"2026-06-23T05:49:55.929364Z","shell.execute_reply":"2026-06-23T05:49:55.939785Z"}},"outputs":[],"execution_count":null},{"id":"CPb0jV5YkIFd","cell_type":"markdown","source":"### Step 2.3 — Download to Runtime","metadata":{"id":"CPb0jV5YkIFd"}},{"id":"-SI2gWOskIFe","cell_type":"code","source":"# ── Auto-detect the real TRAIN_DIR from Kaggle's mounted structure ──\n# Kaggle can mount data in two ways depending on the competition zip layout:\n#   Option A: /kaggle/input/.../train/train/Type_1/   (double-nested)\n#   Option B: /kaggle/input/.../train/Type_1/          (single-nested)\n# We try both and update TRAIN_DIR to whichever actually exists.\n\npossible_train_dirs = [\n    os.path.join(KAGGLE_INPUT, 'train', 'train'),  # double-nested (original zip)\n    os.path.join(KAGGLE_INPUT, 'train'),            # single-nested (Kaggle mount)\n]\n\nTRAIN_DIR = None\nfor candidate in possible_train_dirs:\n    if os.path.exists(os.path.join(candidate, 'Type_1')):\n        TRAIN_DIR = candidate\n        break\n\nif TRAIN_DIR:\n    print(f\"✅ Dataset found at: {TRAIN_DIR}\")\n    for cls in CLASS_NAMES:\n        folder = os.path.join(TRAIN_DIR, cls)\n        n = len(os.listdir(folder)) if os.path.exists(folder) else 0\n        print(f\"   {cls}: {n} images\")\n    print(\"\\n   ℹ️  test/ folder is ignored — only train/ and additional_*/ are read\")\nelse:\n    # Show what IS in the input dir so the user can debug\n    print(f\"❌ Could not find Type_1/Type_2/Type_3 under any known path.\")\n    print(f\"   Contents of {KAGGLE_INPUT}:\")\n    for item in sorted(os.listdir(KAGGLE_INPUT)):\n        print(f\"      {item}/\")\n    print(\"\\n   Fix: click '+ Add Data' → search 'intel mobileodt cervical cancer' → Add\")\n    raise FileNotFoundError(\"Training data not found. See message above.\")","metadata":{"id":"-SI2gWOskIFe","outputId":"c783114d-dcb0-40df-c2e0-775483bdea57","trusted":true,"execution":{"iopub.status.busy":"2026-06-23T05:49:55.941597Z","iopub.execute_input":"2026-06-23T05:49:55.94192Z","iopub.status.idle":"2026-06-23T05:49:56.019906Z","shell.execute_reply.started":"2026-06-23T05:49:55.941891Z","shell.execute_reply":"2026-06-23T05:49:56.0193Z"}},"outputs":[],"execution_count":null},{"id":"7kPJx-6rkIFe","cell_type":"markdown","source":"### Step 3.1 — Build Master File List + Controlled Additional Sampling","metadata":{"id":"7kPJx-6rkIFe"}},{"id":"gGb7eNP9kIFe","cell_type":"code","source":"# ── Main training images ──────────────────────────────────────────\nmain_records = []\nfor class_name in CLASS_NAMES:\n    folder = os.path.join(TRAIN_DIR, class_name)\n    for fpath in glob.glob(os.path.join(folder, '*.jpg')):\n        main_records.append({\n            'filepath': fpath,\n            'filename': os.path.basename(fpath),\n            'original_label': class_name,\n            'source': 'main'\n        })\n\nmain_df = pd.DataFrame(main_records)\nprint(f\"Main images: {len(main_df)}\")\nprint(main_df['original_label'].value_counts().to_string())\n\n# ── Additional images — try multiple folder name patterns ─────────\ndef find_additional_folder(dataset_dir, class_name):\n    \"\"\"Kaggle dataset uses 'additional_Type_X'; Colab-extracted zip used '_v2' suffix.\"\"\"\n    candidates = [\n        os.path.join(dataset_dir, f'additional_{class_name}', class_name),\n        os.path.join(dataset_dir, f'additional_{class_name}_v2', class_name),\n        os.path.join(dataset_dir, 'additional', class_name),\n    ]\n    for p in candidates:\n        if os.path.exists(p):\n            return p\n    return None\n\nadd_records = []\nfor class_name in CLASS_NAMES:\n    add_folder = find_additional_folder(DATASET_DIR, class_name)\n    if add_folder:\n        all_add = glob.glob(os.path.join(add_folder, '*.jpg'))\n        main_count = len(main_df[main_df['original_label'] == class_name])\n        cap = main_count * 2\n        sampled = random.sample(all_add, min(len(all_add), cap))\n        for fpath in sampled:\n            add_records.append({\n                'filepath': fpath,\n                'filename': os.path.basename(fpath),\n                'original_label': class_name,\n                'source': 'additional'\n            })\n        print(f\"  additional_{class_name}: {len(all_add)} available → {len(sampled)} sampled (cap={cap})\")\n    else:\n        print(f\"  ⚠️ No additional data found for {class_name} — using main only\")\n\nadd_df   = pd.DataFrame(add_records) if add_records else pd.DataFrame()\nimage_df = pd.concat([main_df, add_df], ignore_index=True) if len(add_df) > 0 else main_df.copy()\n\nprint(f\"\\n✅ Total images: {len(image_df)}\")\nprint(f\"   Main: {len(main_df)} | Additional (sampled): {len(add_df)}\")","metadata":{"id":"gGb7eNP9kIFe","outputId":"bdeb4bdb-5418-43eb-bdaf-027c1c01af97","trusted":true,"execution":{"iopub.status.busy":"2026-06-23T05:49:56.020642Z","iopub.execute_input":"2026-06-23T05:49:56.021041Z","iopub.status.idle":"2026-06-23T05:49:56.192627Z","shell.execute_reply.started":"2026-06-23T05:49:56.021019Z","shell.execute_reply":"2026-06-23T05:49:56.1918Z"}},"outputs":[],"execution_count":null},{"id":"rA0j3I3mkIFe","cell_type":"markdown","source":"### Step 3.2 — Load and Merge Corrected Medical Labels","metadata":{"id":"rA0j3I3mkIFe"}},{"id":"PmMA90hDkIFe","cell_type":"code","source":"# Load corrected labels if available; fall back to original Kaggle labels\nCSV_CANDIDATES = [\n    os.path.join(DATASET_DIR, 'fixed_labels_v2.csv'),  # if uploaded as dataset\n    os.path.join(BASE_DIR, 'fixed_labels_v2.csv'),      # if saved to working dir\n]\ncsv_found = next((p for p in CSV_CANDIDATES if os.path.exists(p)), None)\n\nif csv_found:\n    labels_df = pd.read_csv(csv_found)\n    master_df  = image_df.merge(labels_df, on='filename', how='left')\n    master_df['final_label'] = master_df['new_label'].fillna(master_df['original_label'])\n    corrected = master_df['new_label'].notna().sum()\n    print(f\"✅ {corrected} labels corrected by medical review ({csv_found})\")\nelse:\n    master_df = image_df.copy()\n    master_df['final_label'] = master_df['original_label']\n    print(\"ℹ️ No corrected labels CSV found — using original Kaggle labels\")\n    print(\"   To add: upload fixed_labels_v2.csv as a Kaggle dataset and attach it\")\n\nprint(master_df['final_label'].value_counts())","metadata":{"id":"PmMA90hDkIFe","outputId":"717f5fc8-31d2-4133-9977-0eed1401a674","trusted":true,"execution":{"iopub.status.busy":"2026-06-23T05:49:56.195207Z","iopub.execute_input":"2026-06-23T05:49:56.195777Z","iopub.status.idle":"2026-06-23T05:49:56.221207Z","shell.execute_reply.started":"2026-06-23T05:49:56.195753Z","shell.execute_reply":"2026-06-23T05:49:56.220606Z"}},"outputs":[],"execution_count":null},{"id":"YFkL7xWykIFf","cell_type":"markdown","source":"### Step 3.3 — Blur Detection Filter","metadata":{"id":"YFkL7xWykIFf"}},{"id":"uLmHfMPHkIFf","cell_type":"code","source":"ImageFile.LOAD_TRUNCATED_IMAGES = True   # PIL handles truncated JPEGs gracefully\n\ndef laplacian_variance(path):\n    \"\"\"\n    Measures image sharpness via Laplacian variance.\n    Uses PIL to open (silently handles 'Premature end of JPEG file' files),\n    then passes the numpy array to cv2.Laplacian for the computation.\n    \"\"\"\n    try:\n        with warnings.catch_warnings():\n            warnings.simplefilter(\"ignore\")\n            with Image.open(path) as img:\n                gray = np.array(img.convert('L'), dtype=np.uint8)\n        return cv2.Laplacian(gray, cv2.CV_64F).var()\n    except Exception:\n        return 0.0   # Treat unreadable images as blurry → they get filtered out\n\nprint(\"Computing sharpness scores (this takes 2–3 mins for full dataset)...\")\nmaster_df['sharpness'] = master_df['filepath'].apply(laplacian_variance)\n\n# Plot sharpness distribution\nfig, axes = plt.subplots(1, 2, figsize=(14, 4))\nmaster_df['sharpness'].hist(bins=50, ax=axes[0], color='steelblue', edgecolor='black')\naxes[0].axvline(100, color='red', linestyle='--', label='Threshold (100)')\naxes[0].set_title('Image Sharpness Distribution (Laplacian Variance)')\naxes[0].set_xlabel('Sharpness Score')\naxes[0].legend()\n\nmaster_df.groupby('source')['sharpness'].hist(\n    bins=40, alpha=0.6, ax=axes[1], label=master_df['source'].unique()\n)\naxes[1].set_title('Sharpness by Source: Main vs Additional')\naxes[1].legend(['Main', 'Additional'])\nplt.tight_layout()\nplt.savefig(os.path.join(LOG_DIR, 'sharpness_distribution.png'), dpi=150)\nplt.show()\n\n# Apply filter — remove bottom 25% by sharpness\nSHARPNESS_THRESHOLD = master_df['sharpness'].quantile(0.25)\nprint(f\"\\nSharpness threshold (25th percentile): {SHARPNESS_THRESHOLD:.1f}\")\nprint(f\"Images before filter: {len(master_df)}\")\n\nmaster_df = master_df[master_df['sharpness'] >= SHARPNESS_THRESHOLD].reset_index(drop=True)\nprint(f\"Images after filter:  {len(master_df)}\")\nprint(f\"Removed: {len(image_df) - len(master_df)} blurry/truncated images\")","metadata":{"id":"uLmHfMPHkIFf","outputId":"9ad35119-9912-4601-ca10-8f202eac2bb3","trusted":true,"execution":{"iopub.status.busy":"2026-06-23T05:49:56.222272Z","iopub.execute_input":"2026-06-23T05:49:56.222588Z","iopub.status.idle":"2026-06-23T06:07:25.313761Z","shell.execute_reply.started":"2026-06-23T05:49:56.222567Z","shell.execute_reply":"2026-06-23T06:07:25.313177Z"}},"outputs":[],"execution_count":null},{"id":"FPyLNltIkIFf","cell_type":"markdown","source":"### Step 3.4 — Corruption Scan","metadata":{"id":"FPyLNltIkIFf"}},{"id":"v6-4_Pc_kIFf","cell_type":"code","source":"corrupted = []\nfor _, row in master_df.iterrows():\n    try:\n        with Image.open(row['filepath']) as img:\n            img.verify()\n    except Exception:\n        corrupted.append(row['filepath'])\n\nprint(f\"✅ {len(corrupted)} corrupted images removed\")\nmaster_df = master_df[~master_df['filepath'].isin(corrupted)].reset_index(drop=True)\nprint(f\"   Final clean dataset: {len(master_df)} images\")","metadata":{"id":"v6-4_Pc_kIFf","outputId":"81b62821-d758-4d43-f300-e2991ea3d6f3","trusted":true,"execution":{"iopub.status.busy":"2026-06-23T06:07:25.314775Z","iopub.execute_input":"2026-06-23T06:07:25.315155Z","iopub.status.idle":"2026-06-23T06:07:29.656137Z","shell.execute_reply.started":"2026-06-23T06:07:25.315132Z","shell.execute_reply":"2026-06-23T06:07:29.655253Z"}},"outputs":[],"execution_count":null},{"id":"pn_3i1kekIFf","cell_type":"markdown","source":"### Step 3.5 — Plot Class Distribution","metadata":{"id":"pn_3i1kekIFf"}},{"id":"jfDdD9ylkIFf","cell_type":"code","source":"counts = master_df['final_label'].value_counts()\ncolors = ['#2196F3', '#4CAF50', '#FF5722']\n\nfig, axes = plt.subplots(1, 2, figsize=(14, 5))\naxes[0].bar(counts.index, counts.values, color=colors, edgecolor='black')\naxes[0].set_title('Class Distribution — Counts', fontweight='bold')\nfor i, v in enumerate(counts.values):\n    axes[0].text(i, v + 5, str(v), ha='center', fontweight='bold')\n\naxes[1].pie(counts.values, labels=counts.index, autopct='%1.1f%%',\n            colors=colors, startangle=90)\naxes[1].set_title('Class Distribution — Proportion', fontweight='bold')\n\nplt.suptitle('Dataset After Blur Filter and Ratio Cap', fontweight='bold')\nplt.tight_layout()\nplt.savefig(os.path.join(LOG_DIR, 'class_distribution.png'), dpi=150)\nplt.show()\nprint(f\"Imbalance ratio: {counts.max()/counts.min():.2f}×\")","metadata":{"id":"jfDdD9ylkIFf","outputId":"0a8fa159-29dc-48a3-caec-7cac8a4b607f","trusted":true,"execution":{"iopub.status.busy":"2026-06-23T06:07:29.657133Z","iopub.execute_input":"2026-06-23T06:07:29.657346Z","iopub.status.idle":"2026-06-23T06:07:30.053813Z","shell.execute_reply.started":"2026-06-23T06:07:29.657327Z","shell.execute_reply":"2026-06-23T06:07:30.053141Z"}},"outputs":[],"execution_count":null},{"id":"AntnHVlzkIFg","cell_type":"markdown","source":"### Step 3.6 — Visual Sanity Check","metadata":{"id":"AntnHVlzkIFg"}},{"id":"VN_r4k_4kIFg","cell_type":"code","source":"fig, axes = plt.subplots(3, 5, figsize=(18, 12))\nfig.suptitle('Visual Sanity Check — 5 Samples Per Type', fontsize=16, fontweight='bold')\nfor row_idx, class_name in enumerate(CLASS_NAMES):\n    samples = master_df[master_df['final_label'] == class_name].sample(5, random_state=SEED)\n    for col_idx, (_, row) in enumerate(samples.iterrows()):\n        img = Image.open(row['filepath']).convert('RGB').resize((200, 200))\n        axes[row_idx, col_idx].imshow(img)\n        axes[row_idx, col_idx].set_title(\n            f\"{class_name}\\nsharpness:{row['sharpness']:.0f}\", fontsize=7)\n        axes[row_idx, col_idx].axis('off')\nplt.tight_layout()\nplt.savefig(os.path.join(LOG_DIR, 'sample_images.png'), dpi=150)\nplt.show()","metadata":{"id":"VN_r4k_4kIFg","outputId":"6e1da272-801b-40fa-f088-0e929c2343fb","trusted":true,"execution":{"iopub.status.busy":"2026-06-23T06:07:30.054715Z","iopub.execute_input":"2026-06-23T06:07:30.055048Z","iopub.status.idle":"2026-06-23T06:07:37.739046Z","shell.execute_reply.started":"2026-06-23T06:07:30.055023Z","shell.execute_reply":"2026-06-23T06:07:37.737721Z"}},"outputs":[],"execution_count":null},{"id":"6Rv3ZArrkIFg","cell_type":"markdown","source":"### Step 4.1 — Understand the Problem Visually","metadata":{"id":"6Rv3ZArrkIFg"}},{"id":"fdRzILtJkIFg","cell_type":"markdown","source":"### Step 4.2 — Implement Smart Cervix Crop Function","metadata":{"id":"fdRzILtJkIFg"}},{"id":"T4M4oLEWkIFg","cell_type":"code","source":"# Show WHY full-image classification fails — visualize background noise\nsample_paths = master_df.sample(4, random_state=SEED)['filepath'].values\n\nfig, axes = plt.subplots(1, 4, figsize=(18, 5))\nfig.suptitle('Why Full-Image Classification Fails:\\nSpeculum, gloves, and lighting dominate the image',\n             fontsize=12, fontweight='bold')\nfor ax, path in zip(axes, sample_paths):\n    img = Image.open(path).convert('RGB')\n    ax.imshow(img)\n    ax.axis('off')\nplt.tight_layout()\nplt.show()","metadata":{"id":"T4M4oLEWkIFg","outputId":"76f81fc3-4d1e-4632-fec8-9365d90761ba","trusted":true,"execution":{"iopub.status.busy":"2026-06-23T06:07:37.740576Z","iopub.execute_input":"2026-06-23T06:07:37.740983Z","iopub.status.idle":"2026-06-23T06:07:40.651837Z","shell.execute_reply.started":"2026-06-23T06:07:37.740942Z","shell.execute_reply":"2026-06-23T06:07:40.651053Z"}},"outputs":[],"execution_count":null},{"id":"k30EFuWzkIFg","cell_type":"code","source":"def crop_cervix(img_path, output_size=CROP_SIZE):\n    \"\"\"\n    Two-stage cervix localization:\n    Stage 1: Centre crop (removes most speculum/border noise)\n    Stage 2: Brightest region crop (the cervical os is typically the\n             brightest central structure in colposcopy images)\n\n    Returns a PIL Image cropped to the region of interest.\n\n    WHY centre crop first:\n    The EVA colposcope (used to take these images) always positions\n    the cervix near the image centre. A 70% centre crop removes the\n    outer ring which is almost always speculum metal or clinician hands.\n\n    WHY brightest region:\n    Acetowhite reaction (the diagnostic marker for SCJ type) is bright.\n    The speculum, once cropped out, is darker than the tissue.\n    \"\"\"\n    img = cv2.imread(img_path)\n    if img is None:\n        return Image.fromarray(np.zeros((*output_size, 3), dtype=np.uint8))\n\n    h, w = img.shape[:2]\n\n    # ── Stage 1: Centre crop to 90% of image (less aggressive) ──\n    # Increased margin to 0.05 from 0.15, meaning 90% center crop\n    margin_h = int(h * 0.05)\n    margin_w = int(w * 0.05)\n    centre_crop = img[margin_h:h-margin_h, margin_w:w-margin_w]\n\n    # Handle cases where centre_crop becomes too small or invalid\n    if centre_crop.shape[0] < output_size[0] or centre_crop.shape[1] < output_size[1]:\n        roi_rgb = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        return Image.fromarray(roi_rgb).resize(output_size)\n\n    # ── Stage 2: Find brightest sub-region (SCJ localization) ─────\n    gray   = cv2.cvtColor(centre_crop, cv2.COLOR_BGR2GRAY)\n    ch, cw = gray.shape\n\n    # Use a sliding window of 70% of the cropped image size (increased from 50%)\n    win_h, win_w = int(ch * 0.7), int(cw * 0.7)\n\n    # Ensure window size is not larger than image dimensions\n    if win_h > ch: win_h = ch\n    if win_w > cw: win_w = cw\n\n    best_sum, best_y, best_x = 0, 0, 0\n    roi = centre_crop # Default to centre_crop if no suitable window is found\n\n    # Ensure there's space for the sliding window\n    if ch - win_h > 0 and cw - win_w > 0:\n        step = 10  # Stride of 10px for finer search (reduced from 20px)\n        for y in range(0, ch - win_h, step):\n            for x in range(0, cw - win_w, step):\n                window_sum = gray[y:y+win_h, x:x+win_w].sum()\n                if window_sum > best_sum:\n                    best_sum = window_sum\n                    best_y, best_x = y, x\n        roi = centre_crop[best_y:best_y+win_h, best_x:best_x+win_w]\n\n    # If roi still has invalid dimensions, use centre_crop resized\n    if roi.shape[0] == 0 or roi.shape[1] == 0:\n        roi = centre_crop\n\n    # Convert BGR→RGB and resize to target size\n    roi_rgb = cv2.cvtColor(roi, cv2.COLOR_BGR2RGB)\n    pil_img = Image.fromarray(roi_rgb).resize(output_size)\n    return pil_img\n\n\n# Test on a few samples\nfig, axes = plt.subplots(3, 6, figsize=(20, 11))\nfig.suptitle('Before vs After Cervix Crop — The Model Now Sees Only Tissue',\n             fontsize=14, fontweight='bold')\n\nfor row_idx, class_name in enumerate(CLASS_NAMES):\n    samples = master_df[master_df['final_label'] == class_name].sample(3, random_state=SEED)\n    for col_pair, (_, sample) in enumerate(samples.iterrows()):\n        # Original\n        axes[row_idx, col_pair*2].imshow(\n            Image.open(sample['filepath']).convert('RGB').resize((200, 200)))\n        axes[row_idx, col_pair*2].set_title(f\"Full: {class_name}\", fontsize=7)\n        axes[row_idx, col_pair*2].axis('off')\n        # Cropped\n        axes[row_idx, col_pair*2+1].imshow(crop_cervix(sample['filepath']))\n        axes[row_idx, col_pair*2+1].set_title(f\"Cropped: {class_name}\", fontsize=7, color='green')\n        axes[row_idx, col_pair*2+1].axis('off')\n\nplt.tight_layout()\nplt.savefig(os.path.join(LOG_DIR, 'cervix_crop_comparison.png'), dpi=150)\nplt.show()","metadata":{"id":"k30EFuWzkIFg","outputId":"a8591a57-f05f-4cbf-c5f4-95bb4f7036ef","trusted":true,"execution":{"iopub.status.busy":"2026-06-23T06:07:40.653018Z","iopub.execute_input":"2026-06-23T06:07:40.653271Z","iopub.status.idle":"2026-06-23T06:10:51.149158Z","shell.execute_reply.started":"2026-06-23T06:07:40.653249Z","shell.execute_reply":"2026-06-23T06:10:51.148097Z"}},"outputs":[],"execution_count":null},{"id":"smlanF-kkIFh","cell_type":"markdown","source":"### Step 4.3 — Pre-Compute and Cache Cropped Images to Runtime","metadata":{"id":"smlanF-kkIFh"}},{"id":"oz3ltZwZkIFh","cell_type":"code","source":"from joblib import Parallel, delayed\nfrom tqdm.notebook import tqdm\n\nCROPPED_DIR = os.path.join(KAGGLE_WORKING, 'cropped_images')\nos.makedirs(CROPPED_DIR, exist_ok=True)\n\nprint(f\"Cropping {len(master_df)} images (parallelized, skips already-done)...\")\n\ndef process_image(idx, row, cropped_dir, crop_size):\n    out_path = os.path.join(cropped_dir, f\"{idx:05d}_{row['final_label']}.jpg\")\n    if os.path.exists(out_path):\n        return out_path, True\n    try:\n        cropped = crop_cervix(row['filepath'], output_size=crop_size)\n        cropped.save(out_path, 'JPEG', quality=95)\n        return out_path, False\n    except Exception:\n        return None, False\n\nresults = Parallel(n_jobs=-1)(delayed(process_image)(\n    idx, row, CROPPED_DIR, CROP_SIZE\n) for idx, row in tqdm(master_df.iterrows(), total=len(master_df)))\n\ncropped_paths = [path for path, _ in results]\nmaster_df['cropped_path'] = cropped_paths\n\ninitial_len = len(master_df)\nmaster_df = master_df.dropna(subset=['cropped_path']).reset_index(drop=True)\nprint(f\"✅ Cropping complete: {len(master_df)} images\")\nprint(f\"   Failed/Removed: {initial_len - len(master_df)}\")","metadata":{"id":"oz3ltZwZkIFh","outputId":"f8c233af-b5a4-460b-f4c9-d2569cd64e5c","trusted":true,"execution":{"iopub.status.busy":"2026-06-23T06:10:51.150817Z","iopub.execute_input":"2026-06-23T06:10:51.151211Z","execution_failed":"2026-06-23T14:16:50.843Z"}},"outputs":[],"execution_count":null},{"id":"1ITL33UJkIFh","cell_type":"markdown","source":"### Step 5.1 — Extract Image Features for Clustering","metadata":{"id":"1ITL33UJkIFh"}},{"id":"u0TlM2LqkIFh","cell_type":"code","source":"from tensorflow.keras.applications.efficientnet import preprocess_input\n\nprint(\"Extracting features for cluster-based split (this takes a few mins)...\")\n\n# Use a lightweight feature extractor — we just need rough visual similarity\nfeature_extractor = keras.applications.EfficientNetB3(\n    weights='imagenet', include_top=False,\n    pooling='avg',            # Global average pooling → flat feature vector\n    input_shape=(*IMAGE_SIZE, 3)\n)\nfeature_extractor.trainable = False\n\n# Collect all preprocessed images first, then predict in batch\nimage_batches = []\nvalid_indices = []\n\nfor idx, row in master_df.iterrows():\n    try:\n        img = keras.preprocessing.image.load_img(\n            row['cropped_path'], target_size=IMAGE_SIZE)\n        arr = keras.preprocessing.image.img_to_array(img)\n        arr = preprocess_input(arr) # Preprocess the image\n        image_batches.append(arr)\n        valid_indices.append(idx)\n    except Exception:\n        continue\n\nif image_batches:\n    image_batches = np.array(image_batches) # Convert list of arrays to a single NumPy array\n    features = feature_extractor.predict(image_batches, verbose=1) # Predict in batch\nelse:\n    features = np.array([]) # Handle case where no images were processed\n\nprint(f\"✅ Feature extraction complete: {len(features)} images × {features.shape[1]} dims\")","metadata":{"id":"u0TlM2LqkIFh","outputId":"9eed897b-68a7-45ae-d323-b2eb88da1ce6","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.845Z"}},"outputs":[],"execution_count":null},{"id":"ydRpkayrkIFh","cell_type":"markdown","source":"### Step 5.2 — K-Means Cluster Then Split 🏆","metadata":{"id":"ydRpkayrkIFh"}},{"id":"Wfau33jekIFh","cell_type":"code","source":"N_CLUSTERS = 100   # 🏆 Winners used 100 clusters, 20 randomly chosen for validation\n\nprint(f\"Running K-means with {N_CLUSTERS} clusters...\")\nkmeans = KMeans(n_clusters=N_CLUSTERS, random_state=SEED, n_init=10)\ncluster_labels = kmeans.fit_predict(features)\n\n# Assign cluster labels back to master_df\nvalid_df = master_df.loc[valid_indices].copy()\nvalid_df['cluster'] = cluster_labels\n\n# 🏆 Pick 20 clusters for validation — this ensures no near-duplicates leak\n# between train and val (images from the same patient cluster together)\nrandom.seed(SEED)\nval_clusters = random.sample(list(range(N_CLUSTERS)), 20)\n\n# CRITICAL: additional images NEVER go into validation\n# (they have duplicates that would contaminate evaluation)\nval_df   = valid_df[\n    (valid_df['cluster'].isin(val_clusters)) &\n    (valid_df['source'] == 'main')          # 🏆 Main images only in validation\n].copy()\n\ntrain_df = valid_df[\n    ~valid_df['cluster'].isin(val_clusters)   # Everything not in val clusters\n].copy()\n\n# Separate held-out test set from main images only\nmain_train = train_df[train_df['source'] == 'main']\ntest_df, remaining_main = train_test_split(\n    main_train, test_size=0.85,\n    stratify=main_train['final_label'], random_state=SEED\n)\n# Add remaining main + all additional back to train\ntrain_df = pd.concat([remaining_main,\n                       train_df[train_df['source'] == 'additional']],\n                      ignore_index=True)\n\nprint(f\"\\n✅ Cluster-based split complete:\")\nprint(f\"   Train:      {len(train_df):>5} (clusters {N_CLUSTERS-20} + additional)\")\nprint(f\"   Validation: {len(val_df):>5}  (20 clusters, main only)\")\nprint(f\"   Test:       {len(test_df):>5}  (held-out, main only)\")\n\n# Verify class distribution\nfor name, df in [('Train', train_df), ('Val', val_df), ('Test', test_df)]:\n    dist = df['final_label'].value_counts()\n    print(f\"\\n  {name}: {dict(dist)}\")","metadata":{"id":"Wfau33jekIFh","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.845Z"}},"outputs":[],"execution_count":null},{"id":"F2y4_KI6kIFi","cell_type":"markdown","source":"### Step 5.3 — Save Split Manifests to Drive","metadata":{"id":"F2y4_KI6kIFi"}},{"id":"6i8ll7l-kIFm","cell_type":"code","source":"train_df.to_csv(os.path.join(BASE_DIR, 'train_manifest_v3.csv'), index=False)\nval_df.to_csv(os.path.join(BASE_DIR,   'val_manifest_v3.csv'),   index=False)\ntest_df.to_csv(os.path.join(BASE_DIR,  'test_manifest_v3.csv'),  index=False)\nprint(\"✅ Manifests saved to Drive\")","metadata":{"id":"6i8ll7l-kIFm","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.845Z"}},"outputs":[],"execution_count":null},{"id":"qnm_W8dwkIFn","cell_type":"markdown","source":"### Step 5.4 — Compute Class Weights","metadata":{"id":"qnm_W8dwkIFn"}},{"id":"JrWs-h5vkIFn","cell_type":"code","source":"class_weights_array = compute_class_weight(\n    'balanced',\n    classes=np.unique(train_df['final_label']),\n    y=train_df['final_label']\n)\nclass_weight_dict = {i: w for i, w in enumerate(class_weights_array)}\nprint(\"✅ Class weights:\")\nfor i, n in enumerate(CLASS_NAMES):\n    print(f\"   {n}: {class_weight_dict[i]:.4f}×\")","metadata":{"id":"JrWs-h5vkIFn","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.846Z"}},"outputs":[],"execution_count":null},{"id":"YBH7iUc_kIFn","cell_type":"markdown","source":"### Step 6.1 — Custom Augmentation Including Red Channel Shifts","metadata":{"id":"YBH7iUc_kIFn"}},{"id":"qmXr3lXskIFn","cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport tensorflow as tf\n\ndef red_channel_augment(img):\n    \"\"\"\n    🏆 1st place winner insight: different cervix types have different\n    blood vessel patterns. Augmenting the red channel teaches the model\n    to recognize vascular patterns independent of absolute colour,\n    making it robust to different colposcope lighting setups.\n    \"\"\"\n    img = img.copy().astype(np.float32)\n    # Randomly scale the red channel by ±20%\n    red_factor = np.random.uniform(0.8, 1.2)\n    img[:, :, 0] = np.clip(img[:, :, 0] * red_factor, 0, 255)\n    return img.astype(np.uint8)\n\n# Training augmentation — aggressive, includes red channel shift\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=20,\n    width_shift_range=0.1,\n    height_shift_range=0.1,\n    shear_range=0.1,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    vertical_flip=False,          # Anatomically wrong — never flip vertically\n    brightness_range=[0.8, 1.2],\n    fill_mode='nearest',\n    preprocessing_function=red_channel_augment   # 🏆 Red channel augmentation\n)\n\n# Validation & Test — only rescale\nval_test_datagen = ImageDataGenerator(rescale=1./255)\n\nprint(\"✅ Augmentation pipeline configured (includes red channel shift)\")","metadata":{"id":"qmXr3lXskIFn","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.846Z"}},"outputs":[],"execution_count":null},{"id":"qLjoIGkakIFn","cell_type":"markdown","source":"### Step 6.2 — Create Generators","metadata":{"id":"qLjoIGkakIFn"}},{"id":"S1DTwxthkIFn","cell_type":"code","source":"train_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    x_col='cropped_path',        # Use pre-cropped images\n    y_col='final_label',\n    target_size=IMAGE_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='categorical',\n    shuffle=True, seed=SEED\n)\nval_generator = val_test_datagen.flow_from_dataframe(\n    dataframe=val_df,\n    x_col='cropped_path',\n    y_col='final_label',\n    target_size=IMAGE_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='categorical',\n    shuffle=False\n)\ntest_generator = val_test_datagen.flow_from_dataframe(\n    dataframe=test_df,\n    x_col='cropped_path',\n    y_col='final_label',\n    target_size=IMAGE_SIZE,\n    batch_size=BATCH_SIZE,\n    class_mode='categorical',\n    shuffle=False\n)\nprint(f\"✅ Generators ready\")\nprint(f\"   Train batches: {len(train_generator)} | Class map: {train_generator.class_indices}\")","metadata":{"id":"S1DTwxthkIFn","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.846Z"}},"outputs":[],"execution_count":null},{"id":"LnV2UlUOkIFo","cell_type":"markdown","source":"### Step 6.3 — Visualize Augmented Crops","metadata":{"id":"LnV2UlUOkIFo"}},{"id":"X0gcQ7GCkIFo","cell_type":"code","source":"aug_imgs, aug_labels = next(train_generator)\nfig, axes = plt.subplots(2, 5, figsize=(18, 8))\nfig.suptitle('What the Model Actually Sees: Cropped + Augmented', fontsize=14, fontweight='bold')\nfor i, ax in enumerate(axes.flatten()):\n    ax.imshow(aug_imgs[i])\n    ax.set_title(CLASS_NAMES[np.argmax(aug_labels[i])], fontsize=9)\n    ax.axis('off')\nplt.tight_layout()\nplt.savefig(os.path.join(LOG_DIR, 'augmented_crops.png'), dpi=150)\nplt.show()","metadata":{"id":"X0gcQ7GCkIFo","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.847Z"}},"outputs":[],"execution_count":null},{"id":"xdongw0rkIFo","cell_type":"markdown","source":"### Step 7.1 — Shared Head Builder","metadata":{"id":"xdongw0rkIFo"}},{"id":"nYaZKzzLkIFo","cell_type":"code","source":"def build_model(base_model, model_name, num_classes=NUM_CLASSES):\n    \"\"\"\n    Consistent head across all base models for fair ensembling.\n    Slimmer than v2 (256→128) to avoid overfitting on ~1500 images.\n    \"\"\"\n    base_model.trainable = False\n    inputs = keras.Input(shape=(*IMAGE_SIZE, 3))\n    x = base_model(inputs, training=False)\n    x = layers.GlobalAveragePooling2D()(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dense(256, activation='relu')(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(0.4)(x)\n    x = layers.Dense(128, activation='relu')(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(0.2)(x)\n    outputs = layers.Dense(num_classes, activation='softmax')(x)\n    return Model(inputs, outputs, name=model_name)\n\ndef compile_model(model, lr=1e-3):\n    model.compile(\n        optimizer=Adam(learning_rate=lr),\n        loss='categorical_crossentropy',\n        metrics=[\n            'accuracy',\n            keras.metrics.AUC(name='auc'),\n            keras.metrics.Recall(name='recall')\n        ]\n    )\n    return model","metadata":{"id":"nYaZKzzLkIFo","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.847Z"}},"outputs":[],"execution_count":null},{"id":"pUdRcwalkIFo","cell_type":"markdown","source":"### Step 7.2 — Define Three Base Models","metadata":{"id":"pUdRcwalkIFo"}},{"id":"xX19DWXHkIFo","cell_type":"code","source":"# EfficientNetB3 at native 300×300\nb3_base = keras.applications.EfficientNetB3(\n    weights='imagenet', include_top=False, input_shape=(*IMAGE_SIZE, 3))\nmodel_b3 = compile_model(build_model(b3_base, 'EfficientNetB3'))\n\n# ResNet50 (top 2% solution used ResNet101 — ResNet50 is our Colab-friendly variant)\nrn_base = keras.applications.ResNet50(\n    weights='imagenet', include_top=False, input_shape=(*IMAGE_SIZE, 3))\nmodel_rn = compile_model(build_model(rn_base, 'ResNet50'))\n\n# InceptionV3 (top 2% solution used InceptionV3 — same architecture)\ninc_base = keras.applications.InceptionV3(\n    weights='imagenet', include_top=False, input_shape=(*IMAGE_SIZE, 3))\nmodel_inc = compile_model(build_model(inc_base, 'InceptionV3'))\n\nprint(f\"✅ Three models built\")\nprint(f\"   EfficientNetB3 params: {model_b3.count_params():,}\")\nprint(f\"   ResNet50 params:       {model_rn.count_params():,}\")\nprint(f\"   InceptionV3 params:    {model_inc.count_params():,}\")","metadata":{"id":"xX19DWXHkIFo","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.847Z"}},"outputs":[],"execution_count":null},{"id":"xmvRYQf3kIFo","cell_type":"markdown","source":"### Step 8.1 — Training Helper Function","metadata":{"id":"xmvRYQf3kIFo"}},{"id":"yj4Hye4hkIFp","cell_type":"code","source":"def get_callbacks(model_name, phase):\n    return [\n        EarlyStopping(\n            monitor='val_auc', mode='max',\n            patience=8, restore_best_weights=True, verbose=1\n        ),\n        ModelCheckpoint(\n            filepath=os.path.join(CHECKPOINT_DIR, f'best_{model_name}_phase{phase}.keras'),\n            monitor='val_auc', mode='max',\n            save_best_only=True, verbose=1\n        ),\n        ReduceLROnPlateau(\n            monitor='val_loss', factor=0.3,\n            patience=4, min_lr=1e-8, verbose=1\n        ),\n        TensorBoard(\n            log_dir=os.path.join(LOG_DIR, f'{model_name}_phase{phase}'),\n            histogram_freq=1\n        )\n    ]\n\ndef plot_history(history, title):\n    fig, axes = plt.subplots(1, 3, figsize=(16, 4))\n    fig.suptitle(title, fontweight='bold')\n    for ax, metric in zip(axes, ['loss', 'accuracy', 'auc']):\n        ax.plot(history.history[metric], label='Train', linewidth=2)\n        ax.plot(history.history[f'val_{metric}'], label='Val', linewidth=2, linestyle='--')\n        ax.set_title(metric.upper())\n        ax.legend()\n        ax.grid(True, alpha=0.3)\n    plt.tight_layout()\n    plt.savefig(os.path.join(LOG_DIR, f'{title.replace(\" \", \"_\")}.png'), dpi=150)\n    plt.show()","metadata":{"id":"yj4Hye4hkIFp","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.847Z"}},"outputs":[],"execution_count":null},{"id":"ypbSU_4xkIFp","cell_type":"markdown","source":"### Step 8.2 — Launch TensorBoard Before Training","metadata":{"id":"ypbSU_4xkIFp"}},{"id":"Hl8Pwv-gkIFp","cell_type":"code","source":"%load_ext tensorboard\n%tensorboard --logdir {LOG_DIR}","metadata":{"id":"Hl8Pwv-gkIFp","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.848Z"}},"outputs":[],"execution_count":null},{"id":"fmPYyNhKkIFp","cell_type":"markdown","source":"### Step 8.3 — Train All Three Models (Phase 1: Frozen Base)","metadata":{"id":"fmPYyNhKkIFp"}},{"id":"Gwi6gkJokIFp","cell_type":"code","source":"histories_p1 = {}\n\nfor model, name in [\n    (model_b3,  'EfficientNetB3'),\n    (model_rn,  'ResNet50'),\n    (model_inc, 'InceptionV3')\n]:\n    print(f\"\\n{'='*55}\")\n    print(f\"  Phase 1: {name} — Head Training (Base Frozen)\")\n    print(f\"{'='*55}\")\n\n    h = model.fit(\n        train_generator,\n        epochs=15,\n        validation_data=val_generator,\n        class_weight=class_weight_dict,\n        callbacks=get_callbacks(name, 1),\n        verbose=1\n    )\n    histories_p1[name] = h\n    plot_history(h, f'{name} Phase 1')\n    print(f\"  ✅ Best val_auc: {max(h.history['val_auc']):.4f}\")\n    print(f\"     Best val_recall: {max(h.history['val_recall']):.4f}\")","metadata":{"id":"Gwi6gkJokIFp","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.848Z"}},"outputs":[],"execution_count":null},{"id":"ORSDR9AUkIFp","cell_type":"markdown","source":"### Step 8.4 — Phase 2 Fine-Tuning for All Three Models 🔧","metadata":{"id":"ORSDR9AUkIFp"}},{"id":"lJ-9BG-rkIFp","cell_type":"code","source":"def fine_tune_model(model, base_model, model_name, top_n_layers=50):\n    ckpt_path = os.path.join(CHECKPOINT_DIR, f'best_{model_name}_phase1.keras')\n    if not os.path.exists(ckpt_path):\n        raise FileNotFoundError(\n            f\"Phase 1 checkpoint not found: {ckpt_path}\\n\"\n            \"Run Step 8.3 first to generate the checkpoint.\")\n    model.load_weights(ckpt_path)\n\n    base_model.trainable = True\n    for layer in base_model.layers[:-top_n_layers]:\n        layer.trainable = False\n\n    frozen    = sum(1 for l in base_model.layers if not l.trainable)\n    trainable = sum(1 for l in base_model.layers if l.trainable)\n    print(f\"  {model_name}: {frozen} frozen | {trainable} trainable base layers\")\n\n    compile_model(model, lr=1e-5)\n    return model\n\nhistories_p2 = {}\n\nfor model, base_model, name in [\n    (model_b3,  b3_base,  'EfficientNetB3'),\n    (model_rn,  rn_base,  'ResNet50'),\n    (model_inc, inc_base, 'InceptionV3')\n]:\n    print(f\"\\n{'='*55}\")\n    print(f\"  Phase 2: {name} — Fine-Tuning (Top 50 Layers)\")\n    print(f\"{'='*55}\")\n\n    fine_tune_model(model, base_model, name)\n\n    h = model.fit(\n        train_generator,\n        epochs=30,\n        validation_data=val_generator,\n        class_weight=class_weight_dict,\n        callbacks=get_callbacks(name, 2),\n        verbose=1\n    )\n    histories_p2[name] = h\n    plot_history(h, f'{name} Phase 2 Fine-Tuning')\n    print(f\"  ✅ Best val_auc: {max(h.history['val_auc']):.4f}\")\n    print(f\"     Best val_recall: {max(h.history['val_recall']):.4f}\")","metadata":{"id":"lJ-9BG-rkIFp","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.848Z"}},"outputs":[],"execution_count":null},{"id":"yCxWECP8kIFq","cell_type":"markdown","source":"### Step 9.1 — Load Best Weights for All Models","metadata":{"id":"yCxWECP8kIFq"}},{"id":"POUXEu0ykIFq","cell_type":"code","source":"for model, name in [\n    (model_b3, 'EfficientNetB3'),\n    (model_rn, 'ResNet50'),\n    (model_inc,'InceptionV3')\n]:\n    model.load_weights(\n        os.path.join(CHECKPOINT_DIR, f'best_{name}_phase2.keras'))\n    print(f\"✅ {name}: best Phase 2 weights loaded\")","metadata":{"id":"POUXEu0ykIFq","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.848Z"}},"outputs":[],"execution_count":null},{"id":"GT_PXZVokIFq","cell_type":"markdown","source":"### Step 9.2 — Test-Time Augmentation (TTA) 🏆🆕","metadata":{"id":"GT_PXZVokIFq"}},{"id":"RO2srxwxkIFq","cell_type":"code","source":"def predict_with_tta(model, image_paths, n_augmentations=10):\n    \"\"\"\n    Test-Time Augmentation: run each test image through N augmented\n    versions and average the softmax predictions.\n\n    WHY: A single forward pass gives one prediction. TTA gives N predictions\n    per image and averages them — reducing variance from lucky/unlucky\n    augmentations and consistently improving accuracy by 2–4%.\n\n    The top 2% solution used 10-crop and 40-crop TTA,\n    achieving 0.85 on the private leaderboard.\n    We use 10 random augmentations per image.\n    \"\"\"\n    tta_datagen = ImageDataGenerator(\n        rescale=1./255,\n        rotation_range=15,\n        zoom_range=0.15,\n        horizontal_flip=True,\n        brightness_range=[0.85, 1.15],\n        fill_mode='nearest'\n    )\n\n    all_preds = []\n    for _ in range(n_augmentations):\n        gen = tta_datagen.flow_from_dataframe(\n            dataframe=pd.DataFrame({'path': image_paths}),\n            x_col='path',\n            y_col=None,\n            target_size=IMAGE_SIZE,\n            batch_size=BATCH_SIZE,\n            class_mode=None,\n            shuffle=False\n        )\n        preds = model.predict(gen, verbose=0)\n        all_preds.append(preds)\n\n    return np.mean(all_preds, axis=0)   # Average over N augmentations\n\n\ntest_paths = test_df['cropped_path'].values\n\nprint(\"Running TTA predictions for all three models (this takes a few mins)...\")\npreds_b3  = predict_with_tta(model_b3,  test_paths, n_augmentations=10)\npreds_rn  = predict_with_tta(model_rn,  test_paths, n_augmentations=10)\npreds_inc = predict_with_tta(model_inc, test_paths, n_augmentations=10)\nprint(\"✅ TTA complete for all three models\")","metadata":{"id":"RO2srxwxkIFq","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.849Z"}},"outputs":[],"execution_count":null},{"id":"ycvotd2bkIFq","cell_type":"markdown","source":"### Step 9.3 — Ensemble by Averaging Softmax Outputs 🏆🆕","metadata":{"id":"ycvotd2bkIFq"}},{"id":"_3icPdh3kIFq","cell_type":"code","source":"# Simple average ensemble — works better than weighted for similar-performing models\nensemble_preds = (preds_b3 + preds_rn + preds_inc) / 3.0\n\ny_pred_ensemble = np.argmax(ensemble_preds, axis=1)\ny_true          = test_generator.classes\n\n# Also generate per-model predictions for comparison\nper_model_preds = {\n    'EfficientNetB3': np.argmax(preds_b3, axis=1),\n    'ResNet50':       np.argmax(preds_rn, axis=1),\n    'InceptionV3':    np.argmax(preds_inc, axis=1),\n    'Ensemble':       y_pred_ensemble\n}\n\nprint(\"✅ Ensemble predictions ready\")","metadata":{"id":"_3icPdh3kIFq","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.849Z"}},"outputs":[],"execution_count":null},{"id":"lETUKczzkIFq","cell_type":"markdown","source":"### Step 10.1 — Per-Model vs Ensemble Accuracy Comparison","metadata":{"id":"lETUKczzkIFq"}},{"id":"MExDS4qAkIFq","cell_type":"code","source":"from sklearn.metrics import accuracy_score\n\nle_map = {v: k for k, v in train_generator.class_indices.items()}\n\nprint(\"=\" * 50)\nprint(\"Model Performance Comparison on Test Set\")\nprint(\"=\" * 50)\nresults_table = {}\nfor model_name, preds in per_model_preds.items():\n    acc = accuracy_score(y_true, preds)\n    results_table[model_name] = acc\n    print(f\"  {model_name:20s}: {acc:.4f} accuracy\")\n\nprint(f\"\\n  Best single model: {max(results_table, key=lambda x: results_table[x] if x != 'Ensemble' else 0)}\")\nprint(f\"  Ensemble gain: +{(results_table['Ensemble'] - max(v for k,v in results_table.items() if k != 'Ensemble')):.4f}\")","metadata":{"id":"MExDS4qAkIFq","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.849Z"}},"outputs":[],"execution_count":null},{"id":"kKg5FJFZkIFr","cell_type":"markdown","source":"### Step 10.2 — Confusion Matrix (Ensemble)","metadata":{"id":"kKg5FJFZkIFr"}},{"id":"1oRu3AC0kIFr","cell_type":"code","source":"cm     = confusion_matrix(y_true, y_pred_ensemble)\ncm_pct = cm.astype(float) / cm.sum(axis=1, keepdims=True) * 100\n\nfig, axes = plt.subplots(1, 2, figsize=(14, 5))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n            xticklabels=CLASS_NAMES, yticklabels=CLASS_NAMES, ax=axes[0])\naxes[0].set_title('Ensemble — Confusion Matrix (Counts)', fontweight='bold')\naxes[0].set_ylabel('Actual')\naxes[0].set_xlabel('Predicted')\n\nsns.heatmap(cm_pct, annot=True, fmt='.1f', cmap='Blues',\n            xticklabels=CLASS_NAMES, yticklabels=CLASS_NAMES, ax=axes[1])\naxes[1].set_title('Ensemble — Confusion Matrix (% Recall per Class)', fontweight='bold')\naxes[1].set_ylabel('Actual')\naxes[1].set_xlabel('Predicted')\n\nplt.suptitle('Ensemble Model — Held-Out Test Set', fontweight='bold')\nplt.tight_layout()\nplt.savefig(os.path.join(LOG_DIR, 'confusion_matrix_ensemble.png'), dpi=150)\nplt.show()","metadata":{"id":"1oRu3AC0kIFr","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.85Z"}},"outputs":[],"execution_count":null},{"id":"VL3AOlfnkIFr","cell_type":"markdown","source":"### Step 10.3 — Classification Report","metadata":{"id":"VL3AOlfnkIFr"}},{"id":"4rSFIj3AkIFr","cell_type":"code","source":"report = classification_report(y_true, y_pred_ensemble, target_names=CLASS_NAMES, digits=4)\nprint(report)\nwith open(os.path.join(LOG_DIR, 'classification_report_ensemble.txt'), 'w') as f:\n    f.write(report)","metadata":{"id":"4rSFIj3AkIFr","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.85Z"}},"outputs":[],"execution_count":null},{"id":"hv59n23qkIFr","cell_type":"markdown","source":"### Step 10.4 — Per-Class ROC Curves (Ensemble)","metadata":{"id":"hv59n23qkIFr"}},{"id":"e6r8mHGCkIFr","cell_type":"code","source":"y_true_bin = label_binarize(y_true, classes=[0, 1, 2])\ncolors     = ['#2196F3', '#4CAF50', '#FF5722']\n\nfig, ax = plt.subplots(figsize=(8, 6))\nfor i, (name, color) in enumerate(zip(CLASS_NAMES, colors)):\n    fpr, tpr, _ = roc_curve(y_true_bin[:, i], ensemble_preds[:, i])\n    roc_auc = auc(fpr, tpr)\n    ax.plot(fpr, tpr, color=color, linewidth=2, label=f'{name} (AUC={roc_auc:.3f})')\n\nax.plot([0,1],[0,1], 'k--', linewidth=1, label='Random baseline')\nax.set_xlabel('False Positive Rate (1 − Specificity)')\nax.set_ylabel('True Positive Rate (Sensitivity)')\nax.set_title('ROC Curves — Ensemble (One-vs-Rest)', fontweight='bold')\nax.legend()\nax.grid(True, alpha=0.3)\nplt.tight_layout()\nplt.savefig(os.path.join(LOG_DIR, 'roc_curves_ensemble.png'), dpi=150)\nplt.show()","metadata":{"id":"e6r8mHGCkIFr","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.85Z"}},"outputs":[],"execution_count":null},{"id":"0IifZ3hdkIFr","cell_type":"markdown","source":"### Step 10.5 — Grad-CAM on Ensemble Winner Model","metadata":{"id":"0IifZ3hdkIFr"}},{"id":"YXkXM4efkIFs","cell_type":"code","source":"def make_gradcam_heatmap(img_array, model, last_conv_layer_name, pred_index=None):\n    grad_model = tf.keras.models.Model(\n        model.inputs,\n        [model.get_layer(last_conv_layer_name).output, model.output]\n    )\n    with tf.GradientTape() as tape:\n        conv_outputs, predictions = grad_model(img_array)\n        if pred_index is None:\n            pred_index = tf.argmax(predictions[0])\n        class_channel = predictions[:, pred_index]\n    grads = tape.gradient(class_channel, conv_outputs)\n    pooled_grads = tf.reduce_mean(grads, axis=(0,1,2))\n    heatmap = conv_outputs[0] @ pooled_grads[..., tf.newaxis]\n    heatmap = tf.squeeze(heatmap)\n    heatmap = tf.maximum(heatmap, 0) / (tf.math.reduce_max(heatmap) + 1e-8)\n    return heatmap.numpy()\n\nimport matplotlib.cm as mpl_cm\n\ndef display_gradcam(img_path, model, conv_layer='top_conv'):\n    img    = keras.preprocessing.image.load_img(img_path, target_size=IMAGE_SIZE)\n    arr    = keras.preprocessing.image.img_to_array(img) / 255.0\n    arr_e  = np.expand_dims(arr, 0)\n    heatmap = make_gradcam_heatmap(arr_e, model, conv_layer)\n    heatmap = np.uint8(255 * heatmap)\n    jet     = mpl_cm.get_cmap('jet')(np.arange(256))[:, :3]\n    jet_hm  = keras.preprocessing.image.array_to_img(jet[heatmap])\n    jet_hm  = jet_hm.resize(IMAGE_SIZE)\n    jet_hm  = keras.preprocessing.image.img_to_array(jet_hm)\n    superimposed = keras.preprocessing.image.array_to_img(jet_hm * 0.4 + arr * 255)\n    return superimposed\n\nfig, axes = plt.subplots(3, 4, figsize=(16, 12))\nfig.suptitle('Grad-CAM (EfficientNetB3) — Is the Model Looking at Cervical Tissue?',\n             fontsize=13, fontweight='bold')\n\nfor row, class_name in enumerate(CLASS_NAMES):\n    samples = test_df[test_df['final_label'] == class_name].sample(2, random_state=SEED)\n    for col, (_, sample) in enumerate(samples.iterrows()):\n        axes[row, col*2].imshow(Image.open(sample['cropped_path']).resize(IMAGE_SIZE))\n        axes[row, col*2].set_title(f\"Cropped: {class_name}\", fontsize=8)\n        axes[row, col*2].axis('off')\n        cam = display_gradcam(sample['cropped_path'], model_b3)\n        axes[row, col*2+1].imshow(cam)\n        axes[row, col*2+1].set_title(f\"Grad-CAM: {class_name}\", fontsize=8, color='red')\n        axes[row, col*2+1].axis('off')\n\nplt.tight_layout()\nplt.savefig(os.path.join(LOG_DIR, 'gradcam_ensemble.png'), dpi=150)\nplt.show()","metadata":{"id":"YXkXM4efkIFs","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.85Z"}},"outputs":[],"execution_count":null},{"id":"HoItQrr2kIFs","cell_type":"markdown","source":"### Step 10.6 — Worst Predictions Analysis","metadata":{"id":"HoItQrr2kIFs"}},{"id":"Pex_lXAMkIFs","cell_type":"code","source":"errors_idx   = np.where(y_pred_ensemble != y_true)[0]\nwrong_conf   = ensemble_preds[errors_idx, y_pred_ensemble[errors_idx]]\nsorted_err   = errors_idx[np.argsort(wrong_conf)[::-1]]\n\nprint(f\"Misclassified: {len(errors_idx)} / {len(y_true)} ({len(errors_idx)/len(y_true)*100:.1f}%)\")\n\nfig, axes = plt.subplots(2, 5, figsize=(18, 8))\nfig.suptitle('Most Confident Wrong Predictions (Ensemble)', fontsize=13, fontweight='bold')\nfor i, ax in enumerate(axes.flatten()):\n    if i >= len(sorted_err):\n        ax.axis('off')\n        continue\n    idx  = sorted_err[i]\n    conf = wrong_conf[np.where(sorted_err == idx)[0][0]]\n    img  = Image.open(test_df.iloc[idx]['cropped_path'])\n    ax.imshow(img)\n    ax.set_title(\n        f\"True: {CLASS_NAMES[y_true[idx]]}\\n\"\n        f\"Pred: {CLASS_NAMES[y_pred_ensemble[idx]]}\\n\"\n        f\"Conf: {conf:.1%}\", fontsize=8, color='red')\n    ax.axis('off')\nplt.tight_layout()\nplt.savefig(os.path.join(LOG_DIR, 'worst_predictions_ensemble.png'), dpi=150)\nplt.show()","metadata":{"id":"Pex_lXAMkIFs","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.85Z"}},"outputs":[],"execution_count":null},{"id":"UJslKH1zkIFs","cell_type":"markdown","source":"### Step 11.1 — Save All Three Models","metadata":{"id":"UJslKH1zkIFs"}},{"id":"i0LtrDagkIFs","cell_type":"code","source":"for model, name in [\n    (model_b3,  'EfficientNetB3'),\n    (model_rn,  'ResNet50'),\n    (model_inc, 'InceptionV3')\n]:\n    path = os.path.join(BASE_DIR, f'cervix_{name}_final.keras')\n    model.save(path)\n    print(f\"✅ {name} saved: {path}\")","metadata":{"id":"i0LtrDagkIFs","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.851Z"}},"outputs":[],"execution_count":null},{"id":"-x4HtGSCkIFs","cell_type":"markdown","source":"### Step 11.2 — Export EfficientNetB3 to TFLite","metadata":{"id":"-x4HtGSCkIFs"}},{"id":"m_eTsIMGkIFs","cell_type":"code","source":"converter    = tf.lite.TFLiteConverter.from_keras_model(model_b3)\ntflite_model = converter.convert()\ntflite_path  = os.path.join(BASE_DIR, 'cervix_b3_classifier.tflite')\nwith open(tflite_path, 'wb') as f:\n    f.write(tflite_model)\nprint(f\"✅ TFLite export: {tflite_path} ({os.path.getsize(tflite_path)/1024**2:.1f} MB)\")","metadata":{"id":"m_eTsIMGkIFs","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.852Z"}},"outputs":[],"execution_count":null},{"id":"GfvZtEjikIFt","cell_type":"markdown","source":"### Step 11.3 — Write Results Summary","metadata":{"id":"GfvZtEjikIFt"}},{"id":"6lJtHyOTkIFt","cell_type":"code","source":"best_single = max(\n    {k: v for k, v in results_table.items() if k != 'Ensemble'},\n    key=results_table.get\n)\n\nsummary = f\"\"\"\n==============================================================\nCERVIX CLASSIFICATION — RESULTS SUMMARY v3.0\n==============================================================\nDate:              {pd.Timestamp.now().strftime('%Y-%m-%d %H:%M')}\nArchitecture:      3-Model Ensemble (EfficientNetB3 + ResNet50 + InceptionV3)\nTTA:               10 augmentations per image\nImage Size:        {IMAGE_SIZE}\nCrop Strategy:     Centre 70% → Brightest region 50%\nValidation:        K-means (100 clusters, 20 held for val, main only)\n\n── DATASET ────────────────────────────────────────────────────\nMain images:       {len(main_df)}\nAdditional (1:2):  {len(add_df)}\nAfter blur filter: {len(master_df)}\n\n── TEST SET ACCURACY ──────────────────────────────────────────\n{chr(10).join(f'{k:20s}: {v:.4f}' for k, v in results_table.items())}\nEnsemble gain:     +{results_table['Ensemble'] - results_table[best_single]:.4f} vs best single\n\n── MISCLASSIFIED (ENSEMBLE) ───────────────────────────────────\n{len(errors_idx)} / {len(y_true)} ({len(errors_idx)/len(y_true)*100:.1f}%)\n\n── KEY TECHNIQUE SOURCES ──────────────────────────────────────\nCervix cropping:   Original (based on winner's bounding box insight)\nCluster split:     1st place team (Team TEST, $50k prize)\nRed channel aug:   1st place team\n1:2 ratio cap:     1st place team\n3-model ensemble:  Top 2% solution (ResNet101 + InceptionV3)\nTTA (10 crops):    Top 2% solution\n==============================================================\n\"\"\"\nprint(summary)\nwith open(os.path.join(BASE_DIR, 'results_summary_v3.txt'), 'w') as f:\n    f.write(summary)","metadata":{"id":"6lJtHyOTkIFt","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.852Z"}},"outputs":[],"execution_count":null},{"id":"negzmB1DkIFt","cell_type":"markdown","source":"## What Changed: v2 → v3 (The Core Insight)","metadata":{"id":"negzmB1DkIFt"}},{"id":"bd2deb2d","cell_type":"markdown","source":"### Step 12.1 — Create a Gradio Interface for Deployment\n\nTo deploy the model as a web application, we'll use Gradio, which allows for quick creation of shareable UI components for machine learning models. We'll load the best performing `EfficientNetB3` model and integrate the `crop_cervix` function for preprocessing.","metadata":{"id":"bd2deb2d"}},{"id":"38d2fda1","cell_type":"code","source":"# Install Gradio\n!pip install gradio --quiet","metadata":{"id":"38d2fda1","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.852Z"}},"outputs":[],"execution_count":null},{"id":"707b7c95","cell_type":"code","source":"import gradio as gr\nfrom tensorflow.keras.models import load_model\n\n# Load the best EfficientNetB3 model (or you could load the ensemble weights and define an ensemble prediction function)\nmodel_path = os.path.join(BASE_DIR, 'cervix_EfficientNetB3_final.keras')\nfinal_model = load_model(model_path, compile=False)\n\n# Re-compile the model with metrics since load_model(compile=False) is used\nfinal_model.compile(\n    optimizer=Adam(learning_rate=1e-3),\n    loss='categorical_crossentropy',\n    metrics=[\n        'accuracy',\n        keras.metrics.AUC(name='auc'),\n        keras.metrics.Recall(name='recall')\n    ]\n)\n\nprint(f\"✅ Loaded model from {model_path}\")","metadata":{"id":"707b7c95","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.852Z"}},"outputs":[],"execution_count":null},{"id":"d772d739","cell_type":"code","source":"def predict_cervix_type(image):\n    \"\"\"\n    Predicts the cervix type from an input image using the loaded model.\n    Applies cropping and preprocessing steps similar to training.\n    \"\"\"\n    if image is None:\n        return {class_name: 0.0 for class_name in CLASS_NAMES}\n\n    # Convert numpy array image to PIL Image for the crop_cervix function if needed\n    if isinstance(image, np.ndarray):\n        image_pil = Image.fromarray(image.astype(np.uint8))\n    else:\n        image_pil = image # Assume it's already a PIL Image if not ndarray\n\n    # Save the image temporarily to a file to use with crop_cervix\n    temp_img_path = '/tmp/gradio_input_image.jpg'\n    image_pil.save(temp_img_path)\n\n    # Apply cervix cropping\n    cropped_img_pil = crop_cervix(temp_img_path, output_size=IMAGE_SIZE)\n\n    # Preprocess for model input\n    img_array = keras.preprocessing.image.img_to_array(cropped_img_pil)\n    img_array = np.expand_dims(img_array, axis=0) # Add batch dimension\n    img_array = img_array / 255.0 # Rescale to [0, 1]\n\n    # Make prediction\n    predictions = final_model.predict(img_array)[0]\n\n    # Map predictions to class names and return as dictionary\n    return {CLASS_NAMES[i]: float(predictions[i]) for i in range(len(CLASS_NAMES))}\n\n# Create Gradio interface\niface = gr.Interface(\n    fn=predict_cervix_type,\n    inputs=gr.Image(type=\"numpy\", label=\"Upload Cervix Image\"),\n    outputs=gr.Label(),\n    title=\"Cervical Cancer Type Classifier (EfficientNetB3)\",\n    description=\"Upload an image of the cervix to classify its type (Type_1, Type_2, or Type_3). The image will be automatically cropped to focus on the cervix.\",\n    examples=[\n        os.path.join(test_df[test_df['final_label'] == 'Type_1'].sample(1, random_state=SEED).iloc[0]['cropped_path']),\n        os.path.join(test_df[test_df['final_label'] == 'Type_2'].sample(1, random_state=SEED).iloc[0]['cropped_path']),\n        os.path.join(test_df[test_df['final_label'] == 'Type_3'].sample(1, random_state=SEED).iloc[0]['cropped_path'])\n    ]\n)\n\nprint(\"✅ Gradio interface created\")","metadata":{"id":"d772d739","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.852Z"}},"outputs":[],"execution_count":null},{"id":"b1733794","cell_type":"code","source":"# Launch the Gradio app\n# The `share=True` argument generates a public link that is valid for 72 hours.\n# To deploy persistently, you'll need to containerize this app.\niface.launch(debug=True, share=True)","metadata":{"id":"b1733794","trusted":true,"execution":{"execution_failed":"2026-06-23T14:16:50.852Z"}},"outputs":[],"execution_count":null},{"id":"0d8f7d92","cell_type":"markdown","source":"### Step 12.2 — Deploying to Google Cloud (Gradio App with TensorFlow Model)\n\nTo deploy this Gradio application with your TensorFlow model to Google Cloud, you would typically follow these steps:\n\n1.  **Containerize Your Application (Dockerfile):** Create a `Dockerfile` that specifies how to build a Docker image for your application. This image will contain your Python code, all necessary libraries (Gradio, TensorFlow, etc.), and your saved model.\n\n    A basic `Dockerfile` might look like this:\n    ```dockerfile\n    FROM python:3.9-slim\n\n    WORKDIR /app\n\n    COPY requirements.txt .\n    RUN pip install -r requirements.txt\n\n    COPY . .\n    RUN pip install gradio tensorflow tensorflow-addons opencv-python-headless Pillow\n\n    # Copy your trained model and any necessary data/scripts\n    COPY cervix_EfficientNetB3_final.keras .\n    COPY fixed_labels_v2.csv .\n    COPY crop_cervix.py .\n\n    ENV PORT 8080\n    EXPOSE 8080\n\n    CMD exec python -m gradio_app.py\n    ```\n\n2.  **Create `requirements.txt`:** List all Python dependencies.\n    ```\n    gradio\n    tensorflow\n    numpy\n    pandas\n    opencv-python-headless\n    Pillow\n    ```\n\n3.  **Refactor Gradio App (`gradio_app.py`):** Organize your Gradio interface creation and prediction logic into a Python script that can be run directly. Ensure all constants (like `IMAGE_SIZE`, `CLASS_NAMES`) and the `crop_cervix` function are accessible.\n\n4.  **Build and Push Docker Image:**\n    *   Build the Docker image: `docker build -t gcr.io/your-gcp-project-id/cervix-classifier:v1 .`\n    *   Push to Google Container Registry (GCR) or Artifact Registry: `docker push gcr.io/your-gcp-project-id/cervix-classifier:v1`\n\n5.  **Deploy to Google Cloud Run:** Google Cloud Run is an excellent choice for stateless containerized applications like Gradio apps. It automatically scales based on traffic.\n    *   Navigate to Cloud Run in the Google Cloud Console.\n    *   Create a new service, select your pushed Docker image.\n    *   Configure resources (CPU, memory) and ensure public access if desired.\n    *   Specify the port (e.g., `8080`) that your container exposes.\n\n6.  **Persistent Storage (if needed):** If your model files are very large or change frequently, consider storing them in Google Cloud Storage and fetching them at container startup, rather than bundling them directly in the Docker image.\n\nThis setup allows your model to be served as a scalable web service accessible via a public URL.","metadata":{"id":"0d8f7d92"}}]}