{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"179340fd-ae3a-4c24-8e55-3b05ae59c182","cell_type":"markdown","source":"# Cassava Leaf Disease Classification — Week 7\n","metadata":{}},{"id":"0a8f11f8-7b31-412f-a469-4e76533f7534","cell_type":"markdown","source":"## Results so far\n\n| Model | Accuracy | Macro F1 |\n|---|---|---|\n| Baseline CNN (Week 3) | 51.46% | 0.2793 |\n| Baseline CNN + Augmentation (Week 4) | 62.40% | 0.2514 |\n| EfficientNetB0 (Week 4) | 66.57% | 0.5347 |\n| EfficientNetB0 + fast pipeline + MixUp + label smoothing (Week 5) | 69.07% | 0.5744 |\n| EfficientNetB0 + focal loss + TTA (Week 6) | 76.23% | 0.5856 |","metadata":{}},{"id":"b3afcb74-39c2-49d2-805b-ed5607aebcea","cell_type":"markdown","source":"## Setup","metadata":{}},{"id":"17cc149c-488e-43e7-b003-46fd3dccf4bb","cell_type":"code","source":"import os\nimport time\nimport json\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport tensorflow as tf\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.applications.efficientnet import preprocess_input\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Input, Dense, Dropout\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix, accuracy_score, f1_score\n\nprint(\"TF version:\", tf.__version__)\ngpus = tf.config.list_physical_devices('GPU')\nprint(\"GPUs available:\", gpus)\n\nSEED = 42\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)\n\nDATA_DIR = \"/kaggle/input/competitions/cassava-leaf-disease-classification\"\nTRAIN_IMG_DIR = os.path.join(DATA_DIR, \"train_images\")\nTEST_IMG_DIR = os.path.join(DATA_DIR, \"test_images\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-16T05:19:01.339288Z","iopub.execute_input":"2026-08-16T05:19:01.339611Z","iopub.status.idle":"2026-08-16T05:19:26.771948Z","shell.execute_reply.started":"2026-08-16T05:19:01.339581Z","shell.execute_reply":"2026-08-16T05:19:26.770809Z"}},"outputs":[],"execution_count":null},{"id":"33c9f8e5-e564-4bab-bb8d-01ef5b7f7aca","cell_type":"code","source":"if len(gpus) > 1:\n    strategy = tf.distribute.MirroredStrategy()\nelse:\n    strategy = tf.distribute.get_strategy()\nprint(\"Devices in use:\", strategy.num_replicas_in_sync)\n\ntf.keras.mixed_precision.set_global_policy(\"mixed_float16\")\nprint(tf.keras.mixed_precision.global_policy())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-16T05:19:26.773884Z","iopub.execute_input":"2026-08-16T05:19:26.774488Z","iopub.status.idle":"2026-08-16T05:19:26.782230Z","shell.execute_reply.started":"2026-08-16T05:19:26.774457Z","shell.execute_reply":"2026-08-16T05:19:26.781213Z"}},"outputs":[],"execution_count":null},{"id":"308ce4bc-6a0c-403e-b4f7-7d8a09b2148d","cell_type":"markdown","source":"## Load labels, recreate train/val split","metadata":{}},{"id":"77fcc1f6-21b8-4df7-b489-91c6c88e5094","cell_type":"code","source":"train_df = pd.read_csv(os.path.join(DATA_DIR, \"train.csv\"))\n\nwith open(os.path.join(DATA_DIR, \"label_num_to_disease_map.json\")) as f:\n    label_map = json.load(f)\nlabel_map = {int(k): v for k, v in label_map.items()}\n\ntrain_df[\"filepath\"] = train_df[\"image_id\"].apply(lambda x: os.path.join(TRAIN_IMG_DIR, x))\n\nNUM_CLASSES = 5\n\ntrain_split, val_split = train_test_split(\n    train_df,\n    test_size=0.15,\n    stratify=train_df[\"label\"],\n    random_state=SEED\n)\nprint(\"Train:\", train_split.shape, \" Val:\", val_split.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-16T05:19:26.784152Z","iopub.execute_input":"2026-08-16T05:19:26.784518Z","iopub.status.idle":"2026-08-16T05:19:26.905881Z","shell.execute_reply.started":"2026-08-16T05:19:26.784489Z","shell.execute_reply":"2026-08-16T05:19:26.904777Z"}},"outputs":[],"execution_count":null},{"id":"903c55ab-f7d3-42c4-8123-b8e612dce274","cell_type":"markdown","source":"## Data pipeline\n\n`build_dataset` now takes `img_size` so Stage 1 can run smaller/faster and Stage 2 can run at full size. Each resolution gets its own cache path. Cutout replaces MixUp by default this week (`USE_CUTOUT = True`, `USE_MIXUP = False`) — flip both to compare.","metadata":{}},{"id":"5ad728b3-3827-4826-aef1-929d87abf8d3","cell_type":"code","source":"PER_REPLICA_BATCH = 32\nAUTOTUNE = tf.data.AUTOTUNE\nUSE_MIXUP = False\nUSE_CUTOUT = True\n\ndef batch_size_for(strategy):\n    return PER_REPLICA_BATCH * max(strategy.num_replicas_in_sync, 1)\n\nBATCH_SIZE = batch_size_for(strategy)\nprint(\"Batch size:\", BATCH_SIZE)\n\ndef decode_img(filepath, label, img_size):\n    img = tf.io.read_file(filepath)\n    img = tf.io.decode_jpeg(img, channels=3)\n    img = tf.image.resize(img, [img_size, img_size], method=\"bilinear\")\n    img = tf.cast(img, tf.uint8)\n    return img, label\n\ndef augment(img):\n    img = tf.image.random_flip_left_right(img)\n    img = tf.image.random_flip_up_down(img)\n    img = tf.image.rot90(img, k=tf.random.uniform([], 0, 4, dtype=tf.int32))\n    img = tf.image.random_brightness(img, 0.15)\n    img = tf.image.random_contrast(img, 0.85, 1.15)\n    img = tf.image.random_saturation(img, 0.85, 1.15)\n    img = tf.clip_by_value(img, 0.0, 255.0)\n    return img\n\ndef cutout(img, pct=0.2):\n    shape = tf.shape(img)\n    h, w = shape[0], shape[1]\n    cut_h = tf.cast(tf.cast(h, tf.float32) * pct, tf.int32)\n    cut_w = tf.cast(tf.cast(w, tf.float32) * pct, tf.int32)\n    cy = tf.random.uniform([], 0, h, dtype=tf.int32)\n    cx = tf.random.uniform([], 0, w, dtype=tf.int32)\n    y1 = tf.clip_by_value(cy - cut_h // 2, 0, h)\n    y2 = tf.clip_by_value(cy + cut_h // 2, 0, h)\n    x1 = tf.clip_by_value(cx - cut_w // 2, 0, w)\n    x2 = tf.clip_by_value(cx + cut_w // 2, 0, w)\n    mask = tf.ones((y2 - y1, x2 - x1, 3), dtype=img.dtype)\n    mask = tf.pad(mask, [[y1, h - y2], [x1, w - x2], [0, 0]])\n    return img * (1.0 - mask)\n\ndef sample_beta(alpha, batch_size):\n    g1 = tf.random.gamma([batch_size], alpha)\n    g2 = tf.random.gamma([batch_size], alpha)\n    return g1 / (g1 + g2)\n\ndef mixup(images, labels, alpha=0.2):\n    images = tf.cast(images, tf.float32)\n    labels = tf.cast(labels, tf.float32)\n    batch_size = tf.shape(images)[0]\n    lam = tf.cast(sample_beta(alpha, batch_size), tf.float32)\n    idx = tf.random.shuffle(tf.range(batch_size))\n    images2 = tf.gather(images, idx)\n    labels2 = tf.gather(labels, idx)\n    lam_x = lam[:, tf.newaxis, tf.newaxis, tf.newaxis]\n    lam_y = lam[:, tf.newaxis]\n    images = lam_x * images + (1.0 - lam_x) * images2\n    labels = lam_y * labels + (1.0 - lam_y) * labels2\n    return images, labels\n\ndef build_dataset(df, training, img_size, cache_path=None, augment_pass=None):\n    do_augment = training if augment_pass is None else augment_pass\n\n    filepaths = df[\"filepath\"].values\n    labels = tf.keras.utils.to_categorical(df[\"label\"].values, num_classes=NUM_CLASSES)\n\n    ds = tf.data.Dataset.from_tensor_slices((filepaths, labels))\n    if training:\n        ds = ds.shuffle(buffer_size=len(df), seed=SEED, reshuffle_each_iteration=True)\n\n    ds = ds.map(lambda fp, lb: decode_img(fp, lb, img_size), num_parallel_calls=AUTOTUNE)\n\n    if cache_path is not None:\n        ds = ds.cache(cache_path)\n\n    ds = ds.map(lambda img, lb: (tf.cast(img, tf.float32), lb), num_parallel_calls=AUTOTUNE)\n\n    if do_augment:\n        ds = ds.map(lambda img, lb: (augment(img), lb), num_parallel_calls=AUTOTUNE)\n        if USE_CUTOUT:\n            ds = ds.map(lambda img, lb: (cutout(img), lb), num_parallel_calls=AUTOTUNE)\n\n    ds = ds.map(lambda img, lb: (preprocess_input(img), lb), num_parallel_calls=AUTOTUNE)\n    ds = ds.batch(BATCH_SIZE)\n\n    if training and USE_MIXUP:\n        ds = ds.map(mixup, num_parallel_calls=AUTOTUNE)\n\n    ds = ds.prefetch(AUTOTUNE)\n    return ds\n\nos.makedirs(\"/kaggle/working/cache\", exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-16T05:19:26.907451Z","iopub.execute_input":"2026-08-16T05:19:26.907740Z","iopub.status.idle":"2026-08-16T05:19:26.930273Z","shell.execute_reply.started":"2026-08-16T05:19:26.907715Z","shell.execute_reply":"2026-08-16T05:19:26.929264Z"}},"outputs":[],"execution_count":null},{"id":"f3b5e486-e62a-44c8-885a-935aa1565f60","cell_type":"markdown","source":"## Focal loss","metadata":{}},{"id":"9df83601-7989-416d-b23a-78d18d019b9b","cell_type":"code","source":"def categorical_focal_loss(gamma=2.0, alpha=0.25):\n    def loss_fn(y_true, y_pred):\n        y_true = tf.cast(y_true, tf.float32)\n        y_pred = tf.cast(y_pred, tf.float32)\n        y_pred = tf.clip_by_value(y_pred, 1e-7, 1.0 - 1e-7)\n        cross_entropy = -y_true * tf.math.log(y_pred)\n        weight = alpha * tf.pow(1.0 - y_pred, gamma)\n        loss = weight * cross_entropy\n        return tf.reduce_sum(loss, axis=-1)\n    return loss_fn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-16T05:19:26.932870Z","iopub.execute_input":"2026-08-16T05:19:26.933335Z","iopub.status.idle":"2026-08-16T05:19:26.958210Z","shell.execute_reply.started":"2026-08-16T05:19:26.933303Z","shell.execute_reply":"2026-08-16T05:19:26.957007Z"}},"outputs":[],"execution_count":null},{"id":"25e553ce-7d53-4259-be77-111d46ba52db","cell_type":"markdown","source":"## Model","metadata":{}},{"id":"cae80c00-97c9-4734-a9ad-30ef71a24542","cell_type":"code","source":"def build_effnet_model(input_shape, num_classes=NUM_CLASSES):\n    base_model = EfficientNetB0(include_top=False, weights=\"imagenet\", input_shape=input_shape)\n    base_model.trainable = False\n\n    inputs = Input(shape=input_shape)\n    x = base_model(inputs, training=False)\n    x = GlobalAveragePooling2D()(x)\n    x = Dropout(0.3)(x)\n    x = Dense(128, activation=\"relu\")(x)\n    x = Dropout(0.3)(x)\n    outputs = Dense(num_classes, activation=\"softmax\", dtype=\"float32\")(x)\n\n    model = Model(inputs, outputs)\n    return model, base_model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-16T05:19:26.959454Z","iopub.execute_input":"2026-08-16T05:19:26.959820Z","iopub.status.idle":"2026-08-16T05:19:26.979645Z","shell.execute_reply.started":"2026-08-16T05:19:26.959762Z","shell.execute_reply":"2026-08-16T05:19:26.978357Z"}},"outputs":[],"execution_count":null},{"id":"4c9b3642-ad85-4b80-b52c-dc12be181da2","cell_type":"markdown","source":"## Stage 1: train head at a smaller resolution (160x160)\n\nSmaller images here cut per-step time noticeably. The base is frozen, so this stage is only training the small head anyway — full resolution isn't needed yet.","metadata":{}},{"id":"6e8d2387-b867-4f92-a359-2f8206251f9c","cell_type":"code","source":"STAGE1_SIZE = 160\n\ntrain_ds_s1 = build_dataset(train_split, training=True, img_size=STAGE1_SIZE, cache_path=\"/kaggle/working/cache/train_160\")\nval_ds_s1 = build_dataset(val_split, training=False, img_size=STAGE1_SIZE, cache_path=\"/kaggle/working/cache/val_160\")\n\nwith strategy.scope():\n    model, effnet_base = build_effnet_model(input_shape=(STAGE1_SIZE, STAGE1_SIZE, 3))\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=1e-3),\n        loss=categorical_focal_loss(gamma=2.0, alpha=0.25),\n        metrics=[\"accuracy\"]\n    )\n\ncallbacks_stage1 = [\n    EarlyStopping(monitor=\"val_loss\", patience=2, restore_best_weights=True),\n    ReduceLROnPlateau(monitor=\"val_loss\", factor=0.5, patience=1, min_lr=1e-6)\n]\n\nstart = time.time()\nhistory_stage1 = model.fit(\n    train_ds_s1,\n    validation_data=val_ds_s1,\n    epochs=6,\n    callbacks=callbacks_stage1\n)\nprint(f\"Stage 1 ({STAGE1_SIZE}x{STAGE1_SIZE}): {time.time() - start:.1f}s\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-16T05:19:26.980894Z","iopub.execute_input":"2026-08-16T05:19:26.981312Z","iopub.status.idle":"2026-08-16T06:26:22.207143Z","shell.execute_reply.started":"2026-08-16T05:19:26.981283Z","shell.execute_reply":"2026-08-16T06:26:22.204228Z"}},"outputs":[],"execution_count":null},{"id":"f547dd8a-964d-4f2c-a852-f1b9bbc23741","cell_type":"markdown","source":"## Stage 2: fine-tune at full resolution (224x224)\n\nRebuilding the model at 224x224 and transferring the Stage 1 weights, since EfficientNet's input shape is fixed at model-build time. Only the base's pretrained + Stage-1-tuned weights carry over — the new Dense head layers keep their Stage 1 training.","metadata":{}},{"id":"9fb07719-f868-4146-bca6-c341a5212969","cell_type":"code","source":"STAGE2_SIZE = 224\n\ntrain_ds_s2 = build_dataset(train_split, training=True, img_size=STAGE2_SIZE, cache_path=\"/kaggle/working/cache/train_224\")\nval_ds_s2 = build_dataset(val_split, training=False, img_size=STAGE2_SIZE, cache_path=\"/kaggle/working/cache/val_224\")\n\nwith strategy.scope():\n    model_full, effnet_base_full = build_effnet_model(input_shape=(STAGE2_SIZE, STAGE2_SIZE, 3))\n    model_full.set_weights(model.get_weights())\n\n    effnet_base_full.trainable = True\n    FINE_TUNE_AT = len(effnet_base_full.layers) - 60\n    for layer in effnet_base_full.layers[:FINE_TUNE_AT]:\n        layer.trainable = False\n\n    model_full.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=1e-5),\n        loss=categorical_focal_loss(gamma=2.0, alpha=0.25),\n        metrics=[\"accuracy\"]\n    )\n\ncallbacks_stage2 = [\n    EarlyStopping(monitor=\"val_loss\", patience=2, restore_best_weights=True),\n    ReduceLROnPlateau(monitor=\"val_loss\", factor=0.5, patience=1, min_lr=1e-7)\n]\n\nstart = time.time()\nhistory_stage2 = model_full.fit(\n    train_ds_s2,\n    validation_data=val_ds_s2,\n    epochs=10,\n    callbacks=callbacks_stage2\n)\nprint(f\"Stage 2 ({STAGE2_SIZE}x{STAGE2_SIZE}): {time.time() - start:.1f}s\")\n\nmodel = model_full\nIMG_SIZE = STAGE2_SIZE","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-16T06:26:22.208150Z","iopub.status.idle":"2026-08-16T06:26:22.208545Z","shell.execute_reply.started":"2026-08-16T06:26:22.208391Z","shell.execute_reply":"2026-08-16T06:26:22.208411Z"}},"outputs":[],"execution_count":null},{"id":"55193e64-c847-4f3e-929b-b8489630a6c9","cell_type":"markdown","source":"## TTA on validation","metadata":{}},{"id":"d4bc0635-ca36-46a0-9b9d-274e9f5a65c6","cell_type":"code","source":"N_TTA_ROUNDS = 3\n\ndef tta_predict(df, cache_path, n_rounds=N_TTA_ROUNDS):\n    preds_sum = None\n    for _ in range(n_rounds):\n        ds = build_dataset(df, training=False, img_size=IMG_SIZE, cache_path=cache_path, augment_pass=True)\n        preds = model.predict(ds, verbose=0)\n        preds_sum = preds if preds_sum is None else preds_sum + preds\n    return preds_sum / n_rounds\n\nstart = time.time()\nval_preds_proba_tta = tta_predict(val_split, cache_path=\"/kaggle/working/cache/val_224\")\nval_preds_tta = np.argmax(val_preds_proba_tta, axis=1)\nval_true = val_split[\"label\"].values\nprint(f\"TTA: {time.time() - start:.1f}s\")\n\ntarget_names = [f\"{i}: {label_map[i]}\" for i in range(NUM_CLASSES)]\nprint(classification_report(val_true, val_preds_tta, target_names=target_names))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-16T06:26:22.210398Z","iopub.status.idle":"2026-08-16T06:26:22.210854Z","shell.execute_reply.started":"2026-08-16T06:26:22.210634Z","shell.execute_reply":"2026-08-16T06:26:22.210665Z"}},"outputs":[],"execution_count":null},{"id":"1aea4568-b2af-4117-9343-57d9c7d443fa","cell_type":"code","source":"cm = confusion_matrix(val_true, val_preds_tta)\nfig, ax = plt.subplots(figsize=(7, 6))\nsns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Reds\", xticklabels=range(NUM_CLASSES), yticklabels=range(NUM_CLASSES), ax=ax)\nax.set_xlabel(\"Predicted\")\nax.set_ylabel(\"Actual\")\nax.set_title(\"Confusion Matrix — Week 7\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-16T06:26:22.213787Z","iopub.status.idle":"2026-08-16T06:26:22.214323Z","shell.execute_reply.started":"2026-08-16T06:26:22.214061Z","shell.execute_reply":"2026-08-16T06:26:22.214108Z"}},"outputs":[],"execution_count":null},{"id":"1b2f0643-fbdf-4a30-a9ba-1152c2d79179","cell_type":"markdown","source":"## Comparison","metadata":{}},{"id":"8b737b0c-20a0-4c37-b239-92f50a1edef1","cell_type":"code","source":"comparison = pd.DataFrame({\n    \"Model\": [\n        \"Baseline CNN (Week 3)\",\n        \"Baseline CNN + Augmentation (Week 4)\",\n        \"EfficientNetB0 (Week 4)\",\n        \"EfficientNetB0 + fast pipeline + MixUp + label smoothing (Week 5)\",\n        \"EfficientNetB0 + focal loss + TTA (Week 6)\",\n        \"EfficientNetB0 + progressive resizing + cutout + focal loss + TTA (Week 7)\"\n    ],\n    \"Accuracy\": [0.5146, 0.6240, 0.6657, 0.6907, 0.7623, accuracy_score(val_true, val_preds_tta)],\n    \"Macro F1\": [0.2793, 0.2514, 0.5347, 0.5744, 0.5856, f1_score(val_true, val_preds_tta, average=\"macro\")]\n})\ncomparison","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-16T06:26:22.215404Z","iopub.status.idle":"2026-08-16T06:26:22.215708Z","shell.execute_reply.started":"2026-08-16T06:26:22.215571Z","shell.execute_reply":"2026-08-16T06:26:22.215588Z"}},"outputs":[],"execution_count":null},{"id":"4682b962-8c35-461f-924c-098d25d37488","cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12, 5))\nx = np.arange(len(comparison))\nwidth = 0.35\nax.bar(x - width/2, comparison[\"Accuracy\"], width, label=\"Accuracy\")\nax.bar(x + width/2, comparison[\"Macro F1\"], width, label=\"Macro F1\")\nax.set_xticks(x)\nax.set_xticklabels(comparison[\"Model\"], rotation=20, ha=\"right\")\nax.set_ylim(0, 1)\nax.set_title(\"Model comparison across all weeks\")\nax.legend()\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-16T06:26:22.217965Z","iopub.status.idle":"2026-08-16T06:26:22.218559Z","shell.execute_reply.started":"2026-08-16T06:26:22.218291Z","shell.execute_reply":"2026-08-16T06:26:22.218326Z"}},"outputs":[],"execution_count":null},{"id":"c0dde01b-8afb-40eb-bb5a-42276390a875","cell_type":"markdown","source":"## Final submission","metadata":{}},{"id":"b66a1127-8e9b-415e-87b6-d4671fcca576","cell_type":"code","source":"test_image_ids = os.listdir(TEST_IMG_DIR)\ntest_df = pd.DataFrame({\"image_id\": test_image_ids})\ntest_df[\"filepath\"] = test_df[\"image_id\"].apply(lambda x: os.path.join(TEST_IMG_DIR, x))\ntest_df[\"label\"] = 0\n\ntest_preds_proba_tta = tta_predict(test_df, cache_path=None)\ntest_preds_tta = np.argmax(test_preds_proba_tta, axis=1)\n\nsubmission = pd.DataFrame({\n    \"image_id\": test_df[\"image_id\"],\n    \"label\": test_preds_tta\n})\nsubmission.to_csv(\"submission.csv\", index=False)\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-16T06:26:22.220817Z","iopub.status.idle":"2026-08-16T06:26:22.221217Z","shell.execute_reply.started":"2026-08-16T06:26:22.221041Z","shell.execute_reply":"2026-08-16T06:26:22.221070Z"}},"outputs":[],"execution_count":null},{"id":"1ec7b152-fa94-496c-a8e4-c6ae3d42e875","cell_type":"markdown","source":"## Cleanup","metadata":{}},{"id":"2970af4e-d08b-4fd5-aab6-a1aeced21866","cell_type":"code","source":"import shutil\ncache_dir = \"/kaggle/working/cache\"\nif os.path.exists(cache_dir):\n    shutil.rmtree(cache_dir)\n    print(\"Cache removed.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-16T06:26:22.223661Z","iopub.status.idle":"2026-08-16T06:26:22.224132Z","shell.execute_reply.started":"2026-08-16T06:26:22.223905Z","shell.execute_reply":"2026-08-16T06:26:22.223926Z"}},"outputs":[],"execution_count":null},{"id":"5dbd608e-1b1b-4876-ba43-7c3800be9f0f","cell_type":"markdown","source":"## Summary\n\nStage 1 trained at 160x160 instead of 224x224 to cut time, then Stage 2 fine-tuned at full 224x224. Swapped MixUp for cutout based on other participants' experience with this dataset. Kept focal loss and TTA from last week.\n\nNext: possibly try a larger backbone now that Stage 1 is faster, or revisit MixUp vs cutout with the actual numbers from this run to see which one this dataset responds to better.","metadata":{}}]}