{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nimport json\nimport time\nimport random\nimport shutil\nfrom pathlib import Path\n\nos.environ.setdefault(\"CUDA_DEVICE_ORDER\", \"PCI_BUS_ID\")\nos.environ.setdefault(\"CUDA_VISIBLE_DEVICES\", \"0\")\nos.environ.setdefault(\"TOKENIZERS_PARALLELISM\", \"false\")\n\nCPU_COUNT = os.cpu_count() or 2\nCPU_THREAD_LIMIT = min(4, CPU_COUNT)\nos.environ.setdefault(\"OMP_NUM_THREADS\", str(CPU_THREAD_LIMIT))\nos.environ.setdefault(\"MKL_NUM_THREADS\", str(CPU_THREAD_LIMIT))\nos.environ.setdefault(\"NUMEXPR_NUM_THREADS\", str(CPU_THREAD_LIMIT))\nos.environ.setdefault(\"OPENBLAS_NUM_THREADS\", str(CPU_THREAD_LIMIT))\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\n\nfrom PIL import Image, ImageFile\nImageFile.LOAD_TRUNCATED_IMAGES = True\n\ntry:\n    import cv2\n    CV2_AVAILABLE = True\n    cv2.setNumThreads(0)\nexcept Exception:\n    cv2 = None\n    CV2_AVAILABLE = False\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\n\nimport torchvision.models as models\nimport torchvision.transforms as transforms\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, f1_score, classification_report, confusion_matrix\n\ntry:\n    torch.set_num_threads(CPU_THREAD_LIMIT)\n    torch.set_num_interop_threads(max(1, CPU_THREAD_LIMIT // 2))\nexcept Exception:\n    pass\n\ndef clean_memory(label: str = \"\", verbose: bool | None = None, kill_workers: bool | None = None, deep: bool | None = None):\n    gc.collect()\n    if torch.cuda.is_available():\n        try:\n            torch.cuda.empty_cache()\n            torch.cuda.ipc_collect()\n        except Exception:\n            pass\n    return None\n\ndef memory_guard(label: str = \"\"):\n    return None\n\ndef cell_cleanup(label: str = \"\"):\n    return None\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:08:09.171420Z","iopub.execute_input":"2026-06-18T16:08:09.172196Z","iopub.status.idle":"2026-06-18T16:08:18.773528Z","shell.execute_reply.started":"2026-06-18T16:08:09.172164Z","shell.execute_reply":"2026-06-18T16:08:18.772726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42\nIMG_SIZE = 224\nNUM_CLASSES = 10\n\nRANDOMIZE_SUBJECT_SPLIT = False\nSUBJECT_SPLIT_SEED = SEED\nVAL_SUBJECT_RATIO = 0.20\nSAVE_SPLIT_INFO = True\n\nUSE_AMP = True\nUSE_CHANNELS_LAST = True\nUSE_OPENCV_IMAGE_READER = True\nUSE_RESIZED_IMAGE_CACHE = True\nREBUILD_IMAGE_CACHE = False\nCACHE_IMAGE_SIZE = 256\nCACHE_JPEG_QUALITY = 95\n\nSHOW_PROGRESS = True\nPROGRESS_MININTERVAL = 3.0\nPROGRESS_MINITERS = 3\n\nBATCH_SIZE = 128\nVAL_BATCH_SIZE = 256\nNUM_WORKERS = min(4, max(2, CPU_COUNT))\nPREFETCH_FACTOR = 2\nPIN_MEMORY = True\nUSE_PERSISTENT_WORKERS = True\nCACHE_NUM_WORKERS = min(4, max(2, CPU_COUNT))\nDATALOADER_WORKER_THREADS = 1\nMAX_TRAIN_BATCHES = None\nMAX_VAL_BATCHES = None\n\nUSE_EARLY_STOPPING = False\nUSE_REDUCE_LR = True\nUSE_GRAD_CLIP = True\nLR_HEAD = 3e-4\nLR_FINE = 1e-4\nLR_BASELINE = 1e-3\nWEIGHT_DECAY = 1e-4\nLABEL_SMOOTHING = 0.03\nDROPOUT = 0.30\nMAX_GRAD_NORM = 1.0\n\nLR_REDUCE_FACTOR = 0.5\nLR_REDUCE_PATIENCE = 2\nLR_REDUCE_THRESHOLD = 1e-4\nMIN_LR = 1e-6\n\nOVERFIT_GAP_WARN = 0.12\nOVERFIT_F1_GAP_WARN = 0.12\nOVERFIT_LOSS_RATIO_WARN = 1.35\n\nENABLE_GRAD_CAM = True\n\nEXPERIMENT_EPOCHS = {\n    \"effb0_pretrained_finetune\": {\"head\": 0, \"fine\": 25},\n    \"effb0_no_pretrain_scratch\": {\"head\": 0, \"fine\": 25},\n    \"effb0_head_only_frozen\": {\"head\": 8, \"fine\": 0},\n    \"effb0_head_warmup\": {\"head\": 5, \"fine\": 20},\n    \"effb0_no_aug\": {\"head\": 0, \"fine\": 15},\n    \"effb0_strong_aug\": {\"head\": 0, \"fine\": 15},\n    \"effb0_low_lr\": {\"head\": 0, \"fine\": 15},\n    \"effb0_high_lr\": {\"head\": 0, \"fine\": 15},\n    \"effb0_no_dropout\": {\"head\": 0, \"fine\": 15},\n    \"effb0_high_dropout\": {\"head\": 0, \"fine\": 15},\n    \"effb0_no_label_smoothing\": {\"head\": 0, \"fine\": 15},\n}\n\nOUTPUT_DIR = Path(\"/kaggle/working/statefarm_effb0_project\") if Path(\"/kaggle/working\").exists() else Path(\"./statefarm_effb0_project\")\nOUTPUT_DIR.mkdir(parents=True, exist_ok=True)\nCACHE_DIR = OUTPUT_DIR / f\"image_cache_{CACHE_IMAGE_SIZE}\"\n\nprint(\"BATCH_SIZE =\", BATCH_SIZE)\nprint(\"VAL_BATCH_SIZE =\", VAL_BATCH_SIZE)\nprint(\"NUM_WORKERS =\", NUM_WORKERS)\nprint(\"OUTPUT_DIR =\", OUTPUT_DIR)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:08:18.776043Z","iopub.execute_input":"2026-06-18T16:08:18.776394Z","iopub.status.idle":"2026-06-18T16:08:18.791915Z","shell.execute_reply.started":"2026-06-18T16:08:18.776369Z","shell.execute_reply":"2026-06-18T16:08:18.791343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def seed_everything(seed: int = 42):\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.benchmark = True\n    torch.backends.cudnn.deterministic = False\n\n    try:\n        torch.backends.cuda.matmul.allow_tf32 = True\n        torch.backends.cudnn.allow_tf32 = True\n    except Exception:\n        pass\n\nseed_everything(SEED)\n\nDEVICE = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nAMP_ENABLED = bool(USE_AMP and DEVICE.type == \"cuda\")\nprint(\"Thiết bị chính đang dùng:\", DEVICE)\nprint(\"Số GPU CUDA PyTorch thấy:\", torch.cuda.device_count())\nprint(\"AMP_ENABLED:\", AMP_ENABLED)\n\nif torch.cuda.is_available():\n    print(\"Tên GPU:\", torch.cuda.get_device_name(0))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:08:18.793799Z","iopub.execute_input":"2026-06-18T16:08:18.794011Z","iopub.status.idle":"2026-06-18T16:08:19.041857Z","shell.execute_reply.started":"2026-06-18T16:08:18.793989Z","shell.execute_reply":"2026-06-18T16:08:19.041174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CANDIDATE_DATA_ROOTS = [\n    Path(\"/kaggle/input/state-farm-distracted-driver-detection\"),\n    Path(\"/kaggle/input/competitions/state-farm-distracted-driver-detection\"),\n    Path(\"/content/data/state-farm-distracted-driver-detection\"),\n    Path(\"./state-farm-distracted-driver-detection\"),\n]\n\nDATA_ROOT = None\nfor candidate in CANDIDATE_DATA_ROOTS:\n    if (candidate / \"driver_imgs_list.csv\").exists():\n        DATA_ROOT = candidate\n        break\n\nif DATA_ROOT is None:\n    raise FileNotFoundError(\"Không tìm thấy driver_imgs_list.csv. Hãy kiểm tra lại đường dẫn dataset trong CANDIDATE_DATA_ROOTS.\")\n\nCSV_PATH = DATA_ROOT / \"driver_imgs_list.csv\"\nTRAIN_DIR = DATA_ROOT / \"imgs\" / \"train\"\nTEST_DIR = DATA_ROOT / \"imgs\" / \"test\"\n\nprint(\"DATA_ROOT:\", DATA_ROOT)\nprint(\"CSV_PATH:\", CSV_PATH)\nprint(\"TRAIN_DIR:\", TRAIN_DIR)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:08:19.043912Z","iopub.execute_input":"2026-06-18T16:08:19.044126Z","iopub.status.idle":"2026-06-18T16:08:19.050936Z","shell.execute_reply.started":"2026-06-18T16:08:19.044097Z","shell.execute_reply":"2026-06-18T16:08:19.050193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(CSV_PATH)\nrequired_cols = {\"subject\", \"classname\", \"img\"}\nmissing_cols = required_cols - set(df.columns)\nif missing_cols:\n    raise ValueError(f\"CSV thiếu cột bắt buộc: {missing_cols}\")\n\nclass_names = [f\"c{i}\" for i in range(NUM_CLASSES)]\nclass_to_idx = {name: idx for idx, name in enumerate(class_names)}\nidx_to_class = {idx: name for name, idx in class_to_idx.items()}\n\nclass_description = {\n    \"c0\": \"safe driving\",\n    \"c1\": \"texting - right\",\n    \"c2\": \"talking on the phone - right\",\n    \"c3\": \"texting - left\",\n    \"c4\": \"talking on the phone - left\",\n    \"c5\": \"operating the radio\",\n    \"c6\": \"drinking\",\n    \"c7\": \"reaching behind\",\n    \"c8\": \"hair and makeup\",\n    \"c9\": \"talking to passenger\",\n}\n\ndf[\"label\"] = df[\"classname\"].map(class_to_idx)\ndf[\"img_path\"] = df.apply(lambda row: str(TRAIN_DIR / row[\"classname\"] / row[\"img\"]), axis=1)\ndf[\"exists\"] = df[\"img_path\"].apply(lambda p: Path(p).exists())\n\nif not df[\"exists\"].all():\n    missing_count = int((~df[\"exists\"]).sum())\n    raise FileNotFoundError(f\"Có {missing_count} ảnh không tồn tại. Hãy kiểm tra TRAIN_DIR.\")\n\ndisplay(df.head())\nprint(\"Tổng số ảnh:\", len(df))\nprint(\"Số tài xế/subject:\", df[\"subject\"].nunique())\nprint(\"Số class:\", df[\"classname\"].nunique())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:08:19.051752Z","iopub.execute_input":"2026-06-18T16:08:19.051963Z","iopub.status.idle":"2026-06-18T16:09:21.749933Z","shell.execute_reply.started":"2026-06-18T16:08:19.051944Z","shell.execute_reply":"2026-06-18T16:09:21.749242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_counts = df[\"classname\"].value_counts().sort_index()\ndriver_counts = df[\"subject\"].value_counts().sort_values(ascending=False)\n\nplt.figure(figsize=(10, 4))\nclass_counts.plot(kind=\"bar\")\nplt.title(\"Phân phối số ảnh theo class\")\nplt.xlabel(\"Class\")\nplt.ylabel(\"Số ảnh\")\nplt.xticks(rotation=0)\nplt.tight_layout()\nplt.show()\n\nplt.figure(figsize=(12, 4))\ndriver_counts.plot(kind=\"bar\")\nplt.title(\"Phân phối số ảnh theo subject/tài xế\")\nplt.xlabel(\"Subject\")\nplt.ylabel(\"Số ảnh\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:09:21.750923Z","iopub.execute_input":"2026-06-18T16:09:21.751435Z","iopub.status.idle":"2026-06-18T16:09:22.252190Z","shell.execute_reply.started":"2026-06-18T16:09:21.751409Z","shell.execute_reply":"2026-06-18T16:09:22.251517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(15, 8))\nfor label in range(NUM_CLASSES):\n    sample_row = df[df[\"label\"] == label].sample(1, random_state=SEED).iloc[0]\n    image = Image.open(sample_row[\"img_path\"]).convert(\"RGB\")\n    plt.subplot(2, 5, label + 1)\n    plt.imshow(image)\n    plt.title(f\"{sample_row['classname']}\\n{class_description[sample_row['classname']]}\")\n    plt.axis(\"off\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:09:22.253111Z","iopub.execute_input":"2026-06-18T16:09:22.253478Z","iopub.status.idle":"2026-06-18T16:09:23.509446Z","shell.execute_reply.started":"2026-06-18T16:09:22.253455Z","shell.execute_reply":"2026-06-18T16:09:23.508528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def subject_split(dataframe: pd.DataFrame, val_ratio: float = 0.2, seed=None):\n\n    subjects = sorted(dataframe[\"subject\"].unique())\n    train_subjects, val_subjects = train_test_split(\n        subjects,\n        test_size=val_ratio,\n        random_state=seed,\n        shuffle=True,\n    )\n    train_part = dataframe[dataframe[\"subject\"].isin(train_subjects)].reset_index(drop=True)\n    val_part = dataframe[dataframe[\"subject\"].isin(val_subjects)].reset_index(drop=True)\n    return train_part, val_part, train_subjects, val_subjects\n\nif RANDOMIZE_SUBJECT_SPLIT:\n    split_seed = random.SystemRandom().randint(0, 2**32 - 1)\nelse:\n    split_seed = SUBJECT_SPLIT_SEED\n\ntrain_df, val_df, train_subjects, val_subjects = subject_split(df, val_ratio=VAL_SUBJECT_RATIO, seed=split_seed)\n\nintersection = set(train_df[\"subject\"]).intersection(set(val_df[\"subject\"]))\nassert len(intersection) == 0, \"Data leakage: subject xuất hiện ở cả train và validation!\"\n\nprint(\"Split mode: RANDOM SUBJECT SPLIT\" if RANDOMIZE_SUBJECT_SPLIT else \"Split mode: FIXED SUBJECT-WISE SPLIT\")\nprint(\"Split seed used:\", split_seed)\nprint(\"Train images:\", len(train_df))\nprint(\"Val images:\", len(val_df))\nprint(\"Train subjects:\", len(train_subjects), sorted(train_subjects))\nprint(\"Val subjects:\", len(val_subjects), sorted(val_subjects))\nprint(\"Subject overlap:\", intersection)\n\nif SAVE_SPLIT_INFO:\n    split_info = {\n        \"randomize_subject_split\": bool(RANDOMIZE_SUBJECT_SPLIT),\n        \"split_seed_used\": split_seed,\n        \"val_subject_ratio\": VAL_SUBJECT_RATIO,\n        \"train_subjects\": sorted([str(x) for x in train_subjects]),\n        \"val_subjects\": sorted([str(x) for x in val_subjects]),\n        \"num_train_images\": int(len(train_df)),\n        \"num_val_images\": int(len(val_df)),\n    }\n    split_path = OUTPUT_DIR / \"subject_split_info.json\"\n    with open(split_path, \"w\", encoding=\"utf-8\") as f:\n        json.dump(split_info, f, ensure_ascii=False, indent=2)\n    print(\"Đã lưu thông tin split vào:\", split_path)\n\nprint(\"Dùng toàn bộ dữ liệu sau subject-wise split để train/evaluate.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:09:23.511178Z","iopub.execute_input":"2026-06-18T16:09:23.511558Z","iopub.status.idle":"2026-06-18T16:09:23.538231Z","shell.execute_reply.started":"2026-06-18T16:09:23.511532Z","shell.execute_reply":"2026-06-18T16:09:23.537426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def safe_cache_filename(row: pd.Series) -> str:\n\n    return f\"{row['classname']}_{Path(row['img']).stem}.jpg\"\n\ndef resize_and_cache_one(src_path: str, dst_path: Path, size: int = CACHE_IMAGE_SIZE) -> bool:\n\n    dst_path.parent.mkdir(parents=True, exist_ok=True)\n    if dst_path.exists() and not REBUILD_IMAGE_CACHE:\n        return False\n\n    image = None\n    if CV2_AVAILABLE:\n        bgr = cv2.imread(str(src_path), cv2.IMREAD_COLOR)\n        if bgr is not None:\n            image = cv2.cvtColor(bgr, cv2.COLOR_BGR2RGB)\n    if image is None:\n        image = np.array(Image.open(src_path).convert(\"RGB\"))\n\n    resized = cv2.resize(image, (size, size), interpolation=cv2.INTER_AREA) if CV2_AVAILABLE else np.array(Image.fromarray(image).resize((size, size)))\n    bgr_out = cv2.cvtColor(resized, cv2.COLOR_RGB2BGR) if CV2_AVAILABLE else resized[:, :, ::-1]\n    ok = cv2.imwrite(str(dst_path), bgr_out, [int(cv2.IMWRITE_JPEG_QUALITY), int(CACHE_JPEG_QUALITY)]) if CV2_AVAILABLE else False\n    if not ok:\n        Image.fromarray(resized).save(dst_path, quality=int(CACHE_JPEG_QUALITY))\n    return True\n\ndef build_resized_cache_for_dataframe(dataframe: pd.DataFrame, split_name: str) -> pd.DataFrame:\n\n    from concurrent.futures import ThreadPoolExecutor, as_completed\n\n    cached_df = dataframe.copy().reset_index(drop=True)\n    cached_paths = [None] * len(cached_df)\n    cache_split_dir = CACHE_DIR / split_name\n    max_workers = int(globals().get(\"CACHE_NUM_WORKERS\", 8))\n\n    def _cache_one(idx: int, row_dict: dict):\n        src_path = row_dict[\"img_path\"]\n        dst_path = cache_split_dir / row_dict[\"classname\"] / f\"{row_dict['classname']}_{Path(row_dict['img']).stem}.jpg\"\n        created = resize_and_cache_one(src_path, dst_path, size=CACHE_IMAGE_SIZE)\n        return idx, str(dst_path), int(created)\n\n    rows = [(idx, cached_df.iloc[idx].to_dict()) for idx in range(len(cached_df))]\n    created_count = 0\n\n    with ThreadPoolExecutor(max_workers=max_workers) as executor:\n        futures = [executor.submit(_cache_one, idx, row_dict) for idx, row_dict in rows]\n        iterator = tqdm(\n            as_completed(futures),\n            total=len(futures),\n            desc=f\"Cache ảnh {split_name} {CACHE_IMAGE_SIZE}px | workers={max_workers}\",\n            dynamic_ncols=True,\n            mininterval=float(globals().get(\"PROGRESS_MININTERVAL\", 5.0)),\n            miniters=int(globals().get(\"PROGRESS_MINITERS\", 5)),\n            disable=not is_progress_enabled() if \"is_progress_enabled\" in globals() else False,\n        )\n        for fut in iterator:\n            idx, cached_path, created = fut.result()\n            cached_paths[idx] = cached_path\n            created_count += created\n\n    cached_df[\"img_path_original\"] = cached_df[\"img_path\"]\n    cached_df[\"img_path\"] = cached_paths\n    print(f\"{split_name}: dùng cache {len(cached_df)} ảnh, tạo mới {created_count} ảnh, workers={max_workers}, thư mục: {cache_split_dir}\")\n    return cached_df.reset_index(drop=True)\n\nif USE_RESIZED_IMAGE_CACHE:\n    print(\"USE_RESIZED_IMAGE_CACHE=True: bắt đầu chuẩn bị cache ảnh resize để tăng tốc các epoch sau.\")\n    train_df = build_resized_cache_for_dataframe(train_df, \"train\")\n    val_df = build_resized_cache_for_dataframe(val_df, \"val\")\nelse:\n    print(\"USE_RESIZED_IMAGE_CACHE=False: đọc trực tiếp ảnh gốc. Nếu train nhiều epoch/model, nên bật True để nhanh hơn sau bước cache ban đầu.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:09:23.539368Z","iopub.execute_input":"2026-06-18T16:09:23.539636Z","iopub.status.idle":"2026-06-18T16:10:39.974165Z","shell.execute_reply.started":"2026-06-18T16:09:23.539614Z","shell.execute_reply":"2026-06-18T16:10:39.973243Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"split_distribution = pd.DataFrame({\n    \"train\": train_df[\"classname\"].value_counts().sort_index(),\n    \"val\": val_df[\"classname\"].value_counts().sort_index(),\n}).fillna(0).astype(int)\n\nsplit_distribution[\"val_ratio_%\"] = (split_distribution[\"val\"] / (split_distribution[\"train\"] + split_distribution[\"val\"]) * 100).round(2)\ndisplay(split_distribution)\n\nsplit_distribution[[\"train\", \"val\"]].plot(kind=\"bar\", figsize=(10, 4))\nplt.title(\"Phân phối class sau khi chia theo subject\")\nplt.xlabel(\"Class\")\nplt.ylabel(\"Số ảnh\")\nplt.xticks(rotation=0)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:10:39.976814Z","iopub.execute_input":"2026-06-18T16:10:39.977098Z","iopub.status.idle":"2026-06-18T16:10:40.181244Z","shell.execute_reply.started":"2026-06-18T16:10:39.977075Z","shell.execute_reply":"2026-06-18T16:10:40.180620Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMAGENET_MEAN = [0.485, 0.456, 0.406]\nIMAGENET_STD = [0.229, 0.224, 0.225]\n\nimproved_train_tfms = transforms.Compose([\n    transforms.Resize((256, 256)),\n    transforms.RandomResizedCrop(IMG_SIZE, scale=(0.90, 1.00), ratio=(0.96, 1.04)),\n    transforms.ColorJitter(brightness=0.15, contrast=0.15, saturation=0.08, hue=0.015),\n    transforms.ToTensor(),\n    transforms.Normalize(IMAGENET_MEAN, IMAGENET_STD),\n    transforms.RandomErasing(p=0.15, scale=(0.02, 0.06), ratio=(0.3, 3.3)),\n])\n\nno_aug_train_tfms = transforms.Compose([\n    transforms.Resize((IMG_SIZE, IMG_SIZE)),\n    transforms.ToTensor(),\n    transforms.Normalize(IMAGENET_MEAN, IMAGENET_STD),\n])\n\nstrong_train_tfms = transforms.Compose([\n    transforms.Resize((256, 256)),\n    transforms.RandomResizedCrop(IMG_SIZE, scale=(0.75, 1.00), ratio=(0.85, 1.15)),\n    transforms.RandomRotation(degrees=15),\n    transforms.ColorJitter(brightness=0.35, contrast=0.35, saturation=0.20, hue=0.04),\n    transforms.RandomPerspective(distortion_scale=0.18, p=0.35),\n    transforms.ToTensor(),\n    transforms.Normalize(IMAGENET_MEAN, IMAGENET_STD),\n    transforms.RandomErasing(p=0.35, scale=(0.03, 0.18), ratio=(0.3, 3.3)),\n])\n\nval_tfms = transforms.Compose([\n    transforms.Resize((IMG_SIZE, IMG_SIZE)),\n    transforms.ToTensor(),\n    transforms.Normalize(IMAGENET_MEAN, IMAGENET_STD),\n])\n\nprint(\"Đã tạo transform train/validation.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:10:40.182091Z","iopub.execute_input":"2026-06-18T16:10:40.182448Z","iopub.status.idle":"2026-06-18T16:10:40.191868Z","shell.execute_reply.started":"2026-06-18T16:10:40.182410Z","shell.execute_reply":"2026-06-18T16:10:40.191206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class StateFarmDataset(Dataset):\n    def __init__(self, dataframe: pd.DataFrame, transform=None):\n\n        dataframe = dataframe.reset_index(drop=True)\n        self.img_paths = dataframe[\"img_path\"].astype(str).tolist()\n        self.labels = dataframe[\"label\"].astype(int).tolist()\n        self.transform = transform\n        self.use_cv2 = bool(globals().get(\"USE_OPENCV_IMAGE_READER\", True)) and CV2_AVAILABLE\n\n    def __len__(self):\n        return len(self.img_paths)\n\n    def read_image_rgb(self, path: str):\n        if self.use_cv2:\n            image_bgr = cv2.imread(path, cv2.IMREAD_COLOR)\n            if image_bgr is not None:\n                image_rgb = cv2.cvtColor(image_bgr, cv2.COLOR_BGR2RGB)\n                return Image.fromarray(image_rgb)\n        return Image.open(path).convert(\"RGB\")\n\n    def __getitem__(self, idx):\n        image = self.read_image_rgb(self.img_paths[idx])\n        label = self.labels[idx]\n        if self.transform is not None:\n            image = self.transform(image)\n        return image, label\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:10:40.192935Z","iopub.execute_input":"2026-06-18T16:10:40.193232Z","iopub.status.idle":"2026-06-18T16:10:40.208613Z","shell.execute_reply.started":"2026-06-18T16:10:40.193197Z","shell.execute_reply":"2026-06-18T16:10:40.207803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def seed_worker(worker_id):\n    worker_seed = SEED + worker_id\n    np.random.seed(worker_seed)\n    random.seed(worker_seed)\n    if CV2_AVAILABLE:\n\n        cv2.setNumThreads(int(globals().get(\"DATALOADER_WORKER_THREADS\", 1)))\n    try:\n        torch.set_num_threads(int(globals().get(\"DATALOADER_WORKER_THREADS\", 1)))\n    except Exception:\n        pass\n\ndef make_loader_kwargs(num_workers: int):\n\n    kwargs = {\n        \"num_workers\": int(num_workers),\n        \"pin_memory\": bool(globals().get(\"PIN_MEMORY\", True)) and (DEVICE.type == \"cuda\"),\n        \"worker_init_fn\": seed_worker,\n    }\n    if num_workers > 0:\n        kwargs[\"persistent_workers\"] = bool(globals().get(\"USE_PERSISTENT_WORKERS\", False))\n        kwargs[\"prefetch_factor\"] = int(globals().get(\"PREFETCH_FACTOR\", 2))\n    return kwargs\n\ndef make_loaders(train_transform, batch_size: int = BATCH_SIZE):\n    generator = torch.Generator()\n    generator.manual_seed(SEED)\n    train_dataset = StateFarmDataset(train_df, transform=train_transform)\n    val_dataset = StateFarmDataset(val_df, transform=val_tfms)\n    loader_kwargs = make_loader_kwargs(NUM_WORKERS)\n\n    train_loader = DataLoader(\n        train_dataset,\n        batch_size=int(batch_size),\n        shuffle=True,\n        generator=generator,\n        drop_last=False,\n        **loader_kwargs,\n    )\n\n    val_loader = DataLoader(\n        val_dataset,\n        batch_size=int(globals().get(\"VAL_BATCH_SIZE\", batch_size)),\n        shuffle=False,\n        drop_last=False,\n        **loader_kwargs,\n    )\n    return train_loader, val_loader\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:10:40.209637Z","iopub.execute_input":"2026-06-18T16:10:40.210257Z","iopub.status.idle":"2026-06-18T16:10:40.225308Z","shell.execute_reply.started":"2026-06-18T16:10:40.210233Z","shell.execute_reply":"2026-06-18T16:10:40.224611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_efficientnet_b0(num_classes: int = NUM_CLASSES, dropout: float = DROPOUT, pretrained: bool = True):\n    weights = models.EfficientNet_B0_Weights.DEFAULT if pretrained else None\n    model = models.efficientnet_b0(weights=weights)\n    in_features = model.classifier[1].in_features\n    model.classifier = nn.Sequential(\n        nn.Dropout(p=dropout, inplace=True),\n        nn.Linear(in_features, num_classes),\n    )\n    return model\n\ndef unwrap_model(model: nn.Module):\n    return model\n\ndef move_model_to_device(model: nn.Module):\n    model = model.to(DEVICE)\n    if bool(globals().get(\"USE_CHANNELS_LAST\", False)) and DEVICE.type == \"cuda\":\n        model = model.to(memory_format=torch.channels_last)\n    return model\n\ndef set_backbone_trainable(model: nn.Module, trainable: bool):\n    base_model = unwrap_model(model)\n    if hasattr(base_model, \"features\"):\n        for param in base_model.features.parameters():\n            param.requires_grad = trainable\n    if hasattr(base_model, \"classifier\"):\n        for param in base_model.classifier.parameters():\n            param.requires_grad = True\n    return model\n\ndef count_trainable_params(model: nn.Module):\n    return sum(p.numel() for p in model.parameters() if p.requires_grad)\n\ndef save_model_state(model: nn.Module, path: Path):\n    torch.save(unwrap_model(model).state_dict(), path)\n\ndef load_model_state(model: nn.Module, path: Path, map_location=None):\n    state_dict = torch.load(path, map_location=map_location or DEVICE)\n    unwrap_model(model).load_state_dict(state_dict)\n    return model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:10:40.226303Z","iopub.execute_input":"2026-06-18T16:10:40.226586Z","iopub.status.idle":"2026-06-18T16:10:40.241992Z","shell.execute_reply.started":"2026-06-18T16:10:40.226556Z","shell.execute_reply":"2026-06-18T16:10:40.241343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def is_progress_enabled():\n    return bool(globals().get(\"SHOW_PROGRESS\", True))\n\ndef maybe_channels_last(images):\n    if bool(globals().get(\"USE_CHANNELS_LAST\", False)) and DEVICE.type == \"cuda\" and images.ndim == 4:\n        return images.contiguous(memory_format=torch.channels_last)\n    return images\n\ndef train_one_epoch(model, loader, criterion, optimizer, scaler, desc=\"Train\", max_batches=None):\n\n    model.train()\n    total_loss = 0.0\n    all_preds = []\n    all_targets = []\n    max_batches = globals().get(\"MAX_TRAIN_BATCHES\", None) if max_batches is None else max_batches\n\n    progress_bar = tqdm(\n        loader,\n        desc=desc,\n        leave=False,\n        dynamic_ncols=True,\n        mininterval=float(globals().get(\"PROGRESS_MININTERVAL\", 3.0)),\n        miniters=int(globals().get(\"PROGRESS_MINITERS\", 3)),\n        disable=not is_progress_enabled(),\n        total=min(len(loader), max_batches) if max_batches is not None else len(loader),\n    )\n\n    for batch_idx, (images, targets) in enumerate(progress_bar, start=1):\n        if max_batches is not None and batch_idx > max_batches:\n            break\n        images = images.to(DEVICE, non_blocking=True)\n        images = maybe_channels_last(images)\n        targets = targets.to(DEVICE, non_blocking=True)\n        optimizer.zero_grad(set_to_none=True)\n\n        with torch.amp.autocast(device_type=DEVICE.type, enabled=AMP_ENABLED):\n            logits = model(images)\n            loss = criterion(logits, targets)\n\n        scaler.scale(loss).backward()\n\n        if bool(globals().get(\"USE_GRAD_CLIP\", False)):\n            scaler.unscale_(optimizer)\n            torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=float(globals().get(\"MAX_GRAD_NORM\", 1.0)))\n\n        scaler.step(optimizer)\n        scaler.update()\n\n        batch_size = images.size(0)\n        total_loss += loss.item() * batch_size\n        preds = logits.argmax(dim=1).detach().cpu().numpy()\n        all_preds.extend(preds.tolist())\n        all_targets.extend(targets.detach().cpu().numpy().tolist())\n\n        running_loss = total_loss / max(1, len(all_targets))\n        progress_bar.set_postfix({\n            \"loss\": f\"{running_loss:.4f}\",\n            \"batch\": f\"{batch_idx}/{max_batches if max_batches is not None else len(loader)}\",\n            \"lr\": f\"{optimizer.param_groups[0]['lr']:.1e}\",\n        })\n\n    avg_loss = total_loss / max(1, len(all_targets))\n    acc = accuracy_score(all_targets, all_preds) if all_targets else 0.0\n    macro_f1 = f1_score(all_targets, all_preds, average=\"macro\", zero_division=0) if all_targets else 0.0\n    return avg_loss, acc, macro_f1\n\n@torch.no_grad()\ndef evaluate(model, loader, criterion, desc=\"Valid\", max_batches=None):\n    model.eval()\n    total_loss = 0.0\n    all_probs = []\n    all_preds = []\n    all_targets = []\n    max_batches = globals().get(\"MAX_VAL_BATCHES\", None) if max_batches is None else max_batches\n\n    progress_bar = tqdm(\n        loader,\n        desc=desc,\n        leave=False,\n        dynamic_ncols=True,\n        mininterval=float(globals().get(\"PROGRESS_MININTERVAL\", 3.0)),\n        miniters=int(globals().get(\"PROGRESS_MINITERS\", 3)),\n        disable=not is_progress_enabled(),\n        total=min(len(loader), max_batches) if max_batches is not None else len(loader),\n    )\n\n    for batch_idx, (images, targets) in enumerate(progress_bar, start=1):\n        if max_batches is not None and batch_idx > max_batches:\n            break\n        images = images.to(DEVICE, non_blocking=True)\n        images = maybe_channels_last(images)\n        targets = targets.to(DEVICE, non_blocking=True)\n\n        with torch.amp.autocast(device_type=DEVICE.type, enabled=AMP_ENABLED):\n            logits = model(images)\n            loss = criterion(logits, targets)\n\n        probs = torch.softmax(logits, dim=1)\n        preds = probs.argmax(dim=1)\n\n        batch_size = images.size(0)\n        total_loss += loss.item() * batch_size\n        all_probs.extend(probs.detach().cpu().numpy().tolist())\n        all_preds.extend(preds.detach().cpu().numpy().tolist())\n        all_targets.extend(targets.detach().cpu().numpy().tolist())\n\n        running_loss = total_loss / max(1, len(all_targets))\n        progress_bar.set_postfix({\n            \"loss\": f\"{running_loss:.4f}\",\n            \"batch\": f\"{batch_idx}/{max_batches if max_batches is not None else len(loader)}\",\n        })\n\n    avg_loss = total_loss / max(1, len(all_targets))\n    acc = accuracy_score(all_targets, all_preds) if all_targets else 0.0\n    macro_f1 = f1_score(all_targets, all_preds, average=\"macro\", zero_division=0) if all_targets else 0.0\n    return avg_loss, acc, macro_f1, np.array(all_targets), np.array(all_preds), np.array(all_probs)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:10:40.242970Z","iopub.execute_input":"2026-06-18T16:10:40.243287Z","iopub.status.idle":"2026-06-18T16:10:40.265393Z","shell.execute_reply.started":"2026-06-18T16:10:40.243264Z","shell.execute_reply":"2026-06-18T16:10:40.264729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_current_lr(optimizer):\n    return float(optimizer.param_groups[0][\"lr\"])\n\ndef make_plateau_scheduler(optimizer):\n    if not bool(globals().get(\"USE_REDUCE_LR\", True)):\n        return None\n    return optim.lr_scheduler.ReduceLROnPlateau(\n        optimizer,\n        mode=\"max\",\n        factor=float(globals().get(\"LR_REDUCE_FACTOR\", 0.5)),\n        patience=int(globals().get(\"LR_REDUCE_PATIENCE\", 2)),\n        threshold=float(globals().get(\"LR_REDUCE_THRESHOLD\", 1e-4)),\n        min_lr=float(globals().get(\"MIN_LR\", 1e-6)),\n    )\n\ndef diagnose_overfitting(train_loss, train_acc, train_f1, val_loss, val_acc, val_f1):\n    warnings = []\n    acc_gap = float(train_acc - val_acc)\n    f1_gap = float(train_f1 - val_f1)\n    loss_ratio = float(val_loss / max(train_loss, 1e-8))\n\n    if acc_gap > float(globals().get(\"OVERFIT_GAP_WARN\", 0.12)):\n        warnings.append(f\"train_acc cao hơn val_acc {acc_gap:.3f}\")\n    if f1_gap > float(globals().get(\"OVERFIT_F1_GAP_WARN\", 0.12)):\n        warnings.append(f\"train_f1 cao hơn val_f1 {f1_gap:.3f}\")\n    if loss_ratio > float(globals().get(\"OVERFIT_LOSS_RATIO_WARN\", 1.35)):\n        warnings.append(f\"val_loss/train_loss={loss_ratio:.2f}\")\n\n    return warnings, acc_gap, f1_gap, loss_ratio\n\ndef run_experiment(exp_name: str, config: dict):\n    print(\"\\n\" + \"=\" * 80)\n    print(f\"Bắt đầu thí nghiệm: {exp_name}\")\n    print(\"Mô tả:\", config[\"description\"])\n    print(\"=\" * 80)\n\n    clean_memory(label=f\"trước experiment {exp_name}\", verbose=False, kill_workers=True)\n    train_loader, val_loader = make_loaders(config[\"train_transform\"], batch_size=BATCH_SIZE)\n    criterion = nn.CrossEntropyLoss(label_smoothing=config.get(\"label_smoothing\", LABEL_SMOOTHING))\n    exp_dropout = float(config.get(\"dropout\", DROPOUT))\n    exp_weight_decay = float(config.get(\"weight_decay\", WEIGHT_DECAY))\n    exp_head_lr = float(config.get(\"head_lr\", LR_HEAD))\n    exp_fine_lr = float(config.get(\"fine_lr\", LR_FINE))\n\n    if config[\"model_type\"] == \"baseline_cnn\":\n        model = move_model_to_device(SmallCNN(num_classes=NUM_CLASSES))\n        optimizer = optim.AdamW(model.parameters(), lr=LR_BASELINE, weight_decay=WEIGHT_DECAY)\n        phases = [(\"baseline_train\", EXPERIMENT_EPOCHS[exp_name][\"fine\"], LR_BASELINE, True)]\n    else:\n        pretrained_flag = bool(config.get(\"pretrained\", True))\n        print(f\"Pretrained ImageNet: {pretrained_flag}\")\n        model = move_model_to_device(build_efficientnet_b0(num_classes=NUM_CLASSES, dropout=exp_dropout, pretrained=pretrained_flag))\n        phases = []\n        if EXPERIMENT_EPOCHS[exp_name][\"head\"] > 0:\n            phases.append((\"head_training\", EXPERIMENT_EPOCHS[exp_name][\"head\"], exp_head_lr, False))\n        if EXPERIMENT_EPOCHS[exp_name][\"fine\"] > 0:\n            phases.append((\"fine_tuning\", EXPERIMENT_EPOCHS[exp_name][\"fine\"], exp_fine_lr, True))\n\n    history = []\n    best_macro_f1 = -1.0\n    best_path = OUTPUT_DIR / f\"best_{exp_name}.pth\"\n    if best_path.exists():\n        best_path.unlink()\n    last_targets = None\n    last_preds = None\n    last_probs = None\n\n    for phase_name, num_epochs, lr, backbone_trainable in phases:\n        if num_epochs <= 0:\n            continue\n\n        if config[\"model_type\"] != \"baseline_cnn\":\n            set_backbone_trainable(model, trainable=backbone_trainable)\n            optimizer = optim.AdamW(filter(lambda p: p.requires_grad, model.parameters()), lr=lr, weight_decay=exp_weight_decay)\n\n        scheduler = make_plateau_scheduler(optimizer)\n        scaler = torch.amp.GradScaler(DEVICE.type, enabled=AMP_ENABLED)\n\n        print(f\"\\nPhase: {phase_name} | epochs={num_epochs} | lr={lr} | trainable_params={count_trainable_params(model):,}\")\n        print(f\"ReduceLROnPlateau: {scheduler is not None} | grad_clip={bool(globals().get('USE_GRAD_CLIP', False))}\")\n        print(f\"Batch giới hạn: train={MAX_TRAIN_BATCHES}, val={MAX_VAL_BATCHES} | channels_last={USE_CHANNELS_LAST}\")\n\n        for epoch in range(1, num_epochs + 1):\n            start_time = time.time()\n            lr_before = get_current_lr(optimizer)\n            train_desc = f\"[{exp_name}][{phase_name}][E{epoch:02d}/{num_epochs:02d}] train\"\n            val_desc = f\"[{exp_name}][{phase_name}][E{epoch:02d}/{num_epochs:02d}] valid\"\n\n            train_loss, train_acc, train_f1 = train_one_epoch(model, train_loader, criterion, optimizer, scaler, desc=train_desc, max_batches=MAX_TRAIN_BATCHES)\n            val_loss, val_acc, val_f1, y_true, y_pred, y_prob = evaluate(model, val_loader, criterion, desc=val_desc, max_batches=MAX_VAL_BATCHES)\n\n            if scheduler is not None:\n                scheduler.step(val_f1)\n            lr_after = get_current_lr(optimizer)\n            elapsed = time.time() - start_time\n\n            overfit_warnings, acc_gap, f1_gap, loss_ratio = diagnose_overfitting(\n                train_loss, train_acc, train_f1, val_loss, val_acc, val_f1\n            )\n\n            row = {\n                \"experiment\": exp_name,\n                \"phase\": phase_name,\n                \"epoch_in_phase\": epoch,\n                \"train_loss\": train_loss,\n                \"train_acc\": train_acc,\n                \"train_macro_f1\": train_f1,\n                \"val_loss\": val_loss,\n                \"val_acc\": val_acc,\n                \"val_macro_f1\": val_f1,\n                \"lr_before\": lr_before,\n                \"lr_after\": lr_after,\n                \"acc_gap\": acc_gap,\n                \"f1_gap\": f1_gap,\n                \"loss_ratio\": loss_ratio,\n                \"overfit_warning\": \" | \".join(overfit_warnings),\n                \"elapsed_sec\": elapsed,\n            }\n            history.append(row)\n\n            print(f\"[{exp_name}][{phase_name}][{epoch:02d}/{num_epochs:02d}] \"\n                  f\"train_loss={train_loss:.4f} train_acc={train_acc:.4f} train_f1={train_f1:.4f} | \"\n                  f\"val_loss={val_loss:.4f} val_acc={val_acc:.4f} val_f1={val_f1:.4f} | \"\n                  f\"lr={lr_after:.2e} | time={elapsed:.1f}s\")\n\n            if lr_after < lr_before:\n                print(f\"  ↳ ReduceLROnPlateau: LR giảm từ {lr_before:.2e} xuống {lr_after:.2e} vì validation macro F1 bị chững.\")\n\n            if overfit_warnings:\n                print(\"  ⚠️ Cảnh báo overfitting:\", \"; \".join(overfit_warnings))\n\n            if val_f1 > best_macro_f1:\n                best_macro_f1 = val_f1\n                save_model_state(model, best_path)\n                last_targets = y_true\n                last_preds = y_pred\n                last_probs = y_prob\n\n            if False:\n                clean_memory(label=f\"sau epoch {exp_name}/{phase_name}/{epoch}\", verbose=False, kill_workers=False, deep=False)\n            else:\n                memory_guard(label=f\"epoch {exp_name}/{phase_name}/{epoch}\")\n\n        clean_memory(label=f\"sau phase {exp_name}/{phase_name}\", verbose=False, kill_workers=False, deep=False)\n\n    if best_path.exists():\n        load_model_state(model, best_path, map_location=DEVICE)\n        val_loss, val_acc, val_f1, last_targets, last_preds, last_probs = evaluate(model, val_loader, criterion, desc=f\"[{exp_name}] final valid\")\n    else:\n        val_loss, val_acc, val_f1, last_targets, last_preds, last_probs = evaluate(model, val_loader, criterion, desc=f\"[{exp_name}] final valid\")\n\n    result = {\n        \"experiment\": exp_name,\n        \"description\": config[\"description\"],\n        \"best_checkpoint\": str(best_path),\n        \"best_val_loss\": float(val_loss),\n        \"best_val_acc\": float(val_acc),\n        \"best_val_macro_f1\": float(val_f1),\n        \"history\": history,\n    }\n\n    history_path = OUTPUT_DIR / f\"history_{exp_name}.csv\"\n    pd.DataFrame(history).to_csv(history_path, index=False)\n\n    del train_loader, val_loader\n    del model, optimizer, scheduler, scaler, criterion\n    clean_memory(\n        label=f\"sau experiment {exp_name}\",\n        verbose=False,\n        kill_workers=True,\n        deep=bool(False),\n    )\n\n    return result, last_targets, last_preds, last_probs\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:10:40.266379Z","iopub.execute_input":"2026-06-18T16:10:40.266629Z","iopub.status.idle":"2026-06-18T16:10:40.289457Z","shell.execute_reply.started":"2026-06-18T16:10:40.266608Z","shell.execute_reply":"2026-06-18T16:10:40.288650Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"EXPERIMENTS = {\n\n    \"effb0_pretrained_finetune\": {\n        \"model_type\": \"efficientnet_b0\",\n        \"pretrained\": True,\n        \"train_transform\": improved_train_tfms,\n        \"label_smoothing\": LABEL_SMOOTHING,\n        \"dropout\": DROPOUT,\n        \"fine_lr\": LR_FINE,\n        \"description\": \"Có pretrained ImageNet, fine-tune trực tiếp toàn bộ EfficientNet-B0.\",\n    },\n\n    \"effb0_no_pretrain_scratch\": {\n        \"model_type\": \"efficientnet_b0\",\n        \"pretrained\": False,\n        \"train_transform\": improved_train_tfms,\n        \"label_smoothing\": LABEL_SMOOTHING,\n        \"dropout\": DROPOUT,\n        \"fine_lr\": LR_FINE,\n        \"description\": \"Không dùng pretrained ImageNet, train EfficientNet-B0 từ đầu để thấy vai trò của transfer learning.\",\n    },\n\n    \"effb0_head_only_frozen\": {\n        \"model_type\": \"efficientnet_b0\",\n        \"pretrained\": True,\n        \"train_transform\": improved_train_tfms,\n        \"label_smoothing\": LABEL_SMOOTHING,\n        \"dropout\": DROPOUT,\n        \"head_lr\": LR_HEAD,\n        \"description\": \"Chỉ train classifier head, giữ backbone pretrained cố định để xem đặc trưng ImageNet có đủ dùng không.\",\n    },\n\n    \"effb0_head_warmup\": {\n        \"model_type\": \"efficientnet_b0\",\n        \"pretrained\": True,\n        \"train_transform\": improved_train_tfms,\n        \"label_smoothing\": LABEL_SMOOTHING,\n        \"dropout\": DROPOUT,\n        \"head_lr\": LR_HEAD,\n        \"fine_lr\": LR_FINE,\n        \"description\": \"Head warm-up rồi fine-tune toàn bộ, kiểm tra việc làm nóng classifier có giúp ổn định hơn không.\",\n    },\n\n    \"effb0_no_aug\": {\n        \"model_type\": \"efficientnet_b0\",\n        \"pretrained\": True,\n        \"train_transform\": no_aug_train_tfms,\n        \"label_smoothing\": LABEL_SMOOTHING,\n        \"dropout\": DROPOUT,\n        \"fine_lr\": LR_FINE,\n        \"description\": \"Có pretrained nhưng không augmentation để quan sát mô hình học thuộc nhanh thế nào.\",\n    },\n\n    \"effb0_strong_aug\": {\n        \"model_type\": \"efficientnet_b0\",\n        \"pretrained\": True,\n        \"train_transform\": strong_train_tfms,\n        \"label_smoothing\": LABEL_SMOOTHING,\n        \"dropout\": DROPOUT,\n        \"fine_lr\": LR_FINE,\n        \"description\": \"Có pretrained nhưng augmentation mạnh để kiểm tra biến đổi quá tay có làm giảm hiệu năng không.\",\n    },\n\n    \"effb0_low_lr\": {\n        \"model_type\": \"efficientnet_b0\",\n        \"pretrained\": True,\n        \"train_transform\": improved_train_tfms,\n        \"label_smoothing\": LABEL_SMOOTHING,\n        \"dropout\": DROPOUT,\n        \"fine_lr\": 3e-5,\n        \"description\": \"Có pretrained nhưng learning rate thấp hơn, kiểm tra mô hình học chậm ra sao.\",\n    },\n\n    \"effb0_high_lr\": {\n        \"model_type\": \"efficientnet_b0\",\n        \"pretrained\": True,\n        \"train_transform\": improved_train_tfms,\n        \"label_smoothing\": LABEL_SMOOTHING,\n        \"dropout\": DROPOUT,\n        \"fine_lr\": 3e-4,\n        \"description\": \"Có pretrained nhưng learning rate cao hơn, kiểm tra nguy cơ dao động hoặc overfit.\",\n    },\n\n    \"effb0_no_dropout\": {\n        \"model_type\": \"efficientnet_b0\",\n        \"pretrained\": True,\n        \"train_transform\": improved_train_tfms,\n        \"label_smoothing\": LABEL_SMOOTHING,\n        \"dropout\": 0.0,\n        \"fine_lr\": LR_FINE,\n        \"description\": \"Bỏ dropout để xem classifier có học thuộc mạnh hơn không.\",\n    },\n\n    \"effb0_high_dropout\": {\n        \"model_type\": \"efficientnet_b0\",\n        \"pretrained\": True,\n        \"train_transform\": improved_train_tfms,\n        \"label_smoothing\": LABEL_SMOOTHING,\n        \"dropout\": 0.50,\n        \"fine_lr\": LR_FINE,\n        \"description\": \"Tăng dropout để xem regularization mạnh có làm mô hình học chậm hơn không.\",\n    },\n\n    \"effb0_no_label_smoothing\": {\n        \"model_type\": \"efficientnet_b0\",\n        \"pretrained\": True,\n        \"train_transform\": improved_train_tfms,\n        \"label_smoothing\": 0.0,\n        \"dropout\": DROPOUT,\n        \"fine_lr\": LR_FINE,\n        \"description\": \"Bỏ label smoothing để xem mô hình tự tin quá mức và overfit thế nào.\",\n    },\n}\n\nSELECTED_EXPERIMENTS = [\n    \"effb0_pretrained_finetune\",\n    \"effb0_no_pretrain_scratch\",\n    \"effb0_head_only_frozen\",\n    \"effb0_head_warmup\",\n    \"effb0_no_aug\",\n    \"effb0_strong_aug\",\n    \"effb0_low_lr\",\n    \"effb0_high_lr\",\n    \"effb0_no_dropout\",\n    \"effb0_high_dropout\",\n    \"effb0_no_label_smoothing\",\n]\n\nprint(\"Thí nghiệm sẽ chạy:\", SELECTED_EXPERIMENTS)\nprint(\"Epoch config:\", json.dumps({k: EXPERIMENT_EPOCHS[k] for k in SELECTED_EXPERIMENTS}, indent=2, ensure_ascii=False))\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:10:40.290564Z","iopub.execute_input":"2026-06-18T16:10:40.290920Z","iopub.status.idle":"2026-06-18T16:10:40.307731Z","shell.execute_reply.started":"2026-06-18T16:10:40.290897Z","shell.execute_reply":"2026-06-18T16:10:40.306922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_results = []\nprediction_cache = {}\n\nfor exp_name in SELECTED_EXPERIMENTS:\n    clean_memory(label=f\"đầu loop experiment {exp_name}\", verbose=False, kill_workers=True)\n    result, y_true, y_pred, y_prob = run_experiment(exp_name, EXPERIMENTS[exp_name])\n    all_results.append(result)\n    prediction_cache[exp_name] = {\"y_true\": y_true, \"y_pred\": y_pred, \"y_prob\": y_prob}\n    clean_memory(label=f\"cuối loop experiment {exp_name}\", verbose=False, kill_workers=True)\n\nsummary_df = pd.DataFrame([\n    {\n        \"experiment\": r[\"experiment\"],\n        \"description\": r[\"description\"],\n        \"best_val_loss\": r[\"best_val_loss\"],\n        \"best_val_acc\": r[\"best_val_acc\"],\n        \"best_val_macro_f1\": r[\"best_val_macro_f1\"],\n        \"best_checkpoint\": r[\"best_checkpoint\"],\n    } for r in all_results\n]).sort_values(\"best_val_macro_f1\", ascending=False)\n\nsummary_path = OUTPUT_DIR / \"experiment_summary_effb0_ablation_playground.csv\"\nsummary_df.to_csv(summary_path, index=False)\ndisplay(summary_df)\nprint(\"Đã lưu bảng kết quả tại:\", summary_path)\n\nif len(summary_df) >= 2:\n    print(\"So sánh nhanh theo Macro F1 validation:\")\n    display(summary_df[[\"experiment\", \"best_val_acc\", \"best_val_macro_f1\", \"best_checkpoint\"]])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T16:10:40.308631Z","iopub.execute_input":"2026-06-18T16:10:40.308988Z","iopub.status.idle":"2026-06-18T20:04:47.824620Z","shell.execute_reply.started":"2026-06-18T16:10:40.308958Z","shell.execute_reply":"2026-06-18T20:04:47.823517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"EXPORT_DIR = OUTPUT_DIR / \"effb0_export\"\nEXPORT_DIR.mkdir(parents=True, exist_ok=True)\n\nfor pattern in [\"*.pth\", \"*.pt\", \"*.csv\", \"*.png\", \"*.json\"]:\n    for f in OUTPUT_DIR.rglob(pattern):\n        if f.is_file():\n            try:\n                shutil.copy2(f, EXPORT_DIR / f.name)\n            except Exception as e:\n                print(\"Không copy được\", f, repr(e))\n\nzip_path = shutil.make_archive(str(OUTPUT_DIR / \"effb0_ablation_export\"), \"zip\", EXPORT_DIR)\nprint(\"Đã export model và kết quả:\", zip_path)\n\ntry:\n    from IPython.display import FileLink\n    display(FileLink(zip_path))\nexcept Exception:\n    pass\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T20:04:47.826267Z","iopub.execute_input":"2026-06-18T20:04:47.826530Z","iopub.status.idle":"2026-06-18T20:04:57.176685Z","shell.execute_reply.started":"2026-06-18T20:04:47.826502Z","shell.execute_reply":"2026-06-18T20:04:57.176144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PLOT_DIR = OUTPUT_DIR / \"plots\"\nPER_EXP_DIR = PLOT_DIR / \"per_experiment\"\nPLOT_DIR.mkdir(parents=True, exist_ok=True)\nPER_EXP_DIR.mkdir(parents=True, exist_ok=True)\n\nplot_catalog = []\nper_class_metric_tables = []\ntop_confusion_tables = []\n\ndef _save_line_plot(x, series_dict, title, xlabel, ylabel, save_path):\n    \"\"\"Vẽ 1 biểu đồ đường và lưu ra file PNG.\"\"\"\n    plt.figure(figsize=(8, 5))\n    for label, values in series_dict.items():\n        plt.plot(x, values, marker=\"o\", label=label)\n    plt.title(title)\n    plt.xlabel(xlabel)\n    plt.ylabel(ylabel)\n    plt.grid(True, alpha=0.3)\n    plt.legend()\n    plt.tight_layout()\n    plt.savefig(save_path, dpi=180, bbox_inches=\"tight\")\n    plt.close()\n    return save_path\n\ndef _save_bar_plot(labels, values, title, xlabel, ylabel, save_path, rotation=30):\n    \"\"\"Vẽ 1 biểu đồ cột và lưu ra file PNG.\"\"\"\n    plt.figure(figsize=(9, 5))\n    plt.bar(labels, values)\n    plt.title(title)\n    plt.xlabel(xlabel)\n    plt.ylabel(ylabel)\n    plt.xticks(rotation=rotation, ha=\"right\")\n    plt.grid(axis=\"y\", alpha=0.3)\n    plt.tight_layout()\n    plt.savefig(save_path, dpi=180, bbox_inches=\"tight\")\n    plt.close()\n    return save_path\n\ndef _save_confusion_matrix_plot(cm, labels, title, save_path, normalize=False):\n    \"\"\"Vẽ confusion matrix an toàn bằng matplotlib thuần.\n\n    Không dùng seaborn để tránh lỗi format `d`/float giữa count matrix và normalized matrix.\n    - normalize=False: ma trận count, hiển thị số nguyên.\n    - normalize=True: ma trận chuẩn hóa theo hàng, hiển thị 2 chữ số thập phân.\n    \"\"\"\n    cm = np.asarray(cm)\n\n    if normalize:\n        cm_float = cm.astype(float)\n        row_sum = cm_float.sum(axis=1, keepdims=True)\n        matrix = np.divide(\n            cm_float,\n            row_sum,\n            out=np.zeros_like(cm_float, dtype=float),\n            where=row_sum != 0,\n        )\n        text_func = lambda v: f\"{float(v):.2f}\"\n    else:\n        matrix = cm.astype(np.int64)\n        text_func = lambda v: f\"{int(v)}\"\n\n    fig, ax = plt.subplots(figsize=(10, 8))\n    im = ax.imshow(matrix, cmap=\"Blues\")\n    fig.colorbar(im, ax=ax, fraction=0.046, pad=0.04)\n\n    ax.set_xticks(np.arange(len(labels)))\n    ax.set_yticks(np.arange(len(labels)))\n    ax.set_xticklabels(labels, rotation=45, ha=\"right\")\n    ax.set_yticklabels(labels)\n\n    max_val = float(np.nanmax(matrix)) if matrix.size else 0.0\n    threshold = max_val / 2.0 if max_val > 0 else 0.0\n    for i in range(matrix.shape[0]):\n        for j in range(matrix.shape[1]):\n            value = matrix[i, j]\n            color = \"white\" if float(value) > threshold else \"black\"\n            ax.text(j, i, text_func(value), ha=\"center\", va=\"center\", color=color, fontsize=8)\n\n    ax.set_title(title)\n    ax.set_xlabel(\"Predicted label\")\n    ax.set_ylabel(\"True label\")\n    fig.tight_layout()\n    fig.savefig(save_path, dpi=180, bbox_inches=\"tight\")\n    plt.close(fig)\n    return save_path\n\ndef _classification_report_to_df(y_true, y_pred, target_names):\n    \"\"\"Chuyển classification report sang DataFrame để lưu CSV và vẽ per-class metrics.\"\"\"\n    report_dict = classification_report(\n        y_true,\n        y_pred,\n        labels=list(range(NUM_CLASSES)),\n        target_names=target_names,\n        output_dict=True,\n        zero_division=0,\n    )\n    report_df = pd.DataFrame(report_dict).T.reset_index().rename(columns={\"index\": \"class\"})\n    return report_df\n\ndef _extract_top_confusions(cm, top_k=10):\n    \"\"\"Lấy các cặp nhầm nhiều nhất từ confusion matrix.\"\"\"\n    rows = []\n    for true_idx in range(NUM_CLASSES):\n        for pred_idx in range(NUM_CLASSES):\n            if true_idx == pred_idx:\n                continue\n            count = int(cm[true_idx, pred_idx])\n            if count > 0:\n                rows.append({\n                    \"true_class\": idx_to_class[true_idx],\n                    \"pred_class\": idx_to_class[pred_idx],\n                    \"count\": count,\n                    \"true_description\": class_description[idx_to_class[true_idx]],\n                    \"pred_description\": class_description[idx_to_class[pred_idx]],\n                })\n    return pd.DataFrame(rows).sort_values(\"count\", ascending=False).head(top_k) if rows else pd.DataFrame(rows)\n\nif \"all_results\" not in globals() or len(all_results) == 0:\n    print(\"Chưa có kết quả thí nghiệm. Hãy chạy cell train trước.\")\nelse:\n    target_names = [f\"{c}: {class_description[c]}\" for c in class_names]\n\n    print(\"=\" * 90)\n    print(\"XUẤT BÁO CÁO RIÊNG CHO TỪNG PHƯƠNG ÁN EFFICIENTNET-B0\")\n    print(\"=\" * 90)\n\n    for result in all_results:\n        exp_name = result[\"experiment\"]\n        exp_dir = PER_EXP_DIR / exp_name\n        exp_dir.mkdir(parents=True, exist_ok=True)\n\n        hist = pd.DataFrame(result.get(\"history\", []))\n        if len(hist) == 0:\n            print(f\"[{exp_name}] Không có history, bỏ qua plot.\")\n            continue\n\n        hist[\"global_epoch\"] = np.arange(1, len(hist) + 1)\n        history_path = exp_dir / f\"{exp_name}_history.csv\"\n        hist.to_csv(history_path, index=False)\n        plot_catalog.append({\"experiment\": exp_name, \"type\": \"history_csv\", \"path\": str(history_path)})\n\n        print(f\"\\n[{exp_name}]\")\n        print(\"Mô tả:\", result.get(\"description\", \"\"))\n        print(f\"Best Val Acc={result['best_val_acc']:.4f} | Best Macro F1={result['best_val_macro_f1']:.4f}\")\n\n        acc_path = exp_dir / f\"{exp_name}_01_accuracy_curve.png\"\n        _save_line_plot(\n            hist[\"global_epoch\"],\n            {\"train_acc\": hist[\"train_acc\"], \"val_acc\": hist[\"val_acc\"]},\n            f\"Accuracy Curve - {exp_name}\",\n            \"Epoch\",\n            \"Accuracy\",\n            acc_path,\n        )\n        plot_catalog.append({\"experiment\": exp_name, \"type\": \"accuracy_curve\", \"path\": str(acc_path)})\n\n        loss_path = exp_dir / f\"{exp_name}_02_loss_curve.png\"\n        _save_line_plot(\n            hist[\"global_epoch\"],\n            {\"train_loss\": hist[\"train_loss\"], \"val_loss\": hist[\"val_loss\"]},\n            f\"Loss Curve - {exp_name}\",\n            \"Epoch\",\n            \"Loss\",\n            loss_path,\n        )\n        plot_catalog.append({\"experiment\": exp_name, \"type\": \"loss_curve\", \"path\": str(loss_path)})\n\n        f1_path = exp_dir / f\"{exp_name}_03_macro_f1_curve.png\"\n        _save_line_plot(\n            hist[\"global_epoch\"],\n            {\"train_macro_f1\": hist[\"train_macro_f1\"], \"val_macro_f1\": hist[\"val_macro_f1\"]},\n            f\"Macro F1 Curve - {exp_name}\",\n            \"Epoch\",\n            \"Macro F1\",\n            f1_path,\n        )\n        plot_catalog.append({\"experiment\": exp_name, \"type\": \"macro_f1_curve\", \"path\": str(f1_path)})\n\n        lr_path = exp_dir / f\"{exp_name}_04_learning_rate_curve.png\"\n        lr_col = \"lr_after\" if \"lr_after\" in hist.columns else \"lr_before\"\n        _save_line_plot(\n            hist[\"global_epoch\"],\n            {\"learning_rate\": hist[lr_col]},\n            f\"Learning Rate - {exp_name}\",\n            \"Epoch\",\n            \"LR\",\n            lr_path,\n        )\n        plot_catalog.append({\"experiment\": exp_name, \"type\": \"learning_rate_curve\", \"path\": str(lr_path)})\n\n        gap_path = exp_dir / f\"{exp_name}_05_train_val_gap_curve.png\"\n        gap_series = {}\n        if \"acc_gap\" in hist.columns:\n            gap_series[\"acc_gap\"] = hist[\"acc_gap\"]\n        if \"f1_gap\" in hist.columns:\n            gap_series[\"f1_gap\"] = hist[\"f1_gap\"]\n        if gap_series:\n            _save_line_plot(\n                hist[\"global_epoch\"],\n                gap_series,\n                f\"Train-Val Gap - {exp_name}\",\n                \"Epoch\",\n                \"Gap\",\n                gap_path,\n            )\n            plot_catalog.append({\"experiment\": exp_name, \"type\": \"train_val_gap_curve\", \"path\": str(gap_path)})\n\n        pred_pack = prediction_cache.get(exp_name)\n        if pred_pack is not None:\n            y_true_exp = pred_pack[\"y_true\"]\n            y_pred_exp = pred_pack[\"y_pred\"]\n\n            report_df = _classification_report_to_df(y_true_exp, y_pred_exp, target_names)\n            report_csv_path = exp_dir / f\"{exp_name}_classification_report.csv\"\n            report_txt_path = exp_dir / f\"{exp_name}_classification_report.txt\"\n            report_df.to_csv(report_csv_path, index=False)\n            with open(report_txt_path, \"w\", encoding=\"utf-8\") as f:\n                f.write(classification_report(y_true_exp, y_pred_exp, labels=list(range(NUM_CLASSES)), target_names=target_names, zero_division=0))\n            plot_catalog.append({\"experiment\": exp_name, \"type\": \"classification_report_csv\", \"path\": str(report_csv_path)})\n            plot_catalog.append({\"experiment\": exp_name, \"type\": \"classification_report_txt\", \"path\": str(report_txt_path)})\n\n            class_rows = report_df[report_df[\"class\"].astype(str).str.startswith(\"c\")].copy()\n            class_rows[\"short_class\"] = class_rows[\"class\"].astype(str).str.split(\":\").str[0]\n            class_rows[\"experiment\"] = exp_name\n            per_class_metric_tables.append(class_rows)\n\n            per_class_f1_path = exp_dir / f\"{exp_name}_06_per_class_f1.png\"\n            _save_bar_plot(\n                class_rows[\"short_class\"],\n                class_rows[\"f1-score\"],\n                f\"Per-class F1 - {exp_name}\",\n                \"Class\",\n                \"F1-score\",\n                per_class_f1_path,\n                rotation=0,\n            )\n            plot_catalog.append({\"experiment\": exp_name, \"type\": \"per_class_f1_bar\", \"path\": str(per_class_f1_path)})\n\n            cm = confusion_matrix(y_true_exp, y_pred_exp, labels=list(range(NUM_CLASSES)))\n            cm_raw_path = exp_dir / f\"{exp_name}_07_confusion_matrix_count.png\"\n            _save_confusion_matrix_plot(\n                cm,\n                class_names,\n                f\"Confusion Matrix Count - {exp_name}\",\n                cm_raw_path,\n                normalize=False,\n            )\n            plot_catalog.append({\"experiment\": exp_name, \"type\": \"confusion_matrix_count\", \"path\": str(cm_raw_path)})\n\n            cm_norm_path = exp_dir / f\"{exp_name}_08_confusion_matrix_normalized.png\"\n            _save_confusion_matrix_plot(\n                cm,\n                class_names,\n                f\"Confusion Matrix Normalized - {exp_name}\",\n                cm_norm_path,\n                normalize=True,\n            )\n            plot_catalog.append({\"experiment\": exp_name, \"type\": \"confusion_matrix_normalized\", \"path\": str(cm_norm_path)})\n\n            top_confusions = _extract_top_confusions(cm, top_k=10)\n            if len(top_confusions) > 0:\n                top_confusions[\"experiment\"] = exp_name\n                top_confusion_tables.append(top_confusions)\n                top_conf_path = exp_dir / f\"{exp_name}_top_confusions.csv\"\n                top_confusions.to_csv(top_conf_path, index=False)\n                plot_catalog.append({\"experiment\": exp_name, \"type\": \"top_confusions_csv\", \"path\": str(top_conf_path)})\n\n                top_conf_bar_path = exp_dir / f\"{exp_name}_09_top_confusions_bar.png\"\n                top_labels = [f\"{r.true_class}->{r.pred_class}\" for r in top_confusions.itertuples()]\n                _save_bar_plot(\n                    top_labels,\n                    top_confusions[\"count\"],\n                    f\"Top Confusions - {exp_name}\",\n                    \"True → Predicted\",\n                    \"Count\",\n                    top_conf_bar_path,\n                    rotation=45,\n                )\n                plot_catalog.append({\"experiment\": exp_name, \"type\": \"top_confusions_bar\", \"path\": str(top_conf_bar_path)})\n\n        print(\"Đã xuất hình/report vào:\", exp_dir)\n\n    plot_catalog_df = pd.DataFrame(plot_catalog)\n    plot_catalog_path = OUTPUT_DIR / \"plot_catalog_per_experiment.csv\"\n    plot_catalog_df.to_csv(plot_catalog_path, index=False)\n\n    if per_class_metric_tables:\n        per_class_metrics_all = pd.concat(per_class_metric_tables, ignore_index=True)\n        per_class_metrics_path = OUTPUT_DIR / \"per_class_metrics_all_experiments.csv\"\n        per_class_metrics_all.to_csv(per_class_metrics_path, index=False)\n        print(\"Đã lưu per-class metrics tổng:\", per_class_metrics_path)\n\n    if top_confusion_tables:\n        top_confusions_all = pd.concat(top_confusion_tables, ignore_index=True)\n        top_confusions_path = OUTPUT_DIR / \"top_confusions_all_experiments.csv\"\n        top_confusions_all.to_csv(top_confusions_path, index=False)\n        print(\"Đã lưu top confusions tổng:\", top_confusions_path)\n\n    print(\"\\nCatalog toàn bộ hình/report đã xuất:\")\n    display(plot_catalog_df)\n    print(\"Đã lưu catalog:\", plot_catalog_path)\n\nif bool(False):\n    clean_memory(label=\"sau xuất biểu đồ từng thí nghiệm\", verbose=False, kill_workers=False, deep=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T20:04:57.177862Z","iopub.execute_input":"2026-06-18T20:04:57.178215Z","iopub.status.idle":"2026-06-18T20:05:29.234548Z","shell.execute_reply.started":"2026-06-18T20:04:57.178191Z","shell.execute_reply":"2026-06-18T20:05:29.233639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"COMPARE_DIR = OUTPUT_DIR / \"plots\" / \"comparison\"\nCOMPARE_DIR.mkdir(parents=True, exist_ok=True)\n\nif \"summary_df\" not in globals() or len(summary_df) == 0:\n    print(\"Chưa có summary_df. Hãy chạy thí nghiệm trước.\")\nelse:\n    compare_df = summary_df.copy().sort_values(\"best_val_macro_f1\", ascending=False)\n\n    rows = []\n    for result in all_results:\n        hist = pd.DataFrame(result.get(\"history\", []))\n        if len(hist) == 0:\n            continue\n        best_idx = hist[\"val_macro_f1\"].idxmax()\n        best_row = hist.loc[best_idx].to_dict()\n        rows.append({\n            \"experiment\": result[\"experiment\"],\n            \"best_epoch_from_history\": int(best_row.get(\"epoch_in_phase\", 0)),\n            \"best_phase\": best_row.get(\"phase\", \"\"),\n            \"best_train_acc_at_best_f1\": best_row.get(\"train_acc\", np.nan),\n            \"best_val_acc_from_history\": best_row.get(\"val_acc\", np.nan),\n            \"best_train_f1_at_best_f1\": best_row.get(\"train_macro_f1\", np.nan),\n            \"best_val_f1_from_history\": best_row.get(\"val_macro_f1\", np.nan),\n            \"acc_gap_at_best_f1\": best_row.get(\"acc_gap\", np.nan),\n            \"f1_gap_at_best_f1\": best_row.get(\"f1_gap\", np.nan),\n            \"loss_ratio_at_best_f1\": best_row.get(\"loss_ratio\", np.nan),\n            \"total_train_time_min\": hist.get(\"elapsed_sec\", pd.Series(dtype=float)).sum() / 60.0,\n            \"last_lr\": best_row.get(\"lr_after\", np.nan),\n        })\n\n    diagnostics_df = pd.DataFrame(rows)\n    compare_full_df = compare_df.merge(diagnostics_df, on=\"experiment\", how=\"left\")\n    compare_full_path = OUTPUT_DIR / \"comparison_summary_full.csv\"\n    compare_full_df.to_csv(compare_full_path, index=False)\n\n    display_cols = [\n        \"experiment\", \"best_val_acc\", \"best_val_macro_f1\",\n        \"best_phase\", \"best_epoch_from_history\",\n        \"acc_gap_at_best_f1\", \"f1_gap_at_best_f1\",\n        \"total_train_time_min\", \"best_checkpoint\",\n    ]\n    display(compare_full_df[[c for c in display_cols if c in compare_full_df.columns]])\n    print(\"Đã lưu bảng so sánh đầy đủ:\", compare_full_path)\n\n    x_labels = compare_full_df[\"experiment\"].tolist()\n\n    acc_cmp_path = COMPARE_DIR / \"comparison_01_best_val_accuracy.png\"\n    plt.figure(figsize=(10, 5))\n    plt.bar(x_labels, compare_full_df[\"best_val_acc\"])\n    plt.title(\"Best Validation Accuracy - EfficientNet-B0 Ablation\")\n    plt.xlabel(\"Experiment\")\n    plt.ylabel(\"Accuracy\")\n    plt.ylim(0, 1)\n    plt.xticks(rotation=30, ha=\"right\")\n    plt.grid(axis=\"y\", alpha=0.3)\n    plt.tight_layout()\n    plt.savefig(acc_cmp_path, dpi=180, bbox_inches=\"tight\")\n    plt.show()\n    print(\"Đã lưu:\", acc_cmp_path)\n\n    f1_cmp_path = COMPARE_DIR / \"comparison_02_best_macro_f1.png\"\n    plt.figure(figsize=(10, 5))\n    plt.bar(x_labels, compare_full_df[\"best_val_macro_f1\"])\n    plt.title(\"Best Validation Macro F1 - EfficientNet-B0 Ablation\")\n    plt.xlabel(\"Experiment\")\n    plt.ylabel(\"Macro F1\")\n    plt.ylim(0, 1)\n    plt.xticks(rotation=30, ha=\"right\")\n    plt.grid(axis=\"y\", alpha=0.3)\n    plt.tight_layout()\n    plt.savefig(f1_cmp_path, dpi=180, bbox_inches=\"tight\")\n    plt.show()\n    print(\"Đã lưu:\", f1_cmp_path)\n\n    if \"f1_gap_at_best_f1\" in compare_full_df.columns:\n        gap_cmp_path = COMPARE_DIR / \"comparison_03_f1_gap_at_best_epoch.png\"\n        plt.figure(figsize=(10, 5))\n        plt.bar(x_labels, compare_full_df[\"f1_gap_at_best_f1\"])\n        plt.title(\"Train-Val Macro F1 Gap at Best Epoch\")\n        plt.xlabel(\"Experiment\")\n        plt.ylabel(\"F1 Gap\")\n        plt.xticks(rotation=30, ha=\"right\")\n        plt.grid(axis=\"y\", alpha=0.3)\n        plt.tight_layout()\n        plt.savefig(gap_cmp_path, dpi=180, bbox_inches=\"tight\")\n        plt.show()\n        print(\"Đã lưu:\", gap_cmp_path)\n\n    if \"total_train_time_min\" in compare_full_df.columns:\n        time_cmp_path = COMPARE_DIR / \"comparison_04_training_time_minutes.png\"\n        plt.figure(figsize=(10, 5))\n        plt.bar(x_labels, compare_full_df[\"total_train_time_min\"])\n        plt.title(\"Training Time - EfficientNet-B0 Ablation\")\n        plt.xlabel(\"Experiment\")\n        plt.ylabel(\"Minutes\")\n        plt.xticks(rotation=30, ha=\"right\")\n        plt.grid(axis=\"y\", alpha=0.3)\n        plt.tight_layout()\n        plt.savefig(time_cmp_path, dpi=180, bbox_inches=\"tight\")\n        plt.show()\n        print(\"Đã lưu:\", time_cmp_path)\n\n    best_experiment_name = compare_full_df.iloc[0][\"experiment\"]\n    print(\"\\nMô hình tốt nhất theo validation Macro F1:\", best_experiment_name)\n    print(\"Checkpoint:\", compare_full_df.iloc[0][\"best_checkpoint\"])\n\nif bool(False):\n    clean_memory(label=\"sau xuất biểu đồ so sánh\", verbose=False, kill_workers=False, deep=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T20:05:29.235837Z","iopub.execute_input":"2026-06-18T20:05:29.236240Z","iopub.status.idle":"2026-06-18T20:05:30.898531Z","shell.execute_reply.started":"2026-06-18T20:05:29.236215Z","shell.execute_reply":"2026-06-18T20:05:30.897752Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"COMPLETE_DIR = OUTPUT_DIR / \"complete_outputs\"\nPER_EXP_DIR = COMPLETE_DIR / \"per_experiment\"\nCOMPARE_DIR = COMPLETE_DIR / \"comparison\"\n\nCOMPLETE_DIR.mkdir(parents=True, exist_ok=True)\nPER_EXP_DIR.mkdir(parents=True, exist_ok=True)\nCOMPARE_DIR.mkdir(parents=True, exist_ok=True)\n\nmanifest_rows = []\n\ndef record_file(path, group, file_type, note=\"\"):\n    path = Path(path)\n    if path.exists():\n        manifest_rows.append({\n            \"group\": group,\n            \"file_type\": file_type,\n            \"path\": str(path),\n            \"note\": note,\n        })\n\ndef pick_col(df, candidates):\n    for c in candidates:\n        if c in df.columns:\n            return c\n    return None\n\ndef safe_divide(a, b):\n    return np.divide(\n        a,\n        np.maximum(b, 1e-8),\n        out=np.zeros_like(a, dtype=float),\n        where=np.maximum(b, 1e-8) != 0,\n    )\n\ndef plot_curve(df, x_col, y_cols, labels, title, ylabel, save_path, ylim=None):\n    plt.figure(figsize=(9, 5))\n\n    plotted = False\n    for col, label in zip(y_cols, labels):\n        if col is not None and col in df.columns:\n            plt.plot(df[x_col], df[col], marker=\"o\", label=label)\n            plotted = True\n\n    if not plotted:\n        plt.text(0.5, 0.5, \"No valid columns found\", ha=\"center\", va=\"center\")\n\n    plt.title(title)\n    plt.xlabel(\"Epoch\")\n    plt.ylabel(ylabel)\n\n    if ylim is not None:\n        plt.ylim(*ylim)\n\n    plt.grid(True, alpha=0.3)\n\n    if plotted:\n        plt.legend()\n\n    plt.tight_layout()\n    plt.savefig(save_path, dpi=200, bbox_inches=\"tight\")\n    plt.close()\n\n    record_file(save_path, \"per_experiment\", \"plot\", title)\n\ndef plot_bar(labels, values, title, ylabel, save_path, rotation=30, ylim=None):\n    plt.figure(figsize=(10, 5))\n    plt.bar(labels, values)\n    plt.title(title)\n    plt.xlabel(\"Class\")\n    plt.ylabel(ylabel)\n\n    if ylim is not None:\n        plt.ylim(*ylim)\n\n    plt.xticks(rotation=rotation, ha=\"right\")\n    plt.grid(axis=\"y\", alpha=0.3)\n    plt.tight_layout()\n    plt.savefig(save_path, dpi=200, bbox_inches=\"tight\")\n    plt.close()\n\n    record_file(save_path, \"per_experiment\", \"plot\", title)\n\ndef plot_confusion_matrix_safe(cm, labels, title, save_path, normalize=False):\n    cm = np.asarray(cm)\n\n    if normalize:\n        cm_float = cm.astype(float)\n        row_sum = cm_float.sum(axis=1, keepdims=True)\n        matrix = np.divide(\n            cm_float,\n            row_sum,\n            out=np.zeros_like(cm_float, dtype=float),\n            where=row_sum != 0,\n        )\n        text_fmt = \".2f\"\n    else:\n        matrix = cm.astype(np.int64)\n        text_fmt = \"d\"\n\n    fig, ax = plt.subplots(figsize=(9, 8))\n    im = ax.imshow(matrix, interpolation=\"nearest\", cmap=\"Blues\")\n    fig.colorbar(im, ax=ax)\n\n    ax.set_title(title)\n    ax.set_xlabel(\"Predicted label\")\n    ax.set_ylabel(\"True label\")\n\n    ax.set_xticks(np.arange(len(labels)))\n    ax.set_yticks(np.arange(len(labels)))\n    ax.set_xticklabels(labels, rotation=45, ha=\"right\")\n    ax.set_yticklabels(labels)\n\n    max_value = matrix.max() if matrix.size > 0 else 0\n    threshold = max_value / 2 if max_value > 0 else 0\n\n    for i in range(matrix.shape[0]):\n        for j in range(matrix.shape[1]):\n            value = format(matrix[i, j], text_fmt)\n            color = \"white\" if matrix[i, j] > threshold else \"black\"\n            ax.text(j, i, value, ha=\"center\", va=\"center\", color=color, fontsize=8)\n\n    plt.tight_layout()\n    plt.savefig(save_path, dpi=200, bbox_inches=\"tight\")\n    plt.close(fig)\n\n    record_file(save_path, \"per_experiment\", \"plot\", title)\n\ndef plot_experiment_dashboard(hist, exp_name, save_path):\n    train_acc_col = pick_col(hist, [\"train_acc\", \"train_accuracy\"])\n    val_acc_col = pick_col(hist, [\"val_acc\", \"val_accuracy\"])\n\n    train_f1_col = pick_col(hist, [\"train_macro_f1\", \"train_f1\", \"train_weighted_f1\"])\n    val_f1_col = pick_col(hist, [\"val_macro_f1\", \"val_f1\", \"val_weighted_f1\"])\n\n    train_loss_col = pick_col(hist, [\"train_loss\", \"loss\"])\n    val_loss_col = pick_col(hist, [\"val_loss\", \"valid_loss\"])\n\n    lr_col = pick_col(hist, [\"lr_after\", \"lr_before\", \"lr\", \"learning_rate\"])\n\n    fig, axes = plt.subplots(2, 2, figsize=(14, 9))\n\n    ax = axes[0, 0]\n    if train_acc_col:\n        ax.plot(hist[\"global_epoch\"], hist[train_acc_col], marker=\"o\", label=\"Train Acc\")\n    if val_acc_col:\n        ax.plot(hist[\"global_epoch\"], hist[val_acc_col], marker=\"o\", label=\"Val Acc\")\n    ax.set_title(\"Accuracy\")\n    ax.set_xlabel(\"Epoch\")\n    ax.set_ylabel(\"Accuracy\")\n    ax.grid(True, alpha=0.3)\n    ax.legend()\n\n    ax = axes[0, 1]\n    if train_loss_col:\n        ax.plot(hist[\"global_epoch\"], hist[train_loss_col], marker=\"o\", label=\"Train Loss\")\n    if val_loss_col:\n        ax.plot(hist[\"global_epoch\"], hist[val_loss_col], marker=\"o\", label=\"Val Loss\")\n    ax.set_title(\"Loss\")\n    ax.set_xlabel(\"Epoch\")\n    ax.set_ylabel(\"Loss\")\n    ax.grid(True, alpha=0.3)\n    ax.legend()\n\n    ax = axes[1, 0]\n    if train_f1_col:\n        ax.plot(hist[\"global_epoch\"], hist[train_f1_col], marker=\"o\", label=\"Train F1\")\n    if val_f1_col:\n        ax.plot(hist[\"global_epoch\"], hist[val_f1_col], marker=\"o\", label=\"Val F1\")\n    ax.set_title(\"F1-score\")\n    ax.set_xlabel(\"Epoch\")\n    ax.set_ylabel(\"F1-score\")\n    ax.grid(True, alpha=0.3)\n    ax.legend()\n\n    ax = axes[1, 1]\n    if lr_col:\n        ax.plot(hist[\"global_epoch\"], hist[lr_col], marker=\"o\", label=\"Learning Rate\")\n    ax.set_title(\"Learning Rate\")\n    ax.set_xlabel(\"Epoch\")\n    ax.set_ylabel(\"LR\")\n    ax.grid(True, alpha=0.3)\n    ax.legend()\n\n    fig.suptitle(f\"Training Dashboard - {exp_name}\", fontsize=14)\n    plt.tight_layout()\n    plt.savefig(save_path, dpi=200, bbox_inches=\"tight\")\n    plt.close(fig)\n\n    record_file(save_path, \"per_experiment\", \"plot\", \"Training dashboard\")\n\nif \"all_results\" not in globals() or \"prediction_cache\" not in globals():\n    print(\"Thiếu all_results hoặc prediction_cache. Hãy chạy train/validation trước.\")\nelse:\n    best_rows = []\n    all_history_rows = []\n\n    for result in all_results:\n        exp_name = result.get(\"experiment\", \"unknown_experiment\")\n        exp_dir = PER_EXP_DIR / exp_name\n        exp_dir.mkdir(parents=True, exist_ok=True)\n\n        print(f\"\\n=== Xuất đầy đủ cho phương án: {exp_name} ===\")\n\n        hist = pd.DataFrame(result.get(\"history\", []))\n        if len(hist) == 0:\n            print(f\"Bỏ qua {exp_name}: không có history.\")\n            continue\n\n        hist = hist.copy()\n        hist.insert(0, \"global_epoch\", np.arange(1, len(hist) + 1))\n\n        hist_path = exp_dir / f\"{exp_name}_00_epoch_history_full.csv\"\n        hist.to_csv(hist_path, index=False)\n        record_file(hist_path, exp_name, \"table\", \"Full epoch history\")\n        all_history_rows.append(hist.assign(experiment=exp_name))\n\n        train_acc_col = pick_col(hist, [\"train_acc\", \"train_accuracy\"])\n        val_acc_col = pick_col(hist, [\"val_acc\", \"val_accuracy\"])\n\n        train_f1_col = pick_col(hist, [\"train_macro_f1\", \"train_f1\", \"train_weighted_f1\"])\n        val_f1_col = pick_col(hist, [\"val_macro_f1\", \"val_f1\", \"val_weighted_f1\"])\n\n        train_loss_col = pick_col(hist, [\"train_loss\", \"loss\"])\n        val_loss_col = pick_col(hist, [\"val_loss\", \"valid_loss\"])\n\n        lr_col = pick_col(hist, [\"lr_after\", \"lr_before\", \"lr\", \"learning_rate\"])\n\n        if val_f1_col is not None:\n            best_idx = hist[val_f1_col].idxmax()\n            best_metric_name = val_f1_col\n        elif val_acc_col is not None:\n            best_idx = hist[val_acc_col].idxmax()\n            best_metric_name = val_acc_col\n        elif val_loss_col is not None:\n            best_idx = hist[val_loss_col].idxmin()\n            best_metric_name = val_loss_col\n        else:\n            best_idx = hist.index[-1]\n            best_metric_name = \"last_epoch\"\n\n        best_row = hist.loc[best_idx].to_dict()\n        best_row.update({\n            \"experiment\": exp_name,\n            \"description\": result.get(\"description\", \"\"),\n            \"best_metric_name\": best_metric_name,\n            \"best_checkpoint\": result.get(\"best_checkpoint\", \"\"),\n            \"best_val_acc_result\": result.get(\"best_val_acc\", np.nan),\n            \"best_val_macro_f1_result\": result.get(\"best_val_macro_f1\", np.nan),\n        })\n        best_rows.append(best_row)\n\n        best_path = exp_dir / f\"{exp_name}_00_best_epoch_detail.csv\"\n        pd.DataFrame([best_row]).to_csv(best_path, index=False)\n        record_file(best_path, exp_name, \"table\", \"Best epoch detail\")\n\n        plot_experiment_dashboard(\n            hist,\n            exp_name,\n            exp_dir / f\"{exp_name}_00_training_dashboard.png\",\n        )\n\n        plot_curve(\n            hist,\n            \"global_epoch\",\n            [train_acc_col, val_acc_col],\n            [\"Train Accuracy\", \"Validation Accuracy\"],\n            f\"Accuracy Curve - {exp_name}\",\n            \"Accuracy\",\n            exp_dir / f\"{exp_name}_01_accuracy_curve.png\",\n            ylim=(0, 1),\n        )\n\n        plot_curve(\n            hist,\n            \"global_epoch\",\n            [train_loss_col, val_loss_col],\n            [\"Train Loss\", \"Validation Loss\"],\n            f\"Loss Curve - {exp_name}\",\n            \"Loss\",\n            exp_dir / f\"{exp_name}_02_loss_curve.png\",\n        )\n\n        plot_curve(\n            hist,\n            \"global_epoch\",\n            [train_f1_col, val_f1_col],\n            [\"Train F1\", \"Validation F1\"],\n            f\"F1 Curve - {exp_name}\",\n            \"F1-score\",\n            exp_dir / f\"{exp_name}_03_f1_curve.png\",\n            ylim=(0, 1),\n        )\n\n        plot_curve(\n            hist,\n            \"global_epoch\",\n            [lr_col],\n            [\"Learning Rate\"],\n            f\"Learning Rate Curve - {exp_name}\",\n            \"Learning Rate\",\n            exp_dir / f\"{exp_name}_04_learning_rate_curve.png\",\n        )\n\n        gap_df = hist[[\"global_epoch\"]].copy()\n\n        if train_acc_col is not None and val_acc_col is not None:\n            gap_df[\"accuracy_gap\"] = hist[train_acc_col] - hist[val_acc_col]\n\n        if train_f1_col is not None and val_f1_col is not None:\n            gap_df[\"f1_gap\"] = hist[train_f1_col] - hist[val_f1_col]\n\n        gap_cols = [c for c in [\"accuracy_gap\", \"f1_gap\"] if c in gap_df.columns]\n\n        if gap_cols:\n            plot_curve(\n                gap_df,\n                \"global_epoch\",\n                gap_cols,\n                gap_cols,\n                f\"Train-Val Gap Curve - {exp_name}\",\n                \"Gap\",\n                exp_dir / f\"{exp_name}_05_train_val_gap_curve.png\",\n            )\n\n        if train_loss_col is not None and val_loss_col is not None:\n            ratio_df = hist[[\"global_epoch\"]].copy()\n            ratio_df[\"val_loss_div_train_loss\"] = hist[val_loss_col] / np.maximum(hist[train_loss_col], 1e-8)\n\n            plot_curve(\n                ratio_df,\n                \"global_epoch\",\n                [\"val_loss_div_train_loss\"],\n                [\"Val Loss / Train Loss\"],\n                f\"Loss Ratio Curve - {exp_name}\",\n                \"Ratio\",\n                exp_dir / f\"{exp_name}_06_loss_ratio_curve.png\",\n            )\n\n        pred_pack = prediction_cache.get(exp_name)\n\n        if pred_pack is None:\n            print(f\"{exp_name}: chưa có prediction_cache, bỏ qua confusion/report.\")\n            continue\n\n        y_true = np.asarray(pred_pack[\"y_true\"])\n        y_pred = np.asarray(pred_pack[\"y_pred\"])\n        y_prob = np.asarray(pred_pack.get(\"y_prob\", []))\n\n        pred_df = pd.DataFrame({\n            \"y_true\": y_true,\n            \"y_pred\": y_pred,\n            \"true_class\": [idx_to_class[int(i)] for i in y_true],\n            \"pred_class\": [idx_to_class[int(i)] for i in y_pred],\n        })\n\n        if y_prob.ndim == 2 and y_prob.shape[1] == NUM_CLASSES:\n            for i, c in enumerate(class_names):\n                pred_df[f\"prob_{c}\"] = y_prob[:, i]\n\n            pred_df[\"confidence\"] = y_prob.max(axis=1)\n            pred_df[\"correct\"] = pred_df[\"y_true\"] == pred_df[\"y_pred\"]\n\n            plt.figure(figsize=(8, 5))\n            plt.hist(pred_df[\"confidence\"], bins=20)\n            plt.title(f\"Prediction Confidence Histogram - {exp_name}\")\n            plt.xlabel(\"Confidence\")\n            plt.ylabel(\"Number of samples\")\n            plt.grid(axis=\"y\", alpha=0.3)\n            plt.tight_layout()\n            confidence_path = exp_dir / f\"{exp_name}_07_confidence_histogram.png\"\n            plt.savefig(confidence_path, dpi=200, bbox_inches=\"tight\")\n            plt.close()\n            record_file(confidence_path, exp_name, \"plot\", \"Confidence histogram\")\n\n            if pred_df[\"correct\"].nunique() > 1:\n                plt.figure(figsize=(8, 5))\n                plt.hist(pred_df[pred_df[\"correct\"] == True][\"confidence\"], bins=20, alpha=0.7, label=\"Correct\")\n                plt.hist(pred_df[pred_df[\"correct\"] == False][\"confidence\"], bins=20, alpha=0.7, label=\"Wrong\")\n                plt.title(f\"Confidence Distribution: Correct vs Wrong - {exp_name}\")\n                plt.xlabel(\"Confidence\")\n                plt.ylabel(\"Number of samples\")\n                plt.legend()\n                plt.grid(axis=\"y\", alpha=0.3)\n                plt.tight_layout()\n                conf_correct_path = exp_dir / f\"{exp_name}_08_confidence_correct_vs_wrong.png\"\n                plt.savefig(conf_correct_path, dpi=200, bbox_inches=\"tight\")\n                plt.close()\n                record_file(conf_correct_path, exp_name, \"plot\", \"Confidence correct vs wrong\")\n\n        pred_path = exp_dir / f\"{exp_name}_validation_predictions_full.csv\"\n        pred_df.to_csv(pred_path, index=False)\n        record_file(pred_path, exp_name, \"table\", \"Validation predictions\")\n\n        target_names = [f\"{c}: {class_description[c]}\" for c in class_names]\n\n        report_dict = classification_report(\n            y_true,\n            y_pred,\n            labels=list(range(NUM_CLASSES)),\n            target_names=target_names,\n            zero_division=0,\n            output_dict=True,\n        )\n\n        report_df = pd.DataFrame(report_dict).T.reset_index().rename(columns={\"index\": \"class\"})\n\n        report_csv = exp_dir / f\"{exp_name}_09_classification_report_full.csv\"\n        report_txt = exp_dir / f\"{exp_name}_09_classification_report_full.txt\"\n\n        report_df.to_csv(report_csv, index=False)\n\n        with open(report_txt, \"w\", encoding=\"utf-8\") as f:\n            f.write(\n                classification_report(\n                    y_true,\n                    y_pred,\n                    labels=list(range(NUM_CLASSES)),\n                    target_names=target_names,\n                    zero_division=0,\n                )\n            )\n\n        record_file(report_csv, exp_name, \"table\", \"Classification report CSV\")\n        record_file(report_txt, exp_name, \"text\", \"Classification report TXT\")\n\n        class_only = report_df[report_df[\"class\"].astype(str).str.startswith(\"c\")].copy()\n        class_only[\"short_class\"] = class_only[\"class\"].astype(str).str.split(\":\").str[0]\n\n        for metric in [\"precision\", \"recall\", \"f1-score\", \"support\"]:\n            if metric in class_only.columns:\n                metric_name = metric.replace(\"-\", \"_\")\n\n                ylim = (0, 1) if metric != \"support\" else None\n\n                plot_bar(\n                    class_only[\"short_class\"],\n                    class_only[metric],\n                    f\"Per-class {metric} - {exp_name}\",\n                    metric,\n                    exp_dir / f\"{exp_name}_10_per_class_{metric_name}.png\",\n                    rotation=0,\n                    ylim=ylim,\n                )\n\n        cm = confusion_matrix(y_true, y_pred, labels=list(range(NUM_CLASSES)))\n\n        cm_count_csv = exp_dir / f\"{exp_name}_11_confusion_matrix_count.csv\"\n        pd.DataFrame(cm, index=class_names, columns=class_names).to_csv(cm_count_csv)\n        record_file(cm_count_csv, exp_name, \"table\", \"Confusion matrix count CSV\")\n\n        plot_confusion_matrix_safe(\n            cm,\n            class_names,\n            f\"Confusion Matrix Count - {exp_name}\",\n            exp_dir / f\"{exp_name}_11_confusion_matrix_count.png\",\n            normalize=False,\n        )\n\n        cm_norm = cm.astype(float) / np.maximum(cm.sum(axis=1, keepdims=True), 1)\n\n        cm_norm_csv = exp_dir / f\"{exp_name}_12_confusion_matrix_normalized.csv\"\n        pd.DataFrame(cm_norm, index=class_names, columns=class_names).to_csv(cm_norm_csv)\n        record_file(cm_norm_csv, exp_name, \"table\", \"Confusion matrix normalized CSV\")\n\n        plot_confusion_matrix_safe(\n            cm_norm,\n            class_names,\n            f\"Confusion Matrix Normalized - {exp_name}\",\n            exp_dir / f\"{exp_name}_12_confusion_matrix_normalized.png\",\n            normalize=True,\n        )\n\n        pairs = []\n        for i, true_cls in enumerate(class_names):\n            for j, pred_cls in enumerate(class_names):\n                if i != j and cm[i, j] > 0:\n                    pairs.append({\n                        \"true_class\": true_cls,\n                        \"pred_class\": pred_cls,\n                        \"count\": int(cm[i, j]),\n                        \"rate_in_true_class\": float(cm_norm[i, j]),\n                    })\n\n        pairs_df = pd.DataFrame(\n            pairs,\n            columns=[\"true_class\", \"pred_class\", \"count\", \"rate_in_true_class\"],\n        )\n\n        if len(pairs_df) > 0:\n            pairs_df = pairs_df.sort_values([\"count\", \"rate_in_true_class\"], ascending=False)\n\n        pairs_csv = exp_dir / f\"{exp_name}_13_top_confusion_pairs.csv\"\n        pairs_df.to_csv(pairs_csv, index=False)\n        record_file(pairs_csv, exp_name, \"table\", \"Top confusion pairs CSV\")\n\n        if len(pairs_df) > 0:\n            top_plot_df = pairs_df.head(10).copy()\n            top_plot_df[\"pair\"] = top_plot_df[\"true_class\"] + \" → \" + top_plot_df[\"pred_class\"]\n\n            plt.figure(figsize=(9, 5))\n            plt.bar(top_plot_df[\"pair\"], top_plot_df[\"count\"])\n            plt.title(f\"Top Confusion Pairs - {exp_name}\")\n            plt.xlabel(\"True → Pred\")\n            plt.ylabel(\"Count\")\n            plt.xticks(rotation=30, ha=\"right\")\n            plt.grid(axis=\"y\", alpha=0.3)\n            plt.tight_layout()\n\n            pairs_png = exp_dir / f\"{exp_name}_13_top_confusion_pairs.png\"\n            plt.savefig(pairs_png, dpi=200, bbox_inches=\"tight\")\n            plt.close()\n            record_file(pairs_png, exp_name, \"plot\", \"Top confusion pairs chart\")\n\n    if all_history_rows:\n        all_hist_df = pd.concat(all_history_rows, ignore_index=True)\n        all_hist_path = COMPLETE_DIR / \"all_experiments_epoch_history_full.csv\"\n        all_hist_df.to_csv(all_hist_path, index=False)\n        record_file(all_hist_path, \"all\", \"table\", \"All experiments epoch history\")\n\n    if best_rows:\n        best_df = pd.DataFrame(best_rows)\n        best_path = COMPLETE_DIR / \"all_experiments_best_epoch_detail.csv\"\n        best_df.to_csv(best_path, index=False)\n        record_file(best_path, \"all\", \"table\", \"Best epoch detail for all experiments\")\n\n        display_cols = [\n            \"experiment\",\n            \"global_epoch\",\n            \"train_acc\",\n            \"val_acc\",\n            \"train_f1\",\n            \"val_f1\",\n            \"train_macro_f1\",\n            \"val_macro_f1\",\n            \"lr\",\n            \"lr_after\",\n            \"loss_ratio\",\n            \"best_checkpoint\",\n        ]\n        display_cols = [c for c in display_cols if c in best_df.columns]\n        display(best_df[display_cols])\n\nif \"summary_df\" in globals() and isinstance(summary_df, pd.DataFrame) and len(summary_df) > 0:\n    summary_plot = summary_df.copy()\n\n    acc_col = pick_col(summary_plot, [\"best_val_acc\", \"val_acc\"])\n    f1_col = pick_col(summary_plot, [\"best_val_macro_f1\", \"best_val_f1\", \"val_macro_f1\", \"val_f1\"])\n    loss_col = pick_col(summary_plot, [\"best_val_loss\", \"val_loss\"])\n\n    if f1_col is not None:\n        summary_plot = summary_plot.sort_values(f1_col, ascending=False)\n    elif acc_col is not None:\n        summary_plot = summary_plot.sort_values(acc_col, ascending=False)\n\n    summary_csv = COMPARE_DIR / \"comparison_00_summary_core.csv\"\n    summary_plot.to_csv(summary_csv, index=False)\n    record_file(summary_csv, \"comparison\", \"table\", \"Core comparison summary\")\n\n    labels = summary_plot[\"experiment\"].astype(str).tolist()\n    x = np.arange(len(labels))\n\n    if acc_col is not None and f1_col is not None:\n        plt.figure(figsize=(12, 5))\n        width = 0.35\n\n        plt.bar(x - width / 2, summary_plot[acc_col], width, label=\"Accuracy\")\n        plt.bar(x + width / 2, summary_plot[f1_col], width, label=\"F1-score\")\n\n        plt.title(\"Best Validation Accuracy and F1-score Comparison\")\n        plt.xlabel(\"Experiment\")\n        plt.ylabel(\"Score\")\n        plt.ylim(0, 1)\n        plt.xticks(x, labels, rotation=30, ha=\"right\")\n        plt.legend()\n        plt.grid(axis=\"y\", alpha=0.3)\n        plt.tight_layout()\n\n        comparison_score_path = COMPARE_DIR / \"comparison_01_accuracy_f1_grouped.png\"\n        plt.savefig(comparison_score_path, dpi=200, bbox_inches=\"tight\")\n        plt.close()\n        record_file(comparison_score_path, \"comparison\", \"plot\", \"Accuracy and F1 comparison\")\n\n    if loss_col is not None:\n        plt.figure(figsize=(12, 5))\n        plt.bar(labels, summary_plot[loss_col])\n        plt.title(\"Best Validation Loss Comparison\")\n        plt.xlabel(\"Experiment\")\n        plt.ylabel(\"Loss\")\n        plt.xticks(rotation=30, ha=\"right\")\n        plt.grid(axis=\"y\", alpha=0.3)\n        plt.tight_layout()\n\n        comparison_loss_path = COMPARE_DIR / \"comparison_02_best_val_loss.png\"\n        plt.savefig(comparison_loss_path, dpi=200, bbox_inches=\"tight\")\n        plt.close()\n        record_file(comparison_loss_path, \"comparison\", \"plot\", \"Best validation loss comparison\")\n\n    compare_display_cols = [\n        \"experiment\",\n        acc_col,\n        f1_col,\n        loss_col,\n        \"best_epoch\",\n        \"total_train_time_min\",\n    ]\n    compare_display_cols = [c for c in compare_display_cols if c is not None and c in summary_plot.columns]\n\n    if compare_display_cols:\n        display(summary_plot[compare_display_cols])\n\n    if \"compare_full_df\" in globals() and isinstance(compare_full_df, pd.DataFrame) and len(compare_full_df) > 0:\n        compare_full_path = COMPARE_DIR / \"comparison_03_summary_full_with_diagnostics.csv\"\n        compare_full_df.to_csv(compare_full_path, index=False)\n        record_file(compare_full_path, \"comparison\", \"table\", \"Full comparison with diagnostics\")\n\n        for col, title, ylabel, fname in [\n            (\"f1_gap_at_best_f1\", \"Train-Val F1 Gap at Best Epoch\", \"F1 Gap\", \"comparison_04_f1_gap.png\"),\n            (\"acc_gap_at_best_f1\", \"Train-Val Accuracy Gap at Best Epoch\", \"Accuracy Gap\", \"comparison_05_acc_gap.png\"),\n            (\"total_train_time_min\", \"Total Training Time\", \"Minutes\", \"comparison_06_training_time.png\"),\n        ]:\n            if col in compare_full_df.columns:\n                plot_df = compare_full_df.copy()\n                if \"best_val_macro_f1\" in plot_df.columns:\n                    plot_df = plot_df.sort_values(\"best_val_macro_f1\", ascending=False)\n\n                plt.figure(figsize=(12, 5))\n                plt.bar(plot_df[\"experiment\"].astype(str), plot_df[col])\n                plt.title(title)\n                plt.xlabel(\"Experiment\")\n                plt.ylabel(ylabel)\n                plt.xticks(rotation=30, ha=\"right\")\n                plt.grid(axis=\"y\", alpha=0.3)\n                plt.tight_layout()\n\n                p = COMPARE_DIR / fname\n                plt.savefig(p, dpi=200, bbox_inches=\"tight\")\n                plt.close()\n                record_file(p, \"comparison\", \"plot\", title)\n\nmanifest_df = pd.DataFrame(manifest_rows).drop_duplicates()\n\nif len(manifest_df) == 0:\n    manifest_df = pd.DataFrame(columns=[\"group\", \"file_type\", \"path\", \"note\"])\n\nmanifest_path = COMPLETE_DIR / \"OUTPUT_MANIFEST.csv\"\nmanifest_df.to_csv(manifest_path, index=False)\n\nchecklist = {\n    \"epoch_history_csv\": manifest_df[manifest_df[\"path\"].str.contains(\"epoch_history\", na=False)].shape[0],\n    \"accuracy_curves\": manifest_df[manifest_df[\"path\"].str.contains(\"accuracy_curve\", na=False)].shape[0],\n    \"loss_curves\": manifest_df[manifest_df[\"path\"].str.contains(\"loss_curve\", na=False)].shape[0],\n    \"f1_curves\": manifest_df[manifest_df[\"path\"].str.contains(\"f1_curve\", na=False)].shape[0],\n    \"learning_rate_curves\": manifest_df[manifest_df[\"path\"].str.contains(\"learning_rate_curve\", na=False)].shape[0],\n    \"gap_curves\": manifest_df[manifest_df[\"path\"].str.contains(\"gap_curve\", na=False)].shape[0],\n    \"loss_ratio_curves\": manifest_df[manifest_df[\"path\"].str.contains(\"loss_ratio\", na=False)].shape[0],\n    \"confidence_histograms\": manifest_df[manifest_df[\"path\"].str.contains(\"confidence_histogram\", na=False)].shape[0],\n    \"classification_reports\": manifest_df[manifest_df[\"path\"].str.contains(\"classification_report\", na=False)].shape[0],\n    \"confusion_matrix_png\": manifest_df[manifest_df[\"path\"].str.contains(\"confusion_matrix.*png\", regex=True, na=False)].shape[0],\n    \"top_confusion_pairs\": manifest_df[manifest_df[\"path\"].str.contains(\"top_confusion_pairs\", na=False)].shape[0],\n    \"comparison_plots\": manifest_df[\n        (manifest_df[\"group\"] == \"comparison\") &\n        (manifest_df[\"file_type\"] == \"plot\")\n    ].shape[0],\n}\n\nchecklist_path = COMPLETE_DIR / \"OUTPUT_CHECKLIST.json\"\n\nwith open(checklist_path, \"w\", encoding=\"utf-8\") as f:\n    json.dump(checklist, f, ensure_ascii=False, indent=2)\n\nprint(\"\\n=== CHECKLIST OUTPUT ĐẦY ĐỦ ===\")\nfor k, v in checklist.items():\n    print(f\"{k}: {v}\")\n\nprint(\"\\nĐã lưu manifest:\", manifest_path)\nprint(\"Đã lưu checklist:\", checklist_path)\nprint(\"Thư mục output:\", COMPLETE_DIR)\n\ndisplay(manifest_df.head(40))\n\nif bool(False):\n    clean_memory(label=\"sau complete_outputs\", verbose=False, kill_workers=False, deep=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T20:05:30.899851Z","iopub.execute_input":"2026-06-18T20:05:30.900438Z","iopub.status.idle":"2026-06-18T20:06:39.408486Z","shell.execute_reply.started":"2026-06-18T20:05:30.900413Z","shell.execute_reply":"2026-06-18T20:06:39.407600Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if \"OUTPUT_DIR\" not in globals():\n    print(\"Chưa có OUTPUT_DIR. Hãy chạy các cell cấu hình trước.\")\nelse:\n    output_counts = {}\n    for ext in [\"*.png\", \"*.csv\", \"*.txt\", \"*.json\", \"*.pth\", \"*.pt\"]:\n        output_counts[ext] = len(list(OUTPUT_DIR.rglob(ext)))\n\n    print(\"=== OUTPUT SANITY CHECK ===\")\n    for ext, count in output_counts.items():\n        print(f\"{ext}: {count}\")\n\n    required_hint = {\n        \"*.png\": \"biểu đồ/report hình ảnh\",\n        \"*.csv\": \"bảng số liệu/history/predictions\",\n        \"*.pth\": \"checkpoint model\",\n    }\n    for ext, note in required_hint.items():\n        if output_counts.get(ext, 0) == 0:\n            print(f\"⚠️ Chưa thấy {ext} ({note}). Kiểm tra đã chạy train/export chưa.\")\n\n    manifest_path = OUTPUT_DIR / \"complete_outputs\" / \"OUTPUT_MANIFEST.csv\"\n    if manifest_path.exists():\n        manifest_preview = pd.read_csv(manifest_path)\n        print(\"\\nManifest tồn tại:\", manifest_path)\n        display(manifest_preview.tail(20))\n    else:\n        print(\"\\nChưa thấy OUTPUT_MANIFEST.csv. Hãy chạy cell complete_outputs_and_manifest trước final export.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T20:06:39.409652Z","iopub.execute_input":"2026-06-18T20:06:39.410340Z","iopub.status.idle":"2026-06-18T20:06:39.629545Z","shell.execute_reply.started":"2026-06-18T20:06:39.410314Z","shell.execute_reply":"2026-06-18T20:06:39.628902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"overfit_tables = []\nfor result in all_results:\n    hist = pd.DataFrame(result[\"history\"])\n    if len(hist) == 0:\n        continue\n    hist[\"experiment\"] = result[\"experiment\"]\n    overfit_tables.append(hist)\n\nif overfit_tables:\n    overfit_df = pd.concat(overfit_tables, ignore_index=True)\n    display_cols = [\n        \"experiment\", \"phase\", \"epoch_in_phase\",\n        \"train_acc\", \"val_acc\", \"acc_gap\",\n        \"train_macro_f1\", \"val_macro_f1\", \"f1_gap\",\n        \"train_loss\", \"val_loss\", \"loss_ratio\",\n        \"lr_before\", \"lr_after\", \"overfit_warning\",\n    ]\n    existing_cols = [col for col in display_cols if col in overfit_df.columns]\n    display(overfit_df[existing_cols].tail(20))\n\n    lr_events = overfit_df[overfit_df[\"lr_after\"] < overfit_df[\"lr_before\"]] if {\"lr_after\", \"lr_before\"}.issubset(overfit_df.columns) else pd.DataFrame()\n    if len(lr_events) > 0:\n        print(\"Các epoch ReduceLROnPlateau đã giảm learning rate:\")\n        display(lr_events[[\"experiment\", \"phase\", \"epoch_in_phase\", \"lr_before\", \"lr_after\", \"val_macro_f1\"]])\n    else:\n        print(\"Chưa có epoch nào giảm learning rate. Điều này bình thường nếu validation macro F1 vẫn cải thiện hoặc số epoch test còn ít.\")\n\n    warning_rows = overfit_df[overfit_df.get(\"overfit_warning\", \"\").astype(str).str.len() > 0] if \"overfit_warning\" in overfit_df.columns else pd.DataFrame()\n    if len(warning_rows) > 0:\n        print(\"Các epoch có dấu hiệu overfitting:\")\n        display(warning_rows[[\"experiment\", \"phase\", \"epoch_in_phase\", \"overfit_warning\"]].tail(20))\n    else:\n        print(\"Chưa thấy cảnh báo overfitting theo ngưỡng hiện tại.\")\nelse:\n    print(\"Chưa có history để phân tích. Hãy chạy cell train thí nghiệm trước.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T20:06:39.630553Z","iopub.execute_input":"2026-06-18T20:06:39.630831Z","iopub.status.idle":"2026-06-18T20:06:39.681602Z","shell.execute_reply.started":"2026-06-18T20:06:39.630807Z","shell.execute_reply":"2026-06-18T20:06:39.681002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if \"summary_df\" in globals() and len(summary_df) > 0:\n    print(\"Top kết quả hiện tại:\")\n    display(summary_df[[\"experiment\", \"best_val_acc\", \"best_val_macro_f1\", \"best_checkpoint\"]].head(10))\nelse:\n    print(\"Chưa có summary_df. Hãy chạy cell train trước.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T20:06:39.682612Z","iopub.execute_input":"2026-06-18T20:06:39.682954Z","iopub.status.idle":"2026-06-18T20:06:39.693021Z","shell.execute_reply.started":"2026-06-18T20:06:39.682911Z","shell.execute_reply":"2026-06-18T20:06:39.692307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if \"PLOT_DIR\" in globals() and PLOT_DIR.exists():\n    output_files = []\n    for ext in [\"*.png\", \"*.csv\", \"*.txt\"]:\n        output_files.extend(PLOT_DIR.rglob(ext))\n\n    output_index_df = pd.DataFrame({\n        \"file\": [f.name for f in output_files],\n        \"folder\": [str(f.parent.relative_to(OUTPUT_DIR)) for f in output_files],\n        \"path\": [str(f) for f in output_files],\n    }).sort_values([\"folder\", \"file\"])\n\n    display(output_index_df)\n    output_index_path = OUTPUT_DIR / \"output_file_index.csv\"\n    output_index_df.to_csv(output_index_path, index=False)\n    print(\"Đã lưu danh sách file:\", output_index_path)\nelse:\n    print(\"Chưa có thư mục plots. Hãy chạy cell xuất biểu đồ trước.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T20:06:39.694187Z","iopub.execute_input":"2026-06-18T20:06:39.694555Z","iopub.status.idle":"2026-06-18T20:06:39.729478Z","shell.execute_reply.started":"2026-06-18T20:06:39.694531Z","shell.execute_reply":"2026-06-18T20:06:39.728903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class GradCAM:\n    def __init__(self, model: nn.Module, target_layer: nn.Module):\n        self.model = model\n        self.target_layer = target_layer\n        self.activations = None\n        self.gradients = None\n        self.forward_handle = target_layer.register_forward_hook(self._forward_hook)\n        self.backward_handle = target_layer.register_full_backward_hook(self._backward_hook)\n\n    def _forward_hook(self, module, input, output):\n        self.activations = output.detach()\n\n    def _backward_hook(self, module, grad_input, grad_output):\n        self.gradients = grad_output[0].detach()\n\n    def __call__(self, input_tensor, class_idx=None):\n        self.model.zero_grad(set_to_none=True)\n        logits = self.model(input_tensor)\n        if class_idx is None:\n            class_idx = int(logits.argmax(dim=1).item())\n        score = logits[:, class_idx].sum()\n        score.backward()\n        weights = self.gradients.mean(dim=(2, 3), keepdim=True)\n        cam = (weights * self.activations).sum(dim=1, keepdim=True)\n        cam = torch.relu(cam)\n        cam = torch.nn.functional.interpolate(cam, size=input_tensor.shape[-2:], mode=\"bilinear\", align_corners=False)\n        cam = cam.squeeze().cpu().numpy()\n        cam = (cam - cam.min()) / (cam.max() - cam.min() + 1e-8)\n        return cam, class_idx\n\n    def close(self):\n        self.forward_handle.remove()\n        self.backward_handle.remove()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T20:06:39.730357Z","iopub.execute_input":"2026-06-18T20:06:39.730607Z","iopub.status.idle":"2026-06-18T20:06:39.738265Z","shell.execute_reply.started":"2026-06-18T20:06:39.730585Z","shell.execute_reply":"2026-06-18T20:06:39.737554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PLOT_DIR = OUTPUT_DIR / \"plots\"\nPLOT_DIR.mkdir(parents=True, exist_ok=True)\n\nif bool(globals().get(\"ENABLE_GRAD_CAM\", True)):\n\n    gradcam_exp_name = \"effb0_pretrained_finetune\"\n    gradcam_rows = summary_df[summary_df[\"experiment\"].astype(str) == gradcam_exp_name]\n    if len(gradcam_rows) > 0:\n        gradcam_checkpoint = gradcam_rows.iloc[0][\"best_checkpoint\"]\n        gradcam_title_name = gradcam_exp_name\n    else:\n        gradcam_checkpoint = summary_df.iloc[0][\"best_checkpoint\"]\n        gradcam_title_name = summary_df.iloc[0][\"experiment\"]\n\n    gradcam_model = build_efficientnet_b0(num_classes=NUM_CLASSES, dropout=DROPOUT, pretrained=False).to(DEVICE)\n    gradcam_model.load_state_dict(torch.load(gradcam_checkpoint, map_location=DEVICE))\n    gradcam_model.eval()\n\n    target_layer = gradcam_model.features[-1]\n    cam_explainer = GradCAM(gradcam_model, target_layer)\n\n    sample_rows = val_df.sample(min(5, len(val_df)), random_state=SEED).reset_index(drop=True)\n\n    fig = plt.figure(figsize=(15, 6))\n\n    for i, row in sample_rows.iterrows():\n        raw_img = Image.open(row[\"img_path\"]).convert(\"RGB\")\n        input_tensor = val_tfms(raw_img).unsqueeze(0).to(DEVICE)\n        heatmap, pred_idx = cam_explainer(input_tensor)\n\n        plt.subplot(2, len(sample_rows), i + 1)\n        plt.imshow(raw_img.resize((IMG_SIZE, IMG_SIZE)))\n        plt.title(f\"True: {row['classname']}\")\n        plt.axis(\"off\")\n\n        plt.subplot(2, len(sample_rows), len(sample_rows) + i + 1)\n        plt.imshow(raw_img.resize((IMG_SIZE, IMG_SIZE)))\n        plt.imshow(heatmap, alpha=0.45, cmap=\"jet\")\n        plt.title(f\"Pred: {idx_to_class[pred_idx]}\")\n        plt.axis(\"off\")\n\n    plt.tight_layout()\n\n    gradcam_path = PLOT_DIR / f\"{gradcam_title_name}_gradcam_grid.png\"\n    plt.savefig(gradcam_path, dpi=180, bbox_inches=\"tight\")\n    plt.show()\n    print(\"Đã lưu Grad-CAM grid:\", gradcam_path)\n\n    cam_explainer.close()\n    clean_memory(label=\"sau Grad-CAM\", verbose=False, kill_workers=False)\n\nelif not bool(globals().get(\"ENABLE_GRAD_CAM\", True)):\n    print(\"ENABLE_GRAD_CAM=False nên bỏ qua Grad-CAM để tiết kiệm thời gian.\")\nelse:\n    print(\"Best model không phải EfficientNet-B0 nên bỏ qua Grad-CAM.\")\n\nif bool(False):\n    clean_memory(label=\"sau Grad-CAM grid\", verbose=False, kill_workers=False, deep=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-18T20:06:39.741301Z","iopub.execute_input":"2026-06-18T20:06:39.741599Z","iopub.status.idle":"2026-06-18T20:06:44.790829Z","shell.execute_reply.started":"2026-06-18T20:06:39.741577Z","shell.execute_reply":"2026-06-18T20:06:44.789878Z"}},"outputs":[],"execution_count":null}]}