{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Download Validation Set\nAs the ILSVRC dataset is only available using kaggle and you can only download the whole DB using the CLI and the web page, I found a way to download only the validation set I needed. ","metadata":{}},{"cell_type":"markdown","source":"1. Lets be sure that the DB is in the folder:  You should see a list of 10 images called `ILSVRC2012_val_xxxxxxxx.JPEG`","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\n\nROOT = Path(\n    \"/kaggle/input/imagenet-object-localization-challenge\"\n)\n\nVAL_DIR = ROOT / \"ILSVRC/Data/CLS-LOC/val\"\nVAL_CSV = ROOT / \"LOC_val_solution.csv\"\nMAPPING = ROOT / \"LOC_synset_mapping.txt\"\n\nprint(\"ROOT:\", ROOT.exists())\nprint(\"VAL_DIR:\", VAL_DIR.exists(), VAL_DIR)\nprint(\"VAL_CSV:\", VAL_CSV.exists(), VAL_CSV)\nprint(\"MAPPING:\", MAPPING.exists(), MAPPING)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-01T12:05:56.111819Z","iopub.execute_input":"2026-10-01T12:05:56.112187Z","iopub.status.idle":"2026-10-01T12:05:56.122924Z","shell.execute_reply.started":"2026-10-01T12:05:56.112154Z","shell.execute_reply":"2026-10-01T12:05:56.121644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import subprocess\n\nresult = subprocess.run(\n    [\n        \"bash\",\n        \"-lc\",\n        f'find \"{VAL_DIR}\" -maxdepth 1 -type f -name \"*.JPEG\" | head -n 5'\n    ],\n    capture_output=True,\n    text=True,\n    check=True\n)\n\nprint(result.stdout)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-01T12:06:49.820426Z","iopub.execute_input":"2026-10-01T12:06:49.820755Z","iopub.status.idle":"2026-10-01T12:06:50.196324Z","shell.execute_reply.started":"2026-10-01T12:06:49.820726Z","shell.execute_reply":"2026-10-01T12:06:50.194911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result = subprocess.run(\n    [\n        \"bash\",\n        \"-lc\",\n        f'find \"{VAL_DIR}\" -maxdepth 1 -type f -name \"*.JPEG\" | wc -l'\n    ],\n    capture_output=True,\n    text=True,\n    check=True\n)\n\nprint(\"Validation image count:\", result.stdout.strip())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-01T12:07:16.666003Z","iopub.execute_input":"2026-10-01T12:07:16.666448Z","iopub.status.idle":"2026-10-01T12:08:33.038308Z","shell.execute_reply.started":"2026-10-01T12:07:16.66641Z","shell.execute_reply":"2026-10-01T12:08:33.036864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndf = pd.read_csv(VAL_CSV)\n\nprint(\"Columns:\", df.columns.tolist())\nprint(\"Rows:\", len(df))\nprint(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-01T12:08:58.831993Z","iopub.execute_input":"2026-10-01T12:08:58.832388Z","iopub.status.idle":"2026-10-01T12:08:58.97702Z","shell.execute_reply.started":"2026-10-01T12:08:58.832352Z","shell.execute_reply":"2026-10-01T12:08:58.975624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(MAPPING, \"r\", encoding=\"utf-8\") as file:\n    mapping_lines = [\n        line.strip()\n        for line in file\n        if line.strip()\n    ]\n\nprint(\"Mapping rows:\", len(mapping_lines))\nprint(\"First five mappings:\")\n\nfor line in mapping_lines[:5]:\n    print(line)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-01T12:09:34.057465Z","iopub.execute_input":"2026-10-01T12:09:34.057873Z","iopub.status.idle":"2026-10-01T12:09:34.070183Z","shell.execute_reply.started":"2026-10-01T12:09:34.057835Z","shell.execute_reply":"2026-10-01T12:09:34.068787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport subprocess\nimport pandas as pd\n\nROOT = Path(\n    \"/kaggle/input/imagenet-object-localization-challenge\"\n)\n\nVAL_DIR = ROOT / \"ILSVRC/Data/CLS-LOC/val\"\nVAL_CSV = ROOT / \"LOC_val_solution.csv\"\nMAPPING = ROOT / \"LOC_synset_mapping.txt\"\n\nprint(\"ROOT exists:\", ROOT.exists())\nprint(\"VAL_DIR exists:\", VAL_DIR.exists())\nprint(\"VAL_CSV exists:\", VAL_CSV.exists())\nprint(\"MAPPING exists:\", MAPPING.exists())\n\nif not VAL_DIR.exists():\n    raise FileNotFoundError(VAL_DIR)\n\nif not VAL_CSV.exists():\n    print(\"\\nRoot-level files:\")\n    for path in ROOT.iterdir():\n        print(path.name)\n    raise FileNotFoundError(VAL_CSV)\n\nif not MAPPING.exists():\n    print(\"\\nRoot-level files:\")\n    for path in ROOT.iterdir():\n        print(path.name)\n    raise FileNotFoundError(MAPPING)\n\ncount_result = subprocess.run(\n    [\n        \"bash\",\n        \"-lc\",\n        f'find \"{VAL_DIR}\" -maxdepth 1 -type f -name \"*.JPEG\" | wc -l'\n    ],\n    capture_output=True,\n    text=True,\n    check=True\n)\n\nprint(\n    \"Validation image count:\",\n    count_result.stdout.strip()\n)\n\ndf = pd.read_csv(VAL_CSV)\n\nprint(\"CSV rows:\", len(df))\nprint(\"CSV columns:\", df.columns.tolist())\nprint(df.head())\n\nwith open(MAPPING, \"r\", encoding=\"utf-8\") as file:\n    mapping_lines = [\n        line.strip()\n        for line in file\n        if line.strip()\n    ]\n\nprint(\"Mapping rows:\", len(mapping_lines))\nprint(\"First mapping:\", mapping_lines[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-01T12:10:09.855211Z","iopub.execute_input":"2026-10-01T12:10:09.855572Z","iopub.status.idle":"2026-10-01T12:10:39.534974Z","shell.execute_reply.started":"2026-10-01T12:10:09.855543Z","shell.execute_reply":"2026-10-01T12:10:39.533622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nSEED = 42\nNUM_CLASSES = 100\nIMAGES_PER_CLASS = 10\n\n# 從 PredictionString 取出第一個 wnid\ndf[\"wnid\"] = df[\"PredictionString\"].str.split().str[0]\n\n# 補上實際 JPEG 檔名\ndf[\"image_name\"] = df[\"ImageId\"] + \".JPEG\"\n\nprint(\"總圖片數：\", len(df))\nprint(\"總類別數：\", df[\"wnid\"].nunique())\nprint(\"\\n各類別圖片數統計：\")\nprint(df.groupby(\"wnid\").size().describe())\n\n# 確認所有 CSV 紀錄都有對應圖片\nmissing_count = sum(\n    not (VAL_DIR / image_name).exists()\n    for image_name in df[\"image_name\"]\n)\n\nprint(\"\\n找不到的圖片數：\", missing_count)\n\nif missing_count != 0:\n    raise RuntimeError(\n        f\"有 {missing_count} 張圖片找不到，先不要繼續\"\n    )\n\n# 固定選取 100 個類別\nrng = np.random.default_rng(SEED)\n\nall_classes = sorted(df[\"wnid\"].unique())\n\nselected_classes = sorted(\n    rng.choice(\n        all_classes,\n        size=NUM_CLASSES,\n        replace=False\n    ).tolist()\n)\n\n# 每個選定類別固定抽取 10 張\nselected_parts = []\n\nfor class_number, wnid in enumerate(selected_classes):\n    class_df = df[df[\"wnid\"] == wnid].copy()\n\n    sampled = class_df.sample(\n        n=IMAGES_PER_CLASS,\n        replace=False,\n        random_state=SEED + class_number\n    )\n\n    selected_parts.append(sampled)\n\nselected = pd.concat(\n    selected_parts,\n    ignore_index=True\n)\n\nselected = selected.sort_values(\n    [\"wnid\", \"image_name\"]\n).reset_index(drop=True)\n\nprint(\"\\n抽樣結果\")\nprint(\"圖片數：\", len(selected))\nprint(\"類別數：\", selected[\"wnid\"].nunique())\nprint(\"\\n每類圖片數分布：\")\nprint(selected.groupby(\"wnid\").size().value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-01T12:13:50.994036Z","iopub.execute_input":"2026-10-01T12:13:50.995163Z","iopub.status.idle":"2026-10-01T12:14:23.121102Z","shell.execute_reply.started":"2026-10-01T12:13:50.995123Z","shell.execute_reply":"2026-10-01T12:14:23.120021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mapping_rows = []\n\nwith open(MAPPING, \"r\", encoding=\"utf-8\") as file:\n    for class_index, line in enumerate(file):\n        line = line.strip()\n\n        if not line:\n            continue\n\n        parts = line.split(\" \", 1)\n\n        wnid = parts[0]\n        class_name = parts[1] if len(parts) > 1 else \"\"\n\n        mapping_rows.append({\n            \"wnid\": wnid,\n            \"class_index\": class_index,\n            \"class_name\": class_name\n        })\n\nmapping_df = pd.DataFrame(mapping_rows)\n\nprint(\"類別對照筆數：\", len(mapping_df))\nprint(mapping_df.head())\n\nif len(mapping_df) != 1000:\n    raise RuntimeError(\n        f\"類別對照應為 1000 筆，目前為 {len(mapping_df)} 筆\"\n    )\n\nselected = selected.merge(\n    mapping_df,\n    on=\"wnid\",\n    how=\"left\",\n    validate=\"many_to_one\"\n)\n\nmissing_mapping = selected[\"class_index\"].isna().sum()\n\nprint(\"\\n找不到類別索引的圖片數：\", missing_mapping)\n\nif missing_mapping != 0:\n    print(\n        selected.loc[\n            selected[\"class_index\"].isna(),\n            \"wnid\"\n        ].unique()\n    )\n    raise RuntimeError(\"部分 wnid 找不到類別索引\")\n\nselected[\"class_index\"] = selected[\"class_index\"].astype(int)\n\nprint(\"\\n抽樣資料預覽：\")\nprint(\n    selected[\n        [\n            \"image_name\",\n            \"wnid\",\n            \"class_index\",\n            \"class_name\"\n        ]\n    ].head(10)\n)\n\nprint(\n    \"\\n類別索引範圍：\",\n    selected[\"class_index\"].min(),\n    \"到\",\n    selected[\"class_index\"].max()\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-01T12:15:05.11194Z","iopub.execute_input":"2026-10-01T12:15:05.113081Z","iopub.status.idle":"2026-10-01T12:15:05.148182Z","shell.execute_reply.started":"2026-10-01T12:15:05.113036Z","shell.execute_reply":"2026-10-01T12:15:05.14694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nimport shutil\n\nOUTPUT_ROOT = Path(\n    \"/kaggle/working/imagenet_val_1000\"\n)\n\nOUTPUT_IMAGES = OUTPUT_ROOT / \"images\"\n\n# 如果之前執行過，先清掉舊輸出，避免混入其他抽樣\nif OUTPUT_ROOT.exists():\n    shutil.rmtree(OUTPUT_ROOT)\n\nOUTPUT_IMAGES.mkdir(\n    parents=True,\n    exist_ok=True\n)\n\nfor row in selected.itertuples(index=False):\n    source = VAL_DIR / row.image_name\n    destination = OUTPUT_IMAGES / row.image_name\n\n    if not source.exists():\n        raise FileNotFoundError(source)\n\n    shutil.copy2(source, destination)\n\n# 保存抽樣與標籤清單\noutput_columns = [\n    \"image_name\",\n    \"wnid\",\n    \"class_index\",\n    \"class_name\"\n]\n\nselected[\n    output_columns\n].to_csv(\n    OUTPUT_ROOT / \"selected_val_1000.csv\",\n    index=False\n)\n\n# 紀錄抽樣設定，讓實驗可以重現\nwith open(\n    OUTPUT_ROOT / \"sampling_info.txt\",\n    \"w\",\n    encoding=\"utf-8\"\n) as file:\n    file.write(f\"seed={SEED}\\n\")\n    file.write(f\"num_classes={NUM_CLASSES}\\n\")\n    file.write(\n        f\"images_per_class={IMAGES_PER_CLASS}\\n\"\n    )\n    file.write(f\"total_images={len(selected)}\\n\")\n\ncopied_images = list(\n    OUTPUT_IMAGES.glob(\"*.JPEG\")\n)\n\nprint(\"已複製圖片數：\", len(copied_images))\nprint(\"輸出目錄：\", OUTPUT_ROOT)\nprint(\"標籤檔：\", OUTPUT_ROOT / \"selected_val_1000.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-01T12:16:34.621336Z","iopub.execute_input":"2026-10-01T12:16:34.621709Z","iopub.status.idle":"2026-10-01T12:16:39.501072Z","shell.execute_reply.started":"2026-10-01T12:16:34.621676Z","shell.execute_reply":"2026-10-01T12:16:39.500014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\nimport pandas as pd\n\ncheck_df = pd.read_csv(\n    OUTPUT_ROOT / \"selected_val_1000.csv\"\n)\n\n# 基本數量檢查\nassert len(check_df) == 1000\nassert check_df[\"wnid\"].nunique() == 100\nassert (check_df.groupby(\"wnid\").size() == 10).all()\nassert check_df[\"class_index\"].between(0, 999).all()\n\n# 檢查 JPEG 是否可以正常讀取\nbad_images = []\n\nfor image_name in check_df[\"image_name\"]:\n    image_path = OUTPUT_IMAGES / image_name\n\n    try:\n        with Image.open(image_path) as image:\n            image.verify()\n    except Exception as error:\n        bad_images.append({\n            \"image_name\": image_name,\n            \"error\": str(error)\n        })\n\nprint(\"CSV 資料筆數：\", len(check_df))\nprint(\"類別數：\", check_df[\"wnid\"].nunique())\nprint(\n    \"JPEG 檔案數：\",\n    len(list(OUTPUT_IMAGES.glob(\"*.JPEG\")))\n)\nprint(\"損壞或無法讀取的圖片：\", len(bad_images))\n\nif bad_images:\n    print(bad_images[:10])\nelse:\n    print(\"全部檢查通過，可以壓縮。\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-01T12:17:27.718425Z","iopub.execute_input":"2026-10-01T12:17:27.718839Z","iopub.status.idle":"2026-10-01T12:17:27.891976Z","shell.execute_reply.started":"2026-10-01T12:17:27.718805Z","shell.execute_reply":"2026-10-01T12:17:27.890905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nfrom pathlib import Path\n\narchive_path = shutil.make_archive(\n    base_name=\"/kaggle/working/imagenet_val_1000\",\n    format=\"gztar\",\n    root_dir=\"/kaggle/working\",\n    base_dir=\"imagenet_val_1000\"\n)\n\narchive_size_mb = (\n    Path(archive_path).stat().st_size\n    / 1024\n    / 1024\n)\n\nprint(\"壓縮檔：\", archive_path)\nprint(f\"檔案大小：{archive_size_mb:.2f} MB\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-10-01T12:18:04.896802Z","iopub.execute_input":"2026-10-01T12:18:04.897477Z","iopub.status.idle":"2026-10-01T12:18:09.786832Z","shell.execute_reply.started":"2026-10-01T12:18:04.897435Z","shell.execute_reply":"2026-10-01T12:18:09.785645Z"}},"outputs":[],"execution_count":null}]}