{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":6799,"databundleVersionId":4225553,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Data Extraction from the Imagenet for learning purpose","metadata":{}},{"cell_type":"markdown","source":"****Just change number of folders to extract below the function and this will give you ImageNet Subset dataset ready****","metadata":{}},{"cell_type":"code","source":"import os\nimport shutil\nimport random\n\ndef extract_and_split_imagenet(\n    src_root: str,\n    mapping_txt: str,\n    dest_root: str,\n    num_classes: int,\n    train_ratio: float = 0.75,\n    seed: int = None,\n) -> None:\n    \"\"\"\n    Randomly pick `num_classes` sub-directories from `src_root`, rename them using\n    the first label in `mapping_txt`, and split their images into train/val.\n\n    If `seed` is provided, the sampling & shuffle will be repeatable; otherwise,\n    each call will be different.\n    \"\"\"\n    # 1) Load mapping: id → first label before any comma\n    id2label = {}\n    with open(mapping_txt, 'r') as f:\n        for line in f:\n            parts = line.strip().split()\n            if len(parts) < 2:\n                continue\n            synset_id = parts[0]\n            labels = \" \".join(parts[1:])\n            primary = labels.split(',')[0].strip().replace(' ', '_')\n            id2label[synset_id] = primary\n\n    # 2) List all class-dirs in src_root\n    all_ids = [d for d in os.listdir(src_root)\n               if os.path.isdir(os.path.join(src_root, d))]\n    if num_classes > len(all_ids):\n        raise ValueError(f\"Requested {num_classes} classes, but only found {len(all_ids)}\")\n\n    # 3) Seed global RNG if requested\n    if seed is not None:\n        random.seed(seed)\n\n    # 4) Sample random IDs using the global random\n    selected = random.sample(all_ids, num_classes)\n\n    # 5) Prepare destination train/val roots\n    train_root = os.path.join(dest_root, 'train')\n    val_root   = os.path.join(dest_root, 'val')\n    os.makedirs(train_root, exist_ok=True)\n    os.makedirs(val_root,   exist_ok=True)\n\n    # 6) Process each selected class\n    for syn_id in selected:\n        if syn_id not in id2label:\n            raise KeyError(f\"Synset ID '{syn_id}' not found in mapping file!\")\n        label = id2label[syn_id]\n\n        # Create label-named subdirs\n        trgt_train = os.path.join(train_root, label)\n        trgt_val   = os.path.join(val_root,   label)\n        os.makedirs(trgt_train, exist_ok=True)\n        os.makedirs(trgt_val,   exist_ok=True)\n\n        # Gather & shuffle images\n        src_dir = os.path.join(src_root, syn_id)\n        imgs = [f for f in os.listdir(src_dir)\n                if os.path.isfile(os.path.join(src_dir, f))]\n        random.shuffle(imgs)\n\n        # Split indices\n        cut = int(len(imgs) * train_ratio)\n        train_imgs = imgs[:cut]\n        val_imgs   = imgs[cut:]\n\n        # Copy files\n        for fn in train_imgs:\n            shutil.copy2(os.path.join(src_dir, fn),\n                         os.path.join(trgt_train, fn))\n        for fn in val_imgs:\n            shutil.copy2(os.path.join(src_dir, fn),\n                         os.path.join(trgt_val, fn))\n\n        print(f\"  • {syn_id} → '{label}' ({len(train_imgs)} train, {len(val_imgs)} val)\")\n\n    print(f\"\\nDone! {num_classes} classes split into:\\n\"\n          f\"  {train_root}/\\n  {val_root}/\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-30T16:30:42.152765Z","iopub.execute_input":"2025-04-30T16:30:42.153109Z","iopub.status.idle":"2025-04-30T16:30:42.168924Z","shell.execute_reply.started":"2025-04-30T16:30:42.153084Z","shell.execute_reply":"2025-04-30T16:30:42.168065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"extract_and_split_imagenet(\n    src_root='/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train',\n    mapping_txt='/kaggle/input/imagenet-object-localization-challenge/LOC_synset_mapping.txt',\n    dest_root='/kaggle/working/imagenet_sample',\n    num_classes=10,\n    seed=None,\n    train_ratio = 0.8\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T16:31:40.086535Z","iopub.execute_input":"2025-04-30T16:31:40.086887Z","iopub.status.idle":"2025-04-30T16:32:56.767817Z","shell.execute_reply.started":"2025-04-30T16:31:40.086862Z","shell.execute_reply":"2025-04-30T16:32:56.76706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nimport os\n\ndef zip_train_val(dest_root: str, zip_root: str = None):\n    \"\"\"\n    Create ZIP archives of the `train` and `val` folders under dest_root.\n\n    Args:\n        dest_root (str): path where 'train' and 'val' live.\n        zip_root  (str): directory where .zip files should be placed.\n                         Defaults to dest_root itself.\n    \"\"\"\n    if zip_root is None:\n        zip_root = dest_root\n\n    for split in ('train', 'val'):\n        folder = os.path.join(dest_root, split)\n        if not os.path.isdir(folder):\n            raise FileNotFoundError(f\"Expected directory not found: {folder}\")\n\n        # make_archive will append .zip for you\n        archive_name = os.path.join(zip_root, split)\n        print(f\"Zipping {folder} → {archive_name}.zip …\")\n        shutil.make_archive(archive_name, 'zip', root_dir=dest_root, base_dir=split)\n\n    print(f\"\\n✅ Zipped 'train' and 'val' into:\\n  {zip_root}/train.zip\\n  {zip_root}/val.zip\")\nzip_train_val(dest_root='/kaggle/working/imagenet_sample')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T16:32:59.593607Z","iopub.execute_input":"2025-04-30T16:32:59.593943Z","iopub.status.idle":"2025-04-30T16:33:51.168016Z","shell.execute_reply.started":"2025-04-30T16:32:59.593917Z","shell.execute_reply":"2025-04-30T16:33:51.167013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}