{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":6799,"databundleVersionId":4225553,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\"\"\"\nKaggle Notebook: Create ImageNet Subset\n\nRun this in a Kaggle notebook with the ImageNet dataset attached.\nThe output will be available in /kaggle/working/ for download.\n\"\"\"\n\nimport os\nimport shutil\nimport random\nfrom pathlib import Path\n\ndef create_imagenet_subset():\n    \"\"\"\n    Create a subset from Kaggle ImageNet targeting ~10GB total size.\n    Sample ~35 images per class to get approximately 35k total images.\n    \"\"\"\n    print(\"Creating ImageNet subset targeting ~10GB...\")\n    \n    # Kaggle ImageNet paths\n    train_dir = '/kaggle/input/imagenet-object-localization-challenge/ILSVRC/Data/CLS-LOC/train'\n    \n    # Output directories\n    output_base = '/kaggle/working/imagenet_subset'\n    output_train = os.path.join(output_base, 'train')\n    \n    # Create output directories\n    os.makedirs(output_train, exist_ok=True)\n    \n    # Get all class directories from training set\n    class_dirs = [d for d in os.listdir(train_dir) if os.path.isdir(os.path.join(train_dir, d))]\n    print(f\"Found {len(class_dirs)} classes in training set\")\n    \n    # Target ~35 images per class for ~10GB total\n    # (35k images * ~300KB average = ~10GB)\n    samples_per_class = 35\n    total_copied = 0\n    \n    print(f\"Sampling {samples_per_class} images per class\")\n    print(f\"Expected total: {len(class_dirs) * samples_per_class} images (~10GB)\")\n    \n    # Process each class\n    for class_name in class_dirs:\n        class_train_dir = os.path.join(train_dir, class_name)\n        \n        # Get all images in this class\n        image_files = [f for f in os.listdir(class_train_dir) \n                      if f.lower().endswith(('.jpg', '.jpeg', '.png'))]\n        \n        if len(image_files) < samples_per_class:\n            print(f\"Warning: Class {class_name} only has {len(image_files)} images, using all\")\n            use_count = len(image_files)\n        else:\n            use_count = samples_per_class\n        \n        # Randomly sample images\n        random.shuffle(image_files)\n        selected_images = image_files[:use_count]\n        \n        # Create class directory in output\n        output_class_dir = os.path.join(output_train, class_name)\n        os.makedirs(output_class_dir, exist_ok=True)\n        \n        # Copy selected images\n        for img_file in selected_images:\n            src = os.path.join(class_train_dir, img_file)\n            dst = os.path.join(output_class_dir, img_file)\n            shutil.copy2(src, dst)\n            total_copied += 1\n        \n        if (len([d for d in os.listdir(output_train) if os.path.isdir(os.path.join(output_train, d))]) % 100) == 0:\n            print(f\"Processed {len([d for d in os.listdir(output_train) if os.path.isdir(os.path.join(output_train, d))])} classes...\")\n    \n    print(f\"Subset creation complete!\")\n    print(f\"Total images: {total_copied}\")\n    print(f\"Expected: {len(class_dirs) * samples_per_class}\")\n    \n    return output_base\n\ndef zip_subset(subset_dir):\n    \"\"\"\n    Zip the subset for download from Kaggle.\n    \"\"\"\n    print(\"Zipping subset...\")\n    \n    zip_path = '/kaggle/working/imagenet_subset.zip'\n    \n    with zipfile.ZipFile(zip_path, 'w', zipfile.ZIP_DEFLATED) as zipf:\n        for root, dirs, files in os.walk(subset_dir):\n            for file in files:\n                file_path = os.path.join(root, file)\n                # Get relative path for zip\n                arcname = os.path.relpath(file_path, '/kaggle/working/')\n                zipf.write(file_path, arcname)\n    \n    print(f\"Zip created: {zip_path}\")\n    \n    # Get zip file size\n    zip_size_mb = os.path.getsize(zip_path) / (1024 * 1024)\n    print(f\"Zip file size: {zip_size_mb:.1f} MB\")\n    \n    return zip_path\n\ndef main():\n    \"\"\"\n    Main function to create subset (no zipping due to space constraints).\n    \"\"\"\n    random.seed(42)  # For reproducibility\n    \n    # Create the subset\n    subset_dir = create_imagenet_subset()\n    \n    print(\"Subset creation complete!\")\n    print(\"Files are in /kaggle/working/imagenet_subset/\")\n    print(\"Use Kaggle's dataset creation feature to save this as a dataset.\")\n    print(\"Then you can access it from Colab without downloading.\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-01T21:12:07.822103Z","iopub.execute_input":"2025-08-01T21:12:07.822491Z","iopub.status.idle":"2025-08-01T21:19:11.550633Z","shell.execute_reply.started":"2025-08-01T21:12:07.82246Z","shell.execute_reply":"2025-08-01T21:19:11.549056Z"}},"outputs":[],"execution_count":null}]}