{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":71549,"databundleVersionId":8561470,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\n# 1. List everything in the main input folder\nprint(\"--- WHAT IS IN YOUR INPUT FOLDER? ---\")\ntry:\n    items = os.listdir(\"/kaggle/input\")\n    print(items)\nexcept FileNotFoundError:\n    print(\"ERROR: /kaggle/input folder does not exist.\")\n\n# 2. If it's empty '[]', the data is NOT added.\nif len(items) == 0:\n    print(\"\\n>>> RESULT: The folder is EMPTY. Please complete Step 2 above.\")\nelse:\n    print(f\"\\n>>> RESULT: Found {len(items)} item(s).\")\n    # If we found something, let's see what the full path is\n    for item in items:\n        print(f\"Path found: /kaggle/input/{item}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T17:02:10.145094Z","iopub.execute_input":"2025-12-03T17:02:10.145450Z","iopub.status.idle":"2025-12-03T17:02:10.152924Z","shell.execute_reply.started":"2025-12-03T17:02:10.145425Z","shell.execute_reply":"2025-12-03T17:02:10.151975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\n\n# --- CONFIGURATION ---\n# We build the path based on what you found\nbase_path = \"/kaggle/input/rsna-2024-lumbar-spine-degenerative-classification\"\ninput_folder = os.path.join(base_path, \"train_images\")\noutput_folder = \"/kaggle/working/train_images_subset\"\ntarget_size_gb = 4\n# ---------------------\n\n# 1. Verification Check\nprint(f\"Checking for images in: {input_folder}...\")\nif not os.path.exists(input_folder):\n    print(\"\\nERROR: 'train_images' folder not found.\")\n    print(f\"Contents of main folder '{base_path}':\")\n    print(os.listdir(base_path))\n    print(\"\\nPlease update the 'input_folder' variable if you see the image folder named differently above.\")\nelse:\n    # 2. Start Copying\n    max_size_bytes = target_size_gb * 1024**3\n    current_size_bytes = 0\n    \n    print(f\"Folder found! Starting copy of ~{target_size_gb}GB...\")\n    \n    # Clean output folder if it exists from previous attempts\n    if os.path.exists(output_folder):\n        shutil.rmtree(output_folder)\n    os.makedirs(output_folder, exist_ok=True)\n\n    # Get list of study folders\n    studies = sorted(os.listdir(input_folder))\n    count = 0\n    \n    for study in studies:\n        if current_size_bytes >= max_size_bytes:\n            break\n            \n        src_path = os.path.join(input_folder, study)\n        dst_path = os.path.join(output_folder, study)\n        \n        if os.path.isdir(src_path):\n            shutil.copytree(src_path, dst_path)\n            count += 1\n            \n            # Calculate size\n            for root, _, files in os.walk(dst_path):\n                current_size_bytes += sum(os.path.getsize(os.path.join(root, name)) for name in files)\n\n            if count % 20 == 0:\n                print(f\"Copied {count} studies... ({current_size_bytes / 1024**2:.0f} MB)\")\n\n    print(f\"Copy Finished. Total Size: {current_size_bytes / 1024**3:.2f} GB\")\n\n    # 3. Zip the files\n    print(\"Zipping files... (Please wait, this takes about 1-2 minutes)\")\n    shutil.make_archive(\"/kaggle/working/subset_4gb\", 'zip', output_folder)\n    \n    print(\"\\nSUCCESS! ===================================================\")\n    print(\"1. Look at the 'Output' section in the right sidebar.\")\n    print(\"2. Find 'subset_4gb.zip'.\")\n    print(\"3. Click the three dots (...) and select 'Download'.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T17:05:30.795063Z","iopub.execute_input":"2025-12-03T17:05:30.795423Z","iopub.status.idle":"2025-12-03T17:13:32.831324Z","shell.execute_reply.started":"2025-12-03T17:05:30.795390Z","shell.execute_reply":"2025-12-03T17:13:32.829739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import FileLink, display\n\n# Create a clickable link for the file\nprint(\"Click the blue link below to start the download:\")\ndisplay(FileLink(r'subset_4gb.zip'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-03T17:23:02.013008Z","iopub.execute_input":"2025-12-03T17:23:02.014082Z","iopub.status.idle":"2025-12-03T17:23:02.025758Z","shell.execute_reply.started":"2025-12-03T17:23:02.014021Z","shell.execute_reply":"2025-12-03T17:23:02.024380Z"}},"outputs":[],"execution_count":null}]}