{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":46105,"databundleVersionId":5087314,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Identifying the top 80 words from the full dataset","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# Folder dir\nINPUT_CSV = '/kaggle/input/asl-signs/train.csv'\nNUM_CLASSES = 80\n\n# 1. Load Original CSV\nprint(\"Reading CSV...\")\ndf = pd.read_csv(INPUT_CSV)\n\n# 2. Find Top 100 Words\nprint(f\"Finding top {NUM_CLASSES} signs...\")\ntop_100_signs = df['sign'].value_counts().head(NUM_CLASSES).index.tolist()\n\n# 3. Filter the DataFrame\ndf_subset = df[df['sign'].isin(top_100_signs)].copy()\n\nprint(f\"Selected {len(df_subset)} files for {len(top_100_signs)} signs.\")\nprint(f\"Signs: {top_100_signs[:10]}...\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-14T06:12:03.543869Z","iopub.execute_input":"2025-12-14T06:12:03.544126Z","iopub.status.idle":"2025-12-14T06:12:03.990216Z","shell.execute_reply.started":"2025-12-14T06:12:03.544105Z","shell.execute_reply":"2025-12-14T06:12:03.989445Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Copying files to output","metadata":{}},{"cell_type":"code","source":"import shutil\nfrom tqdm.notebook import tqdm\nimport os\n\nSOURCE_ROOT = '/kaggle/input/asl-signs'\n\n# CHANGE: Build the dataset in the TEMP directory first\n# This prevents your 19GB quota from filling up with loose files\nDEST_ROOT = '/kaggle/temp/asl_subset_100' \n\n# 1. Reset/Create Directory\nif os.path.exists(DEST_ROOT):\n    print(\"Removing old subset folder in temp...\")\n    shutil.rmtree(DEST_ROOT)\n    \nos.makedirs(DEST_ROOT, exist_ok=True)\nprint(f\"Created temp output folder: {DEST_ROOT}\")\n\n# 2. Copy Loop\nprint(\"Starting Copy Process (This keeps all Facial Features)...\")\n\ncopied_count = 0\n\nfor index, row in tqdm(df_subset.iterrows(), total=len(df_subset)):\n    # Original Path\n    src_path = os.path.join(SOURCE_ROOT, row['path'])\n    \n    # Destination Path (in Temp)\n    dest_path = os.path.join(DEST_ROOT, row['path'])\n    \n    # Make sure the sub-folder exists\n    os.makedirs(os.path.dirname(dest_path), exist_ok=True)\n    \n    # Copy the file\n    shutil.copy2(src_path, dest_path)\n    copied_count += 1\n\nprint(f\"Successfully copied {copied_count} files to Temp storage.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T06:12:08.757871Z","iopub.execute_input":"2025-12-14T06:12:08.758425Z","iopub.status.idle":"2025-12-14T06:20:49.110482Z","shell.execute_reply.started":"2025-12-14T06:12:08.758399Z","shell.execute_reply":"2025-12-14T06:20:49.109652Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Creating index csv","metadata":{}},{"cell_type":"code","source":"# 1. Save the filtered CSV inside the temp folder\ncsv_save_path = os.path.join(DEST_ROOT, 'train.csv')\ndf_subset.to_csv(csv_save_path, index=False)\nprint(f\"Saved index file to: {csv_save_path}\")\n\n# 2. ZIP THE DATASET\n# We zip from '/kaggle/temp/asl_subset_100' -> To -> '/kaggle/working/asl_data_compressed'\nprint(\"Zipping dataset... (This allows us to bypass the 500-file upload limit)\")\n\nshutil.make_archive(\n    '/kaggle/working/asl_subset_100', # Destination (Output folder)\n    'zip',                                 # Format\n    DEST_ROOT                              # Source (Temp folder)\n)\n\nprint(\"------------------------------------------------\")\nprint(\"SUCCESS! A file named 'asl_data_compressed.zip' has been created.\")\nprint(\"You can now Save & Run All. Use this ZIP file to create your dataset.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T06:20:58.003034Z","iopub.execute_input":"2025-12-14T06:20:58.003323Z","iopub.status.idle":"2025-12-14T06:45:51.615167Z","shell.execute_reply.started":"2025-12-14T06:20:58.003300Z","shell.execute_reply":"2025-12-14T06:45:51.614523Z"}},"outputs":[],"execution_count":null}]}