{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":4117,"databundleVersionId":46665,"sourceType":"competition"}],"dockerImageVersionId":31239,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport subprocess\nfrom tqdm import tqdm\n\n# =========================\n# CONFIG\n# =========================\nCLASS_ID = 2\nFAMILY_NAME = \"Lollipop\"\n\nCHUNK_SIZE = 500\nCHUNK_ID = 1\n\nSUB_CHUNK_SIZE = 250\nSUB_CHUNK_ID = 1    # ✅ SECOND HALF (750–999)\n\n# =========================\n# PATHS\n# =========================\nINPUT_DIR = \"/kaggle/input/malware-classification\"\nARCHIVE_PATH = os.path.join(INPUT_DIR, \"train.7z\")\nLABEL_CSV = os.path.join(INPUT_DIR, \"trainLabels.csv\")\n\nWORK_DIR = \"/kaggle/working\"\nOUTPUT_DIR = os.path.join(WORK_DIR, \"asm_labeled\", FAMILY_NAME)\n\nARCHIVE_OUT = os.path.join(\n    WORK_DIR,\n    \"Lollipop_ASM_chunk_1_sub_1.7z\"\n)\n\n# =========================\n# PREPARE OUTPUT\n# =========================\nos.makedirs(OUTPUT_DIR, exist_ok=True)\nprint(\"✅ Output folder ready (Chunk 1 / Sub 1)\")\n\n# =========================\n# LOAD CSV\n# =========================\ndf = pd.read_csv(LABEL_CSV)\ndf = df[df[\"Class\"] == CLASS_ID].reset_index(drop=True)\n\nTOTAL = len(df)\n\nCHUNK_START = CHUNK_ID * CHUNK_SIZE        # 500\nCHUNK_END = min(CHUNK_START + CHUNK_SIZE, TOTAL)\n\nSUB_START = CHUNK_START + (SUB_CHUNK_ID * SUB_CHUNK_SIZE)  # 750\nSUB_END = min(SUB_START + SUB_CHUNK_SIZE, CHUNK_END)       # 1000\n\nchunk_df = df.iloc[SUB_START:SUB_END]\n\nprint(f\"📂 Processing CSV index {SUB_START} → {SUB_END - 1}\")\n\n# =========================\n# EXTRACTION\n# =========================\nfor _, row in tqdm(chunk_df.iterrows(), total=len(chunk_df), desc=\"Extracting ASM\"):\n    file_id = row[\"Id\"]\n    asm_path = f\"train/{file_id}.asm\"\n\n    subprocess.run(\n        [\"7z\", \"x\", ARCHIVE_PATH, asm_path, f\"-o{OUTPUT_DIR}\", \"-y\"],\n        stdout=subprocess.DEVNULL,\n        stderr=subprocess.DEVNULL\n    )\n\n# =========================\n# COMPRESS\n# =========================\nprint(\"📦 Creating 7z archive...\")\nsubprocess.run([\"7z\", \"a\", \"-t7z\", \"-y\", ARCHIVE_OUT, OUTPUT_DIR])\n\n# =========================\n# CLEANUP\n# =========================\nsubprocess.run([\"rm\", \"-rf\", OUTPUT_DIR])\n\nprint(\"✅ DONE → Lollipop_ASM_chunk_1_sub_1.7z\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-18T17:09:30.475898Z","iopub.execute_input":"2025-12-18T17:09:30.476096Z","iopub.status.idle":"2025-12-18T17:52:48.052008Z","shell.execute_reply.started":"2025-12-18T17:09:30.476077Z","shell.execute_reply":"2025-12-18T17:52:48.050614Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}