{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":4117,"databundleVersionId":46665,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport subprocess\nfrom collections import defaultdict\nfrom multiprocessing import Pool, cpu_count\nimport math\nfrom tqdm import tqdm\n\n# =========================\n# PATHS\n# =========================\nINPUT_DIR = \"/kaggle/input/malware-classification\"\nARCHIVE_PATH = os.path.join(INPUT_DIR, \"train.7z\")\nLABEL_CSV = os.path.join(INPUT_DIR, \"trainLabels.csv\")\nOUTPUT_DIR = \"/kaggle/working/asm_labeled\"\n\n# =========================\n# LABEL MAP\n# =========================\nlabel_to_family = {\n    8: \"Obfuscator.ACY\"\n}\n\n# =========================\n# PREPARE OUTPUT FOLDERS\n# =========================\nfor fam in label_to_family.values():\n    os.makedirs(os.path.join(OUTPUT_DIR, fam), exist_ok=True)\n\n# =========================\n# READ & FILTER CSV\n# =========================\ndf = pd.read_csv(LABEL_CSV)\ndf = df[df[\"Class\"].isin(label_to_family.keys())].reset_index(drop=True)\nprint(f\"📊 CSV selected samples: {len(df)}\")\n\n# =========================\n# CHUNKING\n# =========================\nnum_chunks = min(3, cpu_count())   # Safe for Kaggle\nchunk_size = math.ceil(len(df) / num_chunks)\nchunks = [df[i * chunk_size : (i + 1) * chunk_size] for i in range(num_chunks)]\n\nprint(f\"⚙️ Using {num_chunks} parallel workers\")\nfor i, c in enumerate(chunks):\n    print(f\"  • Chunk {i+1}: {len(c)} files\")\n\n# =========================\n# EXTRACTION FUNCTION\n# =========================\ndef extract_files(df_chunk):\n    extracted_ids = set()\n    family_extracted = defaultdict(int)\n    failed_ids = []\n\n    for _, row in tqdm(\n        df_chunk.iterrows(),\n        total=len(df_chunk),\n        desc=f\"Worker PID {os.getpid()}\",\n        leave=False\n    ):\n        file_id = row[\"Id\"]\n        family = label_to_family[row[\"Class\"]]\n        out_dir = os.path.join(OUTPUT_DIR, family)\n        asm_in_7z = f\"train/{file_id}.asm\"\n\n        cmd = [\n            \"7z\", \"x\",\n            ARCHIVE_PATH,\n            asm_in_7z,\n            f\"-o{out_dir}\",\n            \"-y\"\n        ]\n\n        result = subprocess.run(\n            cmd,\n            stdout=subprocess.DEVNULL,\n            stderr=subprocess.DEVNULL\n        )\n\n        if result.returncode == 0:\n            extracted_ids.add(file_id)\n            family_extracted[family] += 1\n        else:\n            failed_ids.append(file_id)\n\n    return extracted_ids, dict(family_extracted), failed_ids\n\n# =========================\n# PARALLEL EXTRACTION\n# =========================\nif __name__ == \"__main__\":\n    with Pool(num_chunks) as pool:\n        results = pool.map(extract_files, chunks)\n\n    # =========================\n    # MERGE RESULTS\n    # =========================\n    all_extracted_ids = set()\n    all_family_extracted = defaultdict(int)\n    all_failed_ids = []\n\n    for extracted_ids, family_extracted, failed_ids in results:\n        all_extracted_ids.update(extracted_ids)\n        for fam, count in family_extracted.items():\n            all_family_extracted[fam] += count\n        all_failed_ids.extend(failed_ids)\n\n    # =========================\n    # FINAL REPORT\n    # =========================\n    print(\"\\n✅ Extraction completed\")\n    print(\"📂 Extracted files per family:\")\n    for fam, cnt in all_family_extracted.items():\n        print(f\"   {fam}: {cnt}\")\n\n    print(f\"\\n❌ Failed extractions: {len(all_failed_ids)}\")\n\n    if all_failed_ids:\n        failed_path = os.path.join(OUTPUT_DIR, \"failed_ids.txt\")\n        with open(failed_path, \"w\") as f:\n            for fid in all_failed_ids:\n                f.write(f\"{fid}\\n\")\n        print(f\"📝 Failed IDs saved to: {failed_path}\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-18T11:04:08.538319Z","iopub.execute_input":"2025-12-18T11:04:08.538497Z","iopub.status.idle":"2025-12-18T12:08:46.411836Z","shell.execute_reply.started":"2025-12-18T11:04:08.538480Z","shell.execute_reply":"2025-12-18T12:08:46.410768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!7z a -tzip /kaggle/working/asm_labeled.zip /kaggle/working/asm_labeled/*","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-18T12:21:26.953765Z","iopub.execute_input":"2025-12-18T12:21:26.954104Z","iopub.status.idle":"2025-12-18T12:23:08.017472Z","shell.execute_reply.started":"2025-12-18T12:21:26.954070Z","shell.execute_reply":"2025-12-18T12:23:08.016813Z"}},"outputs":[],"execution_count":null}]}