{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":4117,"databundleVersionId":46665,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-10T01:15:44.791109Z","iopub.execute_input":"2025-09-10T01:15:44.791371Z","iopub.status.idle":"2025-09-10T01:15:47.215085Z","shell.execute_reply.started":"2025-09-10T01:15:44.791346Z","shell.execute_reply":"2025-09-10T01:15:47.213974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install py7zr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T01:25:14.147258Z","iopub.execute_input":"2025-09-10T01:25:14.147682Z","iopub.status.idle":"2025-09-10T01:25:21.823879Z","shell.execute_reply.started":"2025-09-10T01:25:14.147642Z","shell.execute_reply":"2025-09-10T01:25:21.822422Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==============================================================================\n# SCRIPT 1: DATA PREPROCESSING (CORRECTED AND FINAL VERSION)\n# ==============================================================================\nimport os\nimport gc\nimport py7zr\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nimport tempfile\nimport _lzma\nfrom sklearn.model_selection import train_test_split\nfrom joblib import Parallel, delayed\n\n# --- Configuration ---\n# ✅ CHANGE THIS VALUE TO INCREASE YOUR TRAINING DATA (e.g., 0.3 for 30%)\nSUBSET_FRACTION = 0.15\nTARGET_LENGTH = 4096\n\n# To process a small TEST subset for quick checks, set a number (e.g., 200).\n# For the FINAL submission, set this to None to process ALL test files.\nTEST_SUBSET_SIZE = 500 \n\n# --- File Paths ---\nARCHIVE_DIR = '/kaggle/input/malware-classification'\nTRAIN_ARCHIVE = os.path.join(ARCHIVE_DIR, 'train.7z')\nTEST_ARCHIVE = os.path.join(ARCHIVE_DIR, 'test.7z')\nLABELS_FILE = os.path.join(ARCHIVE_DIR, 'trainLabels.csv')\nOUTPUT_TRAIN_FILE = '/kaggle/working/processed_train_subset.npz'\nOUTPUT_TEST_FILE = '/kaggle/working/processed_test_data.npz'\n\n# --- Helper and Worker functions (no changes needed here) ---\ndef resize_sequence(sequence, target_length):\n    # ... (code is correct)\n    original_length = len(sequence)\n    if original_length == 0: return np.zeros(target_length, dtype=np.uint8)\n    original_index = np.linspace(0, original_length - 1, num=original_length)\n    target_index = np.linspace(0, original_length - 1, num=target_length)\n    resized = np.interp(target_index, original_index, sequence)\n    return np.round(resized).astype(np.uint8)\n\ndef process_file_from_memory(byte_content):\n    # ... (code is correct)\n    sequence = []\n    lines = byte_content.decode('utf-8', errors='ignore').splitlines()\n    for line in lines:\n        parts = line.strip().split()[1:]\n        hex_bytes = [int(b, 16) for b in parts if b != '??']\n        sequence.extend(hex_bytes)\n    return np.array(sequence)\n\ndef process_single_file(file_id, archive_path, labels_dict):\n    # ... (code is correct)\n    dir_prefix = \"train/\" if labels_dict else \"test/\"\n    filename = f\"{dir_prefix}{file_id}.bytes\"\n    try:\n        with py7zr.SevenZipFile(archive_path, mode='r') as z:\n            with tempfile.TemporaryDirectory() as temp_dir:\n                z.extract(targets=[filename], path=temp_dir)\n                temp_file_path = os.path.join(temp_dir, filename)\n                with open(temp_file_path, 'rb') as f:\n                    byte_content = f.read()\n        raw_sequence = process_file_from_memory(byte_content)\n        resized = resize_sequence(raw_sequence, TARGET_LENGTH)\n        info = labels_dict.get(file_id) if labels_dict else file_id\n        return resized, info\n    except (py7zr.exceptions.CrcError, _lzma.LZMAError, KeyError, FileNotFoundError) as e:\n        return None\n\n# --- MAIN LOGIC (Using joblib) ---\nif __name__ == '__main__':\n    # --- 1. Process Training Data Subset ---\n    print(\"--- Starting Training Set Preprocessing ---\")\n    labels_df = pd.read_csv(LABELS_FILE)\n    _, subset_df = train_test_split(labels_df, test_size=SUBSET_FRACTION, stratify=labels_df['Class'], random_state=42)\n    subset_file_ids = subset_df['Id'].tolist()\n    labels_dict = {row['Id']: row['Class'] - 1 for _, row in subset_df.iterrows()}\n    \n    num_cores = os.cpu_count()\n    print(f\"Starting parallel processing on training data with {num_cores} cores...\")\n    \n    results = Parallel(n_jobs=num_cores)(\n        delayed(process_single_file)(file_id, TRAIN_ARCHIVE, labels_dict) for file_id in tqdm(subset_file_ids)\n    )\n    \n    successful_results = [r for r in results if r is not None]\n    X_train_sub, y_train_sub = zip(*successful_results)\n    \n    print(f\"\\nSaving training subset to '{OUTPUT_TRAIN_FILE}'...\")\n    np.savez_compressed(OUTPUT_TRAIN_FILE, X=np.array(X_train_sub), y=np.array(y_train_sub))\n    del X_train_sub, y_train_sub, successful_results\n    gc.collect()\n\n    # --- 2. Process Test Data (Corrected and Unified) ---\n    print(\"\\n--- Starting Test Set Preprocessing ---\")\n    sample_submission_df = pd.read_csv(os.path.join(ARCHIVE_DIR, 'sampleSubmission.csv'))\n    \n    if TEST_SUBSET_SIZE is not None:\n        print(f\"Processing a random subset of {TEST_SUBSET_SIZE} test files.\")\n        test_df = sample_submission_df.sample(n=TEST_SUBSET_SIZE, random_state=42)\n        test_file_ids = test_df['Id'].tolist()\n    else:\n        print(\"Processing the FULL test set.\")\n        test_file_ids = sample_submission_df['Id'].tolist()\n    \n    results = Parallel(n_jobs=num_cores)(\n        delayed(process_single_file)(file_id, TEST_ARCHIVE, None) for file_id in tqdm(test_file_ids)\n    )\n\n    successful_results = [r for r in results if r is not None]\n    X_test, test_ids_processed = zip(*successful_results)\n\n    print(f\"\\nSaving test set to '{OUTPUT_TEST_FILE}'...\")\n    np.savez_compressed(OUTPUT_TEST_FILE, X=np.array(X_test), ids=np.array(test_ids_processed))\n    \n    print(\"\\nPreprocessing complete! 🎉\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-10T01:26:37.578659Z","iopub.execute_input":"2025-09-10T01:26:37.578997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}