{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":4117,"databundleVersionId":46665,"sourceType":"competition"},{"sourceId":12374624,"sourceType":"datasetVersion","datasetId":7802589}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-12T14:45:09.574829Z","iopub.execute_input":"2025-07-12T14:45:09.575205Z","iopub.status.idle":"2025-07-12T14:45:10.933321Z","shell.execute_reply.started":"2025-07-12T14:45:09.575181Z","shell.execute_reply":"2025-07-12T14:45:10.932484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Chunk 1: Library Imports and Initial Setup\n\nimport pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score, roc_curve, confusion_matrix, classification_report, f1_score\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport gc # For garbage collection\nfrom tqdm.notebook import tqdm # For progress bars with iterators\nimport json # To read .jsonl files\nimport re # For regular expressions in .asm parsing\nfrom collections import Counter\nfrom sklearn.preprocessing import LabelEncoder, label_binarize\nfrom itertools import cycle\n\n# Set a consistent style for plots\nsns.set_style(\"whitegrid\")\nplt.rcParams['figure.dpi'] = 100 # Adjust figure resolution for better clarity\n\nprint(\"Libraries imported successfully.\")\nprint(\"Initial setup complete.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:12:05.719915Z","iopub.execute_input":"2025-07-12T15:12:05.720472Z","iopub.status.idle":"2025-07-12T15:12:05.731284Z","shell.execute_reply.started":"2025-07-12T15:12:05.720446Z","shell.execute_reply":"2025-07-12T15:12:05.730247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n--- EMBER-2018 Dataset: Loading and Initial EDA (Revised) ---\")\n\ndef load_ember_jsonl_revised(filepaths, process_limit=None):\n    \"\"\"\n    Revised function to load EMBER features from .jsonl files based on the provided structure.\n    It extracts features from 'histogram', 'byteentropy', 'strings', 'general', 'header', 'section', 'imports', 'exports'.\n    \"\"\"\n    all_features_list = []\n    ids_list = []\n    labels_list = []\n    family_ids_list = []\n    \n    # Keep track of all feature names collected to ensure consistency\n    feature_names_set = set()\n\n    print(f\"Loading {len(filepaths)} EMBER .jsonl files...\")\n    num_processed = 0\n    for fp in filepaths:\n        print(f\"Processing file: {os.path.basename(fp)}\")\n        with open(fp, 'r') as f:\n            for line_num, line in enumerate(tqdm(f, desc=f\"Reading {os.path.basename(fp)}\")):\n                if process_limit is not None and num_processed >= process_limit:\n                    print(f\"Reached processing limit of {process_limit} samples for EMBER. Stopping.\")\n                    break # Stop processing this file\n                try:\n                    entry = json.loads(line)\n                    \n                    current_sample_features = {}\n\n                    # Basic metadata\n                    ids_list.append(entry.get('sha256', entry.get('id')))\n                    labels_list.append(entry.get('label', -1))\n                    family_ids_list.append(entry.get('family_id', -1)) # family_id might not always be present\n\n                    # --- Extract Numerical Features ---\n\n                    # 1. From 'histogram'\n                    histogram = entry.get('histogram', [])\n                    for i, val in enumerate(histogram):\n                        current_sample_features[f'hist_{i}'] = val\n                    \n                    # 2. From 'byteentropy'\n                    byteentropy = entry.get('byteentropy', [])\n                    for i, val in enumerate(byteentropy):\n                        current_sample_features[f'byteentropy_{i}'] = val\n\n                    # 3. From 'strings' (nested dictionary)\n                    strings = entry.get('strings', {})\n                    current_sample_features['str_numstrings'] = strings.get('numstrings', 0)\n                    current_sample_features['str_avlength'] = strings.get('avlength', 0.0)\n                    current_sample_features['str_printables'] = strings.get('printables', 0)\n                    current_sample_features['str_entropy'] = strings.get('entropy', 0.0)\n                    current_sample_features['str_paths'] = strings.get('paths', 0)\n                    current_sample_features['str_urls'] = strings.get('urls', 0)\n                    current_sample_features['str_registry'] = strings.get('registry', 0)\n                    current_sample_features['str_MZ'] = strings.get('MZ', 0)\n                    # Note: 'printabledist' is a list, can be added if needed, but increases dimensionality\n                    # For now, let's omit it to keep feature count manageable.\n\n                    # 4. From 'general'\n                    general = entry.get('general', {})\n                    current_sample_features['gen_size'] = general.get('size', 0)\n                    current_sample_features['gen_vsize'] = general.get('vsize', 0)\n                    current_sample_features['gen_has_debug'] = general.get('has_debug', 0)\n                    current_sample_features['gen_exports'] = general.get('exports', 0)\n                    current_sample_features['gen_imports_count'] = general.get('imports', 0) # Renamed to avoid clash with 'imports' dict\n                    current_sample_features['gen_has_relocations'] = general.get('has_relocations', 0)\n                    current_sample_features['gen_has_resources'] = general.get('has_resources', 0)\n                    current_sample_features['gen_has_signature'] = general.get('has_signature', 0)\n                    current_sample_features['gen_has_tls'] = general.get('has_tls', 0)\n                    current_sample_features['gen_symbols'] = general.get('symbols', 0)\n\n                    # 5. From 'header' -> 'coff' and 'optional'\n                    header = entry.get('header', {})\n                    coff = header.get('coff', {})\n                    current_sample_features['hdr_coff_timestamp'] = coff.get('timestamp', 0)\n                    # 'machine' and 'characteristics' are categorical, might need one-hot encoding if used directly.\n                    # For simplicity, we might skip them or count specific characteristics for now.\n                    # E.g., count total characteristics:\n                    current_sample_features['hdr_coff_char_count'] = len(coff.get('characteristics', []))\n\n                    optional = header.get('optional', {})\n                    current_sample_features['hdr_opt_major_image_version'] = optional.get('major_image_version', 0)\n                    current_sample_features['hdr_opt_minor_image_version'] = optional.get('minor_image_version', 0)\n                    current_sample_features['hdr_opt_major_linker_version'] = optional.get('major_linker_version', 0)\n                    current_sample_features['hdr_opt_minor_linker_version'] = optional.get('minor_linker_version', 0)\n                    current_sample_features['hdr_opt_major_os_version'] = optional.get('major_operating_system_version', 0)\n                    current_sample_features['hdr_opt_minor_os_version'] = optional.get('minor_operating_system_version', 0)\n                    current_sample_features['hdr_opt_major_subsystem_version'] = optional.get('major_subsystem_version', 0)\n                    current_sample_features['hdr_opt_minor_subsystem_version'] = optional.get('minor_subsystem_version', 0)\n                    current_sample_features['hdr_opt_sizeof_code'] = optional.get('sizeof_code', 0)\n                    current_sample_features['hdr_opt_sizeof_headers'] = optional.get('sizeof_headers', 0)\n                    current_sample_features['hdr_opt_sizeof_heap_commit'] = optional.get('sizeof_heap_commit', 0)\n                    # 'subsystem', 'dll_characteristics', 'magic' are categorical\n\n                    # 6. From 'section'\n                    section = entry.get('section', {})\n                    sections_list = section.get('sections', [])\n                    current_sample_features['sec_count'] = len(sections_list)\n                    # Aggregate features from each section\n                    total_sec_size = 0\n                    total_sec_vsize = 0\n                    total_sec_entropy = 0\n                    executable_sections = 0\n                    writable_sections = 0\n                    for s in sections_list:\n                        total_sec_size += s.get('size', 0)\n                        total_sec_vsize += s.get('vsize', 0)\n                        total_sec_entropy += s.get('entropy', 0.0)\n                        props = s.get('props', [])\n                        if 'MEM_EXECUTE' in props:\n                            executable_sections += 1\n                        if 'MEM_WRITE' in props:\n                            writable_sections += 1\n                    current_sample_features['sec_total_size'] = total_sec_size\n                    current_sample_features['sec_total_vsize'] = total_sec_vsize\n                    current_sample_features['sec_avg_entropy'] = total_sec_entropy / current_sample_features['sec_count'] if current_sample_features['sec_count'] > 0 else 0.0\n                    current_sample_features['sec_executable_count'] = executable_sections\n                    current_sample_features['sec_writable_count'] = writable_sections\n\n                    # 7. From 'imports' and 'exports' (count unique DLLs/APIs)\n                    imports = entry.get('imports', {})\n                    current_sample_features['imp_dll_count'] = len(imports) # Number of unique DLLs imported\n                    total_imported_funcs = 0\n                    for dll, funcs in imports.items():\n                        total_imported_funcs += len(funcs)\n                    current_sample_features['imp_func_count'] = total_imported_funcs\n\n                    exports = entry.get('exports', [])\n                    current_sample_features['exp_count'] = len(exports) # Number of functions exported\n\n                    # Add the extracted features for this sample to the list\n                    all_features_list.append(current_sample_features)\n                    num_processed += 1\n                    \n                except json.JSONDecodeError as e:\n                    print(f\"Skipping malformed JSON line {line_num+1} in {os.path.basename(fp)}: {e}\")\n                except Exception as e: # Catch any other unexpected errors during parsing\n                    print(f\"Skipping line {line_num+1} in {os.path.basename(fp)} due to unexpected error: {e}\")\n            if process_limit is not None and num_processed >= process_limit:\n                break # Stop processing further files","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:12:09.834060Z","iopub.execute_input":"2025-07-12T15:12:09.834965Z","iopub.status.idle":"2025-07-12T15:12:09.857237Z","shell.execute_reply.started":"2025-07-12T15:12:09.834939Z","shell.execute_reply":"2025-07-12T15:12:09.856108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# load_ember_jsonl_revised Function Definition\n\ndef load_ember_jsonl_revised(filepaths, process_limit=None):\n    \"\"\"\n    Revised function to load EMBER features from .jsonl files based on the provided structure.\n    It extracts features from 'histogram', 'byteentropy', 'strings', 'general', 'header', 'section', 'imports', 'exports'.\n    \"\"\"\n    all_features_list = [] # This list will hold dictionaries of features for each sample\n    ids_list = []\n    labels_list = []\n    family_ids_list = []\n    \n    print(f\"Loading {len(filepaths)} EMBER .jsonl files...\")\n    num_processed = 0\n    for fp in filepaths:\n        print(f\"Processing file: {os.path.basename(fp)}\")\n        with open(fp, 'r') as f:\n            for line_num, line in enumerate(tqdm(f, desc=f\"Reading {os.path.basename(fp)}\")):\n                if process_limit is not None and num_processed >= process_limit:\n                    print(f\"Reached processing limit of {process_limit} samples for EMBER. Stopping.\")\n                    break # Stop processing this file and subsequent files\n                try:\n                    entry = json.loads(line)\n                    \n                    current_sample_features = {} # Dictionary to hold features for the current sample\n\n                    # Basic metadata\n                    ids_list.append(entry.get('sha256', entry.get('id')))\n                    labels_list.append(entry.get('label', -1))\n                    family_ids_list.append(entry.get('family_id', -1)) # family_id might not always be present\n\n                    # --- Extract Numerical Features ---\n\n                    # 1. From 'histogram' (list of 256 values)\n                    histogram = entry.get('histogram', [])\n                    for i, val in enumerate(histogram):\n                        current_sample_features[f'hist_{i}'] = val\n                    \n                    # Ensure histogram has 256 features, fill with 0 if shorter\n                    for i in range(len(histogram), 256):\n                        current_sample_features[f'hist_{i}'] = 0\n\n                    # 2. From 'byteentropy' (list of 256 values)\n                    byteentropy = entry.get('byteentropy', [])\n                    for i, val in enumerate(byteentropy):\n                        current_sample_features[f'byteentropy_{i}'] = val\n                    \n                    # Ensure byteentropy has 256 features, fill with 0 if shorter\n                    for i in range(len(byteentropy), 256):\n                        current_sample_features[f'byteentropy_{i}'] = 0\n\n                    # 3. From 'strings' (nested dictionary)\n                    strings = entry.get('strings', {})\n                    current_sample_features['str_numstrings'] = strings.get('numstrings', 0)\n                    current_sample_features['str_avlength'] = strings.get('avlength', 0.0)\n                    current_sample_features['str_printables'] = strings.get('printables', 0)\n                    current_sample_features['str_entropy'] = strings.get('entropy', 0.0)\n                    current_sample_features['str_paths'] = strings.get('paths', 0)\n                    current_sample_features['str_urls'] = strings.get('urls', 0)\n                    current_sample_features['str_registry'] = strings.get('registry', 0)\n                    current_sample_features['str_MZ'] = strings.get('MZ', 0)\n                    \n                    printabledist = strings.get('printabledist', [])\n                    if printabledist:\n                        current_sample_features['str_printabledist_mean'] = np.mean(printabledist)\n                        current_sample_features['str_printabledist_std'] = np.std(printabledist)\n                    else:\n                        current_sample_features['str_printabledist_mean'] = 0.0\n                        current_sample_features['str_printabledist_std'] = 0.0\n\n\n                    # 4. From 'general'\n                    general = entry.get('general', {})\n                    current_sample_features['gen_size'] = general.get('size', 0)\n                    current_sample_features['gen_vsize'] = general.get('vsize', 0)\n                    current_sample_features['gen_has_debug'] = general.get('has_debug', 0)\n                    current_sample_features['gen_exports'] = general.get('exports', 0)\n                    current_sample_features['gen_imports_count'] = general.get('imports', 0) # Renamed to avoid clash with 'imports' dict\n                    current_sample_features['gen_has_relocations'] = general.get('has_relocations', 0)\n                    current_sample_features['gen_has_resources'] = general.get('has_resources', 0)\n                    current_sample_features['gen_has_signature'] = general.get('has_signature', 0)\n                    current_sample_features['gen_has_tls'] = general.get('has_tls', 0)\n                    current_sample_features['gen_symbols'] = general.get('symbols', 0)\n\n                    # 5. From 'header' -> 'coff' and 'optional'\n                    header = entry.get('header', {})\n                    coff = header.get('coff', {})\n                    current_sample_features['hdr_coff_timestamp'] = coff.get('timestamp', 0)\n                    current_sample_features['hdr_coff_char_count'] = len(coff.get('characteristics', []))\n\n                    optional = header.get('optional', {})\n                    current_sample_features['hdr_opt_major_image_version'] = optional.get('major_image_version', 0)\n                    current_sample_features['hdr_opt_minor_image_version'] = optional.get('minor_image_version', 0)\n                    current_sample_features['hdr_opt_major_linker_version'] = optional.get('major_linker_version', 0)\n                    current_sample_features['hdr_opt_minor_linker_version'] = optional.get('minor_linker_version', 0)\n                    current_sample_features['hdr_opt_major_os_version'] = optional.get('major_operating_system_version', 0)\n                    current_sample_features['hdr_opt_minor_os_version'] = optional.get('minor_operating_system_version', 0)\n                    current_sample_features['hdr_opt_major_subsystem_version'] = optional.get('major_subsystem_version', 0)\n                    current_sample_features['hdr_opt_minor_subsystem_version'] = optional.get('minor_subsystem_version', 0)\n                    current_sample_features['hdr_opt_sizeof_code'] = optional.get('sizeof_code', 0)\n                    current_sample_features['hdr_opt_sizeof_headers'] = optional.get('sizeof_headers', 0)\n                    current_sample_features['hdr_opt_sizeof_heap_commit'] = optional.get('sizeof_heap_commit', 0)\n\n                    # 6. From 'section'\n                    section = entry.get('section', {})\n                    sections_list = section.get('sections', [])\n                    current_sample_features['sec_count'] = len(sections_list)\n                    # Aggregate features from each section\n                    total_sec_size = 0\n                    total_sec_vsize = 0\n                    total_sec_entropy = 0\n                    executable_sections = 0\n                    writable_sections = 0\n                    for s in sections_list:\n                        total_sec_size += s.get('size', 0)\n                        total_sec_vsize += s.get('vsize', 0)\n                        total_sec_entropy += s.get('entropy', 0.0)\n                        props = s.get('props', [])\n                        if 'MEM_EXECUTE' in props:\n                            executable_sections += 1\n                        if 'MEM_WRITE' in props:\n                            writable_sections += 1\n                    current_sample_features['sec_total_size'] = total_sec_size\n                    current_sample_features['sec_total_vsize'] = total_sec_vsize\n                    current_sample_features['sec_avg_entropy'] = total_sec_entropy / current_sample_features['sec_count'] if current_sample_features['sec_count'] > 0 else 0.0\n                    current_sample_features['sec_executable_count'] = executable_sections\n                    current_sample_features['sec_writable_count'] = writable_sections\n\n                    # 7. From 'imports' and 'exports' (count unique DLLs/APIs)\n                    imports = entry.get('imports', {})\n                    current_sample_features['imp_dll_count'] = len(imports) # Number of unique DLLs imported\n                    total_imported_funcs = 0\n                    for dll, funcs in imports.items():\n                        total_imported_funcs += len(funcs)\n                    current_sample_features['imp_func_count'] = total_imported_funcs\n\n                    exports = entry.get('exports', [])\n                    current_sample_features['exp_count'] = len(exports) # Number of functions exported\n\n                    # Add the extracted features for this sample to the list\n                    all_features_list.append(current_sample_features)\n                    num_processed += 1\n                    \n                except json.JSONDecodeError as e:\n                    print(f\"Skipping malformed JSON line {line_num+1} in {os.path.basename(fp)}: {e}\")\n                except Exception as e: # Catch any other unexpected errors during parsing\n                    print(f\"Skipping line {line_num+1} in {os.path.basename(fp)} due to unexpected error: {e}\")\n            if process_limit is not None and num_processed >= process_limit:\n                break # Stop processing further files\n\n    # Convert list of dictionaries to DataFrame, handling varying keys by filling missing with 0\n    features_df = pd.DataFrame(all_features_list)\n    features_df = features_df.fillna(0) # Fill NaN values with 0 where some features might be missing for a sample\n\n    # Convert features to a NumPy array for LightGBM\n    X_features = features_df.to_numpy(dtype=np.float32)\n    \n    # Create a DataFrame for metadata (ID, label, family_id)\n    metadata_df = pd.DataFrame({\n        'id': ids_list,\n        'label': labels_list,\n        'family_id': family_ids_list\n    })\n    \n    return X_features, metadata_df, list(features_df.columns) # Return feature names too\n\nprint(\"`load_ember_jsonl_revised` function defined.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:12:21.886233Z","iopub.execute_input":"2025-07-12T15:12:21.886568Z","iopub.status.idle":"2025-07-12T15:12:21.912326Z","shell.execute_reply.started":"2025-07-12T15:12:21.886546Z","shell.execute_reply":"2025-07-12T15:12:21.910865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# EMBER Data Loading and Initial EDA\n\nprint(\"\\n--- EMBER-2018 Dataset: Loading and Initial EDA ---\")\n\n# Define the paths to your EMBER .jsonl files\nember_train_filepaths = [\n    '/kaggle/input/multistage-malware-detection-and-classification/ember/train_features_0.jsonl',\n    '/kaggle/input/multistage-malware-detection-and-classification/ember/train_features_1.jsonl',\n    '/kaggle/input/multistage-malware-detection-and-classification/ember/train_features_2.jsonl',\n    '/kaggle/input/multistage-malware-detection-and-classification/ember/train_features_3.jsonl',\n    '/kaggle/input/multistage-malware-detection-and-classification/ember/train_features_4.jsonl',\n    '/kaggle/input/multistage-malware-detection-and-classification/ember/train_features_5.jsonl'\n]\nember_test_filepath = '/kaggle/input/multistage-malware-detection-and-classification/ember/test_features.jsonl'\n\n# !!! IMPORTANT: Adjust this limit !!!\n# Set a limit for processing samples. Start small (e.g., 10000) for testing.\n# Set to None to load the full dataset once you confirm it works.\nprocess_limit_ember = 100000 # For faster testing and to prevent crashes\n\ntry:\n    X_full_ember, metadata_full_ember, ember_feature_names = load_ember_jsonl_revised(\n        ember_train_filepaths + [ember_test_filepath],\n        process_limit=process_limit_ember\n    )\n    print(\"\\nEMBER-2018 dataset loaded successfully with revised parser.\")\n    print(f\"Total samples loaded: {len(metadata_full_ember)}\")\n    print(f\"Features shape: {X_full_ember.shape}\")\n    print(f\"Metadata head:\\n{metadata_full_ember.head()}\")\n    print(f\"Number of extracted EMBER features: {len(ember_feature_names)}\")\n\nexcept Exception as e:\n    print(f\"Critical Error loading EMBER-2018 dataset: {e}. Please check the data format or path.\")\n    # If data loading fails critically, you might want to stop here to debug\n    raise # Re-raise the exception to stop execution and show the error\n\n# Distribution of labels\nplt.figure(figsize=(7, 5))\nsns.countplot(x='label', data=metadata_full_ember, palette='viridis')\nplt.title('Distribution of Labels in Raw EMBER-2018 Dataset')\nplt.xlabel('Label (0: Benign, 1: Malware, -1: Unlabeled)')\nplt.ylabel('Count')\nplt.show()\n\n# Drop unlabeled samples for malware identification training\ninitial_samples = len(metadata_full_ember)\nmetadata_filtered_ember = metadata_full_ember[metadata_full_ember['label'] != -1].copy()\n# Filter X_full_ember using the same boolean mask\nX_filtered_ember = X_full_ember[metadata_full_ember['label'] != -1]\n\nprint(f\"Samples before filtering unlabeled: {initial_samples}\")\nprint(f\"Samples after removing unlabeled data: {len(metadata_filtered_ember)}\")\n\n# Verify the distribution after filtering\nplt.figure(figsize=(7, 5))\nsns.countplot(x='label', data=metadata_filtered_ember, palette='viridis')\nplt.title('Distribution of Malware (1) and Benign (0) Samples (Filtered)')\nplt.xlabel('Label (0: Benign, 1: Malware)')\nplt.ylabel('Count')\nplt.show()\n\n# Check for any NaN values in features\nif np.isnan(X_filtered_ember).any():\n    print(\"Warning: NaN values found in EMBER features. LightGBM can handle them, but consider imputation if they are significant.\")\nelse:\n    print(\"No NaN values found in EMBER features (as expected).\")\n\n# Display a small part of the feature array to understand structure (first 5 rows, first 10 columns)\nprint(\"\\nFirst 5 rows and first 10 columns of EMBER features (from revised parser):\")\nprint(pd.DataFrame(X_filtered_ember[:5, :10], columns=ember_feature_names[:10]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:12:29.788530Z","iopub.execute_input":"2025-07-12T15:12:29.788854Z","iopub.status.idle":"2025-07-12T15:13:46.432032Z","shell.execute_reply.started":"2025-07-12T15:12:29.788833Z","shell.execute_reply":"2025-07-12T15:13:46.431068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Chunk 4: EMBER Data Preparation for Model Training\n\nprint(\"\\nPreparing data for EMBER-2018 Malware Identification model...\")\n\nX_ember_id = X_filtered_ember\ny_ember_id = metadata_filtered_ember['label']\n\nX_train_ember_id, X_test_ember_id, y_train_ember_id, y_test_ember_id = train_test_split(\n    X_ember_id, y_ember_id, test_size=0.2, random_state=42, stratify=y_ember_id\n)\n\nprint(f\"X_train_ember_id shape: {X_train_ember_id.shape}\")\nprint(f\"X_test_ember_id shape: {X_test_ember_id.shape}\")\nprint(f\"y_train_ember_id shape: {y_train_ember_id.shape}\")\nprint(f\"y_test_ember_id shape: {y_test_ember_id.shape}\")\n\n# Clear full dataframes to save memory\ndel X_full_ember, metadata_full_ember, X_filtered_ember, metadata_filtered_ember, X_ember_id, y_ember_id\ngc.collect()\nprint(\"Memory cleaned after data preparation.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:13:51.943876Z","iopub.execute_input":"2025-07-12T15:13:51.944259Z","iopub.status.idle":"2025-07-12T15:13:52.279804Z","shell.execute_reply.started":"2025-07-12T15:13:51.944233Z","shell.execute_reply":"2025-07-12T15:13:52.278734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc # Import garbage collection module\n\nprint(f\"Shape of X_train_ember_id: {X_train_ember_id.shape}\")\nprint(f\"Shape of y_train_ember_id: {y_train_ember_id.shape}\")\nprint(f\"Shape of X_test_ember_id (validation): {X_test_ember_id.shape}\")\nprint(f\"Shape of y_test_ember_id (validation): {y_test_ember_id.shape}\")\n\n# Aggressively clear memory from the full filtered dataset now that train/test/validation sets are created\nif 'X_full_ember' in locals():\n    del X_full_ember\nif 'metadata_full_ember' in locals():\n    del metadata_full_ember\nif 'X_filtered_ember' in locals():\n    del X_filtered_ember\nif 'metadata_filtered_ember' in locals():\n    del metadata_filtered_ember\ngc.collect() # Force garbage collection\nprint(\"Memory for original full and filtered training data cleared.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:13:57.607363Z","iopub.execute_input":"2025-07-12T15:13:57.608192Z","iopub.status.idle":"2025-07-12T15:13:57.749013Z","shell.execute_reply.started":"2025-07-12T15:13:57.608170Z","shell.execute_reply":"2025-07-12T15:13:57.748146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport gc\nimport os\nfrom tqdm.notebook import tqdm\nimport json # Ensure json is imported if not already globally accessible\n\n# Assuming load_ember_jsonl_revised function is defined from Chunk 1/2.\n# If you don't have this function defined in your current kernel,\n# you will need to re-run the cell where 'load_ember_jsonl_revised' is defined.\n\nprint(\"\\n--- Loading Dedicated EMBER Test Dataset (200k samples) ---\")\n\nember_test_filepath = '/kaggle/input/multistage-malware-detection-and-classification/ember/test_features.jsonl'\n\ntry:\n    # Load the test data specifically, without any process limit, for the final evaluation\n    # This will create the X_test_filtered and y_true_test variables your prediction code expects.\n    X_test_raw_dedicated, metadata_test_raw_dedicated, _ = load_ember_jsonl_revised(\n        [ember_test_filepath], # Pass as a list, even for a single file\n        process_limit=None # Load ALL samples from the dedicated test file (200k)\n    )\n    print(f\"Raw Dedicated Test data loaded. Shape: {X_test_raw_dedicated.shape}, Metadata: {len(metadata_test_raw_dedicated)}\")\n\n    # Filter out unlabeled samples if any (EMBER test set should be fully labeled, but good practice)\n    metadata_test_filtered_dedicated = metadata_test_raw_dedicated[metadata_test_raw_dedicated['label'] != -1].copy()\n    X_test_filtered = X_test_raw_dedicated[metadata_test_raw_dedicated['label'] != -1]\n    y_true_test = metadata_test_filtered_dedicated['label']\n\n    print(f\"Samples after unlabeled filtering in dedicated test set: {len(metadata_test_filtered_dedicated)}\")\n    print(f\"Shape of X_test_filtered (dedicated test set): {X_test_filtered.shape}\")\n    print(f\"Shape of y_true_test (dedicated test set): {y_true_test.shape}\")\n\n    # Aggressively clear the raw test data after creating the filtered versions\n    del X_test_raw_dedicated\n    del metadata_test_raw_dedicated\n    del metadata_test_filtered_dedicated\n    gc.collect()\n    print(\"Memory for raw dedicated test data cleared.\")\n\nexcept Exception as e:\n    print(f\"Error loading dedicated test data: {e}\")\n    print(\"The kernel likely restarted during this step due to memory limitations, or the file path is incorrect.\")\n    # Consider reducing the test set size or preprocessing it to .npy/.parquet if this repeatedly fails.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:14:01.477761Z","iopub.execute_input":"2025-07-12T15:14:01.478108Z","iopub.status.idle":"2025-07-12T15:16:41.060674Z","shell.execute_reply.started":"2025-07-12T15:14:01.478061Z","shell.execute_reply":"2025-07-12T15:16:41.057899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Confirmation after loading dedicated test set\nprint(f\"Is X_test_filtered defined? {'X_test_filtered' in locals()}\")\nif 'X_test_filtered' in locals():\n    print(f\"Shape of X_test_filtered (dedicated test set): {X_test_filtered.shape}\")\n    print(f\"Shape of y_true_test (dedicated test set): {y_true_test.shape}\")\nelse:\n    print(\"X_test_filtered is still not defined. This indicates a memory issue or an error during the loading of the 200k test set.\")\n\nimport gc\ngc.collect() # Final memory clear before model prediction","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:16:54.400446Z","iopub.execute_input":"2025-07-12T15:16:54.400748Z","iopub.status.idle":"2025-07-12T15:16:54.556214Z","shell.execute_reply.started":"2025-07-12T15:16:54.400728Z","shell.execute_reply":"2025-07-12T15:16:54.554651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Part of Chunk 4, or a new mini-chunk after it, before Chunk 5\n\n# Calculate class counts for the training set\nbenign_count = (y_train_ember_id == 0).sum()\nmalware_count = (y_train_ember_id == 1).sum()\n\n# Calculate scale_pos_weight\n# Avoid division by zero if for some reason malware_count is 0 (unlikely with sufficient data)\nif malware_count > 0:\n    scale_pos_weight_value = benign_count / malware_count\nelse:\n    scale_pos_weight_value = 1 # Or handle as an error if no malware in training set\n    print(\"Warning: No malware samples in training set. scale_pos_weight set to 1.\")\n\nprint(f\"\\nTraining set class distribution: Benign={benign_count}, Malware={malware_count}\")\nprint(f\"Calculated scale_pos_weight: {scale_pos_weight_value:.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:16:58.417853Z","iopub.execute_input":"2025-07-12T15:16:58.418185Z","iopub.status.idle":"2025-07-12T15:16:58.427237Z","shell.execute_reply.started":"2025-07-12T15:16:58.418162Z","shell.execute_reply":"2025-07-12T15:16:58.425781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Chunk 5: LightGBM Model Training (EMBER Identification)\n\nprint(\"\\nTraining LightGBM model for malware identification...\")\n\nlgb_train_id = lgb.Dataset(X_train_ember_id, y_train_ember_id)\nlgb_eval_id = lgb.Dataset(X_test_ember_id, y_test_ember_id, reference=lgb_train_id)\n\nparams_id = {\n    'objective': 'binary',\n    'metric': 'auc',\n    'boosting_type': 'gbdt',\n    'num_leaves': 31,\n    'learning_rate': 0.05,\n    'feature_fraction': 0.9,\n    'verbose': -1,\n    'n_jobs': -1,\n    'seed': 42,\n    'zero_as_missing': True,\n    'scale_pos_weight': scale_pos_weight_value # <-- ADD THIS LINE\n}\n\n# Train the model with early stopping\nmodel_ember_id = lgb.train(\n    params_id,\n    lgb_train_id,\n    num_boost_round=1000,             # Max number of boosting rounds\n    valid_sets=lgb_eval_id,\n    callbacks=[lgb.early_stopping(100, verbose=False)], # Stop if no improvement for 100 rounds\n)\n\nprint(\"\\nLightGBM model training for malware identification complete.\")\nprint(f\"Best iteration: {model_ember_id.best_iteration}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:01:55.045634Z","iopub.execute_input":"2025-07-12T15:01:55.045959Z","iopub.status.idle":"2025-07-12T15:04:00.622025Z","shell.execute_reply.started":"2025-07-12T15:01:55.045937Z","shell.execute_reply":"2025-07-12T15:04:00.620599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Chunk 6: LightGBM Model Evaluation and Feature Importance (UPDATED)\n\nprint(\"\\nEvaluating EMBER-2018 Malware Identification model...\")\n\n# Predict probabilities and classes\ny_pred_proba_ember_id = model_ember_id.predict(X_test_ember_id, num_iteration=model_ember_id.best_iteration)\ny_pred_ember_id = (y_pred_proba_ember_id > 0.5).astype(int) # Convert probabilities to binary predictions\n\n# ROC AUC Score\nroc_auc = roc_auc_score(y_test_ember_id, y_pred_proba_ember_id)\nprint(f\"ROC AUC Score (EMBER-2018 Malware ID): {roc_auc:.4f}\")\n\n# Classification Report\nprint(\"\\nClassification Report (EMBER-2018 Malware ID):\")\nprint(classification_report(y_test_ember_id, y_pred_ember_id, target_names=['Benign', 'Malware'], zero_division=0))\n\n# Confusion Matrix\ncm_id = confusion_matrix(y_test_ember_id, y_pred_ember_id)\nplt.figure(figsize=(6, 5))\nsns.heatmap(cm_id, annot=True, fmt='d', cmap='Blues',\n            xticklabels=['Predicted Benign', 'Predicted Malware'],\n            yticklabels=['Actual Benign', 'Actual Malware'])\nplt.xlabel('Predicted Label')\nplt.ylabel('True Label')\nplt.title('Confusion Matrix for EMBER-2018 Malware Identification')\nplt.show()\n\n# ROC Curve\nfpr, tpr, thresholds = roc_curve(y_test_ember_id, y_pred_proba_ember_id)\nplt.figure(figsize=(7, 6))\nplt.plot(fpr, tpr, color='darkorange', lw=2, label=f'ROC curve (area = {roc_auc:.2f})')\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--') # Random classifier line\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic (ROC) - EMBER-2018 Malware Identification')\nplt.legend(loc=\"lower right\")\nplt.grid(True)\nplt.show()\n\n# Feature Importance\nprint(\"\\nFeature Importance (Top 20) for EMBER-2018 Malware Identification model:\")\nfeature_importances_id = pd.DataFrame({\n    'feature': ember_feature_names,\n    'importance': model_ember_id.feature_importance()\n}).sort_values(by='importance', ascending=False)\n\nplt.figure(figsize=(10, 8))\nsns.barplot(x='importance', y='feature', data=feature_importances_id.head(20), palette='viridis')\nplt.title('Top 20 Feature Importances - EMBER-2018 Malware Identification')\nplt.xlabel('Importance (Gain)')\nplt.ylabel('Feature Name')\nplt.tight_layout()\nplt.show()\n\n# Clean up EMBER data to free up memory for the next part\n# IMPORTANT: Removed 'model_ember_id' from the deletion list\ndel X_train_ember_id, X_test_ember_id, y_train_ember_id, y_test_ember_id, lgb_train_id, lgb_eval_id, feature_importances_id\ngc.collect()\nprint(\"Memory cleaned after EMBER-2018 model evaluation (keeping model_ember_id).\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:05:14.053051Z","iopub.execute_input":"2025-07-12T15:05:14.053614Z","iopub.status.idle":"2025-07-12T15:05:16.002471Z","shell.execute_reply.started":"2025-07-12T15:05:14.053589Z","shell.execute_reply":"2025-07-12T15:05:16.001514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Chunk 7: Load and Evaluate on the Dedicated Test Dataset (UPDATED)\n\nprint(\"\\n--- Testing the model on the dedicated EMBER Test Dataset ---\")\n\n# Define the path to the test features file\nember_test_filepath_only = [\n    '/kaggle/input/multistage-malware-detection-and-classification/ember/test_features.jsonl'\n]\n\n# Load ONLY the test dataset. Set process_limit to None to load all test samples.\ntry:\n    # We don't need the feature names from the test load, so we use _ for the third return value\n    X_test_ember_full, metadata_test_ember_full, _ = load_ember_jsonl_revised(\n        ember_test_filepath_only,\n        process_limit=None # Load all samples from the test file\n    )\n    print(\"\\nDedicated EMBER test dataset loaded successfully.\")\n    print(f\"Total test samples loaded: {len(metadata_test_ember_full)}\")\n    print(f\"Test features shape: {X_test_ember_full.shape}\")\n\nexcept Exception as e:\n    print(f\"Critical Error loading dedicated EMBER test dataset: {e}. Cannot proceed with final testing.\")\n    raise # Re-raise the exception to stop execution if test data loading fails\n\n# Filter out unlabeled samples (-1) from the test set, if any\ninitial_test_samples = len(metadata_test_ember_full)\nmetadata_test_filtered = metadata_test_ember_full[metadata_test_ember_full['label'] != -1].copy()\nX_test_filtered = X_test_ember_full[metadata_test_ember_full['label'] != -1]\n\ny_true_test = metadata_test_filtered['label']\n\nprint(f\"Test samples before filtering unlabeled: {initial_test_samples}\")\nprint(f\"Test samples after removing unlabeled data: {len(metadata_test_filtered)}\")\nprint(f\"X_test_filtered shape for prediction: {X_test_filtered.shape}\")\n\n# Ensure the columns/features match between training and test (Crucial!)\n# We use len(ember_feature_names) which was retained from Chunk 3's loading\n# to check consistency with the features the model was trained on.\ntrained_feature_count = len(ember_feature_names) # Use the length of feature names from training\nif X_test_filtered.shape[1] != trained_feature_count:\n    print(f\"WARNING: Feature count mismatch between training ({trained_feature_count}) and test ({X_test_filtered.shape[1]}).\")\n    print(\"This might happen if your initial `process_limit_ember` for training was too small and missed some features that appear in the test set.\")\n    print(\"Consider re-running Chunk 3 with `process_limit_ember = None` to load all possible features from the entire dataset for consistency, then retrain (Chunks 4-6).\")\n    # For now, we proceed, but this is a serious warning for model integrity.\n\n# Make predictions on the dedicated test dataset\nprint(\"\\nMaking predictions on the dedicated test dataset...\")\n# Ensure model_ember_id is available (it should be if you ran Chunk 5 and the updated Chunk 6)\ny_pred_proba_final_test = model_ember_id.predict(X_test_filtered, num_iteration=model_ember_id.best_iteration)\ny_pred_final_test = (y_pred_proba_final_test > 0.5).astype(int)\n\n# Evaluate the model on the dedicated test dataset\nprint(\"\\n--- Final Model Evaluation on Dedicated Test Set ---\")\n\n# ROC AUC Score\nroc_auc_final = roc_auc_score(y_true_test, y_pred_proba_final_test)\nprint(f\"ROC AUC Score (Dedicated Test Set): {roc_auc_final:.4f}\")\n\n# Classification Report\nprint(\"\\nClassification Report (Dedicated Test Set):\")\nprint(classification_report(y_true_test, y_pred_final_test, target_names=['Benign', 'Malware'], zero_division=0))\n\n# Confusion Matrix\ncm_final = confusion_matrix(y_true_test, y_pred_final_test)\nplt.figure(figsize=(6, 5))\nsns.heatmap(cm_final, annot=True, fmt='d', cmap='Blues',\n            xticklabels=['Predicted Benign', 'Predicted Malware'],\n            yticklabels=['Actual Benign', 'Actual Malware'])\nplt.xlabel('Predicted Label')\nplt.ylabel('True Label')\nplt.title('Confusion Matrix for EMBER-2018 Malware Identification (Dedicated Test Set)')\nplt.show()\n\n# ROC Curve\nfpr_final, tpr_final, thresholds_final = roc_curve(y_true_test, y_pred_proba_final_test)\nplt.figure(figsize=(7, 6))\nplt.plot(fpr_final, tpr_final, color='darkorange', lw=2, label=f'ROC curve (area = {roc_auc_final:.2f})')\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic (ROC) - EMBER-2018 Malware Identification (Dedicated Test Set)')\nplt.legend(loc=\"lower right\")\nplt.grid(True)\nplt.show()\n\n# Clean up memory\ndel X_test_ember_full, metadata_test_ember_full, X_test_filtered, metadata_test_filtered, y_true_test\ndel y_pred_proba_final_test, y_pred_final_test, cm_final, fpr_final, tpr_final, thresholds_final\n# If you are done with the model entirely, you can uncomment the line below:\n# del model_ember_id\ngc.collect()\nprint(\"Memory cleaned after dedicated test set evaluation.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:05:28.034304Z","iopub.execute_input":"2025-07-12T15:05:28.034927Z","iopub.status.idle":"2025-07-12T15:08:31.435722Z","shell.execute_reply.started":"2025-07-12T15:05:28.034902Z","shell.execute_reply":"2025-07-12T15:08:31.430162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport gc # For garbage collection\n\n# --- Part A: (Run this ONCE to preprocess and save) ---\n# Assuming load_ember_jsonl_revised function is defined and paths are correct\n\nprint(\"--- Loading full EMBER training data (this may take a long time) ---\")\nember_train_filepaths = [\n    '/kaggle/input/multistage-malware-detection-and-classification/ember/train_features_0.jsonl',\n    '/kaggle/input/multistage-malware-detection-and-classification/ember/train_features_1.jsonl',\n    '/kaggle/input/multistage-malware-detection-and-classification/ember/train_features_2.jsonl',\n    '/kaggle/input/multistage-malware-detection-and-classification/ember/train_features_3.jsonl',\n    '/kaggle/input/multistage-malware-detection-and-classification/ember/train_features_4.jsonl',\n    '/kaggle/input/multistage-malware-detection-and-classification/ember/train_features_5.jsonl'\n]\n\n# Set process_limit to None to load ALL available training samples\ntry:\n    X_full_ember, metadata_full_ember, ember_feature_names = load_ember_jsonl_revised(\n        ember_train_filepaths,\n        process_limit=100000\n    )\n    print(\"Full EMBER training data loaded successfully.\")\n    print(f\"Total samples loaded (raw): {len(metadata_full_ember)}\")\n    print(f\"Feature names count: {len(ember_feature_names)}\")\n\n    # Filter out unlabeled data (-1)\n    initial_samples = len(metadata_full_ember)\n    metadata_filtered_ember = metadata_full_ember[metadata_full_ember['label'] != -1].copy()\n    X_filtered_ember = X_full_ember[metadata_full_ember['label'] != -1]\n\n    print(f\"Samples after removing unlabeled data: {len(metadata_filtered_ember)}\")\n\n    # --- Save the processed data ---\n    output_dir = '/kaggle/working/' # Or any other writable directory\n\n    np.save(f'{output_dir}X_ember_train_processed.npy', X_filtered_ember)\n    metadata_filtered_ember.to_parquet(f'{output_dir}metadata_ember_train_processed.parquet', index=False)\n    # You might also want to save feature names if not always consistent\n    pd.Series(ember_feature_names).to_csv(f'{output_dir}ember_feature_names.csv', index=False, header=False)\n\n\n    print(f\"Processed training features saved to: {output_dir}X_ember_train_processed.npy\")\n    print(f\"Processed training metadata saved to: {output_dir}metadata_ember_train_processed.parquet\")\n    print(\"--- Preprocessing and saving complete. ---\")\n\n    # Clean up memory after saving\n    del X_full_ember, metadata_full_ember, metadata_filtered_ember, X_filtered_ember\n    gc.collect()\n\nexcept Exception as e:\n    print(f\"An error occurred during full data loading: {e}\")\n    # Handle error or raise","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:09:16.835937Z","iopub.execute_input":"2025-07-12T15:09:16.836274Z","iopub.status.idle":"2025-07-12T15:10:44.190937Z","shell.execute_reply.started":"2025-07-12T15:09:16.836252Z","shell.execute_reply":"2025-07-12T15:10:44.188681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import necessary libraries (ensure all are imported if you restart kernel)\nimport xgboost as xgb\nfrom sklearn.metrics import roc_auc_score, accuracy_score, classification_report, confusion_matrix\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport time\nimport numpy as np\nimport pandas as pd\nimport gc\n\nprint(\"--- Starting XGBoost Classifier Training with Learning Curve Tracking ---\")\n\n# Calculate scale_pos_weight for XGBoost based on your training set\nneg_count = np.sum(y_train_ember_id == 0)\npos_count = np.sum(y_train_ember_id == 1)\nscale_pos_weight_xgboost = neg_count / pos_count\nprint(f\"Calculated scale_pos_weight for XGBoost (from training set): {scale_pos_weight_xgboost:.2f}\")\n\n# Initialize XGBoost Classifier\n# Modified: eval_metric now includes 'logloss' and 'error'\nxgb_model = xgb.XGBClassifier(\n    objective='binary:logistic',\n    eval_metric=['logloss', 'error', 'auc'], # Track loss, error (1-accuracy), and AUC\n    use_label_encoder=False,\n    enable_categorical=False,\n    n_estimators=1000,\n    learning_rate=0.05,\n    max_depth=7,\n    subsample=0.7,\n    colsample_bytree=0.7,\n    gamma=0.1,\n    n_jobs=-1,\n    random_state=42,\n    scale_pos_weight=scale_pos_weight_xgboost,\n    tree_method='hist',\n)\n\n# Set up early stopping (still monitors AUC)\neval_set = [(X_train_ember_id, y_train_ember_id), (X_test_ember_id, y_test_ember_id)]\ncallbacks = [xgb.callback.EarlyStopping(rounds=50, metric_name='auc', data_name='validation_1', save_best=True)]\n\nstart_time = time.time()\nxgb_model.fit(X_train_ember_id, y_train_ember_id,\n              eval_set=eval_set,\n              callbacks=callbacks,\n              verbose=False)\n\ntraining_time_xgb = time.time() - start_time\nprint(f\"\\nXGBoost Model Training Time: {training_time_xgb:.2f} seconds\")\n\n\n# --- Plotting Learning Curves (Loss, Accuracy, AUC) ---\nprint(\"\\n--- Plotting XGBoost Learning Curves ---\")\n\n# Access the evaluation results from the fitted model\nevals_result = xgb_model.evals_result()\n\nepochs = len(evals_result['validation_0']['logloss']) # Number of boosting rounds\n\nplt.figure(figsize=(18, 5))\n\n# Plot LogLoss\nplt.subplot(1, 3, 1)\nplt.plot(range(epochs), evals_result['validation_0']['logloss'], label='Train LogLoss')\nplt.plot(range(epochs), evals_result['validation_1']['logloss'], label='Validation LogLoss')\nplt.title('XGBoost LogLoss Learning Curve')\nplt.xlabel('Boosting Rounds')\nplt.ylabel('LogLoss')\nplt.legend()\nplt.grid(True)\n\n# Plot Accuracy (from error metric)\nplt.subplot(1, 3, 2)\nplt.plot(range(epochs), [1 - x for x in evals_result['validation_0']['error']], label='Train Accuracy')\nplt.plot(range(epochs), [1 - x for x in evals_result['validation_1']['error']], label='Validation Accuracy')\nplt.title('XGBoost Accuracy Learning Curve')\nplt.xlabel('Boosting Rounds')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.grid(True)\n\n# Plot AUC\nplt.subplot(1, 3, 3)\nplt.plot(range(epochs), evals_result['validation_0']['auc'], label='Train AUC')\nplt.plot(range(epochs), evals_result['validation_1']['auc'], label='Validation AUC')\nplt.title('XGBoost AUC Learning Curve')\nplt.xlabel('Boosting Rounds')\nplt.ylabel('AUC')\nplt.legend()\nplt.grid(True)\n\nplt.tight_layout()\nplt.show()\n\n\n# --- Continue with Evaluation on Validation and Test Set (as before) ---\nprint(\"\\n--- XGBoost Performance on Validation Set ---\")\ny_pred_proba_val_xgb = xgb_model.predict_proba(X_test_ember_id)[:, 1]\ny_pred_val_xgb = (y_pred_proba_val_xgb > 0.5).astype(int)\n\naccuracy_val_xgb = accuracy_score(y_test_ember_id, y_pred_val_xgb)\nroc_auc_val_xgb = roc_auc_score(y_test_ember_id, y_pred_proba_val_xgb)\n\nprint(f\"Accuracy (Validation): {accuracy_val_xgb:.4f}\")\nprint(f\"ROC AUC (Validation): {roc_auc_val_xgb:.4f}\")\nprint(\"\\nClassification Report (Validation):\")\nprint(classification_report(y_test_ember_id, y_pred_val_xgb, target_names=['Benign', 'Malware'], zero_division=0))\n\nprint(\"\\nConfusion Matrix (Validation):\")\ncm_val_xgb = confusion_matrix(y_test_ember_id, y_pred_val_xgb)\nplt.figure(figsize=(6, 5))\nsns.heatmap(cm_val_xgb, annot=True, fmt='d', cmap='Blues',\n            xticklabels=['Predicted Benign', 'Predicted Malware'],\n            yticklabels=['Actual Benign', 'Actual Malware'])\nplt.title('XGBoost Confusion Matrix (Validation Set)')\nplt.ylabel('Actual Label')\nplt.xlabel('Predicted Label')\nplt.show()\n\n\nprint(\"\\n--- XGBoost Performance on Dedicated EMBER Test Set (200k samples) ---\")\n# Ensure X_test_filtered and y_true_test are loaded from your dedicated test set (Chunk 4)\n\ny_pred_proba_test_xgb = xgb_model.predict_proba(X_test_filtered)[:, 1]\ny_pred_test_xgb = (y_pred_proba_test_xgb > 0.5).astype(int)\n\naccuracy_test_xgb = accuracy_score(y_true_test, y_pred_test_xgb)\nroc_auc_test_xgb = roc_auc_score(y_true_test, y_pred_proba_test_xgb)\n\nprint(f\"Accuracy (Test Set): {accuracy_test_xgb:.4f}\")\nprint(f\"ROC AUC (Test Set): {roc_auc_test_xgb:.4f}\")\nprint(\"\\nClassification Report (Test Set):\")\nprint(classification_report(y_true_test, y_pred_test_xgb, target_names=['Benign', 'Malware'], zero_division=0))\n\nprint(\"\\nConfusion Matrix (Test Set):\")\ncm_test_xgb = confusion_matrix(y_true_test, y_pred_test_xgb)\nplt.figure(figsize=(6, 5))\nsns.heatmap(cm_test_xgb, annot=True, fmt='d', cmap='Blues',\n            xticklabels=['Predicted Benign', 'Predicted Malware'],\n            yticklabels=['Actual Benign', 'Actual Malware'])\nplt.title('XGBoost Confusion Matrix (Dedicated Test Set)')\nplt.ylabel('Actual Label')\nplt.xlabel('Predicted Label')\nplt.show()\n\n# Store XGBoost results for comparison\nxgb_results = {\n    'Algorithm': 'XGBoost',\n    'Training Time (s)': training_time_xgb,\n    'Validation Accuracy': accuracy_val_xgb,\n    'Validation ROC AUC': roc_auc_val_xgb,\n    'Test Accuracy': accuracy_test_xgb,\n    'Test ROC AUC': roc_auc_test_xgb,\n    'Test Malware Recall': cm_test_xgb[1, 1] / (cm_test_xgb[1, 0] + cm_test_xgb[1, 1]),\n    'Test Benign Recall': cm_test_xgb[0, 0] / (cm_test_xgb[0, 0] + cm_test_xgb[0, 1])\n}\n\nprint(\"\\nXGBoost Training and Evaluation Complete.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:17:17.665880Z","iopub.execute_input":"2025-07-12T15:17:17.666235Z","iopub.status.idle":"2025-07-12T15:21:56.341485Z","shell.execute_reply.started":"2025-07-12T15:17:17.666210Z","shell.execute_reply":"2025-07-12T15:21:56.340577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(\"Listing all directories under /kaggle/input/ :\")\nfor dirname in os.listdir('/kaggle/input/'):\n    print(os.path.join('/kaggle/input/', dirname))\n\n# After running this, look for a directory name that seems to contain the malware challenge data.\n# It might be 'microsoft-malware-prediction', 'microsoft-malware-classification-challenge', or similar.\n# Once you find it, use that specific path.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T15:23:19.711770Z","iopub.execute_input":"2025-07-12T15:23:19.712493Z","iopub.status.idle":"2025-07-12T15:23:19.718216Z","shell.execute_reply.started":"2025-07-12T15:23:19.712465Z","shell.execute_reply":"2025-07-12T15:23:19.717430Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(\"Contents of /kaggle/input/malware-classification/:\")\ntry:\n    print(os.listdir('/kaggle/input/malware-classification/'))\nexcept FileNotFoundError:\n    print(\"Directory not found. Please double-check the path.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T17:51:38.055216Z","iopub.execute_input":"2025-07-12T17:51:38.055534Z","iopub.status.idle":"2025-07-12T17:51:38.062802Z","shell.execute_reply.started":"2025-07-12T17:51:38.055512Z","shell.execute_reply":"2025-07-12T17:51:38.061206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install py7zr","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T17:47:25.587026Z","iopub.execute_input":"2025-07-12T17:47:25.587347Z","iopub.status.idle":"2025-07-12T17:47:25.598748Z","shell.execute_reply.started":"2025-07-12T17:47:25.587323Z","shell.execute_reply":"2025-07-12T17:47:25.597442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ndata_path = \"/kaggle/input/malware-classification\"\n\n# List available files\nprint(os.listdir(data_path))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T17:47:31.361142Z","iopub.execute_input":"2025-07-12T17:47:31.361459Z","iopub.status.idle":"2025-07-12T17:47:31.369646Z","shell.execute_reply.started":"2025-07-12T17:47:31.361436Z","shell.execute_reply":"2025-07-12T17:47:31.368499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Path to the dataset\ndata_path = \"/kaggle/input/malware-classification\"\n\n# Load the labels\nlabels_df = pd.read_csv(f\"{data_path}/trainLabels.csv\")\n\n# Show first 5 rows\nlabels_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T17:47:34.679787Z","iopub.execute_input":"2025-07-12T17:47:34.680117Z","iopub.status.idle":"2025-07-12T17:47:34.703378Z","shell.execute_reply.started":"2025-07-12T17:47:34.680096Z","shell.execute_reply":"2025-07-12T17:47:34.702042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndata_path = \"/kaggle/input/malware-classification\"\ndf = pd.read_csv(f\"{data_path}/trainLabels.csv\")\ndf.head(10)  # shows 10 samples with Id and Class","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T17:47:38.015760Z","iopub.execute_input":"2025-07-12T17:47:38.016090Z","iopub.status.idle":"2025-07-12T17:47:38.041093Z","shell.execute_reply.started":"2025-07-12T17:47:38.016068Z","shell.execute_reply":"2025-07-12T17:47:38.040044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Mapping class numbers to family names\nfamily_map = {\n    1: \"Ramnit\",\n    2: \"Lollipop\",\n    3: \"Kelihos_ver3\",\n    4: \"Vundo\",\n    5: \"Simda\",\n    6: \"Tracur\",\n    7: \"Kelihos_ver1\",\n    8: \"Obfuscator.ACY\",\n    9: \"Gatak\"\n}\n\n# Add a new column for family name\nlabels_df[\"Family\"] = labels_df[\"Class\"].map(family_map)\n\n# Preview updated labels\nlabels_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T17:47:42.175397Z","iopub.execute_input":"2025-07-12T17:47:42.175750Z","iopub.status.idle":"2025-07-12T17:47:42.189383Z","shell.execute_reply.started":"2025-07-12T17:47:42.175725Z","shell.execute_reply":"2025-07-12T17:47:42.188343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load labels\ndata_path = \"/kaggle/input/malware-classification\"\ndf = pd.read_csv(f\"{data_path}/trainLabels.csv\")\n\n# Optional: Map class to family names\nfamily_map = {\n    1: \"Ramnit\", 2: \"Lollipop\", 3: \"Kelihos_ver3\", 4: \"Vundo\",\n    5: \"Simda\", 6: \"Tracur\", 7: \"Kelihos_ver1\", 8: \"Obfuscator.ACY\", 9: \"Gatak\"\n}\ndf[\"Family\"] = df[\"Class\"].map(family_map)\n\n# Smart sampling: get min(10, count) rows per class\nsampled_df = df.groupby(\"Class\", group_keys=False).apply(\n    lambda x: x.sample(n=min(10, len(x)), random_state=42)\n).reset_index(drop=True)\n\n# Show count per class in the sample\nprint(sampled_df[\"Class\"].value_counts())\n\n# Preview result\nsampled_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T17:47:46.192359Z","iopub.execute_input":"2025-07-12T17:47:46.193388Z","iopub.status.idle":"2025-07-12T17:47:46.232523Z","shell.execute_reply.started":"2025-07-12T17:47:46.193360Z","shell.execute_reply":"2025-07-12T17:47:46.231424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for cls in sorted(sampled_df[\"Class\"].unique()):\n    ids = sampled_df[sampled_df[\"Class\"] == cls][\"Id\"].tolist()\n    print(f\"Class {cls} Sample IDs:\\n\", ids, \"\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T17:47:50.805799Z","iopub.execute_input":"2025-07-12T17:47:50.806137Z","iopub.status.idle":"2025-07-12T17:47:50.822035Z","shell.execute_reply.started":"2025-07-12T17:47:50.806115Z","shell.execute_reply":"2025-07-12T17:47:50.820577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels_df[\"Family\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T17:47:56.913479Z","iopub.execute_input":"2025-07-12T17:47:56.913803Z","iopub.status.idle":"2025-07-12T17:47:56.924271Z","shell.execute_reply.started":"2025-07-12T17:47:56.913780Z","shell.execute_reply":"2025-07-12T17:47:56.922931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(10, 6))\nsns.countplot(data=labels_df, y=\"Family\", order=labels_df[\"Family\"].value_counts().index)\nplt.title(\"Malware Family Distribution\")\nplt.xlabel(\"Number of Samples\")\nplt.ylabel(\"Malware Family\")\nplt.show()\n\nimport os\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\ndef extract_byte_histogram(file_path):\n    counts = np.zeros(256, dtype=int)\n    try:\n        with open(file_path, 'r') as f:\n            for line in f:\n                parts = line.strip().split()[1:]  # skip the address\n                for byte_str in parts:\n                    if byte_str != '??':\n                        try:\n                            byte_val = int(byte_str, 16)\n                            counts[byte_val] += 1\n                        except ValueError:\n                            continue\n    except Exception as e:\n        print(f\"Error reading {file_path}: {e}\")\n    return counts\n\n# Path to dataset\ndata_path = \"/kaggle/input/malware-classification\"\nbytes_path = os.path.join(data_path, \"train\")\n\n# Load labels\nlabel_df = pd.read_csv(os.path.join(data_path, \"trainLabels.csv\"))\nsample_ids = label_df[\"Id\"].tolist()[:100]  # First 100 samples for demo\nlabel_map = dict(zip(label_df[\"Id\"], label_df[\"Class\"]))\n\nX = []\ny = []\nfile_names = []\n\nfor file_id in tqdm(sample_ids):\n    file_path = os.path.join(bytes_path, file_id + \".bytes\")\n    if os.path.exists(file_path):\n        hist = extract_byte_histogram(file_path)\n        X.append(hist)\n        y.append(label_map[file_id])\n        file_names.append(file_id)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T17:55:21.093173Z","iopub.execute_input":"2025-07-12T17:55:21.093504Z","iopub.status.idle":"2025-07-12T17:55:21.339530Z","shell.execute_reply.started":"2025-07-12T17:55:21.093482Z","shell.execute_reply":"2025-07-12T17:55:21.338354Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create the feature DataFrame\ndf_features = pd.DataFrame(X, columns=[f'byte_{i:02X}' for i in range(256)])\ndf_features[\"label\"] = y\ndf_features[\"Id\"] = file_names\n\n# Map label → malware family\nfamily_map = {\n    1: \"Ramnit\", 2: \"Lollipop\", 3: \"Kelihos_ver3\", 4: \"Vundo\",\n    5: \"Simda\", 6: \"Tracur\", 7: \"Kelihos_ver1\", 8: \"Obfuscator.ACY\", 9: \"Gatak\"\n}\ndf_features[\"Family\"] = df_features[\"label\"].map(family_map)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T17:55:37.353556Z","iopub.execute_input":"2025-07-12T17:55:37.354114Z","iopub.status.idle":"2025-07-12T17:55:37.366684Z","shell.execute_reply.started":"2025-07-12T17:55:37.354080Z","shell.execute_reply":"2025-07-12T17:55:37.365617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count NaN values\nprint(\"Total NaNs:\", df_features.isna().sum().sum())\n\n# Check data types\nprint(\"Data types:\\n\", df_features.dtypes.value_counts())\n\n# Preview suspicious rows\nprint(df_features[df_features.isna().any(axis=1)].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T17:48:19.996352Z","iopub.execute_input":"2025-07-12T17:48:19.996807Z","iopub.status.idle":"2025-07-12T17:48:20.015144Z","shell.execute_reply.started":"2025-07-12T17:48:19.996777Z","shell.execute_reply":"2025-07-12T17:48:20.013302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check if any rows are all NaN or all zeros\nprint(df_features.isnull().sum().sum())         # Total NaNs\nprint((df_features.drop(columns=[\"label\"], errors='ignore') == 0).all(axis=1).sum())  # All-zero rows\n\n# Optionally drop NaNs\ndf_features = df_features.dropna()\n\ndf_features.head()\ndf_features.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T17:48:24.536281Z","iopub.execute_input":"2025-07-12T17:48:24.536612Z","iopub.status.idle":"2025-07-12T17:48:24.561034Z","shell.execute_reply.started":"2025-07-12T17:48:24.536590Z","shell.execute_reply":"2025-07-12T17:48:24.559910Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!mkdir -p /kaggle/working/bytes\n!7z l /kaggle/input/malware-classification/train.7z | grep '.bytes' | awk '{print $NF}' | head -n 2000 > /kaggle/working/bytes/bytes_list.txt\n\n# Now extract just these 2000 files\n!7z e /kaggle/input/malware-classification/train.7z -o/kaggle/working/bytes -i@/kaggle/working/bytes/bytes_list.txt\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:25:31.710261Z","iopub.execute_input":"2025-07-12T18:25:31.710883Z","iopub.status.idle":"2025-07-12T18:28:35.180213Z","shell.execute_reply.started":"2025-07-12T18:25:31.710836Z","shell.execute_reply":"2025-07-12T18:28:35.178327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nfrom tqdm import tqdm\n\ndef extract_byte_histogram(file_path):\n    try:\n        with open(file_path, 'r') as file:\n            hex_lines = file.readlines()\n        bytes_list = []\n        for line in hex_lines:\n            parts = line.strip().split()\n            bytes_seq = parts[1:]  # ignore address part\n            bytes_list.extend([b for b in bytes_seq if b != '??'])\n        byte_vals = [int(b, 16) for b in bytes_list if len(b) == 2]\n        hist = np.histogram(byte_vals, bins=256, range=(0, 255))[0]\n        return hist\n    except:\n        return np.zeros(256)\n\n# Extract features for a small sample of files\nfile_dir = '/kaggle/working/bytes'\nsample_files = os.listdir(file_dir)[:2000]  # You can increase to 1000+\n\nX = []\nfile_ids = []\n\nfor fname in tqdm(sample_files):\n    if fname.endswith('.bytes'):\n        f_id = fname.replace(\".bytes\", \"\")\n        hist = extract_byte_histogram(os.path.join(file_dir, fname))\n        X.append(hist)\n        file_ids.append(f_id)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:34:13.080337Z","iopub.execute_input":"2025-07-12T18:34:13.080821Z","iopub.status.idle":"2025-07-12T18:48:18.108774Z","shell.execute_reply.started":"2025-07-12T18:34:13.080793Z","shell.execute_reply":"2025-07-12T18:48:18.107394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_features = pd.DataFrame(X, columns=[f'byte_{i:02X}' for i in range(256)])\ndf_features[\"Id\"] = file_ids\ndf_features = df_features.merge(labels_df, on=\"Id\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:50:35.857329Z","iopub.execute_input":"2025-07-12T18:50:35.857703Z","iopub.status.idle":"2025-07-12T18:50:37.345211Z","shell.execute_reply.started":"2025-07-12T18:50:35.857678Z","shell.execute_reply":"2025-07-12T18:50:37.344052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_features.describe()\n\n# Correlation heatmap (optional)\nimport seaborn as sns\nplt.figure(figsize=(12, 8))\nsns.heatmap(df_features.drop(columns=[\"Id\", \"Class\", \"Family\"]).corr(), cmap=\"viridis\")\nplt.title(\"Feature Correlation\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:50:49.257850Z","iopub.execute_input":"2025-07-12T18:50:49.258885Z","iopub.status.idle":"2025-07-12T18:50:51.111466Z","shell.execute_reply.started":"2025-07-12T18:50:49.258849Z","shell.execute_reply":"2025-07-12T18:50:51.110407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_features.to_csv(\"/kaggle/working/byte_histogram_features.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:51:32.332209Z","iopub.execute_input":"2025-07-12T18:51:32.332574Z","iopub.status.idle":"2025-07-12T18:51:32.550114Z","shell.execute_reply.started":"2025-07-12T18:51:32.332553Z","shell.execute_reply":"2025-07-12T18:51:32.548933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Create correlation matrix\ncorr_matrix = df_features.drop(columns=[\"Id\", \"Class\", \"Family\"]).corr().abs()\n\n# Select upper triangle of correlation matrix\nupper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))\n\n# Find features with correlation greater than 0.95\nto_drop = [column for column in upper.columns if any(upper[column] > 0.95)]\n\n# Drop those features\ndf_reduced = df_features.drop(columns=to_drop)\nprint(f\"Dropped {len(to_drop)} highly correlated features.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:54:58.488618Z","iopub.execute_input":"2025-07-12T18:54:58.489105Z","iopub.status.idle":"2025-07-12T18:54:59.043927Z","shell.execute_reply.started":"2025-07-12T18:54:58.489076Z","shell.execute_reply":"2025-07-12T18:54:59.042800Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nle = LabelEncoder()\ndf_features[\"Family\"] = le.fit_transform(df_features[\"Family\"])  # You can also use df_reduced if used earlier\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:56:56.722828Z","iopub.execute_input":"2025-07-12T18:56:56.724261Z","iopub.status.idle":"2025-07-12T18:56:56.862156Z","shell.execute_reply.started":"2025-07-12T18:56:56.724218Z","shell.execute_reply":"2025-07-12T18:56:56.860457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = df_reduced.drop(columns=[\"Id\", \"Class\", \"Family\"])\ny = df_reduced[\"Family\"]\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:57:20.132673Z","iopub.execute_input":"2025-07-12T18:57:20.133121Z","iopub.status.idle":"2025-07-12T18:57:20.275257Z","shell.execute_reply.started":"2025-07-12T18:57:20.133097Z","shell.execute_reply":"2025-07-12T18:57:20.274045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, classification_report\n\nclf = RandomForestClassifier(n_estimators=100, random_state=42)\nclf.fit(X_train, y_train)\n\ny_pred = clf.predict(X_test)\n\n# Evaluate\nprint(\"Accuracy:\", accuracy_score(y_test, y_pred))\nprint(classification_report(y_test, y_pred, target_names=le.classes_))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T18:57:34.671810Z","iopub.execute_input":"2025-07-12T18:57:34.673099Z","iopub.status.idle":"2025-07-12T18:57:36.917473Z","shell.execute_reply.started":"2025-07-12T18:57:34.673064Z","shell.execute_reply":"2025-07-12T18:57:36.916405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import learning_curve\nimport numpy as np\nimport matplotlib.pyplot as plt\n\ntrain_sizes, train_scores, val_scores = learning_curve(\n    estimator=clf,  # your trained model\n    X=X, y=y,\n    train_sizes=np.linspace(0.1, 1.0, 10),\n    cv=5,\n    scoring='accuracy',\n    n_jobs=-1\n)\n\ntrain_mean = np.mean(train_scores, axis=1)\nval_mean = np.mean(val_scores, axis=1)\n\nplt.figure(figsize=(10,6))\nplt.plot(train_sizes, train_mean, label='Training Accuracy')\nplt.plot(train_sizes, val_mean, label='Validation Accuracy')\nplt.xlabel('Training Set Size')\nplt.ylabel('Accuracy')\nplt.title('Learning Curve')\nplt.legend()\nplt.grid()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T19:00:32.005050Z","iopub.execute_input":"2025-07-12T19:00:32.005392Z","iopub.status.idle":"2025-07-12T19:00:55.489621Z","shell.execute_reply.started":"2025-07-12T19:00:32.005369Z","shell.execute_reply":"2025-07-12T19:00:55.488434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\nimport matplotlib.pyplot as plt\n\n# Create confusion matrix\ncm = confusion_matrix(y_test, y_pred)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=le.classes_)\n\n# Set figure size before plotting\nplt.figure(figsize=(12, 10))\ndisp.plot(cmap='Blues', xticks_rotation=90)\nplt.title('Confusion Matrix')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T19:02:10.164999Z","iopub.execute_input":"2025-07-12T19:02:10.166019Z","iopub.status.idle":"2025-07-12T19:02:10.883438Z","shell.execute_reply.started":"2025-07-12T19:02:10.165959Z","shell.execute_reply":"2025-07-12T19:02:10.882296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\nprint(classification_report(y_test, y_pred, target_names=le.classes_))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T19:02:32.454535Z","iopub.execute_input":"2025-07-12T19:02:32.454936Z","iopub.status.idle":"2025-07-12T19:02:32.487441Z","shell.execute_reply.started":"2025-07-12T19:02:32.454909Z","shell.execute_reply":"2025-07-12T19:02:32.484858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"importances = clf.feature_importances_\nindices = np.argsort(importances)[-20:]  # Top 20 important features\nplt.figure(figsize=(10, 6))\nplt.barh(range(len(indices)), importances[indices], align='center')\nplt.yticks(range(len(indices)), [X.columns[i] for i in indices])\nplt.xlabel('Importance')\nplt.title('Top 20 Feature Importances')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T19:02:56.778076Z","iopub.execute_input":"2025-07-12T19:02:56.778437Z","iopub.status.idle":"2025-07-12T19:02:57.114786Z","shell.execute_reply.started":"2025-07-12T19:02:56.778414Z","shell.execute_reply":"2025-07-12T19:02:57.113423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create legend mapping color -> label\nimport matplotlib.patches as mpatches\nimport numpy as np\n\nunique_labels = np.unique(y_encoded)\npatches = [mpatches.Patch(color=scatter.cmap(scatter.norm(i)), label=le.inverse_transform([i])[0]) for i in unique_labels]\n\nplt.figure(figsize=(10, 6))\nscatter = plt.scatter(X_2d[:, 0], X_2d[:, 1], c=y_encoded, cmap='tab20', s=10)\nplt.title(\"t-SNE visualization of Malware Families\")\nplt.legend(handles=patches, bbox_to_anchor=(1.05, 1), loc='upper left', title=\"Family\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T19:04:40.134352Z","iopub.execute_input":"2025-07-12T19:04:40.134776Z","iopub.status.idle":"2025-07-12T19:04:40.496832Z","shell.execute_reply.started":"2025-07-12T19:04:40.134747Z","shell.execute_reply":"2025-07-12T19:04:40.495398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---------------------------------------------\n# 0)  PREP  (run once)\n# ---------------------------------------------\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split, cross_val_score\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.ensemble import RandomForestClassifier, ExtraTreesClassifier, GradientBoostingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.neural_network import MLPClassifier\nimport warnings, matplotlib.pyplot as plt\nwarnings.filterwarnings('ignore')\n\n# --- features & labels ---\nX = df_features.drop(columns=[\"Id\", \"Class\", \"Family\"])\ny = y_encoded                         # already LabelEncoded earlier\n\n# Train‑test split just for final hold‑out evaluation\nX_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y\n)\n\n# ---------------------------------------------\n# 1)  DEFINE A SMALL MODEL ZOO\n# ---------------------------------------------\nmodels = {\n    \"RandomForest\": RandomForestClassifier(n_estimators=300, random_state=42, n_jobs=-1),\n    \"ExtraTrees\"  : ExtraTreesClassifier(n_estimators=400, random_state=42, n_jobs=-1),\n    \n    # algorithms that need scaling are wrapped in a Pipeline\n    \"LogReg\" : Pipeline([\n        (\"scaler\", StandardScaler()), \n        (\"clf\", LogisticRegression(max_iter=500, multi_class=\"multinomial\"))\n    ]),\n    \n    \"SVM‑RBF\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", SVC(kernel=\"rbf\", C=5, gamma=\"scale\"))\n    ]),\n    \n    \"kNN‑10\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", KNeighborsClassifier(n_neighbors=10))\n    ]),\n    \n    \"GradBoost\": GradientBoostingClassifier(random_state=42),\n    \n    \"MLP‑128x64\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", MLPClassifier(hidden_layer_sizes=(128,64), max_iter=200, random_state=42))\n    ]),\n}\n\n# (Optional) add XGBoost or LightGBM if their libraries are installed:\n# import xgboost as xgb\n# models[\"XGBoost\"] = xgb.XGBClassifier(\n#     n_estimators=500, max_depth=7, learning_rate=0.1,\n#     subsample=0.8, colsample_bytree=0.8, objective=\"multi:softprob\",\n#     num_class=len(np.unique(y)), tree_method=\"hist\", random_state=42\n# )\n\n# ---------------------------------------------\n# 2)  CROSS‑VALIDATE EACH MODEL\n# ---------------------------------------------\nresults = {}\nfor name, clf in models.items():\n    cv_scores = cross_val_score(clf, X_train, y_train, cv=5, scoring=\"accuracy\", n_jobs=-1)\n    results[name] = {\n        \"CV mean\":  cv_scores.mean(),\n        \"CV std\" :  cv_scores.std()\n    }\n    print(f\"{name:12s}  |  CV accuracy = {cv_scores.mean():.4f} ± {cv_scores.std():.4f}\")\n\n# ---------------------------------------------\n# 3)  RANKED SUMMARY\n# ---------------------------------------------\nsummary = (pd.DataFrame(results)\n           .T.sort_values(\"CV mean\", ascending=False)\n           .style.format({\"CV mean\":\"{:.4f}\", \"CV std\":\"{:.4f}\"}))\ndisplay(summary)\n\n# ---------------------------------------------\n# 4)  TRAIN THE BEST MODEL ON FULL TRAIN SET, EVALUATE ON HOLD‑OUT\n# ---------------------------------------------\nbest_name = summary.data.index[0]\nbest_model = models[best_name]\nbest_model.fit(X_train, y_train)\ny_pred = best_model.predict(X_test)\n\nprint(f\"\\n🏆 Best model: {best_name}\")\nprint(\"Hold‑out accuracy:\", accuracy_score(y_test, y_pred))\nprint(classification_report(y_test, y_pred, target_names=le.classes_))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T19:18:06.592856Z","iopub.execute_input":"2025-07-12T19:18:06.593639Z","iopub.status.idle":"2025-07-12T19:23:15.106098Z","shell.execute_reply.started":"2025-07-12T19:18:06.593609Z","shell.execute_reply":"2025-07-12T19:23:15.104559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"importances = best_model.feature_importances_\nfeat_names = X.columns\n\nfeat_imp = pd.Series(importances, index=feat_names).sort_values(ascending=False)\ntop_features = feat_imp.head(20)\n\nplt.figure(figsize=(10,6))\ntop_features.plot(kind=\"barh\", color='steelblue')\nplt.gca().invert_yaxis()\nplt.title(\"Top 20 Feature Importances (ExtraTrees)\")\nplt.xlabel(\"Importance Score\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T19:33:21.145575Z","iopub.execute_input":"2025-07-12T19:33:21.146056Z","iopub.status.idle":"2025-07-12T19:33:21.642404Z","shell.execute_reply.started":"2025-07-12T19:33:21.146026Z","shell.execute_reply":"2025-07-12T19:33:21.641329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import learning_curve\n\ntrain_sizes, train_scores, test_scores = learning_curve(\n    best_model, X, y, cv=5, train_sizes=np.linspace(0.1, 1.0, 10), n_jobs=-1\n)\n\ntrain_mean = train_scores.mean(axis=1)\ntest_mean = test_scores.mean(axis=1)\n\nplt.figure(figsize=(8, 5))\nplt.plot(train_sizes, train_mean, label=\"Train Score\", marker=\"o\")\nplt.plot(train_sizes, test_mean, label=\"CV Score\", marker=\"s\")\nplt.title(\"Learning Curve for ExtraTreesClassifier\")\nplt.xlabel(\"Training Set Size\")\nplt.ylabel(\"Accuracy\")\nplt.legend()\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T19:33:37.056095Z","iopub.execute_input":"2025-07-12T19:33:37.056468Z","iopub.status.idle":"2025-07-12T19:34:08.385278Z","shell.execute_reply.started":"2025-07-12T19:33:37.056443Z","shell.execute_reply":"2025-07-12T19:34:08.383784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import validation_curve\n\nparam_range = [50, 100, 200, 300, 400, 500]\ntrain_scores, test_scores = validation_curve(\n    ExtraTreesClassifier(random_state=42),\n    X, y, param_name=\"n_estimators\", param_range=param_range,\n    cv=5, scoring=\"accuracy\", n_jobs=-1\n)\n\ntrain_mean = train_scores.mean(axis=1)\ntest_mean = test_scores.mean(axis=1)\n\nplt.plot(param_range, train_mean, label=\"Train\", marker='o')\nplt.plot(param_range, test_mean, label=\"CV\", marker='s')\nplt.title(\"Validation Curve for n_estimators\")\nplt.xlabel(\"Number of Estimators\")\nplt.ylabel(\"Accuracy\")\nplt.legend()\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T19:35:33.010893Z","iopub.execute_input":"2025-07-12T19:35:33.011435Z","iopub.status.idle":"2025-07-12T19:35:47.734653Z","shell.execute_reply.started":"2025-07-12T19:35:33.011408Z","shell.execute_reply":"2025-07-12T19:35:47.733604Z"}},"outputs":[],"execution_count":null}]}