{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":4117,"databundleVersionId":46665,"sourceType":"competition"}],"dockerImageVersionId":31234,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# IMPORTS","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom collections import Counter\nimport warnings\nimport random\nwarnings.filterwarnings('ignore')\n\nRANDOM_SEED = 42\n\n# Set all random seeds for reproducibility\nrandom.seed(RANDOM_SEED)\nnp.random.seed(RANDOM_SEED)\nos.environ['PYTHONHASHSEED'] = str(RANDOM_SEED)\n\n# TensorFlow/Keras seeds (if used)\ntry:\n    import tensorflow as tf\n    tf.random.set_seed(RANDOM_SEED)\nexcept ImportError:\n    pass\n\n\n# ML imports\nfrom sklearn.model_selection import train_test_split, cross_val_score\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.ensemble import (RandomForestClassifier, ExtraTreesClassifier, \n                              GradientBoostingClassifier, VotingClassifier)\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.feature_selection import SelectKBest, chi2\nfrom sklearn.metrics import (accuracy_score, classification_report, confusion_matrix, \n                             ConfusionMatrixDisplay, precision_score, recall_score, \n                             f1_score, precision_recall_fscore_support)\n\n# Imbalanced data handling\nfrom imblearn.over_sampling import SMOTE, ADASYN\nfrom imblearn.combine import SMOTETomek\n\n# XGBoost\ntry:\n    from xgboost import XGBClassifier\n    XGBOOST_AVAILABLE = True\nexcept ImportError:\n    print(\"⚠️ XGBoost not available. Install with: pip install xgboost\")\n    XGBOOST_AVAILABLE = False\n\n# LightGBM\ntry:\n    import lightgbm as lgb\n    LGBM_AVAILABLE = True\nexcept ImportError:\n    print(\"⚠️ LightGBM not available. Install with: pip install lightgbm\")\n    LGBM_AVAILABLE = False\n\nprint(\"\\n✅ All imports successful\")\nprint(f\"   XGBoost: {'Available' if XGBOOST_AVAILABLE else 'Not Available'}\")\nprint(f\"   LightGBM: {'Available' if LGBM_AVAILABLE else 'Not Available'}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T09:52:50.650034Z","iopub.execute_input":"2026-01-10T09:52:50.651789Z","iopub.status.idle":"2026-01-10T09:52:50.670886Z","shell.execute_reply.started":"2026-01-10T09:52:50.651741Z","shell.execute_reply":"2026-01-10T09:52:50.668326Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DATA STATS","metadata":{}},{"cell_type":"code","source":"\ndata_path = \"/kaggle/input/malware-classification\"\nlabels_df = pd.read_csv(f\"{data_path}/trainLabels.csv\")\n\nfamily_map = {\n    1: \"Ramnit\", 2: \"Lollipop\", 3: \"Kelihos_ver3\", 4: \"Vundo\",\n    5: \"Simda\", 6: \"Tracur\", 7: \"Kelihos_ver1\", 8: \"Obfuscator.ACY\", 9: \"Gatak\"\n}\nlabels_df[\"Family\"] = labels_df[\"Class\"].map(family_map)\n\nprint(\"\\nLAST 5 rows:\")\nprint(labels_df.tail())\nprint(f\"\\nTotal samples: {len(labels_df)}\")\n\nprint(\"\\n\" + \"=\"*80)\nprint(\"CLASS DISTRIBUTION ANALYSIS\")\nprint(\"=\"*80)\n\n# Use seeded sampling for reproducibility\nsampled_df = labels_df.groupby(\"Class\", group_keys=False).apply(\n    lambda x: x.sample(n=min(10, len(x)), random_state=RANDOM_SEED)\n).reset_index(drop=True)\n\nprint(\"\\nSample count per class (with seed={RANDOM_SEED}):\")\nprint(sampled_df[\"Class\"].value_counts().sort_index())\n\nfor cls in sorted(sampled_df[\"Class\"].unique()):\n    ids = sampled_df[sampled_df[\"Class\"] == cls][\"Id\"].tolist()\n    print(f\"Class {cls} ({family_map[cls]}) Sample IDs:\\n\", ids[:5], \"...\\n\")\n\nprint(\"\\n⚠️ Malware family distribution (BEFORE BALANCING):\")\ndistribution = labels_df[\"Family\"].value_counts()\nprint(distribution)\n\n# Calculate imbalance statistics\nmax_samples = distribution.max()\nmin_samples = distribution.min()\nimbalance_ratio = max_samples / min_samples\nprint(f\"\\nImbalance Ratio: {imbalance_ratio:.1f}:1 (Majority:Minority)\")\nprint(f\"   Majority class: {distribution.idxmax()} ({max_samples} samples)\")\nprint(f\"   Minority class: {distribution.idxmin()} ({min_samples} samples)\")\n\n# Visualization\nplt.figure(figsize=(10, 6))\nsns.countplot(data=labels_df, y=\"Family\", order=labels_df[\"Family\"].value_counts().index)\nplt.title(f\"Malware Family Distribution (Original - Imbalanced)\\nImbalance Ratio: {imbalance_ratio:.1f}:1\")\nplt.xlabel(\"Number of Samples\")\nplt.ylabel(\"Malware Family\")\nplt.tight_layout()\nplt.savefig('/kaggle/working/class_distribution_original.png', dpi=300, bbox_inches='tight')\nplt.show()\n\nprint(\"\\nDistribution analysis complete\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T09:52:54.956140Z","iopub.execute_input":"2026-01-10T09:52:54.956905Z","iopub.status.idle":"2026-01-10T09:52:55.688212Z","shell.execute_reply.started":"2026-01-10T09:52:54.956872Z","shell.execute_reply":"2026-01-10T09:52:55.687147Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# FEATURE EXTRACTION","metadata":{}},{"cell_type":"code","source":"!mkdir -p /kaggle/working/bytes\n!7z l /kaggle/input/malware-classification/train.7z | grep '.bytes' | awk '{print $NF}' | head -n 2000 > /kaggle/working/bytes/bytes_list.txt\n!7z e /kaggle/input/malware-classification/train.7z -o/kaggle/working/bytes -i@/kaggle/working/bytes/bytes_list.txt\n\nprint(\"File extraction complete\")\n\n\nprint(\"\\n\" + \"=\"*80)\nprint(\"🔬 BYTE HISTOGRAM FEATURE EXTRACTION\")\nprint(\"=\"*80)\n\ndef extract_byte_histogram(file_path):\n    \"\"\"\n    Extract 256-dimensional byte frequency histogram from .bytes file\n    \n    Args:\n        file_path: Path to .bytes file containing hex representation\n    \n    Returns:\n        numpy array of shape (256,) containing byte frequencies\n    \"\"\"\n    try:\n        with open(file_path, 'r') as file:\n            hex_lines = file.readlines()\n        \n        bytes_list = []\n        for line in hex_lines:\n            parts = line.strip().split()\n            bytes_seq = parts[1:]  # Skip address column\n            bytes_list.extend([b for b in bytes_seq if b != '??'])\n        \n        # Convert hex to integers\n        byte_vals = [int(b, 16) for b in bytes_list if len(b) == 2]\n        \n        # Create histogram (256 bins for byte values 0-255)\n        hist = np.histogram(byte_vals, bins=256, range=(0, 255))[0]\n        \n        return hist\n    except Exception as e:\n        print(f\"Error reading {file_path}: {e}\")\n        return np.zeros(256)\n\n# Extract features from all files\nfile_dir = '/kaggle/working/bytes'\nsample_files = sorted(os.listdir(file_dir))[:2500]  # Sort for reproducibility\n\nX = []\nfile_ids = []\n\nprint(f\"Extracting features from {len(sample_files)} files...\")\nfor fname in tqdm(sample_files):\n    if fname.endswith('.bytes'):\n        f_id = fname.replace(\".bytes\", \"\")\n        hist = extract_byte_histogram(os.path.join(file_dir, fname))\n        X.append(hist)\n        file_ids.append(f_id)\n\n# Create DataFrame\ndf_features = pd.DataFrame(X, columns=[f'byte_{i:02X}' for i in range(256)])\ndf_features[\"Id\"] = file_ids\ndf_features = df_features.merge(labels_df, on=\"Id\")\n\nprint(\"\\nFeature extraction complete\")\nprint(f\"   Shape: {df_features.shape}\")\nprint(f\"   Features: 256 byte positions\")\nprint(f\"   Samples: {len(df_features)}\")\n\nprint(\"\\nFeature statistics:\")\nprint(df_features.describe())\n\n# Save features\ndf_features.to_csv(\"/kaggle/working/byte_histogram_features.csv\", index=False)\nprint(\"\\nFeatures saved to: byte_histogram_features.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T09:54:52.314570Z","iopub.execute_input":"2026-01-10T09:54:52.314986Z","iopub.status.idle":"2026-01-10T10:11:42.398392Z","shell.execute_reply.started":"2026-01-10T09:54:52.314955Z","shell.execute_reply":"2026-01-10T10:11:42.396483Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DATASET OVERVIEW","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)\npd.set_option('display.width', None)\npd.set_option('display.max_colwidth', 20)\n\n# Show first 10 samples with selected features\nprint(\"\\nSample Data (First 10 rows with selected features):\\n\")\n\n# Select important columns to show\ndisplay_cols = ['Id', 'byte_00', 'byte_01', 'byte_02', 'byte_10', 'byte_20', \n                'byte_50', 'byte_A0', 'byte_FF', 'Class', 'Family']\n\ndf_sample = df_features[display_cols].head(10)\ndisplay(df_sample)  \n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T10:13:19.892513Z","iopub.execute_input":"2026-01-10T10:13:19.892981Z","iopub.status.idle":"2026-01-10T10:13:19.913550Z","shell.execute_reply.started":"2026-01-10T10:13:19.892936Z","shell.execute_reply":"2026-01-10T10:13:19.912183Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# TRAIN-TEST","metadata":{}},{"cell_type":"code","source":"# Label encoding\nle = LabelEncoder()\ndf_features[\"Family_Encoded\"] = le.fit_transform(df_features[\"Family\"]).astype(int)\nfamily_names = le.classes_\n\nprint(f\"Encoded {len(family_names)} malware families:\")\nfor idx, name in enumerate(family_names):\n    print(f\"   {idx}: {name}\")\n\n# Separate features and labels\nX = df_features.drop(columns=[\"Id\", \"Class\", \"Family\", \"Family_Encoded\"])\ny = df_features[\"Family_Encoded\"]\n\nprint(f\"\\nFeature matrix shape: {X.shape}\")\nprint(f\"Label vector shape: {y.shape}\")\n\n# Use random_state for reproducibility\nX_train, X_test, y_train, y_test = train_test_split(\n    X, y, \n    test_size=0.2, \n    random_state=RANDOM_SEED,  \n    stratify=y  # Maintain class distribution\n)\n\nprint(f\"Training set: {X_train.shape[0]} samples ({X_train.shape[0]/len(X)*100:.1f}%)\")\nprint(f\"Test set:     {X_test.shape[0]} samples ({X_test.shape[0]/len(X)*100:.1f}%)\")\n\n\n\nprint(\"\\nTRAINING SET DISTRIBUTION (BEFORE SMOTE):\")\nprint(\"-\"*80)\ntrain_dist = pd.Series(y_train).value_counts().sort_index()\n\ntrain_distribution_data = []\nfor idx, name in enumerate(family_names):\n    count = (y_train == idx).sum()\n    percentage = (count / len(y_train)) * 100\n    train_distribution_data.append({\n        'Class': idx,\n        'Family': name,\n        'Count': count,\n        'Percentage': percentage\n    })\n    print(f\"  {name:15s}: {count:4d} samples ({percentage:5.2f}%)\")\n\n# Calculate imbalance ratio\nmax_class = train_dist.max()\nmin_class = train_dist.min()\nimbalance_ratio = max_class / min_class\nprint(f\"\\n Training Set Imbalance Ratio: {imbalance_ratio:.1f}:1 (Majority:Minority)\")\nprint(f\"   Majority: {family_names[train_dist.idxmax()]} ({max_class} samples)\")\nprint(f\"   Minority: {family_names[train_dist.idxmin()]} ({min_class} samples)\")\n\nprint(\"\\nTEST SET DISTRIBUTION (UNCHANGED - Original Distribution):\")\nprint(\"-\"*80)\ntest_dist = pd.Series(y_test).value_counts().sort_index()\n\nfor idx, name in enumerate(family_names):\n    count = (y_test == idx).sum()\n    percentage = (count / len(y_test)) * 100\n    print(f\"  {name:15s}: {count:4d} samples ({percentage:5.2f}%)\")\n\n# Store original distribution for later comparison\noriginal_train_dist = train_dist.copy()\noriginal_test_dist = test_dist.copy()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T10:14:12.139030Z","iopub.execute_input":"2026-01-10T10:14:12.139853Z","iopub.status.idle":"2026-01-10T10:14:12.187126Z","shell.execute_reply.started":"2026-01-10T10:14:12.139820Z","shell.execute_reply":"2026-01-10T10:14:12.185372Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# FEATURE SELECTION","metadata":{}},{"cell_type":"code","source":"print(\"WHY: Byte histograms are count data - perfect for chi-square!\")\nprint(\"GOAL: Identify which byte positions are most discriminative\")\nprint(f\"SEED: {RANDOM_SEED} (for reproducible feature selection)\")\n\n# Select top 100 most important features\nN_FEATURES = 100\n\n# CRITICAL: Feature selection BEFORE SMOTE to avoid data leakage\nchi2_selector = SelectKBest(chi2, k=N_FEATURES)\nX_train_selected = chi2_selector.fit_transform(X_train, y_train)\nX_test_selected = chi2_selector.transform(X_test)\n\nprint(f\"\\nSelected {N_FEATURES} most discriminative features out of 256\")\nprint(f\"   Training shape: {X_train.shape} → {X_train_selected.shape}\")\nprint(f\"   Test shape:     {X_test.shape} → {X_test_selected.shape}\")\n\n# Get feature scores\nfeature_scores = pd.DataFrame({\n    'Feature': X.columns,\n    'Chi2_Score': chi2_selector.scores_,\n    'Selected': chi2_selector.get_support()\n}).sort_values('Chi2_Score', ascending=False)\n\n\nprint(feature_scores.head(20)[['Feature', 'Chi2_Score']].to_string(index=False))\n\n# Visualize top features\nplt.figure(figsize=(12, 6))\ntop_20 = feature_scores.head(20)\nplt.barh(top_20['Feature'], top_20['Chi2_Score'], color='#3498db')\nplt.xlabel('Chi-Square Score', fontsize=12)\nplt.ylabel('Byte Position', fontsize=12)\nplt.title('Top 20 Most Discriminative Byte Positions for Malware Classification', fontsize=14)\nplt.gca().invert_yaxis()\nplt.tight_layout()\nplt.savefig('/kaggle/working/chi_square_features.png', dpi=300, bbox_inches='tight')\nplt.show()\n\n# Save selected feature names for interpretation\nselected_features = feature_scores[feature_scores['Selected']]['Feature'].tolist()\nfeature_scores.to_csv('/kaggle/working/feature_scores.csv', index=False)\n\nprint(f\"\\nFeature reduction: 256 → {N_FEATURES} ({(N_FEATURES/256)*100:.1f}% retained)\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T10:14:49.820939Z","iopub.execute_input":"2026-01-10T10:14:49.821494Z","iopub.status.idle":"2026-01-10T10:14:50.836192Z","shell.execute_reply.started":"2026-01-10T10:14:49.821463Z","shell.execute_reply":"2026-01-10T10:14:50.834791Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SMOTE","metadata":{}},{"cell_type":"code","source":"print(f\"WHY: {imbalance_ratio:.1f}:1 imbalance ratio is SEVERE!\")\nprint(\"SMOTE: Creates synthetic samples for minority classes\")\nprint(f\"SEED: {RANDOM_SEED} (for reproducible synthetic samples)\")\n\n# Store pre-SMOTE counts\npre_smote_counts = {\n    'total': len(y_train),\n    'per_class': {family_names[i]: (y_train == i).sum() for i in range(len(family_names))}\n}\n\nprint(\"\\nBEFORE SMOTE:\")\nprint(\"-\"*80)\nfor idx, name in enumerate(family_names):\n    count = pre_smote_counts['per_class'][name]\n    percentage = (count / pre_smote_counts['total']) * 100\n    print(f\"  {name:15s}: {count:4d} samples ({percentage:5.2f}%)\")\n\nprint(f\"\\nTotal: {pre_smote_counts['total']} samples\")\n\n# ========== STRATEGIC OVERSAMPLING ==========\nprint(\"\\n\" + \"=\"*80)\nprint(\"STRATEGIC OVERSAMPLING (Not Extreme)\")\nprint(\"=\"*80)\n\nimport numpy as np\nfrom imblearn.over_sampling import SMOTE\n\n# Calculate statistics\nclass_counts = np.bincount(y_train)\nmedian_count = int(np.median(class_counts[class_counts > 0]))\nq3_count = int(np.percentile(class_counts[class_counts > 0], 75))  # 75th percentile\n\nprint(f\"Statistics:\")\nprint(f\"  Maximum: {max(class_counts)} (Kelihos_ver3)\")\nprint(f\"  Q3 (75th percentile): {q3_count}\")\nprint(f\"  Median: {median_count}\")\nprint(f\"  Minimum: {min(class_counts)} (Simda)\")\n\n# Define SMART target: Use Q3 (75th percentile) NOT maximum\n# This balances without extreme oversampling\nsmart_target = q3_count\nprint(f\"\\nSmart target: {smart_target} samples per class (Q3)\")\nprint(f\"(Not {max(class_counts)} which would oversample too much)\")\n\n# SMART sampling strategy\nsampling_strategy = {}\nprint(\"\\nOversampling Strategy:\")\nprint(\"-\"*80)\nfor idx, name in enumerate(family_names):\n    current = pre_smote_counts['per_class'][name]\n    \n    if current < 10:  # Extreme minority (Simda: 8)\n        # Cap at reasonable multiple (6x, max 50)\n        target = min(50, current * 6)\n        sampling_strategy[idx] = target\n        print(f\"  {name:15s}: {current:3d} → {target:3d} (extreme minority, 6x cap)\")\n    \n    elif current < median_count:  # Below median\n        # Bring up to median level\n        target = median_count\n        sampling_strategy[idx] = target\n        print(f\"  {name:15s}: {current:3d} → {target:3d} (below median)\")\n    \n    elif current < smart_target:  # Below Q3\n        # Bring up to Q3 level\n        target = smart_target\n        sampling_strategy[idx] = target\n        print(f\"  {name:15s}: {current:3d} → {target:3d} (below Q3)\")\n    \n    else:  # Already at or above Q3\n        # Keep as is (no oversampling)\n        sampling_strategy[idx] = current\n        print(f\"  {name:15s}: {current:3d} → {current:3d} (already sufficient)\")\n\n# Adaptive k_neighbors for safety\nmin_class_count = min(class_counts)\nsafe_k_neighbors = min(3, max(1, min_class_count - 2))\nprint(f\"\\nUsing k_neighbors={safe_k_neighbors} (safe for min class with {min_class_count} samples)\")\n\n# Apply SMART SMOTE\nsmote = SMOTE(\n    random_state=RANDOM_SEED,\n    k_neighbors=safe_k_neighbors,\n    sampling_strategy=sampling_strategy\n)\n\nX_train_resampled, y_train_resampled = smote.fit_resample(X_train_selected, y_train)\n\n# Store post-SMOTE counts\npost_smote_counts = {\n    'total': len(y_train_resampled),\n    'per_class': {family_names[i]: (y_train_resampled == i).sum() for i in range(len(family_names))}\n}\n\nprint(\"\\nAFTER STRATEGIC OVERSAMPLING:\")\nprint(\"-\"*80)\nfor idx, name in enumerate(family_names):\n    count = post_smote_counts['per_class'][name]\n    percentage = (count / post_smote_counts['total']) * 100\n    increase = count - pre_smote_counts['per_class'][name]\n    change_symbol = \"+\" if increase > 0 else \"\"\n    print(f\"  {name:15s}: {count:4d} samples ({percentage:5.2f}%) [{change_symbol}{increase:4d}]\")\n\nprint(f\"\\nTotal: {post_smote_counts['total']} samples\")\n\n# Calculate changes\nsamples_added = post_smote_counts['total'] - pre_smote_counts['total']\nincrease_percentage = (samples_added / pre_smote_counts['total']) * 100\n\nprint(f\"\\nTraining samples: {pre_smote_counts['total']} → {post_smote_counts['total']}\")\nprint(f\"Synthetic samples added: {samples_added} (+{increase_percentage:.1f}%)\")\nprint(f\"Test samples: {len(y_test)} (UNCHANGED)\")\n\n# Calculate NEW imbalance ratio\nnew_counts = list(post_smote_counts['per_class'].values())\nnew_imbalance_ratio = max(new_counts) / min(new_counts)\nprint(f\"\\nImbalance ratio improved: {imbalance_ratio:.1f}:1 → {new_imbalance_ratio:.1f}:1\")\n\n# Synthetic data analysis\nprint(f\"\\nSYNTHETIC DATA ANALYSIS:\")\nprint(\"-\"*80)\ntotal_synthetic = 0\nfor idx, name in enumerate(family_names):\n    original = pre_smote_counts['per_class'][name]\n    current = post_smote_counts['per_class'][name]\n    synthetic = current - original\n    if synthetic > 0:\n        synth_percentage = (synthetic / current) * 100\n        total_synthetic += synthetic\n        print(f\"  {name:15s}: {synthetic:3d} synthetic ({synth_percentage:5.1f}% of class)\")\n\nsynth_overall_percentage = (total_synthetic / post_smote_counts['total']) * 100\nprint(f\"\\nOverall: {total_synthetic} synthetic samples ({synth_overall_percentage:.1f}% of training data)\")\n\n# Visualize SMART SMOTE effect\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(18, 7))\n\n# Before SMOTE\nbefore_data = pd.DataFrame({\n    'Family': family_names,\n    'Count': [pre_smote_counts['per_class'][name] for name in family_names],\n    'Type': ['Original'] * len(family_names)\n}).sort_values('Count', ascending=True)\n\nax1.barh(before_data['Family'], before_data['Count'], color='#e74c3c')\nax1.set_xlabel('Number of Samples', fontsize=12)\nax1.set_title(f'Before: Severely Imbalanced\\nTotal: {pre_smote_counts[\"total\"]} samples\\nImbalance: {imbalance_ratio:.1f}:1', \n              fontsize=14, fontweight='bold')\nax1.grid(axis='x', alpha=0.3)\nax1.set_xlim(0, max(pre_smote_counts['per_class'].values()) * 1.1)\n\n# Add value labels\nfor i, (family, count) in enumerate(zip(before_data['Family'], before_data['Count'])):\n    ax1.text(count + 5, i, f'{count}', va='center', fontweight='bold')\n\n# After SMART SMOTE\nafter_data = pd.DataFrame({\n    'Family': family_names,\n    'Total': [post_smote_counts['per_class'][name] for name in family_names],\n    'Original': [pre_smote_counts['per_class'][name] for name in family_names]\n})\nafter_data['Synthetic'] = after_data['Total'] - after_data['Original']\nafter_data = after_data.sort_values('Total', ascending=True)\n\n# Stacked bar chart\nax2.barh(after_data['Family'], after_data['Original'], color='#3498db', label='Original')\nax2.barh(after_data['Family'], after_data['Synthetic'], left=after_data['Original'], \n         color='#2ecc71', label='Synthetic')\n\nax2.set_xlabel('Number of Samples', fontsize=12)\nax2.set_title(f'After: Strategically Balanced\\nTotal: {post_smote_counts[\"total\"]} samples\\nImbalance: {new_imbalance_ratio:.1f}:1', \n              fontsize=14, fontweight='bold')\nax2.grid(axis='x', alpha=0.3)\nax2.legend(loc='lower right')\n\n# Add value labels with breakdown\nfor i, (family, original, synthetic, total) in enumerate(zip(after_data['Family'], \n                                                              after_data['Original'], \n                                                              after_data['Synthetic'], \n                                                              after_data['Total'])):\n    if synthetic > 0:\n        label = f'{original}+{synthetic}={total}'\n    else:\n        label = f'{total}'\n    ax2.text(total + 5, i, label, va='center', fontsize=9)\n\n# Add median and Q3 lines\nax2.axvline(x=median_count, color='orange', linestyle='--', alpha=0.7, linewidth=1)\nax2.axvline(x=smart_target, color='red', linestyle='--', alpha=0.7, linewidth=1)\nax2.text(median_count, len(family_names)-0.5, f' Median: {median_count}', \n         color='orange', va='center', fontsize=8)\nax2.text(smart_target, len(family_names)-1.5, f' Q3 Target: {smart_target}', \n         color='red', va='center', fontsize=8)\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/strategic_smote_comparison.png', dpi=300, bbox_inches='tight')\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T10:24:10.981619Z","iopub.execute_input":"2026-01-10T10:24:10.982447Z","iopub.status.idle":"2026-01-10T10:24:12.594462Z","shell.execute_reply.started":"2026-01-10T10:24:10.982408Z","shell.execute_reply":"2026-01-10T10:24:12.593363Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# MODELS","metadata":{}},{"cell_type":"code","source":"models = {\n    \"RandomForest\": RandomForestClassifier(\n        n_estimators=600,\n        max_depth=15,\n        min_samples_split=5,\n        min_samples_leaf=2,\n        max_features=\"sqrt\",\n        bootstrap=True,\n        class_weight='balanced', \n        random_state=RANDOM_SEED,  \n        n_jobs=-1\n    ),\n    \n    \"ExtraTrees\": ExtraTreesClassifier(\n        n_estimators=800,\n        max_depth=12,\n        min_samples_split=4,\n        min_samples_leaf=2,\n        max_features=\"sqrt\",\n        class_weight='balanced',  \n        random_state=RANDOM_SEED,  \n        n_jobs=-1\n    ),\n    \n    \"LogReg\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", LogisticRegression(\n            solver=\"lbfgs\",\n            C=2.0,\n            penalty=\"l2\",\n            max_iter=1000,\n            multi_class=\"multinomial\",\n            class_weight='balanced',  \n            n_jobs=-1,\n            random_state=RANDOM_SEED  \n        ))\n    ]),\n    \n    \"SVM-RBF\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", SVC(\n            kernel=\"rbf\",\n            C=10,\n            gamma=\"scale\",\n            class_weight=\"balanced\",\n            probability=True,\n            random_state=RANDOM_SEED  \n        ))\n    ]),\n    \n    \"kNN-15\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", KNeighborsClassifier(\n            n_neighbors=15,\n            weights=\"distance\",\n            metric=\"minkowski\",\n            p=2\n        ))\n    ]),\n    \n    \"GradBoost\": GradientBoostingClassifier(\n        n_estimators=300,\n        learning_rate=0.05,\n        max_depth=3,\n        subsample=0.85,\n        random_state=RANDOM_SEED,  \n        verbose=0\n    ),\n    \n    \"MLP-256x128\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", MLPClassifier(\n            hidden_layer_sizes=(256, 128),\n            activation=\"relu\",\n            solver=\"adam\",\n            alpha=1e-4,\n            learning_rate=\"adaptive\",\n            max_iter=400,\n            early_stopping=True,\n            random_state=RANDOM_SEED,  \n            verbose=False\n        ))\n    ]),\n}\n\n# Add XGBoost if available\nif XGBOOST_AVAILABLE:\n    models[\"XGBoost\"] = XGBClassifier(\n        n_estimators=500,\n        max_depth=7,\n        learning_rate=0.05,\n        subsample=0.8,\n        colsample_bytree=0.8,\n        scale_pos_weight=10,  # ADDED for 10:1 imbalance\n        random_state=RANDOM_SEED,  \n        n_jobs=-1,\n        eval_metric='mlogloss',\n        verbosity=0\n    )\n    print(f\"✅ XGBoost added with seed={RANDOM_SEED}\")\n\n# Add LightGBM if available\nif LGBM_AVAILABLE:\n    models[\"LightGBM\"] = lgb.LGBMClassifier(\n        n_estimators=500,\n        learning_rate=0.05,\n        max_depth=7,\n        num_leaves=31,\n        subsample=0.8,\n        colsample_bytree=0.8,\n        class_weight='balanced',  # ADDED\n        random_state=RANDOM_SEED,\n        n_jobs=-1,\n        verbose=-1\n    )\n    print(f\"✅ LightGBM added with seed={RANDOM_SEED}\")\n\n# Ensemble models\nensemble_models = {\n    \"Ensemble_RF_ET_GB_Hard\": VotingClassifier(\n        estimators=[\n            ('rf', models['RandomForest']),\n            ('et', models['ExtraTrees']),\n            ('gb', models['GradBoost'])\n        ],\n        voting='hard',\n        n_jobs=-1\n    ),\n    \n    \"Ensemble_RF_ET_GB_Soft\": VotingClassifier(\n        estimators=[\n            ('rf', models['RandomForest']),\n            ('et', models['ExtraTrees']),\n            ('gb', models['GradBoost'])\n        ],\n        voting='soft',\n        n_jobs=-1\n    ),\n}\n\nif LGBM_AVAILABLE:\n    ensemble_models[\"Ensemble_RF_ET_LGBM_Soft\"] = VotingClassifier(\n        estimators=[\n            ('rf', models['RandomForest']),\n            ('et', models['ExtraTrees']),\n            ('lgbm', models['LightGBM'])\n        ],\n        voting='soft',\n        n_jobs=-1\n    )\n    \n    ensemble_models[\"Ensemble_ALL_Soft\"] = VotingClassifier(\n        estimators=[\n            ('rf', models['RandomForest']),\n            ('et', models['ExtraTrees']),\n            ('gb', models['GradBoost']),\n            ('lgbm', models['LightGBM'])\n        ],\n        voting='soft',\n        n_jobs=-1\n    )\n\nprint(f\"\\nTotal models defined: {len(models)} individual + {len(ensemble_models)} ensemble\")\nprint(\"All models configured with random_state and class weights for imbalance\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T10:28:54.944967Z","iopub.execute_input":"2026-01-10T10:28:54.945605Z","iopub.status.idle":"2026-01-10T10:28:54.976117Z","shell.execute_reply.started":"2026-01-10T10:28:54.945568Z","shell.execute_reply":"2026-01-10T10:28:54.972931Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# MODEL TRAIN","metadata":{}},{"cell_type":"code","source":"print(f\"Seed: {RANDOM_SEED} - Results should be identical across runs\")\n\nall_models = {**models, **ensemble_models}\nfinal_results = []\noverfitting_analysis = []\n\nfor name, clf in all_models.items():\n    print(f\"\\n{'='*80}\")\n    print(f\"Training {name}...\")\n    print(f\"{'='*80}\")\n    \n    # Train on SMOTE-balanced selected features\n    clf.fit(X_train_resampled, y_train_resampled)\n    \n    \n    # Training set performance\n    y_train_pred = clf.predict(X_train_resampled)\n    train_accuracy = accuracy_score(y_train_resampled, y_train_pred)\n    train_f1 = f1_score(y_train_resampled, y_train_pred, average='weighted', zero_division=0)\n    \n    # Test set performance\n    y_test_pred = clf.predict(X_test_selected)\n    test_accuracy = accuracy_score(y_test, y_test_pred)\n    test_f1 = f1_score(y_test, y_test_pred, average='weighted', zero_division=0)\n    \n    # Calculate overfitting metrics\n    accuracy_gap = train_accuracy - test_accuracy\n    f1_gap = train_f1 - test_f1\n    \n    # Overfitting severity classification\n    if accuracy_gap < 0.02:\n        overfit_level = \"✅ No Overfitting\"\n    elif accuracy_gap < 0.05:\n        overfit_level = \"⚠️ Slight Overfitting\"\n    elif accuracy_gap < 0.10:\n        overfit_level = \"🟠 Moderate Overfitting\"\n    else:\n        overfit_level = \"🔴 Severe Overfitting\"\n    \n    # Calculate all metrics for test set\n    precision_macro = precision_score(y_test, y_test_pred, average='macro', zero_division=0)\n    precision_weighted = precision_score(y_test, y_test_pred, average='weighted', zero_division=0)\n    recall_macro = recall_score(y_test, y_test_pred, average='macro', zero_division=0)\n    recall_weighted = recall_score(y_test, y_test_pred, average='weighted', zero_division=0)\n    f1_macro = f1_score(y_test, y_test_pred, average='macro', zero_division=0)\n    f1_weighted = f1_score(y_test, y_test_pred, average='weighted', zero_division=0)\n    \n    # Store results\n    final_results.append({\n        'Model': name,\n        'Train_Accuracy': train_accuracy,\n        'Test_Accuracy': test_accuracy,\n        'Accuracy_Gap': accuracy_gap,\n        'Train_F1': train_f1,\n        'Test_F1': test_f1,\n        'F1_Gap': f1_gap,\n        'Precision_Macro': precision_macro,\n        'Precision_Weighted': precision_weighted,\n        'Recall_Macro': recall_macro,\n        'Recall_Weighted': recall_weighted,\n        'F1_Macro': f1_macro,\n        'F1_Weighted': f1_weighted,\n        'Overfit_Level': overfit_level\n    })\n    \n    # Print performance summary\n    print(f\"\\n📊 Performance Summary:\")\n    print(f\"   Training Accuracy:  {train_accuracy:.4f} ({train_accuracy*100:.2f}%)\")\n    print(f\"   Test Accuracy:      {test_accuracy:.4f} ({test_accuracy*100:.2f}%)\")\n    print(f\"   Accuracy Gap:       {accuracy_gap:.4f} ({accuracy_gap*100:.2f}%)\")\n    print(f\"   \")\n    print(f\"   Training F1:        {train_f1:.4f}\")\n    print(f\"   Test F1:            {test_f1:.4f}\")\n    print(f\"   F1 Gap:             {f1_gap:.4f}\")\n    print(f\"   \")\n    print(f\"   {overfit_level}\")\n    \n    # Detailed overfitting analysis\n    if accuracy_gap > 0.05:\n        print(f\"\\n   ⚠️ WARNING: Significant train-test gap detected!\")\n        print(f\"   This model may be overfitting the training data.\")\n\nprint(\"\\n\" + \"=\"*80)\nprint(\"✅ ALL MODELS TRAINED AND EVALUATED\")\nprint(\"=\"*80)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T10:31:00.894658Z","iopub.execute_input":"2026-01-10T10:31:00.895640Z","iopub.status.idle":"2026-01-10T10:46:19.757289Z","shell.execute_reply.started":"2026-01-10T10:31:00.895600Z","shell.execute_reply":"2026-01-10T10:46:19.756209Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# RESULT TABLE","metadata":{}},{"cell_type":"code","source":"# Create DataFrame from results\ndf_results = pd.DataFrame(final_results)\n\n# Sort by Test F1-Score (best first)\ndf_results = df_results.sort_values('Test_F1', ascending=False).reset_index(drop=True)\n\nprint(\"=\"*100)\nprint(f\"{'Rank':<6} {'Model':<25} {'Test Acc':>10} {'Test F1':>10} {'Precision':>10} {'Recall':>10} {'F1 Macro':>10}\")\nprint(\"=\"*100)\n\nfor idx, row in df_results.iterrows():\n    print(f\"{idx+1:<6} {row['Model']:<25} \"\n          f\"{row['Test_Accuracy']:>10.4f} \"\n          f\"{row['Test_F1']:>10.4f} \"\n          f\"{row['Precision_Weighted']:>10.4f} \"\n          f\"{row['Recall_Weighted']:>10.4f} \"\n          f\"{row['F1_Macro']:>10.4f}\")\n\nprint(\"=\"*100)\n\n# Add summary statistics\nprint(f\"\\n{'TOP 3 MODELS':^100}\")\nprint(\"-\"*100)\nfor i in range(min(3, len(df_results))):\n    row = df_results.iloc[i]\n    print(f\"{i+1}. {row['Model']:<30} \"\n          f\"Acc: {row['Test_Accuracy']:.4f} | \"\n          f\"F1: {row['Test_F1']:.4f} | \"\n          f\"Prec: {row['Precision_Weighted']:.4f} | \"\n          f\"Rec: {row['Recall_Weighted']:.4f}\")\n\n# Save to CSV\ndf_results.to_csv('/kaggle/working/model_results_detailed.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T10:47:44.286133Z","iopub.execute_input":"2026-01-10T10:47:44.286513Z","iopub.status.idle":"2026-01-10T10:47:44.304943Z","shell.execute_reply.started":"2026-01-10T10:47:44.286488Z","shell.execute_reply":"2026-01-10T10:47:44.303739Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# BEST MODEL","metadata":{}},{"cell_type":"code","source":"best_model_name = df_results.iloc[0]['Model']\nbest_model = all_models[best_model_name]\n\n\nbest_metrics = df_results.iloc[0]\n\nprint(f\"\\nOverall Metrics:\")\nprint(f\"   Test Accuracy:           {best_metrics['Test_Accuracy']:.4f} ({best_metrics['Test_Accuracy']*100:.2f}%)\")\nprint(f\"   Train Accuracy:          {best_metrics['Train_Accuracy']:.4f} ({best_metrics['Train_Accuracy']*100:.2f}%)\")\nprint(f\"   Accuracy Gap:            {best_metrics['Accuracy_Gap']:.4f} ({best_metrics['Accuracy_Gap']*100:.2f}%)\")\nprint(f\"   \")\nprint(f\"   Precision (Weighted):    {best_metrics['Precision_Weighted']:.4f}\")\nprint(f\"   Recall (Weighted):       {best_metrics['Recall_Weighted']:.4f}\")\nprint(f\"   F1-Score (Weighted):     {best_metrics['Test_F1']:.4f}\")\nprint(f\"   F1-Score (Macro):        {best_metrics['F1_Macro']:.4f}\")\nprint(f\"   \")\nprint(f\"   Overfitting Status:      {best_metrics['Overfit_Level']}\")\n\n# Get predictions for best model\ny_pred_best = best_model.predict(X_test_selected)\n\n\nprint(f\"Seed: {RANDOM_SEED} - Report is reproducible\\n\")\nprint(classification_report(y_test, y_pred_best, target_names=family_names, digits=4))\n\n\ncm = confusion_matrix(y_test, y_pred_best)\n\nfig, ax = plt.subplots(figsize=(12, 10))\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=family_names)\ndisp.plot(cmap=\"Blues\", ax=ax, xticks_rotation=45, values_format='d')\n\n# Add title with overfitting info\ngap_percent = best_metrics['Accuracy_Gap'] * 100\ntitle = f\"Confusion Matrix - {best_model_name}\\n\"\ntitle += f\"Test Accuracy: {best_metrics['Test_Accuracy']*100:.2f}% | \"\ntitle += f\"Train-Test Gap: {gap_percent:.2f}% | {best_metrics['Overfit_Level']}\\n\"\ntitle += f\"(Trained on SMOTE + Chi-Square selected features, Seed={RANDOM_SEED})\"\n\nplt.title(title, fontsize=13)\nplt.tight_layout()\nplt.savefig('/kaggle/working/confusion_matrix.png', dpi=300, bbox_inches='tight')\nplt.show()\n\nprint(\"Confusion matrix saved\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T10:47:56.294182Z","iopub.execute_input":"2026-01-10T10:47:56.294617Z","iopub.status.idle":"2026-01-10T10:47:59.349478Z","shell.execute_reply.started":"2026-01-10T10:47:56.294590Z","shell.execute_reply":"2026-01-10T10:47:59.347639Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PER-CLASS PERFORMANCE","metadata":{}},{"cell_type":"code","source":"\nprint(f\"Best Model: {best_model_name}\")\nprint(f\"Seed: {RANDOM_SEED} - Results are reproducible\\n\")\n\nper_class_data = []\n\nfor idx, name in enumerate(family_names):\n    mask = (y_test == idx)\n\n    if mask.sum() > 0:\n        y_true_class = y_test[mask]\n        y_pred_class = y_pred_best[mask]\n\n        correct = (y_pred_class == y_true_class).sum()\n        total = len(y_true_class)\n        class_acc = correct / total\n\n        # Calculate precision, recall, F1 for this class\n        prec = precision_score(\n            y_test, y_pred_best,\n            labels=[idx],\n            average=None,\n            zero_division=0\n        )[0]\n\n        rec = recall_score(\n            y_test, y_pred_best,\n            labels=[idx],\n            average=None,\n            zero_division=0\n        )[0]\n\n        f1 = f1_score(\n            y_test, y_pred_best,\n            labels=[idx],\n            average=None,\n            zero_division=0\n        )[0]\n\n        per_class_data.append({\n            'Malware_Family': name,\n            'Test_Samples': total,\n            'Correct': correct,\n            'Incorrect': total - correct,\n            'Accuracy': class_acc,\n            'Precision': prec,\n            'Recall': rec,\n            'F1_Score': f1\n        })\n\ndf_per_class = pd.DataFrame(per_class_data)\n\n# Display results\nprint(f\"{'Family':<20} {'Samples':>8} {'Correct':>8} {'Wrong':>6} {'Accuracy':>10} {'Precision':>10} {'Recall':>10} {'F1':>10}\")\nprint(\"-\"*100)\n\nfor _, row in df_per_class.iterrows():\n    print(f\"{row['Malware_Family']:<20} {row['Test_Samples']:>8} {row['Correct']:>8} {row['Incorrect']:>6} \"\n          f\"{row['Accuracy']:>10.4f} {row['Precision']:>10.4f} {row['Recall']:>10.4f} {row['F1_Score']:>10.4f}\")\n\nprint(\"=\"*100)\n\n# Save per-class results\ndf_per_class.to_csv(\"/kaggle/working/per_class_performance.csv\", index=False)\nprint(\"\\n✅ Per-class results saved to: per_class_performance.csv\")\n\n\n# Sort by recall (ability to detect)\ndf_per_class_sorted = df_per_class.sort_values('Recall', ascending=True)\n\nprint(\"\\n⚠️ Classes with Lowest Recall (Hardest to Detect):\")\nprint(\"-\"*80)\nfor i, row in enumerate(df_per_class_sorted.head(3).itertuples(), 1):\n    print(f\"   {i}. {row.Malware_Family:<20} | Recall: {row.Recall:.4f} ({row.Recall*100:.2f}%) | \"\n          f\"Missed: {row.Incorrect}/{row.Test_Samples}\")\n\n# Sort by precision (false positive rate)\ndf_per_class_sorted_prec = df_per_class.sort_values('Precision', ascending=True)\n\nprint(\"\\n⚠️ Classes with Lowest Precision (Most False Positives):\")\nprint(\"-\"*80)\nfor i, row in enumerate(df_per_class_sorted_prec.head(3).itertuples(), 1):\n    print(f\"   {i}. {row.Malware_Family:<20} | Precision: {row.Precision:.4f} ({row.Precision*100:.2f}%)\")\n\n# Highlight minority class performance\n\nprint(\"These classes had very few training samples before SMOTE:\")\n\nminority_classes = ['Simda', 'Vundo', 'Kelihos_ver1', 'Tracur']\nfor name in minority_classes:\n    row = df_per_class[df_per_class['Malware_Family'] == name]\n    if not row.empty:\n        row = row.iloc[0]\n        original_train_count = pre_smote_counts['per_class'][name]\n        smote_train_count = post_smote_counts['per_class'][name]\n        \n        print(f\"\\n{name}:\")\n        print(f\"   Original training samples: {original_train_count}\")\n        print(f\"   After SMOTE: {smote_train_count} (+{smote_train_count - original_train_count})\")\n        print(f\"   Test samples: {row['Test_Samples']}\")\n        print(f\"   Test Recall: {row['Recall']:.4f} ({row['Recall']*100:.2f}%)\")\n        print(f\"   Correctly detected: {row['Correct']}/{row['Test_Samples']}\")\n        \n        if original_train_count < 50:\n            if row['Recall'] > 0.8:\n                print(f\"   EXCELLENT - SMOTE successfully enabled detection!\")\n            elif row['Recall'] > 0.5:\n                print(f\"   MODERATE - SMOTE helped but still challenging\")\n            else:\n                print(f\"   POOR - Class remains difficult despite SMOTE\")\n\nprint(\"\\n\" + \"=\"*80)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T10:48:23.791970Z","iopub.execute_input":"2026-01-10T10:48:23.793499Z","iopub.status.idle":"2026-01-10T10:48:23.921685Z","shell.execute_reply.started":"2026-01-10T10:48:23.793452Z","shell.execute_reply":"2026-01-10T10:48:23.920157Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SHAP","metadata":{}},{"cell_type":"code","source":"import shap\nimport matplotlib.pyplot as plt\nimport numpy as np\n\nprint(\"\\n1. GLOBAL FEATURE IMPORTANCE\")\nprint(\"-\" * 50)\n\n# Use RandomForest instead of ensemble for SHAP compatibility\nbest_model = models[\"RandomForest\"]\n\n# Prepare data\nif hasattr(best_model, 'named_steps'):\n    model_for_shap = best_model.named_steps['clf']\n    X_test_shap = best_model.named_steps['scaler'].transform(X_test_selected)\nelse:\n    model_for_shap = best_model\n    X_test_shap = X_test_selected\n\n# Get feature names\nfeature_names = [f\"Byte_{i}\" for i in range(X_test_selected.shape[1])]\n\n# Create explainer\nexplainer = shap.TreeExplainer(model_for_shap)\n\n# Compute SHAP values for samples\nn_samples = min(100, len(X_test_shap))\nX_sample = X_test_shap[:n_samples]\nshap_values = explainer.shap_values(X_sample)\n\n# Handle SHAP values format\nif isinstance(shap_values, list):\n    shap_importance = np.abs(np.array(shap_values)).mean(axis=(0, 1))\nelse:\n    if shap_values.ndim == 3:\n        shap_importance = np.abs(shap_values).mean(axis=(0, 2))\n    else:\n        shap_importance = np.abs(shap_values).mean(axis=0)\n\n# Ensure we have correct number of features\nif len(shap_importance) != len(feature_names):\n    min_len = min(len(shap_importance), len(feature_names))\n    shap_importance = shap_importance[:min_len]\n    feature_names = feature_names[:min_len]\n\n# Global importance plot\nplt.figure(figsize=(12, 8))\n\n# Top features\ntop_n = min(15, len(shap_importance))\ntop_indices = np.argsort(shap_importance)[-top_n:][::-1]\ntop_features = [feature_names[i] for i in top_indices]\ntop_importance = [shap_importance[i] for i in top_indices]\n\n# Plot\nplt.barh(range(len(top_features)), top_importance, color='skyblue')\nplt.yticks(range(len(top_features)), top_features)\nplt.xlabel('Mean |SHAP Value| (Impact on Prediction)', fontsize=12)\nplt.title('Top 15 Most Important Byte Positions\\n(Global Feature Importance)', \n         fontsize=14, fontweight='bold', pad=20)\nplt.gca().invert_yaxis()\nplt.grid(axis='x', alpha=0.3)\nplt.tight_layout()\nplt.savefig('/kaggle/working/shap_global_importance_simple.png', dpi=300, bbox_inches='tight')\nplt.show()\n\nprint(\"Saved: shap_global_importance_simple.png\")\n\nprint(\"\\n2. WATERFALL PLOT - CORRECTLY CLASSIFIED SAMPLE\")\nprint(\"-\" * 50)\n\n# Find a correctly classified sample\ny_test_pred = best_model.predict(X_test_selected)\ncorrect_indices = np.where(y_test_pred == y_test)[0]\n\nif len(correct_indices) > 0:\n    sample_idx = correct_indices[0]\n    true_label = y_test.iloc[sample_idx] if hasattr(y_test, 'iloc') else y_test[sample_idx]\n    true_class_name = family_names[true_label]\n    \n    print(f\"Sample Index: {sample_idx}\")\n    print(f\"True Class: {true_class_name}\")\n    \n    # Get SHAP values for this sample\n    X_single = X_test_shap[sample_idx:sample_idx+1]\n    \n    # Recompute SHAP for single sample\n    sample_shap = explainer.shap_values(X_single)\n    \n    if isinstance(sample_shap, list):\n        sample_shap_values = sample_shap[true_label][0]\n        base_value = explainer.expected_value[true_label]\n    else:\n        if sample_shap.ndim == 3:\n            sample_shap_values = sample_shap[0, :, true_label]\n        else:\n            sample_shap_values = sample_shap[0, :]\n        base_value = explainer.expected_value[true_label] if isinstance(explainer.expected_value, (list, np.ndarray)) else explainer.expected_value\n    \n    # Create waterfall plot\n    plt.figure(figsize=(12, 6))\n    \n    explanation = shap.Explanation(\n        values=sample_shap_values,\n        base_values=base_value,\n        data=X_single.flatten(),\n        feature_names=feature_names\n    )\n    \n    shap.plots.waterfall(explanation, max_display=15, show=False)\n    \n    plt.title(f'Correct Prediction: {true_class_name}\\nFeature Contributions', \n             fontsize=14, fontweight='bold', pad=20)\n    plt.xlabel('SHAP Value (Impact on Model Output)', fontsize=11)\n    plt.tight_layout()\n    plt.savefig('/kaggle/working/shap_waterfall_correct.png', dpi=300, bbox_inches='tight')\n    plt.show()\n    \n    print(\"Saved: shap_waterfall_correct.png\")\n    \nelse:\n    print(\"No correctly classified samples found\")\n\nprint(\"\\n3. WATERFALL PLOT - MISCLASSIFIED SAMPLE\")\nprint(\"-\" * 50)\n\n# Find a misclassified sample\nwrong_indices = np.where(y_test_pred != y_test)[0]\n\nif len(wrong_indices) > 0:\n    sample_idx = wrong_indices[0]\n    true_label = y_test.iloc[sample_idx] if hasattr(y_test, 'iloc') else y_test[sample_idx]\n    pred_label = y_test_pred[sample_idx]\n    \n    true_class_name = family_names[true_label]\n    pred_class_name = family_names[pred_label]\n    \n    print(f\"Sample Index: {sample_idx}\")\n    print(f\"True Class: {true_class_name}\")\n    print(f\"Predicted Class: {pred_class_name}\")\n    \n    # Get SHAP values for this sample\n    X_single = X_test_shap[sample_idx:sample_idx+1]\n    sample_shap = explainer.shap_values(X_single)\n    \n    if isinstance(sample_shap, list):\n        sample_shap_values = sample_shap[pred_label][0]\n        base_value = explainer.expected_value[pred_label]\n    else:\n        if sample_shap.ndim == 3:\n            sample_shap_values = sample_shap[0, :, pred_label]\n        else:\n            sample_shap_values = sample_shap[0, :]\n        base_value = explainer.expected_value[pred_label] if isinstance(explainer.expected_value, (list, np.ndarray)) else explainer.expected_value\n    \n    # Create waterfall plot\n    plt.figure(figsize=(12, 6))\n    \n    explanation = shap.Explanation(\n        values=sample_shap_values,\n        base_values=base_value,\n        data=X_single.flatten(),\n        feature_names=feature_names\n    )\n    \n    shap.plots.waterfall(explanation, max_display=15, show=False)\n    \n    plt.title(f'Misclassification: {true_class_name} as {pred_class_name}\\nFeature Contributions', \n             fontsize=14, fontweight='bold', pad=20)\n    plt.xlabel('SHAP Value (Impact on Model Output)', fontsize=11)\n    plt.tight_layout()\n    plt.savefig('/kaggle/working/shap_waterfall_wrong.png', dpi=300, bbox_inches='tight')\n    plt.show()\n    \n    print(\"Saved: shap_waterfall_wrong.png\")\n    \nelse:\n    print(\"All samples correctly classified\")\n\nprint(\"\\nSHAP analysis complete.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T10:55:36.786305Z","iopub.execute_input":"2026-01-10T10:55:36.786779Z","iopub.status.idle":"2026-01-10T10:55:45.177514Z","shell.execute_reply.started":"2026-01-10T10:55:36.786747Z","shell.execute_reply":"2026-01-10T10:55:45.175712Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# MODEL COMPARISON","metadata":{}},{"cell_type":"code","source":"# Simple Model Comparison\nfig, axes = plt.subplots(2, 2, figsize=(16, 12))\n\nmetrics_to_plot = [\n    ('Test_Accuracy', 'Test Accuracy'),\n    ('Precision_Weighted', 'Precision (Weighted)'),\n    ('Recall_Weighted', 'Recall (Weighted)'),\n    ('Test_F1', 'F1-Score (Weighted)')\n]\n\nfor idx, (col_name, display_name) in enumerate(metrics_to_plot):\n    ax = axes[idx // 2, idx % 2]\n    \n    data = df_results.sort_values(col_name, ascending=True)\n    \n    ax.barh(data['Model'], data[col_name], color='steelblue')\n    ax.set_xlabel(display_name, fontsize=12)\n    ax.set_title(f'{display_name} Comparison', fontsize=14, fontweight='bold')\n    ax.set_xlim([0.7, 1.05])\n    \n    for i, v in enumerate(data[col_name]):\n        ax.text(v + 0.01, i, f'{v:.3f}', va='center', fontsize=9)\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/model_comparison.png', dpi=300, bbox_inches='tight')\nplt.show()\n\nprint(\"Model Performance Comparison Complete\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T10:58:30.106801Z","iopub.execute_input":"2026-01-10T10:58:30.108027Z","iopub.status.idle":"2026-01-10T10:58:33.043567Z","shell.execute_reply.started":"2026-01-10T10:58:30.107983Z","shell.execute_reply":"2026-01-10T10:58:33.042262Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# TESTING MORE EXPERIMENTS BASED ON THE BASELINE MODEL","metadata":{}},{"cell_type":"code","source":"# ============================================================================\n# 16. ADVANCED EXPERIMENTS WITH REPRODUCIBILITY\n# ============================================================================\n\nprint(\"\\n\" + \"=\"*80*2)\nprint(\"=\"*80*2)\nprint(\"🧪 ADVANCED EXPERIMENTS - REPRODUCIBLE ALTERNATIVE APPROACHES\")\nprint(\"=\"*80*2)\nprint(\"=\"*80*2)\nprint(f\"Seed: {RANDOM_SEED} - All experiments are reproducible\")\n\n# Store baseline for comparison\nbaseline_acc = best_metrics['Test_Accuracy']\nbaseline_f1 = best_metrics['Test_F1']\nbaseline_gap = best_metrics['Accuracy_Gap']\n\nexperimental_results = []\n\n# Add baseline\nexperimental_results.append({\n    'Method': f'Baseline ({best_model_name})',\n    'Train_Accuracy': best_metrics['Train_Accuracy'],\n    'Test_Accuracy': baseline_acc,\n    'Accuracy_Gap': baseline_gap,\n    'Train_F1': best_metrics['Train_F1'],\n    'Test_F1': baseline_f1,\n    'F1_Gap': best_metrics['F1_Gap'],\n    'Gain_vs_Baseline': 0.0,\n    'Overfit_Status': best_metrics['Overfit_Level']\n})\n\n# ============================================================================\n# EXPERIMENT 1: ADASYN (Adaptive Synthetic Sampling)\n# ============================================================================\n\nprint(\"\\n\" + \"=\"*80)\nprint(\"🧪 EXPERIMENT 1: ADASYN - Adaptive Synthetic Sampling\")\nprint(\"=\"*80)\nprint(\"CONCEPT: Generates more synthetic samples for harder-to-learn examples\")\nprint(\"vs SMOTE: SMOTE generates uniformly, ADASYN focuses on boundary cases\")\nprint(f\"Seed: {RANDOM_SEED}\")\n\ntry:\n    adasyn = ADASYN(random_state=RANDOM_SEED, n_neighbors=5)\n    X_train_adasyn, y_train_adasyn = adasyn.fit_resample(X_train_selected, y_train)\n    \n    print(f\"\\nTraining samples: {len(y_train)} → {len(y_train_adasyn)}\")\n    \n    # Train RandomForest with ADASYN data\n    rf_adasyn = RandomForestClassifier(\n        n_estimators=600, max_depth=15, min_samples_split=5,\n        min_samples_leaf=2, random_state=RANDOM_SEED, n_jobs=-1\n    )\n    \n    print(\"Training RandomForest with ADASYN...\")\n    rf_adasyn.fit(X_train_adasyn, y_train_adasyn)\n    \n    # Training performance\n    y_train_pred_adasyn = rf_adasyn.predict(X_train_adasyn)\n    train_acc_adasyn = accuracy_score(y_train_adasyn, y_train_pred_adasyn)\n    train_f1_adasyn = f1_score(y_train_adasyn, y_train_pred_adasyn, average='weighted')\n    \n    # Test performance\n    y_test_pred_adasyn = rf_adasyn.predict(X_test_selected)\n    test_acc_adasyn = accuracy_score(y_test, y_test_pred_adasyn)\n    test_f1_adasyn = f1_score(y_test, y_test_pred_adasyn, average='weighted')\n    \n    # Overfitting metrics\n    gap_adasyn = train_acc_adasyn - test_acc_adasyn\n    f1_gap_adasyn = train_f1_adasyn - test_f1_adasyn\n    \n    if gap_adasyn < 0.02:\n        overfit_adasyn = \"✅ No Overfitting\"\n    elif gap_adasyn < 0.05:\n        overfit_adasyn = \"⚠️ Slight Overfitting\"\n    elif gap_adasyn < 0.10:\n        overfit_adasyn = \"🟠 Moderate Overfitting\"\n    else:\n        overfit_adasyn = \"🔴 Severe Overfitting\"\n    \n    print(f\"\\n✅ ADASYN Results:\")\n    print(f\"   Train Accuracy:   {train_acc_adasyn:.4f} ({train_acc_adasyn*100:.2f}%)\")\n    print(f\"   Test Accuracy:    {test_acc_adasyn:.4f} ({test_acc_adasyn*100:.2f}%)\")\n    print(f\"   Accuracy Gap:     {gap_adasyn:.4f} ({gap_adasyn*100:.2f}%)\")\n    print(f\"   Test F1-Score:    {test_f1_adasyn:.4f}\")\n    print(f\"   Overfitting:      {overfit_adasyn}\")\n    print(f\"   vs Baseline:      {(test_acc_adasyn - baseline_acc)*100:+.2f}%\")\n    \n    experimental_results.append({\n        'Method': 'ADASYN Sampling',\n        'Train_Accuracy': train_acc_adasyn,\n        'Test_Accuracy': test_acc_adasyn,\n        'Accuracy_Gap': gap_adasyn,\n        'Train_F1': train_f1_adasyn,\n        'Test_F1': test_f1_adasyn,\n        'F1_Gap': f1_gap_adasyn,\n        'Gain_vs_Baseline': (test_acc_adasyn - baseline_acc)*100,\n        'Overfit_Status': overfit_adasyn\n    })\nexcept Exception as e:\n    print(f\"❌ ADASYN failed: {e}\")\n\n# ============================================================================\n# EXPERIMENT 2: SMOTE + TOMEK LINKS (HYBRID)\n# ============================================================================\n\nprint(\"\\n\" + \"=\"*80)\nprint(\"🧪 EXPERIMENT 2: SMOTE-TOMEK (Hybrid Approach)\")\nprint(\"=\"*80)\nprint(\"CONCEPT: SMOTE oversample + Tomek Links clean overlapping samples\")\nprint(\"WHY: Removes ambiguous samples at class boundaries\")\nprint(f\"Seed: {RANDOM_SEED}\")\n\ntry:\n    smote_tomek = SMOTETomek(random_state=RANDOM_SEED)\n    X_train_hybrid, y_train_hybrid = smote_tomek.fit_resample(X_train_selected, y_train)\n    \n    print(f\"\\nTraining samples: {len(y_train)} → {len(y_train_hybrid)}\")\n    \n    rf_hybrid = RandomForestClassifier(\n        n_estimators=600, max_depth=15, min_samples_split=5,\n        min_samples_leaf=2, random_state=RANDOM_SEED, n_jobs=-1\n    )\n    \n    print(\"Training RandomForest with SMOTE-Tomek...\")\n    rf_hybrid.fit(X_train_hybrid, y_train_hybrid)\n    \n    # Training performance\n    y_train_pred_hybrid = rf_hybrid.predict(X_train_hybrid)\n    train_acc_hybrid = accuracy_score(y_train_hybrid, y_train_pred_hybrid)\n    train_f1_hybrid = f1_score(y_train_hybrid, y_train_pred_hybrid, average='weighted')\n    \n    # Test performance\n    y_test_pred_hybrid = rf_hybrid.predict(X_test_selected)\n    test_acc_hybrid = accuracy_score(y_test, y_test_pred_hybrid)\n    test_f1_hybrid = f1_score(y_test, y_test_pred_hybrid, average='weighted')\n    \n    gap_hybrid = train_acc_hybrid - test_acc_hybrid\n    f1_gap_hybrid = train_f1_hybrid - test_f1_hybrid\n    \n    if gap_hybrid < 0.02:\n        overfit_hybrid = \"✅ No Overfitting\"\n    elif gap_hybrid < 0.05:\n        overfit_hybrid = \"⚠️ Slight Overfitting\"\n    elif gap_hybrid < 0.10:\n        overfit_hybrid = \"🟠 Moderate Overfitting\"\n    else:\n        overfit_hybrid = \"🔴 Severe Overfitting\"\n    \n    print(f\"\\n✅ SMOTE-Tomek Results:\")\n    print(f\"   Train Accuracy:   {train_acc_hybrid:.4f} ({train_acc_hybrid*100:.2f}%)\")\n    print(f\"   Test Accuracy:    {test_acc_hybrid:.4f} ({test_acc_hybrid*100:.2f}%)\")\n    print(f\"   Accuracy Gap:     {gap_hybrid:.4f} ({gap_hybrid*100:.2f}%)\")\n    print(f\"   Test F1-Score:    {test_f1_hybrid:.4f}\")\n    print(f\"   Overfitting:      {overfit_hybrid}\")\n    print(f\"   vs Baseline:      {(test_acc_hybrid - baseline_acc)*100:+.2f}%\")\n    \n    experimental_results.append({\n        'Method': 'SMOTE-Tomek Hybrid',\n        'Train_Accuracy': train_acc_hybrid,\n        'Test_Accuracy': test_acc_hybrid,\n        'Accuracy_Gap': gap_hybrid,\n        'Train_F1': train_f1_hybrid,\n        'Test_F1': test_f1_hybrid,\n        'F1_Gap': f1_gap_hybrid,\n        'Gain_vs_Baseline': (test_acc_hybrid - baseline_acc)*100,\n        'Overfit_Status': overfit_hybrid\n    })\nexcept Exception as e:\n    print(f\"❌ SMOTE-Tomek failed: {e}\")\n\n# ============================================================================\n# EXPERIMENT 3: XGBOOST WITH CLASS WEIGHTS\n# ============================================================================\n\nprint(\"\\n\" + \"=\"*80)\nprint(\"🧪 EXPERIMENT 3: XGBoost with Sample Weighting\")\nprint(\"=\"*80)\nprint(\"CONCEPT: Use gradient boosting with class-aware weighting\")\nprint(\"WHY: XGBoost can handle imbalance natively without synthetic samples\")\nprint(f\"Seed: {RANDOM_SEED}\")\n\nif XGBOOST_AVAILABLE:\n    try:\n        # Calculate sample weights\n        class_counts = Counter(y_train)\n        max_count = max(class_counts.values())\n        sample_weights = np.array([max_count / class_counts[y] for y in y_train])\n        \n        xgb_clf = XGBClassifier(\n            n_estimators=500,\n            max_depth=7,\n            learning_rate=0.05,\n            subsample=0.8,\n            colsample_bytree=0.8,\n            random_state=RANDOM_SEED,\n            n_jobs=-1,\n            eval_metric='mlogloss',\n            verbosity=0\n        )\n        \n        print(\"Training XGBoost with weighted samples...\")\n        xgb_clf.fit(X_train_selected, y_train, sample_weight=sample_weights)\n        \n        # Training performance\n        y_train_pred_xgb = xgb_clf.predict(X_train_selected)\n        train_acc_xgb = accuracy_score(y_train, y_train_pred_xgb)\n        train_f1_xgb = f1_score(y_train, y_train_pred_xgb, average='weighted')\n        \n        # Test performance\n        y_test_pred_xgb = xgb_clf.predict(X_test_selected)\n        test_acc_xgb = accuracy_score(y_test, y_test_pred_xgb)\n        test_f1_xgb = f1_score(y_test, y_test_pred_xgb, average='weighted')\n        \n        gap_xgb = train_acc_xgb - test_acc_xgb\n        f1_gap_xgb = train_f1_xgb - test_f1_xgb\n        \n        if gap_xgb < 0.02:\n            overfit_xgb = \"✅ No Overfitting\"\n        elif gap_xgb < 0.05:\n            overfit_xgb = \"⚠️ Slight Overfitting\"\n        elif gap_xgb < 0.10:\n            overfit_xgb = \"🟠 Moderate Overfitting\"\n        else:\n            overfit_xgb = \"🔴 Severe Overfitting\"\n        \n        print(f\"\\n✅ XGBoost Results:\")\n        print(f\"   Train Accuracy:   {train_acc_xgb:.4f} ({train_acc_xgb*100:.2f}%)\")\n        print(f\"   Test Accuracy:    {test_acc_xgb:.4f} ({test_acc_xgb*100:.2f}%)\")\n        print(f\"   Accuracy Gap:     {gap_xgb:.4f} ({gap_xgb*100:.2f}%)\")\n        print(f\"   Test F1-Score:    {test_f1_xgb:.4f}\")\n        print(f\"   Overfitting:      {overfit_xgb}\")\n        print(f\"   vs Baseline:      {(test_acc_xgb - baseline_acc)*100:+.2f}%\")\n        print(f\"   Advantage:        No synthetic data needed!\")\n        \n        experimental_results.append({\n            'Method': 'XGBoost + Weights',\n            'Train_Accuracy': train_acc_xgb,\n            'Test_Accuracy': test_acc_xgb,\n            'Accuracy_Gap': gap_xgb,\n            'Train_F1': train_f1_xgb,\n            'Test_F1': test_f1_xgb,\n            'F1_Gap': f1_gap_xgb,\n            'Gain_vs_Baseline': (test_acc_xgb - baseline_acc)*100,\n            'Overfit_Status': overfit_xgb\n        })\n    except Exception as e:\n        print(f\"❌ XGBoost failed: {e}\")\nelse:\n    print(\"❌ XGBoost not available - skipping experiment\")\n\nprint(\"\\n\" + \"=\"*80)\nprint(\"✅ ALL EXPERIMENTS COMPLETE\")\nprint(\"=\"*80)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T10:59:03.790744Z","iopub.execute_input":"2026-01-10T10:59:03.791258Z","iopub.status.idle":"2026-01-10T10:59:32.634259Z","shell.execute_reply.started":"2026-01-10T10:59:03.791188Z","shell.execute_reply":"2026-01-10T10:59:32.633468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create DataFrame and sort by Test Accuracy\ndf_experiments = pd.DataFrame(experimental_results).sort_values('Test_Accuracy', ascending=False)\n\n# Define column widths\ncol_rank = 6\ncol_method = 40\ncol_test_acc = 15\ncol_test_f1 = 10\n\n# Print table header\nprint(\"=\" * (col_rank + col_method + col_test_acc + col_test_f1 + 10))\nprint(f\"{'Rank':<{col_rank}} {'Method':<{col_method}} {'Test Accuracy':>{col_test_acc}} {'Test F1':>{col_test_f1}}\")\nprint(\"=\" * (col_rank + col_method + col_test_acc + col_test_f1 + 10))\n\n# Print table rows\nfor rank, row in enumerate(df_experiments.itertuples(), 1):\n    print(f\"{rank:<{col_rank}} {row.Method:<{col_method}} {row.Test_Accuracy*100:>{col_test_acc-1}.2f}% {row.Test_F1:>{col_test_f1}.4f}\")\n\nprint(\"=\" * (col_rank + col_method + col_test_acc + col_test_f1 + 10))\n\n# Save experimental results\ndf_experiments.to_csv(\"/kaggle/working/experimental_results.csv\", index=False)\nprint(\"\\nExperimental results saved to: experimental_results.csv\")\n\n# Find best approach\nbest_experiment = df_experiments.iloc[0]\n\nprint(\"\\n\" + \"=\"*80)\nprint(\"WINNER: \" + best_experiment['Method'])\nprint(\"=\"*80)\nprint(f\"   Test Accuracy: {best_experiment['Test_Accuracy']*100:.2f}%\")\nprint(f\"   Test F1-Score: {best_experiment['Test_F1']:.4f}\")\n\nprint(\"STATISTICAL INSIGHTS\")\nprint(\"=\"*80)\n\nprint(f\"\\n📊 Across {len(experimental_results)} approaches:\")\nprint(f\"   Average test accuracy: {df_experiments['Test_Accuracy'].mean()*100:.2f}%\")\nprint(f\"   Best test accuracy:    {df_experiments['Test_Accuracy'].max()*100:.2f}%\")\nprint(f\"   Worst test accuracy:   {df_experiments['Test_Accuracy'].min()*100:.2f}%\")\nprint(f\"   Standard deviation:    {df_experiments['Test_Accuracy'].std()*100:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T10:59:56.480893Z","iopub.execute_input":"2026-01-10T10:59:56.481263Z","iopub.status.idle":"2026-01-10T10:59:56.501288Z","shell.execute_reply.started":"2026-01-10T10:59:56.481209Z","shell.execute_reply":"2026-01-10T10:59:56.500193Z"}},"outputs":[],"execution_count":null}]}