{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":4117,"databundleVersionId":46665,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\ndata_path = \"/kaggle/input/malware-classification\"\n\n# List available files\nprint(os.listdir(data_path))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:42:47.738707Z","iopub.execute_input":"2025-08-21T08:42:47.738877Z","iopub.status.idle":"2025-08-21T08:42:47.747124Z","shell.execute_reply.started":"2025-08-21T08:42:47.738860Z","shell.execute_reply":"2025-08-21T08:42:47.746400Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Path to the dataset\ndata_path = \"/kaggle/input/malware-classification\"\n\n# Load the labels\nlabels_df = pd.read_csv(f\"{data_path}/trainLabels.csv\")\n\n# Show first 5 rows\nlabels_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:42:52.963794Z","iopub.execute_input":"2025-08-21T08:42:52.964579Z","iopub.status.idle":"2025-08-21T08:42:53.278884Z","shell.execute_reply.started":"2025-08-21T08:42:52.964542Z","shell.execute_reply":"2025-08-21T08:42:53.278062Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndata_path = \"/kaggle/input/malware-classification\"\ndf = pd.read_csv(f\"{data_path}/trainLabels.csv\")\ndf.head(10)  # shows 10 samples with Id and Class","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:42:54.527161Z","iopub.execute_input":"2025-08-21T08:42:54.527596Z","iopub.status.idle":"2025-08-21T08:42:54.544232Z","shell.execute_reply.started":"2025-08-21T08:42:54.527573Z","shell.execute_reply":"2025-08-21T08:42:54.543569Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Mapping class numbers to family names\nfamily_map = {\n    1: \"Ramnit\",\n    2: \"Lollipop\",\n    3: \"Kelihos_ver3\",\n    4: \"Vundo\",\n    5: \"Simda\",\n    6: \"Tracur\",\n    7: \"Kelihos_ver1\",\n    8: \"Obfuscator.ACY\",\n    9: \"Gatak\"\n}\n\n# Add a new column for family name\nlabels_df[\"Family\"] = labels_df[\"Class\"].map(family_map)\n\n# Preview updated labels\nlabels_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:42:59.681202Z","iopub.execute_input":"2025-08-21T08:42:59.681824Z","iopub.status.idle":"2025-08-21T08:42:59.699394Z","shell.execute_reply.started":"2025-08-21T08:42:59.681802Z","shell.execute_reply":"2025-08-21T08:42:59.698708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load labels\ndata_path = \"/kaggle/input/malware-classification\"\ndf = pd.read_csv(f\"{data_path}/trainLabels.csv\")\n\n# Optional: Map class to family names\nfamily_map = {\n    1: \"Ramnit\", 2: \"Lollipop\", 3: \"Kelihos_ver3\", 4: \"Vundo\",\n    5: \"Simda\", 6: \"Tracur\", 7: \"Kelihos_ver1\", 8: \"Obfuscator.ACY\", 9: \"Gatak\"\n}\ndf[\"Family\"] = df[\"Class\"].map(family_map)\n\n# Smart sampling: get min(10, count) rows per class\nsampled_df = df.groupby(\"Class\", group_keys=False).apply(\n    lambda x: x.sample(n=min(10, len(x)), random_state=42)\n).reset_index(drop=True)\n\n# Show count per class in the sample\nprint(sampled_df[\"Class\"].value_counts())\n\n# Preview result\nsampled_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:43:14.369943Z","iopub.execute_input":"2025-08-21T08:43:14.370488Z","iopub.status.idle":"2025-08-21T08:43:14.412456Z","shell.execute_reply.started":"2025-08-21T08:43:14.370460Z","shell.execute_reply":"2025-08-21T08:43:14.411906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for cls in sorted(sampled_df[\"Class\"].unique()):\n    ids = sampled_df[sampled_df[\"Class\"] == cls][\"Id\"].tolist()\n    print(f\"Class {cls} Sample IDs:\\n\", ids, \"\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:43:27.778714Z","iopub.execute_input":"2025-08-21T08:43:27.778976Z","iopub.status.idle":"2025-08-21T08:43:27.789885Z","shell.execute_reply.started":"2025-08-21T08:43:27.778957Z","shell.execute_reply":"2025-08-21T08:43:27.789070Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels_df[\"Family\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:43:29.076317Z","iopub.execute_input":"2025-08-21T08:43:29.076565Z","iopub.status.idle":"2025-08-21T08:43:29.083520Z","shell.execute_reply.started":"2025-08-21T08:43:29.076547Z","shell.execute_reply":"2025-08-21T08:43:29.082780Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(10, 6))\nsns.countplot(data=labels_df, y=\"Family\", order=labels_df[\"Family\"].value_counts().index)\nplt.title(\"Malware Family Distribution\")\nplt.xlabel(\"Number of Samples\")\nplt.ylabel(\"Malware Family\")\nplt.show()\n\nimport os\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\ndef extract_byte_histogram(file_path):\n    counts = np.zeros(256, dtype=int)\n    try:\n        with open(file_path, 'r') as f:\n            for line in f:\n                parts = line.strip().split()[1:]  # skip the address\n                for byte_str in parts:\n                    if byte_str != '??':\n                        try:\n                            byte_val = int(byte_str, 16)\n                            counts[byte_val] += 1\n                        except ValueError:\n                            continue\n    except Exception as e:\n        print(f\"Error reading {file_path}: {e}\")\n    return counts\n\n# Path to dataset\ndata_path = \"/kaggle/input/malware-classification\"\nbytes_path = os.path.join(data_path, \"train\")\n\n# Load labels\nlabel_df = pd.read_csv(os.path.join(data_path, \"trainLabels.csv\"))\nsample_ids = label_df[\"Id\"].tolist()[:100]  # First 100 samples for demo\nlabel_map = dict(zip(label_df[\"Id\"], label_df[\"Class\"]))\n\nX = []\ny = []\nfile_names = []\n\nfor file_id in tqdm(sample_ids):\n    file_path = os.path.join(bytes_path, file_id + \".bytes\")\n    if os.path.exists(file_path):\n        hist = extract_byte_histogram(file_path)\n        X.append(hist)\n        y.append(label_map[file_id])\n        file_names.append(file_id)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:43:34.431504Z","iopub.execute_input":"2025-08-21T08:43:34.431806Z","iopub.status.idle":"2025-08-21T08:43:35.559379Z","shell.execute_reply.started":"2025-08-21T08:43:34.431783Z","shell.execute_reply":"2025-08-21T08:43:35.558691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create the feature DataFrame\ndf_features = pd.DataFrame(X, columns=[f'byte_{i:02X}' for i in range(256)])\ndf_features[\"label\"] = y\ndf_features[\"Id\"] = file_names\n\n# Map label → malware family\nfamily_map = {\n    1: \"Ramnit\", 2: \"Lollipop\", 3: \"Kelihos_ver3\", 4: \"Vundo\",\n    5: \"Simda\", 6: \"Tracur\", 7: \"Kelihos_ver1\", 8: \"Obfuscator.ACY\", 9: \"Gatak\"\n}\ndf_features[\"Family\"] = df_features[\"label\"].map(family_map)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:43:37.183618Z","iopub.execute_input":"2025-08-21T08:43:37.183931Z","iopub.status.idle":"2025-08-21T08:43:37.193209Z","shell.execute_reply.started":"2025-08-21T08:43:37.183910Z","shell.execute_reply":"2025-08-21T08:43:37.192529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count NaN values\nprint(\"Total NaNs:\", df_features.isna().sum().sum())\n\n# Check data types\nprint(\"Data types:\\n\", df_features.dtypes.value_counts())\n\n# Preview suspicious rows\nprint(df_features[df_features.isna().any(axis=1)].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:43:40.651142Z","iopub.execute_input":"2025-08-21T08:43:40.651375Z","iopub.status.idle":"2025-08-21T08:43:40.662050Z","shell.execute_reply.started":"2025-08-21T08:43:40.651359Z","shell.execute_reply":"2025-08-21T08:43:40.661213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!mkdir -p /kaggle/working/bytes\n!7z l /kaggle/input/malware-classification/train.7z | grep '.bytes' | awk '{print $NF}' | head -n 2000 > /kaggle/working/bytes/bytes_list.txt\n\n# Now extract just these 2000 files\n!7z e /kaggle/input/malware-classification/train.7z -o/kaggle/working/bytes -i@/kaggle/working/bytes/bytes_list.txt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:43:45.562750Z","iopub.execute_input":"2025-08-21T08:43:45.562988Z","iopub.status.idle":"2025-08-21T08:45:50.556199Z","shell.execute_reply.started":"2025-08-21T08:43:45.562972Z","shell.execute_reply":"2025-08-21T08:45:50.555470Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nfrom tqdm import tqdm\n\ndef extract_byte_histogram(file_path):\n    try:\n        with open(file_path, 'r') as file:\n            hex_lines = file.readlines()\n        bytes_list = []\n        for line in hex_lines:\n            parts = line.strip().split()\n            bytes_seq = parts[1:]  # ignore address part\n            bytes_list.extend([b for b in bytes_seq if b != '??'])\n        byte_vals = [int(b, 16) for b in bytes_list if len(b) == 2]\n        hist = np.histogram(byte_vals, bins=256, range=(0, 255))[0]\n        return hist\n    except:\n        return np.zeros(256)\n\n# Extract features for a small sample of files\nfile_dir = '/kaggle/working/bytes'\nsample_files = os.listdir(file_dir)[:2500]  # You can increase to 1000+\n\nX = []\nfile_ids = []\n\nfor fname in tqdm(sample_files):\n    if fname.endswith('.bytes'):\n        f_id = fname.replace(\".bytes\", \"\")\n        hist = extract_byte_histogram(os.path.join(file_dir, fname))\n        X.append(hist)\n        file_ids.append(f_id)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:45:50.557601Z","iopub.execute_input":"2025-08-21T08:45:50.558406Z","iopub.status.idle":"2025-08-21T08:57:41.587955Z","shell.execute_reply.started":"2025-08-21T08:45:50.558378Z","shell.execute_reply":"2025-08-21T08:57:41.587348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_features = pd.DataFrame(X, columns=[f'byte_{i:02X}' for i in range(256)])\ndf_features[\"Id\"] = file_ids\ndf_features = df_features.merge(labels_df, on=\"Id\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:57:41.588599Z","iopub.execute_input":"2025-08-21T08:57:41.588804Z","iopub.status.idle":"2025-08-21T08:57:42.556259Z","shell.execute_reply.started":"2025-08-21T08:57:41.588788Z","shell.execute_reply":"2025-08-21T08:57:42.555398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_features.describe()\n\n# Correlation heatmap (optional)\nimport seaborn as sns\nplt.figure(figsize=(12, 8))\nsns.heatmap(df_features.drop(columns=[\"Id\", \"Class\", \"Family\"]).corr(), cmap=\"viridis\")\nplt.title(\"Feature Correlation\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:57:42.557881Z","iopub.execute_input":"2025-08-21T08:57:42.558080Z","iopub.status.idle":"2025-08-21T08:57:43.985119Z","shell.execute_reply.started":"2025-08-21T08:57:42.558065Z","shell.execute_reply":"2025-08-21T08:57:43.984338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_features.to_csv(\"/kaggle/working/byte_histogram_features.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:57:43.986073Z","iopub.execute_input":"2025-08-21T08:57:43.986347Z","iopub.status.idle":"2025-08-21T08:57:44.147457Z","shell.execute_reply.started":"2025-08-21T08:57:43.986321Z","shell.execute_reply":"2025-08-21T08:57:44.146835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Create correlation matrix\ncorr_matrix = df_features.drop(columns=[\"Id\", \"Class\", \"Family\"]).corr().abs()\n\n# Select upper triangle of correlation matrix\nupper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))\n\n# Find features with correlation greater than 0.95\nto_drop = [column for column in upper.columns if any(upper[column] > 0.95)]\n\n# Drop those features\ndf_reduced = df_features.drop(columns=to_drop)\nprint(f\"Dropped {len(to_drop)} highly correlated features.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:57:44.148290Z","iopub.execute_input":"2025-08-21T08:57:44.148501Z","iopub.status.idle":"2025-08-21T08:57:44.549589Z","shell.execute_reply.started":"2025-08-21T08:57:44.148484Z","shell.execute_reply":"2025-08-21T08:57:44.548879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nle = LabelEncoder()\ndf_features[\"Family\"] = le.fit_transform(df_features[\"Family\"])  # You can also use df_reduced if used earlier","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:57:44.550629Z","iopub.execute_input":"2025-08-21T08:57:44.550932Z","iopub.status.idle":"2025-08-21T08:57:44.678252Z","shell.execute_reply.started":"2025-08-21T08:57:44.550909Z","shell.execute_reply":"2025-08-21T08:57:44.677722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = df_reduced.drop(columns=[\"Id\", \"Class\", \"Family\"])\ny = df_reduced[\"Family\"]\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:57:44.678894Z","iopub.execute_input":"2025-08-21T08:57:44.679144Z","iopub.status.idle":"2025-08-21T08:57:44.894387Z","shell.execute_reply.started":"2025-08-21T08:57:44.679118Z","shell.execute_reply":"2025-08-21T08:57:44.893506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, classification_report\n\nclf = RandomForestClassifier(n_estimators=100, random_state=42)\nclf.fit(X_train, y_train)\n\ny_pred = clf.predict(X_test)\n\n# Evaluate\nprint(\"Accuracy:\", accuracy_score(y_test, y_pred))\nprint(classification_report(y_test, y_pred, target_names=le.classes_))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:57:44.895362Z","iopub.execute_input":"2025-08-21T08:57:44.895593Z","iopub.status.idle":"2025-08-21T08:57:46.652972Z","shell.execute_reply.started":"2025-08-21T08:57:44.895575Z","shell.execute_reply":"2025-08-21T08:57:46.652117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import learning_curve\nimport numpy as np\nimport matplotlib.pyplot as plt\n\ntrain_sizes, train_scores, val_scores = learning_curve(\n    estimator=clf,  # your trained model\n    X=X, y=y,\n    train_sizes=np.linspace(0.1, 1.0, 10),\n    cv=5,\n    scoring='accuracy',\n    n_jobs=-1\n)\n\ntrain_mean = np.mean(train_scores, axis=1)\nval_mean = np.mean(val_scores, axis=1)\n\nplt.figure(figsize=(10,6))\nplt.plot(train_sizes, train_mean, label='Training Accuracy')\nplt.plot(train_sizes, val_mean, label='Validation Accuracy')\nplt.xlabel('Training Set Size')\nplt.ylabel('Accuracy')\nplt.title('Learning Curve')\nplt.legend()\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:57:46.655623Z","iopub.execute_input":"2025-08-21T08:57:46.655949Z","iopub.status.idle":"2025-08-21T08:58:02.001572Z","shell.execute_reply.started":"2025-08-21T08:57:46.655930Z","shell.execute_reply":"2025-08-21T08:58:02.000781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\nimport matplotlib.pyplot as plt\n\n# Create confusion matrix\ncm = confusion_matrix(y_test, y_pred)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=le.classes_)\n\n# Set figure size before plotting\nplt.figure(figsize=(12, 10))\ndisp.plot(cmap='Blues', xticks_rotation=90)\nplt.title('Confusion Matrix')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:58:02.002482Z","iopub.execute_input":"2025-08-21T08:58:02.002787Z","iopub.status.idle":"2025-08-21T08:58:02.328078Z","shell.execute_reply.started":"2025-08-21T08:58:02.002763Z","shell.execute_reply":"2025-08-21T08:58:02.327353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\nprint(classification_report(y_test, y_pred, target_names=le.classes_))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:58:02.328838Z","iopub.execute_input":"2025-08-21T08:58:02.329015Z","iopub.status.idle":"2025-08-21T08:58:02.350400Z","shell.execute_reply.started":"2025-08-21T08:58:02.329001Z","shell.execute_reply":"2025-08-21T08:58:02.349857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"importances = clf.feature_importances_\nindices = np.argsort(importances)[-25:]  # Top 25 important features\nplt.figure(figsize=(10, 6))\nplt.barh(range(len(indices)), importances[indices], align='center')\nplt.yticks(range(len(indices)), [X.columns[i] for i in indices])\nplt.xlabel('Importance')\nplt.title('Top 25 Feature Importances')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:58:02.351196Z","iopub.execute_input":"2025-08-21T08:58:02.351423Z","iopub.status.idle":"2025-08-21T08:58:02.644436Z","shell.execute_reply.started":"2025-08-21T08:58:02.351406Z","shell.execute_reply":"2025-08-21T08:58:02.643833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split, learning_curve\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix, ConfusionMatrixDisplay\nfrom sklearn.manifold import TSNE\nimport matplotlib.patches as mpatches # For t-SNE legend\nimport gc # For memory management\n\n# --- Your initial setup code (assuming data_path and labels_df are defined) ---\ndata_path = \"/kaggle/input/malware-classification\"\nlabels_df = pd.read_csv(f\"{data_path}/trainLabels.csv\")\n\n# Define the directory where you extracted the .bytes files (from the 7z extraction)\nfile_dir = '/kaggle/working/bytes'\n\n# --- Feature Extraction Function ---\ndef extract_byte_histogram(file_path):\n    try:\n        with open(file_path, 'r') as file:\n            hex_lines = file.readlines()\n        bytes_list = []\n        for line in hex_lines:\n            parts = line.strip().split()\n            bytes_seq = parts[1:]  # ignore address part\n            bytes_list.extend([b for b in bytes_seq if b != '??'])\n        byte_vals = [int(b, 16) for b in bytes_list if len(b) == 2]\n        hist = np.histogram(byte_vals, bins=256, range=(0, 255))[0]\n        return hist\n    except Exception as e:\n        # print(f\"Error extracting histogram from {file_path}: {e}\") # Uncomment for debugging\n        return np.zeros(256)\n\n# --- Collect extracted file IDs and filter labels ---\nsample_files = [f for f in os.listdir(file_dir) if f.endswith('.bytes')]\nsample_ids_extracted = [f.replace(\".bytes\", \"\") for f in sample_files]\nfiltered_labels_df = labels_df[labels_df['Id'].isin(sample_ids_extracted)].copy()\n\nX_hist = []\nextracted_file_ids = []\n\nprint(f\"Starting feature extraction for {len(sample_files)} files...\")\nfor fname in tqdm(sample_files):\n    f_id = fname.replace(\".bytes\", \"\")\n    hist = extract_byte_histogram(os.path.join(file_dir, fname))\n    X_hist.append(hist)\n    extracted_file_ids.append(f_id)\nprint(\"Feature extraction complete.\")\n\n# --- Create Feature DataFrame and Merge Labels ---\ndf_features = pd.DataFrame(X_hist, columns=[f'byte_{i:02X}' for i in range(256)])\ndf_features[\"Id\"] = extracted_file_ids\ndf_features = df_features.merge(filtered_labels_df, on=\"Id\")\n\n# --- DATA PREPARATION AND LABEL ENCODING (CRITICAL SECTION) ---\n\n# Map class numbers to family names (ensure 'Family' column with string names exists)\nfamily_map = {\n    1: \"Ramnit\", 2: \"Lollipop\", 3: \"Kelihos_ver3\", 4: \"Vundo\",\n    5: \"Simda\", 6: \"Tracur\", 7: \"Kelihos_ver1\", 8: \"Obfuscator.ACY\", 9: \"Gatak\"\n}\nif \"Family\" not in df_features.columns:\n    df_features[\"Family\"] = df_features[\"Class\"].map(family_map)\n\n# 1. Initialize LabelEncoder\nle = LabelEncoder()\n\n# 2. Apply LabelEncoder to the 'Family' column and create a NEW encoded column\ndf_features[\"Family_Encoded\"] = le.fit_transform(df_features[\"Family\"])\n\n# 3. CRITICAL FIX: Explicitly ensure the encoded column is of integer type\ndf_features[\"Family_Encoded\"] = df_features[\"Family_Encoded\"].astype(int)\n\n# --- Feature Correlation and Reduction ---\ncorr_matrix = df_features.drop(columns=[\"Id\", \"Class\", \"Family\", \"Family_Encoded\"]).corr().abs()\nupper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))\nto_drop = [column for column in upper.columns if any(upper[column] > 0.95)]\n\ndf_reduced = df_features.drop(columns=to_drop)\nprint(f\"Dropped {len(to_drop)} highly correlated features.\")\n\n# --- Define X (features) and y (encoded labels) for Model Training ---\nX = df_reduced.drop(columns=[\"Id\", \"Class\", \"Family\", \"Family_Encoded\"])\ny = df_reduced[\"Family_Encoded\"] # This 'y' now definitively contains integers\n\nX = X.fillna(0) # Handle potential NaN values\n\n# --- Train-Test Split ---\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)\n\n# --- Random Forest Classifier ---\nprint(\"\\n--- Training RandomForestClassifier ---\")\nclf = RandomForestClassifier(n_estimators=100, random_state=42, n_jobs=-1)\nclf.fit(X_train, y_train)\n\ny_pred = clf.predict(X_test)\n\n# Evaluate\nprint(\"Accuracy:\", accuracy_score(y_test, y_pred))\nprint(classification_report(y_test, y_pred, target_names=le.classes_))\n\n# --- Learning Curve ---\nprint(\"\\n--- Generating Learning Curve ---\")\ntrain_sizes, train_scores, val_scores = learning_curve(\n    estimator=clf, X=X, y=y, # Use full X and y for learning curve\n    train_sizes=np.linspace(0.1, 1.0, 10), cv=5, scoring='accuracy', n_jobs=-1\n)\ntrain_mean = np.mean(train_scores, axis=1)\nval_mean = np.mean(val_scores, axis=1)\n\nplt.figure(figsize=(10,6))\nplt.plot(train_sizes, train_mean, label='Training Accuracy')\nplt.plot(train_sizes, val_mean, label='Validation Accuracy')\nplt.xlabel('Training Set Size')\nplt.ylabel('Accuracy')\nplt.title('Learning Curve')\nplt.legend()\nplt.grid()\nplt.show()\n\n# --- Confusion Matrix ---\nprint(\"\\n--- Generating Confusion Matrix ---\")\ncm = confusion_matrix(y_test, y_pred)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=le.classes_)\nplt.figure(figsize=(12, 10))\ndisp.plot(cmap='Blues', xticks_rotation=90, ax=plt.gca())\nplt.title('Confusion Matrix')\nplt.show()\n\n# --- Feature Importances ---\nprint(\"\\n--- Generating Feature Importances Plot ---\")\nimportances = clf.feature_importances_\nindices = np.argsort(importances)[-20:]\nplt.figure(figsize=(10, 6))\nplt.barh(range(len(indices)), importances[indices], align='center')\nplt.yticks(range(len(indices)), [X.columns[i] for i in indices])\nplt.xlabel('Importance')\nplt.title('Top 20 Feature Importances')\nplt.tight_layout()\nplt.show()\n\n# --- t-SNE Visualization ---\nprint(\"\\n--- Starting t-SNE Visualization ---\")\ntsne_sample_size = 2000 # Adjust based on your extracted data size and memory\n\nX_tsne_input = X\ny_tsne_input = y\n\nif len(X) > tsne_sample_size:\n    print(f\"Sampling {tsne_sample_size} data points for t-SNE visualization...\")\n    sample_indices = np.random.choice(X.index, tsne_sample_size, replace=False)\n    X_tsne_input = X.loc[sample_indices]\n    y_tsne_input = y.loc[sample_indices]\nelse:\n    print(\"Using full dataset for t-SNE as it's within sample limit.\")\n\nprint(\"t-SNE computation may take a while...\")\ntsne = TSNE(n_components=2, random_state=42, perplexity=30, n_iter=1000, learning_rate=200, n_jobs=-1)\nX_2d = tsne.fit_transform(X_tsne_input)\n\nprint(\"t-SNE computation complete.\")\n\n# Create patches for the legend\nunique_labels = np.unique(y_tsne_input) # This should now contain integers\n\ncmap = plt.colormaps.get_cmap('tab20')\n\npatches = [mpatches.Patch(color=cmap(i / (len(unique_labels) - 1)) if len(unique_labels) > 1 else cmap(0.5),\n                          label=le.inverse_transform([i])[0])\n           for i in unique_labels]\n\nplt.figure(figsize=(12, 10))\nscatter = plt.scatter(X_2d[:, 0], X_2d[:, 1], c=y_tsne_input, cmap=cmap, s=10, alpha=0.7)\n\nplt.title(\"t-SNE Visualization of Malware Families\")\nplt.xlabel(\"t-SNE Component 1\")\nplt.ylabel(\"t-SNE Component 2\")\nplt.legend(handles=patches, bbox_to_anchor=(1.02, 1), loc='upper left', title=\"Malware Family\")\nplt.grid(True, linestyle='--', alpha=0.6)\nplt.tight_layout(rect=[0, 0, 0.88, 1])\nplt.show()\n\n# Clean up memory\n# del X_hist, X, y, X_train, X_test, y_train, y_test, df_features, df_reduced, tsne, X_2d, X_tsne_input, y_tsne_input\n# gc.collect()\n\nprint(\"\\n--- Malware Family Classification workflow complete ---\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T08:58:02.645263Z","iopub.execute_input":"2025-08-21T08:58:02.645484Z","iopub.status.idle":"2025-08-21T09:10:21.835711Z","shell.execute_reply.started":"2025-08-21T08:58:02.645468Z","shell.execute_reply":"2025-08-21T09:10:21.834717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---------------------------------------------\n# 0)  PREP  (run once)\n# ---------------------------------------------\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split, cross_val_score\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.ensemble import RandomForestClassifier, ExtraTreesClassifier, GradientBoostingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.neural_network import MLPClassifier\nimport warnings, matplotlib.pyplot as plt\nwarnings.filterwarnings('ignore')\n\n# --- features & labels ---\n# Ensure df_features is defined from previous steps (containing 'Family_Encoded' and other features)\nX = df_features.drop(columns=[\"Id\", \"Class\", \"Family\", \"Family_Encoded\"]) # Drop 'Family_Encoded' from X\ny = df_features[\"Family_Encoded\"] # y already holds the encoded labels\n\n# Train-test split just for final hold-out evaluation\nX_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y\n)\n# The rest of your model zoo, cross-validation, and evaluation code follows","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T09:10:21.836598Z","iopub.execute_input":"2025-08-21T09:10:21.837096Z","iopub.status.idle":"2025-08-21T09:10:21.883429Z","shell.execute_reply.started":"2025-08-21T09:10:21.837074Z","shell.execute_reply":"2025-08-21T09:10:21.882530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---------------------------------------------\n# 1)  DEFINE A SMALL MODEL ZOO\n# ---------------------------------------------\nmodels = {\n    \"RandomForest\": RandomForestClassifier(n_estimators=300, random_state=42, n_jobs=-1),\n    \"ExtraTrees\"  : ExtraTreesClassifier(n_estimators=400, random_state=42, n_jobs=-1),\n    \n    # algorithms that need scaling are wrapped in a Pipeline\n    \"LogReg\" : Pipeline([\n        (\"scaler\", StandardScaler()), \n        (\"clf\", LogisticRegression(max_iter=500, multi_class=\"multinomial\"))\n    ]),\n    \n    \"SVM‑RBF\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", SVC(kernel=\"rbf\", C=5, gamma=\"scale\"))\n    ]),\n    \n    \"kNN‑10\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", KNeighborsClassifier(n_neighbors=10))\n    ]),\n    \n    \"GradBoost\": GradientBoostingClassifier(random_state=42),\n    \n    \"MLP‑128x64\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", MLPClassifier(hidden_layer_sizes=(128,64), max_iter=200, random_state=42))\n    ]),\n}\n\n# (Optional) add XGBoost or LightGBM if their libraries are installed:\n# import xgboost as xgb\n# models[\"XGBoost\"] = xgb.XGBClassifier(\n#     n_estimators=500, max_depth=7, learning_rate=0.1,\n#     subsample=0.8, colsample_bytree=0.8, objective=\"multi:softprob\",\n#     num_class=len(np.unique(y)), tree_method=\"hist\", random_state=42\n# )\n\n# ---------------------------------------------\n# 2)  CROSS‑VALIDATE EACH MODEL\n# ---------------------------------------------\nresults = {}\nfor name, clf in models.items():\n    cv_scores = cross_val_score(clf, X_train, y_train, cv=5, scoring=\"accuracy\", n_jobs=-1)\n    results[name] = {\n        \"CV mean\":  cv_scores.mean(),\n        \"CV std\" :  cv_scores.std()\n    }\n    print(f\"{name:12s}  |  CV accuracy = {cv_scores.mean():.4f} ± {cv_scores.std():.4f}\")\n\n# ---------------------------------------------\n# 3)  RANKED SUMMARY\n# ---------------------------------------------\nsummary = (pd.DataFrame(results)\n           .T.sort_values(\"CV mean\", ascending=False)\n           .style.format({\"CV mean\":\"{:.4f}\", \"CV std\":\"{:.4f}\"}))\ndisplay(summary)\n\n# ---------------------------------------------\n# 4)  TRAIN THE BEST MODEL ON FULL TRAIN SET, EVALUATE ON HOLD‑OUT\n# ---------------------------------------------\nbest_name = summary.data.index[0]\nbest_model = models[best_name]\nbest_model.fit(X_train, y_train)\ny_pred = best_model.predict(X_test)\n\nprint(f\"\\n🏆 Best model: {best_name}\")\nprint(\"Hold‑out accuracy:\", accuracy_score(y_test, y_pred))\nprint(classification_report(y_test, y_pred, target_names=le.classes_))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T09:10:21.884452Z","iopub.execute_input":"2025-08-21T09:10:21.884790Z","iopub.status.idle":"2025-08-21T09:13:52.341404Z","shell.execute_reply.started":"2025-08-21T09:10:21.884764Z","shell.execute_reply":"2025-08-21T09:13:52.340440Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"importances = best_model.feature_importances_\nfeat_names = X.columns\n\nfeat_imp = pd.Series(importances, index=feat_names).sort_values(ascending=False)\ntop_features = feat_imp.head(25)\n\nplt.figure(figsize=(10,6))\ntop_features.plot(kind=\"barh\", color='steelblue')\nplt.gca().invert_yaxis()\nplt.title(\"Top 25 Feature Importances (ExtraTrees)\")\nplt.xlabel(\"Importance Score\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T09:13:52.342567Z","iopub.execute_input":"2025-08-21T09:13:52.343036Z","iopub.status.idle":"2025-08-21T09:13:52.769714Z","shell.execute_reply.started":"2025-08-21T09:13:52.343012Z","shell.execute_reply":"2025-08-21T09:13:52.768970Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import learning_curve\n\ntrain_sizes, train_scores, test_scores = learning_curve(\n    best_model, X, y, cv=5, train_sizes=np.linspace(0.1, 1.0, 10), n_jobs=-1\n)\n\ntrain_mean = train_scores.mean(axis=1)\ntest_mean = test_scores.mean(axis=1)\n\nplt.figure(figsize=(8, 5))\nplt.plot(train_sizes, train_mean, label=\"Train Score\", marker=\"o\")\nplt.plot(train_sizes, test_mean, label=\"CV Score\", marker=\"s\")\nplt.title(\"Learning Curve for ExtraTreesClassifier\")\nplt.xlabel(\"Training Set Size\")\nplt.ylabel(\"Accuracy\")\nplt.legend()\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T09:13:52.770390Z","iopub.execute_input":"2025-08-21T09:13:52.770580Z","iopub.status.idle":"2025-08-21T09:14:14.556511Z","shell.execute_reply.started":"2025-08-21T09:13:52.770565Z","shell.execute_reply":"2025-08-21T09:14:14.555719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import validation_curve\n\nparam_range = [50, 100, 200, 300, 400, 500]\ntrain_scores, test_scores = validation_curve(\n    ExtraTreesClassifier(random_state=42),\n    X, y, param_name=\"n_estimators\", param_range=param_range,\n    cv=5, scoring=\"accuracy\", n_jobs=-1\n)\n\ntrain_mean = train_scores.mean(axis=1)\ntest_mean = test_scores.mean(axis=1)\n\nplt.plot(param_range, train_mean, label=\"Train\", marker='o')\nplt.plot(param_range, test_mean, label=\"CV\", marker='s')\nplt.title(\"Validation Curve for n_estimators\")\nplt.xlabel(\"Number of Estimators\")\nplt.ylabel(\"Accuracy\")\nplt.legend()\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T09:14:14.557334Z","iopub.execute_input":"2025-08-21T09:14:14.557619Z","iopub.status.idle":"2025-08-21T09:14:25.656212Z","shell.execute_reply.started":"2025-08-21T09:14:14.557599Z","shell.execute_reply":"2025-08-21T09:14:25.655465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# Ensure 'results' dictionary is available by running the model training and\n# cross-validation section of your notebook before this code block.\n# Example 'results' dictionary structure:\n# results = {\n#     \"RandomForest\": {\"CV mean\": 0.9587, \"CV std\": 0.0109},\n#     \"ExtraTrees\": {\"CV mean\": 0.9619, \"CV std\": 0.0087},\n#     \"LogReg\": {\"CV mean\": 0.8580, \"CV std\": 0.0192},\n#     \"SVM‑RBF\": {\"CV mean\": 0.7830, \"CV std\": 0.0181},\n#     \"kNN‑10\": {\"CV mean\": 0.8199, \"CV std\": 0.0115},\n#     \"GradBoost\": {\"CV mean\": 0.9481, \"CV std\": 0.0210},\n#     \"MLP‑128x64\": {\"CV mean\": 0.9131, \"CV std\": 0.0164},\n# }\n\n\n# Convert results to a DataFrame for easier plotting and sorting\ndf_results = pd.DataFrame(results).T\ndf_results = df_results.sort_values(by=\"CV mean\", ascending=False)\n\n# Create the bar plot\nplt.figure(figsize=(12, 7))\nbars = plt.bar(df_results.index, df_results[\"CV mean\"], yerr=df_results[\"CV std\"], capsize=5, color='skyblue')\n\n# Add accuracy values on top of the bars\nfor bar in bars:\n    yval = bar.get_height()\n    # Adjust position for text based on max standard deviation to avoid overlap\n    plt.text(bar.get_x() + bar.get_width()/2, yval + df_results[\"CV std\"].max() * 0.02,\n             f'{yval:.4f}', ha='center', va='bottom', fontsize=9)\n\nplt.xlabel('Model')\nplt.ylabel('Cross-Validation Accuracy')\nplt.title('Model Performance Comparison (Cross-Validation Accuracy)')\n# Adjust y-limits to make space for text labels and error bars\nplt.ylim(min(df_results[\"CV mean\"]) * 0.95, max(df_results[\"CV mean\"]) * 1.05 + df_results[\"CV std\"].max())\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.xticks(rotation=45, ha='right') # Rotate model names for better readability\nplt.tight_layout() # Adjust layout to prevent labels from overlapping\nplt.show() # Display the plot in the Kaggle notebook","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T09:14:25.656887Z","iopub.execute_input":"2025-08-21T09:14:25.657180Z","iopub.status.idle":"2025-08-21T09:14:25.894444Z","shell.execute_reply.started":"2025-08-21T09:14:25.657152Z","shell.execute_reply":"2025-08-21T09:14:25.893888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---------------------------------------------\n# 0)  PREP  (Ensure df_features and le are defined from previous cells)\n# ---------------------------------------------\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split, cross_val_score\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder # Make sure LabelEncoder is imported and 'le' is defined\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.ensemble import RandomForestClassifier, ExtraTreesClassifier, GradientBoostingClassifier, VotingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.neural_network import MLPClassifier\nimport warnings, matplotlib.pyplot as plt\nwarnings.filterwarnings('ignore')\n\n# --- features & labels ---\n# Assuming df_features and le (LabelEncoder instance) are already defined from previous data preparation steps.\n# Example:\n# df_features = ... (your DataFrame with extracted features and merged labels including 'Family_Encoded')\n# le = ... (your trained LabelEncoder instance)\n\n# Make sure X and y are defined from your df_features.\n# If 'Family_Encoded' is already the integer column, use it directly.\nX = df_features.drop(columns=[\"Id\", \"Class\", \"Family\", \"Family_Encoded\"])\ny = df_features[\"Family_Encoded\"] # This 'y' already holds the encoded integer labels\n\n# Train-test split just for final hold-out evaluation\nX_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y\n)\n\n# ---------------------------------------------\n# 1) DEFINE A SMALL MODEL ZOO (Individual Classifiers)\n# ---------------------------------------------\nmodels = {\n    \"RandomForest\": RandomForestClassifier(n_estimators=300, random_state=42, n_jobs=-1),\n    \"ExtraTrees\": ExtraTreesClassifier(n_estimators=400, random_state=42, n_jobs=-1),\n\n    # algorithms that need scaling are wrapped in a Pipeline\n    \"LogReg\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", LogisticRegression(max_iter=500, multi_class=\"multinomial\"))\n    ]),\n\n    \"SVM‑RBF\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", SVC(kernel=\"rbf\", C=5, gamma=\"scale\")) # Note: For soft voting, need probability=True\n    ]),\n\n    \"kNN‑10\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", KNeighborsClassifier(n_neighbors=10))\n    ]),\n\n    \"GradBoost\": GradientBoostingClassifier(random_state=42),\n\n    \"MLP‑128x64\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", MLPClassifier(hidden_layer_sizes=(128,64), max_iter=200, random_state=42))\n    ]),\n}\n\nprint(\"Individual models defined.\")\n\n# ---------------------------------------------\n# 2) DEFINE ENSEMBLE COMBINATIONS\n# ---------------------------------------------\nensemble_models = {\n    \"Ensemble_RF_ET_GB_Hard\": VotingClassifier(\n        estimators=[\n            ('rf', models['RandomForest']),\n            ('et', models['ExtraTrees']),\n            ('gb', models['GradBoost'])\n        ],\n        voting='hard',\n        n_jobs=-1\n    ),\n    \"Ensemble_RF_ET_GB_Soft\": VotingClassifier(\n        estimators=[\n            ('rf', models['RandomForest']),\n            ('et', models['ExtraTrees']),\n            ('gb', models['GradBoost'])\n        ],\n        voting='soft',\n        n_jobs=-1\n    ),\n    \"Ensemble_RF_ET_GB_MLP_Soft\": VotingClassifier(\n        estimators=[\n            ('rf', models['RandomForest']),\n            ('et', models['ExtraTrees']),\n            ('gb', models['GradBoost']),\n            ('mlp', models['MLP‑128x64'])\n        ],\n        voting='soft',\n        n_jobs=-1\n    ),\n}\n\nprint(\"Ensemble models defined.\")\n\n# ---------------------------------------------\n# 3) CROSS-VALIDATE ALL MODELS (Individual + Ensembles)\n# ---------------------------------------------\n# Start with results from individual models\n# Note: if you have already run this part and have a 'results' dict, you can use it.\n# Otherwise, it will be populated here.\nif 'results' not in locals(): # Check if 'results' from individual models is already defined\n    results = {}\n\nprint(\"\\n--- Cross-validating individual models ---\")\nfor name, clf in models.items():\n    if name not in results: # Avoid re-running if already done\n        cv_scores = cross_val_score(clf, X_train, y_train, cv=5, scoring=\"accuracy\", n_jobs=-1)\n        results[name] = {\n            \"CV mean\": cv_scores.mean(),\n            \"CV std\": cv_scores.std()\n        }\n        print(f\"{name:12s} | CV accuracy = {cv_scores.mean():.4f} ± {cv_scores.std():.4f}\")\n    else:\n        print(f\"{name:12s} | CV accuracy = {results[name]['CV mean']:.4f} ± {results[name]['CV std']:.4f} (already computed)\")\n\n\nprint(\"\\n--- Cross-validating ensemble models ---\")\nensemble_results = {}\nfor name, clf in ensemble_models.items():\n    cv_scores = cross_val_score(clf, X_train, y_train, cv=5, scoring=\"accuracy\", n_jobs=-1)\n    ensemble_results[name] = {\n        \"CV mean\": cv_scores.mean(),\n        \"CV std\": cv_scores.std()\n    }\n    print(f\"{name:25s} | CV accuracy = {cv_scores.mean():.4f} ± {cv_scores.std():.4f}\")\n\n\n# ---------------------------------------------\n# 4) COMBINE AND RANK ALL MODEL RESULTS (Individual + Ensembles)\n# ---------------------------------------------\ndf_individual_results = pd.DataFrame(results).T\ndf_ensemble_results = pd.DataFrame(ensemble_results).T\n\nall_results_df = pd.concat([df_individual_results, df_ensemble_results])\n\nsummary_all_models = (all_results_df\n                      .sort_values(\"CV mean\", ascending=False)\n                      .style.format({\"CV mean\":\"{:.4f}\", \"CV std\":\"{:.4f}\"}))\n\nprint(\"\\n--- Ranked Summary of All Models (Individual + Ensembles) ---\")\ndisplay(summary_all_models)\n\n# ---------------------------------------------\n# 5) TRAIN THE OVERALL BEST MODEL ON FULL TRAIN SET, EVALUATE ON HOLD-OUT\n# ---------------------------------------------\nbest_overall_name = summary_all_models.data.index[0]\n# Retrieve the best model object, checking both 'ensemble_models' and 'models' dictionaries\nbest_overall_model = ensemble_models.get(best_overall_name, models.get(best_overall_name))\n\nif best_overall_model:\n    print(f\"\\n🏆 Overall Best Model: {best_overall_name}\")\n    print(f\"Training {best_overall_name} on the full training set (X_train, y_train)...\")\n    best_overall_model.fit(X_train, y_train)\n    y_pred_overall = best_overall_model.predict(X_test)\n\n    print(\"\\nHold-out accuracy:\", accuracy_score(y_test, y_pred_overall))\n    # 'le' (LabelEncoder) should be available from your initial data preparation steps\n    print(classification_report(y_test, y_pred_overall, target_names=le.classes_))\nelse:\n    print(f\"Error: Best model '{best_overall_name}' not found in defined models. Check dictionary keys.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T09:14:25.895547Z","iopub.execute_input":"2025-08-21T09:14:25.895851Z","iopub.status.idle":"2025-08-21T09:26:00.791052Z","shell.execute_reply.started":"2025-08-21T09:14:25.895832Z","shell.execute_reply":"2025-08-21T09:26:00.790179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Modified code to use the specified data split: X_train, X_test, y_train, y_test\n\nimport xgboost as xgb\nimport lightgbm as lgb\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport time\nimport seaborn as sns # Import seaborn for confusion matrix visualization\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix, roc_auc_score, RocCurveDisplay # Import evaluation metrics and display function\nfrom collections import Counter\nfrom sklearn.preprocessing import label_binarize, LabelEncoder # Import LabelEncoder\n\n# --- Assume data is already split like this ---\n# X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)\n\nprint(\"\\n--- Two-Stage Multiclass Algorithm Implementation ---\")\n\n# NEW: Convert string labels to numerical labels\nle = LabelEncoder()\ny_train_encoded = le.fit_transform(y_train)\ny_test_encoded = le.transform(y_test)\n\n# Calculate class weights for imbalance handling\n# This is a crucial step since the Microsoft Malware dataset is highly imbalanced\nunique_classes, class_counts = np.unique(y_train_encoded, return_counts=True)\ntotal_samples = len(y_train_encoded)\nclass_weights_dict = {cls: total_samples / (len(unique_classes) * count) for cls, count in zip(unique_classes, class_counts)}\nprint(f\"Calculated class_weights for Multiclass: {class_weights_dict}\")\n\n# Stage 1: Train a base XGBoost model\nxgb_model = xgb.XGBClassifier(\n    objective='multi:softprob', # Changed for multiclass probability output\n    eval_metric=['mlogloss', 'merror'], # Changed for multiclass metrics\n    use_label_encoder=False,\n    enable_categorical=False,\n    n_estimators=1000,\n    learning_rate=0.05,\n    max_depth=7,\n    subsample=0.7,\n    colsample_bytree=0.7,\n    gamma=0.1,\n    random_state=42,\n    num_class=len(unique_classes), # New parameter: number of classes\n    tree_method='hist'\n)\n# Update eval_set with new variable names\neval_set_xgb = [(X_train, y_train_encoded), (X_test, y_test_encoded)]\ncallbacks = [xgb.callback.EarlyStopping(rounds=50, metric_name='mlogloss', data_name='validation_1', save_best=True)]\n\nstart_time = time.time()\n# Use sample weights for XGBoost to handle class imbalance\nsample_weights = np.array([class_weights_dict[label] for label in y_train_encoded])\nxgb_model.fit(X_train, y_train_encoded, eval_set=eval_set_xgb, callbacks=callbacks, verbose=False, sample_weight=sample_weights)\ntraining_time_xgb = time.time() - start_time\nprint(f\"\\nXGBoost Model Training Time: {training_time_xgb:.2f} seconds\")\n\n# --- Plotting XGBoost Learning Curves (Multiclass) ---\nresults = xgb_model.evals_result()\nepochs = len(results['validation_0']['mlogloss'])\nx_axis = range(0, epochs)\n\nplt.style.use('ggplot')\nplt.figure(figsize=(18, 5))\nplt.subplot(1, 3, 1) # Changed to 1x3 subplot\nplt.plot(x_axis, results['validation_0']['mlogloss'], label='Train mlogloss')\nplt.plot(x_axis, results['validation_1']['mlogloss'], label='Validation mlogloss')\nplt.grid(True)\nplt.legend()\nplt.title('XGBoost Multiclass LogLoss')\nplt.xlabel('Boosting Rounds')\nplt.ylabel('Multiclass LogLoss')\n\nplt.subplot(1, 3, 2) # Added new subplot for Error\nplt.plot(x_axis, results['validation_0']['merror'], label='Train Error')\nplt.plot(x_axis, results['validation_1']['merror'], label='Validation Error')\nplt.grid(True)\nplt.legend()\nplt.title('XGBoost Multiclass Error Rate')\nplt.xlabel('Boosting Rounds')\nplt.ylabel('Error Rate')\n\nplt.subplot(1, 3, 3) # Added new subplot for Accuracy\nplt.plot(x_axis, [1 - x for x in results['validation_0']['merror']], label='Train Accuracy')\nplt.plot(x_axis, [1 - x for x in results['validation_1']['merror']], label='Validation Accuracy')\nplt.grid(True)\nplt.legend()\nplt.title('XGBoost Multiclass Accuracy')\nplt.xlabel('Boosting Rounds')\nplt.ylabel('Accuracy')\n\nplt.tight_layout()\nplt.show()\n\n# Stage 2 & 3: Feature generation and combination (unchanged from original logic)\nX_train_xgb_features = xgb_model.apply(X_train)\nX_test_xgb_features = xgb_model.apply(X_test)\nprint(f\"\\nStage 2: XGBoost leaf features created. Train shape: {X_train_xgb_features.shape}, Test shape: {X_test_xgb_features.shape}\")\n\nfeature_importances = xgb_model.feature_importances_\nk = 5\ntop_k_original_features_indices = np.argsort(feature_importances)[-k:]\n# FIX: Convert pandas DataFrame to a NumPy array before slicing\nX_train_selected_original = X_train.values[:, top_k_original_features_indices]\nX_test_selected_original = X_test.values[:, top_k_original_features_indices]\nprint(f\"Novelty: Top {k} original features selected.\")\n\nX_train_combined_refined = np.concatenate([X_train_selected_original, X_train_xgb_features], axis=1)\nX_test_combined_refined = np.concatenate([X_test_selected_original, X_test_xgb_features], axis=1)\nprint(f\"Stage 3: Refined combined data created. Train shape: {X_train_combined_refined.shape}, Test shape: {X_test_combined_refined.shape}\")\n\n# Stage 4: Train the LightGBM model on the refined combined data.\nparams_id = {\n    'objective': 'multiclass',\n    'metric': ['multi_logloss', 'multi_error'], # Added 'multi_error' for accuracy plotting\n    'boosting_type': 'gbdt',\n    'num_leaves': 31,\n    'learning_rate': 0.05,\n    'verbose': -1,\n    'n_jobs': -1,\n    'seed': 42,\n    'zero_as_missing': True,\n    'num_class': len(unique_classes),\n    'class_weight': class_weights_dict\n}\nlgbm_model = lgb.LGBMClassifier(**params_id)\nnum_original_features = X_train_selected_original.shape[1]\ncategorical_feature_columns = list(range(num_original_features, X_train_combined_refined.shape[1]))\n# Update eval_set with new variable names\neval_set_lgbm = [(X_train_combined_refined, y_train_encoded), (X_test_combined_refined, y_test_encoded)]\n\nstart_time = time.time()\nlgbm_model.fit(X_train_combined_refined, y_train_encoded,\n               eval_set=eval_set_lgbm,\n               eval_metric=['multi_logloss', 'multi_error'], # Added 'multi_error' to eval_metric\n               callbacks=[lgb.early_stopping(stopping_rounds=50, verbose=False)],\n               categorical_feature=categorical_feature_columns)\ntraining_time_lgbm = time.time() - start_time\nprint(f\"\\nStage 4: LightGBM Model Training Time: {training_time_lgbm:.2f} seconds\")\n\n# --- Plotting LightGBM Learning Curves (Multiclass) ---\nresults_lgbm = lgbm_model.evals_result_\nepochs_lgbm = len(results_lgbm['training']['multi_logloss'])\nx_axis_lgbm = range(0, epochs_lgbm)\n\nplt.style.use('ggplot')\nplt.figure(figsize=(18, 5))\nplt.subplot(1, 2, 1)\nplt.plot(x_axis_lgbm, results_lgbm['training']['multi_logloss'], label='Train multi_logloss')\nplt.plot(x_axis_lgbm, results_lgbm['valid_1']['multi_logloss'], label='Validation multi_logloss')\nplt.grid(True)\nplt.legend()\nplt.title('LightGBM Multiclass LogLoss')\nplt.xlabel('Boosting Rounds')\nplt.ylabel('Multiclass LogLoss')\n\nplt.subplot(1, 2, 2) # Added new subplot for Accuracy\nplt.plot(x_axis_lgbm, [1 - x for x in results_lgbm['training']['multi_error']], label='Train Accuracy')\nplt.plot(x_axis_lgbm, [1 - x for x in results_lgbm['valid_1']['multi_error']], label='Validation Accuracy')\nplt.grid(True)\nplt.legend()\nplt.title('LightGBM Multiclass Accuracy')\nplt.xlabel('Boosting Rounds')\nplt.ylabel('Accuracy')\n\nplt.tight_layout()\nplt.show()\n\n# Stage 5: Get predictions on the test set.\ny_pred_proba = lgbm_model.predict_proba(X_test_combined_refined)\ny_pred = np.argmax(y_pred_proba, axis=1)\n\nprint(f\"\\nSample predicted probabilities:\\n{y_pred_proba[:5]}\")\nprint(f\"Sample predicted labels:\\n{y_pred[:5]}\")\n\n# --- Stage 6: Evaluate the final model's performance ---\nprint(\"\\n--- Model Evaluation ---\")\n\n# Calculate and print accuracy\naccuracy = accuracy_score(y_test_encoded, y_pred)\nprint(f\"Accuracy: {accuracy:.4f}\\n\")\n\n# Calculate and print ROC-AUC score\n# Using 'ovr' (One-vs-Rest) strategy and 'weighted' average for multiclass\nroc_auc = roc_auc_score(y_test_encoded, y_pred_proba, multi_class='ovr', average='weighted')\nprint(f\"ROC-AUC Score: {roc_auc:.4f}\\n\")\n\n# Print the classification report\nprint(\"Classification Report:\")\nprint(classification_report(y_test_encoded, y_pred))\n\n# Plot the confusion matrix as a heatmap\nconf_matrix = confusion_matrix(y_test_encoded, y_pred)\nplt.figure(figsize=(10, 8))\nsns.heatmap(conf_matrix, annot=True, fmt='d', cmap='Blues')\nplt.title('Confusion Matrix')\nplt.xlabel('Predicted Label')\nplt.ylabel('True Label')\nplt.show()\n\n# --- Plot ROC curves for each class ---\nprint(\"\\n--- ROC-AUC Curve Visualization ---\")\ny_test_binarized = label_binarize(y_test_encoded, classes=unique_classes)\nn_classes = len(unique_classes)\n\nfig, ax = plt.subplots(figsize=(10, 8))\n\n# Plot ROC curve for each class\nfor i in range(n_classes):\n    RocCurveDisplay.from_predictions(\n        y_test_binarized[:, i],\n        y_pred_proba[:, i],\n        name=f\"ROC curve for class {le.classes_[i]}\",\n        ax=ax,\n    )\n\nplt.plot([0, 1], [0, 1], linestyle=\"--\", lw=2, color=\"gray\", label=\"Chance level\")\nplt.title(\"ROC-AUC Curve (One-vs-Rest)\")\nplt.xlabel(\"False Positive Rate\")\nplt.ylabel(\"True Positive Rate\")\nplt.grid(True)\nplt.legend(loc=\"lower right\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-21T10:12:31.796247Z","iopub.execute_input":"2025-08-21T10:12:31.796542Z","iopub.status.idle":"2025-08-21T10:12:58.841243Z","shell.execute_reply.started":"2025-08-21T10:12:31.796516Z","shell.execute_reply":"2025-08-21T10:12:58.840558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}