{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":4117,"databundleVersionId":46665,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:21:38.167595Z","iopub.execute_input":"2025-07-21T05:21:38.168294Z","iopub.status.idle":"2025-07-21T05:21:38.188956Z","shell.execute_reply.started":"2025-07-21T05:21:38.168271Z","shell.execute_reply":"2025-07-21T05:21:38.188121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\ndata_path = \"/kaggle/input/malware-classification\"\n\n# List available files\nprint(os.listdir(data_path))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:21:41.568660Z","iopub.execute_input":"2025-07-21T05:21:41.569191Z","iopub.status.idle":"2025-07-21T05:21:41.574334Z","shell.execute_reply.started":"2025-07-21T05:21:41.569172Z","shell.execute_reply":"2025-07-21T05:21:41.573606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Path to the dataset\ndata_path = \"/kaggle/input/malware-classification\"\n\n# Load the labels\nlabels_df = pd.read_csv(f\"{data_path}/trainLabels.csv\")\n\n# Show first 5 rows\nlabels_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:21:46.125888Z","iopub.execute_input":"2025-07-21T05:21:46.126555Z","iopub.status.idle":"2025-07-21T05:21:46.228310Z","shell.execute_reply.started":"2025-07-21T05:21:46.126530Z","shell.execute_reply":"2025-07-21T05:21:46.227668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndata_path = \"/kaggle/input/malware-classification\"\ndf = pd.read_csv(f\"{data_path}/trainLabels.csv\")\ndf.head(10)  # shows 10 samples with Id and Class","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:21:49.330116Z","iopub.execute_input":"2025-07-21T05:21:49.330795Z","iopub.status.idle":"2025-07-21T05:21:49.350016Z","shell.execute_reply.started":"2025-07-21T05:21:49.330769Z","shell.execute_reply":"2025-07-21T05:21:49.349173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Mapping class numbers to family names\nfamily_map = {\n    1: \"Ramnit\",\n    2: \"Lollipop\",\n    3: \"Kelihos_ver3\",\n    4: \"Vundo\",\n    5: \"Simda\",\n    6: \"Tracur\",\n    7: \"Kelihos_ver1\",\n    8: \"Obfuscator.ACY\",\n    9: \"Gatak\"\n}\n\n# Add a new column for family name\nlabels_df[\"Family\"] = labels_df[\"Class\"].map(family_map)\n\n# Preview updated labels\nlabels_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:22:13.721395Z","iopub.execute_input":"2025-07-21T05:22:13.721718Z","iopub.status.idle":"2025-07-21T05:22:13.757945Z","shell.execute_reply.started":"2025-07-21T05:22:13.721695Z","shell.execute_reply":"2025-07-21T05:22:13.756878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load labels\ndata_path = \"/kaggle/input/malware-classification\"\ndf = pd.read_csv(f\"{data_path}/trainLabels.csv\")\n\n# Optional: Map class to family names\nfamily_map = {\n    1: \"Ramnit\", 2: \"Lollipop\", 3: \"Kelihos_ver3\", 4: \"Vundo\",\n    5: \"Simda\", 6: \"Tracur\", 7: \"Kelihos_ver1\", 8: \"Obfuscator.ACY\", 9: \"Gatak\"\n}\ndf[\"Family\"] = df[\"Class\"].map(family_map)\n\n# Smart sampling: get min(10, count) rows per class\nsampled_df = df.groupby(\"Class\", group_keys=False).apply(\n    lambda x: x.sample(n=min(10, len(x)), random_state=42)\n).reset_index(drop=True)\n\n# Show count per class in the sample\nprint(sampled_df[\"Class\"].value_counts())\n\n# Preview result\nsampled_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:22:17.826652Z","iopub.execute_input":"2025-07-21T05:22:17.827140Z","iopub.status.idle":"2025-07-21T05:22:17.882428Z","shell.execute_reply.started":"2025-07-21T05:22:17.827115Z","shell.execute_reply":"2025-07-21T05:22:17.881596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for cls in sorted(sampled_df[\"Class\"].unique()):\n    ids = sampled_df[sampled_df[\"Class\"] == cls][\"Id\"].tolist()\n    print(f\"Class {cls} Sample IDs:\\n\", ids, \"\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:22:23.547318Z","iopub.execute_input":"2025-07-21T05:22:23.547584Z","iopub.status.idle":"2025-07-21T05:22:23.559826Z","shell.execute_reply.started":"2025-07-21T05:22:23.547564Z","shell.execute_reply":"2025-07-21T05:22:23.558531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels_df[\"Family\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:22:28.436790Z","iopub.execute_input":"2025-07-21T05:22:28.437414Z","iopub.status.idle":"2025-07-21T05:22:28.444098Z","shell.execute_reply.started":"2025-07-21T05:22:28.437390Z","shell.execute_reply":"2025-07-21T05:22:28.443288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(10, 6))\nsns.countplot(data=labels_df, y=\"Family\", order=labels_df[\"Family\"].value_counts().index)\nplt.title(\"Malware Family Distribution\")\nplt.xlabel(\"Number of Samples\")\nplt.ylabel(\"Malware Family\")\nplt.show()\n\nimport os\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\ndef extract_byte_histogram(file_path):\n    counts = np.zeros(256, dtype=int)\n    try:\n        with open(file_path, 'r') as f:\n            for line in f:\n                parts = line.strip().split()[1:]  # skip the address\n                for byte_str in parts:\n                    if byte_str != '??':\n                        try:\n                            byte_val = int(byte_str, 16)\n                            counts[byte_val] += 1\n                        except ValueError:\n                            continue\n    except Exception as e:\n        print(f\"Error reading {file_path}: {e}\")\n    return counts\n\n# Path to dataset\ndata_path = \"/kaggle/input/malware-classification\"\nbytes_path = os.path.join(data_path, \"train\")\n\n# Load labels\nlabel_df = pd.read_csv(os.path.join(data_path, \"trainLabels.csv\"))\nsample_ids = label_df[\"Id\"].tolist()[:100]  # First 100 samples for demo\nlabel_map = dict(zip(label_df[\"Id\"], label_df[\"Class\"]))\n\nX = []\ny = []\nfile_names = []\n\nfor file_id in tqdm(sample_ids):\n    file_path = os.path.join(bytes_path, file_id + \".bytes\")\n    if os.path.exists(file_path):\n        hist = extract_byte_histogram(file_path)\n        X.append(hist)\n        y.append(label_map[file_id])\n        file_names.append(file_id)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:22:31.554842Z","iopub.execute_input":"2025-07-21T05:22:31.555393Z","iopub.status.idle":"2025-07-21T05:22:32.026393Z","shell.execute_reply.started":"2025-07-21T05:22:31.555370Z","shell.execute_reply":"2025-07-21T05:22:32.025661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create the feature DataFrame\ndf_features = pd.DataFrame(X, columns=[f'byte_{i:02X}' for i in range(256)])\ndf_features[\"label\"] = y\ndf_features[\"Id\"] = file_names\n\n# Map label → malware family\nfamily_map = {\n    1: \"Ramnit\", 2: \"Lollipop\", 3: \"Kelihos_ver3\", 4: \"Vundo\",\n    5: \"Simda\", 6: \"Tracur\", 7: \"Kelihos_ver1\", 8: \"Obfuscator.ACY\", 9: \"Gatak\"\n}\ndf_features[\"Family\"] = df_features[\"label\"].map(family_map)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:22:38.260965Z","iopub.execute_input":"2025-07-21T05:22:38.261570Z","iopub.status.idle":"2025-07-21T05:22:38.271310Z","shell.execute_reply.started":"2025-07-21T05:22:38.261546Z","shell.execute_reply":"2025-07-21T05:22:38.270463Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count NaN values\nprint(\"Total NaNs:\", df_features.isna().sum().sum())\n\n# Check data types\nprint(\"Data types:\\n\", df_features.dtypes.value_counts())\n\n# Preview suspicious rows\nprint(df_features[df_features.isna().any(axis=1)].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:22:42.045471Z","iopub.execute_input":"2025-07-21T05:22:42.045970Z","iopub.status.idle":"2025-07-21T05:22:42.061437Z","shell.execute_reply.started":"2025-07-21T05:22:42.045947Z","shell.execute_reply":"2025-07-21T05:22:42.060492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check if any rows are all NaN or all zeros\nprint(df_features.isnull().sum().sum())         # Total NaNs\nprint((df_features.drop(columns=[\"label\"], errors='ignore') == 0).all(axis=1).sum())  # All-zero rows\n\n# Optionally drop NaNs\ndf_features = df_features.dropna()\n\ndf_features.head()\ndf_features.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:22:46.466503Z","iopub.execute_input":"2025-07-21T05:22:46.466787Z","iopub.status.idle":"2025-07-21T05:22:46.487159Z","shell.execute_reply.started":"2025-07-21T05:22:46.466767Z","shell.execute_reply":"2025-07-21T05:22:46.486284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!mkdir -p /kaggle/working/bytes\n!7z l /kaggle/input/malware-classification/train.7z | grep '.bytes' | awk '{print $NF}' | head -n 2000 > /kaggle/working/bytes/bytes_list.txt\n\n# Now extract just these 2000 files\n!7z e /kaggle/input/malware-classification/train.7z -o/kaggle/working/bytes -i@/kaggle/working/bytes/bytes_list.txt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:22:53.727184Z","iopub.execute_input":"2025-07-21T05:22:53.727891Z","iopub.status.idle":"2025-07-21T05:25:20.602994Z","shell.execute_reply.started":"2025-07-21T05:22:53.727862Z","shell.execute_reply":"2025-07-21T05:25:20.602268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nfrom tqdm import tqdm\n\ndef extract_byte_histogram(file_path):\n    try:\n        with open(file_path, 'r') as file:\n            hex_lines = file.readlines()\n        bytes_list = []\n        for line in hex_lines:\n            parts = line.strip().split()\n            bytes_seq = parts[1:]  # ignore address part\n            bytes_list.extend([b for b in bytes_seq if b != '??'])\n        byte_vals = [int(b, 16) for b in bytes_list if len(b) == 2]\n        hist = np.histogram(byte_vals, bins=256, range=(0, 255))[0]\n        return hist\n    except:\n        return np.zeros(256)\n\n# Extract features for a small sample of files\nfile_dir = '/kaggle/working/bytes'\nsample_files = os.listdir(file_dir)[:2000]  # You can increase to 1000+\n\nX = []\nfile_ids = []\n\nfor fname in tqdm(sample_files):\n    if fname.endswith('.bytes'):\n        f_id = fname.replace(\".bytes\", \"\")\n        hist = extract_byte_histogram(os.path.join(file_dir, fname))\n        X.append(hist)\n        file_ids.append(f_id)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:25:46.221343Z","iopub.execute_input":"2025-07-21T05:25:46.222060Z","iopub.status.idle":"2025-07-21T05:37:50.479620Z","shell.execute_reply.started":"2025-07-21T05:25:46.222031Z","shell.execute_reply":"2025-07-21T05:37:50.478911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_features = pd.DataFrame(X, columns=[f'byte_{i:02X}' for i in range(256)])\ndf_features[\"Id\"] = file_ids\ndf_features = df_features.merge(labels_df, on=\"Id\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:44:01.390252Z","iopub.execute_input":"2025-07-21T05:44:01.390530Z","iopub.status.idle":"2025-07-21T05:44:02.329789Z","shell.execute_reply.started":"2025-07-21T05:44:01.390508Z","shell.execute_reply":"2025-07-21T05:44:02.329247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_features.describe()\n\n# Correlation heatmap (optional)\nimport seaborn as sns\nplt.figure(figsize=(12, 8))\nsns.heatmap(df_features.drop(columns=[\"Id\", \"Class\", \"Family\"]).corr(), cmap=\"viridis\")\nplt.title(\"Feature Correlation\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:44:08.881698Z","iopub.execute_input":"2025-07-21T05:44:08.882192Z","iopub.status.idle":"2025-07-21T05:44:10.218699Z","shell.execute_reply.started":"2025-07-21T05:44:08.882171Z","shell.execute_reply":"2025-07-21T05:44:10.218061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_features.to_csv(\"/kaggle/working/byte_histogram_features.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:44:16.739691Z","iopub.execute_input":"2025-07-21T05:44:16.740383Z","iopub.status.idle":"2025-07-21T05:44:16.900511Z","shell.execute_reply.started":"2025-07-21T05:44:16.740361Z","shell.execute_reply":"2025-07-21T05:44:16.899966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Create correlation matrix\ncorr_matrix = df_features.drop(columns=[\"Id\", \"Class\", \"Family\"]).corr().abs()\n\n# Select upper triangle of correlation matrix\nupper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))\n\n# Find features with correlation greater than 0.95\nto_drop = [column for column in upper.columns if any(upper[column] > 0.95)]\n\n# Drop those features\ndf_reduced = df_features.drop(columns=to_drop)\nprint(f\"Dropped {len(to_drop)} highly correlated features.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:44:21.387179Z","iopub.execute_input":"2025-07-21T05:44:21.387453Z","iopub.status.idle":"2025-07-21T05:44:21.772777Z","shell.execute_reply.started":"2025-07-21T05:44:21.387431Z","shell.execute_reply":"2025-07-21T05:44:21.772012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nle = LabelEncoder()\ndf_features[\"Family\"] = le.fit_transform(df_features[\"Family\"])  # You can also use df_reduced if used earlier","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:45:27.927218Z","iopub.execute_input":"2025-07-21T05:45:27.927756Z","iopub.status.idle":"2025-07-21T05:45:27.933091Z","shell.execute_reply.started":"2025-07-21T05:45:27.927734Z","shell.execute_reply":"2025-07-21T05:45:27.932402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = df_reduced.drop(columns=[\"Id\", \"Class\", \"Family\"])\ny = df_reduced[\"Family\"]\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:45:38.290451Z","iopub.execute_input":"2025-07-21T05:45:38.290749Z","iopub.status.idle":"2025-07-21T05:45:38.300167Z","shell.execute_reply.started":"2025-07-21T05:45:38.290730Z","shell.execute_reply":"2025-07-21T05:45:38.299644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, classification_report\n\nclf = RandomForestClassifier(n_estimators=100, random_state=42)\nclf.fit(X_train, y_train)\n\ny_pred = clf.predict(X_test)\n\n# Evaluate\nprint(\"Accuracy:\", accuracy_score(y_test, y_pred))\nprint(classification_report(y_test, y_pred, target_names=le.classes_))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:45:44.458415Z","iopub.execute_input":"2025-07-21T05:45:44.458700Z","iopub.status.idle":"2025-07-21T05:45:45.655576Z","shell.execute_reply.started":"2025-07-21T05:45:44.458670Z","shell.execute_reply":"2025-07-21T05:45:45.654798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import learning_curve\nimport numpy as np\nimport matplotlib.pyplot as plt\n\ntrain_sizes, train_scores, val_scores = learning_curve(\n    estimator=clf,  # your trained model\n    X=X, y=y,\n    train_sizes=np.linspace(0.1, 1.0, 10),\n    cv=5,\n    scoring='accuracy',\n    n_jobs=-1\n)\n\ntrain_mean = np.mean(train_scores, axis=1)\nval_mean = np.mean(val_scores, axis=1)\n\nplt.figure(figsize=(10,6))\nplt.plot(train_sizes, train_mean, label='Training Accuracy')\nplt.plot(train_sizes, val_mean, label='Validation Accuracy')\nplt.xlabel('Training Set Size')\nplt.ylabel('Accuracy')\nplt.title('Learning Curve')\nplt.legend()\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:45:53.843701Z","iopub.execute_input":"2025-07-21T05:45:53.843987Z","iopub.status.idle":"2025-07-21T05:46:09.218747Z","shell.execute_reply.started":"2025-07-21T05:45:53.843966Z","shell.execute_reply":"2025-07-21T05:46:09.217984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\nimport matplotlib.pyplot as plt\n\n# Create confusion matrix\ncm = confusion_matrix(y_test, y_pred)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=le.classes_)\n\n# Set figure size before plotting\nplt.figure(figsize=(12, 10))\ndisp.plot(cmap='Blues', xticks_rotation=90)\nplt.title('Confusion Matrix')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:47:18.129337Z","iopub.execute_input":"2025-07-21T05:47:18.129949Z","iopub.status.idle":"2025-07-21T05:47:18.628646Z","shell.execute_reply.started":"2025-07-21T05:47:18.129925Z","shell.execute_reply":"2025-07-21T05:47:18.627857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\nprint(classification_report(y_test, y_pred, target_names=le.classes_))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:47:24.693494Z","iopub.execute_input":"2025-07-21T05:47:24.693774Z","iopub.status.idle":"2025-07-21T05:47:24.713492Z","shell.execute_reply.started":"2025-07-21T05:47:24.693755Z","shell.execute_reply":"2025-07-21T05:47:24.712863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"importances = clf.feature_importances_\nindices = np.argsort(importances)[-20:]  # Top 20 important features\nplt.figure(figsize=(10, 6))\nplt.barh(range(len(indices)), importances[indices], align='center')\nplt.yticks(range(len(indices)), [X.columns[i] for i in indices])\nplt.xlabel('Importance')\nplt.title('Top 20 Feature Importances')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:47:32.851394Z","iopub.execute_input":"2025-07-21T05:47:32.851969Z","iopub.status.idle":"2025-07-21T05:47:33.113098Z","shell.execute_reply.started":"2025-07-21T05:47:32.851947Z","shell.execute_reply":"2025-07-21T05:47:33.112271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split, learning_curve\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix, ConfusionMatrixDisplay\nfrom sklearn.manifold import TSNE\nimport matplotlib.patches as mpatches # For t-SNE legend\nimport gc # For memory management\n\n# --- Your initial setup code (assuming data_path and labels_df are defined) ---\ndata_path = \"/kaggle/input/malware-classification\"\nlabels_df = pd.read_csv(f\"{data_path}/trainLabels.csv\")\n\n# Define the directory where you extracted the .bytes files (from the 7z extraction)\nfile_dir = '/kaggle/working/bytes'\n\n# --- Feature Extraction Function ---\ndef extract_byte_histogram(file_path):\n    try:\n        with open(file_path, 'r') as file:\n            hex_lines = file.readlines()\n        bytes_list = []\n        for line in hex_lines:\n            parts = line.strip().split()\n            bytes_seq = parts[1:]  # ignore address part\n            bytes_list.extend([b for b in bytes_seq if b != '??'])\n        byte_vals = [int(b, 16) for b in bytes_list if len(b) == 2]\n        hist = np.histogram(byte_vals, bins=256, range=(0, 255))[0]\n        return hist\n    except Exception as e:\n        # print(f\"Error extracting histogram from {file_path}: {e}\") # Uncomment for debugging\n        return np.zeros(256)\n\n# --- Collect extracted file IDs and filter labels ---\nsample_files = [f for f in os.listdir(file_dir) if f.endswith('.bytes')]\nsample_ids_extracted = [f.replace(\".bytes\", \"\") for f in sample_files]\nfiltered_labels_df = labels_df[labels_df['Id'].isin(sample_ids_extracted)].copy()\n\nX_hist = []\nextracted_file_ids = []\n\nprint(f\"Starting feature extraction for {len(sample_files)} files...\")\nfor fname in tqdm(sample_files):\n    f_id = fname.replace(\".bytes\", \"\")\n    hist = extract_byte_histogram(os.path.join(file_dir, fname))\n    X_hist.append(hist)\n    extracted_file_ids.append(f_id)\nprint(\"Feature extraction complete.\")\n\n# --- Create Feature DataFrame and Merge Labels ---\ndf_features = pd.DataFrame(X_hist, columns=[f'byte_{i:02X}' for i in range(256)])\ndf_features[\"Id\"] = extracted_file_ids\ndf_features = df_features.merge(filtered_labels_df, on=\"Id\")\n\n# --- DATA PREPARATION AND LABEL ENCODING (CRITICAL SECTION) ---\n\n# Map class numbers to family names (ensure 'Family' column with string names exists)\nfamily_map = {\n    1: \"Ramnit\", 2: \"Lollipop\", 3: \"Kelihos_ver3\", 4: \"Vundo\",\n    5: \"Simda\", 6: \"Tracur\", 7: \"Kelihos_ver1\", 8: \"Obfuscator.ACY\", 9: \"Gatak\"\n}\nif \"Family\" not in df_features.columns:\n    df_features[\"Family\"] = df_features[\"Class\"].map(family_map)\n\n# 1. Initialize LabelEncoder\nle = LabelEncoder()\n\n# 2. Apply LabelEncoder to the 'Family' column and create a NEW encoded column\ndf_features[\"Family_Encoded\"] = le.fit_transform(df_features[\"Family\"])\n\n# 3. CRITICAL FIX: Explicitly ensure the encoded column is of integer type\ndf_features[\"Family_Encoded\"] = df_features[\"Family_Encoded\"].astype(int)\n\n# --- Feature Correlation and Reduction ---\ncorr_matrix = df_features.drop(columns=[\"Id\", \"Class\", \"Family\", \"Family_Encoded\"]).corr().abs()\nupper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))\nto_drop = [column for column in upper.columns if any(upper[column] > 0.95)]\n\ndf_reduced = df_features.drop(columns=to_drop)\nprint(f\"Dropped {len(to_drop)} highly correlated features.\")\n\n# --- Define X (features) and y (encoded labels) for Model Training ---\nX = df_reduced.drop(columns=[\"Id\", \"Class\", \"Family\", \"Family_Encoded\"])\ny = df_reduced[\"Family_Encoded\"] # This 'y' now definitively contains integers\n\nX = X.fillna(0) # Handle potential NaN values\n\n# --- Train-Test Split ---\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)\n\n# --- Random Forest Classifier ---\nprint(\"\\n--- Training RandomForestClassifier ---\")\nclf = RandomForestClassifier(n_estimators=100, random_state=42, n_jobs=-1)\nclf.fit(X_train, y_train)\n\ny_pred = clf.predict(X_test)\n\n# Evaluate\nprint(\"Accuracy:\", accuracy_score(y_test, y_pred))\nprint(classification_report(y_test, y_pred, target_names=le.classes_))\n\n# --- Learning Curve ---\nprint(\"\\n--- Generating Learning Curve ---\")\ntrain_sizes, train_scores, val_scores = learning_curve(\n    estimator=clf, X=X, y=y, # Use full X and y for learning curve\n    train_sizes=np.linspace(0.1, 1.0, 10), cv=5, scoring='accuracy', n_jobs=-1\n)\ntrain_mean = np.mean(train_scores, axis=1)\nval_mean = np.mean(val_scores, axis=1)\n\nplt.figure(figsize=(10,6))\nplt.plot(train_sizes, train_mean, label='Training Accuracy')\nplt.plot(train_sizes, val_mean, label='Validation Accuracy')\nplt.xlabel('Training Set Size')\nplt.ylabel('Accuracy')\nplt.title('Learning Curve')\nplt.legend()\nplt.grid()\nplt.show()\n\n# --- Confusion Matrix ---\nprint(\"\\n--- Generating Confusion Matrix ---\")\ncm = confusion_matrix(y_test, y_pred)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=le.classes_)\nplt.figure(figsize=(12, 10))\ndisp.plot(cmap='Blues', xticks_rotation=90, ax=plt.gca())\nplt.title('Confusion Matrix')\nplt.show()\n\n# --- Feature Importances ---\nprint(\"\\n--- Generating Feature Importances Plot ---\")\nimportances = clf.feature_importances_\nindices = np.argsort(importances)[-20:]\nplt.figure(figsize=(10, 6))\nplt.barh(range(len(indices)), importances[indices], align='center')\nplt.yticks(range(len(indices)), [X.columns[i] for i in indices])\nplt.xlabel('Importance')\nplt.title('Top 20 Feature Importances')\nplt.tight_layout()\nplt.show()\n\n# --- t-SNE Visualization ---\nprint(\"\\n--- Starting t-SNE Visualization ---\")\ntsne_sample_size = 2000 # Adjust based on your extracted data size and memory\n\nX_tsne_input = X\ny_tsne_input = y\n\nif len(X) > tsne_sample_size:\n    print(f\"Sampling {tsne_sample_size} data points for t-SNE visualization...\")\n    sample_indices = np.random.choice(X.index, tsne_sample_size, replace=False)\n    X_tsne_input = X.loc[sample_indices]\n    y_tsne_input = y.loc[sample_indices]\nelse:\n    print(\"Using full dataset for t-SNE as it's within sample limit.\")\n\nprint(\"t-SNE computation may take a while...\")\ntsne = TSNE(n_components=2, random_state=42, perplexity=30, n_iter=1000, learning_rate=200, n_jobs=-1)\nX_2d = tsne.fit_transform(X_tsne_input)\n\nprint(\"t-SNE computation complete.\")\n\n# Create patches for the legend\nunique_labels = np.unique(y_tsne_input) # This should now contain integers\n\ncmap = plt.colormaps.get_cmap('tab20')\n\npatches = [mpatches.Patch(color=cmap(i / (len(unique_labels) - 1)) if len(unique_labels) > 1 else cmap(0.5),\n                          label=le.inverse_transform([i])[0])\n           for i in unique_labels]\n\nplt.figure(figsize=(12, 10))\nscatter = plt.scatter(X_2d[:, 0], X_2d[:, 1], c=y_tsne_input, cmap=cmap, s=10, alpha=0.7)\n\nplt.title(\"t-SNE Visualization of Malware Families\")\nplt.xlabel(\"t-SNE Component 1\")\nplt.ylabel(\"t-SNE Component 2\")\nplt.legend(handles=patches, bbox_to_anchor=(1.02, 1), loc='upper left', title=\"Malware Family\")\nplt.grid(True, linestyle='--', alpha=0.6)\nplt.tight_layout(rect=[0, 0, 0.88, 1])\nplt.show()\n\n# Clean up memory\n# del X_hist, X, y, X_train, X_test, y_train, y_test, df_features, df_reduced, tsne, X_2d, X_tsne_input, y_tsne_input\n# gc.collect()\n\nprint(\"\\n--- Malware Family Classification workflow complete ---\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T05:48:07.463059Z","iopub.execute_input":"2025-07-21T05:48:07.463738Z","iopub.status.idle":"2025-07-21T06:00:35.738090Z","shell.execute_reply.started":"2025-07-21T05:48:07.463715Z","shell.execute_reply":"2025-07-21T06:00:35.737153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---------------------------------------------\n# 0)  PREP  (run once)\n# ---------------------------------------------\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split, cross_val_score\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.ensemble import RandomForestClassifier, ExtraTreesClassifier, GradientBoostingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.neural_network import MLPClassifier\nimport warnings, matplotlib.pyplot as plt\nwarnings.filterwarnings('ignore')\n\n# --- features & labels ---\n# Ensure df_features is defined from previous steps (containing 'Family_Encoded' and other features)\nX = df_features.drop(columns=[\"Id\", \"Class\", \"Family\", \"Family_Encoded\"]) # Drop 'Family_Encoded' from X\ny = df_features[\"Family_Encoded\"] # y already holds the encoded labels\n\n# Train-test split just for final hold-out evaluation\nX_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y\n)\n# The rest of your model zoo, cross-validation, and evaluation code follows","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T06:05:10.027656Z","iopub.execute_input":"2025-07-21T06:05:10.028324Z","iopub.status.idle":"2025-07-21T06:05:10.040172Z","shell.execute_reply.started":"2025-07-21T06:05:10.028300Z","shell.execute_reply":"2025-07-21T06:05:10.039468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---------------------------------------------\n# 1)  DEFINE A SMALL MODEL ZOO\n# ---------------------------------------------\nmodels = {\n    \"RandomForest\": RandomForestClassifier(n_estimators=300, random_state=42, n_jobs=-1),\n    \"ExtraTrees\"  : ExtraTreesClassifier(n_estimators=400, random_state=42, n_jobs=-1),\n    \n    # algorithms that need scaling are wrapped in a Pipeline\n    \"LogReg\" : Pipeline([\n        (\"scaler\", StandardScaler()), \n        (\"clf\", LogisticRegression(max_iter=500, multi_class=\"multinomial\"))\n    ]),\n    \n    \"SVM‑RBF\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", SVC(kernel=\"rbf\", C=5, gamma=\"scale\"))\n    ]),\n    \n    \"kNN‑10\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", KNeighborsClassifier(n_neighbors=10))\n    ]),\n    \n    \"GradBoost\": GradientBoostingClassifier(random_state=42),\n    \n    \"MLP‑128x64\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", MLPClassifier(hidden_layer_sizes=(128,64), max_iter=200, random_state=42))\n    ]),\n}\n\n# (Optional) add XGBoost or LightGBM if their libraries are installed:\n# import xgboost as xgb\n# models[\"XGBoost\"] = xgb.XGBClassifier(\n#     n_estimators=500, max_depth=7, learning_rate=0.1,\n#     subsample=0.8, colsample_bytree=0.8, objective=\"multi:softprob\",\n#     num_class=len(np.unique(y)), tree_method=\"hist\", random_state=42\n# )\n\n# ---------------------------------------------\n# 2)  CROSS‑VALIDATE EACH MODEL\n# ---------------------------------------------\nresults = {}\nfor name, clf in models.items():\n    cv_scores = cross_val_score(clf, X_train, y_train, cv=5, scoring=\"accuracy\", n_jobs=-1)\n    results[name] = {\n        \"CV mean\":  cv_scores.mean(),\n        \"CV std\" :  cv_scores.std()\n    }\n    print(f\"{name:12s}  |  CV accuracy = {cv_scores.mean():.4f} ± {cv_scores.std():.4f}\")\n\n# ---------------------------------------------\n# 3)  RANKED SUMMARY\n# ---------------------------------------------\nsummary = (pd.DataFrame(results)\n           .T.sort_values(\"CV mean\", ascending=False)\n           .style.format({\"CV mean\":\"{:.4f}\", \"CV std\":\"{:.4f}\"}))\ndisplay(summary)\n\n# ---------------------------------------------\n# 4)  TRAIN THE BEST MODEL ON FULL TRAIN SET, EVALUATE ON HOLD‑OUT\n# ---------------------------------------------\nbest_name = summary.data.index[0]\nbest_model = models[best_name]\nbest_model.fit(X_train, y_train)\ny_pred = best_model.predict(X_test)\n\nprint(f\"\\n🏆 Best model: {best_name}\")\nprint(\"Hold‑out accuracy:\", accuracy_score(y_test, y_pred))\nprint(classification_report(y_test, y_pred, target_names=le.classes_))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T06:06:38.496544Z","iopub.execute_input":"2025-07-21T06:06:38.496799Z","iopub.status.idle":"2025-07-21T06:10:21.507642Z","shell.execute_reply.started":"2025-07-21T06:06:38.496784Z","shell.execute_reply":"2025-07-21T06:10:21.506875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"importances = best_model.feature_importances_\nfeat_names = X.columns\n\nfeat_imp = pd.Series(importances, index=feat_names).sort_values(ascending=False)\ntop_features = feat_imp.head(20)\n\nplt.figure(figsize=(10,6))\ntop_features.plot(kind=\"barh\", color='steelblue')\nplt.gca().invert_yaxis()\nplt.title(\"Top 20 Feature Importances (ExtraTrees)\")\nplt.xlabel(\"Importance Score\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T06:11:30.574120Z","iopub.execute_input":"2025-07-21T06:11:30.574549Z","iopub.status.idle":"2025-07-21T06:11:30.960740Z","shell.execute_reply.started":"2025-07-21T06:11:30.574528Z","shell.execute_reply":"2025-07-21T06:11:30.959940Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import learning_curve\n\ntrain_sizes, train_scores, test_scores = learning_curve(\n    best_model, X, y, cv=5, train_sizes=np.linspace(0.1, 1.0, 10), n_jobs=-1\n)\n\ntrain_mean = train_scores.mean(axis=1)\ntest_mean = test_scores.mean(axis=1)\n\nplt.figure(figsize=(8, 5))\nplt.plot(train_sizes, train_mean, label=\"Train Score\", marker=\"o\")\nplt.plot(train_sizes, test_mean, label=\"CV Score\", marker=\"s\")\nplt.title(\"Learning Curve for ExtraTreesClassifier\")\nplt.xlabel(\"Training Set Size\")\nplt.ylabel(\"Accuracy\")\nplt.legend()\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T06:11:35.808670Z","iopub.execute_input":"2025-07-21T06:11:35.809050Z","iopub.status.idle":"2025-07-21T06:11:58.118341Z","shell.execute_reply.started":"2025-07-21T06:11:35.809027Z","shell.execute_reply":"2025-07-21T06:11:58.117360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import validation_curve\n\nparam_range = [50, 100, 200, 300, 400, 500]\ntrain_scores, test_scores = validation_curve(\n    ExtraTreesClassifier(random_state=42),\n    X, y, param_name=\"n_estimators\", param_range=param_range,\n    cv=5, scoring=\"accuracy\", n_jobs=-1\n)\n\ntrain_mean = train_scores.mean(axis=1)\ntest_mean = test_scores.mean(axis=1)\n\nplt.plot(param_range, train_mean, label=\"Train\", marker='o')\nplt.plot(param_range, test_mean, label=\"CV\", marker='s')\nplt.title(\"Validation Curve for n_estimators\")\nplt.xlabel(\"Number of Estimators\")\nplt.ylabel(\"Accuracy\")\nplt.legend()\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T06:12:24.099137Z","iopub.execute_input":"2025-07-21T06:12:24.099960Z","iopub.status.idle":"2025-07-21T06:12:36.048163Z","shell.execute_reply.started":"2025-07-21T06:12:24.099936Z","shell.execute_reply":"2025-07-21T06:12:36.047270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# Ensure 'results' dictionary is available by running the model training and\n# cross-validation section of your notebook before this code block.\n# Example 'results' dictionary structure:\n# results = {\n#     \"RandomForest\": {\"CV mean\": 0.9587, \"CV std\": 0.0109},\n#     \"ExtraTrees\": {\"CV mean\": 0.9619, \"CV std\": 0.0087},\n#     \"LogReg\": {\"CV mean\": 0.8580, \"CV std\": 0.0192},\n#     \"SVM‑RBF\": {\"CV mean\": 0.7830, \"CV std\": 0.0181},\n#     \"kNN‑10\": {\"CV mean\": 0.8199, \"CV std\": 0.0115},\n#     \"GradBoost\": {\"CV mean\": 0.9481, \"CV std\": 0.0210},\n#     \"MLP‑128x64\": {\"CV mean\": 0.9131, \"CV std\": 0.0164},\n# }\n\n\n# Convert results to a DataFrame for easier plotting and sorting\ndf_results = pd.DataFrame(results).T\ndf_results = df_results.sort_values(by=\"CV mean\", ascending=False)\n\n# Create the bar plot\nplt.figure(figsize=(12, 7))\nbars = plt.bar(df_results.index, df_results[\"CV mean\"], yerr=df_results[\"CV std\"], capsize=5, color='skyblue')\n\n# Add accuracy values on top of the bars\nfor bar in bars:\n    yval = bar.get_height()\n    # Adjust position for text based on max standard deviation to avoid overlap\n    plt.text(bar.get_x() + bar.get_width()/2, yval + df_results[\"CV std\"].max() * 0.02,\n             f'{yval:.4f}', ha='center', va='bottom', fontsize=9)\n\nplt.xlabel('Model')\nplt.ylabel('Cross-Validation Accuracy')\nplt.title('Model Performance Comparison (Cross-Validation Accuracy)')\n# Adjust y-limits to make space for text labels and error bars\nplt.ylim(min(df_results[\"CV mean\"]) * 0.95, max(df_results[\"CV mean\"]) * 1.05 + df_results[\"CV std\"].max())\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.xticks(rotation=45, ha='right') # Rotate model names for better readability\nplt.tight_layout() # Adjust layout to prevent labels from overlapping\nplt.show() # Display the plot in the Kaggle notebook","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T06:14:38.232359Z","iopub.execute_input":"2025-07-21T06:14:38.232672Z","iopub.status.idle":"2025-07-21T06:14:38.595048Z","shell.execute_reply.started":"2025-07-21T06:14:38.232653Z","shell.execute_reply":"2025-07-21T06:14:38.594289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---------------------------------------------\n# 0)  PREP  (Ensure df_features and le are defined from previous cells)\n# ---------------------------------------------\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split, cross_val_score\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder # Make sure LabelEncoder is imported and 'le' is defined\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.ensemble import RandomForestClassifier, ExtraTreesClassifier, GradientBoostingClassifier, VotingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.neural_network import MLPClassifier\nimport warnings, matplotlib.pyplot as plt\nwarnings.filterwarnings('ignore')\n\n# --- features & labels ---\n# Assuming df_features and le (LabelEncoder instance) are already defined from previous data preparation steps.\n# Example:\n# df_features = ... (your DataFrame with extracted features and merged labels including 'Family_Encoded')\n# le = ... (your trained LabelEncoder instance)\n\n# Make sure X and y are defined from your df_features.\n# If 'Family_Encoded' is already the integer column, use it directly.\nX = df_features.drop(columns=[\"Id\", \"Class\", \"Family\", \"Family_Encoded\"])\ny = df_features[\"Family_Encoded\"] # This 'y' already holds the encoded integer labels\n\n# Train-test split just for final hold-out evaluation\nX_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y\n)\n\n# ---------------------------------------------\n# 1) DEFINE A SMALL MODEL ZOO (Individual Classifiers)\n# ---------------------------------------------\nmodels = {\n    \"RandomForest\": RandomForestClassifier(n_estimators=300, random_state=42, n_jobs=-1),\n    \"ExtraTrees\": ExtraTreesClassifier(n_estimators=400, random_state=42, n_jobs=-1),\n\n    # algorithms that need scaling are wrapped in a Pipeline\n    \"LogReg\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", LogisticRegression(max_iter=500, multi_class=\"multinomial\"))\n    ]),\n\n    \"SVM‑RBF\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", SVC(kernel=\"rbf\", C=5, gamma=\"scale\")) # Note: For soft voting, need probability=True\n    ]),\n\n    \"kNN‑10\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", KNeighborsClassifier(n_neighbors=10))\n    ]),\n\n    \"GradBoost\": GradientBoostingClassifier(random_state=42),\n\n    \"MLP‑128x64\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", MLPClassifier(hidden_layer_sizes=(128,64), max_iter=200, random_state=42))\n    ]),\n}\n\nprint(\"Individual models defined.\")\n\n# ---------------------------------------------\n# 2) DEFINE ENSEMBLE COMBINATIONS\n# ---------------------------------------------\nensemble_models = {\n    \"Ensemble_RF_ET_GB_Hard\": VotingClassifier(\n        estimators=[\n            ('rf', models['RandomForest']),\n            ('et', models['ExtraTrees']),\n            ('gb', models['GradBoost'])\n        ],\n        voting='hard',\n        n_jobs=-1\n    ),\n    \"Ensemble_RF_ET_GB_Soft\": VotingClassifier(\n        estimators=[\n            ('rf', models['RandomForest']),\n            ('et', models['ExtraTrees']),\n            ('gb', models['GradBoost'])\n        ],\n        voting='soft',\n        n_jobs=-1\n    ),\n    \"Ensemble_RF_ET_GB_MLP_Soft\": VotingClassifier(\n        estimators=[\n            ('rf', models['RandomForest']),\n            ('et', models['ExtraTrees']),\n            ('gb', models['GradBoost']),\n            ('mlp', models['MLP‑128x64'])\n        ],\n        voting='soft',\n        n_jobs=-1\n    ),\n}\n\nprint(\"Ensemble models defined.\")\n\n# ---------------------------------------------\n# 3) CROSS-VALIDATE ALL MODELS (Individual + Ensembles)\n# ---------------------------------------------\n# Start with results from individual models\n# Note: if you have already run this part and have a 'results' dict, you can use it.\n# Otherwise, it will be populated here.\nif 'results' not in locals(): # Check if 'results' from individual models is already defined\n    results = {}\n\nprint(\"\\n--- Cross-validating individual models ---\")\nfor name, clf in models.items():\n    if name not in results: # Avoid re-running if already done\n        cv_scores = cross_val_score(clf, X_train, y_train, cv=5, scoring=\"accuracy\", n_jobs=-1)\n        results[name] = {\n            \"CV mean\": cv_scores.mean(),\n            \"CV std\": cv_scores.std()\n        }\n        print(f\"{name:12s} | CV accuracy = {cv_scores.mean():.4f} ± {cv_scores.std():.4f}\")\n    else:\n        print(f\"{name:12s} | CV accuracy = {results[name]['CV mean']:.4f} ± {results[name]['CV std']:.4f} (already computed)\")\n\n\nprint(\"\\n--- Cross-validating ensemble models ---\")\nensemble_results = {}\nfor name, clf in ensemble_models.items():\n    cv_scores = cross_val_score(clf, X_train, y_train, cv=5, scoring=\"accuracy\", n_jobs=-1)\n    ensemble_results[name] = {\n        \"CV mean\": cv_scores.mean(),\n        \"CV std\": cv_scores.std()\n    }\n    print(f\"{name:25s} | CV accuracy = {cv_scores.mean():.4f} ± {cv_scores.std():.4f}\")\n\n\n# ---------------------------------------------\n# 4) COMBINE AND RANK ALL MODEL RESULTS (Individual + Ensembles)\n# ---------------------------------------------\ndf_individual_results = pd.DataFrame(results).T\ndf_ensemble_results = pd.DataFrame(ensemble_results).T\n\nall_results_df = pd.concat([df_individual_results, df_ensemble_results])\n\nsummary_all_models = (all_results_df\n                      .sort_values(\"CV mean\", ascending=False)\n                      .style.format({\"CV mean\":\"{:.4f}\", \"CV std\":\"{:.4f}\"}))\n\nprint(\"\\n--- Ranked Summary of All Models (Individual + Ensembles) ---\")\ndisplay(summary_all_models)\n\n# ---------------------------------------------\n# 5) TRAIN THE OVERALL BEST MODEL ON FULL TRAIN SET, EVALUATE ON HOLD-OUT\n# ---------------------------------------------\nbest_overall_name = summary_all_models.data.index[0]\n# Retrieve the best model object, checking both 'ensemble_models' and 'models' dictionaries\nbest_overall_model = ensemble_models.get(best_overall_name, models.get(best_overall_name))\n\nif best_overall_model:\n    print(f\"\\n🏆 Overall Best Model: {best_overall_name}\")\n    print(f\"Training {best_overall_name} on the full training set (X_train, y_train)...\")\n    best_overall_model.fit(X_train, y_train)\n    y_pred_overall = best_overall_model.predict(X_test)\n\n    print(\"\\nHold-out accuracy:\", accuracy_score(y_test, y_pred_overall))\n    # 'le' (LabelEncoder) should be available from your initial data preparation steps\n    print(classification_report(y_test, y_pred_overall, target_names=le.classes_))\nelse:\n    print(f\"Error: Best model '{best_overall_name}' not found in defined models. Check dictionary keys.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T06:16:19.730504Z","iopub.execute_input":"2025-07-21T06:16:19.731105Z"}},"outputs":[],"execution_count":null}]}