{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":4117,"databundleVersionId":46665,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ============================================\n# 1. SETUP AND DATA LOADING\n# ============================================\n# WHAT'S HAPPENING: Loading all the tools (libraries) we need for the project\n# Think of these like importing different tool boxes\n\nimport os  # For working with files and folders on the computer\nimport numpy as np  # For math operations on arrays/matrices\nimport pandas as pd  # For working with data in table format (like Excel)\nfrom tqdm import tqdm  # Shows progress bars (so you know how long tasks will take)\nimport seaborn as sns  # For making pretty charts\nimport matplotlib.pyplot as plt  # For creating visualizations/plots\nfrom sklearn.model_selection import train_test_split, cross_val_score  # For splitting data and testing models\nfrom sklearn.metrics import accuracy_score, classification_report  # For measuring how good our model is\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder  # For preparing data for ML models\nfrom sklearn.pipeline import Pipeline  # For chaining multiple steps together\nfrom sklearn.ensemble import RandomForestClassifier, ExtraTreesClassifier, GradientBoostingClassifier, VotingClassifier  # Different ML algorithms\nfrom sklearn.linear_model import LogisticRegression  # Another ML algorithm\nfrom sklearn.svm import SVC  # Support Vector Machine algorithm\nfrom sklearn.neighbors import KNeighborsClassifier  # k-Nearest Neighbors algorithm\nfrom sklearn.neural_network import MLPClassifier  # Neural network algorithm\nimport warnings\nwarnings.filterwarnings('ignore')  # Hide warning messages (keeps output clean)\n\n# Data paths\ndata_path = \"/kaggle/input/malware-classification\"\n# EXPLANATION: This tells the program where to find the malware data files on Kaggle's servers\nprint(\"Available files:\", os.listdir(data_path))\n# EXPLANATION: List all files in the folder so we can see what we're working with\n\n# ============================================\n# 2. LOAD AND PREPARE LABELS\n# ============================================\n# WHAT'S HAPPENING: Loading the \"answer sheet\" that tells us which malware belongs to which family\n\nlabels_df = pd.read_csv(f\"{data_path}/trainLabels.csv\")\n# EXPLANATION: Read a CSV file (like an Excel spreadsheet) that has:\n#   - Column 1: \"Id\" - the name/ID of each malware sample\n#   - Column 2: \"Class\" - a number (1-9) representing which malware family it belongs to\n# We store this in 'labels_df' (df = DataFrame, which is like a table)\n\n# Mapping class numbers to family names\n# EXPLANATION: Numbers are hard to remember, so we create a dictionary (like a translation guide)\n# that converts: 1 → \"Ramnit\", 2 → \"Lollipop\", etc.\nfamily_map = {\n    1: \"Ramnit\", 2: \"Lollipop\", 3: \"Kelihos_ver3\", 4: \"Vundo\",\n    5: \"Simda\", 6: \"Tracur\", 7: \"Kelihos_ver1\", 8: \"Obfuscator.ACY\", 9: \"Gatak\"\n}\n\nlabels_df[\"Family\"] = labels_df[\"Class\"].map(family_map)\n# EXPLANATION: Create a new column called \"Family\" that shows the actual malware name\n# instead of just the number. For example, if Class=1, Family will be \"Ramnit\"\n\nprint(\"\\nFirst 5 rows:\")\nprint(labels_df.head())\n# EXPLANATION: Show the first 5 rows so we can verify everything looks correct\n\n# Smart sampling: get min(10, count) rows per class\n# EXPLANATION: This is taking a small sample from each malware family\n# WHY? Because we might have thousands of samples and want to see a quick preview\n# It takes up to 10 samples from each class (or all of them if there are fewer than 10)\nsampled_df = labels_df.groupby(\"Class\", group_keys=False).apply(\n    lambda x: x.sample(n=min(10, len(x)), random_state=42)\n).reset_index(drop=True)\n# BREAKDOWN:\n# - groupby(\"Class\") → separate the data by each malware class\n# - apply(lambda x: ...) → for each group, do something\n# - x.sample(n=min(10, len(x))) → take 10 samples, or all if less than 10\n# - random_state=42 → makes the random selection reproducible (same results every time)\n\nprint(\"\\nSample count per class:\")\nprint(sampled_df[\"Class\"].value_counts())\n# EXPLANATION: Show how many samples we got from each class\n\n# Display sample IDs per class\n# EXPLANATION: For each malware class, print out the IDs of the samples we selected\n# This is useful for debugging or if you want to look up specific files later\nfor cls in sorted(sampled_df[\"Class\"].unique()):\n    ids = sampled_df[sampled_df[\"Class\"] == cls][\"Id\"].tolist()\n    print(f\"Class {cls} Sample IDs:\\n\", ids, \"\\n\")\n\n# ============================================\n# 3. VISUALIZE CLASS DISTRIBUTION\n# ============================================\n# WHAT'S HAPPENING: Creating visualizations to understand our data better\n\nprint(\"\\nMalware family distribution:\")\nprint(labels_df[\"Family\"].value_counts())\n# EXPLANATION: Count how many samples we have of each malware family\n# This shows if our data is balanced (equal amounts) or imbalanced\n\n# Create a bar chart showing the distribution\nplt.figure(figsize=(10, 6))  # Set the size of the chart\nsns.countplot(data=labels_df, y=\"Family\", order=labels_df[\"Family\"].value_counts().index)\n# EXPLANATION: countplot creates a bar chart showing counts for each category\n# - data=labels_df → use this data\n# - y=\"Family\" → put malware families on the y-axis (vertical)\n# - order=... → order bars from most common to least common\nplt.title(\"Malware Family Distribution\")\nplt.xlabel(\"Number of Samples\")\nplt.ylabel(\"Malware Family\")\nplt.show()  # Display the chart\n\n# ============================================\n# 4. EXTRACT FILES FROM ARCHIVE\n# ============================================\n# WHAT'S HAPPENING: The malware files are compressed in a .7z archive (like a .zip file)\n# We need to extract them before we can analyze them\n\n# These are Linux commands (the ! tells Jupyter to run them as system commands)\n!mkdir -p /kaggle/working/bytes\n# EXPLANATION: Create a folder called \"bytes\" to store extracted files\n# -p means \"create parent directories if needed, and don't error if it already exists\"\n\n!7z l /kaggle/input/malware-classification/train.7z | grep '.bytes' | awk '{print $NF}' | head -n 2000 > /kaggle/working/bytes/bytes_list.txt\n# EXPLANATION: This is a complex command, let's break it down:\n# - 7z l ... → list all files in the archive\n# - | grep '.bytes' → filter to only show .bytes files\n# - | awk '{print $NF}' → extract just the filename from each line\n# - | head -n 2000 → take only the first 2000 files (to keep dataset manageable)\n# - > bytes_list.txt → save this list to a text file\n\n!7z e /kaggle/input/malware-classification/train.7z -o/kaggle/working/bytes -i@/kaggle/working/bytes/bytes_list.txt\n# EXPLANATION: Extract the files we listed\n# - 7z e ... → extract from archive\n# - -o/kaggle/working/bytes → output to this folder\n# - -i@bytes_list.txt → only extract files listed in this text file\n\n# ============================================\n# 5. FEATURE EXTRACTION (BYTE HISTOGRAM)\n# ============================================\n# WHAT'S HAPPENING: Converting raw malware files into numbers that ML algorithms can understand\n# This is the MOST IMPORTANT part for understanding the project!\n\ndef extract_byte_histogram(file_path):\n    \"\"\"\n    Extract byte histogram from .bytes file\n    \n    WHAT IS A BYTE HISTOGRAM?\n    - Computers store everything as numbers (bytes: 0-255)\n    - A histogram counts how often each byte value appears\n    - Think of it like: \"This malware uses byte 0x4A fifty times, byte 0xFF ten times, etc.\"\n    - Different malware families have different \"byte patterns\" (like fingerprints)\n    \n    EXAMPLE: If a file contains: [0x41, 0x41, 0x42, 0x41]\n    The histogram would be: [0, ..., 3 at position 0x41, 1 at position 0x42, ...]\n    \"\"\"\n    try:\n        # Open and read the .bytes file\n        with open(file_path, 'r') as file:\n            hex_lines = file.readlines()\n        \n        # Each line in the file looks like: \"0040F0A0  48 8B 4C 24 ?? 48 8B\"\n        # We need to extract just the hex values (48, 8B, 4C, etc.)\n        bytes_list = []\n        for line in hex_lines:\n            parts = line.strip().split()  # Split line into parts\n            bytes_seq = parts[1:]  # Skip the first part (the address like \"0040F0A0\")\n            bytes_list.extend([b for b in bytes_seq if b != '??'])  # Keep only valid hex, skip \"??\"\n        \n        # Convert hex strings to actual numbers\n        # \"48\" → 72, \"8B\" → 139, etc.\n        byte_vals = [int(b, 16) for b in bytes_list if len(b) == 2]\n        \n        # Create histogram: count how many times each value (0-255) appears\n        hist = np.histogram(byte_vals, bins=256, range=(0, 255))[0]\n        # Result: array of 256 numbers, where hist[0] = count of 0x00, hist[1] = count of 0x01, etc.\n        \n        return hist\n    except Exception as e:\n        print(f\"Error reading {file_path}: {e}\")\n        return np.zeros(256)  # If error, return array of zeros\n\n# Extract features from sample files\nfile_dir = '/kaggle/working/bytes'\nsample_files = os.listdir(file_dir)[:2500]  # Get first 2500 files\n\n# EXPLANATION: We'll process each file and convert it to features\nX = []  # This will store our features (the 256 byte counts for each file)\nfile_ids = []  # This will store which file each row corresponds to\n\n# Loop through each file and extract its byte histogram\nfor fname in tqdm(sample_files):  # tqdm shows a progress bar\n    if fname.endswith('.bytes'):  # Only process .bytes files\n        f_id = fname.replace(\".bytes\", \"\")  # Remove extension to get ID\n        hist = extract_byte_histogram(os.path.join(file_dir, fname))  # Get the histogram\n        X.append(hist)  # Add to our features list\n        file_ids.append(f_id)  # Remember which file this was\n\n# RESULT: X is now a list of lists, where each inner list has 256 numbers\n# Example: X[0] = [3, 5, 2, 100, ...] means first file had 3 0x00 bytes, 5 0x01 bytes, etc.\n\n# Create feature DataFrame\n# EXPLANATION: Convert our lists into a nice table format\ndf_features = pd.DataFrame(X, columns=[f'byte_{i:02X}' for i in range(256)])\n# This creates columns named: byte_00, byte_01, byte_02, ..., byte_FF\n# Each row is one malware sample, each column is the count of that specific byte\n\ndf_features[\"Id\"] = file_ids  # Add a column for the file ID\ndf_features = df_features.merge(labels_df, on=\"Id\")  # Add the labels (which family each belongs to)\n# RESULT: Now we have a table with 256 feature columns + Id + Class + Family\n\nprint(\"\\nFeature statistics:\")\nprint(df_features.describe())\n# EXPLANATION: Show basic stats (mean, min, max, etc.) for each byte column\n\n# Optional: Correlation heatmap\n# EXPLANATION: This shows which byte counts tend to move together\n# If byte_4A and byte_5F are highly correlated, they appear together frequently\nplt.figure(figsize=(12, 8))\nsns.heatmap(df_features.drop(columns=[\"Id\", \"Class\", \"Family\"]).corr(), cmap=\"viridis\")\n# corr() calculates correlation between all pairs of columns\n# Heatmap visualizes it: bright colors = high correlation, dark = low correlation\nplt.title(\"Feature Correlation\")\nplt.show()\n\n# Save features\ndf_features.to_csv(\"/kaggle/working/byte_histogram_features.csv\", index=False)\n# EXPLANATION: Save our processed features to a CSV file so we don't have to re-extract later\n\n# ============================================\n# 6. ENCODE LABELS\n# ============================================\n# WHAT'S HAPPENING: Converting text labels to numbers for machine learning\n# WHY? ML algorithms only understand numbers, not text like \"Ramnit\" or \"Simda\"\n\nle = LabelEncoder()\n# LabelEncoder is a tool that converts categories to numbers\n# Example: [\"Ramnit\", \"Simda\", \"Ramnit\"] → [0, 1, 0]\n\ndf_features[\"Family_Encoded\"] = le.fit_transform(df_features[\"Family\"]).astype(int)\n# BREAKDOWN:\n# - le.fit_transform() → learns the mapping and applies it\n#   \"Ramnit\" might become 0, \"Lollipop\" becomes 1, etc.\n# - .astype(int) → make sure it's an integer type (not float)\n# RESULT: New column with numbers instead of names\n\n# IMPORTANT: Store the original family names for later use in reports\nfamily_names = le.classes_\n# This saves the mapping: [0: \"Ramnit\", 1: \"Lollipop\", etc.]\n\n# ============================================\n# 7. PREPARE TRAIN/TEST SPLIT\n# ============================================\n# WHAT'S HAPPENING: Splitting our data into two groups\n# WHY? We need to test if our model actually works on NEW data it hasn't seen\n\n# X = features (the input data - the 256 byte counts)\nX = df_features.drop(columns=[\"Id\", \"Class\", \"Family\", \"Family_Encoded\"])\n# Drop the columns we don't want to use for prediction\n# We keep only the 256 byte_XX columns\n\n# y = labels (the answers we're trying to predict)\ny = df_features[\"Family_Encoded\"]\n\n# Split the data: 80% for training, 20% for testing\nX_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y\n)\n# BREAKDOWN:\n# - test_size=0.2 → 20% goes to testing\n# - random_state=42 → makes the split reproducible (same split every time)\n# - stratify=y → ensures each split has proportional representation of each class\n#   (if 30% of data is Ramnit, then 30% of train AND test will be Ramnit)\n\n# RESULT:\n# X_train: 80% of features for training\n# y_train: 80% of labels for training  \n# X_test: 20% of features for testing\n# y_test: 20% of labels for testing\n\n# ============================================\n# 8. DEFINE MODELS\n# ============================================\n# WHAT'S HAPPENING: Setting up different ML algorithms to compare their performance\n# Think of these as different \"students\" taking the same exam - we'll see who scores best!\n\nmodels = {\n    # RANDOM FOREST: Creates many decision trees and combines their votes\n    # Like asking 300 experts and taking the majority opinion\n    \"RandomForest\": RandomForestClassifier(n_estimators=300, random_state=42, n_jobs=-1),\n    \n    # EXTRA TREES: Similar to Random Forest but even more random (faster but sometimes less accurate)\n    \"ExtraTrees\": ExtraTreesClassifier(n_estimators=400, random_state=42, n_jobs=-1),\n    \n    # LOGISTIC REGRESSION: A simple, fast algorithm (good baseline)\n    # Needs data to be scaled (standardized) first, so we use a Pipeline\n    \"LogReg\": Pipeline([\n        (\"scaler\", StandardScaler()),  # Step 1: Scale the data (make all features similar range)\n        (\"clf\", LogisticRegression(max_iter=500, multi_class=\"multinomial\"))  # Step 2: Train model\n    ]),\n    \n    # SUPPORT VECTOR MACHINE (SVM): Tries to find the best boundary between classes\n    # Also needs scaling\n    \"SVM‑RBF\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", SVC(kernel=\"rbf\", C=5, gamma=\"scale\", probability=True))\n        # probability=True allows it to be used in soft voting ensembles\n    ]),\n    \n    # K-NEAREST NEIGHBORS: Classifies based on the 10 closest training examples\n    # \"You are the average of your 10 nearest neighbors\"\n    \"kNN‑10\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", KNeighborsClassifier(n_neighbors=10))\n    ]),\n    \n    # GRADIENT BOOSTING: Builds trees sequentially, each correcting the previous one's mistakes\n    \"GradBoost\": GradientBoostingClassifier(random_state=42),\n    \n    # MULTI-LAYER PERCEPTRON: A neural network with 2 hidden layers (128 and 64 neurons)\n    \"MLP‑128x64\": Pipeline([\n        (\"scaler\", StandardScaler()),\n        (\"clf\", MLPClassifier(hidden_layer_sizes=(128,64), max_iter=200, random_state=42))\n    ]),\n}\n# COMMON PARAMETERS EXPLAINED:\n# - n_estimators: number of trees to build\n# - random_state: seed for reproducibility\n# - n_jobs=-1: use all CPU cores (faster training)\n\n# ============================================\n# 9. DEFINE ENSEMBLE MODELS\n# ============================================\n# WHAT'S HAPPENING: Combining multiple models to potentially get better results\n# Think of it as a \"committee\" of experts making decisions together\n\nensemble_models = {\n    # HARD VOTING: Each model votes for a class, and the majority wins\n    # Example: RF says \"Ramnit\", ET says \"Ramnit\", GB says \"Simda\" → Final: \"Ramnit\" (2 votes)\n    \"Ensemble_RF_ET_GB_Hard\": VotingClassifier(\n        estimators=[\n            ('rf', models['RandomForest']),\n            ('et', models['ExtraTrees']),\n            ('gb', models['GradBoost'])\n        ],\n        voting='hard',  # Majority vote\n        n_jobs=-1\n    ),\n    \n    # SOFT VOTING: Each model gives probability for each class, then we average\n    # Example: RF: 70% Ramnit, ET: 60% Ramnit, GB: 50% Simda → Average and pick highest\n    # Usually performs better than hard voting\n    \"Ensemble_RF_ET_GB_Soft\": VotingClassifier(\n        estimators=[\n            ('rf', models['RandomForest']),\n            ('et', models['ExtraTrees']),\n            ('gb', models['GradBoost'])\n        ],\n        voting='soft',  # Average probabilities\n        n_jobs=-1\n    ),\n    \n    # SOFT VOTING WITH NEURAL NETWORK: Adding MLP to the committee\n    \"Ensemble_RF_ET_GB_MLP_Soft\": VotingClassifier(\n        estimators=[\n            ('rf', models['RandomForest']),\n            ('et', models['ExtraTrees']),\n            ('gb', models['GradBoost']),\n            ('mlp', models['MLP‑128x64'])\n        ],\n        voting='soft',\n        n_jobs=-1\n    ),\n}\n\nprint(\"All models defined.\")\n# We now have 7 individual models + 3 ensemble models = 10 total to compare!\n\n# ============================================\n# 10. CROSS-VALIDATE ALL MODELS\n# ============================================\n# WHAT'S HAPPENING: Testing each model to see how well it performs\n# CROSS-VALIDATION EXPLAINED:\n# Instead of just one train/test split, we do this 5 times (5-fold CV):\n# - Split 1: Train on 80%, test on 20% (different 20%)\n# - Split 2: Train on 80%, test on 20% (different 20%)\n# - ... repeat 5 times\n# Then average the 5 scores to get a more reliable estimate of performance\n# WHY? Single split might be lucky/unlucky; averaging reduces randomness\n\nresults = {}\n\nprint(\"\\n--- Cross-validating individual models ---\")\nfor name, clf in models.items():\n    # cross_val_score does the 5-fold CV automatically\n    cv_scores = cross_val_score(clf, X_train, y_train, cv=5, scoring=\"accuracy\", n_jobs=-1)\n    # cv_scores is an array with 5 accuracy values (one for each fold)\n    \n    results[name] = {\n        \"CV mean\": cv_scores.mean(),  # Average of the 5 scores\n        \"CV std\": cv_scores.std()      # Standard deviation (how much variance in scores)\n    }\n    print(f\"{name:12s} | CV accuracy = {cv_scores.mean():.4f} ± {cv_scores.std():.4f}\")\n    # Example output: \"RandomForest | CV accuracy = 0.9234 ± 0.0123\"\n    # This means: 92.34% average accuracy, with ±1.23% variation across folds\n\nprint(\"\\n--- Cross-validating ensemble models ---\")\nensemble_results = {}\nfor name, clf in ensemble_models.items():\n    cv_scores = cross_val_score(clf, X_train, y_train, cv=5, scoring=\"accuracy\", n_jobs=-1)\n    ensemble_results[name] = {\n        \"CV mean\": cv_scores.mean(),\n        \"CV std\": cv_scores.std()\n    }\n    print(f\"{name:25s} | CV accuracy = {cv_scores.mean():.4f} ± {cv_scores.std():.4f}\")\n# Now we have scores for all 10 models (7 individual + 3 ensemble)\n\n# ============================================\n# 11. RANK AND EVALUATE BEST MODEL\n# ============================================\n# WHAT'S HAPPENING: Finding which model performed best and testing it on hold-out data\n\n# Combine all results into one table\ndf_individual_results = pd.DataFrame(results).T  # Convert dict to DataFrame, .T transposes it\ndf_ensemble_results = pd.DataFrame(ensemble_results).T\nall_results_df = pd.concat([df_individual_results, df_ensemble_results])  # Stack them together\n\n# Sort by average CV score (highest first)\nsummary_all_models = all_results_df.sort_values(\"CV mean\", ascending=False)\nprint(\"\\n--- Ranked Summary of All Models ---\")\nprint(summary_all_models.to_string())\n# This will show a leaderboard of all models, best to worst\n\n# Train and evaluate best model\nbest_overall_name = summary_all_models.index[0]  # Get the name of the top model\nbest_overall_model = ensemble_models.get(best_overall_name, models.get(best_overall_name))\n# Look for the model in ensemble_models first, then in models\n\nif best_overall_model:\n    print(f\"\\n🏆 Overall Best Model: {best_overall_name}\")\n    print(f\"Training {best_overall_name} on the full training set...\")\n    \n    # THE FINAL TEST:\n    # Train the best model on ALL training data (X_train, y_train)\n    best_overall_model.fit(X_train, y_train)\n    \n    # Now predict on the hold-out test set (data it has NEVER seen before)\n    y_pred_overall = best_overall_model.predict(X_test)\n    \n    # Calculate accuracy: what % did we get correct?\n    print(\"\\nHold-out accuracy:\", accuracy_score(y_test, y_pred_overall))\n    # Example: 0.9500 means 95% correct predictions\n    \n    # Detailed report showing precision, recall, F1-score for each malware family\n    print(\"\\nClassification Report:\")\n    print(classification_report(y_test, y_pred_overall, target_names=family_names))\n    # FIXED: Using family_names (the actual string names) instead of le.classes_\n    # This shows:\n    # - Precision: Of all times we predicted \"Ramnit\", how often was it actually Ramnit?\n    # - Recall: Of all actual Ramnit samples, how many did we correctly identify?\n    # - F1-score: Harmonic mean of precision and recall (balanced metric)\n    # - Support: How many samples of each class in the test set\nelse:\n    print(f\"Error: Best model '{best_overall_name}' not found in defined models.\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-28T05:12:36.043154Z","iopub.execute_input":"2025-11-28T05:12:36.043433Z","iopub.status.idle":"2025-11-28T05:40:16.428626Z","shell.execute_reply.started":"2025-11-28T05:12:36.043410Z","shell.execute_reply":"2025-11-28T05:40:16.427654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# SUMMARY OF THE ENTIRE PROCESS:\n# ============================================\n# 1. Loaded malware file labels (which family each belongs to)\n# 2. Extracted malware files from compressed archive\n# 3. Converted each file to a 256-number \"fingerprint\" (byte histogram)\n# 4. Split data: 80% training, 20% testing\n# 5. Trained 10 different ML models\n# 6. Used cross-validation to fairly compare them\n# 7. Picked the best model and tested it on fresh data\n# 8. Got a final accuracy score showing how well we can classify malware!","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}