{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":4117,"databundleVersionId":46665,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":13014290,"sourceType":"datasetVersion","datasetId":8239405},{"sourceId":13612889,"sourceType":"datasetVersion","datasetId":8651026}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n#hello\n\n# DATA_DIR = \"/kaggle/input/malware-classification/\"\n\nprint(os.listdir(\"/kaggle/input/malware-classification/\"))\nprint(os.listdir(\"/kaggle/working\"))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-04T16:00:09.248058Z","iopub.execute_input":"2025-11-04T16:00:09.248211Z","iopub.status.idle":"2025-11-04T16:00:09.255615Z","shell.execute_reply.started":"2025-11-04T16:00:09.248195Z","shell.execute_reply":"2025-11-04T16:00:09.254874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!apt-get install -y p7zip-full\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-04T16:00:16.451371Z","iopub.execute_input":"2025-11-04T16:00:16.451644Z","iopub.status.idle":"2025-11-04T16:00:19.091891Z","shell.execute_reply.started":"2025-11-04T16:00:16.451623Z","shell.execute_reply":"2025-11-04T16:00:19.091222Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!mkdir -p /kaggle/working/sampleData\n!7z e /kaggle/input/malware-classification/dataSample.7z -o/kaggle/working/sampleData","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-04T16:00:19.093564Z","iopub.execute_input":"2025-11-04T16:00:19.093805Z","iopub.status.idle":"2025-11-04T16:00:19.929801Z","shell.execute_reply.started":"2025-11-04T16:00:19.093782Z","shell.execute_reply":"2025-11-04T16:00:19.928469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install py7zr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-04T16:00:19.931056Z","iopub.execute_input":"2025-11-04T16:00:19.931362Z","iopub.status.idle":"2025-11-04T16:00:25.528639Z","shell.execute_reply.started":"2025-11-04T16:00:19.931326Z","shell.execute_reply":"2025-11-04T16:00:25.527980Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"PREPROCESS","metadata":{}},{"cell_type":"markdown","source":"TRAINING","metadata":{}},{"cell_type":"code","source":"# ==============================================================================\n# FINAL SCRIPT: MODEL TRAINING AND VALIDATION\n# ==============================================================================\nimport os\nimport gc\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\n\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import TensorDataset, DataLoader, random_split\n\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# --- Configuration ---\nNUM_CLASSES = 9\nD_MODEL = 128\nN_HEAD = 8\nNUM_LAYERS = 3\nBATCH_SIZE = 64\nEPOCHS = 50\nLEARNING_RATE = 1e-4\n\n# --- File Paths ---\n# Assumes you've created a Kaggle Dataset and named it 'malware-preprocessed-full'\n# Change this name to match your dataset's folder in /kaggle/input/\nTRAIN_DATA_FILE = '/kaggle/input/malware-preprocessed-full/processed_train_full.npz'\nOUTPUT_MODEL_FILE = '/kaggle/working/best_malware_model.pth'\n\n# --- Device Configuration ---\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n\n# --- 1. Load Preprocessed Data ---\nprint(\"Loading preprocessed data...\")\ntrain_data = np.load(TRAIN_DATA_FILE)\nX_data, y_data = train_data['X'], train_data['y']\nprint(f\"Data loaded. Total samples: {len(X_data)}\")\n\n# --- 2. Model Architecture ---\nclass MalHybridModel(nn.Module):\n    def __init__(self, num_classes=9, input_length=4096, d_model=128, nhead=8, num_layers=3):\n        super().__init__()\n        self.embedding = nn.Embedding(num_embeddings=256, embedding_dim=d_model)\n        self.cnn_frontend = nn.Sequential(\n            nn.Conv1d(in_channels=d_model, out_channels=128, kernel_size=5, stride=1, padding=2),\n            nn.BatchNorm1d(128), nn.LeakyReLU(), nn.MaxPool1d(kernel_size=2, stride=2),\n            nn.Conv1d(in_channels=128, out_channels=256, kernel_size=5, stride=1, padding=2),\n            nn.BatchNorm1d(256), nn.LeakyReLU(), nn.MaxPool1d(kernel_size=2, stride=2),\n            nn.Conv1d(in_channels=256, out_channels=d_model, kernel_size=5, stride=1, padding=2),\n            nn.AdaptiveAvgPool1d(50)\n        )\n        encoder_layers = nn.TransformerEncoderLayer(d_model=d_model, nhead=nhead, dim_feedforward=512, dropout=0.1, batch_first=True)\n        self.transformer_encoder = nn.TransformerEncoder(encoder_layers, num_layers=num_layers)\n        self.cls_token = nn.Parameter(torch.randn(1, 1, d_model))\n        self.output_layer = nn.Linear(d_model, num_classes)\n    def forward(self, x):\n        x = self.embedding(x)\n        cnn_features = self.cnn_frontend(x.permute(0, 2, 1)).permute(0, 2, 1)\n        cls_tokens = self.cls_token.expand(x.size(0), -1, -1)\n        transformer_input = torch.cat((cls_tokens, cnn_features), dim=1)\n        encoded = self.transformer_encoder(transformer_input)[:, 0, :]\n        return self.output_layer(encoded)\n\n# --- 3. Dataset, DataLoaders, and Class Weights ---\nX_tensor = torch.from_numpy(X_data).long()\ny_tensor = torch.from_numpy(y_data).long()\ndataset = TensorDataset(X_tensor, y_tensor)\n\n# Here is the CORRECT 80/20 split\ntrain_size = int(0.8 * len(dataset))\nval_size = len(dataset) - train_size\ntrain_dataset, val_dataset = random_split(dataset, [train_size, val_size])\nprint(f\"Data split. Training samples: {len(train_dataset)}, Validation samples: {len(val_dataset)}\")\n\ntrain_loader = DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=BATCH_SIZE, shuffle=False)\n\n# Calculate weights based on the full dataset's distribution\nclass_counts = np.bincount(y_data) \nclass_weights = 1. / torch.tensor(class_counts, dtype=torch.float).to(device)\nprint(\"DataLoaders and class weights created.\")\n\n# --- 4. Training Loop ---\nmodel = MalHybridModel(num_classes=NUM_CLASSES, d_model=D_MODEL, nhead=N_HEAD, num_layers=NUM_LAYERS).to(device)\ncriterion = nn.CrossEntropyLoss(weight=class_weights)\noptimizer = torch.optim.AdamW(model.parameters(), lr=LEARNING_RATE)\nbest_val_accuracy = 0.0\nprint(\"Starting training...\")\nfor epoch in range(EPOCHS):\n    model.train()\n    # Train on the 80%\n    for data, target in tqdm(train_loader, desc=f\"Epoch {epoch+1}/{EPOCHS} [Train]\"):\n        data, target = data.to(device), target.to(device)\n        optimizer.zero_grad()\n        loss = criterion(model(data), target)\n        loss.backward()\n        optimizer.step()\n    \n    model.eval()\n    correct, total = 0, 0\n    # Validate on the 20%\n    with torch.no_grad():\n        for data, target in val_loader:\n            data, target = data.to(device), target.to(device)\n            outputs = model(data)\n            _, predicted = torch.max(outputs.data, 1)\n            total += target.size(0)\n            correct += (predicted == target).sum().item()\n    \n    val_accuracy = correct / total\n    print(f\"Epoch {epoch+1}/{EPOCHS} -> Val Accuracy: {val_accuracy:.4f}\")\n    \n    # Save the model only if its validation accuracy is the best\n    if val_accuracy > best_val_accuracy:\n        best_val_accuracy = val_accuracy\n        torch.save(model.state_dict(), OUTPUT_MODEL_FILE)\n        print(f\"New best model saved!\")\n\nprint(f\"\\nTraining finished. Best model saved to '{OUTPUT_MODEL_FILE}'\")\n\n# --- 5. Final Analysis on the Best Model ---\nprint(\"\\n--- Loading best model for final validation analysis ---\")\nmodel.load_state_dict(torch.load(OUTPUT_MODEL_FILE))\nmodel.eval()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-04T17:25:17.711397Z","iopub.execute_input":"2025-11-04T17:25:17.712125Z","iopub.status.idle":"2025-11-04T17:43:51.598952Z","shell.execute_reply.started":"2025-11-04T17:25:17.712099Z","shell.execute_reply":"2025-11-04T17:43:51.598376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Validation\n\nall_preds = []\nall_labels = []\n# Run the analysis on the same 'val_loader' (the 20% unseen data)\nwith torch.no_grad():\n    for data, target in tqdm(val_loader, desc=\"Final Validation\"):\n        data, target = data.to(device), target.to(device)\n        outputs = model(data)\n        _, predicted = torch.max(outputs.data, 1)\n        all_preds.extend(predicted.cpu().numpy())\n        all_labels.extend(target.cpu().numpy())\n\nprint(\"\\n--- Final Classification Report on Validation Set ---\")\nclass_names = [f\"Class {i+1}\" for i in range(NUM_CLASSES)]\nprint(classification_report(all_labels, all_preds, target_names=class_names, zero_division=0))\n\nprint(\"\\n--- Confusion Matrix ---\")\ncm = confusion_matrix(all_labels, all_preds)\nplt.figure(figsize=(10, 8))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=class_names, yticklabels=class_names)\nplt.xlabel('Predicted Label')\nplt.ylabel('True Label')\nplt.title('Confusion Matrix on Validation Set')\nplt.show()\n\n# ==============================================================================\n# ✨ NEW SECTION YOU REQUESTED ✨\n# ==============================================================================\nprint(\"\\n--- Per-Class Accuracy (Raw Counts) ---\")\n# The diagonal of the confusion matrix has the correct predictions\nclass_correct = cm.diagonal()\n# The sum of each row is the total number of samples for that class\nclass_total = cm.sum(axis=1)\n\nfor i in range(NUM_CLASSES):\n    # Handle cases where a class might have 0 samples in the val set\n    if class_total[i] > 0:\n        accuracy = class_correct[i] / class_total[i]\n        print(f\"**{class_names[i]}:** {class_correct[i]} / {class_total[i]}  (Accuracy: {accuracy:.4f})\")\n    else:\n        print(f\"**{class_names[i]}:** {class_correct[i]} / {class_total[i]}  (No samples in validation set)\")\n# ==============================================================================","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-04T17:45:25.112248Z","iopub.execute_input":"2025-11-04T17:45:25.113036Z","iopub.status.idle":"2025-11-04T17:45:26.953371Z","shell.execute_reply.started":"2025-11-04T17:45:25.113012Z","shell.execute_reply":"2025-11-04T17:45:26.952510Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Results Comparison","metadata":{}},{"cell_type":"code","source":"# ==============================================================================\n# SCRIPT: CLASSICAL ML BASELINE COMPARISON (EXPANDED - FULL DATASET)\n# ==============================================================================\nimport os\nimport gc\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nimport time\n\n# Import classical models and metrics\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import classification_report\n\n# --- Configuration ---\nDATA_FILE = '/kaggle/input/malware-preprocessed-full/processed_train_full.npz'\nRANDOM_STATE = 42\n\n# --- 1. Load Full Data ---\nprint(\"Loading full preprocessed dataset...\")\ndata = np.load(DATA_FILE)\nX_full, y_full = data['X'], data['y']\nprint(f\"Full dataset loaded. Total samples: {len(X_full)}\")\n\n# --- 2. Feature Engineering: Bag-of-Bytes ---\n# We engineer features on the FULL dataset\nprint(\"Engineering 'Bag-of-Bytes' (BoB) features for all samples...\")\n\ndef create_bob_features(data):\n    bob_counts = []\n    bob_freqs = []\n    for sequence in tqdm(data, desc=\"Creating BoB features\"):\n        counts = np.bincount(sequence, minlength=256)\n        # Avoid division by zero if a sequence is somehow empty\n        freq = counts / len(sequence) if len(sequence) > 0 else counts \n        bob_counts.append(counts)\n        bob_freqs.append(freq)\n    return np.array(bob_counts), np.array(bob_freqs)\n\nX_counts_ml, X_freqs_ml = create_bob_features(X_full)\nprint(f\"BoB features created. Counts shape: {X_counts_ml.shape}, Freqs shape: {X_freqs_ml.shape}\")\n\n# Clean up the original 4096-length data to save RAM\ndel X_full, data\ngc.collect()\n\n# --- 3. Split Data for Classical ML ---\n# We now split the FULL feature set into 80/20 train/validation\nprint(\"Splitting 100% of data into 80/20 train/validation sets...\")\nX_train_freq, X_val_freq, y_train, y_val = train_test_split(\n    X_freqs_ml, y_full, test_size=0.2, stratify=y_full, random_state=RANDOM_STATE\n)\nX_train_counts, X_val_counts, _, _ = train_test_split(\n    X_counts_ml, y_full, test_size=0.2, stratify=y_full, random_state=RANDOM_STATE\n)\nprint(f\"Data split. Training samples: {len(y_train)}, Validation samples: {len(y_val)}\")\n\n# --- 4. Define and Train Baseline Models ---\nmodels = {\n    \"Multinomial Naive Bayes (MNB)\": {\n        \"model\": MultinomialNB(), \n        \"X_train\": X_train_counts, \"X_val\": X_val_counts\n    },\n    \"K-Nearest Neighbors (KNN)\": {\n        \"model\": KNeighborsClassifier(n_neighbors=5, n_jobs=-1),\n        \"X_train\": X_train_freq, \"X_val\": X_val_freq\n    },\n    \"Logistic Regression\": {\n        \"model\": LogisticRegression(max_iter=1000, random_state=RANDOM_STATE, n_jobs=-1),\n        \"X_train\": X_train_freq, \"X_val\": X_val_freq\n    },\n    \"Random Forest\": {\n        \"model\": RandomForestClassifier(n_estimators=100, random_state=RANDOM_STATE, n_jobs=-1),\n        \"X_train\": X_train_freq, \"X_val\": X_val_freq\n    }\n}\n\nclass_names = [f\"Class {i+1}\" for i in range(9)]\n\nfor name, config in models.items():\n    print(\"\\n\" + \"=\"*50)\n    print(f\"Training {name} on {len(config['X_train'])} samples...\")\n    start_time = time.time()\n    \n    # Train the model on its specified feature set\n    model = config[\"model\"]\n    model.fit(config[\"X_train\"], y_train)\n    \n    end_time = time.time()\n    print(f\"Training finished in {end_time - start_time:.2f} seconds.\")\n    \n    # Evaluate the model\n    print(f\"\\n--- Validation Report for {name} ---\")\n    preds = model.predict(config[\"X_val\"])\n    print(classification_report(y_val, preds, target_names=class_names, zero_division=0))\n    print(\"=\"*50 + \"\\n\")\n\nprint(\"Baseline comparison complete.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-04T17:47:09.723234Z","iopub.execute_input":"2025-11-04T17:47:09.723504Z","iopub.status.idle":"2025-11-04T17:47:14.030772Z","shell.execute_reply.started":"2025-11-04T17:47:09.723483Z","shell.execute_reply":"2025-11-04T17:47:14.030008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}