{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":4117,"databundleVersionId":46665,"isSourceIdPinned":false},{"sourceType":"kernelVersion","sourceId":311606830,"isSourceIdPinned":false}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# MS Malware Classification: Hybrid Architecture Pipeline\n\n**Author:** Nguyen Le Minh Quan, Phan Ngoc Thuc, Tong Phuc Thien\n\n**Objective:** A robust, zero-leakage Hybrid Pipeline. Integrates an Adaptive PyTorch CNN (spatial feature extractor) and an Optuna-tuned XGBoost classifier, strictly evaluated via Stratified K-Fold Cross-Validation.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"## Data Ingestion & Preprocessing\n\nLoad tabular features and images. Images are kept at their original aspect ratio.","metadata":{}},{"cell_type":"code","source":"import os\nimport glob\nimport gc\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport joblib\nimport warnings\nwarnings.filterwarnings('ignore')\n\nfrom imblearn.over_sampling import SMOTE\nimport optuna\nimport xgboost as xgb\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\nfrom sklearn.metrics import classification_report, accuracy_score, log_loss\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")\n\n# --- Directories ---\nTABULAR_DIR = \"/kaggle/input/notebooks/nguynlminhqun/ms-mal-classification-feature-engineering\" \nIMAGE_DIR = os.path.join(TABULAR_DIR, \"train_malware_images\")\nLABEL_PATH = \"/kaggle/input/competitions/malware-classification/trainLabels.csv\"\n\n# --- Load Tabular Data ---\nprint(\"Loading tabular features...\")\nall_csv_files = glob.glob(os.path.join(TABULAR_DIR, \"*tabular_features_batch_*.csv\"))\ndf_features = pd.concat([pd.read_csv(f) for f in all_csv_files], ignore_index=True)\ndf_labels = pd.read_csv(LABEL_PATH)\ndf_full = pd.merge(df_features, df_labels, on='Id', how='inner').sort_values('Id').reset_index(drop=True)\n\ndf_full = df_full.dropna(subset=[col for col in df_full.columns if col not in ['Id', 'Class']])\n\n# --- Encode Labels ---\nencoder = LabelEncoder()\ny = encoder.fit_transform(df_full['Class'])\nnum_classes = len(np.unique(y))\njoblib.dump(encoder, 'label_encoder.pkl')\n\n# --- Prepare Tabular Data (NO SCALING YET) ---\nX_tabular = df_full.drop(columns=['Id', 'Class']).values \nfeature_names = df_full.drop(columns=['Id', 'Class']).columns.tolist()\nimage_ids = df_full['Id'].tolist()\n\ndel df_full, df_features\ngc.collect()\nprint(f\"Loaded {len(y)} samples with {X_tabular.shape[1]} tabular features.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## The CNN Spatial Feature Extractor (Adaptive Architecture)\n\nSince we preserved the original image structure in Phase 1, we use an `AdaptiveAvgPool2d` layer. This allows the CNN to ingest images of varying heights and output a fixed-size 256-dimensional embedding. We train this CNN briefly, then extract features for the XGBoost hybrid.","metadata":{}},{"cell_type":"code","source":"# --- Custom Dataset for Variable Sized Images ---\nclass MalwareImageDataset(Dataset):\n    def __init__(self, image_dir, image_ids, labels=None):\n        self.image_dir = image_dir\n        self.image_ids = image_ids\n        self.labels = labels\n        self.transform = transforms.Compose([\n            transforms.ToTensor()\n        ])\n\n    def __len__(self):\n        return len(self.image_ids)\n\n    def __getitem__(self, idx):\n        img_path = os.path.join(self.image_dir, f\"{self.image_ids[idx]}.png\")\n        image = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)\n        if image is None:\n            image = np.zeros((224, 224), dtype=np.uint8)\n            \n        image = self.transform(image)\n        \n        if self.labels is not None:\n            label = torch.tensor(self.labels[idx], dtype=torch.long)\n            return image, label\n        return image\n\n# --- Adaptive CNN Architecture ---\nclass AdaptiveMalwareCNN(nn.Module):\n    def __init__(self, num_classes):\n        super(AdaptiveMalwareCNN, self).__init__()\n        self.features = nn.Sequential(\n            nn.Conv2d(1, 32, kernel_size=3, padding=1), nn.ReLU(), nn.MaxPool2d(2, 2),\n            nn.Conv2d(32, 64, kernel_size=3, padding=1), nn.ReLU(), nn.MaxPool2d(2, 2),\n            nn.Conv2d(64, 128, kernel_size=3, padding=1), nn.ReLU(), nn.MaxPool2d(2, 2)\n        )\n        self.adaptive_pool = nn.AdaptiveAvgPool2d((4, 4)) \n        \n        self.flatten_size = 128 * 4 * 4 # = 2048\n        \n        self.classifier = nn.Sequential(\n            nn.Linear(self.flatten_size, 256),\n            nn.ReLU(),\n            nn.Dropout(0.5),\n            nn.Linear(256, num_classes)\n        )\n\n    def forward(self, x):\n        x = self.features(x)\n        x = self.adaptive_pool(x)\n        x = x.view(x.size(0), -1)\n        x = self.classifier(x)\n        return x\n\n    def extract_features(self, x):\n        x = self.features(x)\n        x = self.adaptive_pool(x)\n        x = x.view(x.size(0), -1)\n        x = self.classifier[0](x)\n        x = self.classifier[1](x)\n        return x\n\n# --- Damped Inverse Frequency Weighting for CNN ---\nclass_counts = np.bincount(y)\ndamped_weights = np.log((len(y) / class_counts) + 1.2)\ntensor_weights = torch.tensor(damped_weights, dtype=torch.float32).to(device)\n\nprint(\"Training Adaptive CNN Feature Extractor...\")\n# Train CNN on a stratified subset to learn basic visual representations\ntrain_idx, _, y_img_t, _ = train_test_split(range(len(y)), y, test_size=0.2, random_state=42, stratify=y)\ntrain_ids = [image_ids[i] for i in train_idx]\n\ntrain_dataset = MalwareImageDataset(IMAGE_DIR, train_ids, y_img_t)\ntrain_loader = DataLoader(train_dataset, batch_size=1, shuffle=True) \n\ncnn_model = AdaptiveMalwareCNN(num_classes).to(device)\ncriterion = nn.CrossEntropyLoss(weight=tensor_weights)\noptimizer = optim.Adam(cnn_model.parameters(), lr=0.001)\n\ncnn_model.train()\nfor epoch in range(2): \n    running_loss = 0.0\n    for i, (inputs, labels) in enumerate(train_loader):\n        inputs, labels = inputs.to(device), labels.to(device)\n        optimizer.zero_grad()\n        outputs = cnn_model(inputs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        running_loss += loss.item()\n        \n    print(f\"Epoch {epoch+1} Loss: {running_loss/len(train_loader):.4f}\")\n\ntorch.save(cnn_model.state_dict(), 'adaptive_cnn_weights.pth')\nprint(\"CNN weights saved.\")\n\n# --- Extract Embeddings for ALL images ---\nprint(\"Extracting 256-D CNN embeddings...\")\ncnn_model.eval()\nfull_dataset = MalwareImageDataset(IMAGE_DIR, image_ids)\nfull_loader = DataLoader(full_dataset, batch_size=1, shuffle=False)\n\nX_cnn_features = []\nwith torch.no_grad():\n    for inputs in full_loader:\n        inputs = inputs.to(device)\n        features = cnn_model.extract_features(inputs).cpu().numpy()\n        X_cnn_features.append(features[0])\n\nX_cnn_features = np.array(X_cnn_features)\nprint(f\"CNN Embeddings Shape: {X_cnn_features.shape}\")\n\ndel cnn_model, train_dataset, train_loader, full_dataset, full_loader\ntorch.cuda.empty_cache()\ngc.collect()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Optuna Hyperparameter Tuning (Zero Leakage)\n\nOptuna executes a Bayesian search for optimal XGBoost parameters. SMOTE and Scaling are strictly applied *within* the validation split to prevent data leakage.","metadata":{}},{"cell_type":"code","source":"X_hybrid_raw = np.hstack((X_cnn_features, X_tabular))\n\noptuna.logging.set_verbosity(optuna.logging.WARNING)\n\ndef objective(trial):\n    X_tr_raw, X_va_raw, y_tr, y_va = train_test_split(X_hybrid_raw, y, test_size=0.2, random_state=42, stratify=y)\n    \n    scaler = StandardScaler()\n    X_tr_scaled = scaler.fit_transform(X_tr_raw)\n    X_va_scaled = scaler.transform(X_va_raw)\n    \n    smote = SMOTE(k_neighbors=3, random_state=42)\n    X_tr_sm, y_tr_sm = smote.fit_resample(X_tr_scaled, y_tr)\n    \n    params = {\n        'objective': 'multi:softprob', 'num_class': num_classes, 'eval_metric': 'mlogloss',\n        'tree_method': 'hist', 'device': 'cuda' if torch.cuda.is_available() else 'cpu',\n        'max_depth': trial.suggest_int('max_depth', 3, 8),\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.1),\n        'subsample': trial.suggest_float('subsample', 0.6, 0.9),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 0.9),\n        'reg_alpha': trial.suggest_float('reg_alpha', 1e-3, 10.0, log=True),\n        'reg_lambda': trial.suggest_float('reg_lambda', 1e-3, 10.0, log=True)\n    }\n    \n    bst = xgb.train(params, xgb.DMatrix(X_tr_sm, label=y_tr_sm), num_boost_round=150)\n    preds = bst.predict(xgb.DMatrix(X_va_scaled))\n    return log_loss(y_va, preds)\n\nprint(\"Running Optuna Optimization...\")\nstudy = optuna.create_study(direction='minimize')\nstudy.optimize(objective, n_trials=15)\nbest_params = study.best_params\nbest_params.update({'objective': 'multi:softprob', 'num_class': num_classes, 'tree_method': 'hist', 'eval_metric': 'mlogloss'})\nprint(f\"Best Log-Loss: {study.best_value:.4f}\")\nprint(f\"Best Params: {best_params}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Strict Stratified K-Fold Cross-Validation\n\nThe model undergoes 5-Fold Cross-Validation. **Zero Data Leakage guarantee**: `StandardScaler` and `SMOTE` are re-initialized and fitted *exclusively* on the training subset of each fold.","metadata":{}},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\noof_preds = np.zeros((len(X_hybrid_raw), num_classes))\n\nprint(\"\\nStarting K-Fold Cross Validation...\")\nfor fold, (train_idx, val_idx) in enumerate(skf.split(X_hybrid_raw, y)):\n    X_tr_raw, y_tr = X_hybrid_raw[train_idx], y[train_idx]\n    X_va_raw, y_va = X_hybrid_raw[val_idx], y[val_idx]\n    \n    scaler = StandardScaler()\n    X_tr_scaled = scaler.fit_transform(X_tr_raw)\n    X_va_scaled = scaler.transform(X_va_raw)\n    \n    smote = SMOTE(k_neighbors=3, random_state=42)\n    X_tr_sm, y_tr_sm = smote.fit_resample(X_tr_scaled, y_tr)\n    \n    xgb_model = xgb.XGBClassifier(n_estimators=400, early_stopping_rounds=30, **best_params)\n    xgb_model.fit(X_tr_sm, y_tr_sm, eval_set=[(X_va_scaled, y_va)], verbose=False)\n    \n    oof_preds[val_idx] = xgb_model.predict_proba(X_va_scaled)\n    fold_logloss = log_loss(y_va, oof_preds[val_idx])\n    print(f\"Fold {fold+1} Log-Loss: {fold_logloss:.4f}\")\n\n# --- Evaluate OOF Predictions ---\noof_labels = np.argmax(oof_preds, axis=1)\nprint(\"\\n--- Out-Of-Fold (OOF) CV Results ---\")\nprint(f\"True Generalized Accuracy: {accuracy_score(y, oof_labels) * 100:.2f}%\")\nprint(f\"True Generalized Log-Loss: {log_loss(y, oof_preds):.4f}\")\n\n# --- Train Final Production Model ---\nprint(\"\\nTraining Final Production Model on 100% Data...\")\nfinal_scaler = StandardScaler()\nX_hybrid_scaled_final = final_scaler.fit_transform(X_hybrid_raw)\njoblib.dump(final_scaler, 'tabular_scaler_final.pkl')\n\nsmote_final = SMOTE(k_neighbors=3, random_state=42)\nX_hybrid_smote_final, y_smote_final = smote_final.fit_resample(X_hybrid_scaled_final, y)\n\nfinal_model = xgb.XGBClassifier(n_estimators=400, **best_params)\nfinal_model.fit(X_hybrid_smote_final, y_smote_final, verbose=False)\nfinal_model.save_model('xgboost_model_final.json')\n\nprint(\"\\n--- Top 15 Feature Importances (Gain) ---\")\nimportance_dict = final_model.get_booster().get_score(importance_type='gain')\nsorted_importance = sorted(importance_dict.items(), key=lambda x: x[1], reverse=True)\n\nfor i, (feat, gain) in enumerate(sorted_importance[:15]):\n    feat_idx = int(feat[1:])\n    if feat_idx < 256:\n        feat_name = f\"CNN_Embedding_{feat_idx}\"\n    else:\n        tabular_idx = feat_idx - 256\n        feat_name = feature_names[tabular_idx]\n    print(f\"{i+1}. {feat_name} - Gain: {gain:.2f}\")\n\nprint(\"Phase 2 Complete. Artifacts Saved.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}