{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":4117,"databundleVersionId":46665,"sourceType":"competition"},{"sourceId":11621845,"sourceType":"datasetVersion","datasetId":7290742}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:00:05.098453Z","iopub.execute_input":"2025-04-30T10:00:05.098625Z","iopub.status.idle":"2025-04-30T10:00:07.076344Z","shell.execute_reply.started":"2025-04-30T10:00:05.0986Z","shell.execute_reply":"2025-04-30T10:00:07.075758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install torch_geometric","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:00:12.093784Z","iopub.execute_input":"2025-04-30T10:00:12.094064Z","iopub.status.idle":"2025-04-30T10:00:17.286254Z","shell.execute_reply.started":"2025-04-30T10:00:12.094042Z","shell.execute_reply":"2025-04-30T10:00:17.285556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset\nfrom torch_geometric.data import Data, DataLoader\nfrom torch_geometric.nn import GCNConv, global_mean_pool\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics.pairwise import cosine_similarity\nimport random","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:00:21.588512Z","iopub.execute_input":"2025-04-30T10:00:21.588796Z","iopub.status.idle":"2025-04-30T10:00:34.655009Z","shell.execute_reply.started":"2025-04-30T10:00:21.588762Z","shell.execute_reply":"2025-04-30T10:00:34.654392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_data(path):\n    df = pd.read_csv(path)\n    X = df.drop(['Category', 'Class'], axis=1).values.astype(np.float32)\n    y = LabelEncoder().fit_transform(df['Class'])\n\n    X_temp, X_test, y_temp, y_test = train_test_split(\n        X, y, test_size=0.2, stratify=y, random_state=42)\n    X_train, X_val, y_train, y_val = train_test_split(\n        X_temp, y_temp, test_size=0.25, stratify=y_temp, random_state=42)\n\n    scaler = StandardScaler()\n    X_train = scaler.fit_transform(X_train)\n    X_val = scaler.transform(X_val)\n    X_test = scaler.transform(X_test)\n\n    return (X_train, y_train), (X_val, y_val), (X_test, y_test)\n\ndef feature_graph(x, k=5):\n    sim = cosine_similarity(x.reshape(-1, 1))\n    np.fill_diagonal(sim, 0)\n    top_k = np.argsort(-sim, axis=1)[:, :k]\n    edge_index = []\n    for i in range(len(x)):\n        for j in top_k[i]:\n            edge_index.append((i, j))\n    edge_index = torch.tensor(edge_index, dtype=torch.long).t().contiguous()\n    return edge_index","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:06:30.129365Z","iopub.execute_input":"2025-04-30T10:06:30.129674Z","iopub.status.idle":"2025-04-30T10:06:30.136295Z","shell.execute_reply.started":"2025-04-30T10:06:30.129654Z","shell.execute_reply":"2025-04-30T10:06:30.135625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class MalwareGraphDataset(Dataset):\n    def __init__(self, X, y, noise_std=0.05):\n        self.X = X\n        self.y = y\n        self.noise_std = noise_std\n\n    def __len__(self):\n        return len(self.X)\n\n    def __getitem__(self, idx):\n        x = self.X[idx]\n        y = self.y[idx]\n        edge_index = feature_graph(x)\n        x_tensor = torch.tensor(x, dtype=torch.float32).unsqueeze(1)\n        x_tensor += torch.randn_like(x_tensor) * self.noise_std\n        return Data(x=x_tensor, edge_index=edge_index, y=torch.tensor(y, dtype=torch.long))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:06:43.090132Z","iopub.execute_input":"2025-04-30T10:06:43.090396Z","iopub.status.idle":"2025-04-30T10:06:43.095356Z","shell.execute_reply.started":"2025-04-30T10:06:43.090377Z","shell.execute_reply":"2025-04-30T10:06:43.094718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class GNNClassifier(nn.Module):\n    def __init__(self, input_dim=1, hidden_dim=64, num_classes=2, dropout=0.3):\n        super().__init__()\n        self.gcn1 = GCNConv(input_dim, hidden_dim)\n        self.gcn2 = GCNConv(hidden_dim, hidden_dim)\n        self.dropout = nn.Dropout(dropout)\n        self.fc = nn.Sequential(\n            nn.Linear(hidden_dim, 64),\n            nn.ReLU(),\n            nn.BatchNorm1d(64),\n            nn.Linear(64, num_classes)\n        )\n\n    def forward(self, data):\n        x, edge_index, batch = data.x, data.edge_index, data.batch\n        x = F.relu(self.gcn1(x, edge_index))\n        x = self.dropout(x)\n        x = F.relu(self.gcn2(x, edge_index))\n        out = global_mean_pool(x, batch)\n        return self.fc(out)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:06:55.174346Z","iopub.execute_input":"2025-04-30T10:06:55.174624Z","iopub.status.idle":"2025-04-30T10:06:55.18002Z","shell.execute_reply.started":"2025-04-30T10:06:55.174604Z","shell.execute_reply":"2025-04-30T10:06:55.179322Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nclass LSTM_CNN_Classifier(nn.Module):\n    def __init__(self, input_dim=1, lstm_hidden=64, num_classes=2, dropout=0.3):\n        super().__init__()\n        self.lstm_hidden = lstm_hidden\n\n        # LSTM for sequence modeling\n        self.lstm = nn.LSTM(input_size=input_dim, hidden_size=lstm_hidden, num_layers=1, batch_first=True)\n\n        # 1D CNN for feature extraction\n        self.conv1 = nn.Conv1d(in_channels=lstm_hidden, out_channels=64, kernel_size=3, padding=1)\n        self.conv2 = nn.Conv1d(in_channels=64, out_channels=128, kernel_size=3, padding=1)\n        self.pool = nn.MaxPool1d(kernel_size=2)\n\n        # Fully connected classifier\n        self.fc = nn.Sequential(\n            nn.Linear(128, 64),\n            nn.ReLU(),\n            nn.Dropout(dropout),\n            nn.Linear(64, num_classes)\n        )\n\n    def forward(self, data):\n        x, batch = data.x, data.batch\n\n        x_lstm = x.view(data.num_graphs, -1, 1)\n\n        lstm_out, _ = self.lstm(x_lstm)\n        x_cnn = lstm_out.permute(0, 2, 1)\n\n        x_cnn = F.relu(self.conv1(x_cnn))  \n        x_cnn = self.pool(x_cnn)         \n        x_cnn = F.relu(self.conv2(x_cnn))  \n        x_cnn = self.pool(x_cnn) \n\n        x_cnn = x_cnn.mean(dim=2)       \n\n        out = self.fc(x_cnn) \n        return out","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:54:32.402694Z","iopub.execute_input":"2025-04-30T10:54:32.403013Z","iopub.status.idle":"2025-04-30T10:54:32.410483Z","shell.execute_reply.started":"2025-04-30T10:54:32.402991Z","shell.execute_reply":"2025-04-30T10:54:32.409687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_epoch(model, loader, optimizer, criterion, device):\n    model.train()\n    total_loss = 0\n    for data in loader:\n        data = data.to(device)\n        optimizer.zero_grad()\n        out = model(data)\n        loss = criterion(out, data.y)\n        loss.backward()\n        nn.utils.clip_grad_norm_(model.parameters(), 1.0)\n        optimizer.step()\n        total_loss += loss.item() * data.num_graphs\n    return total_loss / len(loader.dataset)\n\ndef evaluate(model, loader, criterion, device):\n    model.eval()\n    total_loss, correct = 0, 0\n    with torch.no_grad():\n        for data in loader:\n            data = data.to(device)\n            out = model(data)\n            loss = criterion(out, data.y)\n            total_loss += loss.item() * data.num_graphs\n            pred = out.argmax(dim=1)\n            correct += (pred == data.y).sum().item()\n    return total_loss / len(loader.dataset), correct / len(loader.dataset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:07:25.219085Z","iopub.execute_input":"2025-04-30T10:07:25.219753Z","iopub.status.idle":"2025-04-30T10:07:25.225533Z","shell.execute_reply.started":"2025-04-30T10:07:25.219728Z","shell.execute_reply":"2025-04-30T10:07:25.224847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nfrom sklearn.metrics import confusion_matrix\n\ndef plot_loss(train_losses, val_losses):\n    plt.figure(figsize=(12, 6))\n    plt.plot(range(1, len(train_losses) + 1), train_losses, label='Train Loss', color='blue')\n    plt.plot(range(1, len(val_losses) + 1), val_losses, label='Validation Loss', color='red')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.title('Train vs Validation Loss')\n    plt.legend()\n    plt.grid(True)\n    plt.show()\n\ndef plot_accuracy(val_accuracies):\n    plt.figure(figsize=(12, 6))\n    plt.plot(range(1, len(val_accuracies) + 1), val_accuracies, label='Validation Accuracy', color='green')\n    plt.xlabel('Epochs')\n    plt.ylabel('Accuracy')\n    plt.title('Validation Accuracy per Epoch')\n    plt.legend()\n    plt.grid(True)\n    plt.show()\n\ndef plot_confusion_matrix(y_true, y_pred, class_labels):\n    cm = confusion_matrix(y_true, y_pred)\n    plt.figure(figsize=(8, 6))\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=class_labels, yticklabels=class_labels)\n    plt.xlabel('Predicted')\n    plt.ylabel('True')\n    plt.title('Confusion Matrix')\n    plt.show()\n\ndef plot_contour_map(feature_maps):\n    plt.figure(figsize=(8, 6))\n    sns.kdeplot(x=feature_maps[:, 0], y=feature_maps[:, 1], cmap=\"Blues\", shade=True)\n    plt.title(\"Contour Map of Feature Maps\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:13:30.114693Z","iopub.execute_input":"2025-04-30T10:13:30.115326Z","iopub.status.idle":"2025-04-30T10:13:30.122937Z","shell.execute_reply.started":"2025-04-30T10:13:30.115301Z","shell.execute_reply":"2025-04-30T10:13:30.122217Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch_geometric.data import DataLoader  # Import DataLoader from PyG\n\ndef run(model_class):\n    device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n    (X_train, y_train), (X_val, y_val), (X_test, y_test) = load_data('/kaggle/input/malware/Obfuscated-MalMem2022.csv')\n\n    train_set = MalwareGraphDataset(X_train, y_train)\n    val_set = MalwareGraphDataset(X_val, y_val, noise_std=0.0)\n    test_set = MalwareGraphDataset(X_test, y_test, noise_std=0.0)\n\n    train_loader = DataLoader(train_set, batch_size=64, shuffle=True)\n    val_loader = DataLoader(val_set, batch_size=64)\n    test_loader = DataLoader(test_set, batch_size=64)\n\n    model = model_class().to(device)\n    optimizer = torch.optim.Adam(model.parameters(), lr=0.001)\n    criterion = nn.CrossEntropyLoss()\n\n    best_val_acc = 0\n    patience, wait = 7, 0\n\n    train_losses, val_losses, val_accuracies = [], [], []\n\n    for epoch in range(1, 51):\n        train_loss = train_epoch(model, train_loader, optimizer, criterion, device)\n        val_loss, val_acc = evaluate(model, val_loader, criterion, device)\n\n        train_losses.append(train_loss)\n        val_losses.append(val_loss)\n        val_accuracies.append(val_acc)\n\n        print(f\"Epoch {epoch:03d} | Train Loss: {train_loss:.4f} | Val Loss: {val_loss:.4f} | Val Acc: {val_acc:.4f}\")\n\n        if val_acc > best_val_acc:\n            best_val_acc = val_acc\n            if isinstance(model_class(), GNNClassifier):\n                torch.save(model.state_dict(), 'gnn_model.pth')\n            else:\n                torch.save(model.state_dict(), 'LstmCNN_model.pth')\n            wait = 0\n        else:\n            wait += 1\n            if wait >= patience:\n                print(f\"Early stopping at epoch {epoch}\")\n                break\n\n    print(\"\\nEvaluating best model...\")\n    model.load_state_dict(torch.load('gnn_model.pth' if isinstance(model_class(), GNNClassifier) else 'LstmCNN_model.pth'))\n    test_loss, test_acc = evaluate(model, test_loader, criterion, device)\n    print(f\"Test Loss: {test_loss:.4f} | Test Accuracy: {test_acc:.4f}\")\n\n    plot_loss(train_losses, val_losses)\n    plot_accuracy(val_accuracies)\n\n    model.eval()\n    y_true, y_pred = [], []\n    with torch.no_grad():\n        for data in test_loader:\n            data = data.to(device)\n            out = model(data)\n            pred = out.argmax(dim=1)\n            y_true.extend(data.y.cpu().numpy())\n            y_pred.extend(pred.cpu().numpy())\n\n    plot_confusion_matrix(y_true, y_pred, class_labels=np.unique(y_true))\n\n    if hasattr(model, 'fc') and isinstance(model, LSTM_CNN_Classifier):\n        feature_maps = []\n        with torch.no_grad():\n            for data in test_loader:\n                data = data.to(device)\n                x, _ = model(data)\n                feature_maps.append(x.cpu().numpy())\n\n        feature_maps = np.concatenate(feature_maps, axis=0)\n        plot_contour_map(feature_maps)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:23:40.654208Z","iopub.execute_input":"2025-04-30T10:23:40.654791Z","iopub.status.idle":"2025-04-30T10:23:40.665484Z","shell.execute_reply.started":"2025-04-30T10:23:40.654766Z","shell.execute_reply":"2025-04-30T10:23:40.664948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if __name__ == '__main__':\n    print(\"\\n--- GNN Only Model ---\")\n    run(lambda: GNNClassifier(input_dim=1, hidden_dim=64, num_classes=2))\n\n    print(\"\\n--- LSTM + CNN Model ---\")\n    run(lambda: LSTM_CNN_Classifier(input_dim=1, lstm_hidden=64, num_classes=2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T10:34:17.057622Z","iopub.execute_input":"2025-04-30T10:34:17.05837Z","iopub.status.idle":"2025-04-30T10:49:50.219114Z","shell.execute_reply.started":"2025-04-30T10:34:17.058345Z","shell.execute_reply":"2025-04-30T10:49:50.218146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}