{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":4117,"databundleVersionId":46665,"isSourceIdPinned":false}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-16T10:57:30.165694Z","iopub.execute_input":"2026-05-16T10:57:30.167140Z","iopub.status.idle":"2026-05-16T10:57:31.162158Z","shell.execute_reply.started":"2026-05-16T10:57:30.166441Z","shell.execute_reply":"2026-05-16T10:57:31.161379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================================================\n# DEEPSMOTE FOR MICROSOFT MALWARE BIG2015\n# =========================================================\n\nimport os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.neighbors import NearestNeighbors\nfrom sklearn.utils import shuffle\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\n\nfrom torchvision import transforms\nimport matplotlib.pyplot as plt\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# =========================================================\n# CONFIG\n# =========================================================\n\nIMG_SIZE = 64\nLATENT_DIM = 128\nBATCH_SIZE = 32\nEPOCHS_AE = 20\nEPOCHS_CLS = 15\n\nDATASET_PATH = \"/kaggle/input/competitions/malware-classification\"\n\n# =========================================================\n# LOAD LABELS\n# =========================================================\n\nlabels_df = pd.read_csv(\n    os.path.join(DATASET_PATH, \"trainLabels.csv\")\n)\n\nprint(labels_df.head())\n\n# =========================================================\n# BYTE FILE TO IMAGE\n# =========================================================\n\ndef byteplot_to_image(byte_file, size=64):\n\n    with open(byte_file, 'r') as f:\n        content = f.readlines()\n\n    hex_values = []\n\n    for line in content:\n        parts = line.strip().split()[1:]\n\n        for p in parts:\n            if p == \"??\":\n                hex_values.append(0)\n            else:\n                hex_values.append(int(p, 16))\n\n    arr = np.array(hex_values, dtype=np.uint8)\n\n    width = 256\n    height = int(np.ceil(len(arr) / width))\n\n    padded = np.pad(\n        arr,\n        (0, width * height - len(arr)),\n        mode='constant'\n    )\n\n    image = padded.reshape(height, width)\n\n    image = cv2.resize(image, (size, size))\n\n    return image\n\n# =========================================================\n# CREATE DATASET\n# =========================================================\n\nimages = []\nlabels = []\n\ntrain_folder = os.path.join(DATASET_PATH, \"train\")\n\nfor idx, row in tqdm(labels_df.iterrows(), total=len(labels_df)):\n\n    file_id = row['Id']\n    label = row['Class']\n\n    byte_path = os.path.join(train_folder, file_id + \".bytes\")\n\n    if os.path.exists(byte_path):\n\n        try:\n            img = byteplot_to_image(byte_path)\n\n            images.append(img)\n            labels.append(label - 1)\n\n        except:\n            pass\n\nimages = np.array(images)\nlabels = np.array(labels)\n\nprint(images.shape)\nprint(labels.shape)\n\n# =========================================================\n# NORMALIZE\n# =========================================================\n\nimages = images.astype(np.float32) / 255.0\nimages = np.expand_dims(images, axis=1)\n\n# =========================================================\n# TRAIN TEST SPLIT\n# =========================================================\n\nX_train, X_test, y_train, y_test = train_test_split(\n    images,\n    labels,\n    test_size=0.2,\n    stratify=labels,\n    random_state=42\n)\n\n# =========================================================\n# DATASET CLASS\n# =========================================================\n\nclass MalwareDataset(Dataset):\n\n    def __init__(self, X, y):\n        self.X = torch.tensor(X, dtype=torch.float32)\n        self.y = torch.tensor(y, dtype=torch.long)\n\n    def __len__(self):\n        return len(self.X)\n\n    def __getitem__(self, idx):\n        return self.X[idx], self.y[idx]\n\ntrain_dataset = MalwareDataset(X_train, y_train)\ntest_dataset = MalwareDataset(X_test, y_test)\n\ntrain_loader = DataLoader(\n    train_dataset,\n    batch_size=BATCH_SIZE,\n    shuffle=True\n)\n\ntest_loader = DataLoader(\n    test_dataset,\n    batch_size=BATCH_SIZE\n)\n\n# =========================================================\n# AUTOENCODER (DeepSMOTE backbone)\n# =========================================================\n\nclass Encoder(nn.Module):\n\n    def __init__(self):\n        super().__init__()\n\n        self.encoder = nn.Sequential(\n\n            nn.Conv2d(1, 32, 4, 2, 1),\n            nn.ReLU(),\n\n            nn.Conv2d(32, 64, 4, 2, 1),\n            nn.BatchNorm2d(64),\n            nn.ReLU(),\n\n            nn.Conv2d(64, 128, 4, 2, 1),\n            nn.BatchNorm2d(128),\n            nn.ReLU(),\n\n            nn.Flatten(),\n\n            nn.Linear(128 * 8 * 8, LATENT_DIM)\n        )\n\n    def forward(self, x):\n        return self.encoder(x)\n\nclass Decoder(nn.Module):\n\n    def __init__(self):\n        super().__init__()\n\n        self.fc = nn.Linear(LATENT_DIM, 128 * 8 * 8)\n\n        self.decoder = nn.Sequential(\n\n            nn.ConvTranspose2d(128, 64, 4, 2, 1),\n            nn.BatchNorm2d(64),\n            nn.ReLU(),\n\n            nn.ConvTranspose2d(64, 32, 4, 2, 1),\n            nn.BatchNorm2d(32),\n            nn.ReLU(),\n\n            nn.ConvTranspose2d(32, 1, 4, 2, 1),\n            nn.Sigmoid()\n        )\n\n    def forward(self, z):\n\n        x = self.fc(z)\n\n        x = x.view(-1, 128, 8, 8)\n\n        return self.decoder(x)\n\nencoder = Encoder().to(device)\ndecoder = Decoder().to(device)\n\n# =========================================================\n# TRAIN AUTOENCODER\n# =========================================================\n\ncriterion = nn.MSELoss()\n\noptimizer = optim.Adam(\n    list(encoder.parameters()) +\n    list(decoder.parameters()),\n    lr=1e-3\n)\n\nfor epoch in range(EPOCHS_AE):\n\n    encoder.train()\n    decoder.train()\n\n    total_loss = 0\n\n    for imgs, _ in train_loader:\n\n        imgs = imgs.to(device)\n\n        optimizer.zero_grad()\n\n        latent = encoder(imgs)\n\n        reconstructed = decoder(latent)\n\n        # Reconstruction loss\n        recon_loss = criterion(reconstructed, imgs)\n\n        # Penalty loss (DeepSMOTE paper idea)\n        perm = torch.randperm(latent.size(0))\n\n        permuted = latent[perm]\n\n        decoded_perm = decoder(permuted)\n\n        penalty_loss = criterion(decoded_perm, imgs)\n\n        loss = recon_loss + 0.5 * penalty_loss\n\n        loss.backward()\n\n        optimizer.step()\n\n        total_loss += loss.item()\n\n    print(f\"AE Epoch {epoch+1}: {total_loss:.4f}\")\n\n# =========================================================\n# ENCODE TRAIN DATA\n# =========================================================\n\nencoder.eval()\n\nlatent_vectors = []\nlatent_labels = []\n\nwith torch.no_grad():\n\n    for imgs, lbls in train_loader:\n\n        imgs = imgs.to(device)\n\n        z = encoder(imgs)\n\n        latent_vectors.append(z.cpu().numpy())\n        latent_labels.append(lbls.numpy())\n\nlatent_vectors = np.concatenate(latent_vectors)\nlatent_labels = np.concatenate(latent_labels)\n\n# =========================================================\n# DEEPSMOTE LATENT SPACE OVERSAMPLING\n# =========================================================\n\ndef deepsmote_generate(features, labels):\n\n    unique_classes = np.unique(labels)\n\n    max_count = max([\n        np.sum(labels == c)\n        for c in unique_classes\n    ])\n\n    synthetic_features = []\n    synthetic_labels = []\n\n    for cls in unique_classes:\n\n        cls_idx = np.where(labels == cls)[0]\n\n        cls_features = features[cls_idx]\n\n        current_count = len(cls_features)\n\n        needed = max_count - current_count\n\n        if needed <= 0:\n            continue\n\n        nbrs = NearestNeighbors(n_neighbors=5)\n        nbrs.fit(cls_features)\n\n        for _ in range(needed):\n\n            idx = np.random.randint(0, current_count)\n\n            x_i = cls_features[idx]\n\n            nn_array = nbrs.kneighbors(\n                [x_i],\n                return_distance=False\n            )[0]\n\n            nn_idx = np.random.choice(nn_array[1:])\n\n            x_nn = cls_features[nn_idx]\n\n            lam = np.random.rand()\n\n            synthetic = x_i + lam * (x_nn - x_i)\n\n            synthetic_features.append(synthetic)\n            synthetic_labels.append(cls)\n\n    return (\n        np.array(synthetic_features),\n        np.array(synthetic_labels)\n    )\n\nsyn_features, syn_labels = deepsmote_generate(\n    latent_vectors,\n    latent_labels\n)\n\nprint(syn_features.shape)\n\n# =========================================================\n# GENERATE SYNTHETIC IMAGES\n# =========================================================\n\ndecoder.eval()\n\nwith torch.no_grad():\n\n    syn_tensor = torch.tensor(\n        syn_features,\n        dtype=torch.float32\n    ).to(device)\n\n    generated_images = decoder(syn_tensor)\n\ngenerated_images = generated_images.cpu().numpy()\n\n# =========================================================\n# VISUALIZE GENERATED MALWARE IMAGES\n# =========================================================\n\nfig, axes = plt.subplots(2, 5, figsize=(12, 5))\n\nfor i, ax in enumerate(axes.flat):\n\n    ax.imshow(\n        generated_images[i][0],\n        cmap='gray'\n    )\n\n    ax.axis('off')\n\nplt.show()\n\n# =========================================================\n# BALANCED DATASET\n# =========================================================\n\nX_balanced = np.concatenate([\n    X_train,\n    generated_images\n])\n\ny_balanced = np.concatenate([\n    y_train,\n    syn_labels\n])\n\nX_balanced, y_balanced = shuffle(\n    X_balanced,\n    y_balanced,\n    random_state=42\n)\n\n# =========================================================\n# CLASSIFIER\n# =========================================================\n\nclass MalwareCNN(nn.Module):\n\n    def __init__(self):\n\n        super().__init__()\n\n        self.features = nn.Sequential(\n\n            nn.Conv2d(1, 32, 3, padding=1),\n            nn.ReLU(),\n            nn.MaxPool2d(2),\n\n            nn.Conv2d(32, 64, 3, padding=1),\n            nn.ReLU(),\n            nn.MaxPool2d(2),\n\n            nn.Conv2d(64, 128, 3, padding=1),\n            nn.ReLU(),\n            nn.AdaptiveAvgPool2d(1)\n        )\n\n        self.classifier = nn.Linear(128, 9)\n\n    def forward(self, x):\n\n        x = self.features(x)\n\n        x = x.view(x.size(0), -1)\n\n        return self.classifier(x)\n\nmodel = MalwareCNN().to(device)\n\n# =========================================================\n# TRAIN CLASSIFIER\n# =========================================================\n\nbalanced_dataset = MalwareDataset(\n    X_balanced,\n    y_balanced\n)\n\nbalanced_loader = DataLoader(\n    balanced_dataset,\n    batch_size=BATCH_SIZE,\n    shuffle=True\n)\n\ncriterion_cls = nn.CrossEntropyLoss()\n\noptimizer_cls = optim.Adam(\n    model.parameters(),\n    lr=1e-3\n)\n\nfor epoch in range(EPOCHS_CLS):\n\n    model.train()\n\n    running_loss = 0\n\n    for imgs, lbls in balanced_loader:\n\n        imgs = imgs.to(device)\n        lbls = lbls.to(device)\n\n        optimizer_cls.zero_grad()\n\n        outputs = model(imgs)\n\n        loss = criterion_cls(outputs, lbls)\n\n        loss.backward()\n\n        optimizer_cls.step()\n\n        running_loss += loss.item()\n\n    print(\n        f\"Classifier Epoch {epoch+1}: \"\n        f\"{running_loss:.4f}\"\n    )\n\n# =========================================================\n# TEST\n# =========================================================\n\nmodel.eval()\n\ncorrect = 0\ntotal = 0\n\nwith torch.no_grad():\n\n    for imgs, lbls in test_loader:\n\n        imgs = imgs.to(device)\n        lbls = lbls.to(device)\n\n        outputs = model(imgs)\n\n        _, preds = torch.max(outputs, 1)\n\n        total += lbls.size(0)\n\n        correct += (preds == lbls).sum().item()\n\naccuracy = 100 * correct / total\n\nprint(f\"Test Accuracy: {accuracy:.2f}%\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}