{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":4117,"databundleVersionId":46665}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-13T18:02:30.197466Z","iopub.execute_input":"2026-05-13T18:02:30.198331Z","iopub.status.idle":"2026-05-13T18:02:30.205040Z","shell.execute_reply.started":"2026-05-13T18:02:30.198300Z","shell.execute_reply":"2026-05-13T18:02:30.204167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!apt-get install -y p7zip-full","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T18:02:34.382222Z","iopub.execute_input":"2026-05-13T18:02:34.382893Z","iopub.status.idle":"2026-05-13T18:02:37.301614Z","shell.execute_reply.started":"2026-05-13T18:02:34.382859Z","shell.execute_reply":"2026-05-13T18:02:37.300892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport torch\nimport random\nimport numpy as np\nimport pandas as pd\n\nfrom PIL import Image\nfrom tqdm import tqdm\n\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom torchvision import transforms\nfrom torchvision.utils import save_image\nfrom torch.utils.data import Dataset, DataLoader","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T18:02:42.838321Z","iopub.execute_input":"2026-05-13T18:02:42.839167Z","iopub.status.idle":"2026-05-13T18:02:42.843346Z","shell.execute_reply.started":"2026-05-13T18:02:42.839135Z","shell.execute_reply":"2026-05-13T18:02:42.842627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!7z x \"/kaggle/input/competitions/malware-classification/train.7z\" -o\"/kaggle/working/big2015\" \"*.bytes\" -r -y","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!rm -rf /kaggle/working/train\n!rm -rf /kaggle/working/sample_data\n!rm -rf /kaggle/working/big2015\n!rm -rf /kaggle/working/malware_images","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-16T12:30:59.776140Z","iopub.execute_input":"2026-05-16T12:30:59.776473Z","iopub.status.idle":"2026-05-16T12:31:00.267402Z","shell.execute_reply.started":"2026-05-16T12:30:59.776444Z","shell.execute_reply":"2026-05-16T12:31:00.266209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nlabels_df = pd.read_csv(\n    \"/kaggle/input/competitions/malware-classification/trainLabels.csv\"\n)\n\nlabels_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-16T12:30:13.465292Z","iopub.execute_input":"2026-05-16T12:30:13.465650Z","iopub.status.idle":"2026-05-16T12:30:13.488952Z","shell.execute_reply.started":"2026-05-16T12:30:13.465614Z","shell.execute_reply":"2026-05-16T12:30:13.488011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"subset = pd.concat([\n\n    labels_df[labels_df['Class'] == 1].sample(400),\n    labels_df[labels_df['Class'] == 2].sample(400),\n    labels_df[labels_df['Class'] == 3].sample(400),\n    labels_df[labels_df['Class'] == 4].sample(300),\n    labels_df[labels_df['Class'] == 5],   # Simda (minority)\n    labels_df[labels_df['Class'] == 6],   # Tracur\n    labels_df[labels_df['Class'] == 7].sample(300),\n    labels_df[labels_df['Class'] == 8],   # Obfuscator\n    labels_df[labels_df['Class'] == 9].sample(300)\n\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-16T12:30:19.182056Z","iopub.execute_input":"2026-05-16T12:30:19.182370Z","iopub.status.idle":"2026-05-16T12:30:19.199107Z","shell.execute_reply.started":"2026-05-16T12:30:19.182342Z","shell.execute_reply":"2026-05-16T12:30:19.198098Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ids = subset['Id'].tolist()\n\ncmd = '!7z e \"/kaggle/input/competitions/malware-classification/train.7z\" '\n\ncmd += '-o\"/kaggle/working/sample_bytes\" '\n\nfor file_id in ids:\n\n    cmd += f'train/{file_id}.bytes '\n\ncmd += '-y'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-16T12:41:11.678051Z","iopub.execute_input":"2026-05-16T12:41:11.678986Z","iopub.status.idle":"2026-05-16T12:41:11.697185Z","shell.execute_reply.started":"2026-05-16T12:41:11.678944Z","shell.execute_reply":"2026-05-16T12:41:11.695888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"eval(cmd)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-16T12:41:18.608456Z","iopub.execute_input":"2026-05-16T12:41:18.608864Z","iopub.status.idle":"2026-05-16T12:41:18.619520Z","shell.execute_reply.started":"2026-05-16T12:41:18.608801Z","shell.execute_reply":"2026-05-16T12:41:18.616843Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_map = {\n    1: \"Ramnit\",\n    2: \"Lollipop\",\n    3: \"Kelihos_ver3\",\n    4: \"Vundo\",\n    5: \"Simda\",\n    6: \"Tracur\",\n    7: \"Kelihos_ver1\",\n    8: \"Obfuscator_ACY\",\n    9: \"Gatak\"\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T18:04:13.094172Z","iopub.execute_input":"2026-05-13T18:04:13.094827Z","iopub.status.idle":"2026-05-13T18:04:13.099050Z","shell.execute_reply.started":"2026-05-13T18:04:13.094796Z","shell.execute_reply":"2026-05-13T18:04:13.098190Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport cv2\nimport os\n\ndef bytes_to_image(byte_file, width=256):\n\n    hex_values = []\n\n    with open(byte_file, 'r', errors='ignore') as f:\n\n        for line in f:\n\n            parts = line.strip().split()\n\n            if len(parts) <= 1:\n                continue\n\n            bytes_data = parts[1:]\n\n            for byte in bytes_data:\n\n                if byte == '??':\n                    hex_values.append(0)\n\n                else:\n                    hex_values.append(int(byte, 16))\n\n    arr = np.array(hex_values, dtype=np.uint8)\n\n    height = len(arr) // width\n\n    arr = arr[:height * width]\n\n    image = arr.reshape((height, width))\n\n    return image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T18:04:18.096358Z","iopub.execute_input":"2026-05-13T18:04:18.097110Z","iopub.status.idle":"2026-05-13T18:04:18.103148Z","shell.execute_reply.started":"2026-05-13T18:04:18.097077Z","shell.execute_reply":"2026-05-13T18:04:18.102383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"output_dir = \"/kaggle/working/malware_images\"\n\nos.makedirs(output_dir, exist_ok=True)\n\nfor name in class_map.values():\n    os.makedirs(os.path.join(output_dir, name), exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T18:04:31.701541Z","iopub.execute_input":"2026-05-13T18:04:31.702243Z","iopub.status.idle":"2026-05-13T18:04:31.707225Z","shell.execute_reply.started":"2026-05-13T18:04:31.702205Z","shell.execute_reply":"2026-05-13T18:04:31.706170Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm import tqdm\n\ntrain_path = \"/kaggle/working/train\"\n\nfor idx, row in tqdm(labels_df.iterrows(), total=len(labels_df)):\n\n    file_id = row['Id']\n    malware_class = class_map[row['Class']]\n\n    byte_file = os.path.join(\n        train_path,\n        file_id + \".bytes\"\n    )\n\n    if not os.path.exists(byte_file):\n        continue\n\n    try:\n\n        image = bytes_to_image(byte_file)\n\n        save_path = os.path.join(\n            output_dir,\n            malware_class,\n            file_id + \".png\"\n        )\n\n        cv2.imwrite(save_path, image)\n\n    except Exception as e:\n        print(\"Error:\", file_id, e)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T18:04:38.126127Z","iopub.execute_input":"2026-05-13T18:04:38.126424Z","iopub.status.idle":"2026-05-13T18:04:38.704295Z","shell.execute_reply.started":"2026-05-13T18:04:38.126390Z","shell.execute_reply":"2026-05-13T18:04:38.703462Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from glob import glob\n\nbyte_files = glob(\n    \"/kaggle/working/**/*.bytes\",\n    recursive=True\n)\n\nprint(\"Found:\", len(byte_files))\n\nprint(byte_files[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T18:22:30.465222Z","iopub.execute_input":"2026-05-13T18:22:30.466063Z","iopub.status.idle":"2026-05-13T18:22:30.474659Z","shell.execute_reply.started":"2026-05-13T18:22:30.466028Z","shell.execute_reply":"2026-05-13T18:22:30.473899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfrom glob import glob\nimport matplotlib.pyplot as plt\nimport cv2\n\nsamples = glob(\n    \"/kaggle/working/malware_images/Ramnit/*.png\"\n)\n\nprint(\"Found images:\", len(samples))\n\nif len(samples) > 0:\n\n    img = cv2.imread(samples[0], 0)\n\n    plt.figure(figsize=(8,8))\n    plt.imshow(img, cmap='gray')\n    plt.title(\"Malware Image\")\n    plt.axis(\"off\")\n    plt.show()\n\nelse:\n\n    print(\"No images found.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-13T18:09:23.766162Z","iopub.execute_input":"2026-05-13T18:09:23.766649Z","iopub.status.idle":"2026-05-13T18:09:23.773091Z","shell.execute_reply.started":"2026-05-13T18:09:23.766616Z","shell.execute_reply":"2026-05-13T18:09:23.772139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import Counter\n\ncounts = Counter(labels_df['Class'])\n\nfor k, v in counts.items():\n    print(class_map[k], \":\", v)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ImageFolder(\n    \"/kaggle/working/malware_images\"\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Self-Attention Block** ","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\n\nclass SelfAttention(nn.Module):\n\n    def __init__(self, in_dim):\n\n        super(SelfAttention, self).__init__()\n\n        self.query_conv = nn.Conv2d(in_dim, in_dim // 8, 1)\n        self.key_conv   = nn.Conv2d(in_dim, in_dim // 8, 1)\n        self.value_conv = nn.Conv2d(in_dim, in_dim, 1)\n\n        self.gamma = nn.Parameter(torch.zeros(1))\n\n        self.softmax = nn.Softmax(dim=-1)\n\n    def forward(self, x):\n\n        batch, C, width, height = x.size()\n\n        query = self.query_conv(x).view(batch, -1, width*height)\n        key   = self.key_conv(x).view(batch, -1, width*height)\n\n        attention = torch.bmm(\n            query.permute(0,2,1),\n            key\n        )\n\n        attention = self.softmax(attention)\n\n        value = self.value_conv(x).view(batch, -1, width*height)\n\n        out = torch.bmm(\n            value,\n            attention.permute(0,2,1)\n        )\n\n        out = out.view(batch, C, width, height)\n\n        out = self.gamma * out + x\n\n        return out","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Generator with Attention**","metadata":{}},{"cell_type":"code","source":"class Generator(nn.Module):\n\n    def __init__(self, latent_dim=256):\n\n        super().__init__()\n\n        self.fc = nn.Linear(latent_dim, 256*16*16)\n\n        self.main = nn.Sequential(\n\n            nn.ConvTranspose2d(256,128,4,2,1),\n            nn.BatchNorm2d(128),\n            nn.ReLU(True),\n\n            nn.ConvTranspose2d(128,64,4,2,1),\n            nn.BatchNorm2d(64),\n            nn.ReLU(True),\n        )\n\n        # ATTENTION MODULE\n        self.attention = SelfAttention(64)\n\n        self.final = nn.Sequential(\n\n            nn.ConvTranspose2d(64,1,4,2,1),\n            nn.Tanh()\n        )\n\n    def forward(self, z):\n\n        x = self.fc(z)\n\n        x = x.view(-1,256,16,16)\n\n        x = self.main(x)\n\n        # APPLY ATTENTION\n        x = self.attention(x)\n\n        x = self.final(x)\n\n        return x","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **DISCRIMINATOR**","metadata":{}},{"cell_type":"code","source":"class Discriminator(nn.Module):\n\n    def __init__(self):\n\n        super().__init__()\n\n        self.model = nn.Sequential(\n\n            nn.Conv2d(1,64,4,2,1),\n            nn.LeakyReLU(0.2),\n\n            nn.Conv2d(64,128,4,2,1),\n            nn.BatchNorm2d(128),\n            nn.LeakyReLU(0.2),\n\n            nn.Flatten(),\n\n            nn.Linear(128*64*64,1),\n            nn.Sigmoid()\n        )\n\n    def forward(self,x):\n\n        return self.model(x)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **DEEPSMOTE IMPLEMENTATION**","metadata":{}},{"cell_type":"code","source":"class Encoder(nn.Module):\n\n    def __init__(self):\n\n        super().__init__()\n\n        self.encoder = nn.Sequential(\n\n            nn.Conv2d(1,32,4,2,1),\n            nn.ReLU(),\n\n            nn.Conv2d(32,64,4,2,1),\n            nn.ReLU(),\n\n            nn.Flatten(),\n\n            nn.Linear(64*64*64,256)\n        )\n\n    def forward(self,x):\n\n        return self.encoder(x)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **DeepSMOTE Function**","metadata":{}},{"cell_type":"code","source":"import random\n\ndef deep_smote(latent_vectors):\n\n    idx1 = random.randint(0, len(latent_vectors)-1)\n    idx2 = random.randint(0, len(latent_vectors)-1)\n\n    z1 = latent_vectors[idx1]\n    z2 = latent_vectors[idx2]\n\n    alpha = random.random()\n\n    synthetic = alpha*z1 + (1-alpha)*z2\n\n    return synthetic","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"latent = encoder(real_images)\n\nsynthetic_latent = deep_smote(latent)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **GAN TRAINING LOOP**","metadata":{}},{"cell_type":"code","source":"for epoch in range(EPOCHS):\n\n    for real_images, labels in loader:\n\n        # ======================\n        # ENCODER\n        # ======================\n\n        latent = encoder(real_images)\n\n        # ======================\n        # DEEPSMOTE\n        # ======================\n\n        synthetic_latent = deep_smote(latent)\n\n        # ======================\n        # GENERATOR\n        # ======================\n\n        fake_images = generator(synthetic_latent)\n\n        # ======================\n        # TRAIN DISCRIMINATOR\n        # ======================\n\n        real_preds = discriminator(real_images)\n\n        fake_preds = discriminator(fake_images.detach())\n\n        d_loss = (\n            criterion(real_preds, torch.ones_like(real_preds))\n            +\n            criterion(fake_preds, torch.zeros_like(fake_preds))\n        )\n\n        d_optimizer.zero_grad()\n\n        d_loss.backward()\n\n        d_optimizer.step()\n\n        # ======================\n        # TRAIN GENERATOR\n        # ======================\n\n        fake_preds = discriminator(fake_images)\n\n        g_loss = criterion(\n            fake_preds,\n            torch.ones_like(fake_preds)\n        )\n\n        g_optimizer.zero_grad()\n\n        g_loss.backward()\n\n        g_optimizer.step()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **GENERATE SYNTHETIC IMAGES**","metadata":{}},{"cell_type":"code","source":"import torch\nimport matplotlib.pyplot as plt\nimport torchvision.utils as vutils\nimport os\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\ngenerator.eval()\n\n# folder to save generated images\nos.makedirs(\"generated_images\", exist_ok=True)\n\nnum_generate = 1000\n\nwith torch.no_grad():\n\n    # random latent vectors\n    noise = torch.randn(num_generate, latent_dim).to(device)\n\n    # generate fake images\n    fake_images = generator(noise)\n\n    # convert [-1,1] -> [0,1]\n    fake_images = (fake_images + 1) / 2\n\n    # save images\n    for i in range(num_generate):\n\n        img = fake_images[i]\n\n        plt.imsave(\n            f\"generated_images/fake_{i}.png\",\n            img.squeeze().cpu().numpy(),\n            cmap='gray'\n        )\n\nprint(\"Synthetic malware images generated successfully.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **VISUALIZE GENERATED IMAGES**","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(10,10))\n\ngrid = vutils.make_grid(\n    fake_images[:64],\n    padding=2,\n    normalize=True\n)\n\nplt.imshow(grid.permute(1,2,0).cpu())\nplt.title(\"Generated Malware Images\")\nplt.axis(\"off\")\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **REAL VS GENERATED COMPARISON**","metadata":{}},{"cell_type":"code","source":"real_batch = next(iter(dataloader))[0][:16]\n\nplt.figure(figsize=(12,6))\n\n# REAL\nplt.subplot(1,2,1)\n\nreal_grid = vutils.make_grid(\n    real_batch,\n    normalize=True\n)\n\nplt.imshow(real_grid.permute(1,2,0).cpu())\nplt.title(\"Real Malware Images\")\nplt.axis(\"off\")\n\n# FAKE\nplt.subplot(1,2,2)\n\nfake_grid = vutils.make_grid(\n    fake_images[:16],\n    normalize=True\n)\n\nplt.imshow(fake_grid.permute(1,2,0).cpu())\nplt.title(\"Generated Malware Images\")\nplt.axis(\"off\")\n\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **BALANCE THE DATASET**","metadata":{}},{"cell_type":"code","source":"balanced_images = []\nbalanced_labels = []\n\n# original data\nfor imgs, labels in dataloader:\n\n    balanced_images.append(imgs)\n    balanced_labels.append(labels)\n\n# synthetic minority class samples\nsynthetic_labels = torch.full(\n    (num_generate,),\n    minority_class_id\n)\n\nbalanced_images.append(fake_images.cpu())\nbalanced_labels.append(synthetic_labels)\n\nX_balanced = torch.cat(balanced_images)\ny_balanced = torch.cat(balanced_labels)\n\nprint(X_balanced.shape)\nprint(y_balanced.shape)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **SIMPLE CNN CLASSIFIER**","metadata":{}},{"cell_type":"code","source":"import torch.nn as nn\n\nclass MalwareCNN(nn.Module):\n\n    def __init__(self, num_classes):\n        super().__init__()\n\n        self.features = nn.Sequential(\n\n            nn.Conv2d(1, 32, 3, padding=1),\n            nn.ReLU(),\n            nn.MaxPool2d(2),\n\n            nn.Conv2d(32, 64, 3, padding=1),\n            nn.ReLU(),\n            nn.MaxPool2d(2),\n\n            nn.Conv2d(64, 128, 3, padding=1),\n            nn.ReLU(),\n            nn.MaxPool2d(2)\n        )\n\n        self.classifier = nn.Sequential(\n\n            nn.Flatten(),\n            nn.Linear(128*8*8, 256),\n            nn.ReLU(),\n            nn.Linear(256, num_classes)\n        )\n\n    def forward(self, x):\n\n        x = self.features(x)\n        x = self.classifier(x)\n\n        return x","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **EVALUATE RESULTS**","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import (\n    accuracy_score,\n    precision_score,\n    recall_score,\n    f1_score\n)\n\naccuracy = accuracy_score(y_true, y_pred)\nprecision = precision_score(y_true, y_pred, average='macro')\nrecall = recall_score(y_true, y_pred, average='macro')\nf1 = f1_score(y_true, y_pred, average='macro')\n\nprint(\"Accuracy:\", accuracy)\nprint(\"Precision:\", precision)\nprint(\"Recall:\", recall)\nprint(\"F1:\", f1)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}