{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":4117,"databundleVersionId":46665,"sourceType":"competition"},{"sourceId":5931326,"sourceType":"datasetVersion","datasetId":3077614},{"sourceId":5963602,"sourceType":"datasetVersion","datasetId":3348292},{"sourceId":8830912,"sourceType":"datasetVersion","datasetId":5313508},{"sourceId":58027,"sourceType":"modelInstanceVersion","modelInstanceId":48649},{"sourceId":58047,"sourceType":"modelInstanceVersion","modelInstanceId":48649}],"dockerImageVersionId":30461,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"model_path = \"/kaggle/working/models/hydra/\"\n!pip install chardet","metadata":{"execution":{"iopub.status.busy":"2024-07-01T08:34:18.788653Z","iopub.execute_input":"2024-07-01T08:34:18.789351Z","iopub.status.idle":"2024-07-01T08:34:32.118015Z","shell.execute_reply.started":"2024-07-01T08:34:18.789310Z","shell.execute_reply":"2024-07-01T08:34:32.116720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_addons as tfa\nimport tensorflow as tf\nimport tensorflow_text as text\nimport json\nimport pandas as pd\nimport numpy as np\nimport math\nimport re\nimport chardet\nimport os\nimport argparse\nimport zipfile\nimport errno\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport lime\nfrom lime import lime_text\n\nfrom sklearn.metrics import confusion_matrix\n\nimport socket\nimport pickle\nimport numpy, struct\nfrom zipfile import ZipFile\n\nfrom sklearn.metrics import accuracy_score\n\nimport argparse\nproject_path = os.path.dirname(os.path.realpath(\"../../../\"))\nimport sys\nimport csv\nsys.path.append(project_path)","metadata":{"execution":{"iopub.status.busy":"2024-07-01T08:37:07.493392Z","iopub.execute_input":"2024-07-01T08:37:07.493805Z","iopub.status.idle":"2024-07-01T08:37:07.502876Z","shell.execute_reply.started":"2024-07-01T08:37:07.493768Z","shell.execute_reply":"2024-07-01T08:37:07.501856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def max_fcount():\n    import os, re\n    checkpoint_dir = \"/kaggle/working/models/hydra/\"\n    # Find all checkpoint files in the directory\n    checkpoint_files = [f for f in os.listdir(checkpoint_dir) if re.match(r'model_\\d+\\.ckpt\\.index', f)]\n    # Extract the numbers from the file names and find the maximum\n    max_number = max([int(re.search(r'model_(\\d+)\\.ckpt\\.index', f).group(1)) for f in checkpoint_files])\n    # Construct the file name with the maximum number\n    max_checkpoint_file = os.path.join('model_{}'.format(max_number))\n    print('Maximum checkpoint file:', max_checkpoint_file)\n    return max_checkpoint_file, max_number","metadata":{"execution":{"iopub.status.busy":"2024-07-01T08:37:15.655217Z","iopub.execute_input":"2024-07-01T08:37:15.656281Z","iopub.status.idle":"2024-07-01T08:37:15.663219Z","shell.execute_reply.started":"2024-07-01T08:37:15.656230Z","shell.execute_reply":"2024-07-01T08:37:15.662210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hydra _ architecture\nclass HYDRA(tf.keras.Model):\n    def __init__(self, parameters):\n        super(HYDRA, self).__init__()\n        self.parameters = parameters\n\n    def build(self, input_shapes):\n        # Bytes component\n        ######################################### Bytes component ######################################################\n        self.bytes_emb = tf.keras.layers.Embedding(self.parameters['bytes']['V'], self.parameters['bytes']['E'],\n                                             input_shape=(None, self.parameters['bytes']['max_bytes_values']))\n\n        self.bytes_conv_1 = tf.keras.layers.Conv2D(filters=self.parameters['bytes']['num_filters'][0],\n                                             kernel_size=[self.parameters['bytes']['kernel_sizes'][0],\n                                                          self.parameters['bytes']['E']],\n                                             strides=(self.parameters['bytes']['strides'][0], 1),\n                                             data_format='channels_last',\n                                             use_bias=True,\n                                             activation=\"relu\")\n\n        self.bytes_conv_2 = tf.keras.layers.Conv2D(filters=self.parameters['bytes']['num_filters'][1],\n                                             kernel_size=[self.parameters['bytes']['kernel_sizes'][1],\n                                                          1],\n                                             strides=(self.parameters['bytes']['strides'][1], 1),\n                                             data_format='channels_last',\n                                             use_bias=True,\n                                             activation=\"relu\")\n\n        self.bytes_max_pool_1 = tf.keras.layers.MaxPooling2D(pool_size=(self.parameters['bytes']['max_pool_size'], 1))\n\n        self.bytes_conv_3 = tf.keras.layers.Conv2D(filters=self.parameters['bytes']['num_filters'][2],\n                                             kernel_size=[self.parameters['bytes']['kernel_sizes'][2],\n                                                          1],\n                                             strides=(self.parameters['bytes']['strides'][2], 1),\n                                             data_format='channels_last',\n                                             use_bias=True,\n                                             activation=\"relu\")\n\n        self.bytes_conv_4 = tf.keras.layers.Conv2D(filters=self.parameters['bytes']['num_filters'][3],\n                                             kernel_size=[self.parameters['bytes']['kernel_sizes'][3],\n                                                          1],\n                                             strides=(self.parameters['bytes']['strides'][3], 1),\n                                             data_format='channels_last',\n                                             use_bias=True,\n                                             activation=\"relu\")\n\n        self.bytes_global_avg_pool = tf.keras.layers.GlobalAvgPool2D()\n\n        self.bytes_drop_1 = tf.keras.layers.Dropout(self.parameters[\"dropout_rate\"])\n        self.bytes_dense_1 = tf.keras.layers.Dense(self.parameters['bytes']['hidden'][0],\n                                             activation=\"selu\")\n\n        self.bytes_drop_2 = tf.keras.layers.Dropout(self.parameters[\"dropout_rate\"])\n        self.bytes_dense_2 = tf.keras.layers.Dense(self.parameters['bytes']['hidden'][1],\n                                             activation=\"selu\")\n\n        self.bytes_drop_3 = tf.keras.layers.Dropout(self.parameters[\"dropout_rate\"])\n\n        self.bytes_dense_3 = tf.keras.layers.Dense(self.parameters['bytes']['hidden'][2],\n                                             activation=\"selu\")\n\n        ####################################### Opcodes component ######################################################\n        self.opcodes_emb = tf.keras.layers.Embedding(self.parameters['opcodes']['V'],\n                                                   self.parameters['opcodes']['E'],\n                                                   input_shape=(None,\n                                                                self.parameters['opcodes']['seq_length']))\n\n        self.opcodes_conv_3 = tf.keras.layers.Conv2D(self.parameters['opcodes']['conv']['num_filters'],\n                                                   (self.parameters['opcodes']['conv']['size'][0],\n                                                    self.parameters['opcodes']['E']),\n                                                   activation=\"relu\",\n                                                   input_shape=(None,\n                                                                self.parameters['opcodes']['seq_length'],\n                                                                self.parameters['opcodes']['E']))\n        self.opcodes_global_max_pooling_3 = tf.keras.layers.GlobalMaxPooling2D()\n\n        self.opcodes_conv_5 = tf.keras.layers.Conv2D(self.parameters['opcodes']['conv']['num_filters'],\n                                                   (self.parameters['opcodes']['conv']['size'][1], self.parameters['opcodes']['E']),\n                                                   activation=\"relu\",\n                                                   input_shape=(None,\n                                                                self.parameters['opcodes']['seq_length'],\n                                                                self.parameters['opcodes']['E']))\n        self.opcodes_global_max_pooling_5 = tf.keras.layers.GlobalMaxPooling2D()\n\n        self.opcodes_conv_7 = tf.keras.layers.Conv2D(self.parameters['opcodes']['conv']['num_filters'],\n                                                   (self.parameters['opcodes']['conv']['size'][2], self.parameters['opcodes']['E']),\n                                                   activation=\"relu\",\n                                                   input_shape=(None,\n                                                                self.parameters['opcodes']['seq_length'],\n                                                                self.parameters['opcodes']['E']))\n        self.opcodes_global_max_pooling_7 = tf.keras.layers.GlobalMaxPooling2D()\n\n        ################################################# APIs Component ###############################################\n        self.apis_input_dropout = tf.keras.layers.Dropout(self.parameters[\"input_dropout_rate\"],\n                                                     input_shape=(None, self.parameters[\"api_features\"]))\n\n        self.apis_hidden_1 = tf.keras.layers.Dense(self.parameters['apis']['hidden'],\n                                        activation=\"relu\",\n                                        input_shape=(None, self.parameters[\"api_features\"]))\n\n        self.bytes_apis_dense_dropout = tf.keras.layers.Dropout(self.parameters[\"dropout_rate\"])\n        self.bytes_apis_dense = tf.keras.layers.Dense(self.parameters['output'], activation=\"selu\")\n\n        self.dense_dropout = tf.keras.layers.Dropout(self.parameters[\"dropout_rate\"])\n        self.dense = tf.keras.layers.Dense(self.parameters['output'], activation=\"selu\")\n\n        self.output_dropout = tf.keras.layers.Dropout(self.parameters[\"dropout_rate\"])\n        self.out = tf.keras.layers.Dense(10,\n                                           activation=\"softmax\")\n\n    def call(self, opcodes_tensor, bytes_tensor, apis_tensor, training=False):\n        # Bytes subcomponent\n        bytes_emb = self.bytes_emb(bytes_tensor)\n        bytes_emb_expanded = tf.keras.backend.expand_dims(bytes_emb, axis=-1)\n\n        bytes_conv_1 = self.bytes_conv_1(bytes_emb_expanded)\n        bytes_conv_2 = self.bytes_conv_2(bytes_conv_1)\n\n        bytes_max_pool_1 = self.bytes_max_pool_1(bytes_conv_2)\n\n        bytes_conv_3 = self.bytes_conv_3(bytes_max_pool_1)\n        bytes_conv_4 = self.bytes_conv_4(bytes_conv_3)\n\n        bytes_features = self.bytes_global_avg_pool(bytes_conv_4)\n\n        bytes_drop_1 = self.bytes_drop_1(bytes_features, training=training)\n        bytes_dense_1 = self.bytes_dense_1(bytes_drop_1)\n\n        bytes_drop_2 = self.bytes_drop_2(bytes_dense_1, training=training)\n        bytes_dense_2 = self.bytes_dense_2(bytes_drop_2)\n\n        bytes_drop_3 = self.bytes_drop_3(bytes_dense_2, training=training)\n        bytes_dense_3 = self.bytes_dense_3(bytes_drop_3)\n\n        # Opcodes subcomponent\n        opcodes_emb = self.opcodes_emb(opcodes_tensor)\n        opcodes_emb_expanded = tf.keras.backend.expand_dims(opcodes_emb, axis=-1)\n\n        opcodes_conv_3 = self.opcodes_conv_3(opcodes_emb_expanded)\n        opcodes_pool_3 = self.opcodes_global_max_pooling_3(opcodes_conv_3)\n\n        opcodes_conv_5 = self.opcodes_conv_5(opcodes_emb_expanded)\n        opcodes_pool_5 = self.opcodes_global_max_pooling_5(opcodes_conv_5)\n\n        opcodes_conv_7 = self.opcodes_conv_7(opcodes_emb_expanded)\n        opcodes_pool_7 = self.opcodes_global_max_pooling_7(opcodes_conv_7)\n\n        #APIs subcomponent\n        apis_input_dropout = self.apis_input_dropout(apis_tensor, training=training)\n        apis_hidden1 = self.apis_hidden_1(apis_input_dropout)\n\n\n        # Features fusion\n        features_api_bytes = tf.keras.layers.concatenate([bytes_dense_3, apis_hidden1])\n        features_api_bytes_dropout = self.bytes_apis_dense_dropout(features_api_bytes, training=training)\n        dense_api_bytes = self.bytes_apis_dense(features_api_bytes_dropout)\n\n        features = tf.keras.layers.concatenate([opcodes_pool_3, opcodes_pool_5, opcodes_pool_7, dense_api_bytes])\n        features_dropout = self.dense_dropout(features, training=training)\n        dense_opcodes_apis_bytes = self.dense(features_dropout)\n\n        features_dropout = self.dense_dropout(dense_opcodes_apis_bytes, training=training)\n        output = self.out(features_dropout)\n\n        return output","metadata":{"_uuid":"4dddbf31-2110-4a96-b46f-78c348322321","_cell_guid":"24e4ef7e-6b1f-46ae-9bc7-c50e174b53f7","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-07-01T08:37:17.070783Z","iopub.execute_input":"2024-07-01T08:37:17.071149Z","iopub.status.idle":"2024-07-01T08:37:17.117342Z","shell.execute_reply.started":"2024-07-01T08:37:17.071104Z","shell.execute_reply":"2024-07-01T08:37:17.116375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _parse_tfrecord_function(example, opcodes_lookup_table, bytes_lookup_table):\n    example_fmt = {\n        'opcodes': tf.io.FixedLenFeature([], tf.string),\n        'bytes': tf.io.FixedLenFeature([], tf.string),\n        'APIs': tf.io.FixedLenFeature([], tf.string),\n        'label': tf.io.FixedLenFeature([], tf.int64)\n        }\n    parsed = tf.io.parse_single_example(example, example_fmt)\n\n    tokenizer = text.WhitespaceTokenizer()\n\n    opcodes_tokens = tokenizer.tokenize(parsed['opcodes'])\n    opcodes_IDs = opcodes_lookup_table.lookup(opcodes_tokens)\n\n    bytes_tokens = tokenizer.tokenize(parsed['bytes'])\n    bytes_IDs = bytes_lookup_table.lookup(bytes_tokens)\n\n    feature_vector = tf.io.decode_raw(parsed['APIs'], tf.float32)\n    return opcodes_IDs, bytes_IDs, feature_vector, parsed['label']\n\n\ndef make_dataset(filepath,\n                 opcodes_lookup_table,\n                 bytes_lookup_table,\n                 SHUFFLE_BUFFER_SIZE=1024,\n                 BATCH_SIZE=32,\n                 EPOCHS=5):\n    dataset = tf.data.TFRecordDataset(filepath)\n    dataset = dataset.shuffle(SHUFFLE_BUFFER_SIZE)\n    dataset = dataset.repeat(EPOCHS)\n    dataset = dataset.map(lambda x: _parse_tfrecord_function(x, opcodes_lookup_table, bytes_lookup_table))\n    dataset = dataset.batch(batch_size=BATCH_SIZE)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2024-07-01T08:37:20.305150Z","iopub.execute_input":"2024-07-01T08:37:20.305603Z","iopub.status.idle":"2024-07-01T08:37:20.318515Z","shell.execute_reply.started":"2024-07-01T08:37:20.305563Z","shell.execute_reply":"2024-07-01T08:37:20.316014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def initialize_TFRecords(tfrecords_filepath, num_tfrecords=10, filename=\"training\"):\n    training_writers = []\n    for i in range(num_tfrecords):\n        training_writers.append(tf.io.TFRecordWriter(tfrecords_filepath + \"{}{}.tfrecords\".format(filename,i)))\n    return training_writers\n\ndef create_lookup_table(vocabulary_mapping, num_oov_buckets):\n    keys = [k for k in vocabulary_mapping.keys()]\n    values = [tf.constant(vocabulary_mapping[k], dtype=tf.int64) for k in keys]\n\n    table = tf.lookup.StaticVocabularyTable(\n        tf.lookup.KeyValueTensorInitializer(\n            keys=keys,\n            values=values\n        ),\n        num_oov_buckets\n    )\n    return table\n\ndef _bytes_feature(values):\n    # Convert list of integers to bytes\n    byte_string = bytes(values)\n    # Create BytesList from bytes\n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=[byte_string]))\n\ndef _int64_feature(value):\n    return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))\n\ndef load_vocabulary(vocabulary_filepath):\n    with open(vocabulary_filepath, \"r\") as vocab_file:\n        vocabulary_dict = json.load(vocab_file)\n    return vocabulary_dict\n\ndef serialize_hydra_example(opcodes, bytes, apis_values, label):\n    feature = {\n        'opcodes': _bytes_feature(opcodes.encode('UTF-8')),\n        'bytes': _bytes_feature(bytes.encode('UTF-8')),\n        'APIs': _bytes_feature(apis_values),\n        'label': _int64_feature(label)\n    }\n    example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example_proto.SerializeToString()\n\ndef load_parameters(parameters_path):\n    with open(parameters_path, \"r\") as param_file:\n        params = json.load(param_file)\n    return params\n\nclass MetaPHOR:\n    def __init__(self, asm_filepath):\n        self.asm_filepath = asm_filepath\n        self.vocab = {}\n\n    def extract_windows_api_calls(self):\n        # Define regular expressions for Windows API calls\n        api_regex = re.compile(r'(call|jmp)\\s+(\\w+)(@.*)?$')\n        winapi_regex = re.compile(r'^(A|W|Nt|Zw)[a-zA-Z]+')\n\n        api_calls = set()\n\n        with open(self.asm_filepath, 'r', encoding='KOI8-R') as f:\n            for line in f:\n                match = api_regex.search(line)\n                if match:\n                    api_name = match.group(2)\n                    if winapi_regex.match(api_name):\n                        api_calls.add(api_name)\n\n        return api_calls\n\n    def count_windows_api_calls(self):\n        # Define regular expression for Windows API calls\n        api_regex = re.compile(r'call\\s+(\\w+)')\n\n        api_counts = {api: 0 for api in ['VirtualAlloc', 'CreateFile', 'ReadFile', 'WriteFile', 'CloseHandle', 'GetModuleHandle', 'GetProcAddress', 'LoadLibrary', 'ExitProcess', 'OpenProcess', 'CreateProcess', 'CreateThread', 'RegOpenKeyEx', 'RegSetValueEx', 'Process32Next', 'Process32First', 'CreateToolhelp32Snapshot', 'LookupPrivilegeValue', 'AdjustTokenPrivileges', 'VirtualProtect', 'WriteProcessMemory', 'NtUnmapViewOfSection', 'NtCreateSection', 'NtMapViewOfSection', 'QueueUserAPC', 'SuspendThread', 'ResumeThread', 'CreateRemoteThread', 'RtlCreateUserThread', 'NtCreateThreadEx', 'GetThreadContext', 'SetThreadContext']}\n\n        with open(self.asm_filepath, 'r', encoding='KOI8-R') as f:\n            try: \n                for line in f:\n                    try: \n                        match = api_regex.search(line)\n                        if match:\n                            api_name = match.group(1)\n                            if api_name in api_counts:\n                                api_counts[api_name] += 1\n                    except:\n                        print(\"Error in line: \", line)\n                        continue\n            except:\n                print(\"Error in file: \", self.asm_filepath)\n                return None\n\n        return list(api_counts.values())\n    \n    def get_hexadecimal_data_as_list(self):\n        hex_data = []\n\n        with open(self.asm_filepath, 'r', encoding='KOI8-R') as asm_file:\n            for line in asm_file:\n                hex_values = re.findall(r'\\b[0-9A-Fa-f]{2}\\b', line)\n                hex_data.extend(hex_values)\n\n        return hex_data\n\n    def get_opcodes_data_as_list(self, vocab_mapping):\n        opcodes = []\n\n        with open(self.asm_filepath, 'r', encoding='KOI8-R') as asm_file:\n            for line in asm_file:\n                opcode_match = re.findall(r'\\b[A-Za-z]+\\b', line) # 30 is to skip the address and hex data\n                opcodes.extend(opcode_match)\n                \n        return opcodes","metadata":{"execution":{"iopub.status.busy":"2024-07-01T08:37:25.042158Z","iopub.execute_input":"2024-07-01T08:37:25.042633Z","iopub.status.idle":"2024-07-01T08:37:25.077879Z","shell.execute_reply.started":"2024-07-01T08:37:25.042591Z","shell.execute_reply":"2024-07-01T08:37:25.076639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class HYDRA_Training():\n    model = None\n    tr_tfrecord = None\n    val_tfrecord = None\n    parameters = None\n    opcodes_vocabulary_mapping_filepath = None\n    bytes_vocabulary_mapping_filepath = None\n    test_tfrecord = None\n    opcodes_vocabulary_mapping = None\n    bytes_vocabulary_mapping = None\n    opcode_lookup_table = None\n    bytes_lookup_table = None\n    loss_func = None\n    optimizer = None\n    accuracy = None\n    epoch_loss_avg = None\n    epoch_accuracy = None\n    val_epoch_loss_avg = None\n    val_epoch_accuracy = None\n    initial_loss = 10\n\n    checkpoint_path = f\"\"\n    validation_loss_results = []\n    validation_accuracy_results = []\n    validation_precision_results = []\n    validation_recall_results = []\n    validation_f1score_results = []\n\n    train_loss_results = []\n    train_accuracy_results = []\n    train_precision_results = []\n    train_recall_results = []\n    train_f1score_results = []\n    \n    test_epoch_precision = []\n    test_epoch_recall = []\n    \n    d_train = None\n    d_val = None\n    d_test = None\n\n    def __init__(self,\n                 model = \"Hydra\",\n                 tr_tfrecord = \"/kaggle/input/hydra-dataset/train-001.tfrecords\",\n                 val_tfrecord = \"/kaggle/input/hydra-dataset/val.tfrecords\",\n                 parameters = \"/kaggle/input/big2015-tfrecords/hydra_parameters.json\",\n                 opcodes_vocabulary_mapping_filepath = \"/kaggle/input/big2015-tfrecords/opcodes.json\",\n                 bytes_vocabulary_mapping_filepath = \"/kaggle/input/big2015-tfrecords/bytes.json\",\n                 test_tfrecord = \"/kaggle/input/hydra-dataset/test.tfrecords\"):\n        self.model = model\n        self.tr_tfrecord = tr_tfrecord\n        self.val_tfrecord = val_tfrecord\n        self.parameters = parameters\n        self.opcodes_vocabulary_mapping_filepath = opcodes_vocabulary_mapping_filepath\n        self.bytes_vocabulary_mapping_filepath = bytes_vocabulary_mapping_filepath\n        self.test_tfrecord = test_tfrecord\n    \n        # num_epochs = self.parameters['epochs']\n        self.num_epochs = 100\n        self.batch_size = 54\n        self.learning_rate = 0.00025\n        self.initial_loss = 10.0\n        \n    def init_model(self):\n        print(\"TensorFlow version: {}\".format(tf.__version__))\n        print(\"Eager execution: {}\".format(tf.executing_eagerly()))\n        print(\"Num GPUs Available: \", len(tf.config.experimental.list_physical_devices('GPU')))\n\n        gpus = tf.config.experimental.list_physical_devices('GPU')\n        print(gpus)\n        \n        if gpus:\n            try:\n                tf.config.experimental.set_visible_devices(gpus[0], 'GPU')\n                logical_gpus = tf.config.experimental.list_logical_devices('GPU')\n                print(len(gpus), \"Physical GPUs,\", len(logical_gpus), \"Logical GPU\")\n            except RuntimeError as e:\n                # Visible devices must be set before GPUs have been initialized\n                # print(e)\n                pass\n\n        #Load vocabulary and create lookup table\n        self.opcodes_vocabulary_mapping = load_vocabulary(self.opcodes_vocabulary_mapping_filepath)\n        self.bytes_vocabulary_mapping = load_vocabulary(self.bytes_vocabulary_mapping_filepath)\n\n        self.opcodes_lookup_table = create_lookup_table(self.opcodes_vocabulary_mapping, 1)\n        self.bytes_lookup_table = create_lookup_table(self.bytes_vocabulary_mapping, 1)\n\n        # Load parameters of the model\n        self.parameters = load_parameters(self.parameters)\n\n        # Specify GPU\n        if \"gpu\" in self.parameters.keys():\n            os.environ[\"CUDA_VISIBLE_DEVICES\"] = self.parameters[\"gpu\"]\n\n\n        self.model = HYDRA(self.parameters)\n\n        self.loss_func = tf.keras.losses.SparseCategoricalCrossentropy()\n        self.accuracy = tf.keras.metrics.SparseCategoricalAccuracy()\n        self.epoch_recall   = tf.keras.metrics.Recall()\n        self.epoch_precision = tf.keras.metrics.Precision()\n\n        self.optimizer = tf.keras.optimizers.Adam(learning_rate=self.learning_rate)\n\n    def load_weights(self):\n        if os.path.isdir(model_path):\n            print(\"LOADING WEIGHTS!!!!\")\n            latest = tf.train.latest_checkpoint(model_path)\n            self.model.load_weights(latest)\n            print(\"LOADED WEIGHTS!!!!\")\n    \n    def train_loop(self, opcodes, bytes, apis, labels, training=False, loss_func=None, optimizer=None):\n        # Define the GradientTape context\n        with tf.GradientTape() as tape:\n            # Get the probabilities\n            predictions = self.model(opcodes, bytes, apis, training)\n            #labels = tf.dtypes.cast(labels, tf.float32)\n            # Calculate the loss\n            loss = self.loss_func(labels, predictions)\n        # Get the gradients\n        gradients = tape.gradient(loss, self.model.trainable_variables)\n        # Update the weights\n        self.optimizer.apply_gradients(zip(gradients, self.model.trainable_variables))\n        return loss, predictions\n    \n    def train(self, init=False):\n        # Training loop\n        # 1/ Iterate each epoch. An epoch is one pass through the dataset\n        # 2/ Whithin an epoch, iterate over each example in the training Dataset.\n        # 3/ Calculate model's loss and gradients\n        # 4/ Use an optimizer to update the model's variables\n        # 5/ Keep track of stats and repeat\n\n        for epoch in range(self.num_epochs):\n            print(\"#### Current epoch: {}\".format(epoch))\n            self.checkpoint_path = \"hydra.ckpt\"\n\n            # Training metrics\n            self.epoch_loss_avg = tf.keras.metrics.Mean()\n            self.epoch_accuracy = tf.keras.metrics.SparseCategoricalAccuracy()\n            self.epoch_recall   = tf.keras.metrics.Recall()\n            self.epoch_precision = tf.keras.metrics.Precision()\n\n            # Validation metrics\n            self.val_epoch_loss_avg = tf.keras.metrics.Mean()\n            self.val_epoch_accuracy = tf.keras.metrics.SparseCategoricalAccuracy()\n            self.val_epoch_recall   = tf.keras.metrics.Recall()\n            self.val_epoch_precision = tf.keras.metrics.Precision()\n\n            tr_step = 0\n\n            # Training loop\n            for step, (opcodes, bytes, apis, y) in enumerate(self.d_train):\n                #print(\"Input: {}\".format(x))\n                print(\"#### Step {} of epoch {}\".format(step, epoch))\n\n                loss, y_ = self.train_loop(opcodes, bytes, apis, y, True)\n                \n                y_pred = tf.math.argmax(y_, axis=1)\n\n                # Track progress\n                self.epoch_loss_avg(loss)\n                self.epoch_accuracy(y, y_)\n                self.epoch_precision(y, y_pred)\n                self.epoch_recall(y, y_pred)\n                self.epoch_f1score = 2 * (self.epoch_precision.result() * self.epoch_recall.result()) / (self.epoch_precision.result() + self.epoch_recall.result())\n\n                print(\"#### Iteration step: {}; Loss: {:.3f}, Accuracy: {:.3%}, Precision: {:.3%}, Recall: {:.3%}, F1Score: {:.3%}\".format(tr_step,\n                                                                                    self.epoch_loss_avg.result(),\n                                                                                    self.epoch_accuracy.result(),\n                                                                                    self.epoch_precision.result(),\n                                                                                    self.epoch_recall.result(),\n                                                                                    self.epoch_f1score\n                                                                                    ))\n                if (init == True):\n                    break\n\n                tr_step += 1\n\n            # End epoch\n            self.train_loss_results.append(self.epoch_loss_avg.result())\n            self.train_accuracy_results.append(self.epoch_accuracy.result())\n            self.train_precision_results.append(self.epoch_precision.result())\n            self.train_recall_results.append(self.epoch_recall.result())\n            self.train_f1score_results.append(self.epoch_f1score)\n\n            # Run a validation loop at the end of each epoch.\n            self.validation(epoch)\n            if (init == True):\n                break\n                \n            # Test model after 10 epoch\n            if (epoch % 10 == 0):\n                self.test()\n\n        # Training is done!\n        print(\"Training is done!\")\n\n        # Test the model\n        _,_,explainations = self.test()\n        # Return weights of the model           \n        self.model.save_weights(self.checkpoint_path) # Save only the weights\n        return self.model.get_weights()\n    \n    def visualize(self):\n        # Assuming `y` contains the labels for your dataset\n        labels = ['Ramnit', 'Lollipop', 'Kelihos_ver3', 'Vundo', 'Simda', 'Tracur', 'Kelihos_ver1', 'Obfuscator.ACY', 'Gatak', 'Benign']\n\n        self.d_train = make_dataset(self.tr_tfrecord,\n                                    self.opcodes_lookup_table,\n                                    self.bytes_lookup_table,\n                                    self.parameters['buffer_size'],\n                                    1,\n                                    1)\n            \n        self.d_val = make_dataset(self.val_tfrecord,\n                                    self.opcodes_lookup_table,\n                                    self.bytes_lookup_table,\n                                    1024,\n                                    1,\n                                    1)\n        \n        self.d_test = make_dataset(self.test_tfrecord,\n                                    self.opcodes_lookup_table,\n                                    self.bytes_lookup_table,\n                                    1,\n                                    1,\n                                    1)\n            \n        train_label_counts = [0] * len(labels)\n        val_label_counts = [0] * len(labels)\n        test_label_counts = [0] * len(labels)\n\n        # Assuming `d_train`, `d_val`, and `d_test` contain the respective datasets\n        for _, _, _, y in self.d_train:\n            train_label_counts[int(y[0])] += 1\n\n        for _, _, _, y in self.d_val:\n            val_label_counts[int(y[0])] += 1\n\n        for _, _, _, y in self.d_test:\n            test_label_counts[int(y[0])] += 1\n\n        # Create a bar plot\n        bar_width = 0.5\n        index = range(len(labels))\n        # Create separate bar plots for each set\n        fig, ax = plt.subplots()\n\n        # Train Set\n        ax.bar(index, train_label_counts, width=bar_width, label='Train')\n\n        # Set labels and title for the train plot\n        ax.set_xlabel('Classes')\n        ax.set_ylabel('Count')\n        ax.set_title('Number of Samples per Class - Train Set')\n        ax.set_xticks(index)\n        ax.set_xticklabels(labels)\n\n        # Show the train plot\n        plt.show()\n\n        # Validation Set\n        fig, ax = plt.subplots()\n\n        ax.bar(index, val_label_counts, width=bar_width, label='Validation')\n\n        # Set labels and title for the validation plot\n        ax.set_xlabel('Classes')\n        ax.set_ylabel('Count')\n        ax.set_title('Number of Samples per Class - Validation Set')\n        ax.set_xticks(index)\n        ax.set_xticklabels(labels)\n\n        # Show the validation plot\n        plt.show()\n\n        # Test Set\n        fig, ax = plt.subplots()\n\n        ax.bar(index, test_label_counts, width=bar_width, label='Test')\n\n        # Set labels and title for the test plot\n        ax.set_xlabel('Classes')\n        ax.set_ylabel('Count')\n        ax.set_title('Number of Samples per Class - Test Set')\n        ax.set_xticks(index)\n        ax.set_xticklabels(labels)\n\n        # Show the test plot\n        plt.show()\n        \n        self.d_train = make_dataset(self.tr_tfrecord,\n                            self.opcodes_lookup_table,\n                            self.bytes_lookup_table,\n                            self.parameters['buffer_size'],\n                            self.batch_size,\n                            1)\n\n    def validation(self, epoch):\n        y_actual_test = []\n        y_pred_test = []\n        \n        for opcodes_batch_val, bytes_batch_val, apis_batch_val, y_batch_val in self.d_val:\n            val_logits = self.model(opcodes_batch_val, bytes_batch_val, apis_batch_val, False)\n            val_loss = self.loss_func(y_batch_val, val_logits)\n\n            # Update metrics\n            y_pred = tf.argmax(val_logits, axis=-1)\n            y_pred_test.extend(y_pred)\n            y_actual_test.extend(y_batch_val)\n            \n            self.val_epoch_loss_avg(val_loss)\n            self.val_epoch_accuracy(y_batch_val, val_logits)\n            self.val_epoch_precision(y_actual_test, y_pred_test)\n            self.val_epoch_recall(y_actual_test, y_pred_test)\n\n        val_acc = self.val_epoch_accuracy.result()\n        val_loss = self.val_epoch_loss_avg.result()\n        val_recall = self.val_epoch_recall.result()\n        val_precision = self.val_epoch_precision.result()\n        val_f1score = 2 * (val_precision * val_recall) / (val_precision + val_recall)\n\n        print('#### Epoch: {}; Validation loss {}; acc: {}, Recall:  {:.3%}, Precision:  {:.3%}, F1Score:  {:.3%}'.format(epoch, val_loss, val_acc, val_recall, val_precision, val_f1score))\n\n        self.validation_loss_results.append(val_loss)\n        self.validation_accuracy_results.append(val_acc)\n        self.validation_precision_results.append(val_precision)\n        self.validation_recall_results.append(val_recall)\n        self.validation_f1score_results.append(val_f1score)\n\n        if float(val_loss) < self.initial_loss:\n            self.initial_loss = float(val_loss)\n            self.model.save_weights(self.checkpoint_path) # Save only the weights\n\n    def test(self):\n        # Load the model\n        self.model.load_weights(self.checkpoint_path)\n        test_epoch_loss_avg = tf.keras.metrics.Mean()\n        test_epoch_accuracy = tf.keras.metrics.SparseCategoricalAccuracy()\n        test_epoch_recall = tf.keras.metrics.Recall()\n        test_epoch_precision = tf.keras.metrics.Precision()\n\n        y_actual_test = []\n        y_pred_test = []\n        explanations = []  # List to store LIME explanations\n\n        # Evaluate model on the test set\n        for opcodes_batch_test, bytes_batch_test, apis_batch_test, y_batch_test in self.d_test:\n            test_logits = self.model(opcodes_batch_test, bytes_batch_test, apis_batch_test, training=False)\n            test_loss = self.loss_func(y_batch_test, test_logits)\n\n            # For the confusion matrix\n            y_pred = tf.argmax(test_logits, axis=-1)\n            y_pred_test.extend(y_pred.numpy())\n            y_actual_test.extend(y_batch_test.numpy())\n\n            # Update metrics\n            test_epoch_loss_avg(test_loss)\n            test_epoch_accuracy(y_batch_test, test_logits)\n            test_epoch_precision(y_actual_test, y_pred_test)\n            test_epoch_recall(y_actual_test, y_pred_test)\n\n#             # Apply LIME for explanation for each input type\n#             explainer_opcodes = lime.lime_text.LimeTextExplainer(class_names=self.labels)\n#             explainer_bytes = lime.lime_text.LimeTextExplainer(class_names=self.labels)\n#             explainer_apis = lime.lime_text.LimeTextExplainer(class_names=self.labels)\n\n#             for i in range(opcodes_batch_test.shape[0]):\n#                 # Create a wrapper function for each input type\n#                 predict_fn_opcodes = lambda x: self.model.new_predict([x, bytes_batch_test[i:i+1], apis_batch_test[i:i+1]])\n#                 predict_fn_bytes = lambda x: self.model.new_predict([opcodes_batch_test[i:i+1], x, apis_batch_test[i:i+1]])\n#                 predict_fn_apis = lambda x: self.model.new_predict([opcodes_batch_test[i:i+1], bytes_batch_test[i:i+1], x])\n\n#                 # Generate LIME explanation for each input type\n#                 explanation_opcodes = explainer_opcodes.explain_instance(opcodes_batch_test[i], predict_fn_opcodes, num_features=5)\n#                 explanation_bytes = explainer_bytes.explain_instance(bytes_batch_test[i], predict_fn_bytes, num_features=5)\n#                 explanation_apis = explainer_apis.explain_instance(apis_batch_test[i], predict_fn_apis, num_features=5)\n\n#                 explanations.append((explanation_opcodes, explanation_bytes, explanation_apis))\n\n#                 # Log the top features affecting the result\n#                 print(\"Top features affecting the result for opcodes:\")\n#                 for feature in explanation_opcodes.as_list():\n#                     print(feature[0])\n\n#                 print(\"Top features affecting the result for bytes:\")\n#                 for feature in explanation_bytes.as_list():\n#                     print(feature[0])\n\n#                 print(\"Top features affecting the result for APIs:\")\n#                 for feature in explanation_apis.as_list():\n#                     print(feature[0])\n\n\n        test_acc = test_epoch_accuracy.result()\n        test_loss = test_epoch_loss_avg.result()\n        test_precision = test_epoch_precision.result()\n        test_recall = test_epoch_recall.result()\n        test_f1score = 2 * (test_precision * test_recall) / (test_precision + test_recall)\n\n        print('Test loss {}; acc: {}, precision: {}, recall: {}, f1score: {}'.format(\n            test_loss, test_acc, test_precision, test_recall, test_f1score))\n\n        cm = confusion_matrix(y_actual_test, y_pred_test)\n        plt.figure(figsize=(10, 8))\n        sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\n        plt.xlabel('Predicted Labels')\n        plt.ylabel('True Labels')\n        plt.title('Confusion Matrix')\n        plt.show()\n\n        return test_acc, test_loss, explanations\n    \n    #######\n    # Updating\n    #######\n    \n#     def custom_test(self, filePath):\n        \n#         opcodes_vocabulary_mapping_filepath='/kaggle/input/big2015-tfrecords/opcodes.json'\n#         bytes_vocabulary_mapping_filepath='/kaggle/input/big2015-tfrecords/bytes.json'\n        \n#         opcodes_vocabulary_mapping = load_vocabulary(opcodes_vocabulary_mapping_filepath)\n#         bytes_vocabulary_mapping = load_vocabulary(bytes_vocabulary_mapping_filepath)\n        \n#         metaPHOR = MetaPHOR(filePath)\n\n#         # Extract opcodes\n#         opcodes = metaPHOR.get_opcodes_data_as_list(opcodes_vocabulary_mapping)\n#         if len(opcodes) < max_mnemonics:\n#             while len(opcodes) < max_mnemonics:\n#                 opcodes.append(\"PAD\")\n#         else:\n#             opcodes = opcodes[:max_mnemonics]\n#         raw_mnemonics = \" \".join(opcodes)\n\n#         # Extract bytes\n#         bytes_sequence = metaPHOR.get_hexadecimal_data_as_list()\n#         for i in range(len(bytes_sequence)):\n#             if bytes_sequence[i] not in bytes_vocabulary_mapping.keys():\n#                 bytes_sequence[i] = \"UNK\"\n#         if len(bytes_sequence) < max_bytes:\n#             while len(bytes_sequence) < max_bytes:\n#                 bytes_sequence.append(\"PAD\")\n#         else:\n#             bytes_sequence = bytes_sequence[:max_bytes]\n#         raw_bytes_sequence = \" \".join(bytes_sequence)\n\n#         # Extract APIs\n#         feature_vector = metaPHOR.count_windows_api_calls()\n\n#         example = serialize_hydra_example(raw_mnemonics,\n#                                           raw_bytes_sequence,\n#                                           feature_vector,\n#                                           9)\n#         tfwriter.write(example)\n#         line += 1\n","metadata":{"execution":{"iopub.status.busy":"2024-07-01T08:37:30.515891Z","iopub.execute_input":"2024-07-01T08:37:30.516631Z","iopub.status.idle":"2024-07-01T08:37:30.599386Z","shell.execute_reply.started":"2024-07-01T08:37:30.516583Z","shell.execute_reply":"2024-07-01T08:37:30.598254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_train(folder: str, filenames: list, logs: bool = False):\n    directory_command = f'\"-o{folder}\"'\n    filenames_command = \"\".join(filenames)  # Use comma as separator\n    if logs:\n        print(filenames_command)\n    command = f\"7z x /kaggle/input/malware-classification/train.7z {directory_command} {filenames_command} -r\"\n    command = command + ' >/dev/null 2>&1'  # Done to hide console output\n    if logs:\n        print(command)\n        print('Started extracting')\n    os.system(command)\n    if logs:\n        print('Finished extracting')","metadata":{"execution":{"iopub.status.busy":"2024-07-01T08:37:39.224996Z","iopub.execute_input":"2024-07-01T08:37:39.225450Z","iopub.status.idle":"2024-07-01T08:37:39.232920Z","shell.execute_reply.started":"2024-07-01T08:37:39.225413Z","shell.execute_reply":"2024-07-01T08:37:39.231137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport tensorflow as tf\n\nimport os\n\npe_filepath='/kaggle/input/hydra-dataset/benignASM/benignASM(2)'\n# Get the list of filenames\nfilenames = os.listdir(pe_filepath)\n\n# Calculate the number of files for each partition\ntotal_files = len(filenames)\npart1_count = int(total_files * 0.7)\npart2_count = int(total_files * 0.15)\n\n# Divide the filenames into three parts\npart1 = filenames[:part1_count]\npart2 = filenames[part1_count:part1_count + part2_count]\npart3 = filenames[part1_count + part2_count:]\n\n\ndef dataset_to_tfrecords(pe_filepath, tfrecords_filepath, filenames,\n                         opcodes_vocabulary_mapping_filepath,\n                         bytes_vocabulary_mapping_filepath,\n                         max_mnemonics=2000000, max_bytes=2000000):\n\n    tfwriter = tf.io.TFRecordWriter(tfrecords_filepath)\n    opcodes_vocabulary_mapping = load_vocabulary(opcodes_vocabulary_mapping_filepath)\n    bytes_vocabulary_mapping = load_vocabulary(bytes_vocabulary_mapping_filepath)\n    \n#     filenames = os.listdir(pe_filepath)\n\n    line = 0\n    for file in filenames:\n        print(\"{};{}\".format(line, file))\n        metaPHOR = MetaPHOR(pe_filepath + \"/\" + file)\n\n        # Extract opcodes\n        opcodes = metaPHOR.get_opcodes_data_as_list(opcodes_vocabulary_mapping)\n        if len(opcodes) < max_mnemonics:\n            while len(opcodes) < max_mnemonics:\n                opcodes.append(\"PAD\")\n        else:\n            opcodes = opcodes[:max_mnemonics]\n        raw_mnemonics = \" \".join(opcodes)\n\n        # Extract bytes\n        bytes_sequence = metaPHOR.get_hexadecimal_data_as_list()\n        for i in range(len(bytes_sequence)):\n            if bytes_sequence[i] not in bytes_vocabulary_mapping.keys():\n                bytes_sequence[i] = \"UNK\"\n        if len(bytes_sequence) < max_bytes:\n            while len(bytes_sequence) < max_bytes:\n                bytes_sequence.append(\"PAD\")\n        else:\n            bytes_sequence = bytes_sequence[:max_bytes]\n        raw_bytes_sequence = \" \".join(bytes_sequence)\n\n        # Extract APIs\n        feature_vector = metaPHOR.count_windows_api_calls()\n\n        example = serialize_hydra_example(raw_mnemonics,\n                                          raw_bytes_sequence,\n                                          feature_vector,\n                                          9)\n        tfwriter.write(example)\n        line += 1","metadata":{"execution":{"iopub.status.busy":"2024-07-01T08:37:41.087538Z","iopub.execute_input":"2024-07-01T08:37:41.087920Z","iopub.status.idle":"2024-07-01T08:37:41.182819Z","shell.execute_reply.started":"2024-07-01T08:37:41.087886Z","shell.execute_reply":"2024-07-01T08:37:41.181836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_to_tfrecords(pe_filepath='/kaggle/input/hydra-dataset/benignASM/benignASM(2)', \n                     tfrecords_filepath='/kaggle/working/benignTrain.tfrecords', \n                     filenames=part1,\n                     opcodes_vocabulary_mapping_filepath='/kaggle/input/big2015-tfrecords/opcodes.json',\n                     bytes_vocabulary_mapping_filepath='/kaggle/input/big2015-tfrecords/bytes.json',\n                     max_mnemonics=50000, max_bytes=50000)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-07-01T08:37:43.816210Z","iopub.execute_input":"2024-07-01T08:37:43.817073Z","iopub.status.idle":"2024-07-01T08:41:52.963747Z","shell.execute_reply.started":"2024-07-01T08:37:43.817036Z","shell.execute_reply":"2024-07-01T08:41:52.962499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_to_tfrecords(pe_filepath='/kaggle/input/hydra-dataset/benignASM/benignASM(2)', \n                     tfrecords_filepath='/kaggle/working/benignVal.tfrecords', \n                     filenames=part2,\n                     opcodes_vocabulary_mapping_filepath='/kaggle/input/big2015-tfrecords/opcodes.json',\n                     bytes_vocabulary_mapping_filepath='/kaggle/input/big2015-tfrecords/bytes.json',\n                     max_mnemonics=50000, max_bytes=50000)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-07-01T08:42:46.367694Z","iopub.execute_input":"2024-07-01T08:42:46.368021Z","iopub.status.idle":"2024-07-01T08:43:38.625471Z","shell.execute_reply.started":"2024-07-01T08:42:46.367989Z","shell.execute_reply":"2024-07-01T08:43:38.624371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_to_tfrecords(pe_filepath='/kaggle/input/hydra-dataset/benignASM/benignASM(2)', \n                     tfrecords_filepath='/kaggle/working/benignTest.tfrecords', \n                     filenames=part3,\n                     opcodes_vocabulary_mapping_filepath='/kaggle/input/big2015-tfrecords/opcodes.json',\n                     bytes_vocabulary_mapping_filepath='/kaggle/input/big2015-tfrecords/bytes.json',\n                     max_mnemonics=50000, max_bytes=50000)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-07-01T08:43:38.627740Z","iopub.execute_input":"2024-07-01T08:43:38.628061Z","iopub.status.idle":"2024-07-01T08:44:32.559153Z","shell.execute_reply.started":"2024-07-01T08:43:38.628028Z","shell.execute_reply":"2024-07-01T08:44:32.558134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def merge_tfrecords(input_filepaths, output_filepath):\n    # Create a new TFRecordWriter for the output file\n    writer = tf.io.TFRecordWriter(output_filepath)\n\n    for input_filepath in input_filepaths:\n        # Create a TFRecordDataset to read the input file\n        dataset = tf.data.TFRecordDataset(input_filepath)\n\n        # Write each example from the input file to the output file\n        for raw_record in dataset:\n            writer.write(raw_record.numpy())\n\n    # Close the writer\n    writer.close()\n\n# Example usage\nmerge_tfrecords([\"/kaggle/input/hydra-dataset/train-001.tfrecords\", \"/kaggle/working/benignTrain.tfrecords\"], \"/kaggle/working/mergedTrain.tfrecords\")\nmerge_tfrecords([\"/kaggle/input/hydra-dataset/test.tfrecords\", \"/kaggle/working/benignTest.tfrecords\"], \"/kaggle/working/mergedTest.tfrecords\")\nmerge_tfrecords([\"/kaggle/input/hydra-dataset/val.tfrecords\", \"/kaggle/working/benignVal.tfrecords\"], \"/kaggle/working/mergedVal.tfrecords\")\n","metadata":{"execution":{"iopub.status.busy":"2024-07-01T08:44:32.560332Z","iopub.execute_input":"2024-07-01T08:44:32.560624Z","iopub.status.idle":"2024-07-01T08:45:15.597123Z","shell.execute_reply.started":"2024-07-01T08:44:32.560597Z","shell.execute_reply":"2024-07-01T08:45:15.595941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train and result","metadata":{}},{"cell_type":"code","source":"model = HYDRA_Training(tr_tfrecord='/kaggle/working/mergedTrain.tfrecords',\n                      test_tfrecord='/kaggle/working/mergedTest.tfrecords',\n                      val_tfrecord='/kaggle/working/mergedVal.tfrecords')\n\nmodel.init_model()\nmodel.visualize()\nweight = model.train()","metadata":{"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.test()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.validation(epoch=5)","metadata":{"execution":{"iopub.status.busy":"2024-05-27T13:01:58.914296Z","iopub.execute_input":"2024-05-27T13:01:58.915253Z","iopub.status.idle":"2024-05-27T13:03:13.475996Z","shell.execute_reply.started":"2024-05-27T13:01:58.915214Z","shell.execute_reply":"2024-05-27T13:03:13.474910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test model again","metadata":{}},{"cell_type":"code","source":"model = HYDRA_Training(tr_tfrecord='/kaggle/working/mergedTrain.tfrecords',\n                      test_tfrecord='/kaggle/working/mergedTest.tfrecords',\n                      val_tfrecord='/kaggle/working/mergedVal.tfrecords')\nmodel.init_model()","metadata":{"execution":{"iopub.status.busy":"2024-07-01T10:26:49.151162Z","iopub.execute_input":"2024-07-01T10:26:49.152079Z","iopub.status.idle":"2024-07-01T10:26:49.185466Z","shell.execute_reply.started":"2024-07-01T10:26:49.152040Z","shell.execute_reply":"2024-07-01T10:26:49.184418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.checkpoint_path = \"/kaggle/input/hydra_model/keras/sth/2/hydra.ckpt\"\n# model.visualize()\n# model.test()\n","metadata":{"execution":{"iopub.status.busy":"2024-07-01T08:45:33.117974Z","iopub.execute_input":"2024-07-01T08:45:33.119061Z","iopub.status.idle":"2024-07-01T08:45:33.123450Z","shell.execute_reply.started":"2024-07-01T08:45:33.119016Z","shell.execute_reply":"2024-07-01T08:45:33.122393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test a specific executable file","metadata":{}},{"cell_type":"code","source":"!objdump -d -M intel /*.exe > /kaggle/working/test.asm","metadata":{"execution":{"iopub.status.busy":"2024-07-01T08:56:46.099860Z","iopub.execute_input":"2024-07-01T08:56:46.100786Z","iopub.status.idle":"2024-07-01T08:56:47.105646Z","shell.execute_reply.started":"2024-07-01T08:56:46.100748Z","shell.execute_reply":"2024-07-01T08:56:47.104433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filePath=\"/kaggle/input/kelihos3/_ex-08.exe.lst\"\nopcodes_vocabulary_mapping_filepath='/kaggle/input/big2015-tfrecords/opcodes.json'\nbytes_vocabulary_mapping_filepath='/kaggle/input/big2015-tfrecords/bytes.json'\n\nopcodes_vocabulary_mapping = load_vocabulary(opcodes_vocabulary_mapping_filepath)\nbytes_vocabulary_mapping = load_vocabulary(bytes_vocabulary_mapping_filepath)\n\nmax_mnemonics=50000\nmax_bytes=50000\n\nmetaPHOR = MetaPHOR(filePath)\n\n# Extract opcodes\nopcodes = metaPHOR.get_opcodes_data_as_list(opcodes_vocabulary_mapping)\nif len(opcodes) < max_mnemonics:\n    while len(opcodes) < max_mnemonics:\n        opcodes.append(\"PAD\")\nelse:\n    opcodes = opcodes[:max_mnemonics]\nraw_mnemonics = \" \".join(opcodes)\n\n# Extract bytes\nbytes_sequence = metaPHOR.get_hexadecimal_data_as_list()\n\nfor i in range(len(bytes_sequence)):\n    if bytes_sequence[i] not in bytes_vocabulary_mapping.keys():\n        bytes_sequence[i] = \"UNK\"\nif len(bytes_sequence) < max_bytes:\n    while len(bytes_sequence) < max_bytes:\n        bytes_sequence.append(\"PAD\")\nelse:\n    bytes_sequence = bytes_sequence[:max_bytes]\nraw_bytes_sequence = \" \".join(bytes_sequence)\n\n# Extract APIs\nfeature_vector = metaPHOR.count_windows_api_calls()\n\n\nfeature = {\n        'opcodes': _bytes_feature(raw_mnemonics.encode('UTF-8')),\n        'bytes': _bytes_feature(raw_bytes_sequence.encode('UTF-8')),\n        'APIs': _bytes_feature(feature_vector)\n    }\nexample_proto = tf.train.Example(features=tf.train.Features(feature=feature))\nexample = example_proto.SerializeToString()\n\n\n","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-07-01T10:26:51.305666Z","iopub.execute_input":"2024-07-01T10:26:51.306318Z","iopub.status.idle":"2024-07-01T10:27:01.402334Z","shell.execute_reply.started":"2024-07-01T10:26:51.306279Z","shell.execute_reply":"2024-07-01T10:27:01.401442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"opcodes_lookup_table = create_lookup_table(opcodes_vocabulary_mapping, 1)\nbytes_lookup_table = create_lookup_table(bytes_vocabulary_mapping, 1)\n\nlabels = ['Ramnit', 'Lollipop', 'Kelihos_ver3', 'Vundo', 'Simda', 'Tracur', 'Kelihos_ver1', 'Obfuscator.ACY', 'Gatak', 'Benign']\n\nexample_fmt = {\n        'opcodes': tf.io.FixedLenFeature([], tf.string),\n        'bytes': tf.io.FixedLenFeature([], tf.string),\n        'APIs': tf.io.FixedLenFeature([], tf.string)\n        }\nparsed = tf.io.parse_single_example(example, example_fmt)\n\ntokenizer = text.WhitespaceTokenizer()\n\nopcodes_tokens = tokenizer.tokenize(parsed['opcodes'])\nopcodes_IDs = opcodes_lookup_table.lookup(opcodes_tokens)\n\nbytes_tokens = tokenizer.tokenize(parsed['bytes'])\nbytes_IDs = bytes_lookup_table.lookup(bytes_tokens)\n\nfeature_vector = tf.io.decode_raw(parsed['APIs'], tf.float32)\n\n\n\nopcode_np = tf.expand_dims(opcodes_IDs,axis=0)\nbytes_np = tf.expand_dims(bytes_IDs,axis=0)\napi_np = tf.expand_dims(feature_vector,axis=0)\n\ntest_logits = model.model(opcode_np,bytes_np,api_np,training=False)\n\nprint(test_logits)\n\nmax_value = tf.argmax(test_logits,axis=1).numpy()[0]\n\nprint(\"Malware of this binary: \", labels[max_value])","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}