{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"model_path = \"/kaggle/working/models/hydra/\"\n!pip install chardet\n!pip install py7zr","metadata":{"execution":{"iopub.status.busy":"2023-06-12T03:51:00.176727Z","iopub.execute_input":"2023-06-12T03:51:00.177237Z","iopub.status.idle":"2023-06-12T03:51:33.702285Z","shell.execute_reply.started":"2023-06-12T03:51:00.177196Z","shell.execute_reply":"2023-06-12T03:51:33.7004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_text as text\nimport json\nimport pandas as pd\nimport numpy as np\nimport math\nimport re\nimport chardet\nimport os\nimport argparse\nimport zipfile\nimport errno\nimport py7zr\nfrom sklearn.metrics import confusion_matrix\n\nimport socket\nimport pickle\nimport numpy, struct\nfrom zipfile import ZipFile\n\nfrom sklearn.metrics import accuracy_score\n\nimport argparse\nproject_path = os.path.dirname(os.path.realpath(\"../../../\"))\nimport sys\nimport csv\nsys.path.append(project_path)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T03:51:33.705305Z","iopub.execute_input":"2023-06-12T03:51:33.705756Z","iopub.status.idle":"2023-06-12T03:51:44.248079Z","shell.execute_reply.started":"2023-06-12T03:51:33.705708Z","shell.execute_reply":"2023-06-12T03:51:44.24666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n# import numpy as np\n\n# # Load CSV data into a DataFrame\n# data = pd.read_csv('/kaggle/input/malware-classification/trainLabels.csv')\n\n# # Calculate the percentage distribution of each label\n# label_counts = data['Class'].value_counts()\n# label_percentages = label_counts / len(data)\n\n# # Define the number of parts to split the data into\n# n = 10\n\n# # Calculate the desired percentage distribution for each label in each part\n# desired_percentages = label_percentages / n\n\n# # Create empty lists to hold the partitions for each part\n# partitions = [[] for _ in range(n)]\n\n# # Iterate through each label and partition the data\n# for label in label_counts.index:\n#     # Get the samples with the current label\n#     label_samples = data[data['Class'] == label]\n    \n#     # Shuffle the samples randomly\n#     label_samples = label_samples.sample(frac=1, random_state=42)\n    \n#     # Split the samples into n parts based on the desired percentage distribution\n#     for i in range(n):\n#         num_samples = int(desired_percentages[label] * len(data))\n#         start_index = i * num_samples\n#         end_index = (i + 1) * num_samples\n#         partitions[i].extend(label_samples[start_index:end_index])\n\n# # Further process or analyze the partitions as needed\n# for i, partition in enumerate(partitions):\n#     partition_data = pd.concat(partition)\n#     # Do something with each partition of data\n#     # For example, save it to a separate CSV file\n#     partition_data.to_csv(f'partition_{i}.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-12T03:51:44.250507Z","iopub.execute_input":"2023-06-12T03:51:44.251844Z","iopub.status.idle":"2023-06-12T03:51:44.259791Z","shell.execute_reply.started":"2023-06-12T03:51:44.251773Z","shell.execute_reply":"2023-06-12T03:51:44.25868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('/kaggle/input/big2015-tfrecords/partition_0.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-12T03:51:44.262495Z","iopub.execute_input":"2023-06-12T03:51:44.262843Z","iopub.status.idle":"2023-06-12T03:51:44.328505Z","shell.execute_reply.started":"2023-06-12T03:51:44.262796Z","shell.execute_reply":"2023-06-12T03:51:44.327398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def max_fcount():\n    import os, re\n    checkpoint_dir = \"/kaggle/working/models/hydra/\"\n    # Find all checkpoint files in the directory\n    checkpoint_files = [f for f in os.listdir(checkpoint_dir) if re.match(r'model_\\d+\\.ckpt\\.index', f)]\n    # Extract the numbers from the file names and find the maximum\n    max_number = max([int(re.search(r'model_(\\d+)\\.ckpt\\.index', f).group(1)) for f in checkpoint_files])\n    # Construct the file name with the maximum number\n    max_checkpoint_file = os.path.join('model_{}'.format(max_number))\n    print('Maximum checkpoint file:', max_checkpoint_file)\n    return max_checkpoint_file, max_number","metadata":{"execution":{"iopub.status.busy":"2023-06-12T03:51:44.330439Z","iopub.execute_input":"2023-06-12T03:51:44.331201Z","iopub.status.idle":"2023-06-12T03:51:44.338722Z","shell.execute_reply.started":"2023-06-12T03:51:44.331158Z","shell.execute_reply":"2023-06-12T03:51:44.337618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hydra _ architecture\nclass HYDRA(tf.keras.Model):\n    def __init__(self, parameters):\n        super(HYDRA, self).__init__()\n        self.parameters = parameters\n\n    def build(self, input_shapes):\n        # Bytes component\n        ######################################### Bytes component ######################################################\n        self.bytes_emb = tf.keras.layers.Embedding(self.parameters['bytes']['V'], self.parameters['bytes']['E'],\n                                             input_shape=(None, self.parameters['bytes']['max_bytes_values']))\n\n        self.bytes_conv_1 = tf.keras.layers.Conv2D(filters=self.parameters['bytes']['num_filters'][0],\n                                             kernel_size=[self.parameters['bytes']['kernel_sizes'][0],\n                                                          self.parameters['bytes']['E']],\n                                             strides=(self.parameters['bytes']['strides'][0], 1),\n                                             data_format='channels_last',\n                                             use_bias=True,\n                                             activation=\"relu\")\n\n        self.bytes_conv_2 = tf.keras.layers.Conv2D(filters=self.parameters['bytes']['num_filters'][1],\n                                             kernel_size=[self.parameters['bytes']['kernel_sizes'][1],\n                                                          1],\n                                             strides=(self.parameters['bytes']['strides'][1], 1),\n                                             data_format='channels_last',\n                                             use_bias=True,\n                                             activation=\"relu\")\n\n        self.bytes_max_pool_1 = tf.keras.layers.MaxPooling2D(pool_size=(self.parameters['bytes']['max_pool_size'], 1))\n\n        self.bytes_conv_3 = tf.keras.layers.Conv2D(filters=self.parameters['bytes']['num_filters'][2],\n                                             kernel_size=[self.parameters['bytes']['kernel_sizes'][2],\n                                                          1],\n                                             strides=(self.parameters['bytes']['strides'][2], 1),\n                                             data_format='channels_last',\n                                             use_bias=True,\n                                             activation=\"relu\")\n\n        self.bytes_conv_4 = tf.keras.layers.Conv2D(filters=self.parameters['bytes']['num_filters'][3],\n                                             kernel_size=[self.parameters['bytes']['kernel_sizes'][3],\n                                                          1],\n                                             strides=(self.parameters['bytes']['strides'][3], 1),\n                                             data_format='channels_last',\n                                             use_bias=True,\n                                             activation=\"relu\")\n\n        self.bytes_global_avg_pool = tf.keras.layers.GlobalAvgPool2D()\n\n        self.bytes_drop_1 = tf.keras.layers.Dropout(self.parameters[\"dropout_rate\"])\n        self.bytes_dense_1 = tf.keras.layers.Dense(self.parameters['bytes']['hidden'][0],\n                                             activation=\"selu\")\n\n        self.bytes_drop_2 = tf.keras.layers.Dropout(self.parameters[\"dropout_rate\"])\n        self.bytes_dense_2 = tf.keras.layers.Dense(self.parameters['bytes']['hidden'][1],\n                                             activation=\"selu\")\n\n        self.bytes_drop_3 = tf.keras.layers.Dropout(self.parameters[\"dropout_rate\"])\n\n        self.bytes_dense_3 = tf.keras.layers.Dense(self.parameters['bytes']['hidden'][2],\n                                             activation=\"selu\")\n\n        ####################################### Opcodes component ######################################################\n        self.opcodes_emb = tf.keras.layers.Embedding(self.parameters['opcodes']['V'],\n                                                   self.parameters['opcodes']['E'],\n                                                   input_shape=(None,\n                                                                self.parameters['opcodes']['seq_length']))\n\n        self.opcodes_conv_3 = tf.keras.layers.Conv2D(self.parameters['opcodes']['conv']['num_filters'],\n                                                   (self.parameters['opcodes']['conv']['size'][0],\n                                                    self.parameters['opcodes']['E']),\n                                                   activation=\"relu\",\n                                                   input_shape=(None,\n                                                                self.parameters['opcodes']['seq_length'],\n                                                                self.parameters['opcodes']['E']))\n        self.opcodes_global_max_pooling_3 = tf.keras.layers.GlobalMaxPooling2D()\n\n        self.opcodes_conv_5 = tf.keras.layers.Conv2D(self.parameters['opcodes']['conv']['num_filters'],\n                                                   (self.parameters['opcodes']['conv']['size'][1], self.parameters['opcodes']['E']),\n                                                   activation=\"relu\",\n                                                   input_shape=(None,\n                                                                self.parameters['opcodes']['seq_length'],\n                                                                self.parameters['opcodes']['E']))\n        self.opcodes_global_max_pooling_5 = tf.keras.layers.GlobalMaxPooling2D()\n\n        self.opcodes_conv_7 = tf.keras.layers.Conv2D(self.parameters['opcodes']['conv']['num_filters'],\n                                                   (self.parameters['opcodes']['conv']['size'][2], self.parameters['opcodes']['E']),\n                                                   activation=\"relu\",\n                                                   input_shape=(None,\n                                                                self.parameters['opcodes']['seq_length'],\n                                                                self.parameters['opcodes']['E']))\n        self.opcodes_global_max_pooling_7 = tf.keras.layers.GlobalMaxPooling2D()\n\n        ################################################# APIs Component ###############################################\n        self.apis_input_dropout = tf.keras.layers.Dropout(self.parameters[\"input_dropout_rate\"],\n                                                     input_shape=(None, self.parameters[\"api_features\"]))\n\n        self.apis_hidden_1 = tf.keras.layers.Dense(self.parameters['apis']['hidden'],\n                                        activation=\"relu\",\n                                        input_shape=(None, self.parameters[\"api_features\"]))\n\n        self.bytes_apis_dense_dropout = tf.keras.layers.Dropout(self.parameters[\"dropout_rate\"])\n        self.bytes_apis_dense = tf.keras.layers.Dense(self.parameters['output'], activation=\"selu\")\n\n        self.dense_dropout = tf.keras.layers.Dropout(self.parameters[\"dropout_rate\"])\n        self.dense = tf.keras.layers.Dense(self.parameters['output'], activation=\"selu\")\n\n        self.output_dropout = tf.keras.layers.Dropout(self.parameters[\"dropout_rate\"])\n        self.out = tf.keras.layers.Dense(self.parameters['output'],\n                                           activation=\"softmax\")\n\n    def call(self, opcodes_tensor, bytes_tensor, apis_tensor, training=False):\n        # Bytes subcomponent\n        bytes_emb = self.bytes_emb(bytes_tensor)\n        bytes_emb_expanded = tf.keras.backend.expand_dims(bytes_emb, axis=-1)\n\n        bytes_conv_1 = self.bytes_conv_1(bytes_emb_expanded)\n        bytes_conv_2 = self.bytes_conv_2(bytes_conv_1)\n\n        bytes_max_pool_1 = self.bytes_max_pool_1(bytes_conv_2)\n\n        bytes_conv_3 = self.bytes_conv_3(bytes_max_pool_1)\n        bytes_conv_4 = self.bytes_conv_4(bytes_conv_3)\n\n        bytes_features = self.bytes_global_avg_pool(bytes_conv_4)\n\n        bytes_drop_1 = self.bytes_drop_1(bytes_features, training=training)\n        bytes_dense_1 = self.bytes_dense_1(bytes_drop_1)\n\n        bytes_drop_2 = self.bytes_drop_2(bytes_dense_1, training=training)\n        bytes_dense_2 = self.bytes_dense_2(bytes_drop_2)\n\n        bytes_drop_3 = self.bytes_drop_3(bytes_dense_2, training=training)\n        bytes_dense_3 = self.bytes_dense_3(bytes_drop_3)\n\n        # Opcodes subcomponent\n        opcodes_emb = self.opcodes_emb(opcodes_tensor)\n        opcodes_emb_expanded = tf.keras.backend.expand_dims(opcodes_emb, axis=-1)\n\n        opcodes_conv_3 = self.opcodes_conv_3(opcodes_emb_expanded)\n        opcodes_pool_3 = self.opcodes_global_max_pooling_3(opcodes_conv_3)\n\n        opcodes_conv_5 = self.opcodes_conv_5(opcodes_emb_expanded)\n        opcodes_pool_5 = self.opcodes_global_max_pooling_5(opcodes_conv_5)\n\n        opcodes_conv_7 = self.opcodes_conv_7(opcodes_emb_expanded)\n        opcodes_pool_7 = self.opcodes_global_max_pooling_7(opcodes_conv_7)\n\n        #APIs subcomponent\n        apis_input_dropout = self.apis_input_dropout(apis_tensor, training=training)\n        apis_hidden1 = self.apis_hidden_1(apis_input_dropout)\n\n\n        # Features fusion\n        features_api_bytes = tf.keras.layers.concatenate([bytes_dense_3, apis_hidden1])\n        features_api_bytes_dropout = self.bytes_apis_dense_dropout(features_api_bytes, training=training)\n        dense_api_bytes = self.bytes_apis_dense(features_api_bytes_dropout)\n\n        features = tf.keras.layers.concatenate([opcodes_pool_3, opcodes_pool_5, opcodes_pool_7, dense_api_bytes])\n        features_dropout = self.dense_dropout(features, training=training)\n        dense_opcodes_apis_bytes = self.dense(features_dropout)\n\n        features_dropout = self.dense_dropout(dense_opcodes_apis_bytes, training=training)\n        output = self.out(features_dropout)\n\n        return output","metadata":{"_uuid":"4dddbf31-2110-4a96-b46f-78c348322321","_cell_guid":"24e4ef7e-6b1f-46ae-9bc7-c50e174b53f7","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-12T03:51:44.34048Z","iopub.execute_input":"2023-06-12T03:51:44.341165Z","iopub.status.idle":"2023-06-12T03:51:44.39263Z","shell.execute_reply.started":"2023-06-12T03:51:44.341127Z","shell.execute_reply":"2023-06-12T03:51:44.391399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _parse_tfrecord_function(example, opcodes_lookup_table, bytes_lookup_table):\n    example_fmt = {\n        'opcodes': tf.io.FixedLenFeature([], tf.string),\n        'bytes': tf.io.FixedLenFeature([], tf.string),\n        'APIs': tf.io.FixedLenFeature([], tf.string),\n        'label': tf.io.FixedLenFeature([], tf.int64)\n        }\n    parsed = tf.io.parse_single_example(example, example_fmt)\n\n    tokenizer = text.WhitespaceTokenizer()\n\n    opcodes_tokens = tokenizer.tokenize(parsed['opcodes'])\n    opcodes_IDs = opcodes_lookup_table.lookup(opcodes_tokens)\n\n    bytes_tokens = tokenizer.tokenize(parsed['bytes'])\n    bytes_IDs = bytes_lookup_table.lookup(bytes_tokens)\n\n    feature_vector = tf.io.decode_raw(parsed['APIs'], tf.float32)\n    return opcodes_IDs, bytes_IDs, feature_vector, parsed['label']\n\n\ndef make_dataset(filepath,\n                 opcodes_lookup_table,\n                 bytes_lookup_table,\n                 SHUFFLE_BUFFER_SIZE=1024,\n                 BATCH_SIZE=32,\n                 EPOCHS=5):\n    dataset = tf.data.TFRecordDataset(filepath)\n    dataset = dataset.shuffle(SHUFFLE_BUFFER_SIZE)\n    dataset = dataset.repeat(EPOCHS)\n    dataset = dataset.map(lambda x: _parse_tfrecord_function(x, opcodes_lookup_table, bytes_lookup_table))\n    dataset = dataset.batch(batch_size=BATCH_SIZE)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2023-06-12T03:51:44.394418Z","iopub.execute_input":"2023-06-12T03:51:44.394767Z","iopub.status.idle":"2023-06-12T03:51:44.41079Z","shell.execute_reply.started":"2023-06-12T03:51:44.394735Z","shell.execute_reply":"2023-06-12T03:51:44.409389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def initialize_TFRecords(tfrecords_filepath, num_tfrecords=10, filename=\"training\"):\n    training_writers = []\n    for i in range(num_tfrecords):\n        training_writers.append(tf.io.TFRecordWriter(tfrecords_filepath + \"{}{}.tfrecords\".format(filename,i)))\n    return training_writers\n\ndef create_lookup_table(vocabulary_mapping, num_oov_buckets):\n    keys = [k for k in vocabulary_mapping.keys()]\n    values = [tf.constant(vocabulary_mapping[k], dtype=tf.int64) for k in keys]\n\n    table = tf.lookup.StaticVocabularyTable(\n        tf.lookup.KeyValueTensorInitializer(\n            keys=keys,\n            values=values\n        ),\n        num_oov_buckets\n    )\n    return table\n\ndef _bytes_feature(values):\n    # Convert list of integers to bytes\n    byte_string = bytes(values)\n    # Create BytesList from bytes\n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=[byte_string]))\n\ndef _int64_feature(value):\n    return tf.train.Feature(int64_list=tf.train.Int64List(value=[value]))\n\ndef load_vocabulary(vocabulary_filepath):\n    \"\"\"\n    It reads and stores in a dictionary-like structure the data from the file passed as argument\n\n    Parameters\n    ----------\n    vocabulary_filepath: str\n        JSON-like file\n\n    Return\n    ------\n    vocabulary_dict: dict\n    \"\"\"\n    with open(vocabulary_filepath, \"r\") as vocab_file:\n        vocabulary_dict = json.load(vocab_file)\n    return vocabulary_dict\n\ndef serialize_mnemonics_example_IDs(mnemonic_IDs, label):\n    \"\"\"\n    Creates a tf.Example message ready to be written to a file\n    :param mnemonics: str -> \"[4,67,109,...,402, 402]\"\n    :param label: int [0,8]\n    :return:\n    \"\"\"\n    feature = {\n        'opcodes': _bytes_feature(np.array(mnemonic_IDs).tostring()),\n        'label': _int64_feature(label)\n    }\n    example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example_proto.SerializeToString()\n\ndef serialize_mnemonics_example(mnemonics, label):\n    \"\"\"\n    Creates a tf.Example message ready to be written to a file\n    :param mnemonics: str -> \"push,pop,...,NONE\"\n    :param label: int [0,8]\n    :return:\n    \"\"\"\n    feature = {\n        'opcodes': _bytes_feature(mnemonics.encode('UTF-8')),\n        'label': _int64_feature(label)\n    }\n    example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example_proto.SerializeToString()\n\ndef serialize_bytes_example(bytes, label):\n    \"\"\"\n    Creates a tf.Example message ready to be written to a file\n    :param bytes: str -> \"00,FF,...,??,NONE\"\n    :param label: int [0,8]\n    :return:\n    \"\"\"\n    feature = {\n        'bytes': _bytes_feature(bytes.encode('UTF-8')),\n        'label': _int64_feature(label)\n    }\n    example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example_proto.SerializeToString()\n\ndef serialize_apis_example(feature_vector, label):\n    feature = {\n        'APIs': _bytes_feature(feature_vector),\n        'label': _int64_feature(label)\n    }\n    example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example_proto.SerializeToString()\n\ndef serialize_hydra_example(opcodes, bytes, apis_values, label):\n    feature = {\n        'opcodes': _bytes_feature(opcodes.encode('UTF-8')),\n        'bytes': _bytes_feature(bytes.encode('UTF-8')),\n        'APIs': _bytes_feature(apis_values),\n        'label': _int64_feature(label)\n    }\n    example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example_proto.SerializeToString()\n\ndef load_parameters(parameters_path):\n    \"\"\"\n    It loads the network parameters\n\n    Parameters\n    ----------\n    parameters_path: str\n        File containing the parameters of the network\n    \"\"\"\n    with open(parameters_path, \"r\") as param_file:\n        params = json.load(param_file)\n    return params\n\nclass MetaPHOR:\n    def __init__(self, asm_filepath):\n        self.asm_filepath = asm_filepath\n        self.vocab = {}\n\n    def extract_windows_api_calls(self):\n        # Define regular expressions for Windows API calls\n        api_regex = re.compile(r'(call|jmp)\\s+(\\w+)(@.*)?$')\n        winapi_regex = re.compile(r'^(A|W|Nt|Zw)[a-zA-Z]+')\n\n        api_calls = set()\n\n        with open(self.asm_filepath, 'r', encoding='KOI8-R') as f:\n            for line in f:\n                match = api_regex.search(line)\n                if match:\n                    api_name = match.group(2)\n                    if winapi_regex.match(api_name):\n                        api_calls.add(api_name)\n\n        return api_calls\n\n    def count_windows_api_calls(self):\n        # Define regular expression for Windows API calls\n        api_regex = re.compile(r'call\\s+(\\w+)')\n\n        api_counts = {\n            'VirtualAlloc': 0,\n            'CreateFile': 0,\n            'ReadFile': 0,\n            'WriteFile': 0,\n            'CloseHandle': 0,\n            'GetModuleHandle': 0,\n            'GetProcAddress': 0,\n            'LoadLibrary': 0,\n            'ExitProcess': 0,\n            'OpenProcess': 0,\n            'CreateProcess': 0,\n            'CreateThread': 0,\n            'RegOpenKeyEx': 0,\n            'RegSetValueEx': 0,\n            'Process32Next': 0,\n            'Process32First': 0,\n            'CreateToolhelp32Snapshot': 0,\n            'LookupPrivilegeValue': 0,\n            'AdjustTokenPrivileges': 0,\n            'VirtualProtect': 0,\n            'WriteProcessMemory': 0,\n            'NtUnmapViewOfSection': 0,\n            'NtCreateSection': 0,\n            'NtMapViewOfSection': 0,\n            'QueueUserAPC': 0,\n            'SuspendThread': 0,\n            'ResumeThread': 0,\n            'CreateRemoteThread': 0,\n            'RtlCreateUserThread': 0,\n            'NtCreateThreadEx': 0,\n            'GetThreadContext': 0,\n            'SetThreadContext': 0\n        }\n\n        with open(self.asm_filepath, 'r', encoding='KOI8-R') as f:\n            try: \n                for line in f:\n                    try: \n                        match = api_regex.search(line)\n                        if match:\n                            api_name = match.group(1)\n                            if api_name in api_counts:\n                                api_counts[api_name] += 1\n                    except:\n                        print(\"Error in line: \", line)\n                        continue\n            except:\n                print(\"Error in file: \", self.asm_filepath)\n                return None\n\n        return list(api_counts.values())\n    \n    def get_hexadecimal_data_as_list(self):\n        hex_data = []\n\n        with open(self.asm_filepath, 'r', encoding='KOI8-R') as asm_file:\n            for line in asm_file:\n                hex_values = re.findall(r'\\b[0-9A-Fa-f]{2}\\b', line)\n                hex_data.extend(hex_values)\n\n        return hex_data\n\n    def get_opcodes_data_as_list(self, vocab_mapping):\n        opcodes = []\n\n        with open(self.asm_filepath, 'r', encoding='KOI8-R') as asm_file:\n            for line in asm_file:\n                opcode_match = re.findall(r'\\b[A-Za-z]+\\b', line)\n                opcodes.extend(opcode_match)\n#                 if opcode_match:\n#                     opcodes.append(opcode_match.group())\n\n        return opcodes","metadata":{"execution":{"iopub.status.busy":"2023-06-12T03:51:44.412487Z","iopub.execute_input":"2023-06-12T03:51:44.413413Z","iopub.status.idle":"2023-06-12T03:51:44.456368Z","shell.execute_reply.started":"2023-06-12T03:51:44.413369Z","shell.execute_reply":"2023-06-12T03:51:44.455041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_train(folder: str, filenames: list, logs: bool = False):\n    directory_command = f'\"-o{folder}\"'\n    filenames_command = \" \".join(filenames)  # Use comma as separator\n    if logs:\n        print(filenames_command)\n    command = f\"7z x /kaggle/input/malware-classification/test.7z {directory_command} {filenames_command} -r\"\n    command = command + ' >/dev/null 2>&1'  # Done to hide console output\n    if logs:\n        print(command)\n        print('Started extracting')\n    os.system(command)\n    if logs:\n        print('Finished extracting')\n\ndef dataset_to_tfrecords(pe_filepath, tfrecords_filepath,\n                         opcodes_vocabulary_mapping_filepath,\n                         bytes_vocabulary_mapping_filepath,\n                         max_mnemonics=2000000, max_bytes=2000000, batch_size=1200):\n\n    tfwriter = tf.io.TFRecordWriter(tfrecords_filepath)\n    opcodes_vocabulary_mapping = load_vocabulary(opcodes_vocabulary_mapping_filepath)\n    bytes_vocabulary_mapping = load_vocabulary(bytes_vocabulary_mapping_filepath)\n\n    line = 0\n    files_processed = 0\n\n    with py7zr.SevenZipFile('/kaggle/input/malware-classification/test.7z', mode='r') as z:\n        file_infos = z.getnames()\n        total_files = len(file_infos)\n\n        while files_processed < total_files:\n            batch_files = file_infos[files_processed:files_processed+batch_size]\n            extract_train(folder='/kaggle/working/tmp', filenames=batch_files)\n\n            for file_info in batch_files:\n                match = re.search(r\"/([^/]+)$\", file_info)\n                if match:\n                    file_name = match.group(1)\n\n                    print(\"{};{}\".format(line, file_name))\n                    metaPHOR = MetaPHOR(pe_filepath + \"/\" + file_name)\n\n                    # Extract opcodes\n                    opcodes = metaPHOR.get_opcodes_data_as_list(opcodes_vocabulary_mapping)\n                    if len(opcodes) < max_mnemonics:\n                        while len(opcodes) < max_mnemonics:\n                            opcodes.append(\"PAD\")\n                    else:\n                        opcodes = opcodes[:max_mnemonics]\n                    raw_mnemonics = \" \".join(opcodes)\n\n                    # Extract bytes\n                    bytes_sequence = metaPHOR.get_hexadecimal_data_as_list()\n                    for i in range(len(bytes_sequence)):\n                        if bytes_sequence[i] not in bytes_vocabulary_mapping.keys():\n                            bytes_sequence[i] = \"UNK\"\n                    if len(bytes_sequence) < max_bytes:\n                        while len(bytes_sequence) < max_bytes:\n                            bytes_sequence.append(\"PAD\")\n                    else:\n                        bytes_sequence = bytes_sequence[:max_bytes]\n                    raw_bytes_sequence = \" \".join(bytes_sequence)\n\n                    # Extract APIs\n                    feature_vector = metaPHOR.count_windows_api_calls()\n\n                    example = serialize_hydra_example(raw_mnemonics,\n                                                      raw_bytes_sequence,\n                                                      feature_vector,\n                                                      0)\n                    tfwriter.write(example)\n                    os.remove(pe_filepath + \"/\" + file_name)\n                    line += 1\n\n            files_processed += batch_size\n\n    tfwriter.close()\ndataset_to_tfrecords(pe_filepath='/kaggle/working/tmp/test', \n                     tfrecords_filepath='/kaggle/working/test.tfrecords', \n                     opcodes_vocabulary_mapping_filepath='/kaggle/input/big2015-tfrecords/opcodes.json',\n                     bytes_vocabulary_mapping_filepath='/kaggle/input/big2015-tfrecords/bytes.json',\n                     max_mnemonics=50000, max_bytes=50000)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-06-12T03:51:44.458352Z","iopub.execute_input":"2023-06-12T03:51:44.458709Z","iopub.status.idle":"2023-06-12T06:00:57.166245Z","shell.execute_reply.started":"2023-06-12T03:51:44.458675Z","shell.execute_reply":"2023-06-12T06:00:57.16047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport tensorflow as tf\nimport py7zr\n\ndef dataset_to_tfrecords(pe_filepath, tfrecords_filepath,\n                         opcodes_vocabulary_mapping_filepath,\n                         bytes_vocabulary_mapping_filepath,\n                         max_mnemonics=2000000, max_bytes=2000000, batch_size=100):\n\n    tfwriter = tf.io.TFRecordWriter(tfrecords_filepath)                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                  \n    opcodes_vocabulary_mapping = load_vocabulary(opcodes_vocabulary_mapping_filepath)\n    bytes_vocabulary_mapping = load_vocabulary(bytes_vocabulary_mapping_filepath)\n\n    line = 0\n    files_processed = 0\n\n    with py7zr.SevenZipFile('/kaggle/input/malware-classification/test.7z', mode='r') as z:\n        file_infos = z.getnames()\n        total_files = len(file_infos)\n\n        while files_processed < total_files:\n            batch_files = file_infos[files_processed:files_processed+batch_size]\n\n            for file_info in batch_files:\n                match = re.search(r\"/([^/]+)$\", file_info)\n                if match:\n                    file_name = match.group(1)\n                    extract_train(folder='/kaggle/working/tmp', filenames=[file_name])\n\n                    print(\"{};{}\".format(line, file_name))\n                    metaPHOR = MetaPHOR(pe_filepath + \"/\" + file_name)\n\n                    \n            files_processed += batch_size\n\n    tfwriter.close()","metadata":{"execution":{"iopub.status.busy":"2023-06-12T06:00:57.170693Z","iopub.status.idle":"2023-06-12T06:00:57.171367Z","shell.execute_reply.started":"2023-06-12T06:00:57.171086Z","shell.execute_reply":"2023-06-12T06:00:57.171123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class HYDRA_Training():\n    model = None\n    tr_tfrecord = None\n    val_tfrecord = None\n    parameters = None\n    opcodes_vocabulary_mapping_filepath = None\n    bytes_vocabulary_mapping_filepath = None\n    test_tfrecord = None\n    opcodes_vocabulary_mapping = None\n    bytes_vocabulary_mapping = None\n    opcode_lookup_table = None\n    bytes_lookup_table = None\n    loss_func = None\n    optimizer = None\n    accuracy = None\n    epoch_loss_avg = None\n    epoch_accuracy = None\n    val_epoch_loss_avg = None\n    val_epoch_accuracy = None\n    initial_loss = 10\n\n    checkpoint_path = f\"\"\n    validation_loss_results = []\n    validation_accuracy_results = []\n\n    train_loss_results = []\n    train_accuracy_results = []\n\n    def __init__(self,\n                 model = \"Hydra\",\n                 tr_tfrecord = \"/kaggle/input/big2015-tfrecords/train.tfrecords\",\n                 val_tfrecord = \"/kaggle/input/big2015-tfrecords/val.tfrecords\",\n                 parameters = \"/kaggle/input/big2015-tfrecords/hydra_parameters.json\",\n                 opcodes_vocabulary_mapping_filepath = \"/kaggle/input/big2015-tfrecords/opcodes.json\",\n                 bytes_vocabulary_mapping_filepath = \"/kaggle/input/big2015-tfrecords/bytes.json\",\n                 test_tfrecord = \"/kaggle/input/big2015-tfrecords/test.tfrecords\"):\n        self.model = model\n        self.tr_tfrecord = tr_tfrecord\n        self.val_tfrecord = val_tfrecord\n        self.parameters = parameters\n        self.opcodes_vocabulary_mapping_filepath = opcodes_vocabulary_mapping_filepath\n        self.bytes_vocabulary_mapping_filepath = bytes_vocabulary_mapping_filepath\n        self.test_tfrecord = test_tfrecord\n        \n    \n        # num_epochs = self.parameters['epochs']\n        self.num_epochs = 10\n        self.batch_size = 128\n        self.learning_rate = 0.0005\n        self.initial_loss = 10.0\n        \n    def init_model(self):\n        print(\"TensorFlow version: {}\".format(tf.__version__))\n        print(\"Eager execution: {}\".format(tf.executing_eagerly()))\n        print(\"Num GPUs Available: \", len(tf.config.experimental.list_physical_devices('GPU')))\n\n        gpus = tf.config.experimental.list_physical_devices('GPU')\n        print(gpus)\n        \n        if gpus:\n            try:\n                tf.config.experimental.set_visible_devices(gpus[0], 'GPU')\n                logical_gpus = tf.config.experimental.list_logical_devices('GPU')\n                print(len(gpus), \"Physical GPUs,\", len(logical_gpus), \"Logical GPU\")\n            except RuntimeError as e:\n                # Visible devices must be set before GPUs have been initialized\n                # print(e)\n                pass\n\n        #Load vocabulary and create lookup table\n        self.opcodes_vocabulary_mapping = load_vocabulary(self.opcodes_vocabulary_mapping_filepath)\n        self.bytes_vocabulary_mapping = load_vocabulary(self.bytes_vocabulary_mapping_filepath)\n\n        self.opcodes_lookup_table = create_lookup_table(self.opcodes_vocabulary_mapping, 1)\n        self.bytes_lookup_table = create_lookup_table(self.bytes_vocabulary_mapping, 1)\n\n        # Load parameters of the model\n        self.parameters = load_parameters(self.parameters)\n\n        # Specify GPU\n        if \"gpu\" in self.parameters.keys():\n            os.environ[\"CUDA_VISIBLE_DEVICES\"] = self.parameters[\"gpu\"]\n\n\n        self.model = HYDRA(self.parameters)\n\n        self.loss_func = tf.keras.losses.SparseCategoricalCrossentropy()\n        self.accuracy = tf.keras.metrics.SparseCategoricalAccuracy()\n        self.optimizer = tf.keras.optimizers.Adam(learning_rate=self.learning_rate)\n\n    def load_weights(self):\n        if os.path.isdir(model_path):\n            print(\"LOADING WEIGHTS!!!!\")\n            latest = tf.train.latest_checkpoint(model_path)\n            self.model.load_weights(latest)\n            print(\"LOADED WEIGHTS!!!!\")\n        \n    def compile_model(self):\n        self.checkpoint_path = f\"{model_path}{max_fcount()[0]}.ckpt\"\n        print(self.checkpoint_path)\n        d_train = make_dataset(self.tr_tfrecord,\n                                    self.opcodes_lookup_table,\n                                    self.bytes_lookup_table,\n                                    self.parameters['buffer_size'],\n                                    self.batch_size,\n                                    1)\n        \n        for step, (opcodes, bytes, apis, y) in enumerate(d_train):\n            loss, y_ = self.train_loop(opcodes, bytes, apis, y, True)\n        \n            with tf.GradientTape() as tape:\n                predictions = self.model(opcodes, bytes, apis, True)\n                loss = self.loss_func(y, predictions)\n\n            gradients = tape.gradient(loss, self.model.trainable_variables)\n            break\n            \n        self.test()\n\n        return self.model.get_weights()\n\n    def load_weights(self, weights):\n        return self.model.load_weights(weights)\n    \n    def train_loop(self, opcodes, bytes, apis, labels, training=False, loss_func=None, optimizer=None):\n        # Define the GradientTape context\n        with tf.GradientTape() as tape:\n            # Get the probabilities\n            predictions = self.model(opcodes, bytes, apis, training)\n            #labels = tf.dtypes.cast(labels, tf.float32)\n            # Calculate the loss\n            loss = self.loss_func(labels, predictions)\n        # Get the gradients\n        gradients = tape.gradient(loss, self.model.trainable_variables)\n        # Update the weights\n        self.optimizer.apply_gradients(zip(gradients, self.model.trainable_variables))\n        return loss, predictions\n    \n    def train(self, init=False):\n        # Training loop\n        # 1/ Iterate each epoch. An epoch is one pass through the dataset\n        # 2/ Whithin an epoch, iterate over each example in the training Dataset.\n        # 3/ Calculate model's loss and gradients\n        # 4/ Use an optimizer to update the model's variables\n        # 5/ Keep track of stats and repeat\n\n        for epoch in range(self.num_epochs):\n            print(\"#### Current epoch: {}\".format(epoch))\n            path,_ = max_fcount()\n            self.checkpoint_path = model_path + path + \".ckpt\"\n            \n            #self.checkpoint_path = model_path + \"model_001.ckpt\"\n            #checkpoint_dir = os.path.dirname(checkpoint_path)\n\n            d_train = make_dataset(self.tr_tfrecord,\n                                    self.opcodes_lookup_table,\n                                    self.bytes_lookup_table,\n                                    self.parameters['buffer_size'],\n                                    self.batch_size,\n                                    1)\n            \n            d_val = make_dataset(self.val_tfrecord,\n                                    self.opcodes_lookup_table,\n                                    self.bytes_lookup_table,\n                                    1024,\n                                    1,\n                                    1)\n\n\n            # Training metrics\n            self.epoch_loss_avg = tf.keras.metrics.Mean()\n            self.epoch_accuracy = tf.keras.metrics.SparseCategoricalAccuracy()\n            # Validation metrics\n            self.val_epoch_loss_avg = tf.keras.metrics.Mean()\n            self.val_epoch_accuracy = tf.keras.metrics.SparseCategoricalAccuracy()\n            tr_step = 0\n\n            # Training loop\n            for step, (opcodes, bytes, apis, y) in enumerate(d_train):\n                #print(\"Input: {}\".format(x))\n                print(\"#### Step {} of epoch {}\".format(step, epoch))\n\n                loss, y_ = self.train_loop(opcodes, bytes, apis, y, True)\n\n                # Track progress\n                self.epoch_loss_avg(loss)\n                self.epoch_accuracy(y, y_)\n                print(\"#### Iteration step: {}; Loss: {:.3f}, Accuracy: {:.3%}\".format(tr_step,\n                                                                                    self.epoch_loss_avg.result(),\n                                                                                    self.epoch_accuracy.result()))\n                if (init == True):\n                    break\n\n                tr_step += 1\n\n            # End epoch\n            self.train_loss_results.append(self.epoch_loss_avg.result())\n            self.train_accuracy_results.append(self.epoch_accuracy.result())\n\n            # Run a validation loop at the end of each epoch.\n            self.validation(d_val, epoch)\n            if (init == True):\n                break\n\n        # Training is done!\n        print(\"Training is done!\")\n        # Test the model\n        self.test()\n        # Return weights of the model           \n        self.model.save_weights(self.checkpoint_path) # Save only the weights\n        return self.model.get_weights()\n\n    def validation(self, d_val, epoch):\n        for opcodes_batch_val, bytes_batch_val, apis_batch_val, y_batch_val in d_val:\n            val_logits = self.model(opcodes_batch_val, bytes_batch_val, apis_batch_val, False)\n            val_loss = self.loss_func(y_batch_val, val_logits)\n\n            # Update metrics\n            self.val_epoch_loss_avg(val_loss)\n            self.val_epoch_accuracy(y_batch_val, val_logits)\n\n        val_acc = self.val_epoch_accuracy.result()\n        val_loss = self.val_epoch_loss_avg.result()\n        print('#### Epoch: {}; Validation loss {}; acc: {}'.format(epoch, val_loss, val_acc))\n\n        self.validation_loss_results.append(val_loss)\n        self.validation_accuracy_results.append(val_acc)\n\n        if float(val_loss) < self.initial_loss:\n            self.initial_loss = float(val_loss)\n            self.model.save_weights(self.checkpoint_path) # Save only the weights\n\n    def test(self):\n        # Load the model\n        self.model.load_weights(self.checkpoint_path)\n        test_epoch_loss_avg = tf.keras.metrics.Mean()\n        test_epoch_accuracy = tf.keras.metrics.SparseCategoricalAccuracy()\n\n        y_actual_test = []\n        y_pred_test = []\n        # Evaluate model on the test set\n        d_test = make_dataset(self.test_tfrecord,\n                            self.opcodes_lookup_table,\n                            self.bytes_lookup_table,\n                            1,\n                            1,\n                            1)\n        \n        file_asm_name = []\n        with py7zr.SevenZipFile('/kaggle/input/malware-classification/test.7z', mode='r') as z:\n            file_infos = z.getnames()\n            asm_files = [filename for filename in file_infos if filename.endswith(\".asm\")]\n            # Split the string by '/'\n            for filename in asm_files:\n                split_parts = filename.split('/')\n                # Get the last part of the split string\n                last_part = split_parts[-1]\n                # Remove the file extension\n                filename_without_extension = last_part.split('.')[0]\n                file_asm_name.append(filename_without_extension)\n                print(filename_without_extension)\n\n        index = 0\n        df = pd.DataFrame({\"Id\": filename_without_extension})\n        \n        try:\n            for opcodes_batch_test, bytes_batch_test, apis_batch_test, y_batch_test in d_test:\n                test_logits = self.model(opcodes_batch_test, bytes_batch_test, apis_batch_test, False)\n\n                prediction = test_logits.numpy().tolist()[0]\n\n                for i in range(9):\n                    column_name = f\"Prediction{i+1}\"\n                    df[column_name] = [prediction[i] for prediction in predictions]\n\n                        \n                test_loss = self.loss_func(y_batch_test, test_logits)\n\n                # For the confusion matrix\n                y_pred = tf.argmax(test_logits, axis=-1)\n                y_pred_test.extend(y_pred)\n                y_actual_test.extend(y_batch_test)\n\n                # Update metrics\n                test_epoch_loss_avg(test_loss)\n                test_epoch_accuracy(y_batch_test, test_logits)\n        except:\n            print(\"Error\")\n\n        df.to_csv(\"predictions.csv\", index=False)\n#         test_acc = test_epoch_accuracy.result()\n#         test_loss = test_epoch_loss_avg.result()\n#         print('Test loss {}; acc: {}'.format(test_loss, test_acc))\n\n#         cm = confusion_matrix(y_actual_test, y_pred_test)\n#         print(\"Confusion Matrix:\\n {}\".format(cm))\n#         return test_acc, test_loss","metadata":{"execution":{"iopub.status.busy":"2023-06-12T06:00:57.173772Z","iopub.status.idle":"2023-06-12T06:00:57.174298Z","shell.execute_reply.started":"2023-06-12T06:00:57.174055Z","shell.execute_reply":"2023-06-12T06:00:57.174085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_train(folder: str, filenames: list, logs: bool = False):\n    directory_command = f'\"-o{folder}\"'\n    filenames_command = \"\".join(filenames)  # Use comma as separator\n    if logs:\n        print(filenames_command)\n    command = f\"7z x /kaggle/input/malware-classification/train.7z {directory_command} {filenames_command} -r\"\n    command = command + ' >/dev/null 2>&1'  # Done to hide console output\n    if logs:\n        print(command)\n        print('Started extracting')\n    os.system(command)\n    if logs:\n        print('Finished extracting')","metadata":{"execution":{"iopub.status.busy":"2023-06-12T06:00:57.17608Z","iopub.status.idle":"2023-06-12T06:00:57.176532Z","shell.execute_reply.started":"2023-06-12T06:00:57.17631Z","shell.execute_reply":"2023-06-12T06:00:57.176334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport tensorflow as tf\n\ndef dataset_to_tfrecords(pe_filepath, tfrecords_filepath, labels_df,\n                         opcodes_vocabulary_mapping_filepath,\n                         bytes_vocabulary_mapping_filepath,\n                         max_mnemonics=2000000, max_bytes=2000000):\n\n    tfwriter = tf.io.TFRecordWriter(tfrecords_filepath)\n    opcodes_vocabulary_mapping = load_vocabulary(opcodes_vocabulary_mapping_filepath)\n    bytes_vocabulary_mapping = load_vocabulary(bytes_vocabulary_mapping_filepath)\n\n    line = 0\n    for j, row in labels_df.iterrows():\n        filenames = f\"{row['Id']}.asm\"\n        extract_train(folder='/kaggle/working/tmp', filenames=filenames)\n\n        \n        print(\"{};{}\".format(line, row['Id']))\n        metaPHOR = MetaPHOR(pe_filepath + \"/\" + row['Id'] + \".asm\")\n\n        # Extract opcodes\n        opcodes = metaPHOR.get_opcodes_data_as_list(opcodes_vocabulary_mapping)\n        if len(opcodes) < max_mnemonics:\n            while len(opcodes) < max_mnemonics:\n                opcodes.append(\"PAD\")\n        else:\n            opcodes = opcodes[:max_mnemonics]\n        raw_mnemonics = \" \".join(opcodes)\n\n        # Extract bytes\n        bytes_sequence = metaPHOR.get_hexadecimal_data_as_list()\n        for i in range(len(bytes_sequence)):\n            if bytes_sequence[i] not in bytes_vocabulary_mapping.keys():\n                bytes_sequence[i] = \"UNK\"\n        if len(bytes_sequence) < max_bytes:\n            while len(bytes_sequence) < max_bytes:\n                bytes_sequence.append(\"PAD\")\n        else:\n            bytes_sequence = bytes_sequence[:max_bytes]\n        raw_bytes_sequence = \" \".join(bytes_sequence)\n\n        # Extract APIs\n        feature_vector = metaPHOR.count_windows_api_calls()\n\n        example = serialize_hydra_example(raw_mnemonics,\n                                          raw_bytes_sequence,\n                                          feature_vector,\n                                          int(row['Class']) - 1)\n        tfwriter.write(example)\n        os.remove(pe_filepath + \"/\" + row['Id'] + \".asm\")\n        line += 1","metadata":{"execution":{"iopub.status.busy":"2023-06-12T06:00:57.179299Z","iopub.status.idle":"2023-06-12T06:00:57.179784Z","shell.execute_reply.started":"2023-06-12T06:00:57.179548Z","shell.execute_reply":"2023-06-12T06:00:57.17958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dataset_to_tfrecords(pe_filepath='/kaggle/working/tmp/train', \n#                      tfrecords_filepath='/kaggle/working/train.tfrecords', \n#                      labels_df=labels,\n#                      opcodes_vocabulary_mapping_filepath='/kaggle/input/big2015-tfrecords/opcodes.json',\n#                      bytes_vocabulary_mapping_filepath='/kaggle/input/big2015-tfrecords/bytes.json',\n#                      max_mnemonics=50000, max_bytes=50000)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-06-12T06:00:57.18159Z","iopub.status.idle":"2023-06-12T06:00:57.182087Z","shell.execute_reply.started":"2023-06-12T06:00:57.181817Z","shell.execute_reply":"2023-06-12T06:00:57.181864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = HYDRA_Training(tr_tfrecord='/kaggle/input/big2015-tfrecords/train (1).tfrecords',\n                      val_tfrecord='/kaggle/input/big2015-tfrecords/train (9).tfrecords',\n                      test_tfrecord='/kaggle/input/big2015-tfrecords/test (1).tfrecords')\nmodel.init_model()\n\nHEADER_SIZE = 10\n\ndef recv(soc, buffer_size=1024, recv_timeout=10):\n    try:\n        header = soc.recv(4)\n\n        # Unpack the header to get the length of the serialized data\n        data_length = struct.unpack(\"!I\", header)[0]\n        print(\"Data Length:\", data_length)\n\n        # Receive the serialized data\n        data = b''\n        while len(data) < data_length:\n            chunk = soc.recv(data_length - len(data))\n            if not chunk:\n                break\n            data += chunk\n        print(\"All data ({data_len} bytes) Received from the Server.\".format(data_len=len(data)))\n\n    except Exception as e:\n        print(\"Error:\", e)\n    \n    try:\n        print(\"Decoding the Client's Data.\\n\")\n        received_data = pickle.loads(data)\n        \n        with open('./model.zip', 'wb') as file:\n            file.write(received_data[\"data\"])\n\n        with ZipFile('./model.zip', 'r') as zip_ref:\n            zip_ref.extractall('./')\n\n    except BaseException as e:\n        print(\"Error Decoding the Client's Data: {msg}.\\n\".format(msg=e))\n\n        return None, 1\n\n    return received_data, 1\n\nsoc = socket.socket(family=socket.AF_INET, type=socket.SOCK_STREAM)\nprint(\"Socket Created.\\n\")\n\nconnected = False\nwhile not connected:\n    try:\n        soc.connect((\"43.207.224.126\", 9999))\n        print(\"Successful Connection to the Server.\\n\")\n        connected = True\n    except BaseException as e:\n        print(\"Error Connecting to the Server: {msg}\".format(msg=e))\n\nsubject = \"echo\"\nGANN_instance = None\n\nwhile True:\n    try:\n        data = {\"subject\": subject, \"data\": GANN_instance}\n        data_byte = pickle.dumps(data)\n\n        data_length = len(data_byte)\n        header = struct.pack(\"!I\", data_length)\n\n        # Send the header and serialized data\n        soc.sendall(header)\n        soc.sendall(data_byte)\n\n        print(\"Receiving Reply from the Server.\")\n        received_data, status = recv(soc=soc, \n                                     buffer_size=1024, \n                                     recv_timeout=10)\n        if status == 0:\n            print(\"Nothing Received from the Server.\")\n            break\n        else:\n            print(\"./\")\n\n        subject = received_data[\"subject\"]\n        if subject == \"model\":\n            print(\"The server sent the model to the client.\")\n        elif subject == \"done\":\n            print(\"The server said the model is trained successfully and no need for further updates its parameters.\")\n            break\n        else:\n            print(\"Unrecognized message type.\")\n            break\n\n        path,_ = max_fcount()\n\n        weights = model.load_weights(weights=model_path + path + \".ckpt\")\n        weights = model.compile_model()\n        weights = model.train()\n        np.save('/kaggle/working/models/model.npy', weights)\n\n        subject = \"model\"\n        print(\"Sending the Updated Model to the Server.\\n\")\n        GANN_instance = weights\n        print(type(weights))\n    except OSError as ose:\n        if ose.errno == errno.EPIPE:\n            print(\"The pipe is broken. Trying to connect again!\")\n            try:\n                soc = socket.socket(family=socket.AF_INET, type=socket.SOCK_STREAM)\n                soc.connect((\"43.207.224.126\", 9999))\n                print(\"Successful Connection to the Server.\\n\")\n            except OSError as ose:\n                if ose.errno == errno.EPIPE:\n                    print(\"Error Connecting to the Server: {msg}\".format(msg=e))\n                    soc.close()\n                    print(\"Socket Closed.\")\n            \n    except Exception as e:\n        subject = \"echo\"\n        print(f\"Error with server. Try again!\\nError: {e}\")\n\nsoc.close()\nprint(\"Socket Closed.\\n\")","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-06-12T06:00:57.184517Z","iopub.status.idle":"2023-06-12T06:00:57.185004Z","shell.execute_reply.started":"2023-06-12T06:00:57.184745Z","shell.execute_reply":"2023-06-12T06:00:57.18477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import shutil\n# shutil.rmtree('/kaggle/working/tmp')","metadata":{"execution":{"iopub.status.busy":"2023-06-12T06:00:57.18656Z","iopub.status.idle":"2023-06-12T06:00:57.187036Z","shell.execute_reply.started":"2023-06-12T06:00:57.18678Z","shell.execute_reply":"2023-06-12T06:00:57.186803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def test(model, data):\n#     # Load the model\n#     model.load_weights(self.checkpoint_path)\n#     test_epoch_loss_avg = tf.keras.metrics.Mean()\n#     test_epoch_accuracy = tf.keras.metrics.SparseCategoricalAccuracy()\n\n#     y_actual_test = []\n#     y_pred_test = []\n#     # Evaluate model on the test set\n#     d_test = make_dataset(self.test_tfrecord,\n#                         self.opcodes_lookup_table,\n#                         self.bytes_lookup_table,\n#                         1,\n#                         1,\n#                         1)\n\n#     for opcodes_batch_test, bytes_batch_test, apis_batch_test, y_batch_test in d_test:\n#         test_logits = self.model(opcodes_batch_test, bytes_batch_test, apis_batch_test, False)\n#         test_loss = self.loss_func(y_batch_test, test_logits)\n\n#         # For the confusion matrix\n#         y_pred = tf.argmax(test_logits, axis=-1)\n#         y_pred_test.extend(y_pred)\n#         y_actual_test.extend(y_batch_test)   \n        \n\n# model = HYDRA_Training(tr_tfrecord='/kaggle/input/big2015-tfrecords/train (1).tfrecords',\n#                       val_tfrecord='/kaggle/input/big2015-tfrecords/train (9).tfrecords',\n#                       test_tfrecord='/kaggle/input/big2015-tfrecords/train (10).tfrecords')\n\n# weights = model.load_weights(weights=model_path + path + \".ckpt\")\n# weights = model.compile_model()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-12T06:00:57.188615Z","iopub.status.idle":"2023-06-12T06:00:57.189089Z","shell.execute_reply.started":"2023-06-12T06:00:57.18885Z","shell.execute_reply":"2023-06-12T06:00:57.188881Z"},"trusted":true},"execution_count":null,"outputs":[]}]}