{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":1997418,"sourceType":"datasetVersion","datasetId":1184281},{"sourceId":5963602,"sourceType":"datasetVersion","datasetId":3348292},{"sourceId":8906530,"sourceType":"datasetVersion","datasetId":5355053}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#!pip install py7zr","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.058943Z","iopub.status.idle":"2024-07-08T13:38:29.059783Z","shell.execute_reply.started":"2024-07-08T13:38:29.059480Z","shell.execute_reply":"2024-07-08T13:38:29.059504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport py7zr as unzip\nimport os\nimport re\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, Dense\nfrom tensorflow.keras.optimizers import Adam","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.061263Z","iopub.status.idle":"2024-07-08T13:38:29.062041Z","shell.execute_reply.started":"2024-07-08T13:38:29.061734Z","shell.execute_reply":"2024-07-08T13:38:29.061759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#file_path = '/kaggle/input/malware-classification/dataSample.7z'\nextract_path = '/kaggle/working/extracted_files/'\nif not os.path.exists(extract_path):\n    os.makedirs(extract_path)","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.063595Z","iopub.status.idle":"2024-07-08T13:38:29.064387Z","shell.execute_reply.started":"2024-07-08T13:38:29.064081Z","shell.execute_reply":"2024-07-08T13:38:29.064109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#with unzip.SevenZipFile(file_path, mode='r') as z:\n    z.extractall(path=extract_path)","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.065962Z","iopub.status.idle":"2024-07-08T13:38:29.066796Z","shell.execute_reply.started":"2024-07-08T13:38:29.066497Z","shell.execute_reply":"2024-07-08T13:38:29.066540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"byte_data_path='/kaggle/working/extracted_files/0ACDbR5M3ZhBJajygTuf.bytes'\nasm_file_path='/kaggle/working/extracted_files/0ACDbR5M3ZhBJajygTuf.asm'","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.068382Z","iopub.status.idle":"2024-07-08T13:38:29.069263Z","shell.execute_reply.started":"2024-07-08T13:38:29.068974Z","shell.execute_reply":"2024-07-08T13:38:29.068999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(asm_file_path, 'r', encoding='latin1') as file:\n    lines = file.readlines()","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.070837Z","iopub.status.idle":"2024-07-08T13:38:29.071682Z","shell.execute_reply.started":"2024-07-08T13:38:29.071384Z","shell.execute_reply":"2024-07-08T13:38:29.071408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" hex_data = []\n\nwith open(asm_file_path, 'r', encoding='KOI8-R') as asm_file:\n    for line in asm_file:\n        hex_values = re.findall(r'\\b[0-9A-Fa-f]{2}\\b', line)\n        hex_data.extend(hex_values)","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.073267Z","iopub.status.idle":"2024-07-08T13:38:29.074138Z","shell.execute_reply.started":"2024-07-08T13:38:29.073829Z","shell.execute_reply":"2024-07-08T13:38:29.073855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"opcodes = []\nwith open(asm_file_path, 'r', encoding='KOI8-R') as asm_file:\n    for line in asm_file:\n        opcode_match = re.findall(r'\\b[A-Za-z]+\\b', line) # 30 is to skip the address and hex data\n        opcodes.extend(opcode_match)","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.075727Z","iopub.status.idle":"2024-07-08T13:38:29.076703Z","shell.execute_reply.started":"2024-07-08T13:38:29.076388Z","shell.execute_reply":"2024-07-08T13:38:29.076414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_hexadecimal_data_as_list(self):\n        hex_data = []\n\n        with open(self.asm_filepath, 'r', encoding='KOI8-R') as asm_file:\n            for line in asm_file:\n                hex_values = re.findall(r'\\b[0-9A-Fa-f]{2}\\b', line)\n                hex_data.extend(hex_values)\n\n        return hex_data","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.078256Z","iopub.status.idle":"2024-07-08T13:38:29.079033Z","shell.execute_reply.started":"2024-07-08T13:38:29.078805Z","shell.execute_reply":"2024-07-08T13:38:29.078829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MetaPHOR:\n    def __init__(self, asm_filepath):\n        self.asm_filepath = asm_filepath\n        self.vocab = {}\n\n    def extract_windows_api_calls(self):\n        # Define regular expressions for Windows API calls\n        api_regex = re.compile(r'(call|jmp)\\s+(\\w+)(@.*)?$')\n        winapi_regex = re.compile(r'^(A|W|Nt|Zw)[a-zA-Z]+')\n\n        api_calls = set()\n\n        with open(self.asm_filepath, 'r', encoding='KOI8-R') as f:\n            for line in f:\n                match = api_regex.search(line)\n                if match:\n                    api_name = match.group(2)\n                    if winapi_regex.match(api_name):\n                        api_calls.add(api_name)\n\n        return api_calls\n\n    def count_windows_api_calls(self):\n        # Define regular expression for Windows API calls\n        api_regex = re.compile(r'call\\s+(\\w+)')\n\n        api_counts = {api: 0 for api in ['VirtualAlloc', 'CreateFile', 'ReadFile', 'WriteFile', 'CloseHandle', 'GetModuleHandle', 'GetProcAddress', 'LoadLibrary', 'ExitProcess', 'OpenProcess', 'CreateProcess', 'CreateThread', 'RegOpenKeyEx', 'RegSetValueEx', 'Process32Next', 'Process32First', 'CreateToolhelp32Snapshot', 'LookupPrivilegeValue', 'AdjustTokenPrivileges', 'VirtualProtect', 'WriteProcessMemory', 'NtUnmapViewOfSection', 'NtCreateSection', 'NtMapViewOfSection', 'QueueUserAPC', 'SuspendThread', 'ResumeThread', 'CreateRemoteThread', 'RtlCreateUserThread', 'NtCreateThreadEx', 'GetThreadContext', 'SetThreadContext']}\n\n        with open(self.asm_filepath, 'r', encoding='KOI8-R') as f:\n            try: \n                for line in f:\n                    try: \n                        match = api_regex.search(line)\n                        if match:\n                            api_name = match.group(1)\n                            if api_name in api_counts:\n                                api_counts[api_name] += 1\n                    except:\n                        print(\"Error in line: \", line)\n                        continue\n            except:\n                print(\"Error in file: \", self.asm_filepath)\n                return None\n\n        return list(api_counts.values())\n    \n    def get_hexadecimal_data_as_list(self):\n        hex_data = []\n\n        with open(self.asm_filepath, 'r', encoding='KOI8-R') as asm_file:\n            for line in asm_file:\n                hex_values = re.findall(r'\\b[0-9A-Fa-f]{2}\\b', line)\n                hex_data.extend(hex_values)\n\n        return hex_data\n\n    def get_opcodes_data_as_list(self, vocab_mapping):\n        opcodes = []\n\n        with open(self.asm_filepath, 'r', encoding='KOI8-R') as asm_file:\n            for line in asm_file:\n                opcode_match = re.findall(r'\\b[A-Za-z]+\\b', line) # 30 is to skip the address and hex data\n                opcodes.extend(opcode_match)\n                \n        return opcodes","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.080623Z","iopub.status.idle":"2024-07-08T13:38:29.081593Z","shell.execute_reply.started":"2024-07-08T13:38:29.081183Z","shell.execute_reply":"2024-07-08T13:38:29.081208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfrecord_file = '/kaggle/input/hydra-dataset/test.tfrecords'","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.083656Z","iopub.status.idle":"2024-07-08T13:38:29.084486Z","shell.execute_reply.started":"2024-07-08T13:38:29.084203Z","shell.execute_reply":"2024-07-08T13:38:29.084228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_dataset = tf.data.TFRecordDataset(tfrecord_file)","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.086028Z","iopub.status.idle":"2024-07-08T13:38:29.086858Z","shell.execute_reply.started":"2024-07-08T13:38:29.086577Z","shell.execute_reply":"2024-07-08T13:38:29.086602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dont know Shit about this tfrecord","metadata":{}},{"cell_type":"code","source":"tfrecord_file = '/kaggle/input/hydra-dataset/train-001.tfrecords'\n\n# Create a TFRecordDataset\nraw_dataset = tf.data.TFRecordDataset(tfrecord_file)\n\n# Function to parse a single record and return feature keys\ndef get_feature_keys(raw_record):\n    example = tf.train.Example()\n    example.ParseFromString(raw_record.numpy())\n    return list(example.features.feature.keys())\n\n# Extract feature keys from the first record\nfor raw_record in raw_dataset.take(1):\n    feature_keys = get_feature_keys(raw_record)\n    print(\"Feature keys:\", feature_keys)","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.088395Z","iopub.status.idle":"2024-07-08T13:38:29.089246Z","shell.execute_reply.started":"2024-07-08T13:38:29.088961Z","shell.execute_reply":"2024-07-08T13:38:29.088986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport pandas as pd\n\n# Function to parse TFRecord file\ndef parse_tfrecord(serialized_example):\n    # Define your feature structure here based on your data\n    feature_description = {\n        'APIs': tf.io.VarLenFeature(tf.string),\n        'bytes': tf.io.VarLenFeature(tf.string),\n        'opcodes': tf.io.VarLenFeature(tf.string),\n        'label': tf.io.VarLenFeature(tf.int64)\n    }\n    example = tf.io.parse_single_example(serialized_example, feature_description)\n    return example\n\n# Path to your TFRecord file (replace with your actual path)\ntfrecord_path = '/kaggle/input/hydra-dataset/train-001.tfrecords'\n\n# Create a TFRecordDataset and parse it\ndataset = tf.data.TFRecordDataset(tfrecord_path)\nparsed_dataset = dataset.map(parse_tfrecord)\n\n# Initialize empty lists to hold parsed data\nAPIs_list = []\nbytes_list = []\nopcodes_list = []\nlabel_list = []\n\n# Iterate over parsed dataset and append to lists\nfor parsed_record in parsed_dataset:\n    APIs_list.append(parsed_record['APIs'].values.numpy())\n    bytes_list.append(parsed_record['bytes'].values.numpy())\n    opcodes_list.append(parsed_record['opcodes'].values.numpy())\n    label_list.append(parsed_record['label'].values.numpy())\n\n# Convert lists to DataFrame\ndata = pd.DataFrame({\n    'APIs': APIs_list,\n    'bytes': bytes_list,\n    'opcodes': opcodes_list,\n    'label': label_list\n})\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.090847Z","iopub.status.idle":"2024-07-08T13:38:29.091704Z","shell.execute_reply.started":"2024-07-08T13:38:29.091398Z","shell.execute_reply":"2024-07-08T13:38:29.091421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['APIs']=data['APIs'].replace(\"[b'\", \"\").replace(\"']\", \"\").replace(\"b'\", \"\").replace(\"'\", \"\")","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.093328Z","iopub.status.idle":"2024-07-08T13:38:29.094207Z","shell.execute_reply.started":"2024-07-08T13:38:29.093901Z","shell.execute_reply":"2024-07-08T13:38:29.093938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['APIs']","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.095814Z","iopub.status.idle":"2024-07-08T13:38:29.096512Z","shell.execute_reply.started":"2024-07-08T13:38:29.096287Z","shell.execute_reply":"2024-07-08T13:38:29.096312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.to_csv('dataset.csv')","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.098144Z","iopub.status.idle":"2024-07-08T13:38:29.098942Z","shell.execute_reply.started":"2024-07-08T13:38:29.098620Z","shell.execute_reply":"2024-07-08T13:38:29.098644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['label']=data['label'].astype(int) #or df['label'] = df['label'].apply(lambda x: int(x))","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.100551Z","iopub.status.idle":"2024-07-08T13:38:29.101401Z","shell.execute_reply.started":"2024-07-08T13:38:29.101115Z","shell.execute_reply":"2024-07-08T13:38:29.101139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom tensorflow.keras.models import Sequential, load_model\nfrom tensorflow.keras.layers import LSTM, Dense, Embedding\n\n# Load your dataset\ndata = pd.read_csv('/kaggle/working/dataset')  # Adjust this to your actual data loading method\n\n# Function to convert binary sequences to numerical sequences\ndef convert_binary_to_sequence(binary_data):\n    # Remove unwanted characters and decode the string to bytes\n    cleaned_data = binary_data.strip(\"[]b'\").replace(\"\\\\x\", \" \").replace(\"\\\\n\", \"\").replace(\"\\\\t\", \"\").replace(\"\\\\r\", \"\")\n    # Split the cleaned string into hex bytes\n    hex_bytes = cleaned_data.split()\n    sequence = []\n    for byte in hex_bytes:\n        try:\n            # Convert each hex byte to an integer\n            sequence.append(int(byte, 16))\n        except ValueError:\n            # Handle non-hexadecimal values gracefully\n            continue\n    return sequence\n\ndata['APIs_seq'] = data['APIs'].apply(convert_binary_to_sequence)\ndata['bytes_seq'] = data['bytes'].apply(convert_binary_to_sequence)\ndata['opcodes_seq'] = data['opcodes'].apply(convert_binary_to_sequence)\n\n# Combine sequences or choose one for the LSTM model (example uses APIs)\nsequences = data['APIs_seq'].tolist()\nlabels = data['label'].tolist()\n\n# Pad sequences\nmax_length = max([len(seq) for seq in sequences])\nX = pad_sequences(sequences, maxlen=max_length, padding='post')\n\n# Convert labels to numpy array\ny = np.array(labels)\n\n# Split data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Define the LSTM model\nmodel = Sequential()\nmodel.add(Embedding(input_dim=256, output_dim=128, input_length=max_length))  # Adjust input_dim as needed\nmodel.add(LSTM(64, return_sequences=True))\nmodel.add(LSTM(64))\nmodel.add(Dense(32, activation='relu'))\nmodel.add(Dense(1, activation='sigmoid'))  # Use 'softmax' for multi-class classification\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\n# Train the model\nmodel.fit(X_train, y_train, epochs=10, batch_size=64, validation_split=0.2)\n\n# Evaluate the model\nloss, accuracy = model.evaluate(X_test, y_test)\nprint(f'Test Accuracy: {accuracy}')\n\n# Save the model\nmodel.save('/kaggle/working/trained_model.h5')\nprint('Model saved to /kaggle/working/trained_model.h5')\n\n# To load the model later, you can use:\n# loaded_model = load_model('/kaggle/working/trained_model.h5')\n# loaded_model.evaluate(X_test, y_test)\n","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.103076Z","iopub.status.idle":"2024-07-08T13:38:29.103952Z","shell.execute_reply.started":"2024-07-08T13:38:29.103644Z","shell.execute_reply":"2024-07-08T13:38:29.103669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LSTM Malware detection [api call only]\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.layers import LSTM, Dense, Dropout, Embedding, SpatialDropout1D\nfrom tensorflow.keras.preprocessing import sequence\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.models import Sequential\nfrom mlxtend.plotting import plot_confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2024-07-08T13:38:29.105541Z","iopub.status.idle":"2024-07-08T13:38:29.106378Z","shell.execute_reply.started":"2024-07-08T13:38:29.106088Z","shell.execute_reply":"2024-07-08T13:38:29.106112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"malware_calls_df = pd.read_csv(\"/kaggle/input/lstmtest/calls/1000_calls.txt\", \n                               sep=\"\\t\", names=[\"API_Calls\"])\n\nmalware_labels_df = pd.read_csv(\"/kaggle/input/lstmtest/types/1000_types.txt\",\n                                sep=\"\\t\", names=[\"API_Labels\"])\n\nmalware_calls_df[\"API_Labels\"] = malware_labels_df.API_Labels\nmalware_calls_df[\"API_Calls\"] = malware_calls_df.API_Calls.apply(lambda x: \" \".join(x.split(\",\")))\n\nmalware_calls_df[\"API_Labels\"] = malware_calls_df.API_Labels.apply(lambda x: 1 if x == \"Virus\" else 0)","metadata":{"execution":{"iopub.status.busy":"2024-07-08T14:16:52.088245Z","iopub.execute_input":"2024-07-08T14:16:52.088774Z","iopub.status.idle":"2024-07-08T14:17:17.585510Z","shell.execute_reply.started":"2024-07-08T14:16:52.088729Z","shell.execute_reply":"2024-07-08T14:17:17.584117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_ones=0\ncount_zeros=0\nfor i in malware_calls_df.API_Labels:\n    if i==1:\n        count_ones+=1\n    count_zeros+=1\nprint(f\"ones:{count_ones}, zeros{count_zeros}\")","metadata":{"execution":{"iopub.status.busy":"2024-07-08T14:19:31.655945Z","iopub.execute_input":"2024-07-08T14:19:31.656460Z","iopub.status.idle":"2024-07-08T14:19:31.667206Z","shell.execute_reply.started":"2024-07-08T14:19:31.656423Z","shell.execute_reply":"2024-07-08T14:19:31.665866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.countplot(x='API_Labels', data=malware_calls_df)\nplt.xlabel('Labels')\nplt.title('Class distribution')\nplt.savefig(\"class_distribution.png\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-07-08T14:24:46.816387Z","iopub.execute_input":"2024-07-08T14:24:46.817152Z","iopub.status.idle":"2024-07-08T14:24:47.182598Z","shell.execute_reply.started":"2024-07-08T14:24:46.817115Z","shell.execute_reply":"2024-07-08T14:24:47.181395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_words = 800\nmax_len = 100\n\nX = malware_calls_df.API_Calls\nY = malware_calls_df.API_Labels.astype('category').cat.codes\n\ntok = Tokenizer(num_words=max_words)\ntok.fit_on_texts(X)\nprint('Found %s unique tokens.' % len(tok.word_index))\nX = tok.texts_to_sequences(X.values)\nX = sequence.pad_sequences(X, maxlen=max_len)\nprint('Shape of data tensor:', X.shape)\n\nX_train, X_test, Y_train, Y_test = train_test_split(X, Y, test_size=0.15)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-07-08T14:25:06.461412Z","iopub.execute_input":"2024-07-08T14:25:06.461852Z","iopub.status.idle":"2024-07-08T14:27:57.260286Z","shell.execute_reply.started":"2024-07-08T14:25:06.461822Z","shell.execute_reply":"2024-07-08T14:27:57.258369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le = LabelEncoder()\nY_train_enc = le.fit_transform(Y_train)\nY_train_enc = to_categorical(Y_train_enc)\n\nY_test_enc = le.transform(Y_test)\nY_test_enc = to_categorical(Y_test_enc)","metadata":{"execution":{"iopub.status.busy":"2024-07-08T14:30:52.319155Z","iopub.execute_input":"2024-07-08T14:30:52.319583Z","iopub.status.idle":"2024-07-08T14:30:52.331244Z","shell.execute_reply.started":"2024-07-08T14:30:52.319550Z","shell.execute_reply":"2024-07-08T14:30:52.330019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def malware_model(act_func=\"softsign\"):\n    model = Sequential()\n    model.add(Embedding(max_words, 300, input_length=max_len))\n    model.add(SpatialDropout1D(0.1))\n    model.add(LSTM(32, dropout=0.1, recurrent_dropout=0.1,\n                   return_sequences=True, activation=act_func))\n    model.add(LSTM(32, dropout=0.1, activation=act_func, return_sequences=True))\n    model.add(LSTM(32, dropout=0.1, activation=act_func))\n    model.add(Dense(128, activation=act_func))\n    model.add(Dropout(0.1))\n    model.add(Dense(256, activation=act_func))\n    model.add(Dropout(0.1))\n    model.add(Dense(128, activation=act_func))\n    model.add(Dropout(0.1))\n    model.add(Dense(1, name='out_layer', activation=\"linear\"))\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-07-08T14:31:21.814996Z","iopub.execute_input":"2024-07-08T14:31:21.815964Z","iopub.status.idle":"2024-07-08T14:31:21.825877Z","shell.execute_reply.started":"2024-07-08T14:31:21.815855Z","shell.execute_reply":"2024-07-08T14:31:21.824614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = malware_model()\nprint(model.summary())\nmodel.compile(loss='mse', optimizer=\"rmsprop\",\n              metrics=['accuracy'])\nhistory = model.fit(X_train, Y_train, batch_size=1000, epochs=10,\n                    validation_data=(X_test, Y_test_enc), verbose=1)","metadata":{"execution":{"iopub.status.busy":"2024-07-08T14:38:42.238975Z","iopub.execute_input":"2024-07-08T14:38:42.239560Z","iopub.status.idle":"2024-07-08T14:40:11.648549Z","shell.execute_reply.started":"2024-07-08T14:38:42.239500Z","shell.execute_reply":"2024-07-08T14:40:11.647183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}