{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59865,"databundleVersionId":6660280,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#%%\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport pathlib\nimport tensorflow as tf\n\nimport dataclasses","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"/#%%\n# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\ninputs = pathlib.Path('/kaggle/input/vpn-classification/dataset_v2')\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nlist(inputs.glob(\"*\"))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_train = inputs / 'train.parq'\nfile_test = inputs / 'test.parq'\nfile_shodan = inputs / 'shodan_df_hashed.csv'\n\ndata_train = pd.read_parquet(file_train)\ndata_test = pd.read_parquet(file_test)\ndata_shodan = pd.read_csv(file_shodan)\n\n#data_train = data_train[:100]\ndata_train = data_train\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train = data_train.drop(\"watcher_as_name\", axis = 1)\ndata_train = data_train.drop(\"attacker_as_name\", axis = 1)\ndata_train = data_train.drop(\"attack_time\", axis = 1)\ndata_train","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test = data_test.drop(\"watcher_as_name\", axis = 1)\ndata_test = data_test.drop(\"attacker_as_name\", axis = 1)\ndata_test","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert the cols to what we want the to be\n\n# Cols: attack_time \twatcher_country \twatcher_as_num \tattacker_country \tattacker_as_num \tattack_type \twatcher_uuid_enum \tattacker_ip_enum \tlabel\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Inputs\n\n# Two layer dense?\n\n# Output one bit\n\ndata_train\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, models\nimport pandas as pd\n\n# Assume `data_train` is a pandas DataFrame containing your data\n\n# Fill NaN values in watcher_as_num and attacker_as_num with 0 (or another value if appropriate)\ndata_train['watcher_as_num'] = data_train['watcher_as_num'].fillna(0).astype(int)\ndata_train['attacker_as_num'] = data_train['attacker_as_num'].fillna(0).astype(int)\n\n# Convert categorical columns to object type, fill NaNs, and convert to strings\nfor col in ['watcher_country', 'attacker_country', 'attack_type', 'watcher_uuid_enum', 'attacker_ip_enum']:\n    data_train[col] = data_train[col].astype(object).fillna('Unknown').astype(str)\n\n# Create vocabulary for watcher_country and attacker_country\nwatcher_country_vocab = data_train['watcher_country'].unique().tolist()\nattacker_country_vocab = data_train['attacker_country'].unique().tolist()\n\ncountry_vocab = list(set(watcher_country_vocab) | set(attacker_country_vocab))\n\nprint(country_vocab)\n\n# Define the StringLookup layers for categorical features with the vocabularies\nlookup_watcher_country = tf.keras.layers.StringLookup(vocabulary=country_vocab, output_mode=\"one_hot\")\nlookup_attacker_country = tf.keras.layers.StringLookup(vocabulary=country_vocab, output_mode=\"one_hot\")\nlookup_attack_type = tf.keras.layers.StringLookup(output_mode=\"one_hot\")\nlookup_watcher = tf.keras.layers.StringLookup(output_mode=\"int\")\nlookup_attacker = tf.keras.layers.StringLookup(output_mode=\"int\")\n\n# Adapt the lookup layers to the data\nlookup_attack_type.adapt(data_train['attack_type'])\nlookup_watcher.adapt(data_train['watcher_uuid_enum'])\nlookup_attacker.adapt(data_train['attacker_ip_enum'])\n\n# Input layers\nwatcher_country_input = layers.Input(shape=(1,), name='watcher_country', dtype=tf.string)\nwatcher_as_num_input = layers.Input(shape=(1,), name='watcher_as_num', dtype=tf.int32)\nattacker_country_input = layers.Input(shape=(1,), name='attacker_country', dtype=tf.string)\nattacker_as_num_input = layers.Input(shape=(1,), name='attacker_as_num', dtype=tf.int32)\nattack_type_input = layers.Input(shape=(1,), name='attack_type', dtype=tf.string)\nwatcher_uuid_enum_input = layers.Input(shape=(1,), name='watcher_uuid_enum', dtype=tf.string)\nattacker_ip_enum_input = layers.Input(shape=(1,), name='attacker_ip_enum', dtype=tf.string)\n\n# Preprocessing layers\nwatcher_country_encoded = lookup_watcher_country(watcher_country_input)\nattacker_country_encoded = lookup_attacker_country(attacker_country_input)\nattack_type_encoded = lookup_attack_type(attack_type_input)\nwatcher_uuid_enum_encoded = lookup_watcher(watcher_uuid_enum_input)\nattacker_ip_enum_encoded = lookup_attacker(attacker_ip_enum_input)\n\n# Concatenate all features\nall_features = layers.concatenate([\n    watcher_country_encoded,\n    watcher_as_num_input,\n    attacker_country_encoded,\n    attacker_as_num_input,\n    attack_type_encoded,\n    watcher_uuid_enum_encoded,\n    attacker_ip_enum_encoded\n])\n\n# Define the neural network\nx = layers.Dense(64, activation='relu')(all_features)\nx = layers.Dense(32, activation='relu')(x)\noutput = layers.Dense(1, activation='sigmoid')(x)\n\n# Create the model\nmodel = models.Model(inputs=[\n    watcher_country_input,\n    watcher_as_num_input,\n    attacker_country_input,\n    attacker_as_num_input,\n    attack_type_input,\n    watcher_uuid_enum_input,\n    attacker_ip_enum_input\n], outputs=output)\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\n# Summary of the model\nmodel.summary()\n\n# Plot the model\ntf.keras.utils.plot_model(model, to_file='model.png', show_shapes=True, show_layer_names=True)\n\n# Assuming `data_train` has a column 'label' for the target variable\n# Split the data into features and labels\nfeatures = {\n    'watcher_country': data_train['watcher_country'],\n    'watcher_as_num': data_train['watcher_as_num'],\n    'attacker_country': data_train['attacker_country'],\n    'attacker_as_num': data_train['attacker_as_num'],\n    'attack_type': data_train['attack_type'],\n    'watcher_uuid_enum': data_train['watcher_uuid_enum'],\n    'attacker_ip_enum': data_train['attacker_ip_enum']\n}\nlabels = data_train['label']\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot model\nmodel.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the model\ntf.keras.utils.plot_model(model, to_file='model.png', show_shapes=True, show_layer_names=True, show_dtype=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model\nmodel.fit(features, labels, epochs=10, batch_size=32, validation_split=0.2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}