{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59865,"databundleVersionId":6660280,"sourceType":"competition"}],"dockerImageVersionId":30626,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\nimport numpy as np \nimport pandas as pd\nimport ast\nimport json\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport gc\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-09T23:03:28.753732Z","iopub.execute_input":"2024-01-09T23:03:28.754112Z","iopub.status.idle":"2024-01-09T23:03:30.557733Z","shell.execute_reply.started":"2024-01-09T23:03:28.754080Z","shell.execute_reply":"2024-01-09T23:03:30.556438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ndf = pd.read_parquet(\"/kaggle/input/vpn-classification/dataset_v2/train.parq\")\ndf.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:03:30.559998Z","iopub.execute_input":"2024-01-09T23:03:30.560712Z","iopub.status.idle":"2024-01-09T23:03:42.850649Z","shell.execute_reply.started":"2024-01-09T23:03:30.560668Z","shell.execute_reply":"2024-01-09T23:03:42.849638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.dtypes","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:03:42.851877Z","iopub.execute_input":"2024-01-09T23:03:42.852436Z","iopub.status.idle":"2024-01-09T23:03:42.860956Z","shell.execute_reply.started":"2024-01-09T23:03:42.852400Z","shell.execute_reply":"2024-01-09T23:03:42.859601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:03:42.863320Z","iopub.execute_input":"2024-01-09T23:03:42.863780Z","iopub.status.idle":"2024-01-09T23:03:43.684528Z","shell.execute_reply.started":"2024-01-09T23:03:42.863749Z","shell.execute_reply":"2024-01-09T23:03:43.683346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:03:43.685748Z","iopub.execute_input":"2024-01-09T23:03:43.686056Z","iopub.status.idle":"2024-01-09T23:03:53.042145Z","shell.execute_reply.started":"2024-01-09T23:03:43.686028Z","shell.execute_reply":"2024-01-09T23:03:53.041066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Visualization\n\nThis section has some visualization that may help undestanding our data. It is clearly an umbalanced problem, with many Non-VPN/Proxy users and few VPN/Proxy users, due to this difference you need to pay attention to the axis (y-axis specially), since it may have a different scale when comparing the Non-VPN/Proxy and VPN/Proxy users.\n","metadata":{}},{"cell_type":"code","source":"COLOR_NON_VPN= '#437F97'\nCOLOR_VPN = '#FF7477'","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:03:53.043812Z","iopub.execute_input":"2024-01-09T23:03:53.044478Z","iopub.status.idle":"2024-01-09T23:03:53.049209Z","shell.execute_reply.started":"2024-01-09T23:03:53.044438Z","shell.execute_reply":"2024-01-09T23:03:53.048003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop_duplicates(['attacker_ip_enum', 'label']).groupby('label')['attacker_ip_enum'].count().plot(kind='bar', color=[COLOR_NON_VPN, COLOR_VPN])","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:03:53.050344Z","iopub.execute_input":"2024-01-09T23:03:53.051187Z","iopub.status.idle":"2024-01-09T23:03:55.682380Z","shell.execute_reply.started":"2024-01-09T23:03:53.051139Z","shell.execute_reply":"2024-01-09T23:03:55.681312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,3, figsize=(20,5))\n\ndf.groupby(['attack_type']).count()['attacker_ip_enum'].plot(kind='bar', ax=ax[0], title='Attack Frequency')\ndf[df['label'] == 0].groupby(['attack_type']).count()['attacker_ip_enum'].plot(kind='bar', ax=ax[1], title='Non-VPN/Proxy', color=COLOR_NON_VPN)\ndf[df['label'] == 1].groupby(['attack_type']).count()['attacker_ip_enum'].plot(kind='bar', ax=ax[2], title='VPN/Proxy', color=COLOR_VPN)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:03:55.684077Z","iopub.execute_input":"2024-01-09T23:03:55.684633Z","iopub.status.idle":"2024-01-09T23:04:10.937638Z","shell.execute_reply.started":"2024-01-09T23:03:55.684562Z","shell.execute_reply":"2024-01-09T23:04:10.936430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Most fequent Watcher Country ","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1,3, figsize=(20,5))\n\ndf.groupby(['watcher_country']).count()['attacker_ip_enum'].sort_values(ascending=False).head(20).plot(kind='bar', ax=ax[0], title=\"Most Frequent Watcher Country\")\ndf[df['label'] == 0].groupby(['watcher_country']).count()['attacker_ip_enum'].sort_values(ascending=False).head(20).plot(kind='bar', ax=ax[1], title=\"Non-VPN/Proxy\", color=COLOR_NON_VPN)\ndf[df['label'] == 1].groupby(['watcher_country']).count()['attacker_ip_enum'].sort_values(ascending=False).head(20).plot(kind='bar', ax=ax[2], title=\"VPN/Proxy\", color=COLOR_VPN)\n\nprint('Num. Unique Watcher Country',  len(df['watcher_country'].unique()) )","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:04:10.939955Z","iopub.execute_input":"2024-01-09T23:04:10.940622Z","iopub.status.idle":"2024-01-09T23:04:27.837356Z","shell.execute_reply.started":"2024-01-09T23:04:10.940561Z","shell.execute_reply":"2024-01-09T23:04:27.835980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Most Frequent Attacker Country","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1,3, figsize=(20,5))\n\ndf.groupby(['attacker_country']).count()['attacker_ip_enum'].sort_values(ascending=False).head(20).plot(kind='bar', ax=ax[0], title=\"Most Frequent Attacker Country\")\ndf[df['label'] == 0].groupby(['attacker_country']).count()['attacker_ip_enum'].sort_values(ascending=False).head(20).plot(kind='bar', ax=ax[1], title=\"Non-VPN/Proxy\", color=COLOR_NON_VPN)\ndf[df['label'] == 1].groupby(['attacker_country']).count()['attacker_ip_enum'].sort_values(ascending=False).head(20).plot(kind='bar', ax=ax[2], title=\"VPN/Proxy\", color=COLOR_VPN)\n\nprint('Num. Unique Attacker Country',  len(df['attacker_country'].unique()) )\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:04:27.842495Z","iopub.execute_input":"2024-01-09T23:04:27.842896Z","iopub.status.idle":"2024-01-09T23:04:41.748255Z","shell.execute_reply.started":"2024-01-09T23:04:27.842863Z","shell.execute_reply":"2024-01-09T23:04:41.747183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Most Frequent Attacker AS","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1,3, figsize=(20,5))\n\ndf.groupby(['attacker_as_name']).count()['attacker_ip_enum'].sort_values(ascending=False).head(20).plot(kind='bar', ax=ax[0], title=\"Most Frequent Attacker AS\")\ndf[df['label'] == 0].groupby(['attacker_as_name']).count()['attacker_ip_enum'].sort_values(ascending=False).head(20).plot(kind='bar', ax=ax[1], title=\"Non-VPN/Proxy\", color=COLOR_NON_VPN)\ndf[df['label'] == 1].groupby(['attacker_as_name']).count()['attacker_ip_enum'].sort_values(ascending=False).head(20).plot(kind='bar', ax=ax[2], title=\"VPN/Proxy\", color=COLOR_VPN)\n\nprint('Num. Unique Attacker AS',  len(df['attacker_as_name'].unique()) )\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:04:41.749814Z","iopub.execute_input":"2024-01-09T23:04:41.750753Z","iopub.status.idle":"2024-01-09T23:04:55.063054Z","shell.execute_reply.started":"2024-01-09T23:04:41.750698Z","shell.execute_reply":"2024-01-09T23:04:55.061926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vpn_attacker_as_name = df[df['label'] == 1]['attacker_as_name'].astype(str).unique()\nnon_vpn_attacker_as_name = df[df['label'] == 0]['attacker_as_name'].astype(str).unique()\n\n\nprint('Num. AS VPN/Proxy Attacker', len(vpn_attacker_as_name))\nprint('Num. AS Non-VPN/Proxy Attacker', len(non_vpn_attacker_as_name))\n\nprint('Num. AS ONLY VPN/Proxy Attacker', len(list(set(vpn_attacker_as_name) - set(non_vpn_attacker_as_name))))\nprint('Num. AS ONLY Non-VPN/Proxy Attacker', len(list(set(non_vpn_attacker_as_name) - set(vpn_attacker_as_name))))\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:04:55.064678Z","iopub.execute_input":"2024-01-09T23:04:55.065521Z","iopub.status.idle":"2024-01-09T23:05:07.754905Z","shell.execute_reply.started":"2024-01-09T23:04:55.065485Z","shell.execute_reply":"2024-01-09T23:05:07.753231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Missing Values","metadata":{}},{"cell_type":"code","source":"# df.isna().sum()\ndf.isna().sum().sort_values(ascending=False).plot(kind='bar', title='Missing Values')\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:05:07.756694Z","iopub.execute_input":"2024-01-09T23:05:07.757060Z","iopub.status.idle":"2024-01-09T23:05:08.876970Z","shell.execute_reply.started":"2024-01-09T23:05:07.757026Z","shell.execute_reply":"2024-01-09T23:05:08.875893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Day/Time visualisation","metadata":{}},{"cell_type":"code","source":"df['attack_time'] = pd.to_datetime(df['attack_time'])\n# df['day_name'] =  df['attack_time'].dt.day_name()\ndf['dayofweek'] =  df['attack_time'].dt.dayofweek\n# df['attack_time'].dt.dayofweek\ndf['attack_hour'] = df['attack_time'].dt.hour\ndf['attack_minute'] = df['attack_time'].dt.minute\ndf['timestamp'] = df['attack_time'].dt.time\n\ndf['date'] = df['attack_time'].dt.date\n\n\nx = df[df['label'] == 1].groupby('dayofweek')['label'].size()\nx = x / x.sum()\n\nx2 = df[df['label'] == 0].groupby('dayofweek')['label'].size()\nx2 = x2 / x2.sum()\n\nfig, ax = plt.subplots(figsize=(10, 5))\n\nax.plot(x, color=COLOR_VPN, label=\"VPN/Proxy\")\nax.fill_between(list(range(len(x))), x-x.std(), x+x.std(), color=COLOR_VPN, alpha=0.2)\n\nax.plot(x2, color=COLOR_NON_VPN, label='NON VPN/Proxy')\nax.fill_between(list(range(len(x2))), x2-x2.std(), x2+x2.std(), color=COLOR_NON_VPN, alpha=0.2)\n\nax.set_title('Relative Number of Attacks per Day of Week')\nax.set_ylabel('Frequency')\nax.set_xlabel('Day of the week')\nax.legend()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:05:08.878687Z","iopub.execute_input":"2024-01-09T23:05:08.879004Z","iopub.status.idle":"2024-01-09T23:07:02.465773Z","shell.execute_reply.started":"2024-01-09T23:05:08.878976Z","shell.execute_reply":"2024-01-09T23:07:02.464616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nx = df[df['label'] == 1].groupby('attack_hour')['label'].size()\nx = x / x.sum()\n\nx2 = df[df['label'] == 0].groupby('attack_hour')['label'].size()\nx2 = x2 / x2.sum()\n\nfig, ax = plt.subplots(figsize=(10, 5))\n\nax.plot(x, color=COLOR_VPN, label=\"VPN/Proxy\")\nax.plot(x2, color=COLOR_NON_VPN, label='NON VPN/Proxy')\n\nax.set_title('Relative Frequency of attacks of each hour')\nax.set_ylabel('Frequency (%)')\nax.set_xlabel('Hour')\nax.set_xticks(list(range(24))) # 24h -> from 0 to 23\nax.legend()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:07:02.467689Z","iopub.execute_input":"2024-01-09T23:07:02.468140Z","iopub.status.idle":"2024-01-09T23:07:12.578036Z","shell.execute_reply.started":"2024-01-09T23:07:02.468087Z","shell.execute_reply":"2024-01-09T23:07:12.576856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Shodan\n","metadata":{}},{"cell_type":"code","source":"def extract_port_numbers(value: dict):\n    return set(key.split(\"/\")[0] for key in value.keys())\n\ndef extract_protocol(value: dict):\n    return set(key.split(\"/\")[1] for key in value.keys())\n\n\ndef extract_headers_hash(info):\n    headers = []\n    for k in info:\n        headers.append( info[k].get('headers_hash', {}) )\n    \n    headers = [h for h in headers if h is not None ]\n    \n    return headers\n\n\ndef count_protocols(value: dict):\n    count = {'tcp': 0, 'udp': 0}\n    for k in value:\n        protocol = k.split('/')[1]\n        count[protocol] += 1\n    \n    return count\n\n\nshodan_df = pd.read_csv(\n    \"/kaggle/input/vpn-classification/dataset_v2/shodan_df_hashed.csv\",\n    dtype={\"attacker_ip_enum\": \"int32\"},\n    index_col=\"attacker_ip_enum\",\n)\n\nshodan_df[\"shodan_info\"] = shodan_df[\"shodan_info\"].map(ast.literal_eval)\nshodan_df[\"shodan_open_ports\"] = shodan_df[\"shodan_info\"].map(extract_port_numbers)\nshodan_df[\"shodan_protocol\"] = shodan_df[\"shodan_info\"].map(extract_protocol)\nshodan_df['num_open_ports'] = shodan_df['shodan_open_ports'].apply(lambda x : len(x))\n\nshodan_df = shodan_df.reset_index()\n\n# Extract the number of \nshodan_df['num_tcp'] = shodan_df['shodan_info'].apply(count_protocols).apply(lambda x : x['tcp'])\nshodan_df['num_udp'] = shodan_df['shodan_info'].apply(count_protocols).apply(lambda x : x['udp'])\n\n# Extract information about JARM and JA3S in the port 433 only\nshodan_df['jarm'] = shodan_df['shodan_info'].apply(lambda x : x.get('443/tcp', {}).get('jarm', np.nan))\nshodan_df['ja3s'] = shodan_df['shodan_info'].apply(lambda x : x.get('443/tcp', {}).get('ja3s', np.nan))\n\n# Split the information of JARM in Head and Tail\nshodan_df['jarm_head'] = shodan_df['jarm'].str[:30]\nshodan_df['jarm_tail'] = shodan_df['jarm'].str[30:]\n\n\n# Extract the headers from all protocols\nshodan_df['headers_hash'] = shodan_df['shodan_info'].apply(extract_headers_hash)\nshodan_df['num_headers_hash'] = shodan_df['headers_hash'].apply(lambda x : len(x))\nshodan_df['num_repeated_headers'] = shodan_df['headers_hash'].apply(lambda x : len(x) - len(set(x)))\n\n\n\n\nshodan_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:07:12.579862Z","iopub.execute_input":"2024-01-09T23:07:12.580210Z","iopub.status.idle":"2024-01-09T23:07:22.942333Z","shell.execute_reply.started":"2024-01-09T23:07:12.580178Z","shell.execute_reply":"2024-01-09T23:07:22.941275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shodan_df[['num_open_ports', 'num_headers_hash', 'num_repeated_headers']].describe()","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:07:22.943773Z","iopub.execute_input":"2024-01-09T23:07:22.944302Z","iopub.status.idle":"2024-01-09T23:07:22.983177Z","shell.execute_reply.started":"2024-01-09T23:07:22.944270Z","shell.execute_reply":"2024-01-09T23:07:22.982079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax = shodan_df[['num_open_ports', 'num_headers_hash', 'num_repeated_headers']].plot(kind='box')\nax.set_xticklabels(['Open Ports', 'All Headers', 'Repeated Headers'])","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:07:22.984771Z","iopub.execute_input":"2024-01-09T23:07:22.985094Z","iopub.status.idle":"2024-01-09T23:07:23.443424Z","shell.execute_reply.started":"2024-01-09T23:07:22.985066Z","shell.execute_reply":"2024-01-09T23:07:23.442202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ports = shodan_df['shodan_open_ports'].explode().dropna().reset_index()\ndf_ports.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:07:23.445066Z","iopub.execute_input":"2024-01-09T23:07:23.445501Z","iopub.status.idle":"2024-01-09T23:07:23.858595Z","shell.execute_reply.started":"2024-01-09T23:07:23.445460Z","shell.execute_reply":"2024-01-09T23:07:23.857459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,3, figsize=(20,5))\n\ndf_ports.groupby('shodan_open_ports').count()['index'].sort_values(ascending=False).head(20).plot(kind='bar', ax=ax[0], title='Frequent Ports')\ndf_ports[ df_ports['index'].isin( df[ df['label'] == 0 ]['attacker_ip_enum'] ) ].groupby('shodan_open_ports').count()['index'].sort_values(ascending=False).head(20).plot(kind='bar', ax=ax[1], title='Non VPN/Proxy', color=COLOR_NON_VPN)\ndf_ports[ df_ports['index'].isin( df[ df['label'] == 1 ]['attacker_ip_enum'] ) ].groupby('shodan_open_ports').count()['index'].sort_values(ascending=False).head(20).plot(kind='bar', ax=ax[2], title='VPN/Proxy', color=COLOR_VPN)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:07:23.860273Z","iopub.execute_input":"2024-01-09T23:07:23.861252Z","iopub.status.idle":"2024-01-09T23:07:34.186373Z","shell.execute_reply.started":"2024-01-09T23:07:23.861212Z","shell.execute_reply":"2024-01-09T23:07:34.185308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nports_counts = df_ports['shodan_open_ports'].value_counts()\n\nfig, axes = plt.subplots(1,2, figsize=(20,7))\nports_counts.value_counts().plot(logy=True, logx=True, ax=axes[0], title='Power law of used ports')\nports_counts[:20].plot(kind='barh', ax=axes[1], title='TOP 20 Ports')\n\nnum_vnp_with_top_ports = len( df[ (df['label'] == 1) & (df['attacker_ip_enum'].isin( df_ports[ df_ports['shodan_open_ports'].isin(ports_counts[:20].index) ]['index'] )) ].groupby('attacker_ip_enum').count() )\nnum_nonvnp_with_top_ports = len( df[ (df['label'] == 0) & (df['attacker_ip_enum'].isin( df_ports[ df_ports['shodan_open_ports'].isin(ports_counts[:20].index) ]['index'] )) ].groupby('attacker_ip_enum').count() )\n\n\n\nprint('Num. IPs using a single port:', (ports_counts == 1).sum())\nprint('Num. IPs using VPN/Proxy with TOP ports open:', num_vnp_with_top_ports)\nprint('Num. IPs using NON-VPN/Proxy with TOP ports open:', num_nonvnp_with_top_ports)\nprint('Fraction of VPN/Proxy using TOP portas:', num_vnp_with_top_ports/(len( df[df['label'] == 1]['attacker_ip_enum'].unique() ) ))\nprint('Fraction of NON-VPN/Proxy using TOP portas:', num_nonvnp_with_top_ports/(len( df[df['label'] == 0]['attacker_ip_enum'].unique() ) ))\n\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:07:34.187865Z","iopub.execute_input":"2024-01-09T23:07:34.188192Z","iopub.status.idle":"2024-01-09T23:07:51.367775Z","shell.execute_reply.started":"2024-01-09T23:07:34.188163Z","shell.execute_reply":"2024-01-09T23:07:51.366950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_headers = shodan_df['headers_hash'].explode().dropna().reset_index()\ndf_headers.columns = ['attacker_ip_enum', 'headers_hash']\n\ndf_headers.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:07:51.369329Z","iopub.execute_input":"2024-01-09T23:07:51.369713Z","iopub.status.idle":"2024-01-09T23:07:51.439845Z","shell.execute_reply.started":"2024-01-09T23:07:51.369681Z","shell.execute_reply":"2024-01-09T23:07:51.438747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,3, figsize=(20,5))\n\ndf_headers.groupby('headers_hash').count()['attacker_ip_enum'].sort_values(ascending=False).head(20).plot(kind='bar', ax=ax[0], title='Frequent Headers')\ndf_headers[ df_headers['attacker_ip_enum'].isin( df[ df['label'] == 0 ]['attacker_ip_enum'] ) ].groupby('headers_hash').count()['attacker_ip_enum'].sort_values(ascending=False).head(20).plot(kind='bar', ax=ax[1], title='Non VPN/Proxy', color=COLOR_NON_VPN)\ndf_headers[ df_headers['attacker_ip_enum'].isin( df[ df['label'] == 1 ]['attacker_ip_enum'] ) ].groupby('headers_hash').count()['attacker_ip_enum'].sort_values(ascending=False).head(20).plot(kind='bar', ax=ax[2], title='VPN/Proxy', color=COLOR_VPN)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:07:51.441271Z","iopub.execute_input":"2024-01-09T23:07:51.441632Z","iopub.status.idle":"2024-01-09T23:08:01.790887Z","shell.execute_reply.started":"2024-01-09T23:07:51.441582Z","shell.execute_reply":"2024-01-09T23:08:01.789807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training the Model","metadata":{}},{"cell_type":"code","source":"attack_types_df = (\n    df.attack_type.str.split(\":\", expand=True)\n    .rename(columns={0: \"service\", 1: \"type\"})\n    .set_index(df[\"attacker_ip_enum\"])\n)\n\none_hot_attack_service_df = pd.get_dummies(\n    attack_types_df.reset_index()\n    .drop_duplicates(subset=[\"attacker_ip_enum\", \"service\"])\n    .set_index(\"attacker_ip_enum\")[\"service\"]\n    # ,sparse=True\n)\none_hot_attack_service_df = one_hot_attack_service_df.groupby(\"attacker_ip_enum\").sum()\none_hot_attack_service_df = (one_hot_attack_service_df >= 1).astype(int)\n# one_hot_attack_service_df.head()\n\none_hot_attack_types_df = pd.get_dummies(\n    attack_types_df[\"type\"],\n    prefix=\"attack_type\"\n)\none_hot_attack_types_df = one_hot_attack_types_df.groupby(\"attacker_ip_enum\").sum()\none_hot_attack_types_df = one_hot_attack_types_df / one_hot_attack_types_df.sum(\n    1\n).values.reshape(-1, 1)\n\none_hot_attack_types_df","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:08:01.792483Z","iopub.execute_input":"2024-01-09T23:08:01.792993Z","iopub.status.idle":"2024-01-09T23:09:22.039626Z","shell.execute_reply.started":"2024-01-09T23:08:01.792952Z","shell.execute_reply":"2024-01-09T23:09:22.038464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we will transform our DataFrames to features to be used by our model. Here I build some functions to encode the values in a \"Multilabel\" way, for example, if a IP uses the ports 80 and 443 while other ip uses the ports 443 and 22 the output would be something like 1, 1, 0 for the fist one and 0, 1, 1 to the second one. You may choose use the `MultiLabelBinarizer` instead.\n\nOne important thing to note, I select the TOP N occurrences from a column to select as features (e.g: Top 20  open Ports). I did this to the Ports, JARM, JARM-Head, JARM-Tail and JA3s.\n\nThe code below select the TOP 50 open ports used and encode it as columns with 1 or 0. We focus in the ports used by VPN/Proxy users, this may add some bias or can even be consided as a data leakage (for training/validation) since it not consider only the training set. (But a didn't notice it at the time =P)  \n","metadata":{}},{"cell_type":"code","source":"NUM_TOP_PORTS = 50\n\n# Restore the index column\nif shodan_df.index.name != 'attacker_ip_enum':\n    shodan_df = shodan_df.set_index('attacker_ip_enum')\n\ndf_ports = shodan_df['shodan_open_ports'].explode().dropna().reset_index()\n# Filter ports used by VPN/Proxy users\ndf_ports = df_ports[ df_ports['attacker_ip_enum'].isin(df[df['label'] == 1]['attacker_ip_enum']) ]\n# top_ports = df_ports.groupby('shodan_open_ports').count().sort_values('attacker_ip_enum', ascending=False).reset_index()['shodan_open_ports'] .values\ntop_ports = df_ports.groupby('shodan_open_ports') \\\n                .count() \\\n                .sort_values('attacker_ip_enum', ascending=False) \\\n                .head(NUM_TOP_PORTS) \\\n                .reset_index()['shodan_open_ports'] \\\n                .values\n\n# Remove duplicates\ntop_ports = set(top_ports)\n\nopen_ports_count = shodan_df[\"shodan_open_ports\"].map(len).rename(\"open_ports_count\")\nreference_ports_df = shodan_df[\"shodan_open_ports\"].map(lambda x: top_ports & x)\n\none_hot_reference_ports_df = pd.get_dummies(reference_ports_df.explode(), prefix=\"port\")\none_hot_reference_ports_df = one_hot_reference_ports_df.groupby(\n    \"attacker_ip_enum\"\n).sum()\nports_features_df = pd.concat([one_hot_reference_ports_df, open_ports_count], axis=1)\n\nports_features_df[(ports_features_df > 0).any(axis=1)].head()","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:09:22.041408Z","iopub.execute_input":"2024-01-09T23:09:22.042392Z","iopub.status.idle":"2024-01-09T23:09:25.225067Z","shell.execute_reply.started":"2024-01-09T23:09:22.042348Z","shell.execute_reply":"2024-01-09T23:09:25.223929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets get the TOP 100 used headers.","metadata":{}},{"cell_type":"code","source":"TOP_HEADERS = 100\n\ndf_headers = shodan_df['headers_hash'].explode().dropna().reset_index()\ndf_headers.columns = ['attacker_ip_enum', 'headers_hash']\n\n#df_headers = df_headers[ df_headers['attacker_ip_enum'].isin(df[df['label'] == 1]['attacker_ip_enum']) ]\n\ntop_headers = set(df_headers.groupby('headers_hash').count().sort_values('attacker_ip_enum', ascending=False).head(TOP_HEADERS).reset_index()['headers_hash'].values)\n\nreference_headers_df = shodan_df[\"headers_hash\"].map(lambda x: top_headers & set(x))\n\none_hot_reference_headers_df = pd.get_dummies(reference_headers_df.explode(), prefix=\"header\")\n# Aggregate the dummies by IP\none_hot_reference_headers_df = one_hot_reference_headers_df.groupby(\n    \"attacker_ip_enum\"\n).sum()\n\nheaders_features_df = pd.concat((one_hot_reference_headers_df, shodan_df[['num_headers_hash', 'num_repeated_headers', 'num_tcp', 'num_udp']]), axis=1)\n\nheaders_features_df[(headers_features_df > 0).any(axis=1)].head()","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:09:25.226438Z","iopub.execute_input":"2024-01-09T23:09:25.226791Z","iopub.status.idle":"2024-01-09T23:09:28.323672Z","shell.execute_reply.started":"2024-01-09T23:09:25.226761Z","shell.execute_reply":"2024-01-09T23:09:28.322849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now lets get the information about the JARM and JA3S and the commons headers used for HTTP (port 80) and HTTPS (port 443). Again, we are focusing the the VPN/Proxy users.\n\nAlso, we divided the JARM information in Head (first 30 chars) and Tails (the remaining chars), the combination of head and tails can be useful to detect the VPN/Proxy users.\n","metadata":{}},{"cell_type":"code","source":"TOP_JARM = 20\nTOP_HEADERS_HTTP = 20\n\nvpn_attackers_ip_enum = df[df['label'] == 1]['attacker_ip_enum']\nindex = shodan_df.index.isin(vpn_attackers_ip_enum)\n\n\ntop_ja3s = shodan_df[ (~shodan_df['ja3s'].isna()) & index ].groupby('ja3s').count()['shodan_info'].sort_values().tail(TOP_JARM).index.values\ntop_jarm_head = shodan_df[(~shodan_df['jarm_head'].isna()) & index ].groupby('jarm_head').count()['shodan_info'].sort_values().tail(TOP_JARM).index.values\ntop_jarm_tail = shodan_df[(~shodan_df['jarm_tail'].isna()) & index ].groupby('jarm_tail').count()['shodan_info'].sort_values().tail(TOP_JARM).index.values\n\n\n# top_hash_headers_443 = shodan_df[(~shodan_df['hash_headers_443'].isna()) & index ].groupby('hash_headers_443').count()['shodan_info'].sort_values().tail(TOP_HEADERS_HTTP).index.values\n# top_hash_headers_80 = shodan_df[(~shodan_df['hash_headers_80'].isna()) & index ].groupby('hash_headers_80').count()['shodan_info'].sort_values().tail(TOP_HEADERS_HTTP).index.values\n\n\nshodan_df_tmp = shodan_df.copy()\nshodan_df_tmp['ja3s'] = shodan_df_tmp['ja3s'].apply(lambda x : x if x in top_ja3s else np.nan)\nshodan_df_tmp['jarm_head'] = shodan_df_tmp['jarm_head'].apply(lambda x : x if x in top_jarm_head else np.nan)\nshodan_df_tmp['jarm_tail'] = shodan_df_tmp['jarm_tail'].apply(lambda x : x if x in top_jarm_tail else np.nan)\n\n# shodan_df_tmp['hash_headers_443'] = shodan_df_tmp['hash_headers_443'].apply(lambda x : x if x in top_hash_headers_443 else np.nan)\n# shodan_df_tmp['hash_headers_80'] = shodan_df_tmp['hash_headers_80'].apply(lambda x : x if x in top_hash_headers_80 else np.nan)\n\n\none_hot_ja3s_df = pd.get_dummies(\n    shodan_df_tmp['ja3s'],\n    prefix='ja3s'\n).groupby('attacker_ip_enum').sum()\n\none_hot_jarm_head_df = pd.get_dummies(\n    shodan_df_tmp['jarm_head'],\n    prefix='jarm_head'\n).groupby('attacker_ip_enum').sum()\n\none_hot_jarm_tail_df = pd.get_dummies(\n    shodan_df_tmp['jarm_tail'],\n    prefix='jarm_tail'\n).groupby('attacker_ip_enum').sum()\n\n# one_hot_hash_headers_443_df = pd.get_dummies(\n#     shodan_df_tmp['hash_headers_443'],\n#     prefix='hash_headers_443'\n# ).groupby('attacker_ip_enum').sum()\n\n\n# one_hot_hash_headers_80_df = pd.get_dummies(\n#     shodan_df_tmp['hash_headers_80'],\n#     prefix='hash_headers_80'\n# ).groupby('attacker_ip_enum').sum()\n\n\n\n# one_hot_jarm = pd.concat([one_hot_ja3s_df, one_hot_jarm_head_df, one_hot_jarm_tail_df, one_hot_hash_headers_443_df, one_hot_hash_headers_80_df], axis=1)\none_hot_jarm = pd.concat([one_hot_ja3s_df, one_hot_jarm_head_df, one_hot_jarm_tail_df], axis=1)\n\n# del one_hot_ja3s_df, one_hot_jarm_head_df, one_hot_jarm_tail_df, one_hot_hash_headers_443_df, one_hot_hash_headers_80_df\n\none_hot_jarm.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:09:28.324834Z","iopub.execute_input":"2024-01-09T23:09:28.325764Z","iopub.status.idle":"2024-01-09T23:09:35.958590Z","shell.execute_reply.started":"2024-01-09T23:09:28.325722Z","shell.execute_reply":"2024-01-09T23:09:35.957472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now lets get the information about the attarker and watcher countries and AS.","metadata":{}},{"cell_type":"code","source":"NUM_TOP_AS = 20\n\nvpn_attacker_as_name = df[df['label'] == 1]['attacker_as_name'].astype(str).unique()\nnon_vpn_attacker_as_name = df[df['label'] == 0]['attacker_as_name'].astype(str).unique()\n\n\ntop_as_vpn_attacker = df[ (df['label'] == 1) & (~df['attacker_as_name'].isin(non_vpn_attacker_as_name)) ].groupby('attacker_as_name').count()['label'].sort_values(ascending=False)\n# top_as_vpn_attacker = df[ (df['label'] == 1) ].groupby('attacker_as_name').count()['label'].sort_values(ascending=False)\ntop_as_vpn_attacker = pd.DataFrame(top_as_vpn_attacker.reset_index(), columns=['attacker_as_name', 'label'])\ntop_as_vpn_attacker.rename(columns={'label': 'count'}, inplace=True)\n\ntop_as_nonvpn_attacker = df[ (df['label'] == 0) & (~df['attacker_as_name'].isin(vpn_attacker_as_name)) ].groupby('attacker_as_name').count()['label'].sort_values(ascending=False)\n# top_as_nonvpn_attacker = df[ (df['label'] == 0) ].groupby('attacker_as_name').count()['label'].sort_values(ascending=False)\ntop_as_nonvpn_attacker = pd.DataFrame(top_as_nonvpn_attacker.reset_index(), columns=['attacker_as_name', 'label'])\ntop_as_nonvpn_attacker.rename(columns={'label': 'count'}, inplace=True)\n\n\ntop_as_attacker = pd.concat((top_as_vpn_attacker.head(NUM_TOP_AS), top_as_nonvpn_attacker.head(NUM_TOP_AS)), axis=0)\ntop_as_attacker = list(top_as_attacker['attacker_as_name'].values) + list(set(vpn_attacker_as_name) - set(non_vpn_attacker_as_name))\n\n\n\n# Cria uma coluna  indicando se a AS está nos mais frequentes, caso não esteja coloca 0\n# A coluna será \"pivotada\" para tidy\ndf_tmp = df.drop_duplicates(subset=[\"attacker_ip_enum\", \"attacker_as_name\"]).copy()\ndf_tmp['top_as_attacker'] = df_tmp['attacker_as_name'].apply(lambda x : x if x in top_as_attacker else 0)\n\n\ntop_attackers_as_df = pd.get_dummies(\n    df_tmp.set_index('attacker_ip_enum')['top_as_attacker'],\n    prefix='as_name'\n).groupby(\n    \"attacker_ip_enum\"\n).sum()\n\ndel df_tmp\n\ntop_attackers_as_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:09:35.966149Z","iopub.execute_input":"2024-01-09T23:09:35.966663Z","iopub.status.idle":"2024-01-09T23:10:13.106788Z","shell.execute_reply.started":"2024-01-09T23:09:35.966633Z","shell.execute_reply":"2024-01-09T23:10:13.105933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM_TOP_WATCHERS = 20\n\n\ntop_watcher_uuid_enum = df.groupby('watcher_uuid_enum')\\\n.count()['label']\\\n.sort_values(ascending=False)\\\n.head(NUM_TOP_WATCHERS)\\\n.reset_index()['watcher_uuid_enum']\\\n.values\n\ndf_tmp = df.drop_duplicates(subset=[\"attacker_ip_enum\", \"watcher_uuid_enum\"]).copy()\ndf_tmp['top_watcher_uuid_enum'] = df_tmp['watcher_uuid_enum'].apply(lambda x : x if x in top_watcher_uuid_enum else 0)\n\ntop_watchers_df = pd.get_dummies(\n    df_tmp.set_index('attacker_ip_enum')['top_watcher_uuid_enum'],\n    prefix='watcher'\n).groupby(\n    \"attacker_ip_enum\"\n).sum()\n\ndel df_tmp\n\n\nknown_watchers = top_watchers_df.columns.values\n\n# top_watchers_df = top_watchers_df.div(top_watchers_df.sum(axis=1), axis=0)\n# top_watchers_df = top_watchers_df.fillna(0)\n\ntop_watchers_df","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:10:13.108080Z","iopub.execute_input":"2024-01-09T23:10:13.109116Z","iopub.status.idle":"2024-01-09T23:11:30.306058Z","shell.execute_reply.started":"2024-01-09T23:10:13.109053Z","shell.execute_reply":"2024-01-09T23:11:30.304939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"attacker_contries_df = df.groupby(['attacker_ip_enum', 'attacker_country']).size().unstack('attacker_country', fill_value=0)\n\nvalid_countries = attacker_contries_df.columns.values\nattacker_contries_df[valid_countries]\n\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:11:30.307651Z","iopub.execute_input":"2024-01-09T23:11:30.308801Z","iopub.status.idle":"2024-01-09T23:11:55.376952Z","shell.execute_reply.started":"2024-01-09T23:11:30.308758Z","shell.execute_reply":"2024-01-09T23:11:55.375543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nwatcher_contries_df = df.groupby(['attacker_ip_enum', 'watcher_country']).size().unstack('watcher_country', fill_value=0)\nwatcher_contries_df = watcher_contries_df.add_prefix('watcher_country_')\n\nvalid_watcher_countries = watcher_contries_df.columns.values\nwatcher_contries_df[valid_watcher_countries]\n\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:11:55.378495Z","iopub.execute_input":"2024-01-09T23:11:55.379579Z","iopub.status.idle":"2024-01-09T23:12:12.259209Z","shell.execute_reply.started":"2024-01-09T23:11:55.379501Z","shell.execute_reply":"2024-01-09T23:12:12.258143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, we need to concatenate the features and split into Train (70%) and Validation (30%).","metadata":{}},{"cell_type":"code","source":"label_df = df.drop_duplicates([\"attacker_ip_enum\", \"label\"]).set_index(\n    \"attacker_ip_enum\"\n)[\"label\"]\n\ndataset = pd.concat(\n    [\n        one_hot_attack_service_df,\n        one_hot_attack_types_df,\n        ports_features_df,\n        top_attackers_as_df,\n        attacker_contries_df,\n#         one_hot_jarm,\n        headers_features_df,\n        top_watchers_df,\n        watcher_contries_df,\n        label_df,\n    ],\n    axis=1,\n    join=\"inner\",\n)\n\ndataset","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:12:12.260702Z","iopub.execute_input":"2024-01-09T23:12:12.261059Z","iopub.status.idle":"2024-01-09T23:12:15.258279Z","shell.execute_reply.started":"2024-01-09T23:12:12.261021Z","shell.execute_reply":"2024-01-09T23:12:15.257149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import f1_score\n\ntrain, test = train_test_split(dataset, test_size=0.3, random_state=42)\nX_train = train.drop([\"label\"], axis=1)\ny_train = train[\"label\"]\nX_test = test.drop([\"label\"], axis=1)\ny_test = test[\"label\"]\n\ndel train, test","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:12:15.259513Z","iopub.execute_input":"2024-01-09T23:12:15.259864Z","iopub.status.idle":"2024-01-09T23:12:16.454836Z","shell.execute_reply.started":"2024-01-09T23:12:15.259834Z","shell.execute_reply":"2024-01-09T23:12:16.453360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we will try some models, with different configurations. You can try looking for best parameters using the grid search or randomized search. Sometimes I got some errors from GridSearchCV and RandomizedSearchCV, they got stuck for hours. Despite that, the default HistGradientBoostingClassifier seems to perform the best.\n","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import ExtraTreesClassifier\nfrom sklearn.model_selection import GridSearchCV, KFold, RandomizedSearchCV\nfrom joblib import parallel_backend\n\n\nparams = {\n    'n_estimators': [100, 200, 500],\n    'criterion': ['gini', 'entropy'],\n    'min_samples_split': [2, 5, 10], \n    'class_weight': [None, {0:3, 1:7}],\n} \n\n# et =  RandomizedSearchCV(ExtraTreesClassifier(random_state=42), params, n_jobs=-1, cv=3, scoring='f1', verbose=10)\n# et =  RandomizedSearchCV(ExtraTreesClassifier(random_state=42), params, cv=5, scoring='f1', verbose=5)\n# et =  GridSearchCV(ExtraTreesClassifier(random_state=42), params, cv=5, scoring='f1', n_jobs=-1, verbose=10)\n# print('Training...')\n# et.fit(X_train, y_train)\n# print('Best score:', et.best_score_)\n# print('Best score:', et.best_params_)\n\net = ExtraTreesClassifier(class_weight={0:3, 1:7}, n_jobs=-1, random_state=42)\nprint('Training...')\net.fit(X_train, y_train)\n\nprint('Predicting...')\ny_pred = et.predict(X_test)\n\nprint('F-score:', f1_score(y_pred=y_pred, y_true=y_test))\nprint(classification_report(y_true=y_test, y_pred=y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:12:16.456441Z","iopub.execute_input":"2024-01-09T23:12:16.457402Z","iopub.status.idle":"2024-01-09T23:13:11.043638Z","shell.execute_reply.started":"2024-01-09T23:12:16.457366Z","shell.execute_reply":"2024-01-09T23:13:11.042463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import make_pipeline\n\n\nparams = {\n    'n_neighbors': [5, 10, 50],\n    'weights': ['uniform', 'distance'],\n    'p': [1, 2], \n} \n\n\n# knn = make_pipeline(StandardScaler(), RandomizedSearchCV(KNeighborsClassifier(), params, n_jobs=-1, cv=5, scoring='f1', verbose=3))\n# knn = make_pipeline(StandardScaler(), GridSearchCV(KNeighborsClassifier(), params, n_jobs=1, cv=3, scoring='f1', verbose=1))\n# print('Training...')\n# knn.fit(X_train, y_train)\n# print('Best score:', knn.best_score_)\n# print('Best score:', knn.best_params_)\n\nknn = make_pipeline(StandardScaler(), KNeighborsClassifier())\nprint('Training...')\nknn.fit(X_train, y_train)\n\nprint('Predicting...')\ny_pred = knn.predict(X_test)\n\nprint('F-score:', f1_score(y_pred=y_pred, y_true=y_test))\nprint(classification_report(y_true=y_test, y_pred=y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:13:11.045216Z","iopub.execute_input":"2024-01-09T23:13:11.046250Z","iopub.status.idle":"2024-01-09T23:14:58.011637Z","shell.execute_reply.started":"2024-01-09T23:13:11.046207Z","shell.execute_reply":"2024-01-09T23:14:58.010582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV, KFold, RepeatedStratifiedKFold, RandomizedSearchCV\nfrom sklearn.ensemble import RandomForestClassifier\n\nparams = {\n    'n_estimators': [10, 100, 200],\n    'criterion': ['gini', 'entropy'],\n    'min_samples_split': [1,2,5],\n    'min_samples_leaf': [1,2,5],\n    'max_leaf_nodes': [4,10,50,None],\n    'class_weight': [None, {0:3,1:7}],\n    \n}\n\n\n\n# rf =  RandomizedSearchCV(RandomForestClassifier(random_state=42), params, n_jobs=-1, cv=3, scoring='f1', verbose=10)\n# rf =  RandomizedSearchCV(RandomForestClassifier(random_state=42), params, cv=5, scoring='f1', verbose=5)\n# rf =  GridSearchCV(RandomForestClassifier(random_state=42), params, cv=5, scoring='f1', n_jobs=-1, verbose=10)\n# print('Training...')\n# rf.fit(X_train, y_train)\n# print('Best score:', rf.best_score_)\n# print('Best score:', rf.best_params_)\n\nrf = RandomForestClassifier(random_state=42)\nprint('Training...')\nrf.fit(X_train, y_train)\n\nprint('Predicting...')\ny_pred = rf.predict(X_test)\n\nprint('F-score:', f1_score(y_pred=y_pred, y_true=y_test))\nprint(classification_report(y_true=y_test, y_pred=y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:14:58.013070Z","iopub.execute_input":"2024-01-09T23:14:58.013550Z","iopub.status.idle":"2024-01-09T23:15:43.283173Z","shell.execute_reply.started":"2024-01-09T23:14:58.013517Z","shell.execute_reply":"2024-01-09T23:15:43.281953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import HistGradientBoostingClassifier\n\n\n\nparams = {\n    'learning_rate': [0.1, 0.01, 1],\n    'max_iter': [100,  50, 200],\n    'max_leaf_nodes': [31, 63, None], \n    'class_weight': [None, {0:3, 1:7}]\n} \n\n\n\n\n# hgb = RandomizedSearchCV(HistGradientBoostingClassifier(), params, n_jobs=-1, cv=5, scoring='f1', verbose=3)\n# hgb = GridSearchCV(HistGradientBoostingClassifier(), params, n_jobs=-1, cv=5, scoring='f1', verbose=3)\n\n# hgb.fit(X_train, y_train)\n# y_pred = hgb.predict(X_test) \n# print('F-score:', f1_score(y_pred=y_pred, y_true=y_test))\n# print(classification_report(y_true=y_test, y_pred=y_pred))\n# print('Best score:', hgb.best_score_)\n# print('Best score:', hgb.best_params_)\n\nhgb = HistGradientBoostingClassifier(class_weight={0:3, 1:7}, random_state=42)\nprint('Training...')\nhgb.fit(X_train, y_train)\nprint('Predicting...')\ny_pred = hgb.predict(X_test)\n\nprint('F-score:', f1_score(y_pred=y_pred, y_true=y_test))\nprint(classification_report(y_true=y_test, y_pred=y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:15:43.284964Z","iopub.execute_input":"2024-01-09T23:15:43.285415Z","iopub.status.idle":"2024-01-09T23:16:07.679782Z","shell.execute_reply.started":"2024-01-09T23:15:43.285370Z","shell.execute_reply":"2024-01-09T23:16:07.678344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission\n","metadata":{}},{"cell_type":"markdown","source":"After choosing the final model we will re-train it using the entire dataset. This will give more data to our model.","metadata":{}},{"cell_type":"code","source":"model = HistGradientBoostingClassifier(class_weight={0:3, 1:7})\nmodel.fit(dataset[sorted(dataset.columns)].drop('label', axis=1), dataset['label'])","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:16:07.681487Z","iopub.execute_input":"2024-01-09T23:16:07.682034Z","iopub.status.idle":"2024-01-09T23:16:27.238047Z","shell.execute_reply.started":"2024-01-09T23:16:07.681979Z","shell.execute_reply":"2024-01-09T23:16:27.236981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Then, we will prepare the features from the test set and make the predictions.","metadata":{}},{"cell_type":"code","source":"df_test = pd.read_parquet(\"/kaggle/input/vpn-classification/dataset_v2/test.parq\")\ndf_test.head()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:16:27.240888Z","iopub.execute_input":"2024-01-09T23:16:27.241738Z","iopub.status.idle":"2024-01-09T23:16:30.567433Z","shell.execute_reply.started":"2024-01-09T23:16:27.241688Z","shell.execute_reply":"2024-01-09T23:16:30.566638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"attack_types_df = (\n    df_test.attack_type.str.split(\":\", expand=True)\n    .rename(columns={0: \"service\", 1: \"type\"})\n    .set_index(df_test[\"attacker_ip_enum\"])\n)\n\none_hot_attack_service_df = pd.get_dummies(\n    # Dropping duplicated service before calling get dummies\n    attack_types_df.reset_index()\n    .drop_duplicates(subset=[\"attacker_ip_enum\", \"service\"])\n    .set_index(\"attacker_ip_enum\")[\"service\"]\n    # ,sparse=True\n)\none_hot_attack_service_df = one_hot_attack_service_df.groupby(\"attacker_ip_enum\").sum()\n\none_hot_attack_types_df = pd.get_dummies(\n    attack_types_df[\"type\"],\n    prefix='attack_type'\n    # sparse=True\n)\none_hot_attack_types_df = one_hot_attack_types_df.groupby(\"attacker_ip_enum\").sum()\n# We group by ip_enum and normalized by the number of attack to get a distribyution\none_hot_attack_types_df = one_hot_attack_types_df / one_hot_attack_types_df.sum(\n    1\n).values.reshape(-1, 1)\n\n\nattacker_contries_df = df_test.groupby(['attacker_ip_enum', 'attacker_country']).size().unstack('attacker_country', fill_value=0)\nattacker_contries_df = attacker_contries_df.div(attacker_contries_df.sum(axis=1), axis=0)\n\n# attacker_contries_df.reset_index()[valid_countries]\n\n# Remove os paises não conhecidos durante o treino e preenche com zero os paises ausentes no conjunto de validação\nval_countries = attacker_contries_df.columns.values\nmissing_countries = list(set(valid_countries) - set(val_countries)) \n\nattacker_contries_df_tmp = attacker_contries_df.copy()\nattacker_contries_df_tmp[missing_countries] = 0.0\nattacker_contries_df = attacker_contries_df_tmp[valid_countries]\n\ndel attacker_contries_df_tmp\nattacker_contries_df\n\n\ndf_tmp = df_test.drop_duplicates(subset=[\"attacker_ip_enum\", \"attacker_as_name\"]).copy()\ndf_tmp['top_as_attacker'] = df_tmp['attacker_as_name'].apply(lambda x : x if x in top_as_attacker else 0)\n\ntop_attackers_as_df = pd.get_dummies(\n    df_tmp.set_index('attacker_ip_enum')['top_as_attacker'],\n    prefix='as_name'\n).groupby(\n    \"attacker_ip_enum\"\n).sum()\n\n# top_attackers_as_df.head()\n\nknwon_as = [ as_name.replace('as_name_', '') for as_name in  top_attackers_as_df.columns ]\n# knwon_as\n\nmissing_as = list(set(top_as_attacker) - set(knwon_as))\nmissing_as = [ f'as_name_{as_name}' for as_name in missing_as]\n\ntop_attackers_as_df_tmp = top_attackers_as_df.copy()\ntop_attackers_as_df_tmp[missing_as] = 0.0\ntop_attackers_as_df = top_attackers_as_df_tmp\n\ndel top_attackers_as_df_tmp\n\n\n\nwatcher_contries_df = df_test.groupby(['attacker_ip_enum', 'watcher_country']).size().unstack('watcher_country', fill_value=0)\nwatcher_contries_df = watcher_contries_df.add_prefix('watcher_country_')\n\nknwon_watcher_countries = watcher_contries_df.columns \nmissing_countries = list(set(valid_watcher_countries) - set(knwon_watcher_countries)) \n\n\nwatcher_contries_df_tmp = watcher_contries_df.copy()\nwatcher_contries_df_tmp[missing_countries] = 0.0\nwatcher_contries_df = watcher_contries_df_tmp[valid_watcher_countries]\n\n\n\ndf_tmp = df_test.drop_duplicates(subset=[\"attacker_ip_enum\", \"watcher_uuid_enum\"]).copy()\ndf_tmp['top_watcher_uuid_enum'] = df_tmp['watcher_uuid_enum'].apply(lambda x : x if x in top_watcher_uuid_enum else 0)\n\ntop_watchers_df = pd.get_dummies(\n    df_tmp.set_index('attacker_ip_enum')['top_watcher_uuid_enum'],\n    prefix='watcher'\n).groupby(\n    \"attacker_ip_enum\"\n).sum()\n\ndel df_tmp\n\n\nknwon_watcher = [ watcher.replace('watcher_', '') for watcher in  top_watchers_df.columns ]\n# knwon_watcher\n\nmissing_watcher = list(set(top_watcher_uuid_enum) - set(knwon_watcher))\nmissing_watcher = [ f'watcher_{watcher}' for watcher in missing_watcher]\n\ntop_watchers_df_tmp = top_watchers_df.copy()\ntop_watchers_df_tmp[missing_watcher] = 0.0\ntop_watchers_df = top_watchers_df_tmp","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:16:30.569153Z","iopub.execute_input":"2024-01-09T23:16:30.569781Z","iopub.status.idle":"2024-01-09T23:17:32.615965Z","shell.execute_reply.started":"2024-01-09T23:16:30.569748Z","shell.execute_reply":"2024-01-09T23:17:32.614676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nX_val_df = pd.concat(\n    [\n        one_hot_attack_service_df,\n        one_hot_attack_types_df,\n        ports_features_df,\n        top_attackers_as_df,\n        attacker_contries_df,\n#         one_hot_jarm,\n        headers_features_df,\n        top_watchers_df,\n        watcher_contries_df,\n    ],\n    axis=1,\n    join=\"inner\",\n)\n\nprint(len(X_val_df))\nX_val_df.head()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:17:32.617850Z","iopub.execute_input":"2024-01-09T23:17:32.618209Z","iopub.status.idle":"2024-01-09T23:17:32.799405Z","shell.execute_reply.started":"2024-01-09T23:17:32.618177Z","shell.execute_reply":"2024-01-09T23:17:32.798093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nprediction = model.predict(X_val_df[sorted(X_val_df.columns)])\nprediction = pd.Series(prediction, index=X_val_df.index).rename(\"prediction\")\nprediction","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:17:32.801372Z","iopub.execute_input":"2024-01-09T23:17:32.802269Z","iopub.status.idle":"2024-01-09T23:17:33.230907Z","shell.execute_reply.started":"2024-01-09T23:17:32.802221Z","shell.execute_reply":"2024-01-09T23:17:33.229743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print((prediction==0).sum(), (prediction == 1).sum())","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:17:33.232892Z","iopub.execute_input":"2024-01-09T23:17:33.233362Z","iopub.status.idle":"2024-01-09T23:17:33.241488Z","shell.execute_reply.started":"2024-01-09T23:17:33.233320Z","shell.execute_reply":"2024-01-09T23:17:33.240004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction.to_csv(\"submission.csv\")\nprint('Done!')","metadata":{"execution":{"iopub.status.busy":"2024-01-09T23:17:33.243513Z","iopub.execute_input":"2024-01-09T23:17:33.244259Z","iopub.status.idle":"2024-01-09T23:17:33.380956Z","shell.execute_reply.started":"2024-01-09T23:17:33.244212Z","shell.execute_reply":"2024-01-09T23:17:33.379732Z"},"trusted":true},"execution_count":null,"outputs":[]}]}