{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59865,"databundleVersionId":6660280,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Binary Classification of VPN IP Address\n\nThis challenge is a binary Classification challenge where we have to predict whether an attack comes using a VPN/Proxy or not.\n\nFor this challenge I used features provided by the organizers, time related features and feature related to the attaccker.\n\nI trained 5 RandomForestClassifiers on 5 folds of the dataset and averaged the predictions.\n\nTo address data imbalance I didn't use augmentation techniques, I only changed the threshold to consider a sample positive, instead of probability>0.5 I used the average of the best validation threshold of the 5 folds considered.\n\nThe folding is not seeded so the results vary from run to run. The averaging keeps the score quite stable, but results may vary.\n\nFor any doubts feel free to leave a comment.\n\n\n## Happy kaggling!","metadata":{}},{"cell_type":"markdown","source":"# Imports\n","metadata":{}},{"cell_type":"code","source":"#Data handling\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import KFold\n\n#Data loading for shodan features\nimport ast\n\n#Evaluation\nfrom sklearn.metrics import f1_score\n\n#Modelling \nfrom sklearn.ensemble import RandomForestClassifier","metadata":{"execution":{"iopub.status.busy":"2024-01-08T15:09:04.987080Z","iopub.execute_input":"2024-01-08T15:09:04.987506Z","iopub.status.idle":"2024-01-08T15:09:07.560586Z","shell.execute_reply.started":"2024-01-08T15:09:04.987474Z","shell.execute_reply":"2024-01-08T15:09:07.559144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Reading","metadata":{}},{"cell_type":"code","source":"df=pd.read_parquet(\"/kaggle/input/vpn-classification/dataset_v2/train.parq\")\ndf[\"attack_time\"]=df[\"attack_time\"].astype(int)# convert timestamp to integer\ndf.describe(include=\"all\")","metadata":{"execution":{"iopub.status.busy":"2024-01-08T15:09:07.563848Z","iopub.execute_input":"2024-01-08T15:09:07.564652Z","iopub.status.idle":"2024-01-08T15:09:37.098646Z","shell.execute_reply.started":"2024-01-08T15:09:07.564595Z","shell.execute_reply":"2024-01-08T15:09:37.097346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Time series features\n\nGenerate time series features, in particular I use 3 features:\n1. attack_time_span: The duration of the attack (max time - min time)\n2. attack_time_variance: the standard devidation","metadata":{}},{"cell_type":"code","source":"def get_time_features(df):\n    df_time_features=df[[\"attacker_ip_enum\",\"attack_time\"]].groupby(\"attacker_ip_enum\").agg(list)\n    df_time_features[\"attack_time_span\"]=df_time_features[\"attack_time\"].apply(lambda x: np.max(x).astype(np.int64)-np.min(x).astype(np.int64))/1000000\n    df_time_features[\"attack_time_std\"]=df_time_features[\"attack_time\"].apply(lambda x: np.std(np.array(x,dtype=np.int64)))/1e10\n    df_time_features[\"count\"]=df_time_features[\"attack_time\"].str.len()\n    df_time_features.drop(\"attack_time\",axis=1,inplace=True)\n    return df_time_features","metadata":{"execution":{"iopub.status.busy":"2024-01-08T15:09:37.100553Z","iopub.execute_input":"2024-01-08T15:09:37.101050Z","iopub.status.idle":"2024-01-08T15:09:37.111821Z","shell.execute_reply.started":"2024-01-08T15:09:37.101008Z","shell.execute_reply":"2024-01-08T15:09:37.110113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Provided features","metadata":{}},{"cell_type":"code","source":"def get_one_hot_features_attack(df):\n    attack_types_df = (\n        df.attack_type.str.split(\":\", expand=True)\n        .rename(columns={0: \"service\", 1: \"type\"})\n        .set_index(df[\"attacker_ip_enum\"])\n    )\n    one_hot_attack_service_df = pd.get_dummies(\n        # Dropping duplicated service before calling get dummies\n        attack_types_df.reset_index()\n        .drop_duplicates(subset=[\"attacker_ip_enum\", \"service\"])\n        .set_index(\"attacker_ip_enum\")[\"service\"]\n        # ,sparse=True\n    )\n    one_hot_attack_service_df = one_hot_attack_service_df.groupby(\"attacker_ip_enum\").sum()\n    # We group by ip_enum and keep only strictly positive values\n    one_hot_attack_service_df = (one_hot_attack_service_df >= 1).astype(int)\n    one_hot_attack_types_df = pd.get_dummies(\n        attack_types_df[\"type\"],\n        # sparse=True\n    )\n    one_hot_attack_types_df = one_hot_attack_types_df.groupby(\"attacker_ip_enum\").sum()\n    # We group by ip_enum and normalized by the number of attack to get a distribution\n    one_hot_attack_types_df = one_hot_attack_types_df / one_hot_attack_types_df.sum(\n        1\n    ).values.reshape(-1, 1)\n    return one_hot_attack_service_df,one_hot_attack_types_df","metadata":{"execution":{"iopub.status.busy":"2024-01-08T15:09:37.115475Z","iopub.execute_input":"2024-01-08T15:09:37.116275Z","iopub.status.idle":"2024-01-08T15:09:37.128146Z","shell.execute_reply.started":"2024-01-08T15:09:37.116221Z","shell.execute_reply":"2024-01-08T15:09:37.126632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Attacker features","metadata":{}},{"cell_type":"code","source":"# use frequency of an attacker (using the name) as a feature can help identify if there is any pattern of frequent attacker name that is actually an attack\n# Since an ip can be linked to more than one attacker name we compute max,min and average of the frequency to capture different characteristics of the various attackers\nfreq_attacker=df[[\"attacker_ip_enum\",\"attacker_as_name\"]].drop_duplicates()[\"attacker_as_name\"].value_counts().reset_index().rename(columns={\"count\":\"frequency\"})\n\nattacker_feats=pd.merge(df[[\"attacker_ip_enum\",\"attacker_as_name\"]],freq_attacker,on=\"attacker_as_name\",how=\"left\")\n\nattacker_feats=attacker_feats[[\"attacker_ip_enum\",\"frequency\"]].groupby(\"attacker_ip_enum\").agg([\"min\",\"max\",\"mean\"])[\"frequency\"].reset_index()\nattacker_feats.columns=['attacker_ip_enum' if 'attacker_ip_enum' in c else f\"attacker_{'_'.join(list(c))}\" for c in attacker_feats.columns]\nattacker_feats=attacker_feats.set_index(\"attacker_ip_enum\")\nattacker_feats.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-01-08T15:09:37.129992Z","iopub.execute_input":"2024-01-08T15:09:37.130538Z","iopub.status.idle":"2024-01-08T15:09:50.158736Z","shell.execute_reply.started":"2024-01-08T15:09:37.130471Z","shell.execute_reply":"2024-01-08T15:09:50.156286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use frequency and average label (TE) of data points of attacker country can provide additional information for the model\n# Since an ip can be linked to more than one attacker name we compute max,min and average of the frequency/TE to capture different characteristics of the various attackers\nfreq_attacker_country=df[[\"attacker_ip_enum\",\"attacker_country\"]].drop_duplicates()[\"attacker_country\"].value_counts().reset_index().rename(columns={\"count\":\"frequency\"})\nTE_attacker_country=df[[\"attacker_ip_enum\",\"attacker_country\",\"label\"]].drop_duplicates()[[\"attacker_country\",\"label\"]].groupby(\"attacker_country\").mean().reset_index().rename(columns={\"label\":\"TE\"})\nfreq_attacker_country=pd.merge(freq_attacker_country,TE_attacker_country,on=\"attacker_country\",how=\"left\")\nattacker_country_feats=pd.merge(df[[\"attacker_ip_enum\",\"attacker_country\"]],freq_attacker_country,on=\"attacker_country\",how=\"left\")\n\nattacker_country_feats=attacker_country_feats[[\"attacker_ip_enum\",\"frequency\",\"TE\"]].groupby(\"attacker_ip_enum\").agg([\"min\",\"max\",\"mean\"])[[\"frequency\",\"TE\"]].reset_index()\nattacker_country_feats.columns=['attacker_ip_enum' if 'attacker_ip_enum' in c else f\"attacker_country_{'_'.join(c)}\" for c in attacker_country_feats.columns]\nattacker_country_feats=attacker_country_feats.set_index(\"attacker_ip_enum\")","metadata":{"execution":{"iopub.status.busy":"2024-01-08T15:09:50.162880Z","iopub.execute_input":"2024-01-08T15:09:50.163442Z","iopub.status.idle":"2024-01-08T15:10:19.052300Z","shell.execute_reply.started":"2024-01-08T15:09:50.163397Z","shell.execute_reply":"2024-01-08T15:10:19.050206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Shodan Features\n\nI added different ports to the list provided using some of the most common ports in the dataset.\n\nI also counted the presence of header_hush, jarm and ja3s in the various data entries. ","metadata":{}},{"cell_type":"code","source":"REFERENCE_PORTS = set((\"21\",\"22\", \"53\", \"80\", \"443\",'465','995','587',\"993\",  \"7777\",\"8000\",\"8080\")) # Most common ports\nREFERENCE_PROTOCOLS = set((\"tcp\",\"udp\")) # most common protocols\nALL_PORTS=set()\n\ndef extract_port_numbers(value: dict):\n    return set(key.split(\"/\")[0] for key in value.keys())\ndef extract_protocol(value: dict):\n    return list(key.split(\"/\")[1] for key in value.keys())\ndef get_distinct_count(value):\n    return len(set(value))\ndef count_key(key):\n    def impl(value):\n        return sum([1 if v[key] is not None else 0 for v in value.values()])\n    return impl\n\nshodan_df = pd.read_csv(\n    \"/kaggle/input/vpn-classification/dataset_v2/shodan_df_hashed.csv\",\n    dtype={\"attacker_ip_enum\": \"int32\"},\n    index_col=\"attacker_ip_enum\",\n)\nshodan_df=shodan_df.loc[shodan_df[\"shodan_info\"]!=\"{}\"]\nshodan_df[\"shodan_info\"] = shodan_df[\"shodan_info\"].map(ast.literal_eval)\nshodan_df.iloc[1,0]","metadata":{"execution":{"iopub.status.busy":"2024-01-08T15:10:19.055632Z","iopub.execute_input":"2024-01-08T15:10:19.056748Z","iopub.status.idle":"2024-01-08T15:10:24.382015Z","shell.execute_reply.started":"2024-01-08T15:10:19.056685Z","shell.execute_reply":"2024-01-08T15:10:24.380637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shodan_df[\"shodan_open_ports\"] = shodan_df[\"shodan_info\"].map(extract_port_numbers)\nshodan_df[\"shodan_protocols\"] = shodan_df[\"shodan_info\"].map(extract_protocol)\n\nopen_ports_count = shodan_df[\"shodan_open_ports\"].map(len).rename(\"open_ports_count\")\nprotocol_count = shodan_df[\"shodan_protocols\"].map(get_distinct_count).rename(\"distinct_protocol_count\")\ntcp_count = shodan_df[\"shodan_protocols\"].apply(lambda x: x.count(\"tcp\")).rename(\"distinct_tcp_count\")\nudp_count = shodan_df[\"shodan_protocols\"].apply(lambda x: x.count(\"udp\")).rename(\"distinct_udp_count\")\nheaders_hash_count = shodan_df[\"shodan_info\"].apply(count_key(\"headers_hash\")).rename(\"headers_hash_count\")\njarm_count = shodan_df[\"shodan_info\"].apply(count_key(\"jarm\")).rename(\"jarm_count\")\nja3s_count = shodan_df[\"shodan_info\"].apply(count_key(\"ja3s\")).rename(\"ja3s_count\")\n# We select the reference port by doing a set intersection\nreference_ports_df = shodan_df[\"shodan_open_ports\"].map(lambda x: REFERENCE_PORTS & x)\n\none_hot_reference_ports_df = pd.get_dummies(reference_ports_df.explode(), prefix=\"port\")\n# Aggregate the dummies by IP\none_hot_reference_ports_df = one_hot_reference_ports_df.groupby(\n    \"attacker_ip_enum\"\n).sum()\n\n\n# Final feature for port: count and one hot table of reference port\nshodan_features = pd.concat([one_hot_reference_ports_df, open_ports_count,tcp_count,udp_count,protocol_count,headers_hash_count,\n                            jarm_count,ja3s_count], axis=1)\n# Checking results\nshodan_features[(shodan_features > 0).any(axis=1)].head()","metadata":{"execution":{"iopub.status.busy":"2024-01-08T15:10:24.383720Z","iopub.execute_input":"2024-01-08T15:10:24.384139Z","iopub.status.idle":"2024-01-08T15:10:26.194163Z","shell.execute_reply.started":"2024-01-08T15:10:24.384101Z","shell.execute_reply":"2024-01-08T15:10:26.192536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Merge features\n\nI combine all features in a single dataset","metadata":{}},{"cell_type":"code","source":"label_df = df.drop_duplicates([\"attacker_ip_enum\", \"label\"]).set_index(\n    \"attacker_ip_enum\"\n)[\"label\"]\n\ndataset = pd.concat(\n    [\n        *get_one_hot_features_attack(df),\n        get_time_features(df),\n        attacker_feats,\n        attacker_country_feats,\n        label_df,\n        \n    ],\n    axis=1,\n    join=\"inner\",\n)\ndataset=dataset.merge(shodan_features,how=\"left\",on=\"attacker_ip_enum\").fillna(-10000)\n\ndataset.to_csv(\"train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-01-08T15:10:26.196032Z","iopub.execute_input":"2024-01-08T15:10:26.196442Z","iopub.status.idle":"2024-01-08T15:13:06.433480Z","shell.execute_reply.started":"2024-01-08T15:10:26.196408Z","shell.execute_reply":"2024-01-08T15:13:06.432002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test set preparation\n\nThe same preprocessing done for train set is performed for test set","metadata":{}},{"cell_type":"code","source":"df = pd.read_parquet(\"/kaggle/input/vpn-classification/dataset_v2/test.parq\")\ndf[\"attack_time\"]=df[\"attack_time\"].astype(int)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-08T15:13:06.438404Z","iopub.execute_input":"2024-01-08T15:13:06.439035Z","iopub.status.idle":"2024-01-08T15:13:10.224041Z","shell.execute_reply.started":"2024-01-08T15:13:06.438989Z","shell.execute_reply":"2024-01-08T15:13:10.223064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"attacker_feats=pd.merge(df[[\"attacker_ip_enum\",\"attacker_as_name\"]],freq_attacker,on=\"attacker_as_name\",how=\"left\")\n\nattacker_feats=attacker_feats[[\"attacker_ip_enum\",\"frequency\"]].groupby(\"attacker_ip_enum\").agg([\"min\",\"max\",\"mean\"])[\"frequency\"].reset_index()\nattacker_feats.columns=['attacker_ip_enum' if 'attacker_ip_enum' in c else f\"attacker_{'_'.join(list(c))}\" for c in attacker_feats.columns]\nattacker_feats=attacker_feats.set_index(\"attacker_ip_enum\")\nattacker_feats.head(10)\n\nattacker_country_feats=pd.merge(df[[\"attacker_ip_enum\",\"attacker_country\"]],freq_attacker_country,on=\"attacker_country\",how=\"left\")\n\nattacker_country_feats=attacker_country_feats[[\"attacker_ip_enum\",\"frequency\",\"TE\"]].groupby(\"attacker_ip_enum\").agg([\"min\",\"max\",\"mean\"])[[\"frequency\",\"TE\"]].reset_index()\nattacker_country_feats.columns=['attacker_ip_enum' if 'attacker_ip_enum' in c else f\"attacker_country_{'_'.join(c)}\" for c in attacker_country_feats.columns]\nattacker_country_feats=attacker_country_feats.set_index(\"attacker_ip_enum\")","metadata":{"execution":{"iopub.status.busy":"2024-01-08T15:13:10.225734Z","iopub.execute_input":"2024-01-08T15:13:10.226915Z","iopub.status.idle":"2024-01-08T15:13:25.654802Z","shell.execute_reply.started":"2024-01-08T15:13:10.226846Z","shell.execute_reply":"2024-01-08T15:13:25.653681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_val_df = pd.concat(\n    [\n       *get_one_hot_features_attack(df),\n        get_time_features(df),\n        attacker_feats,\n        attacker_country_feats\n    ],\n    axis=1,\n    join=\"inner\",\n)\n\nX_val_df=X_val_df.merge(shodan_features,how=\"left\",on=\"attacker_ip_enum\").fillna(-10000)\n\nX_val_df.to_csv(\"test.csv\")\n\ntest=X_val_df\ntest_index=test.index","metadata":{"execution":{"iopub.status.busy":"2024-01-08T15:13:25.656348Z","iopub.execute_input":"2024-01-08T15:13:25.657374Z","iopub.status.idle":"2024-01-08T15:14:12.355515Z","shell.execute_reply.started":"2024-01-08T15:13:25.657340Z","shell.execute_reply":"2024-01-08T15:14:12.354021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Training & inference\n\nI trained 5 RandomForestClassifiers on 5 folds of the dataset to use as much data as possible and averaged the results.\n\nOne particular aspect of the solution is that for each fold I compute what prediction threshold provides the highest f1 score, and I use the average of those threshold to binarize the final prediction.","metadata":{}},{"cell_type":"code","source":"# Parameters found with cvgridsearch\nbest_params={'class_weight': None, 'max_depth': 25, 'max_features': 'sqrt'}\n# Best model trained\nestimator_class=RandomForestClassifier","metadata":{"execution":{"iopub.status.busy":"2024-01-08T15:14:12.357691Z","iopub.execute_input":"2024-01-08T15:14:12.358174Z","iopub.status.idle":"2024-01-08T15:14:12.364710Z","shell.execute_reply.started":"2024-01-08T15:14:12.358129Z","shell.execute_reply":"2024-01-08T15:14:12.363319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction= 0\naverage_T=0\n\nsplits=5\nkf = KFold(n_splits=splits)\n\nfor i, (train_index, valid_index) in enumerate(kf.split(dataset)):\n    best_score=0\n    # Split in train and validation\n    X_train,y_train=dataset.iloc[train_index].drop([\"label\"], axis=1),dataset[\"label\"].iloc[train_index]\n    X_valid,y_valid=dataset.iloc[valid_index].drop([\"label\"], axis=1),dataset[\"label\"].iloc[valid_index]\n    \n    #Scale data (useful for other models)\n    sc = StandardScaler()\n    X_train = sc.fit_transform(X_train)\n    X_valid = sc.transform(X_valid)\n    \n    # Code to account for some models that don't have n_jobs parameters\n    try:\n        model = estimator_class(n_jobs=-1, **best_params)\n    except:\n        model = estimator_class(**best_params)       \n        \n    # Train the model\n    model.fit(X_train,y_train)\n    \n    #Predict the output for validation set\n    y_pred = model.predict_proba(X_valid)[:,1]\n    \n    # Find the best threshold for result binarization\n    for j in range(20,80):\n        score=f1_score(y_pred=(y_pred>j/100), y_true=y_valid)\n        if score>best_score:\n            best_score=score\n            T=j/100\n    # Print best threshold for the binarization of the current fold\n    print(i,\" threshold: \",T,\"--->\",score)\n        \n    # Prepare test data with equal transformation\n    X_test=sc.transform(test)\n    \n    # Perform prediction on the test set\n    prediction += pd.Series(model.predict_proba(X_test)[:,1]/splits, index=test_index).rename(\"prediction\")\n    \n    # Update threshold\n    average_T+=T/splits\n    \n# Binarize the prediction and convert data type\nprediction=(prediction>average_T).astype(\"uint8\")","metadata":{"execution":{"iopub.status.busy":"2024-01-08T15:28:34.425730Z","iopub.execute_input":"2024-01-08T15:28:34.426647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save predictions for submission","metadata":{}},{"cell_type":"code","source":"prediction.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-01-08T15:14:57.372044Z","iopub.execute_input":"2024-01-08T15:14:57.372983Z","iopub.status.idle":"2024-01-08T15:14:57.530157Z","shell.execute_reply.started":"2024-01-08T15:14:57.372930Z","shell.execute_reply":"2024-01-08T15:14:57.528835Z"},"trusted":true},"execution_count":null,"outputs":[]}]}