{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from sklearn.datasets import load_iris\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import preprocessing as pre\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, RobustScaler\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import classification_report\nimport tracemalloc\nfrom tabulate import tabulate\nfrom sklearn.decomposition import *\nfrom sklearn.tree import *\nfrom sklearn.feature_extraction.text import *\nfrom sklearn import preprocessing\nfrom sklearn import utils\nimport tensorflow as tf\nfrom keras.models import Sequential\nfrom keras.layers import Dense, LSTM\nfrom sklearn.metrics import mean_squared_error\n# from sklearn.metrics import mean_absolute_percentage_error\nfrom sklearn.model_selection import TimeSeriesSplit\nfrom keras.layers import SimpleRNN\nfrom keras.layers import Dropout\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.cluster import KMeans\nfrom sklearn.tree import DecisionTreeClassifier #Decision Tree\nfrom sklearn import metrics #accuracy measure\nfrom datetime import datetime, time\nimport seaborn as sn","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-05-20T08:18:12.407547Z","iopub.execute_input":"2023-05-20T08:18:12.408286Z","iopub.status.idle":"2023-05-20T08:18:16.580351Z","shell.execute_reply.started":"2023-05-20T08:18:12.408213Z","shell.execute_reply":"2023-05-20T08:18:16.579504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df = pd.read_csv(\"/kaggle/input/ip-network-traffic-flows-labeled-with-87-apps/Dataset-Unicauca-Version2-87Atts.csv\", nrows = 300000)\ndf = pd.read_csv(\"/kaggle/input/ip-network-traffic-flows-labeled-with-87-apps/Dataset-Unicauca-Version2-87Atts.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2023-05-20T08:18:16.5819Z","iopub.execute_input":"2023-05-20T08:18:16.582193Z","iopub.status.idle":"2023-05-20T08:19:24.095021Z","shell.execute_reply.started":"2023-05-20T08:18:16.582146Z","shell.execute_reply":"2023-05-20T08:19:24.094107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the display options\npd.set_option('display.max_rows', None)  # Print all rows\npd.set_option('display.max_columns', None)  # Print all columns\ndf","metadata":{"execution":{"iopub.status.busy":"2023-05-20T08:19:24.096302Z","iopub.execute_input":"2023-05-20T08:19:24.096602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df = df.apply(pd.to_numeric, errors='coerce')\n# df = df.dropna()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = df.drop(['Timestamp', 'Label', 'Flow.ID', 'Bwd.Packet.Length.Min', 'Flow.IAT.Mean', 'Flow.IAT.Std', 'Flow.IAT.Min', 'Fwd.IAT.Std', 'Min.Packet.Length', 'RST.Flag.Count', 'PSH.Flag.Count', 'CWE.Flag.Count', 'ECE.Flag.Count', 'Down.Up.Ratio', 'Source.IP', 'Source.Port', 'Destination.IP', 'Destination.Port', 'Flow.Duration', 'Total.Fwd.Packets', 'Total.Backward.Packets', 'Total.Length.of.Fwd.Packets', 'Total.Length.of.Bwd.Packets', 'Fwd.PSH.Flags', 'Bwd.PSH.Flags', 'Fwd.URG.Flags', 'Bwd.URG.Flags', 'Fwd.Avg.Bytes.Bulk','Fwd.Avg.Packets.Bulk', 'Fwd.Avg.Bulk.Rate', 'Bwd.Avg.Bytes.Bulk', 'Bwd.Avg.Packets.Bulk'], axis = 1)\n#drop 30 columns, 57 left\ndata['ProtocolName'] = pd.to_numeric(df['ProtocolName'], errors='coerce')\n# heat = data.head(100000)\n# print(heat.dtypes)\ncorr_matrix = data.corr()\nplt.figure(figsize=(100, 100))\nsn.heatmap(corr_matrix, cmap='coolwarm', annot=True, fmt=\".2f\", linewidths=0.5)\nplt.xlabel('X-axis', fontsize=20, color='white')\nplt.ylabel('Y-axis', fontsize=20, color='white')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"protocol = df['ProtocolName'].unique()\nprotocol","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Cái này có thể là một khía cạnh khai thác sau này","metadata":{}},{"cell_type":"code","source":"a =[]\nfor i in range (0,len(protocol)-1):\n    a.append(df[df['ProtocolName'] == protocol[i]].shape[0])\n    print(a[i], protocol[i])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ip = df[['Flow.ID', 'Source.IP', 'Source.Port', 'Destination.IP', 'Destination.Port', 'Timestamp', 'Flow.Duration', 'Total.Fwd.Packets', 'Total.Backward.Packets', 'Total.Length.of.Fwd.Packets', 'Total.Length.of.Bwd.Packets', 'Fwd.Packet.Length.Max', 'Bwd.Packet.Length.Max', 'Flow.IAT.Max', 'Bwd.IAT.Min', 'Max.Packet.Length', 'Max.Packet.Length', 'Average.Packet.Size', 'Idle.Mean']]\nip","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ip = ip.apply(pd.to_numeric, errors='coerce')\n# ip = ip.dropna()\n# print(ip)\n# corr_matrix = ip.corr()\n# plt.figure(figsize=(100, 100))\n# sn.heatmap(corr_matrix, cmap='coolwarm', annot=True, fmt=\".2f\", linewidths=0.5)\n# plt.xlabel('X-axis', fontsize=20, color='white')\n# plt.ylabel('Y-axis', fontsize=20, color='white')\n# plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ip.isnull().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for j in range(0,len(ip)):\n    i = ip.at[j, \"Timestamp\"]\n    i = i[:10] + \" \" + i[10:]\n    ip.at[j, \"Timestamp\"] = i","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ip['Date'] = pd.to_datetime(ip['Timestamp']).dt.date\nip['Time'] = pd.to_datetime(ip['Timestamp']).dt.time\nip['Date'] = pd.to_datetime(ip['Date']) \nip","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"source_ip_number = ip['Source.IP'].unique()\nlen(source_ip_number)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = 1\nn = 3\nipn = ip[ip['Source.IP']== source_ip_number[n]]\ndes_ip_number = ipn['Destination.IP'].unique()\n\nplt.figure(figsize=(20,10))\nipn.plot(x='Time',y='Total.Fwd.Packets')\nplt.title('Traffic by number of packets', fontsize=15)\nplt.ylabel('Number of packets')\nplt.show()\n\nplt.figure(figsize=(20,10))\nipn.plot(x='Time',y='Total.Length.of.Fwd.Packets')\nplt.title('Traffic by total length of packets', fontsize=15)\nplt.ylabel('Total length of packets')\nplt.show()\n\nplt.figure(figsize=(20,10))\nipn.plot(x='Time',y='Total.Backward.Packets')\nplt.title('Traffic by number of packets', fontsize=15)\nplt.ylabel('Number of packets')\nplt.show()\n\nplt.figure(figsize=(20,10))\nipn.plot(x='Time',y='Total.Length.of.Bwd.Packets')\nplt.title('Traffic by total length of packets', fontsize=15)\nplt.ylabel('Total length of packets')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Ta nhận thấy mật độ dữ liệu ở là không đều nhau và có sự chênh lệch lớn -> sử dụng mean alsolute error sẽ tốt hơn là mean square error","metadata":{}},{"cell_type":"code","source":"import statsmodels.api as sm\nfrom statsmodels.stats.diagnostic import linear_reset\n\nfor col in df.columns:\n    try:\n        # Define dependent and independent variables\n        y = df[col]\n        X = df[['Source.Port','Destination.Port', 'Flow.Duration', 'Total.Fwd.Packets', 'Total.Backward.Packets', 'Total.Length.of.Fwd.Packets', 'Total.Length.of.Bwd.Packets']]\n\n        # Fit linear regression model\n        model = sm.OLS(y, sm.add_constant(X)).fit()\n\n        # Run Ramsey RESET test with a power transformation of order 2\n        reset = linear_reset(model, power=2)\n\n        # Print test results\n        print(col)\n        print(reset.summary())\n    except:\n        pass # doing nothing on exception","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The test statistic is typically an F-statistic, and the p-value associated with the test is the probability of observing such a large F-statistic under the null hypothesis. If the p-value is less than a pre-specified significance level, typically 0.05 or 0.01, we reject the null hypothesis and conclude that there is evidence of omitted variable bias or nonlinearity in the model.","metadata":{}},{"cell_type":"code","source":"ip = ip[['Source.IP', 'Source.Port', 'Destination.IP', 'Destination.Port', 'Flow.Duration', 'Total.Fwd.Packets', 'Total.Backward.Packets', 'Total.Length.of.Fwd.Packets', 'Total.Length.of.Bwd.Packets', 'Date', 'Time']]\nsource_ip_number = ip['Source.IP'].unique()\nprint(len(source_ip_number))\nm = 1\nn = 3\nipn = ip[ip['Source.IP']== source_ip_number[n]]\ndes_ip_number = ipn['Destination.IP'].unique()\nipn","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# count = ip.groupby('Destination.IP')['Destination.IP'].count()\ncount = ip.groupby('Source.IP')['Source.IP'].count()\nprint(count.max(), count.idxmax())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ipmax = ip[ip['Destination.IP']== '10.200.7.8']\nipmax = ip[ip['Source.IP']== '10.200.7.218']\nipmax","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data = ipn[['Source.Port', 'Destination.Port', 'Flow.Duration', 'Total.Fwd.Packets', 'Total.Backward.Packets', 'Total.Length.of.Fwd.Packets', 'Total.Length.of.Bwd.Packets']]\ndata = ip[['Source.IP','Destination.IP','Source.Port', 'Destination.Port', 'Flow.Duration', 'Total.Fwd.Packets', 'Total.Backward.Packets', 'Total.Length.of.Fwd.Packets', 'Total.Length.of.Bwd.Packets', 'Date', 'Time']]\ndata['Date'] = pd.to_datetime(data['Date'])\n# data['Time'] = data['Time'].apply(pd.Timestamp)\n\ndata['Day'] = data['Date'].dt.day\ndata['Month'] = data['Date'].dt.month\ndata['Year'] = data['Date'].dt.year\ndata = data.drop(['Date'], axis = 1)\n\n# separate the time column into hour, minute, and second columns\n# data['Hour'] = data['Time'].dt.strftime('%H').astype(int)\n# data['Minute'] = data['Time'].dt.strftime('%M').astype(int)\n# data['Second'] = data['Time'].dt.strftime('%S').astype(int)\ndata['Time_str'] = data['Time'].apply(lambda t: t.strftime('%H:%M:%S'))\ndata['Time_float'] = data['Time_str'].apply(lambda s: (datetime.strptime(s, '%H:%M:%S')))\ndata = data.drop(['Time', 'Time_str'], axis = 1)\n\ndata","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nimport collections\n\nimport matplotlib\nimport matplotlib.pyplot as plt\n\n# import torch\n# import torch.nn as nn\n# import torch.nn.functional as F\n\n# from torch_geometric.data import Data\n# # from torch_geometric.transforms import AddTrainValTestMask as masking\n# from torch_geometric.utils.convert import to_networkx\n# from torch_geometric.nn import GCNConv\n\nimport networkx as nx","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"G1 = nx.from_pandas_edgelist(ipn, source='Source.IP', target='Destination.IP', edge_attr='Total.Fwd.Packets', create_using=nx.DiGraph())\nG2 = nx.from_pandas_edgelist(ip, source='Source.IP', target='Destination.IP', edge_attr='Total.Backward.Packets', create_using=nx.DiGraph())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# limit = 30000\n# ipx = data[data['Destination.IP']== source_ip_number[1]]\n# ipx = ipx.drop(['Destination.IP', 'Source.IP'], axis = 1)\n# if ipx.shape[0] >= limit:\n#     ipx = ipx.head(limit)\n# ipx.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ipx = data[data['Source.IP']== source_ip_number[0]]\n# ipx['Date'] = pd.to_datetime(ipx['Date']).astype('int64')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs=50\nbatch_size=32\nverbose = 1\nlimit = 30000 \n\nipx = data[data['Source.IP']== source_ip_number[3]]\n# ipx = data[data['Source.IP']== '10.200.7.218']\nipx = ipx.drop(['Destination.IP', 'Source.IP', 'Year', 'Time_float'], axis = 1) # only drop year column this time to reduce workload\n\nX = ipx\nX = X.drop(['Total.Fwd.Packets'],1)\ny = ipx\ny = y['Total.Fwd.Packets']\nscaler = MinMaxScaler(feature_range=(0, 1))\nX = scaler.fit_transform(X)\ny = scaler.fit_transform(y.values.reshape(-1, 1))\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# if ipx.shape[0] >= limit:\n#     ipx = ipx.head(limit)\n# # Normalize the data\n# scaler = MinMaxScaler(feature_range=(0, 1))\n# ipx = scaler.fit_transform(ipx.values.reshape(-1, 1))\n\n# # Split the data into training and testing sets\n# train_size = int(len(ipx) * 0.8)\n# test_size = len(ipx) - train_size\n# train, test = ipx[0:train_size, :], ipx[train_size:len(ipx), :]\n\n# # Function to create the input sequences and labels for training the LSTM model\n# def create_dataset(dataset, look_back=1):\n#     X, Y = [], []\n#     for i in range(len(dataset)-look_back):\n#         a = dataset[i:(i+look_back), 0]\n#         X.append(a)\n#         Y.append(dataset[i + look_back, 0])\n#     return np.array(X), np.array(Y)\n\n# look_back = 50  # Number of time steps to look back for making predictions\n\n# # Create the training dataset\n# X_train, y_train = create_dataset(train, look_back)\n\n# Reshape the input data to be 3D for LSTM (samples, time steps, features)\nX_train = np.reshape(X_train, (X_train.shape[0], 1, X_train.shape[1]))\n\n# X_test, y_test = create_dataset(test, look_back)\nX_test = np.reshape(X_test, (X_test.shape[0], 1, X_test.shape[1]))\n\n# Create the LSTM model\n\n# model = Sequential()\n# model.add(LSTM(units=64, return_sequences=True, recurrent_activation = \"sigmoid\", input_shape=(1, look_back)))\n# model.add(LSTM(units=32, return_sequences=True))\n# model.add(LSTM(units=16, return_sequences=True))\n# model.add(LSTM(units=8))\n# model.add(Dense(units=1))\n# model.compile(optimizer=\"adam\", loss=\"mae\")\n\nmodel = Sequential()\nmodel.add(LSTM(64, activation='relu', input_shape=(X_train.shape[1], X_train.shape[2]), return_sequences=True))\nmodel.add(LSTM(32, activation='relu', return_sequences=False))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(units=1))\nmodel.compile(optimizer=\"adam\", loss=\"mae\")\n\n# Train the LSTM model\nmodel.fit(X_train, y_train, epochs=6, batch_size=4, verbose=2)\n\n# Make predictions using the trained LSTM model\ny_predicted = model.predict(X_test)\ny_predicted = scaler.inverse_transform(y_predicted)\n# print(y_test, y_predicted)\nmae = mean_absolute_error(y_test, y_predicted)\n\nprint(\"Mean Absolute Error:\", mae)\n# Plot the actual and predicted traffic volume\nplt.plot(scaler.inverse_transform(y_test.reshape(-1, 1)), label='Actual Traffic Volume')\nplt.plot(y_predicted, label='Predicted Traffic Volume')\nplt.xlabel('Time Step')\nplt.ylabel('Traffic Volume')\nplt.legend()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thời gian thực thi là RẤT lâu, ta mất khoảng 4h cho việc dự đoán sự trao đổi giữa các cặp source-destination IP chính -> khi tổng hợp thành hàm chạy hàng loạt cặp IP có thể sẽ lâu hơn.","metadata":{}},{"cell_type":"markdown","source":"Còn phần việc chọn ra bộ tham số phù hợp khi thực hiện cho hàng loại cặp IP","metadata":{}},{"cell_type":"markdown","source":"Note: thay đổi từ việc dự đoán một thuộc tính số packet đến dựa trên các thuộc tính còn lại, ta dự đoán toàn bộ dựa trên các thông tin đã có vì trong ứng dụng việc có trước các thuộc tính đầu vào mà lại không có thuộc tính đầu ra là không thực tế.","metadata":{}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeRegressor\nfrom sklearn.model_selection import GridSearchCV, train_test_split\nfrom sklearn.metrics import mean_absolute_error\n\nipx = data[data['Source.IP']== source_ip_number[3]]\n# ipx = data[data['Source.IP']== '10.200.7.218']\nipx = ipx.drop(['Destination.IP', 'Source.IP', 'Year', 'Time_float'], axis = 1)\n\nX = ipx\nX = X.drop(['Total.Fwd.Packets'],1)\ny = ipx\ny = y['Total.Fwd.Packets']\nscaler = MinMaxScaler(feature_range=(0, 1))\nX = scaler.fit_transform(X)\ny = scaler.fit_transform(y.values.reshape(-1, 1))\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\nparams = {\n    \"max_depth\": [None, 10, 20, 30],\n    \"min_samples_split\": [2, 5, 10],\n    \"min_samples_leaf\": [1, 2, 4],\n    \"max_features\": [\"auto\", \"sqrt\", \"log2\"],\n}\n\ntree_reg = DecisionTreeRegressor(random_state=42)\ngrid_search = GridSearchCV(tree_reg, params, cv=5, scoring=\"neg_mean_absolute_error\")\ngrid_search.fit(X_train, y_train)\nprint(\"Best parameters: \", grid_search.best_params_)\ny_prediction = grid_search.predict(X_test)\nbest_tree_reg = DecisionTreeRegressor(max_depth=10, max_features='auto', min_samples_leaf=2, min_samples_split=5, random_state=42)\nbest_tree_reg.fit(X, y)\n\n# print(y_test, '\\t', y_prediction)\n# Evaluate the model using mean squared error (MSE)\nmae = mean_absolute_error(y_test, y_pred)\nprint(\"Mean Absolute Error: {:.10f}\".format(mae))\n\n# Plot the actual and predicted traffic volume\nplt.plot(scaler.inverse_transform(y_test.reshape(-1, 1)), label='Actual Traffic Volume')\nplt.plot(y_predicted, label='Predicted Traffic Volume')\nplt.xlabel('Time Step')\nplt.ylabel('Traffic Volume')\nplt.legend()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs=50\nbatch_size=32\nverbose = 1\nlimit = 30000 \n\n# ipx = data[data['Source.IP']== source_ip_number[3]]\nipx = data[data['Source.IP']== '10.200.7.218']\nipx = ipx.drop(['Destination.IP', 'Source.IP', 'Year', 'Time_float'], axis = 1) # only drop year column this time to reduce workload\n\nX = ipx\nX = X.drop(['Total.Fwd.Packets'],1)\ny = ipx\ny = y['Total.Fwd.Packets']\nscaler = MinMaxScaler(feature_range=(0, 1))\nX = scaler.fit_transform(X)\ny = scaler.fit_transform(y.values.reshape(-1, 1))\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# if ipx.shape[0] >= limit:\n#     ipx = ipx.head(limit)\n# # Normalize the data\n# scaler = MinMaxScaler(feature_range=(0, 1))\n# ipx = scaler.fit_transform(ipx.values.reshape(-1, 1))\n\n# # Split the data into training and testing sets\n# train_size = int(len(ipx) * 0.8)\n# test_size = len(ipx) - train_size\n# train, test = ipx[0:train_size, :], ipx[train_size:len(ipx), :]\n\n# # Function to create the input sequences and labels for training the LSTM model\n# def create_dataset(dataset, look_back=1):\n#     X, Y = [], []\n#     for i in range(len(dataset)-look_back):\n#         a = dataset[i:(i+look_back), 0]\n#         X.append(a)\n#         Y.append(dataset[i + look_back, 0])\n#     return np.array(X), np.array(Y)\n\n# look_back = 50  # Number of time steps to look back for making predictions\n\n# # Create the training dataset\n# X_train, y_train = create_dataset(train, look_back)\n\n# Reshape the input data to be 3D for LSTM (samples, time steps, features)\nX_train = np.reshape(X_train, (X_train.shape[0], 1, X_train.shape[1]))\n\n# X_test, y_test = create_dataset(test, look_back)\nX_test = np.reshape(X_test, (X_test.shape[0], 1, X_test.shape[1]))\n\n# Create the LSTM model\n\n# model = Sequential()\n# model.add(LSTM(units=64, return_sequences=True, recurrent_activation = \"sigmoid\", input_shape=(1, look_back)))\n# model.add(LSTM(units=32, return_sequences=True))\n# model.add(LSTM(units=16, return_sequences=True))\n# model.add(LSTM(units=8))\n# model.add(Dense(units=1))\n# model.compile(optimizer=\"adam\", loss=\"mae\")\n\nmodel = Sequential()\nmodel.add(LSTM(64, activation='relu', input_shape=(X_train.shape[1], X_train.shape[2]), return_sequences=True))\nmodel.add(LSTM(32, activation='relu', return_sequences=False))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(units=1))\nmodel.compile(optimizer=\"adam\", loss=\"mae\")\n\n# Train the LSTM model\nmodel.fit(X_train, y_train, epochs=6, batch_size=4, verbose=2)\n\n# Make predictions using the trained LSTM model\ny_predicted = model.predict(X_test)\ny_predicted = scaler.inverse_transform(y_predicted)\n# print(y_test, y_predicted)\nmae = mean_absolute_error(y_test, y_predicted)\n\nprint(\"Mean Absolute Error:\", mae)\n# Plot the actual and predicted traffic volume\nplt.plot(scaler.inverse_transform(y_test.reshape(-1, 1)), label='Actual Traffic Volume')\nplt.plot(y_predicted, label='Predicted Traffic Volume')\nplt.xlabel('Time Step')\nplt.ylabel('Traffic Volume')\nplt.legend()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def LSTM_prediction_IP(ip):\n    epochs = 20\n    batch_size = 12\n    verbose = 2\n    limit = 30000 \n\n    ipx = ip[ip['Source.IP']== ip]\n    ipx = ipx.drop(['Destination.IP', 'Source.IP'], axis = 1)\n    if ipx.shape[0] >= limit:\n        ipx = ipx.head(limit)\n    # Normalize the data\n    scaler = MinMaxScaler(feature_range=(0, 1))\n    ipx = scaler.fit_transform(ipx.values.reshape(-1, 1))\n\n    # Split the data into training and testing sets\n    train_size = int(len(ipx) * 0.8)\n    test_size = len(ipx) - train_size\n    train, test = ipx[0:train_size, :], ipx[train_size:len(ipx), :]\n\n    # Function to create the input sequences and labels for training the LSTM model\n    def create_dataset(dataset, look_back=1):\n        X, Y = [], []\n        for i in range(len(dataset)-look_back):\n            a = dataset[i:(i+look_back), 0]\n            X.append(a)\n            Y.append(dataset[i + look_back, 0])\n        return np.array(X), np.array(Y)\n\n    look_back = 50  # Number of time steps to look back for making predictions\n\n    # Create the training dataset\n    X_train, y_train = create_dataset(train, look_back)\n\n    # Reshape the input data to be 3D for LSTM (samples, time steps, features)\n    X_train = np.reshape(X_train, (X_train.shape[0], 1, X_train.shape[1]))\n\n    # Create the LSTM model\n    model = Sequential()\n    model.add(LSTM(units=64, return_sequences=True, recurrent_activation = \"sigmoid\", input_shape=(1, look_back)))\n    model.add(LSTM(units=32, return_sequences=True))\n    model.add(LSTM(units=16, return_sequences=True))\n    model.add(LSTM(units=8))\n    model.add(Dense(units=1))\n    model.compile(optimizer=\"adam\", loss=\"mae\")\n    # Train the LSTM model\n    model.fit(X_train, y_train, epochs = 15, batch_size = 8, verbose = 2)\n\n    # Create the testing dataset\n    X_test, y_test = create_dataset(test, look_back)\n    X_test = np.reshape(X_test, (X_test.shape[0], 1, X_test.shape[1]))\n\n    # Make predictions using the trained LSTM model\n    y_predicted = model.predict(X_test)\n    y_predicted = scaler.inverse_transform(y_predicted)\n    print(y_test, y_predicted)\n    # Plot the actual and predicted traffic volume\n    plt.plot(scaler.inverse_transform(y_test.reshape(-1, 1)), label='Actual Traffic Volume')\n    plt.plot(y_predicted, label='Predicted Traffic Volume')\n    plt.xlabel('Time Step')\n    plt.ylabel('Traffic Volume')\n    plt.legend()\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def LSTM_prediction(n):\n    epochs = 20\n    batch_size = 12\n    verbose = 2\n    limit = 30000 \n\n    ipx = ip[ip['Source.IP']== source_ip_number[n]]\n    ipx = ipx.drop(['Destination.IP', 'Source.IP'], axis = 1)\n    if ipx.shape[0] >= limit:\n        ipx = ipx.head(limit)\n    # Normalize the data\n    scaler = MinMaxScaler(feature_range=(0, 1))\n    ipx = scaler.fit_transform(ipx.values.reshape(-1, 1))\n\n    # Split the data into training and testing sets\n    train_size = int(len(ipx) * 0.8)\n    test_size = len(ipx) - train_size\n    train, test = ipx[0:train_size, :], ipx[train_size:len(ipx), :]\n\n    # Function to create the input sequences and labels for training the LSTM model\n    def create_dataset(dataset, look_back=1):\n        X, Y = [], []\n        for i in range(len(dataset)-look_back):\n            a = dataset[i:(i+look_back), 0]\n            X.append(a)\n            Y.append(dataset[i + look_back, 0])\n        return np.array(X), np.array(Y)\n\n    look_back = 50  # Number of time steps to look back for making predictions\n\n    # Create the training dataset\n    X_train, y_train = create_dataset(train, look_back)\n\n    # Reshape the input data to be 3D for LSTM (samples, time steps, features)\n    X_train = np.reshape(X_train, (X_train.shape[0], 1, X_train.shape[1]))\n\n    # Create the LSTM model\n    model = Sequential()\n    model.add(LSTM(units=64, return_sequences=True, recurrent_activation = \"sigmoid\", input_shape=(1, look_back)))\n    model.add(LSTM(units=32, return_sequences=True))\n    model.add(LSTM(units=16, return_sequences=True))\n    model.add(LSTM(units=8))\n    model.add(Dense(units=1))\n    model.compile(optimizer=\"adam\", loss=\"mae\")\n    # Train the LSTM model\n    model.fit(X_train, y_train, epochs = 15, batch_size = 8, verbose = 2)\n\n    # Create the testing dataset\n    X_test, y_test = create_dataset(test, look_back)\n    X_test = np.reshape(X_test, (X_test.shape[0], 1, X_test.shape[1]))\n\n    # Make predictions using the trained LSTM model\n    y_predicted = model.predict(X_test)\n    y_predicted = scaler.inverse_transform(y_predicted)\n    print(y_test, y_predicted)\n    # Plot the actual and predicted traffic volume\n    plt.plot(scaler.inverse_transform(y_test.reshape(-1, 1)), label='Actual Traffic Volume')\n    plt.plot(y_predicted, label='Predicted Traffic Volume')\n    plt.xlabel('Time Step')\n    plt.ylabel('Traffic Volume')\n    plt.legend()\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ipx = data[data['Source.IP']== source_ip_number[1]]\nipx = ipx.drop(['Destination.IP', 'Source.IP'], axis = 1)\n\n# define input and output variables\nX1 = ipx.drop(['Total.Length.of.Fwd.Packets'], axis = 1)\nX1 = X1.values\nX2 = ipx.drop(['Total.Length.of.Bwd.Packets'], axis = 1)\nX2 = X2.values\nY1 = ipx['Total.Length.of.Fwd.Packets'].values\nY2 = ipx['Total.Length.of.Bwd.Packets'].values\n\n# split the data into training and testing sets\ntrain_size = int(len(X1) * 0.8)\ntest_size = len(X1) - train_size\ntrain_X1, test_X1 = X1[0:train_size], X1[train_size:len(X1)]\ntrain_X2, test_X2 = X2[0:train_size], X2[train_size:len(X2)]\ntrain_Y1, test_Y1 = Y1[0:train_size], Y1[train_size:len(Y1)]\ntrain_Y2, test_Y2 = Y2[0:train_size], Y2[train_size:len(Y2)]\n\n# reshape the data\ntrain_X = np.column_stack((train_X1, train_X2))\ntest_X = np.column_stack((test_X1, test_X2))\ntrain_Y = np.column_stack((train_Y1, train_Y2))\ntest_Y = np.column_stack((test_Y1, test_Y2))\ntrain_X = np.reshape(train_X, (train_X.shape[0], 1, train_X.shape[1]))\ntest_X = np.reshape(test_X, (test_X.shape[0], 1, test_X.shape[1]))\n\n# define the LSTM model\nmodel = Sequential()\nmodel.add(LSTM(50, input_shape=(train_X.shape[1], train_X.shape[2])))\nmodel.add(Dense(2))\nmodel.compile(loss='mae', optimizer='adam')\n\n# train the model\nmodel.fit(train_X, train_Y, epochs=10, batch_size=2, validation_data=(test_X, test_Y), verbose=2, shuffle=False)\n\n# evaluate the model\nscore = model.evaluate(test_X, test_Y, verbose=0)\nprint('Test RMSE: %.3f' % score)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}