{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\npd.set_option('display.max_columns', None)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T13:45:00.014664Z","iopub.execute_input":"2022-08-11T13:45:00.015194Z","iopub.status.idle":"2022-08-11T13:45:01.283085Z","shell.execute_reply.started":"2022-08-11T13:45:00.015089Z","shell.execute_reply":"2022-08-11T13:45:01.281778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing Dataset","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv(\"../input/predict-potential-spammers-on-fiverr/train.csv\")\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:45:28.491304Z","iopub.execute_input":"2022-08-11T13:45:28.491809Z","iopub.status.idle":"2022-08-11T13:45:31.289201Z","shell.execute_reply.started":"2022-08-11T13:45:28.491776Z","shell.execute_reply":"2022-08-11T13:45:31.284600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:51:00.014689Z","iopub.execute_input":"2022-08-11T13:51:00.016937Z","iopub.status.idle":"2022-08-11T13:51:00.116347Z","shell.execute_reply.started":"2022-08-11T13:51:00.016837Z","shell.execute_reply":"2022-08-11T13:51:00.114587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Dataset Shape:**","metadata":{}},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:52:23.560571Z","iopub.execute_input":"2022-08-11T13:52:23.561147Z","iopub.status.idle":"2022-08-11T13:52:23.570804Z","shell.execute_reply.started":"2022-08-11T13:52:23.561098Z","shell.execute_reply":"2022-08-11T13:52:23.569336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.label.value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:52:10.669122Z","iopub.execute_input":"2022-08-11T13:52:10.669745Z","iopub.status.idle":"2022-08-11T13:52:10.687992Z","shell.execute_reply.started":"2022-08-11T13:52:10.669700Z","shell.execute_reply":"2022-08-11T13:52:10.686717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Checking duplicates","metadata":{}},{"cell_type":"code","source":"data.duplicated().value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:56:45.790680Z","iopub.execute_input":"2022-08-11T13:56:45.791150Z","iopub.status.idle":"2022-08-11T13:56:46.650797Z","shell.execute_reply.started":"2022-08-11T13:56:45.791117Z","shell.execute_reply":"2022-08-11T13:56:46.649576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Missing values","metadata":{}},{"cell_type":"code","source":"data.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:52:36.389495Z","iopub.execute_input":"2022-08-11T13:52:36.389968Z","iopub.status.idle":"2022-08-11T13:52:36.444024Z","shell.execute_reply.started":"2022-08-11T13:52:36.389926Z","shell.execute_reply":"2022-08-11T13:52:36.442827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.X13.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:55:08.615527Z","iopub.execute_input":"2022-08-11T13:55:08.616042Z","iopub.status.idle":"2022-08-11T13:55:08.634692Z","shell.execute_reply.started":"2022-08-11T13:55:08.616002Z","shell.execute_reply":"2022-08-11T13:55:08.633379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Replace missing values with the median**","metadata":{}},{"cell_type":"code","source":"data['X13'].fillna(value=round(data['X13'].median()), inplace=True)\ndata.X13.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:55:19.372354Z","iopub.execute_input":"2022-08-11T13:55:19.372745Z","iopub.status.idle":"2022-08-11T13:55:19.399258Z","shell.execute_reply.started":"2022-08-11T13:55:19.372715Z","shell.execute_reply":"2022-08-11T13:55:19.398007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Features Selection","metadata":{}},{"cell_type":"code","source":"features = [col for col in data if col.startswith('X')]\nprint(features)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:02:50.781392Z","iopub.execute_input":"2022-08-11T14:02:50.781841Z","iopub.status.idle":"2022-08-11T14:02:50.788694Z","shell.execute_reply.started":"2022-08-11T14:02:50.781804Z","shell.execute_reply":"2022-08-11T14:02:50.787370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Checking Columns with no Variation","metadata":{}},{"cell_type":"code","source":"no_variation_cols = list()\nfor col in features:\n    if data[col].nunique() == 1:\n        no_variation_cols.append(col)\n\nprint(no_variation_cols)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:02:52.958000Z","iopub.execute_input":"2022-08-11T14:02:52.958392Z","iopub.status.idle":"2022-08-11T14:02:53.112881Z","shell.execute_reply.started":"2022-08-11T14:02:52.958360Z","shell.execute_reply":"2022-08-11T14:02:53.111711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = [col for col in features if col not in no_variation_cols]\nprint(features)\n# print(len(features))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:02:55.142353Z","iopub.execute_input":"2022-08-11T14:02:55.142772Z","iopub.status.idle":"2022-08-11T14:02:55.148711Z","shell.execute_reply.started":"2022-08-11T14:02:55.142736Z","shell.execute_reply":"2022-08-11T14:02:55.147593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Correlation matrix","metadata":{}},{"cell_type":"code","source":"correlation = data[features].corr()\ncorrelation","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:03:51.610155Z","iopub.execute_input":"2022-08-11T14:03:51.610975Z","iopub.status.idle":"2022-08-11T14:03:54.489267Z","shell.execute_reply.started":"2022-08-11T14:03:51.610930Z","shell.execute_reply":"2022-08-11T14:03:54.488280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Plotting Correlation Matrix**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize =(32,24))\nsns.heatmap(correlation, cmap=\"Blues\" ,annot = True )\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:04:30.167461Z","iopub.execute_input":"2022-08-11T14:04:30.167915Z","iopub.status.idle":"2022-08-11T14:04:38.347699Z","shell.execute_reply.started":"2022-08-11T14:04:30.167879Z","shell.execute_reply":"2022-08-11T14:04:38.343932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"upper_correlation = correlation.where(np.triu(np.ones(correlation.shape),k=1).astype(bool))\nupper_correlation","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:05:37.216093Z","iopub.execute_input":"2022-08-11T14:05:37.216502Z","iopub.status.idle":"2022-08-11T14:05:37.310513Z","shell.execute_reply.started":"2022-08-11T14:05:37.216469Z","shell.execute_reply":"2022-08-11T14:05:37.309334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dropping columns with high correlation (>0.8)","metadata":{}},{"cell_type":"code","source":"high_corr_cols = [column for column in upper_correlation.columns if any(upper_correlation[column] > 0.8)]\nprint(high_corr_cols)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:07:27.191974Z","iopub.execute_input":"2022-08-11T14:07:27.192379Z","iopub.status.idle":"2022-08-11T14:07:27.208173Z","shell.execute_reply.started":"2022-08-11T14:07:27.192347Z","shell.execute_reply":"2022-08-11T14:07:27.206609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = [col for col in features if col not in high_corr_cols]\nprint(features)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:09:38.327618Z","iopub.execute_input":"2022-08-11T14:09:38.328058Z","iopub.status.idle":"2022-08-11T14:09:38.335077Z","shell.execute_reply.started":"2022-08-11T14:09:38.328013Z","shell.execute_reply":"2022-08-11T14:09:38.333565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We have reduced the number of features from 51 to 40 features**","metadata":{}},{"cell_type":"code","source":"print(len(features))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:10:57.983172Z","iopub.execute_input":"2022-08-11T14:10:57.984212Z","iopub.status.idle":"2022-08-11T14:10:57.990767Z","shell.execute_reply.started":"2022-08-11T14:10:57.984165Z","shell.execute_reply":"2022-08-11T14:10:57.989761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Preparing Training and Testing sets","metadata":{}},{"cell_type":"code","source":"X = data[features]\nX.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:14:19.563850Z","iopub.execute_input":"2022-08-11T14:14:19.564349Z","iopub.status.idle":"2022-08-11T14:14:19.626830Z","shell.execute_reply.started":"2022-08-11T14:14:19.564302Z","shell.execute_reply":"2022-08-11T14:14:19.625193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y = data['label'].values\nY.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:14:26.190251Z","iopub.execute_input":"2022-08-11T14:14:26.190912Z","iopub.status.idle":"2022-08-11T14:14:26.200829Z","shell.execute_reply.started":"2022-08-11T14:14:26.190842Z","shell.execute_reply":"2022-08-11T14:14:26.199032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Train / Test Split**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(X, Y, test_size=0.15, random_state=42)\n\nprint(\"Training Dataset count: \" + str(X_train.shape[0]))\nprint(\"Test Dataset count: \" + str(X_test.shape[0]))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:16:42.698261Z","iopub.execute_input":"2022-08-11T14:16:42.698687Z","iopub.status.idle":"2022-08-11T14:16:42.934573Z","shell.execute_reply.started":"2022-08-11T14:16:42.698651Z","shell.execute_reply":"2022-08-11T14:16:42.933112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Scaling features with StandardScaler","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler(with_mean=True, with_std=True)\nscaler = scaler.fit(X)\n\nX_train = scaler.transform(X_train)\nX_test = scaler.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:22:13.470331Z","iopub.execute_input":"2022-08-11T14:22:13.470777Z","iopub.status.idle":"2022-08-11T14:22:13.841886Z","shell.execute_reply.started":"2022-08-11T14:22:13.470743Z","shell.execute_reply":"2022-08-11T14:22:13.840700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Now, our features are ready to build and apply some Machine Learning/ Deep Learning Classification Algorithms.**</br>\n**Because that our dataset is imbalanced, I suggest applying the following techniques :**\n* **Undersampling/ Oversampling methods.**\n* **Cost-Sensitive Algorithms/ Class weighing.**\n* **Anomaly detection Algorithms.**\n* . . .","metadata":{}}]}