{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Import Libraries","metadata":{}},{"cell_type":"code","source":"\nimport pandas as pd \nimport numpy as np \nimport seaborn as sns \nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:15:35.964946Z","iopub.execute_input":"2022-08-07T21:15:35.965273Z","iopub.status.idle":"2022-08-07T21:15:36.444145Z","shell.execute_reply.started":"2022-08-07T21:15:35.965249Z","shell.execute_reply":"2022-08-07T21:15:36.443230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### import dataset","metadata":{}},{"cell_type":"code","source":"data =pd.read_csv('/kaggle/input/predict-potential-spammers-on-fiverr/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:15:36.445665Z","iopub.execute_input":"2022-08-07T21:15:36.445948Z","iopub.status.idle":"2022-08-07T21:15:38.026346Z","shell.execute_reply.started":"2022-08-07T21:15:36.445918Z","shell.execute_reply":"2022-08-07T21:15:38.025108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:15:38.028618Z","iopub.execute_input":"2022-08-07T21:15:38.028960Z","iopub.status.idle":"2022-08-07T21:15:38.056738Z","shell.execute_reply.started":"2022-08-07T21:15:38.028928Z","shell.execute_reply":"2022-08-07T21:15:38.056058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### exploratory data Analysis","metadata":{}},{"cell_type":"code","source":"# label indicate \n# 1 spammer \n# 0 not spammer \n\nprint('unique value in label', data['label'].unique())\nprint('---'*20)\nprint('')\n\nprint('number of each value in label \\n' ,data['label'].value_counts()  )\nprint('---'*20)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:15:38.057849Z","iopub.execute_input":"2022-08-07T21:15:38.058494Z","iopub.status.idle":"2022-08-07T21:15:38.076938Z","shell.execute_reply.started":"2022-08-07T21:15:38.058467Z","shell.execute_reply":"2022-08-07T21:15:38.076081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:15:38.078656Z","iopub.execute_input":"2022-08-07T21:15:38.079385Z","iopub.status.idle":"2022-08-07T21:15:38.120997Z","shell.execute_reply.started":"2022-08-07T21:15:38.079357Z","shell.execute_reply":"2022-08-07T21:15:38.119769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X13 is only column having float value\ndata['X13']","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:15:38.122070Z","iopub.execute_input":"2022-08-07T21:15:38.122340Z","iopub.status.idle":"2022-08-07T21:15:38.130043Z","shell.execute_reply.started":"2022-08-07T21:15:38.122315Z","shell.execute_reply":"2022-08-07T21:15:38.129211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:15:38.132060Z","iopub.execute_input":"2022-08-07T21:15:38.132555Z","iopub.status.idle":"2022-08-07T21:15:38.651064Z","shell.execute_reply.started":"2022-08-07T21:15:38.132477Z","shell.execute_reply":"2022-08-07T21:15:38.650199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.describe().round().T","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:15:38.652121Z","iopub.execute_input":"2022-08-07T21:15:38.652377Z","iopub.status.idle":"2022-08-07T21:15:39.188372Z","shell.execute_reply.started":"2022-08-07T21:15:38.652355Z","shell.execute_reply":"2022-08-07T21:15:39.186980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:15:39.189368Z","iopub.execute_input":"2022-08-07T21:15:39.189593Z","iopub.status.idle":"2022-08-07T21:15:39.195631Z","shell.execute_reply.started":"2022-08-07T21:15:39.189572Z","shell.execute_reply":"2022-08-07T21:15:39.194523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### we find the relationship between \nlabel and user id \n\nlabel and other parameters ","metadata":{}},{"cell_type":"code","source":"sns.barplot(data['label'] , data['user_id'] ,hue=data['label'] )  \nplt.title('relationship between user_id and label' , size=22)\nplt.xlabel('label')\nplt.ylabel('User_id')\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:15:39.196902Z","iopub.execute_input":"2022-08-07T21:15:39.197122Z","iopub.status.idle":"2022-08-07T21:15:45.603788Z","shell.execute_reply.started":"2022-08-07T21:15:39.197101Z","shell.execute_reply":"2022-08-07T21:15:45.602590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i ,var in enumerate(data.iloc[:,2:]):\n    plt.figure(figsize=(10, 110))\n    plt.subplot(54,1,i+1)\n    \n    \n    sns.barplot(data['label'] , data[var] ,hue=data['label'] )  \nplt.title('relationship between other parameter and label' , size=22)\n    \n        ","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:15:45.606425Z","iopub.execute_input":"2022-08-07T21:15:45.606670Z","iopub.status.idle":"2022-08-07T21:21:24.816782Z","shell.execute_reply.started":"2022-08-07T21:15:45.606648Z","shell.execute_reply":"2022-08-07T21:21:24.816055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n### drop value\nwe will drop these columns\n\n'X27','X29', 'X30', 'X33', 'X46', 'X47', 'X48 '\n","metadata":{}},{"cell_type":"code","source":"data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:21:24.818267Z","iopub.execute_input":"2022-08-07T21:21:24.818600Z","iopub.status.idle":"2022-08-07T21:21:24.850201Z","shell.execute_reply.started":"2022-08-07T21:21:24.818566Z","shell.execute_reply":"2022-08-07T21:21:24.849615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# before Missing value\nsns.heatmap(data.isnull())","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:21:24.851380Z","iopub.execute_input":"2022-08-07T21:21:24.853410Z","iopub.status.idle":"2022-08-07T21:21:42.869092Z","shell.execute_reply.started":"2022-08-07T21:21:24.853377Z","shell.execute_reply":"2022-08-07T21:21:42.868123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### handle missing value","metadata":{}},{"cell_type":"code","source":"print('mode of X13 =' ,data['X13'].mode()[0])\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:21:42.870430Z","iopub.execute_input":"2022-08-07T21:21:42.870674Z","iopub.status.idle":"2022-08-07T21:21:42.881781Z","shell.execute_reply.started":"2022-08-07T21:21:42.870648Z","shell.execute_reply":"2022-08-07T21:21:42.880851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['X13']=data['X13'].fillna(data['X13'].mode()[0])","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:21:42.883023Z","iopub.execute_input":"2022-08-07T21:21:42.883391Z","iopub.status.idle":"2022-08-07T21:21:42.899111Z","shell.execute_reply.started":"2022-08-07T21:21:42.883356Z","shell.execute_reply":"2022-08-07T21:21:42.897912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(data.isnull())","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:21:42.900520Z","iopub.execute_input":"2022-08-07T21:21:42.901338Z","iopub.status.idle":"2022-08-07T21:22:02.042691Z","shell.execute_reply.started":"2022-08-07T21:21:42.901312Z","shell.execute_reply":"2022-08-07T21:22:02.041417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(35,35))\nsns.heatmap(data.corr() , annot=True,cmap=plt.cm.Reds, fmt='.2f')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:22:02.044093Z","iopub.execute_input":"2022-08-07T21:22:02.044509Z","iopub.status.idle":"2022-08-07T21:22:10.830369Z","shell.execute_reply.started":"2022-08-07T21:22:02.044478Z","shell.execute_reply":"2022-08-07T21:22:10.829381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop(columns=['X27','X29', 'X30', 'X33', 'X46', 'X47', 'X48'], inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:22:10.833999Z","iopub.execute_input":"2022-08-07T21:22:10.834431Z","iopub.status.idle":"2022-08-07T21:22:10.872149Z","shell.execute_reply.started":"2022-08-07T21:22:10.834404Z","shell.execute_reply":"2022-08-07T21:22:10.871056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Correlation graph","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(40,40))\nsns.heatmap(data.corr() , annot=True,cmap=plt.cm.Reds, fmt='.2f')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:22:10.874404Z","iopub.execute_input":"2022-08-07T21:22:10.874686Z","iopub.status.idle":"2022-08-07T21:22:18.979478Z","shell.execute_reply.started":"2022-08-07T21:22:10.874664Z","shell.execute_reply":"2022-08-07T21:22:18.978411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" 1 shows the strong relationshiop\n    \n 0 shows no relatopnship\n\n-1 shows negative relationship\n\n    ","metadata":{}},{"cell_type":"code","source":"for i in data:\n    print(i, 'unique value')\n    print(data[i].unique())\n    print('')\n    print('value_counts')\n    print(data[i].value_counts('\\n'))\n    print('')\n    print('')\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:22:18.980638Z","iopub.execute_input":"2022-08-07T21:22:18.981420Z","iopub.status.idle":"2022-08-07T21:22:19.351040Z","shell.execute_reply.started":"2022-08-07T21:22:18.981295Z","shell.execute_reply":"2022-08-07T21:22:19.349756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X 13 is a float value or contain a lot of zero value \n# so float value cannot be trained if the have lot of 0 value \n# we have 2 method to tackle with \n# drop value \n# change data type \n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:22:19.352456Z","iopub.execute_input":"2022-08-07T21:22:19.352770Z","iopub.status.idle":"2022-08-07T21:22:19.357949Z","shell.execute_reply.started":"2022-08-07T21:22:19.352744Z","shell.execute_reply":"2022-08-07T21:22:19.356705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data1= data.drop(columns='X13' , axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:22:19.360015Z","iopub.execute_input":"2022-08-07T21:22:19.360424Z","iopub.status.idle":"2022-08-07T21:22:19.403293Z","shell.execute_reply.started":"2022-08-07T21:22:19.360396Z","shell.execute_reply":"2022-08-07T21:22:19.402243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### split data","metadata":{}},{"cell_type":"code","source":"x = data1.iloc[:,1:]\ny=data1.iloc[:,:1]","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:22:19.404511Z","iopub.execute_input":"2022-08-07T21:22:19.405360Z","iopub.status.idle":"2022-08-07T21:22:19.411593Z","shell.execute_reply.started":"2022-08-07T21:22:19.405327Z","shell.execute_reply":"2022-08-07T21:22:19.410661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:22:19.412515Z","iopub.execute_input":"2022-08-07T21:22:19.412812Z","iopub.status.idle":"2022-08-07T21:22:19.436400Z","shell.execute_reply.started":"2022-08-07T21:22:19.412781Z","shell.execute_reply":"2022-08-07T21:22:19.435490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:22:19.437554Z","iopub.execute_input":"2022-08-07T21:22:19.437790Z","iopub.status.idle":"2022-08-07T21:22:19.445771Z","shell.execute_reply.started":"2022-08-07T21:22:19.437767Z","shell.execute_reply":"2022-08-07T21:22:19.444760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('shape of total data set  =', data.shape)\nprint('Shape of x data set  =', x.shape)\nprint('Shape of y data set  =', y.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:22:19.447104Z","iopub.execute_input":"2022-08-07T21:22:19.447860Z","iopub.status.idle":"2022-08-07T21:22:19.457785Z","shell.execute_reply.started":"2022-08-07T21:22:19.447809Z","shell.execute_reply":"2022-08-07T21:22:19.456873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test train data","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_train , x_test , y_train ,y_test = train_test_split(x,y ,test_size=0.2 , random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:22:19.459764Z","iopub.execute_input":"2022-08-07T21:22:19.460012Z","iopub.status.idle":"2022-08-07T21:22:19.792617Z","shell.execute_reply.started":"2022-08-07T21:22:19.459990Z","shell.execute_reply":"2022-08-07T21:22:19.791472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('shape of x_train =' , x_train.shape)\nprint('shape of x_test =' , x_test.shape)\nprint('shape of y_train =' , y_train.shape)\nprint('shape of y_test =' , y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:22:19.794361Z","iopub.execute_input":"2022-08-07T21:22:19.794669Z","iopub.status.idle":"2022-08-07T21:22:19.800229Z","shell.execute_reply.started":"2022-08-07T21:22:19.794644Z","shell.execute_reply":"2022-08-07T21:22:19.798950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model, predict and solve\n","metadata":{}},{"cell_type":"markdown","source":"Now we are ready to train a model and predict the required solution. There are 60+ predictive modelling algorithms to choose from. We must understand the type of problem and solution requirement to narrow down to a select few models which we can evaluate. Our problem is a classification and regression problem. We want to identify relationship between output (spam order  or not)\n\nKNN or k-Nearest Neighbors\n\nSupport Vector Machines\n\nNaive Bayes classifier\n\nDecision Tree\n","metadata":{}},{"cell_type":"markdown","source":"### k-Nearest Neighbors\n\nk-Nearest Neighbors algorithm (or k-NN for short) is a non-parametric method used for classification and regression. A sample is classified by a majority vote of its neighbors, with the sample being assigned to the class most common among its k nearest neighbors (k is a positive integer, typically small). If k = 1, then the object is simply assigned to the class of that single nearest neighbor. Reference Wikipedia.","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nmodel = KNeighborsClassifier(n_neighbors=11).fit(x_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:22:19.801493Z","iopub.execute_input":"2022-08-07T21:22:19.801746Z","iopub.status.idle":"2022-08-07T21:22:19.962680Z","shell.execute_reply.started":"2022-08-07T21:22:19.801722Z","shell.execute_reply":"2022-08-07T21:22:19.961542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:22:19.964134Z","iopub.execute_input":"2022-08-07T21:22:19.964826Z","iopub.status.idle":"2022-08-07T21:22:19.971961Z","shell.execute_reply.started":"2022-08-07T21:22:19.964799Z","shell.execute_reply":"2022-08-07T21:22:19.971166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kn_pred=model.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:22:19.972918Z","iopub.execute_input":"2022-08-07T21:22:19.973423Z","iopub.status.idle":"2022-08-07T21:31:19.765789Z","shell.execute_reply.started":"2022-08-07T21:22:19.973398Z","shell.execute_reply":"2022-08-07T21:31:19.764639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Error/Score","metadata":{}},{"cell_type":"code","source":"from sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:19.767482Z","iopub.execute_input":"2022-08-07T21:31:19.768588Z","iopub.status.idle":"2022-08-07T21:31:19.773577Z","shell.execute_reply.started":"2022-08-07T21:31:19.768555Z","shell.execute_reply":"2022-08-07T21:31:19.772419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metrics.accuracy_score(y_test,kn_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:19.774979Z","iopub.execute_input":"2022-08-07T21:31:19.775255Z","iopub.status.idle":"2022-08-07T21:31:19.793317Z","shell.execute_reply.started":"2022-08-07T21:31:19.775225Z","shell.execute_reply":"2022-08-07T21:31:19.792644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Our(KNN) accuracy score is 97.22% percent, which is extra-ordinary ","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:19.794327Z","iopub.execute_input":"2022-08-07T21:31:19.794693Z","iopub.status.idle":"2022-08-07T21:31:19.802501Z","shell.execute_reply.started":"2022-08-07T21:31:19.794670Z","shell.execute_reply":"2022-08-07T21:31:19.801516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Precision\n\nPrecision can be defined as the percentage of correctly predicted positive outcomes out of all the predicted positive outcomes. It can be given as the ratio of true positives (TP) to the sum of true and false positives (TP + FP).\n\nSo, Precision identifies the proportion of correctly predicted positive outcome. It is more concerned with the positive class than the negative class.\n\nMathematically, precision can be defined as the ratio of TP to (TP + FP).\n\n### Recall\n\nRecall can be defined as the percentage of correctly predicted positive outcomes out of all the actual positive outcomes. It can be given as the ratio of true positives (TP) to the sum of true positives and false negatives (TP + FN). Recall is also called Sensitivity.\n\nRecall identifies the proportion of correctly predicted actual positives.\n\nMathematically, recall can be given as the ratio of TP to (TP + FN).\n\n### f1-score\n\n\nf1-score is the weighted harmonic mean of precision and recall. The best possible f1-score would be 1.0 and the worst would be 0.0. f1-score is the harmonic mean of precision and recall. So, f1-score is always lower than accuracy measures as they embed precision and recall into their computation. The weighted average of f1-score should be used to compare classifier models, not global accuracy.\n\n### Support\n\nSupport is the actual number of occurrences of the class in our dataset.\n\n","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(y_test,kn_pred))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:19.805272Z","iopub.execute_input":"2022-08-07T21:31:19.806422Z","iopub.status.idle":"2022-08-07T21:31:19.901093Z","shell.execute_reply.started":"2022-08-07T21:31:19.806392Z","shell.execute_reply":"2022-08-07T21:31:19.899913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# precision rate , recall , f1-score are high means customer are not spammer \n\n# precision rate , recall , f1-score are lower means customer are spammer","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:19.902400Z","iopub.execute_input":"2022-08-07T21:31:19.902739Z","iopub.status.idle":"2022-08-07T21:31:19.907651Z","shell.execute_reply.started":"2022-08-07T21:31:19.902706Z","shell.execute_reply":"2022-08-07T21:31:19.906388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## confusion matrix\n\nA confusion matrix is a tool for summarizing the performance of a classification algorithm. A confusion matrix will give us a clear picture of classification model performance and the types of errors produced by the model. It gives us a summary of correct and incorrect predictions broken down by each category. The summary is represented in a tabular form.\n\nFour types of outcomes are possible while evaluating a classification model performance. These four outcomes are described below:-\n\n### True Positives\nTrue Positives (TP) – True Positives occur when we predict an observation belongs to a certain class and the observation actually belongs to that class.\n\n### True Negatives\nTrue Negatives (TN) – True Negatives occur when we predict an observation does not belong to a certain class and the observation actually does not belong to that class.\n\n### False Positives (Type I error)\nFalse Positives (FP) – False Positives occur when we predict an observation belongs to a certain class but the observation actually does not belong to that class. This type of error is called Type I error.\n\n### False Negatives (Type II error)\nFalse Negatives (FN) – False Negatives occur when we predict an observation does not belong to a certain class but the observation actually belongs to that class. This is a very serious error and it is called Type II error.","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import plot_confusion_matrix\ncm = metrics.confusion_matrix(y_test , kn_pred)\ncm","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:19.908849Z","iopub.execute_input":"2022-08-07T21:31:19.909127Z","iopub.status.idle":"2022-08-07T21:31:19.934318Z","shell.execute_reply.started":"2022-08-07T21:31:19.909100Z","shell.execute_reply":"2022-08-07T21:31:19.933306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('\\nTrue Positives(TP) = ', cm[0,0])\n\nprint('\\nTrue Negatives(TN) = ', cm[1,1])\n\nprint('\\nFalse Positives(FP) = ', cm[0,1])\n\nprint('\\nFalse Negatives(FN) = ', cm[1,0])","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:19.941980Z","iopub.execute_input":"2022-08-07T21:31:19.942408Z","iopub.status.idle":"2022-08-07T21:31:19.947072Z","shell.execute_reply.started":"2022-08-07T21:31:19.942383Z","shell.execute_reply":"2022-08-07T21:31:19.946469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"knn_score = metrics.accuracy_score(y_test, kn_pred)\nprint('accutracy score is =', knn_score * 100,'%' )","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:19.948243Z","iopub.execute_input":"2022-08-07T21:31:19.948894Z","iopub.status.idle":"2022-08-07T21:31:19.963709Z","shell.execute_reply.started":"2022-08-07T21:31:19.948859Z","shell.execute_reply":"2022-08-07T21:31:19.962222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,12))\nsns.heatmap(cm ,annot=True, fmt='.3f' , linewidths=.5 ,square=True , cmap='Spectral' )\nplt.ylabel('Actual value')\nplt.xlabel('Predecited value')\nall_sample_title ='KNN model accuracy : {0}% '.format( knn_score * 100)\nplt.title(all_sample_title, size  =17)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:19.965556Z","iopub.execute_input":"2022-08-07T21:31:19.965807Z","iopub.status.idle":"2022-08-07T21:31:20.157365Z","shell.execute_reply.started":"2022-08-07T21:31:19.965783Z","shell.execute_reply":"2022-08-07T21:31:20.156418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# how to read confusion matrix\n\nTrue positive (88847)  this result shows that actual value(no spammer ) and predicted value (no spammer)-----True prediction\n\nTrue negative (365)  this result shows that the actual value (spamer) and predicted value (spammer)-------True prediction\n\nFalse positive (450)  this result shows that the actual value (spammer) and predicted value (no spammer)----Type-1 error\n\nFalse negative (2098)  this result shows that the actual value (no spammer) and predicted value (spammer)----Type-2 error","metadata":{}},{"cell_type":"markdown","source":"## Decision Tree\n\ndecision tree as a predictive model which maps features (tree branches) to conclusions about the target value (tree leaves). Tree models where the target variable can take a finite set of values are called classification trees; in these tree structures, leaves represent class labels and branches represent conjunctions of features that lead to those class labels. Decision trees where the target variable can take continuous values (typically real numbers) are called regression trees. Reference Wikipedia.","metadata":{}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier  \n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:20.158540Z","iopub.execute_input":"2022-08-07T21:31:20.159676Z","iopub.status.idle":"2022-08-07T21:31:20.205618Z","shell.execute_reply.started":"2022-08-07T21:31:20.159646Z","shell.execute_reply":"2022-08-07T21:31:20.204636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dt_model=DecisionTreeClassifier().fit(x_train,y_train)\ndt_model","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:20.207246Z","iopub.execute_input":"2022-08-07T21:31:20.207644Z","iopub.status.idle":"2022-08-07T21:31:29.731796Z","shell.execute_reply.started":"2022-08-07T21:31:20.207612Z","shell.execute_reply":"2022-08-07T21:31:29.730728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dt_pred= dt_model.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:29.732933Z","iopub.execute_input":"2022-08-07T21:31:29.733198Z","iopub.status.idle":"2022-08-07T21:31:29.765513Z","shell.execute_reply.started":"2022-08-07T21:31:29.733158Z","shell.execute_reply":"2022-08-07T21:31:29.763746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Error/Score","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\ndt_score = accuracy_score(y_test,dt_pred)\ndt_score","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:29.767677Z","iopub.execute_input":"2022-08-07T21:31:29.768021Z","iopub.status.idle":"2022-08-07T21:31:29.780873Z","shell.execute_reply.started":"2022-08-07T21:31:29.767995Z","shell.execute_reply":"2022-08-07T21:31:29.780151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Our(decision Tree) accuracy score is 97.45% percent, which is extra-ordinary ","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:29.782458Z","iopub.execute_input":"2022-08-07T21:31:29.782766Z","iopub.status.idle":"2022-08-07T21:31:29.787404Z","shell.execute_reply.started":"2022-08-07T21:31:29.782740Z","shell.execute_reply":"2022-08-07T21:31:29.786366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(y_test,dt_pred))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:29.788895Z","iopub.execute_input":"2022-08-07T21:31:29.789166Z","iopub.status.idle":"2022-08-07T21:31:29.888987Z","shell.execute_reply.started":"2022-08-07T21:31:29.789142Z","shell.execute_reply":"2022-08-07T21:31:29.888303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import plot_confusion_matrix\ncm = metrics.confusion_matrix(y_test , dt_pred)\ncm","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:29.890038Z","iopub.execute_input":"2022-08-07T21:31:29.890457Z","iopub.status.idle":"2022-08-07T21:31:29.906627Z","shell.execute_reply.started":"2022-08-07T21:31:29.890432Z","shell.execute_reply":"2022-08-07T21:31:29.905913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('\\nTrue Positives(TP) = ', cm[0,0])\n\nprint('\\nTrue Negatives(TN) = ', cm[1,1])\n\nprint('\\nFalse Positives(FP) = ', cm[0,1])\n\nprint('\\nFalse Negatives(FN) = ', cm[1,0])","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:29.907652Z","iopub.execute_input":"2022-08-07T21:31:29.908284Z","iopub.status.idle":"2022-08-07T21:31:29.913795Z","shell.execute_reply.started":"2022-08-07T21:31:29.908258Z","shell.execute_reply":"2022-08-07T21:31:29.912715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score = metrics.accuracy_score(y_test, dt_pred)\nprint('accutracy score is =', score * 100,'%' )","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:29.915130Z","iopub.execute_input":"2022-08-07T21:31:29.915623Z","iopub.status.idle":"2022-08-07T21:31:29.931736Z","shell.execute_reply.started":"2022-08-07T21:31:29.915594Z","shell.execute_reply":"2022-08-07T21:31:29.930491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,12))\nsns.heatmap(cm ,annot=True, fmt='.3f' , linewidths=.5 ,square=True , cmap='Spectral' )\nplt.ylabel('Actual value')\nplt.xlabel('Predecited value')\nall_sample_title ='Decision tree model accuracy : {0}% '.format( score * 100)\nplt.title(all_sample_title, size  =17)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:29.933575Z","iopub.execute_input":"2022-08-07T21:31:29.934146Z","iopub.status.idle":"2022-08-07T21:31:30.135053Z","shell.execute_reply.started":"2022-08-07T21:31:29.934120Z","shell.execute_reply":"2022-08-07T21:31:30.134007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### how to read confusion matrix\n\nTrue positive (88020)  this result shows that actual value(no spammer ) and predicted value (no spammer)-----True prediction\n\nTrue negative (1403)  this result shows that the actual value (spamer) and predicted value (spammer)-------True prediction\n\nFalse positive (1277)  this result shows that the actual value (spammer) and predicted value (no spammer)----Type-1 error\n\nFalse negative (1060)  this result shows that the actual value (no spammer) and predicted value (spammer)----Type-2 error","metadata":{}},{"cell_type":"markdown","source":"## Support vector machine\n\nSupport Vector Machines which are supervised learning models with associated learning algorithms that analyze data used for classification and regression analysis. Given a set of training samples, each marked as belonging to one or the other of two categories, an SVM training algorithm builds a model that assigns new test samples to one category or the other, making it a non-probabilistic binary linear classifier. Reference Wikipedia.","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import SVC\n\n\nsvm_model = SVC().fit(x_train,y_train)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:31:30.136626Z","iopub.execute_input":"2022-08-07T21:31:30.136872Z","iopub.status.idle":"2022-08-07T21:42:18.964618Z","shell.execute_reply.started":"2022-08-07T21:31:30.136849Z","shell.execute_reply":"2022-08-07T21:42:18.963756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"svm_pred= svm_model.predict(x_test)\nsvm_pred","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:42:18.965819Z","iopub.execute_input":"2022-08-07T21:42:18.967250Z","iopub.status.idle":"2022-08-07T21:44:56.617395Z","shell.execute_reply.started":"2022-08-07T21:42:18.967212Z","shell.execute_reply":"2022-08-07T21:44:56.616355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Error/Score","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\nsvm_score = accuracy_score(y_test , svm_pred)\nprint('Accuracy score =' , svm_score)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:56.618346Z","iopub.execute_input":"2022-08-07T21:44:56.618565Z","iopub.status.idle":"2022-08-07T21:44:56.630001Z","shell.execute_reply.started":"2022-08-07T21:44:56.618543Z","shell.execute_reply":"2022-08-07T21:44:56.628819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Our(Support Vector Machine) accuracy score is 97.31% percent, which is extra-ordinary ","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:56.631690Z","iopub.execute_input":"2022-08-07T21:44:56.632060Z","iopub.status.idle":"2022-08-07T21:44:56.636790Z","shell.execute_reply.started":"2022-08-07T21:44:56.632027Z","shell.execute_reply":"2022-08-07T21:44:56.635812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom sklearn.metrics import confusion_matrix\ncm = confusion_matrix(y_test , svm_pred)\ncm","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:56.638240Z","iopub.execute_input":"2022-08-07T21:44:56.638593Z","iopub.status.idle":"2022-08-07T21:44:56.661943Z","shell.execute_reply.started":"2022-08-07T21:44:56.638559Z","shell.execute_reply":"2022-08-07T21:44:56.661016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('\\nTrue Positives(TP) = ', cm[0,0])\n\nprint('\\nTrue Negatives(TN) = ', cm[1,1])\n\nprint('\\nFalse Positives(FP) = ', cm[0,1])\n\nprint('\\nFalse Negatives(FN) = ', cm[1,0])","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:56.663293Z","iopub.execute_input":"2022-08-07T21:44:56.663523Z","iopub.status.idle":"2022-08-07T21:44:56.669728Z","shell.execute_reply.started":"2022-08-07T21:44:56.663501Z","shell.execute_reply":"2022-08-07T21:44:56.668739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,12))\nsns.heatmap(cm , annot=True , linewidths=5 , square=True,cmap='Spectral')\nplt.xlabel('Actual value ')\nplt.ylabel('predicted value')\nall_sample_title = 'SVM model accuray ( in %) :{0}'.format(svm_score*100)\nprint(all_sample_title)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:56.670753Z","iopub.execute_input":"2022-08-07T21:44:56.670998Z","iopub.status.idle":"2022-08-07T21:44:56.877945Z","shell.execute_reply.started":"2022-08-07T21:44:56.670974Z","shell.execute_reply":"2022-08-07T21:44:56.876617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" True positive (89297) this result shows that actual value(no spammer ) and predicted value (no spammer)-----True prediction\n\nTrue negative (0) this result shows that the actual value (spamer) and predicted value (spammer)-------True prediction\n\nFalse positive (0) this result shows that the actual value (spammer) and predicted value (no spammer)----Type-1 error\n\nFalse negative (2463) this result shows that the actual value (no spammer) and predicted value (spammer)----Type-2 error","metadata":{}},{"cell_type":"markdown","source":"## naive Bayes classifiers\n\nnaive Bayes classifiers are a family of simple probabilistic classifiers based on applying Bayes' theorem with strong (naive) independence assumptions between the features. Naive Bayes classifiers are highly scalable, requiring a number of parameters linear in the number of variables (features) in a learning problem. Reference Wikipedia","metadata":{}},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:56.879559Z","iopub.execute_input":"2022-08-07T21:44:56.879813Z","iopub.status.idle":"2022-08-07T21:44:56.887685Z","shell.execute_reply.started":"2022-08-07T21:44:56.879789Z","shell.execute_reply":"2022-08-07T21:44:56.886592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nb_model = GaussianNB().fit(x_train,y_train)\nnb_model","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:56.888699Z","iopub.execute_input":"2022-08-07T21:44:56.889751Z","iopub.status.idle":"2022-08-07T21:44:57.160380Z","shell.execute_reply.started":"2022-08-07T21:44:56.889704Z","shell.execute_reply":"2022-08-07T21:44:57.158547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nb_pred=nb_model.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:57.161806Z","iopub.execute_input":"2022-08-07T21:44:57.162105Z","iopub.status.idle":"2022-08-07T21:44:57.205915Z","shell.execute_reply.started":"2022-08-07T21:44:57.162080Z","shell.execute_reply":"2022-08-07T21:44:57.204418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Error/Score","metadata":{}},{"cell_type":"code","source":"from sklearn import metrics\nnb_score = metrics.accuracy_score(y_test, nb_pred)\nnb_score","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:57.207068Z","iopub.execute_input":"2022-08-07T21:44:57.207393Z","iopub.status.idle":"2022-08-07T21:44:57.221788Z","shell.execute_reply.started":"2022-08-07T21:44:57.207368Z","shell.execute_reply":"2022-08-07T21:44:57.220057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Our(Naive Bayes) accuracy score is 97.75% percent, which is extra-ordinary ","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:57.223498Z","iopub.execute_input":"2022-08-07T21:44:57.223871Z","iopub.status.idle":"2022-08-07T21:44:57.228866Z","shell.execute_reply.started":"2022-08-07T21:44:57.223844Z","shell.execute_reply":"2022-08-07T21:44:57.227774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(y_test,nb_pred))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:57.229888Z","iopub.execute_input":"2022-08-07T21:44:57.230322Z","iopub.status.idle":"2022-08-07T21:44:57.332030Z","shell.execute_reply.started":"2022-08-07T21:44:57.230288Z","shell.execute_reply":"2022-08-07T21:44:57.330782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# precision rate , recall , f1-score are high means customer are not spammer\n\n# precision rate , recall , f1-score are lower means customer are spammer","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:57.333375Z","iopub.execute_input":"2022-08-07T21:44:57.333709Z","iopub.status.idle":"2022-08-07T21:44:57.337065Z","shell.execute_reply.started":"2022-08-07T21:44:57.333684Z","shell.execute_reply":"2022-08-07T21:44:57.336421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm = metrics.confusion_matrix(y_test , nb_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:57.339007Z","iopub.execute_input":"2022-08-07T21:44:57.340092Z","iopub.status.idle":"2022-08-07T21:44:57.362390Z","shell.execute_reply.started":"2022-08-07T21:44:57.340051Z","shell.execute_reply":"2022-08-07T21:44:57.361226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:57.363348Z","iopub.execute_input":"2022-08-07T21:44:57.363571Z","iopub.status.idle":"2022-08-07T21:44:57.370259Z","shell.execute_reply.started":"2022-08-07T21:44:57.363549Z","shell.execute_reply":"2022-08-07T21:44:57.369520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('\\nTrue Positives(TP) = ', cm[0,0])\n\nprint('\\nTrue Negatives(TN) = ', cm[1,1])\n\nprint('\\nFalse Positives(FP) = ', cm[0,1])\n\nprint('\\nFalse Negatives(FN) = ', cm[1,0])","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:57.371419Z","iopub.execute_input":"2022-08-07T21:44:57.371655Z","iopub.status.idle":"2022-08-07T21:44:57.383130Z","shell.execute_reply.started":"2022-08-07T21:44:57.371633Z","shell.execute_reply":"2022-08-07T21:44:57.381740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nplt.figure(figsize=(12,12))\nsns.heatmap(cm ,annot=True, fmt='.3f' , linewidths=.5 ,square=True , cmap='Spectral' )\nplt.ylabel('Actual value')\nplt.xlabel('Predecited value')\nall_sample_title ='Naive Bayse model accuracy : {0}% '.format( nb_score * 100)\nplt.title(all_sample_title, size  =17)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:57.384891Z","iopub.execute_input":"2022-08-07T21:44:57.385167Z","iopub.status.idle":"2022-08-07T21:44:57.593780Z","shell.execute_reply.started":"2022-08-07T21:44:57.385143Z","shell.execute_reply":"2022-08-07T21:44:57.592223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### how to read confusion matrix\n\nTrue positive (88946) this result shows that actual value(no spammer ) and predicted value (no spammer)-----True prediction\n\nTrue negative (757) this result shows that the actual value (spamer) and predicted value (spammer)-------True prediction\n\nFalse positive (351) this result shows that the actual value (spammer) and predicted value (no spammer)----Type-1 error\n\nFalse negative (1706) this result shows that the actual value (no spammer) and predicted value (spammer)----Type-2 error","metadata":{}},{"cell_type":"markdown","source":"### combine classification result","metadata":{}},{"cell_type":"code","source":"models = pd.DataFrame({\n    'Model': ['Support Vector Machines', 'KNN',  \n               'Naive Bayes', \n              'Decision Tree'],\n    'Score': [svm_score , knn_score,  \n             nb_score ,\n              dt_score]})\nmodels.sort_values(by='Score', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:57.596389Z","iopub.execute_input":"2022-08-07T21:44:57.597752Z","iopub.status.idle":"2022-08-07T21:44:57.614867Z","shell.execute_reply.started":"2022-08-07T21:44:57.597698Z","shell.execute_reply":"2022-08-07T21:44:57.613948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### this shows that all model predicted good result but ''Navie Bayes'' give best possible result","metadata":{}},{"cell_type":"markdown","source":"### prediction based on test data","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv('../input/predict-potential-spammers-on-fiverr/test.csv')\ntest.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:57.615906Z","iopub.execute_input":"2022-08-07T21:44:57.616295Z","iopub.status.idle":"2022-08-07T21:44:57.725780Z","shell.execute_reply.started":"2022-08-07T21:44:57.616261Z","shell.execute_reply":"2022-08-07T21:44:57.725156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test=test.drop(columns=['user_id'])\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:57.726980Z","iopub.execute_input":"2022-08-07T21:44:57.727218Z","iopub.status.idle":"2022-08-07T21:44:57.734494Z","shell.execute_reply.started":"2022-08-07T21:44:57.727167Z","shell.execute_reply":"2022-08-07T21:44:57.733445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:57.736249Z","iopub.execute_input":"2022-08-07T21:44:57.736867Z","iopub.status.idle":"2022-08-07T21:44:57.755138Z","shell.execute_reply.started":"2022-08-07T21:44:57.736841Z","shell.execute_reply":"2022-08-07T21:44:57.754500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### missing value graph","metadata":{}},{"cell_type":"code","source":"sns.heatmap(test.isnull())","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:57.756239Z","iopub.execute_input":"2022-08-07T21:44:57.757110Z","iopub.status.idle":"2022-08-07T21:44:59.013090Z","shell.execute_reply.started":"2022-08-07T21:44:57.757085Z","shell.execute_reply":"2022-08-07T21:44:59.012220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.drop(columns=['X27','X29', 'X30', 'X33', 'X46', 'X47', 'X48'], inplace=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:59.014119Z","iopub.execute_input":"2022-08-07T21:44:59.014373Z","iopub.status.idle":"2022-08-07T21:44:59.022550Z","shell.execute_reply.started":"2022-08-07T21:44:59.014350Z","shell.execute_reply":"2022-08-07T21:44:59.021259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### correlation heatmap \n\n1 shows the strong relationshiop\n\n0 shows no relatopnship\n\n-1 shows negative relationship","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(35,35))\nsns.heatmap(test.corr() , annot=True,cmap=plt.cm.Reds, fmt='.2f')","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:44:59.024395Z","iopub.execute_input":"2022-08-07T21:44:59.024689Z","iopub.status.idle":"2022-08-07T21:45:04.716307Z","shell.execute_reply.started":"2022-08-07T21:44:59.024665Z","shell.execute_reply":"2022-08-07T21:45:04.715212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### prediction on \"test\" data set","metadata":{}},{"cell_type":"code","source":"y_pred = nb_model.predict(test)\ny_pred[:5]","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:45:04.717540Z","iopub.execute_input":"2022-08-07T21:45:04.717975Z","iopub.status.idle":"2022-08-07T21:45:04.734753Z","shell.execute_reply.started":"2022-08-07T21:45:04.717949Z","shell.execute_reply":"2022-08-07T21:45:04.733848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = pd.DataFrame(y_pred,columns=['predicted_label'])\nprediction.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:45:04.736227Z","iopub.execute_input":"2022-08-07T21:45:04.736476Z","iopub.status.idle":"2022-08-07T21:45:04.744765Z","shell.execute_reply.started":"2022-08-07T21:45:04.736453Z","shell.execute_reply":"2022-08-07T21:45:04.744117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission File","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv('../input/predict-potential-spammers-on-fiverr/sample_submission.csv')\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:45:04.745795Z","iopub.execute_input":"2022-08-07T21:45:04.746091Z","iopub.status.idle":"2022-08-07T21:45:04.779975Z","shell.execute_reply.started":"2022-08-07T21:45:04.746068Z","shell.execute_reply":"2022-08-07T21:45:04.778773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:45:04.780983Z","iopub.execute_input":"2022-08-07T21:45:04.781752Z","iopub.status.idle":"2022-08-07T21:45:04.786131Z","shell.execute_reply.started":"2022-08-07T21:45:04.781725Z","shell.execute_reply":"2022-08-07T21:45:04.785600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head(10)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:45:04.787245Z","iopub.execute_input":"2022-08-07T21:45:04.787570Z","iopub.status.idle":"2022-08-07T21:45:04.807965Z","shell.execute_reply.started":"2022-08-07T21:45:04.787539Z","shell.execute_reply":"2022-08-07T21:45:04.806657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['predicted_label']=prediction","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:45:04.809068Z","iopub.execute_input":"2022-08-07T21:45:04.810641Z","iopub.status.idle":"2022-08-07T21:45:04.817850Z","shell.execute_reply.started":"2022-08-07T21:45:04.810582Z","shell.execute_reply":"2022-08-07T21:45:04.816712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:45:04.819696Z","iopub.execute_input":"2022-08-07T21:45:04.820427Z","iopub.status.idle":"2022-08-07T21:45:04.833075Z","shell.execute_reply.started":"2022-08-07T21:45:04.820392Z","shell.execute_reply":"2022-08-07T21:45:04.831671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.columns = ['user_id','label','predicted_label']\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:45:04.834935Z","iopub.execute_input":"2022-08-07T21:45:04.835250Z","iopub.status.idle":"2022-08-07T21:45:04.840988Z","shell.execute_reply.started":"2022-08-07T21:45:04.835221Z","shell.execute_reply":"2022-08-07T21:45:04.839670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prediction","metadata":{}},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:45:04.842416Z","iopub.execute_input":"2022-08-07T21:45:04.843326Z","iopub.status.idle":"2022-08-07T21:45:04.858928Z","shell.execute_reply.started":"2022-08-07T21:45:04.843298Z","shell.execute_reply":"2022-08-07T21:45:04.857113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission=submission.drop(columns='label'  ,  axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:51:43.610804Z","iopub.execute_input":"2022-08-07T21:51:43.611155Z","iopub.status.idle":"2022-08-07T21:51:43.616892Z","shell.execute_reply.started":"2022-08-07T21:51:43.611129Z","shell.execute_reply":"2022-08-07T21:51:43.616146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:51:51.371584Z","iopub.execute_input":"2022-08-07T21:51:51.371914Z","iopub.status.idle":"2022-08-07T21:51:51.383552Z","shell.execute_reply.started":"2022-08-07T21:51:51.371889Z","shell.execute_reply":"2022-08-07T21:51:51.382536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:53:49.805117Z","iopub.execute_input":"2022-08-07T21:53:49.805497Z","iopub.status.idle":"2022-08-07T21:53:49.836191Z","shell.execute_reply.started":"2022-08-07T21:53:49.805469Z","shell.execute_reply":"2022-08-07T21:53:49.835474Z"},"trusted":true},"execution_count":null,"outputs":[]}]}