{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### I will be taking you through the steps I followed on predicting potential spammers using Fiverr dataset. Will try to keep it simple as possible. Please upvote the notebook if you find it helful","metadata":{}},{"cell_type":"markdown","source":"Problem type: Binary Classification","metadata":{}},{"cell_type":"code","source":"#importing the necessary packages\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nimport h2o\nfrom h2o.automl import H2OAutoML\n\nfrom sklearn.metrics import confusion_matrix, classification_report\nfrom sklearn.metrics import accuracy_score, f1_score, recall_score, precision_score, precision_recall_curve","metadata":{"execution":{"iopub.status.busy":"2022-08-05T01:40:40.383990Z","iopub.execute_input":"2022-08-05T01:40:40.384383Z","iopub.status.idle":"2022-08-05T01:40:41.738332Z","shell.execute_reply.started":"2022-08-05T01:40:40.384297Z","shell.execute_reply":"2022-08-05T01:40:41.737044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#importing the dataset\n\ndata = pd.read_csv('/kaggle/input/predict-potential-spammers-on-fiverr/train.csv')\nprint(data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T01:40:41.740593Z","iopub.execute_input":"2022-08-05T01:40:41.741455Z","iopub.status.idle":"2022-08-05T01:40:44.135005Z","shell.execute_reply.started":"2022-08-05T01:40:41.741419Z","shell.execute_reply":"2022-08-05T01:40:44.133655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#let's check for any missing values\n\nnn = data.isna().any()\nnull_cols = nn.index[nn].tolist()\ndata[null_cols].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T01:40:44.136801Z","iopub.execute_input":"2022-08-05T01:40:44.137400Z","iopub.status.idle":"2022-08-05T01:40:44.173692Z","shell.execute_reply.started":"2022-08-05T01:40:44.137367Z","shell.execute_reply":"2022-08-05T01:40:44.172425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Only the attribute X13 is having null records and as this is a small amount of records I am not going to drop the records. Null replacement, dropping or not dropping the records will not add any significance improvement to the model","metadata":{}},{"cell_type":"code","source":"#checking the class distribution across the dataset\n\ndata.label.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T01:40:44.177213Z","iopub.execute_input":"2022-08-05T01:40:44.177723Z","iopub.status.idle":"2022-08-05T01:40:44.197501Z","shell.execute_reply.started":"2022-08-05T01:40:44.177675Z","shell.execute_reply":"2022-08-05T01:40:44.196003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset is highly imbalanced with around 3% data on the predictive class","metadata":{"execution":{"iopub.status.busy":"2022-08-04T05:44:45.712704Z","iopub.execute_input":"2022-08-04T05:44:45.713082Z","iopub.status.idle":"2022-08-04T05:44:45.719325Z","shell.execute_reply.started":"2022-08-04T05:44:45.713055Z","shell.execute_reply":"2022-08-04T05:44:45.718118Z"}}},{"cell_type":"code","source":"#Identifying categorical and numerical attributes\n\ncol_vals = data.nunique()\ncol_vals","metadata":{"execution":{"iopub.status.busy":"2022-08-05T01:40:44.199534Z","iopub.execute_input":"2022-08-05T01:40:44.200005Z","iopub.status.idle":"2022-08-05T01:40:44.383698Z","shell.execute_reply.started":"2022-08-05T01:40:44.199962Z","shell.execute_reply":"2022-08-05T01:40:44.382264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are few attributes with only 1 unique value. These does not contribute to the model performance. Therefore these column will be removed from the dataset. ","metadata":{}},{"cell_type":"code","source":"cols_to_remove = list()\nfor attr in data.columns:\n    if data[attr].nunique() == 1:\n        cols_to_remove.append(attr)\n        \ncols_to_remove","metadata":{"execution":{"iopub.status.busy":"2022-08-05T01:40:44.385608Z","iopub.execute_input":"2022-08-05T01:40:44.386200Z","iopub.status.idle":"2022-08-05T01:40:44.609686Z","shell.execute_reply.started":"2022-08-05T01:40:44.386156Z","shell.execute_reply":"2022-08-05T01:40:44.608330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I am not going to do further EDA on this dataset. Will try with EDA in the Feature Engineering part. Let's train the data with H20 autoML and develop a baseline model first.","metadata":{}},{"cell_type":"markdown","source":"## Model Development | AutoML | H20","metadata":{}},{"cell_type":"code","source":"h2o.init()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T01:40:44.611395Z","iopub.execute_input":"2022-08-05T01:40:44.611852Z","iopub.status.idle":"2022-08-05T01:40:52.343479Z","shell.execute_reply.started":"2022-08-05T01:40:44.611809Z","shell.execute_reply":"2022-08-05T01:40:52.342118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#converting the dataframes into H20  dataframes\n\ndata_h = h2o.H2OFrame(data)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T01:40:52.349470Z","iopub.execute_input":"2022-08-05T01:40:52.352462Z","iopub.status.idle":"2022-08-05T01:41:14.243779Z","shell.execute_reply.started":"2022-08-05T01:40:52.352400Z","shell.execute_reply":"2022-08-05T01:41:14.242221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the dataset into train and test\ntrain, test = data_h.split_frame(ratios = [.999], seed = 1234)\n\nprint('training dataset size: ', train.shape)\nprint('test dataset size: ', test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T01:41:14.245576Z","iopub.execute_input":"2022-08-05T01:41:14.245981Z","iopub.status.idle":"2022-08-05T01:41:15.425331Z","shell.execute_reply.started":"2022-08-05T01:41:14.245945Z","shell.execute_reply":"2022-08-05T01:41:15.424076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#removing unwanted attribues: user_id, label and cols_to_remove from training data\n\nx = train.columns\nprint(x)\nprint(len(x))\ny = 'label'\nz = 'user_id'\n\ncols_to_remove.append(y)\ncols_to_remove.append(z)\ncols_to_remove","metadata":{"execution":{"iopub.status.busy":"2022-08-05T01:41:15.434707Z","iopub.execute_input":"2022-08-05T01:41:15.437537Z","iopub.status.idle":"2022-08-05T01:41:15.445900Z","shell.execute_reply.started":"2022-08-05T01:41:15.437485Z","shell.execute_reply":"2022-08-05T01:41:15.445032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#adding this in the latest version to check whether not dropping columns manually will affect the model scores\n\ncols_to_remove_upd = []\ncols_to_remove_upd.append(y)\ncols_to_remove_upd.append(z)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#removing unwanted attribues: user_id, label and cols_to_remove from training data\n# for cols in cols_to_remove:\n#     x.remove(cols)\n\nfor cols in cols_to_remove_upd:\n    x.remove(cols)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T01:41:15.447336Z","iopub.execute_input":"2022-08-05T01:41:15.447994Z","iopub.status.idle":"2022-08-05T01:41:15.456793Z","shell.execute_reply.started":"2022-08-05T01:41:15.447950Z","shell.execute_reply":"2022-08-05T01:41:15.455932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(x)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T01:41:15.458964Z","iopub.execute_input":"2022-08-05T01:41:15.459986Z","iopub.status.idle":"2022-08-05T01:41:15.470164Z","shell.execute_reply.started":"2022-08-05T01:41:15.459951Z","shell.execute_reply":"2022-08-05T01:41:15.468890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"9 attributes from the original dataset will be neglected while training the model.","metadata":{}},{"cell_type":"code","source":"# For binary classification, response should be a factor\ntrain[y] = train[y].asfactor()\ntest[y] = test[y].asfactor()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T01:41:15.471284Z","iopub.execute_input":"2022-08-05T01:41:15.473765Z","iopub.status.idle":"2022-08-05T01:41:15.480308Z","shell.execute_reply.started":"2022-08-05T01:41:15.473718Z","shell.execute_reply":"2022-08-05T01:41:15.479322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Run AutoML for 20 base models\naml = H2OAutoML(max_models=10, \n                #balance_classes=True, \n                seed=1, \n                #nfolds=5, \n                #keep_cross_validation_predictions=True\n               )\naml.train(x=x, y=y, training_frame=train)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T01:41:15.481979Z","iopub.execute_input":"2022-08-05T01:41:15.482815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# View the AutoML Leaderboard\nlb = aml.leaderboard\nlb.head(rows=lb.nrows)  # Print all rows instead of default (10 rows)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's save our best model","metadata":{}},{"cell_type":"code","source":"best_model = aml.leader\nbest_model_path = h2o.save_model(model=best_model,path='fiverr_spammer_best_model', force=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's dig into the best model and check its performance !","metadata":{}},{"cell_type":"code","source":"print(best_model)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predicting on the test dataset","metadata":{}},{"cell_type":"code","source":"#loading the best model\n\nour_model = h2o.load_model(path = best_model_path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_test = our_model.predict(test)\nprint(preds_test)\nprint(preds_test.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking the testing performance\n\ntest_labels = test['label']\npred_labels = preds_test['predict']\nprint(type(test_labels))\nprint(type(pred_labels))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred_df = test_labels.concat(pred_labels)\ntest_pred_df = test_pred_df.as_data_frame()\nconf_matrix = confusion_matrix(test_pred_df['label'], test_pred_df['predict'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print the confusion matrix using Matplotlib\n#\nfig, ax = plt.subplots(figsize=(4, 4))\nax.matshow(conf_matrix, cmap=plt.cm.Blues, alpha=0.3)\nfor i in range(conf_matrix.shape[0]):\n    for j in range(conf_matrix.shape[1]):\n        ax.text(x=j, y=i,s=conf_matrix[i, j], va='center', ha='center', size='xx-large')\n \nplt.xlabel('Predictions', fontsize=18)\nplt.ylabel('Actuals', fontsize=18)\nplt.title('Confusion Matrix', fontsize=18)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(test_pred_df['label'], test_pred_df['predict']))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predictions for the final evaluation","metadata":{}},{"cell_type":"code","source":"to_pred = pd.read_csv('/kaggle/input/predict-potential-spammers-on-fiverr/test.csv')\nto_pred.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#arranging the dataset \n\npred_user_id = to_pred['user_id']\nto_pred.drop(['X27', 'X29', 'X30', 'X33', 'X46', 'X47', 'X48', 'user_id'], axis = 1)\n\n#converting to H2O dataframe\nto_pred_h = h2o.H2OFrame(to_pred)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predicting for the final dataset and converting to a dataframe\n\npreds_val = our_model.predict(to_pred_h)\npreds_val.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_val","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#h2o.export_file(preds_val, path = 'predicted_values.csv', force = True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_predictions = preds_val['predict']\nfinal_predictions.rename(columns = {'predict': 'prediction'})\nfinal_predictions.head(3)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(type(pred_user_id))\nprint(type(final_predictions))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_id_df = pred_user_id.to_frame()\nuser_id_dh = h2o.H2OFrame(user_id_df)\nprint(type(user_id_dh))\nprint(type(final_predictions))\nsubmission = user_id_dh.concat(final_predictions)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(submission.shape)\nsubmission.head(3)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"h2o.export_file(submission, path = 'submission.csv', force = True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###                                                Hope you found this informative and useful ! Cheers !!","metadata":{}}]}