{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-06T13:22:45.680885Z","iopub.execute_input":"2022-08-06T13:22:45.682091Z","iopub.status.idle":"2022-08-06T13:22:45.718854Z","shell.execute_reply.started":"2022-08-06T13:22:45.681976Z","shell.execute_reply":"2022-08-06T13:22:45.717864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#importing libraries\nimport xgboost\nfrom sklearn import metrics\nfrom sklearn.model_selection import RandomizedSearchCV, GridSearchCV","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:22:45.721108Z","iopub.execute_input":"2022-08-06T13:22:45.721465Z","iopub.status.idle":"2022-08-06T13:22:47.195106Z","shell.execute_reply.started":"2022-08-06T13:22:45.721433Z","shell.execute_reply":"2022-08-06T13:22:47.193922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Let's take test and train dataset as input\ntrain_data = pd.read_csv(\"/kaggle/input/predict-potential-spammers-on-fiverr/train.csv\")\ntest_data = pd.read_csv(\"/kaggle/input/predict-potential-spammers-on-fiverr/test.csv\")\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:22:47.197455Z","iopub.execute_input":"2022-08-06T13:22:47.198016Z","iopub.status.idle":"2022-08-06T13:22:49.835600Z","shell.execute_reply.started":"2022-08-06T13:22:47.197966Z","shell.execute_reply":"2022-08-06T13:22:49.834597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"#removing user_id column from both training and test set\ntrain_data = train_data.drop(columns = [\"user_id\"])\ntest_data = test_data.drop(columns = [\"user_id\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:22:49.837057Z","iopub.execute_input":"2022-08-06T13:22:49.837666Z","iopub.status.idle":"2022-08-06T13:22:49.912753Z","shell.execute_reply.started":"2022-08-06T13:22:49.837629Z","shell.execute_reply":"2022-08-06T13:22:49.911765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#now split the train dataset into X and Y\nX = train_data.drop(columns = [\"label\"])\nY = train_data[[\"label\"]]","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:22:49.915535Z","iopub.execute_input":"2022-08-06T13:22:49.916150Z","iopub.status.idle":"2022-08-06T13:22:49.992766Z","shell.execute_reply.started":"2022-08-06T13:22:49.916113Z","shell.execute_reply":"2022-08-06T13:22:49.991706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nparams = {\n    \"learning_rate\" : [0.05,0.1,0.15,0.20,0.25,0.30],\n    \"max_depth\"  : [3, 4, 5, 6, 8, 10, 12, 15],\n    \"min_child_weight\" : [1, 3, 5, 7],\n    \"gamma\" : [0.0, 0.1, 0.2, 0.3, 0.4],\n    \"colsmaple_bytree\" : [0.3, 0.4, 0.5, 0.7]\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:22:49.994955Z","iopub.execute_input":"2022-08-06T13:22:49.996156Z","iopub.status.idle":"2022-08-06T13:22:50.010373Z","shell.execute_reply.started":"2022-08-06T13:22:49.996106Z","shell.execute_reply":"2022-08-06T13:22:50.007706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#let's use customized parameters\nxgb_model1 = xgboost.XGBClassifier()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:22:50.012781Z","iopub.execute_input":"2022-08-06T13:22:50.013770Z","iopub.status.idle":"2022-08-06T13:22:50.023554Z","shell.execute_reply.started":"2022-08-06T13:22:50.013718Z","shell.execute_reply":"2022-08-06T13:22:50.022119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_model1.fit(X,Y)\nxgb_model1","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:22:50.025155Z","iopub.execute_input":"2022-08-06T13:22:50.025647Z","iopub.status.idle":"2022-08-06T13:23:50.596354Z","shell.execute_reply.started":"2022-08-06T13:22:50.025601Z","shell.execute_reply":"2022-08-06T13:23:50.595193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict1 = xgb_model1.predict(test_data)\npredict1","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:23:50.597883Z","iopub.execute_input":"2022-08-06T13:23:50.598262Z","iopub.status.idle":"2022-08-06T13:23:50.651286Z","shell.execute_reply.started":"2022-08-06T13:23:50.598227Z","shell.execute_reply":"2022-08-06T13:23:50.645577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.unique(predict1, return_counts= True)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:23:50.653147Z","iopub.execute_input":"2022-08-06T13:23:50.653902Z","iopub.status.idle":"2022-08-06T13:23:50.662004Z","shell.execute_reply.started":"2022-08-06T13:23:50.653858Z","shell.execute_reply":"2022-08-06T13:23:50.661127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_train = xgb_model1.predict(X)\nprint(metrics.classification_report(Y, predict_train))\nprint(metrics.confusion_matrix(Y, predict_train))","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:23:50.664078Z","iopub.execute_input":"2022-08-06T13:23:50.664941Z","iopub.status.idle":"2022-08-06T13:23:52.207119Z","shell.execute_reply.started":"2022-08-06T13:23:50.664891Z","shell.execute_reply":"2022-08-06T13:23:52.205922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(\"/kaggle/input/predict-potential-spammers-on-fiverr/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:23:52.208560Z","iopub.execute_input":"2022-08-06T13:23:52.209045Z","iopub.status.idle":"2022-08-06T13:23:52.226923Z","shell.execute_reply.started":"2022-08-06T13:23:52.209011Z","shell.execute_reply":"2022-08-06T13:23:52.225813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:23:52.228346Z","iopub.execute_input":"2022-08-06T13:23:52.228730Z","iopub.status.idle":"2022-08-06T13:23:52.239328Z","shell.execute_reply.started":"2022-08-06T13:23:52.228696Z","shell.execute_reply":"2022-08-06T13:23:52.238144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.label = predict1","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:23:52.243225Z","iopub.execute_input":"2022-08-06T13:23:52.243826Z","iopub.status.idle":"2022-08-06T13:23:52.252245Z","shell.execute_reply.started":"2022-08-06T13:23:52.243788Z","shell.execute_reply":"2022-08-06T13:23:52.250916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.columns = ['user_id', 'prediction']","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:23:52.253290Z","iopub.execute_input":"2022-08-06T13:23:52.253675Z","iopub.status.idle":"2022-08-06T13:23:52.264245Z","shell.execute_reply.started":"2022-08-06T13:23:52.253641Z","shell.execute_reply":"2022-08-06T13:23:52.262918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.prediction.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:23:52.265881Z","iopub.execute_input":"2022-08-06T13:23:52.266244Z","iopub.status.idle":"2022-08-06T13:23:52.280152Z","shell.execute_reply.started":"2022-08-06T13:23:52.266211Z","shell.execute_reply":"2022-08-06T13:23:52.279043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:23:52.281177Z","iopub.execute_input":"2022-08-06T13:23:52.281547Z","iopub.status.idle":"2022-08-06T13:23:52.321818Z","shell.execute_reply.started":"2022-08-06T13:23:52.281488Z","shell.execute_reply":"2022-08-06T13:23:52.320774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n","metadata":{},"execution_count":null,"outputs":[]}]}