{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-01T18:55:25.628630Z","iopub.execute_input":"2022-08-01T18:55:25.628996Z","iopub.status.idle":"2022-08-01T18:55:25.643777Z","shell.execute_reply.started":"2022-08-01T18:55:25.628964Z","shell.execute_reply":"2022-08-01T18:55:25.642635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '../input/predict-potential-spammers-on-fiverr/train.csv'\ndf = pd.read_csv(path,index_col = 'user_id')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:55:42.713163Z","iopub.execute_input":"2022-08-01T18:55:42.713635Z","iopub.status.idle":"2022-08-01T18:55:45.283568Z","shell.execute_reply.started":"2022-08-01T18:55:42.713587Z","shell.execute_reply":"2022-08-01T18:55:45.282100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfeatures = df.columns.drop('label')\nX = df[features]\ny = df['label']\n\nX_train,X_val,y_train,y_val = train_test_split(X,y,random_state = 0) \nX.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:55:48.181177Z","iopub.execute_input":"2022-08-01T18:55:48.182616Z","iopub.status.idle":"2022-08-01T18:55:48.831924Z","shell.execute_reply.started":"2022-08-01T18:55:48.182560Z","shell.execute_reply":"2022-08-01T18:55:48.831026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import xgboost as xgb\nmodel = xgb.XGBClassifier(n_estimators=500,learning_rate = 0.05)\nmodel.fit(X_train,y_train,early_stopping_rounds = 5, eval_set = [(X_val,y_val)],verbose = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:58:15.560388Z","iopub.execute_input":"2022-08-01T18:58:15.562216Z","iopub.status.idle":"2022-08-01T19:02:14.769637Z","shell.execute_reply.started":"2022-08-01T18:58:15.562158Z","shell.execute_reply":"2022-08-01T19:02:14.768419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error\npred = model.predict(X_val)\nmean_absolute_error(y_val,pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T19:02:24.468066Z","iopub.execute_input":"2022-08-01T19:02:24.468507Z","iopub.status.idle":"2022-08-01T19:02:25.202041Z","shell.execute_reply.started":"2022-08-01T19:02:24.468472Z","shell.execute_reply":"2022-08-01T19:02:25.200682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '../input/predict-potential-spammers-on-fiverr/test.csv'\ndf = pd.read_csv(path,index_col = 'user_id')\ndf.head()\npred = model.predict(df)\ndf.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-01T19:02:27.797521Z","iopub.execute_input":"2022-08-01T19:02:27.797945Z","iopub.status.idle":"2022-08-01T19:02:28.075986Z","shell.execute_reply.started":"2022-08-01T19:02:27.797912Z","shell.execute_reply":"2022-08-01T19:02:28.074525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({'user_id': df.index,'prediction' : pred})\noutput.to_csv('submission.csv',index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T19:02:40.198301Z","iopub.execute_input":"2022-08-01T19:02:40.198781Z","iopub.status.idle":"2022-08-01T19:02:40.256114Z","shell.execute_reply.started":"2022-08-01T19:02:40.198742Z","shell.execute_reply":"2022-08-01T19:02:40.254798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = './submission.csv'\ndf = pd.read_csv(path,index_col = 'user_id')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T19:03:08.481896Z","iopub.execute_input":"2022-08-01T19:03:08.482367Z","iopub.status.idle":"2022-08-01T19:03:08.507843Z","shell.execute_reply.started":"2022-08-01T19:03:08.482331Z","shell.execute_reply":"2022-08-01T19:03:08.506801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}