{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-20T12:08:13.426268Z","iopub.execute_input":"2022-04-20T12:08:13.426684Z","iopub.status.idle":"2022-04-20T12:08:13.442931Z","shell.execute_reply.started":"2022-04-20T12:08:13.426594Z","shell.execute_reply":"2022-04-20T12:08:13.442146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n","metadata":{"execution":{"iopub.status.busy":"2022-04-20T12:08:18.875302Z","iopub.execute_input":"2022-04-20T12:08:18.875966Z","iopub.status.idle":"2022-04-20T12:08:19.709813Z","shell.execute_reply.started":"2022-04-20T12:08:18.875927Z","shell.execute_reply":"2022-04-20T12:08:19.708953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### df = pd.read_csv('/kaggle/input/talkingdata-adtracking-fraud-detection/train_sample.csv')\ndf.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-04-18T14:00:26.608488Z","iopub.execute_input":"2022-04-18T14:00:26.609096Z","iopub.status.idle":"2022-04-18T14:00:26.830219Z","shell.execute_reply.started":"2022-04-18T14:00:26.60905Z","shell.execute_reply":"2022-04-18T14:00:26.829452Z"}}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/talkingdata-adtracking-fraud-detection/train_sample.csv')\ndf.head(20)","metadata":{"execution":{"iopub.status.busy":"2022-04-20T12:08:48.422441Z","iopub.execute_input":"2022-04-20T12:08:48.422798Z","iopub.status.idle":"2022-04-20T12:08:48.572331Z","shell.execute_reply.started":"2022-04-20T12:08:48.422767Z","shell.execute_reply":"2022-04-20T12:08:48.57117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('precision', 2)\ndf.describe()","metadata":{"execution":{"iopub.status.busy":"2022-04-20T12:33:49.695996Z","iopub.execute_input":"2022-04-20T12:33:49.696775Z","iopub.status.idle":"2022-04-20T12:33:49.761002Z","shell.execute_reply.started":"2022-04-20T12:33:49.696722Z","shell.execute_reply":"2022-04-20T12:33:49.759983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.nunique(dropna=True)","metadata":{"execution":{"iopub.status.busy":"2022-04-20T12:34:10.419561Z","iopub.execute_input":"2022-04-20T12:34:10.419965Z","iopub.status.idle":"2022-04-20T12:34:10.499556Z","shell.execute_reply.started":"2022-04-20T12:34:10.419925Z","shell.execute_reply":"2022-04-20T12:34:10.498438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_points,data_features = df.shape\n\n(data_points,data_features)","metadata":{"execution":{"iopub.status.busy":"2022-04-20T12:34:14.816119Z","iopub.execute_input":"2022-04-20T12:34:14.816831Z","iopub.status.idle":"2022-04-20T12:34:14.822799Z","shell.execute_reply.started":"2022-04-20T12:34:14.816785Z","shell.execute_reply":"2022-04-20T12:34:14.821858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"percentage = (df.is_attributed.values == 0).mean()\npercentage","metadata":{"execution":{"iopub.status.busy":"2022-04-20T12:34:18.460343Z","iopub.execute_input":"2022-04-20T12:34:18.461127Z","iopub.status.idle":"2022-04-20T12:34:18.468685Z","shell.execute_reply.started":"2022-04-20T12:34:18.46108Z","shell.execute_reply":"2022-04-20T12:34:18.467774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nplot  = sns.barplot(['Not Downloaded','Downloaded'],[percentage*100, (1-percentage)*100])\nplot.set(ylabel = 'Propotion')\n\nfor i in range(2):\n    a = plot.patches[i]\n    height = a.get_height()\n    value = abs(percentage - i)\n    plot.text(a.get_x()+a.get_width()/2.,height+0.5,round(value*100,2),ha=\"center\")","metadata":{"execution":{"iopub.status.busy":"2022-04-20T12:34:21.754748Z","iopub.execute_input":"2022-04-20T12:34:21.755446Z","iopub.status.idle":"2022-04-20T12:34:21.932227Z","shell.execute_reply.started":"2022-04-20T12:34:21.755408Z","shell.execute_reply":"2022-04-20T12:34:21.931069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-06-29T06:40:10.95756Z","iopub.execute_input":"2021-06-29T06:40:10.958004Z","iopub.status.idle":"2021-06-29T06:40:11.013559Z","shell.execute_reply.started":"2021-06-29T06:40:10.957959Z","shell.execute_reply":"2021-06-29T06:40:11.012355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_values_1 = df.nunique(dropna=True)\nunique_values_1","metadata":{"execution":{"iopub.status.busy":"2022-04-20T12:34:25.546129Z","iopub.execute_input":"2022-04-20T12:34:25.546657Z","iopub.status.idle":"2022-04-20T12:34:25.606651Z","shell.execute_reply.started":"2022-04-20T12:34:25.546624Z","shell.execute_reply":"2022-04-20T12:34:25.605695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import xgboost as xgb\nfrom xgboost import plot_importance\nfrom sklearn.model_selection import cross_val_score\nimport gc\n\nimport os\nprint(os.listdir(\"../input/talkingdata-adtracking-fraud-detection\"))","metadata":{"execution":{"iopub.status.busy":"2022-04-20T12:34:29.307157Z","iopub.execute_input":"2022-04-20T12:34:29.307543Z","iopub.status.idle":"2022-04-20T12:34:29.622398Z","shell.execute_reply.started":"2022-04-20T12:34:29.307503Z","shell.execute_reply":"2022-04-20T12:34:29.621436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_columns = ['ip', 'app', 'device', 'os', 'channel', 'click_time', 'is_attributed']\ntest_columns  = ['ip', 'app', 'device', 'os', 'channel', 'click_time', 'click_id']\ndtypes = {\n        'ip'            : 'uint32',\n        'app'           : 'uint16',\n        'device'        : 'uint16',\n        'os'            : 'uint16',\n        'channel'       : 'uint16',\n        'is_attributed' : 'uint8',\n        'click_id'      : 'uint32'\n        }\n\ntrain_df = pd.read_csv(\"../input/talkingdata-adtracking-fraud-detection/train.csv\", skiprows=range(1,123903891), nrows=6100000, usecols=train_columns, dtype=dtypes)\ntest_df = pd.read_csv(\"../input/talkingdata-adtracking-fraud-detection/test.csv\", usecols=test_columns, dtype=dtypes)\n\ntrain_y = train_df['is_attributed']\ntrain_df.drop(['is_attributed'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-04-20T12:34:32.747024Z","iopub.execute_input":"2022-04-20T12:34:32.74738Z","iopub.status.idle":"2022-04-20T12:37:19.233104Z","shell.execute_reply.started":"2022-04-20T12:34:32.747349Z","shell.execute_reply":"2022-04-20T12:37:19.231626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(10)\n","metadata":{"execution":{"iopub.status.busy":"2022-04-20T12:37:38.457923Z","iopub.execute_input":"2022-04-20T12:37:38.458315Z","iopub.status.idle":"2022-04-20T12:37:38.471216Z","shell.execute_reply.started":"2022-04-20T12:37:38.45828Z","shell.execute_reply":"2022-04-20T12:37:38.470499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-04-20T12:37:43.350948Z","iopub.execute_input":"2022-04-20T12:37:43.351499Z","iopub.status.idle":"2022-04-20T12:37:43.364892Z","shell.execute_reply.started":"2022-04-20T12:37:43.351461Z","shell.execute_reply":"2022-04-20T12:37:43.363832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def timeconvert(df):\n    df['datetime'] = pd.to_datetime(df['click_time'])\n    df[\"dayofweek\"] = df[\"datetime\"].dt.dayofweek\n    df[\"dayofyear\"] = df[\"datetime\"].dt.dayofyear\n    df[\"hourofday\"] = df[\"datetime\"].dt.hour\n    df.drop(['click_time', 'datetime'], axis=1, inplace=True)\n    \n    return df\n","metadata":{"execution":{"iopub.status.busy":"2022-04-20T12:38:13.11003Z","iopub.execute_input":"2022-04-20T12:38:13.110394Z","iopub.status.idle":"2022-04-20T12:38:13.116452Z","shell.execute_reply.started":"2022-04-20T12:38:13.110362Z","shell.execute_reply":"2022-04-20T12:38:13.115354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def nb_click_per_ip_per_day(df):\n    df['dateclick'] = pd.to_datetime(df['click_time']).dt.date\n    return df.groupby(['ip', 'dateclick']).size().reset_index().rename(columns={0:'nbclick'})","metadata":{"execution":{"iopub.status.busy":"2022-04-20T12:38:21.515704Z","iopub.execute_input":"2022-04-20T12:38:21.51611Z","iopub.status.idle":"2022-04-20T12:38:21.521863Z","shell.execute_reply.started":"2022-04-20T12:38:21.516076Z","shell.execute_reply":"2022-04-20T12:38:21.520971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nrow_train = train_df.shape[0]\nmerge = pd.concat([train_df, test_df])\n\n# Count the average number of clicks per day by ip\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-04-20T12:38:25.837064Z","iopub.execute_input":"2022-04-20T12:38:25.837441Z","iopub.status.idle":"2022-04-20T12:38:26.926305Z","shell.execute_reply.started":"2022-04-20T12:38:25.837409Z","shell.execute_reply":"2022-04-20T12:38:26.925575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merge['dateclick'] = pd.to_datetime(merge['click_time']).dt.date\nip_count1 = merge.groupby(['ip', 'dateclick']).size().reset_index().rename(columns={0:'nbclick'})","metadata":{"execution":{"iopub.status.busy":"2022-04-20T12:39:24.740046Z","iopub.execute_input":"2022-04-20T12:39:24.740431Z","iopub.status.idle":"2022-04-20T12:39:50.978609Z","shell.execute_reply.started":"2022-04-20T12:39:24.740398Z","shell.execute_reply":"2022-04-20T12:39:50.977602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nip_count = ip_count1.groupby(['ip']).mean().reset_index()\nip_count.columns = ['ip', 'avg_clicks_per_day_by_ip']\nmerge = pd.merge(merge, ip_count, on='ip', how='left', sort=False)\n#merge.drop('ip', axis=1, inplace=True)\nmerge.drop('dateclick', axis=1, inplace=True)\n\ntrain_df = merge[:nrow_train]\ntest_df = merge[nrow_train:]\n\nmerge.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-15T11:40:32.716415Z","iopub.execute_input":"2021-08-15T11:40:32.71678Z","iopub.status.idle":"2021-08-15T11:40:43.033268Z","shell.execute_reply.started":"2021-08-15T11:40:32.71675Z","shell.execute_reply":"2021-08-15T11:40:43.03217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.drop('click_id', axis=1, inplace=True) #can even do a dropna\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-15T11:48:18.116763Z","iopub.execute_input":"2021-08-15T11:48:18.117149Z","iopub.status.idle":"2021-08-15T11:48:18.348181Z","shell.execute_reply.started":"2021-08-15T11:48:18.117117Z","shell.execute_reply":"2021-08-15T11:48:18.347024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = timeconvert(train_df)\ntrain_df.head()\n\n#ignore the warnings below","metadata":{"execution":{"iopub.status.busy":"2021-08-15T11:49:18.0798Z","iopub.execute_input":"2021-08-15T11:49:18.080164Z","iopub.status.idle":"2021-08-15T11:49:21.568728Z","shell.execute_reply.started":"2021-08-15T11:49:18.080134Z","shell.execute_reply":"2021-08-15T11:49:21.567593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtrain = xgb.DMatrix(train_df, train_y)\n\n# Set the params for xgboost model\n# i got this parameters from other notebooks directly i am using it over here , you can even find this out\nparams = {'eta': 0.3,\n          'tree_method': \"hist\",\n          'grow_policy': \"lossguide\",\n          'max_leaves': 1400,  \n          'max_depth': 0, \n          'subsample': 0.9, \n          'colsample_bytree': 0.7, \n          'colsample_bylevel':0.7,\n          'min_child_weight':0,\n          'alpha':4,\n          'objective': 'binary:logistic', \n          'scale_pos_weight':9,\n          'eval_metric': 'auc', \n          'nthread':8,\n          'random_state': 99, \n          'silent': True}\n\n#model_xgb = xgb.XGBClassifier(**params)\n\nwatchlist = [(dtrain, 'train')]\n\nmodel_xgb = xgb.train(params, dtrain, 200, watchlist, maximize=True, early_stopping_rounds = 25, verbose_eval=5)\n\nplot_importance(model_xgb)","metadata":{"execution":{"iopub.status.busy":"2021-08-15T11:50:35.880584Z","iopub.execute_input":"2021-08-15T11:50:35.880946Z","iopub.status.idle":"2021-08-15T11:52:02.457017Z","shell.execute_reply.started":"2021-08-15T11:50:35.880908Z","shell.execute_reply":"2021-08-15T11:52:02.455023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#above i have done till 200 you can go higher too (for more better results )\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.DataFrame()\nsub_df['click_id'] = test_df['click_id'].astype('int')\ntest_df.drop(['click_id'], axis=1, inplace=True)\ntest_df = timeconvert(test_df)\n\n#ignore warnings","metadata":{"execution":{"iopub.status.busy":"2021-06-29T06:52:25.098673Z","iopub.execute_input":"2021-06-29T06:52:25.099122Z","iopub.status.idle":"2021-06-29T06:52:41.404347Z","shell.execute_reply.started":"2021-06-29T06:52:25.099081Z","shell.execute_reply":"2021-06-29T06:52:41.403051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtest = xgb.DMatrix(test_df)\ntest_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2021-06-29T06:52:47.839851Z","iopub.execute_input":"2021-06-29T06:52:47.840462Z","iopub.status.idle":"2021-06-29T06:52:52.469277Z","shell.execute_reply.started":"2021-06-29T06:52:47.840404Z","shell.execute_reply":"2021-06-29T06:52:52.468597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df['is_attributed'] = model_xgb.predict(dtest)\nsub_df.to_csv('xgb_sub.csv', float_format='%.8f', index=False) \n","metadata":{"execution":{"iopub.status.busy":"2021-06-29T06:52:56.262447Z","iopub.execute_input":"2021-06-29T06:52:56.263123Z","iopub.status.idle":"2021-06-29T06:56:37.181511Z","shell.execute_reply.started":"2021-06-29T06:52:56.263056Z","shell.execute_reply":"2021-06-29T06:56:37.180303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res = pd.read_csv('xgb_sub.csv')\nres['is_attributed'] = res['is_attributed'].astype('int64') \nres.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-06-29T07:04:03.877271Z","iopub.execute_input":"2021-06-29T07:04:03.877683Z","iopub.status.idle":"2021-06-29T07:04:08.435703Z","shell.execute_reply.started":"2021-06-29T07:04:03.87765Z","shell.execute_reply":"2021-06-29T07:04:08.434674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#I also have removed some analysis codes from these , i taught they would make it complex if you want let me know i will add them up here.","metadata":{},"execution_count":null,"outputs":[]}]}