{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport xgboost as xgb\nfrom xgboost import plot_importance\nfrom sklearn.model_selection import cross_val_score\nimport gc\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":1,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"collapsed":true},"cell_type":"code","source":"\ntrain_columns = ['ip', 'app', 'device', 'os', 'channel', 'click_time', 'is_attributed']\ntest_columns  = ['ip', 'app', 'device', 'os', 'channel', 'click_time', 'click_id']\ndtypes = {\n        'ip'            : 'uint32',\n        'app'           : 'uint16',\n        'device'        : 'uint16',\n        'os'            : 'uint16',\n        'channel'       : 'uint16',\n        'is_attributed' : 'uint8',\n        'click_id'      : 'uint32'\n        }\n\ntrain_df = pd.read_csv(\"../input/train.csv\", skiprows=range(1,123903891), nrows=6100000, usecols=train_columns, dtype=dtypes)\ntest_df = pd.read_csv(\"../input/test.csv\", usecols=test_columns, dtype=dtypes)\n\ntrain_y = train_df['is_attributed']\ntrain_df.drop(['is_attributed'], axis=1, inplace=True)\n#train_df.drop(['attributed_time'], axis=1, inplace=True)\n\n","execution_count":2,"outputs":[]},{"metadata":{"_cell_guid":"aff384d6-7045-4757-a75f-1da072f5efb5","_uuid":"b312d3df6043acc13c46d77bdf7f7b595e720253","trusted":false,"collapsed":true},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"f2990372-2706-475f-9a7b-b97f392105d6","collapsed":true,"_uuid":"09d26fadaa9210612c3d0aaa245284835f4dee56","trusted":true},"cell_type":"code","source":"# Convert datetime feature to : dayofweek + dayofyear + hour of day\n\ndef timeconvert(df):\n    df['datetime'] = pd.to_datetime(df['click_time'])\n    df[\"dayofweek\"] = df[\"datetime\"].dt.dayofweek\n    df[\"dayofyear\"] = df[\"datetime\"].dt.dayofyear\n    df[\"hourofday\"] = df[\"datetime\"].dt.hour\n    df.drop(['click_time', 'datetime'], axis=1, inplace=True)\n    \n    return df\n","execution_count":3,"outputs":[]},{"metadata":{"_cell_guid":"e45177b5-ad52-43d5-99e8-74bbe90fd90a","collapsed":true,"_uuid":"3b2886dfea9135c39eb1edaa4d46330f4270ef5b","trusted":false},"cell_type":"code","source":"# get the nb of click per IP per day\n\ndef nb_click_per_ip_per_day(df):\n    df['dateclick'] = pd.to_datetime(df['click_time']).dt.date\n    return df.groupby(['ip', 'dateclick']).size().reset_index().rename(columns={0:'nbclick'})\n    ","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"06515a72-5176-4c97-af98-2c2272ef340b","_uuid":"20b65618f8f716e376731267ffda503c6369ec26","trusted":true},"cell_type":"code","source":"nrow_train = train_df.shape[0]\nmerge = pd.concat([train_df, test_df])\n\n# Count the average number of clicks per day by ip\ngc.collect()\n\n","execution_count":4,"outputs":[]},{"metadata":{"_cell_guid":"53c05495-72a4-4255-9032-a33826d8936b","collapsed":true,"_uuid":"0f3e51cec0bced1e0c1204bd517d9431c0a91115","trusted":true},"cell_type":"code","source":"merge['dateclick'] = pd.to_datetime(merge['click_time']).dt.date\nip_count1 = merge.groupby(['ip', 'dateclick']).size().reset_index().rename(columns={0:'nbclick'})","execution_count":5,"outputs":[]},{"metadata":{"_cell_guid":"ed681663-3f87-491d-814d-df20098c80a8","_uuid":"da5386824f9b179c14339b1bbefaf22edb0daca0","trusted":true},"cell_type":"code","source":"gc.collect()\nip_count = ip_count1.groupby(['ip']).mean().reset_index()\nip_count.columns = ['ip', 'avg_clicks_per_day_by_ip']\nmerge = pd.merge(merge, ip_count, on='ip', how='left', sort=False)\n#merge.drop('ip', axis=1, inplace=True)\nmerge.drop('dateclick', axis=1, inplace=True)\n\ntrain_df = merge[:nrow_train]\ntest_df = merge[nrow_train:]\n\nmerge.head()","execution_count":6,"outputs":[]},{"metadata":{"_cell_guid":"1bcecea7-6ff8-4fa2-bd70-7f387ba4d225","_uuid":"1ff91177684f3e895ff61904c6094bc5a61fd143","trusted":true},"cell_type":"code","source":"train_df.drop('click_id', axis=1, inplace=True)\ntrain_df.head()","execution_count":7,"outputs":[]},{"metadata":{"_cell_guid":"d6d0d3ec-0a6d-42fd-be87-62237d473bf0","collapsed":true,"_uuid":"a03dce52edcef1dd498a7f4f57963f20ea92364e","trusted":false},"cell_type":"code","source":"\n#train_df2 = nb_click_per_ip_per_day(train_df)\n#train_df2.head()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"a536c1ec-41a4-4f80-bd3d-479364eab00e","collapsed":true,"_uuid":"0498556392e95b594eb366fd4e32c89fde39004c","trusted":false},"cell_type":"code","source":"#train_df2['nbclick'].unique()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"3c4b8d53-c28c-4382-bb46-cdea12edf8d9","collapsed":true,"_uuid":"17bc069d48449652ee03652f58ba6db9bd5a8a32","trusted":false},"cell_type":"code","source":"#train_df2.loc[train_df2['nbclick'] < 10]","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"4474dc23-d2b3-4b80-8d1a-b97150dc5a49","collapsed":true,"_uuid":"2f376121c17e5e97c71f73ada3cbdb9156c2f8a0","trusted":false},"cell_type":"code","source":"#train_all.loc[train_df['ip'] == 364084].groupby(['is_attributed']).size().reset_index().rename(columns={0:'count'})","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"61d8f759-649e-4c3e-8aa2-be67d035c0a6","collapsed":true,"_uuid":"5a78e4e594f9ce72c170f0a227e32c60541be12b","trusted":false},"cell_type":"code","source":"#train_y[31]","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"86bf626e-5c40-41ea-a6b8-47b0a46661a7","_uuid":"14085090bbd7e62fb0c82a6cab55825725177dad","trusted":true},"cell_type":"code","source":"#Prepare the data\n\ntrain_df = timeconvert(train_df)\ntrain_df.head()","execution_count":8,"outputs":[]},{"metadata":{"_cell_guid":"70b434bc-f157-4178-98f8-9f591305e296","_uuid":"54e11131ce8723138cd02b9d776cd4184f3beaaa","trusted":true},"cell_type":"code","source":"dtrain = xgb.DMatrix(train_df, train_y)\n\n# Set the params(this params from Pranav kernel) for xgboost model\nparams = {'eta': 0.3,\n          'tree_method': \"hist\",\n          'grow_policy': \"lossguide\",\n          'max_leaves': 1400,  \n          'max_depth': 0, \n          'subsample': 0.9, \n          'colsample_bytree': 0.7, \n          'colsample_bylevel':0.7,\n          'min_child_weight':0,\n          'alpha':4,\n          'objective': 'binary:logistic', \n          'scale_pos_weight':9,\n          'eval_metric': 'auc', \n          'nthread':8,\n          'random_state': 99, \n          'silent': True}\n\n#model_xgb = xgb.XGBClassifier(**params)\n\nwatchlist = [(dtrain, 'train')]\n\nmodel_xgb = xgb.train(params, dtrain, 200, watchlist, maximize=True, early_stopping_rounds = 25, verbose_eval=5)\n\nplot_importance(model_xgb)\n\n'''\nresults = cross_val_score(model_xgb, train_df, train_y, cv=10)\nprint(\"XGB score: %.4f (%.4f)\" % (results.mean()*100, results.std()*100))\nprint(results)\n'''","execution_count":14,"outputs":[]},{"metadata":{"_cell_guid":"726c5712-9fa2-41e1-9d05-5ba0aeb3c917","scrolled":true,"_uuid":"6d793dd6cbf5ee23cf637ac10fc1bc94b18b64ae","trusted":true},"cell_type":"code","source":"sub_df = pd.DataFrame()\nsub_df['click_id'] = test_df['click_id'].astype('int')\ntest_df.drop(['click_id'], axis=1, inplace=True)\ntest_df = timeconvert(test_df)","execution_count":10,"outputs":[]},{"metadata":{"_cell_guid":"132e5092-2168-42f1-8032-61a5c493f23c","_uuid":"dfce00955cd3ea343e46aeb2cbee44ef61cd8357","trusted":true},"cell_type":"code","source":"dtest = xgb.DMatrix(test_df)\ntest_df.head()","execution_count":12,"outputs":[]},{"metadata":{"_cell_guid":"b18c0933-4a0f-4619-ab15-41b57306985b","_uuid":"fd0bdd02e131652824643c207028ab3e5e9ea20e","trusted":true},"cell_type":"code","source":"# Save the predictions\nsub_df['is_attributed'] = model_xgb.predict(dtest)\nsub_df.to_csv('xgb_sub.csv', float_format='%.8f', index=False)","execution_count":15,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}