{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","collapsed":true,"trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in\n#thanks Joao for your great notebook!\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport time\nimport gc\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"17ba2466-c31d-4d2c-bbc0-94861b2d15c1","_uuid":"9f4548af7695dd39e2cfad4024e7961c383c5922","collapsed":true,"trusted":true},"cell_type":"code","source":"is_valid = False","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","collapsed":true,"trusted":true},"cell_type":"code","source":"start_time = time.time()\ntrain_columns = ['ip', 'app', 'device', 'os', 'channel', 'click_time', 'is_attributed']\ntest_columns  = ['ip', 'app', 'device', 'os', 'channel', 'click_time', 'click_id']\ndtypes = {\n        'ip'            : 'uint32',\n        'app'           : 'uint16',\n        'device'        : 'uint16',\n        'os'            : 'uint16',\n        'channel'       : 'uint16',\n        'is_attributed' : 'uint8',\n        'click_id'      : 'uint32'\n        }\ntrain=pd.read_csv(\"../input/train.csv\", skiprows=range(1,129903891), nrows=61000000, usecols=train_columns, dtype=dtypes)\ntest=pd.read_csv(\"../input/test_supplement.csv\",usecols=test_columns, dtype=dtypes)\nprint('loading of data is completed in [{}] seconds'.format(time.time() - start_time))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"989e9971-a6a3-4d0a-8b7d-feb0bf44dc62","_uuid":"8f5feac9f5b53c168757f067ae4e2b7e51a05629","collapsed":true,"trusted":true},"cell_type":"code","source":"def datatimeFeatures(df):\n    df['datetime'] = pd.to_datetime(df['click_time'])\n    df['dow']      = df['datetime'].dt.dayofweek\n    df['doy']      = df['datetime'].dt.dayofyear\n    df.drop(['click_time', 'datetime'], axis=1, inplace=True)\n    return df","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"2baabe00-1146-49cb-ba43-6c8baa86f809","_uuid":"85e2a10bc6b9feff15b5e18dfcd9c82e08c53b89","collapsed":true,"trusted":true},"cell_type":"code","source":"#select the target variable\ny = train['is_attributed']\n\ntrain.drop(['is_attributed'], axis=1, inplace=True)\nsub = pd.DataFrame()\ntest.drop('click_id', axis=1, inplace=True)\ngc.collect()\n\n# Some feature engineering\nnrow_train = train.shape[0]\nmerge = pd.concat([train, test])\ndel train, test\ngc.collect()\n\n# Count the number of clicks by ip and app\nip_count = merge.groupby(['ip'])['channel'].count().reset_index()\nip_count.columns = ['ip', 'clicks_by_ip']\nmerge = pd.merge(merge, ip_count, on='ip', how='left', sort=False)\nmerge['clicks_by_ip'] = merge['clicks_by_ip'].astype('uint16')\nmerge.drop('ip', axis=1, inplace=True)\n\ntrain = merge[:nrow_train]\ntest = merge[nrow_train:]\ndel test, merge\ngc.collect()\n\n# Make new feature with datatimeFeatures function\ntrain = datatimeFeatures(train)\ngc.collect()\n\nprint('preprocessing is completed in [{}] seconds'.format(time.time() - start_time))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"9c0c688e-1e84-4399-8b89-c5c5a9173007","_uuid":"75435ef9b5102a825598c4cc3db3096c7a6ee890","collapsed":true,"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport xgboost as xgb\nparams = {'eta': 0.3,\n          'tree_method': \"hist\",\n          'grow_policy': \"lossguide\",\n          'max_leaves': 1400,  \n          'max_depth': 0, \n          'subsample': 0.9, \n          'colsample_bytree': 0.7, \n          'colsample_bylevel':0.7,\n          'min_child_weight':0,\n          'alpha':4,\n          'objective': 'binary:logistic', \n          'scale_pos_weight':9,\n          'eval_metric': 'auc', \n          'nthread':8,\n          'random_state': 99, \n          'silent': True}","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"bb9cfd0f-e742-441c-947f-0d7dd6284c1e","_uuid":"c7a2621ab34a658600e836e53b851293e966f8f8","collapsed":true,"trusted":true},"cell_type":"code","source":"if (is_valid == True):\n    # Get 10% of train dataset to use as validation\n    x1, x2, y1, y2 = train_test_split(train, y, test_size=0.1, random_state=99)\n    dtrain = xgb.DMatrix(x1, y1)\n    dvalid = xgb.DMatrix(x2, y2)\n    del x1, y2, x2, y2 \n    gc.collect()\n    watchlist = [(dtrain, 'train'), (dvalid, 'valid')]\n    model = xgb.train(params, dtrain, 200, watchlist, maximize=True, early_stopping_rounds = 25, verbose_eval=5)\n    del dvalid\nelse:\n    dtrain = xgb.DMatrix(train, y)\n    del train, y\n    gc.collect()\n    watchlist = [(dtrain, 'train')]\n    model = xgb.train(params, dtrain, 30, watchlist, maximize=True, verbose_eval=1)\n\ndel dtrain\ngc.collect()\nprint('XGBoost Training is finished in [{}] seconds'.format(time.time() - start_time))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"90161fc5-e886-4008-878f-331ba7235bb0","_uuid":"562302172a26f31fd287ae5bba0b6111990dd222","collapsed":true,"trusted":true},"cell_type":"code","source":"# Load the test for predict \ntest = pd.read_csv(\"../input/test.csv\", usecols=test_columns, dtype=dtypes)\ntest = pd.merge(test, ip_count, on='ip', how='left', sort=False)\ndel ip_count\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"f44c482fff986224ee48d31c1dbd067b309361a5"},"cell_type":"code","source":"sub['click_id'] = test['click_id'].astype('int')","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"db6bc2b9-bf70-495a-8d7e-c2c5aa1e4cc0","_uuid":"3af9efb73c680d0df6281bb936d78cb97d76e038","collapsed":true,"trusted":true},"cell_type":"code","source":"test['clicks_by_ip'] = test['clicks_by_ip'].astype('uint16')\ntest = datatimeFeatures(test)\ntest.drop(['click_id', 'ip'], axis=1, inplace=True)\ndtest = xgb.DMatrix(test)\ndel test\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"f506c097-a28f-4bb7-92f4-cd0592b37557","_uuid":"73d3049a4459d5c41295b9d953b21d09230bd1fe","collapsed":true,"trusted":true},"cell_type":"code","source":"# Save the predictions\nsub['is_attributed'] = model.predict(dtest, ntree_limit=model.best_ntree_limit)\nsub.to_csv('improved_xgb_sub_today.csv',float_format='%.8f',index=False)\nprint('submission is done in [{}] seconds'.format(time.time() - start_time))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"6c56b4dc-158d-4867-96a1-281326aabbe2","_uuid":"dadc32c782ead3afb871e2b3f0c81fc09150d9ac","collapsed":true,"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}