{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","collapsed":true,"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":false},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score\n\nimport lightgbm as lgb\nimport xgboost as xgb\n\nimport gc\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","collapsed":true,"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":false},"cell_type":"code","source":"path = '../input/' \npath_train = path + 'train.csv'\npath_test = path + 'test.csv'\n\n#nsamples = 100000\nunbalance_fact = 10\n\ntrain_cols = ['ip', 'app', 'device', 'os', 'channel', 'click_time', 'is_attributed']\ntest_cols = ['ip', 'app', 'device', 'os', 'channel', 'click_time' ]\n\ndtypes = {\n        'ip'            : 'uint64',\n        'app'           : 'uint16',\n        'device'        : 'uint16',\n        'os'            : 'uint16',\n        'channel'       : 'uint16',\n        'is_attributed' : 'uint8',\n        'click_id'      : 'uint64'\n        }","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e8f0c0391e3c70a1fdef37e009d8fce3a95d3153","_cell_guid":"5f0a6e37-388b-4b2b-a1be-2f48dbb1a925","trusted":false,"collapsed":true},"cell_type":"code","source":"print(\"Loading Data\")\ntrain = pd.read_csv(path_train, usecols=train_cols, dtype=dtypes, parse_dates=[\"click_time\"])\nprint(\"Loading is done\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d410c946c0c9c8c299a200edf64cf7aef5cf7bda","_cell_guid":"9f22b9aa-2e03-472c-9f1b-3e926cef3460","trusted":false,"collapsed":true},"cell_type":"code","source":"print(len(train))\ntrain_pos = train[train[\"is_attributed\"] == 1]\nnum_attr = len(train_pos)\nprint(\"Number of attributed: {:d}\".format(num_attr))\ntrain_neg = train[train[\"is_attributed\"] == 0].sample(num_attr )\ntrain = pd.concat([train_pos, train_neg])\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9591201d62d1bfbffe46501818a99a792836cdbe","collapsed":true,"_cell_guid":"c54ff418-dcd3-4934-be83-3fc8237e0e25","trusted":false},"cell_type":"code","source":"def prepare_dataframe(df):\n    print(\"Display frames:\")\n    display(df.head())\n    display(df.dtypes)\n    display(df.shape)\n    \n    print(\"Creating new time features:\")\n    df['hour'] = df[\"click_time\"].dt.hour.astype('uint8')\n    df['day'] = df[\"click_time\"].dt.day.astype('uint8')\n    df[\"minute\"] = df[\"click_time\"].dt.minute.astype('uint8')\n    \n    print(\"## Get counts per cat.\")\n    n_chans = df[['ip','day','hour','channel']].groupby(by=['ip','day',\n          'hour'])[['channel']].count().reset_index().rename(columns={'channel': 'ip_day_hour'})\n    df = df.merge(n_chans, on=['ip','day','hour'], how='left')\n    del n_chans\n    gc.collect()\n    \n    n_chans = df[['ip','app', 'channel']].groupby(by=['ip', \n          'app'])[['channel']].count().reset_index().rename(columns={'channel': 'ip_app_count'})\n    df = df.merge(n_chans, on=['ip','app'], how='left')\n    del n_chans\n    gc.collect()\n    \n    n_chans = df[['ip','app', 'os', 'channel']].groupby(by=['ip', 'app', \n          'os'])[['channel']].count().reset_index().rename(columns={'channel': 'ip_app_os_count'})\n    df = df.merge(n_chans, on=['ip','app', 'os'], how='left')\n    del n_chans\n    gc.collect()\n    \n    n_chans = df[['ip','channel']].groupby(by=['ip'])[['channel']].count().reset_index().rename(columns={'channel': 'count_by_ip'})\n    print('Merging the channels data with the main data set...')\n    df = df.merge(n_chans, on=['ip'], how='left')\n\n    # Count by IP HOUR CHANNEL\n    n_chans = df[['ip','hour','channel','os']].groupby(by=['ip','hour','channel'\n               ])[['os']].count().reset_index().rename(columns={'os': 'ip_hour_channel'})\n    df = df.merge(n_chans, on=['ip','hour','channel'], how='left')\n    del n_chans\n    gc.collect()\n\n    # Count by IP HOUR Device\n    n_chans = df[['ip','hour','channel','os']].groupby(by=['ip','hour','os'\n               ])[['channel']].count().reset_index().rename(columns={'channel': 'ip_hour_os'})\n    df = df.merge(n_chans, on=['ip','hour','os'], how='left')\n    del n_chans\n    gc.collect()\n\n    n_chans = df[['ip','hour','channel','app']].groupby(by=['ip','hour','app'\n               ])[['channel']].count().reset_index().rename(columns={'channel': 'ip_hour_app'})\n    df = df.merge(n_chans, on=['ip','hour','app'], how='left')\n    del n_chans\n    gc.collect()\n\n    n_chans = df[['ip','hour','channel','device']].groupby(by=['ip','hour','device'\n               ])[['channel']].count().reset_index().rename(columns={'channel': 'ip_hour_device'})\n    df = df.merge(n_chans, on=['ip','hour','device'], how='left')\n    del n_chans\n    gc.collect()\n    \n    print(\"Adjusting the data types of the new count features... \")\n    df.info()\n    df['ip_day_hour'] = df['ip_day_hour'].astype('uint8')\n    df['ip_app_count'] = df['ip_app_count'].astype('uint8')\n    df['ip_app_os_count'] = df['ip_app_os_count'].astype('uint8')\n\n    # Added..\n    df['count_by_ip'] = df['count_by_ip'].astype('uint16')\n    df['ip_hour_channel'] = df['ip_hour_channel'].astype('uint16')\n    df['ip_hour_os'] = df['ip_hour_os'].astype('uint16')\n    df['ip_hour_app'] = df['ip_hour_app'].astype('uint16')\n    df['ip_hour_device'] = df['ip_hour_device'].astype('uint16')\n    \n    return df\n    ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2a27cea98622c41f13735fc57958bd588f7ff107","_cell_guid":"c4b44dd7-8204-491d-8e19-a502b0c42bc0","trusted":false,"collapsed":true},"cell_type":"code","source":"train = prepare_dataframe(train)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ebe622bbde6543414f70697545d5b221f32dd748","collapsed":true,"_cell_guid":"6c0cf6c1-bc24-48b4-ab13-0bd43153d276","trusted":false},"cell_type":"code","source":"train_df, test_df = train_test_split(train, test_size=0.2, stratify=train[\"is_attributed\"])\ntrain_df, valid_df = train_test_split(train_df, test_size=0.2, stratify=train_df[\"is_attributed\"])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a7c35f112300277f01b2bdb2bdad6c61e188c6b2","_cell_guid":"25e464e0-6155-45b6-a282-cb8ed9401ee5","trusted":false,"collapsed":true},"cell_type":"code","source":"print(len(train_df))\nprint(len(valid_df))\nprint(len(test_df))\ndel train\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cb8ee4b95b3a9ab928efb5c27566befe40ec02d6","_cell_guid":"895c05a7-5133-466e-bd4e-f4ecb8829199","trusted":false,"collapsed":true},"cell_type":"code","source":"print(len(train_df))\nprint(len(valid_df))\nprint(len(test_df))\n\nprint(train_df[\"is_attributed\"].sum())\nprint(valid_df[\"is_attributed\"].sum())\nprint(test_df[\"is_attributed\"].sum())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b535f3ed364c6748c1a1cb2701f97825a347b19b","_cell_guid":"68f2da8a-6cb5-48fe-bc39-5c24314ca678","trusted":false,"collapsed":true},"cell_type":"code","source":"target = 'is_attributed'\n\npredictors = ['ip', 'device', 'app', 'os', 'channel', 'hour', \"minute\", # Starter Vars, Then new features below\n              'ip_day_hour','count_by_ip','ip_app_count', 'ip_app_os_count',\n              \"ip_hour_channel\", \"ip_hour_os\", \"ip_hour_app\",\"ip_hour_device\"]\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8d6e007d4d0e697235281f3546cb943d6e5b0bd5","_cell_guid":"930950e0-cfc2-45de-a083-56befeb55888","trusted":false,"collapsed":true},"cell_type":"code","source":"# params = {'eta': 0.3,\n#           'tree_method': \"hist\",\n#           'grow_policy': \"lossguide\",\n#           'max_leaves': 1400,  \n#           'max_depth': 0, \n#           'subsample': 0.9, \n#           'colsample_bytree': 0.7, \n#           'colsample_bylevel':0.7,\n#           'min_child_weight':0,\n#           'alpha':4,\n#           'objective': 'binary:logistic', \n#           'scale_pos_weight':9,\n#           'eval_metric': 'auc', \n#           'nthread':8,\n#           'random_state': 99, \n#           'silent': True}\n\nparams = {'eta': 0.1,\n          'objective': 'binary:logistic', \n          'scale_pos_weight':1.0 / unbalance_fact,\n          'eval_metric': 'auc', \n          'nthread':8,\n          'silent': True}\n\nprint(train_df[target].nunique())\nprint(pd.unique(train_df[target]))\ndtrain = xgb.DMatrix(train_df[predictors], train_df[target])\ndvalid = xgb.DMatrix(valid_df[predictors], valid_df[target])\n#del train_df, valid_df \ngc.collect()\nwatchlist = [(dtrain, 'train'), (dvalid, 'valid')]\nmodel = xgb.train(params, dtrain, 500, watchlist, early_stopping_rounds = 50, verbose_eval=5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ef5b4740c05a8d9889291bb8b944b7e815f93d85","_cell_guid":"1094170c-5778-48e4-bde0-28e3a5d16177","trusted":false,"collapsed":true},"cell_type":"code","source":"# Nick's Feature Importance Plot\nf, ax = plt.subplots(figsize=[7,10])\nxgb.plot_importance(model, ax=ax, max_num_features=len(predictors))\nplt.title(\"XGboost Feature Importance\")\nplt.savefig('feature_import.png')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e2b0b7937a89ae95178d96756d35fe89b12760c4","_cell_guid":"eaba3104-9628-47d0-8f87-b9dca4ac5229","trusted":false,"collapsed":true},"cell_type":"code","source":"## Testing accuracy on test split.\nprint(\"Testing against test split\")\ndtest = xgb.DMatrix(test_df[predictors])\ntest_res = model.predict(dtest, ntree_limit=model.best_ntree_limit)\ntest_score = roc_auc_score(test_df[target].values, test_res)\nprint(\"Test score = {:f}\".format(test_score))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"60fe2a34e15da1723c7aa4e1720f3e32e413c274","_cell_guid":"fbe78732-6940-4f36-8d73-ceab694127b4","trusted":false,"collapsed":true},"cell_type":"code","source":"print(\"Loading test data for generating submission.\")\ntest = pd.read_csv(path_test, dtype=dtypes, parse_dates=[\"click_time\"])\ntest = prepare_dataframe(test)\n\nprint(\"Preparing data for submission...\")\ndtest = xgb.DMatrix(test[predictors])\ntest['is_attributed'] = model.predict(dtest, ntree_limit=model.best_ntree_limit)\n\nprint(\"Writing the submission data into a csv file...\")\ntest[[\"click_id\",\"is_attributed\"]].to_csv(\"submission_xgb_v2.csv\",index=False)\nprint(\"All Done...\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"428509f27806331a045a253f89c45a7af1e2eb09","collapsed":true,"_cell_guid":"c759eeb1-df36-4d90-9553-7c596ad3225c","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":1}