{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":1,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import time\nimport xgboost as xgb\nfrom sklearn.cross_validation import train_test_split","execution_count":2,"outputs":[]},{"metadata":{"_cell_guid":"2697c45d-bbf4-442e-b4c2-69e7ac2b7211","_uuid":"da12df57bf597550b18a43deb78d0b54230f7fe5","collapsed":true,"trusted":true},"cell_type":"code","source":"start_time=time.time()\ncolumns=['ip','app','device','os','channel','click_time','is_attributed']\ndtypes={\n        'ip'            : 'uint32',\n        'app'           : 'uint16',\n        'device'        : 'uint16',\n        'os'            : 'uint16',\n        'channel'       : 'uint16',\n        'is_attributed' : 'uint8',\n}","execution_count":3,"outputs":[]},{"metadata":{"_cell_guid":"6398a1ca-a1bd-4cde-8e4d-9199782252c3","_uuid":"906b3775a12edb0ea18906b687c2c0723bcc10c4","trusted":true},"cell_type":"code","source":"train=pd.read_csv('../input/train.csv',skiprows=range(1,149903891),nrows=35000000,usecols=columns,dtype=dtypes)\ntest=pd.read_csv('../input/test.csv')\nprint('[{}] Finished to load data'.format(time.time() - start_time))","execution_count":4,"outputs":[]},{"metadata":{"_cell_guid":"92c37b8a-7285-41fa-a7db-152cab6a4b0a","_uuid":"eaa97424b7c7bdee465409ba4f95616edc0de96e","collapsed":true,"trusted":true},"cell_type":"code","source":"sub = pd.DataFrame()\nsub['click_id'] = test['click_id']","execution_count":5,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"289e67bc4f88d480344695b4988342f63625c91b"},"cell_type":"code","source":"def dataPreProcessTime(df):\n    # Transform click_time in two columns(one with date and another with time)\n    df['date_click'] = pd.to_datetime(df['click_time']).dt.date\n    df['date_click'] = df['date_click'].apply(lambda x: x.strftime('%Y%m%d')).astype(int)\n    \n    df['time_click'] = pd.to_datetime(df['click_time']).dt.time\n    df['time_click'] = df['time_click'].apply(lambda x: x.strftime('%H%M%S')).astype(int)   \n    \n    df.drop('click_time', axis=1, inplace=True)\n    return df\n","execution_count":6,"outputs":[]},{"metadata":{"_cell_guid":"9e9d290e-fdd0-494a-8495-2b104ece03f4","_uuid":"1303a59f15b6c8c2dc0a4e8f3dbc5cce24b9c8a0","trusted":true},"cell_type":"code","source":"#数据的统计信息\nprint(train['is_attributed'].value_counts())\nprint(train[train['is_attributed']==1]['is_attributed'].sum()/len(train))","execution_count":7,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"2bb8727c9929c55bb1d98f6a3d13edbce7427e8d"},"cell_type":"code","source":"train = dataPreProcessTime(train)\ntest = dataPreProcessTime(test)","execution_count":8,"outputs":[]},{"metadata":{"_cell_guid":"05f9cc4e-d474-4e4e-be98-7a5d4f002d50","_uuid":"b7fc9c0aadecfd7c18fc94212795b880e10fcb51","collapsed":true,"trusted":true},"cell_type":"code","source":"y=train['is_attributed']\n#'click_time','is_attributed','attributed_timed'\ntrain.drop(['is_attributed'],axis=1,inplace=True)#inplace=True代表更改原内存的值\n#'click_id','click_time'\ntest.drop(['click_id'],axis=1,inplace=True)","execution_count":9,"outputs":[]},{"metadata":{"_cell_guid":"dd67cb4e-65ab-4926-be5d-2f4fdb3eee34","_uuid":"4513ee77ea009713d058a35ac9c07ce7bd0fbee8","collapsed":true,"trusted":true},"cell_type":"code","source":"# Some feature engineering\nnrow_train = train.shape[0]\nmerge = pd.concat([train, test])","execution_count":10,"outputs":[]},{"metadata":{"_cell_guid":"043babc4-b2a7-4dde-8efc-943466824288","_uuid":"e30fa106b27bcc2b37ba518ab8654ead9ae571f7","collapsed":true,"trusted":true},"cell_type":"code","source":"# Count the number of clicks by ip\nip_count = merge.groupby('ip')['app'].count().reset_index()\nip_count.columns = ['ip', 'clicks_by_ip']\nip_count.tail()\nmerge = pd.merge(merge, ip_count, on='ip', how='left', sort=False)\nmerge.drop('ip', axis=1, inplace=True)","execution_count":11,"outputs":[]},{"metadata":{"_cell_guid":"237f5bce-f4fd-46c0-a378-0d82ceb7bd50","_uuid":"2b62449fb0d47669be912c59e60a133180cdb9fe","trusted":true},"cell_type":"code","source":"train = merge[:nrow_train]\ntest = merge[nrow_train:]\ntest.head()","execution_count":12,"outputs":[]},{"metadata":{"_cell_guid":"27248fed-d426-4e91-835d-91efea3c8a99","_uuid":"c3cfbb0eaef06fc49be370daf5b4f7095a2e14b8","collapsed":true,"trusted":true},"cell_type":"code","source":"# Set the params(this params from Pranav kernel) for xgboost model\nparams = {'eta': 0.6,\n          'tree_method': \"hist\",\n          'grow_policy': \"lossguide\",\n          'max_leaves': 1400,  \n          'max_depth': 0, \n          'subsample': 0.9, \n          'colsample_bytree': 0.7, \n          'colsample_bylevel':0.7,\n          'min_child_weight':0,\n          'alpha':4,\n          'objective': 'binary:logistic', \n          'scale_pos_weight':9,\n          'eval_metric': 'auc', \n          'nthread':8,\n          'random_state': 99, \n          'silent': True}","execution_count":13,"outputs":[]},{"metadata":{"_cell_guid":"8a5bde47-e917-451b-8797-d4027195da6c","_uuid":"8f2b2cf330cd3c5b9e197dd47213982c25822347","trusted":true},"cell_type":"code","source":"watchlist = [(xgb.DMatrix(train, y), 'train')]\nmodel = xgb.train(params, xgb.DMatrix(train, y), 15, watchlist, maximize=True, verbose_eval=1)","execution_count":14,"outputs":[]},{"metadata":{"_cell_guid":"d9b9d9e5-b8c7-4f53-af10-792038140e74","_uuid":"dba8dc6a70fb4dbf1dac6e3e6a155026e59102f1","trusted":true},"cell_type":"code","source":"print('[{}] Finish XGBoost Training'.format(time.time() - start_time))","execution_count":15,"outputs":[]},{"metadata":{"_cell_guid":"6e4bb6c6-89fd-40e7-a814-c200350d9869","_uuid":"44468991d6b617081ca81ce52dfd2044afdf0434","collapsed":true,"trusted":true},"cell_type":"code","source":"sub['is_attributed'] = model.predict(xgb.DMatrix(test), ntree_limit=model.best_ntree_limit)\nsub.to_csv('xgb_sub.csv',index=False)","execution_count":16,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e05da8076ee797dcf9ea3a76b13cb0101878e462"},"cell_type":"code","source":"sub['is_attributed'].head()","execution_count":17,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.4"}},"nbformat":4,"nbformat_minor":1}