{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true,"_kg_hide-input":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":2,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1a308f6e3a5b0db3fc14a8dafe58b0619870d0c0"},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\ndata = pd.read_csv('../input/train.csv',nrows=1 * (10**7))\ntest = pd.read_csv('../input/test.csv')\ndata['click_time'] = pd.to_datetime(data['click_time'])\ndata['hour'] = data['click_time'].apply(lambda x: x.hour)\ndata['weekday'] = data['click_time'].apply(lambda x: x.weekday())\ndata['day'] = data['click_time'].apply(lambda x: x.day)\ntest['click_time'] = pd.to_datetime(test['click_time'])\ntest['hour'] = test['click_time'].apply(lambda x: x.hour)\ntest['weekday'] = test['click_time'].apply(lambda x: x.weekday())\ntest['day'] = test['click_time'].apply(lambda x: x.day)\n\nimport xgboost as xgb\nfrom sklearn.model_selection import train_test_split\npars = {'eta' : 0.3,\n       'max_depth' : 6,\n       'objective' : 'binary:logistic',\n       'eval_metric' : 'auc',\n       'random_state' : 42}\n\nx_train, x_test, y_train, y_test = train_test_split(data.drop(['attributed_time','is_attributed','click_time'],axis=1),\n                                                    data['is_attributed'],test_size=0.3,random_state=42)\nls = [(xgb.DMatrix(x_train,y_train), 'train'), (xgb.DMatrix(x_test,y_test),'valid')]\nreport = {}\nclf = xgb.train(pars, xgb.DMatrix(x_train,y_train),250,ls,maximize=True,evals_result=report)\n\npred = clf.predict(xgb.DMatrix(test.drop(['click_id','click_time'],axis=1)),ntree_limit=clf.best_ntree_limit)\nsub = pd.DataFrame({'click_id' : test.click_id.values, 'is_attributed': pred})\nsub.to_csv('submission.csv',index=False)","execution_count":1,"outputs":[]},{"metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"trusted":true,"_uuid":"311d37bc550cd96e187b7a8de80f06c7378a7371"},"cell_type":"code","source":"","execution_count":25,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"774912091c0d7526f8b55a1fe40a866e7e992249"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}