{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import roc_auc_score\nfrom collections import Counter\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":95,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"e236728ffbe54c3f7cb327f2ae5c3d8ffe101bbd"},"cell_type":"code","source":"# load in the sample data\ntrain = pd.read_csv(\"../input/train_sample.csv\")","execution_count":96,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1a240798e87a9df435ef6045aa4119e553a0b1ca"},"cell_type":"code","source":"# let's take a look\ntrain.head()","execution_count":97,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a1c59132a605940b287d1bbc8180b9a30dba97bf"},"cell_type":"code","source":"# How imbalanced is the dataset\nprint(Counter(train['is_attributed']))\n","execution_count":98,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"3210d9896b73f4bf937ea39f73a5ed553438eff2"},"cell_type":"code","source":"# Let's split the time data into separate columns\n#train['year'] = pd.DatetimeIndex(train['click_time']).year\n#train['month'] = pd.DatetimeIndex(train['click_time']).month\n#train['day'] = pd.DatetimeIndex(train['click_time']).day\n#train['hour'] = pd.DatetimeIndex(train['click_time']).hour\n#train['minute'] = pd.DatetimeIndex(train['click_time']).minute\n#train['second'] = pd.DatetimeIndex(train['click_time']).second","execution_count":32,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","collapsed":true,"trusted":true},"cell_type":"code","source":"# let's drop the time column, and I think attribuited_time is only there if attributed is true - so drop this too\ntrain = train.drop(['click_time','attributed_time'],1)","execution_count":99,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"86ea5dbbe6df8b7428ed57cca2c24fa791e61f49"},"cell_type":"code","source":"# Let's check it\ntrain.head()","execution_count":100,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"413468396b7e7df43914d7c854bded1250cd560c"},"cell_type":"code","source":"# Year and month are constants, so get rid and split into X and y\nX = train.drop(['is_attributed'],1)\ny = train['is_attributed']","execution_count":101,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"1ba5aa43de49dd727de6502da8ac51b5fc1eef48"},"cell_type":"code","source":"# Create a test train split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.33, random_state=42)","execution_count":102,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"24ec8b7f4e569eeb44f16111616ed82a0e393294"},"cell_type":"code","source":"# Let's try random forests first\nclf = RandomForestClassifier(n_estimators=500 , max_depth=20)","execution_count":103,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1de3391df660c30af34fc77fb71e298ca22b139f"},"cell_type":"code","source":"clf.fit(X_train,y_train)","execution_count":104,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5a3fb4ee84eb01f110911314066c95f3b9537f8a"},"cell_type":"code","source":"preds = clf.predict(X_test)\nroc_auc_score(y_test, preds)","execution_count":105,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"a01e969e564d00925618b9967318093af620bba1"},"cell_type":"code","source":"# OK, let's try XGBoost!\nimport xgboost as xgb","execution_count":106,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f116296cfeb255eac532554f4711cdf2eb78793a"},"cell_type":"code","source":"clf = xgb.XGBClassifier(n_estimators=500,max_depth=20,learning_rate=0.01)\nclf.fit(X_train,y_train)","execution_count":107,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3bf56a25b75782bba81d2ee20e9e94ec45e097fc"},"cell_type":"code","source":"preds = clf.predict(X_test)\nroc_auc_score(y_test, preds)","execution_count":108,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"eb61aaaecbae20c8f67732a4cc3ca20285a1719c"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}