{"cells":[{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":12,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","collapsed":true,"trusted":true},"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nimport gc","execution_count":13,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ce37e39d292f3d36730b77550875deb821d832cc"},"cell_type":"code","source":"!wc -l ../input/train.csv","execution_count":14,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"571cecc5e2fffb41ceddcd4b46fdd3c034a2f47e"},"cell_type":"code","source":"!wc -l ../input/test.csv","execution_count":15,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"58feaac179f2b8d0b088e5778cbd39e69ca1425d"},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\", nrows = 10000000)\n# train = pd.read_csv(\"../input/train_sample.csv\")\n\ntrain.head()","execution_count":16,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c6ad4312e76b5aa2b2795feca0e6e804dabbbf42"},"cell_type":"code","source":"test = pd.read_csv(\"../input/test.csv\")\ntest.head()","execution_count":17,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a593ef4c832853e7cffa4c613c0c64bf2226981b"},"cell_type":"code","source":"sub = pd.read_csv(\"../input/sample_submission.csv\")\nsub.head()","execution_count":18,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b82ddcfae1ed29ff568be9b65e4b59a218c4283"},"cell_type":"code","source":"y_train = train['is_attributed']\n# x_train = train[['ip', 'app', 'device', 'os', 'channel', 'click_time']]\n# x_test  = test[['ip', 'app', 'device', 'os', 'channel', 'click_time']]\nx_train = train[['ip', 'app', 'device', 'os', 'channel']]\nx_test  = test[['ip', 'app', 'device', 'os', 'channel']]\n\ndel train\ndel test\ngc.collect()","execution_count":19,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"c8f607e049ef8bb36a0c4e6e803f3c75fbae44c2"},"cell_type":"code","source":"clf = RandomForestClassifier(max_depth=6, random_state=0)\nclf.fit(x_train, y_train)\n# y_pred = clf.predict(x_test)\ny_pred = clf.predict_proba(x_test)","execution_count":20,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b07dc8336511ca45409064f75b1ee0d867fb8fcd"},"cell_type":"code","source":"sub['is_attributed'] = y_pred[:,1]\nsub.head()","execution_count":23,"outputs":[]},{"metadata":{"collapsed":true,"trusted":true,"_uuid":"1d5d6540c260a47d603bcaac7ae6faa4129e3933"},"cell_type":"code","source":"sub.to_csv('sub_rf.csv', index=False)","execution_count":22,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}