{"cells":[{"metadata":{"trusted":true,"_uuid":"ea9350ad74d12e6bfd1791278a8558391c192202"},"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn import metrics","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"23f676852366d3dd8076c898de01311825d5193f"},"cell_type":"code","source":"X = train.iloc[:,1:]\ny = train.iloc[:,0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"70f1208a2bc9ac84d5f7f34ad6b61c02af2ad536"},"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, random_state=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c2cc934aa3470110280330cd1157e25e278b9c80"},"cell_type":"code","source":"model = RandomForestClassifier(n_estimators=1000)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"468edd5a4d263caa66c400936c7a1e7666151ff2"},"cell_type":"code","source":"model.fit(X_train, y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dbd501aabbae0a7cc0553b738a2f8b8355ac8e94"},"cell_type":"code","source":"y_pred = model.predict(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"64562c504fa6ab9b28b1e47c7e6d59e1b545ddc3"},"cell_type":"code","source":"print(metrics.classification_report(y_pred, y_test))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1ba7767f4af05795c496e1fdda6c5046fd62bc9f"},"cell_type":"code","source":"y_pred_test = model.predict(test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fdb6d29cecf6a862314e5cbfd3ef02f0478e7030"},"cell_type":"code","source":"submission = pd.DataFrame({'ImageId': [i + 1 for i in range(len(y_pred_test))], 'Label': y_pred_test})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7e7028cb6e72d8158dd81135a0b99b63c57044a3"},"cell_type":"code","source":"submission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"69c0e58c2c5be6ecb1f3d9998208b76621643eef"},"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"91d425c4d71ba964de114bcbb396ad18dbcf1e93"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}