{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\nimport matplotlib.pyplot as plt\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-26T02:00:00.248662Z","iopub.execute_input":"2022-07-26T02:00:00.249062Z","iopub.status.idle":"2022-07-26T02:00:00.260566Z","shell.execute_reply.started":"2022-07-26T02:00:00.249029Z","shell.execute_reply":"2022-07-26T02:00:00.259346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/digit-recognizer/train.csv\")\ntest_df = pd.read_csv(\"../input/digit-recognizer/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-26T02:00:01.286166Z","iopub.execute_input":"2022-07-26T02:00:01.287108Z","iopub.status.idle":"2022-07-26T02:00:07.051307Z","shell.execute_reply.started":"2022-07-26T02:00:01.287061Z","shell.execute_reply":"2022-07-26T02:00:07.050333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T02:00:08.242245Z","iopub.execute_input":"2022-07-26T02:00:08.242625Z","iopub.status.idle":"2022-07-26T02:00:08.270795Z","shell.execute_reply.started":"2022-07-26T02:00:08.242590Z","shell.execute_reply":"2022-07-26T02:00:08.269713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T02:00:12.177149Z","iopub.execute_input":"2022-07-26T02:00:12.177509Z","iopub.status.idle":"2022-07-26T02:00:12.195314Z","shell.execute_reply.started":"2022-07-26T02:00:12.177479Z","shell.execute_reply":"2022-07-26T02:00:12.194507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ref. https://www.kaggle.com/code/prashant111/random-forest-classifier-tutorial","metadata":{"execution":{"iopub.status.busy":"2022-07-26T02:00:13.771589Z","iopub.execute_input":"2022-07-26T02:00:13.772622Z","iopub.status.idle":"2022-07-26T02:00:13.777100Z","shell.execute_reply.started":"2022-07-26T02:00:13.772571Z","shell.execute_reply":"2022-07-26T02:00:13.776173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_df.iloc[:, 1:].values\ny = train_df[\"label\"].values","metadata":{"execution":{"iopub.status.busy":"2022-07-26T02:00:24.643943Z","iopub.execute_input":"2022-07-26T02:00:24.644362Z","iopub.status.idle":"2022-07-26T02:00:24.653779Z","shell.execute_reply.started":"2022-07-26T02:00:24.644327Z","shell.execute_reply":"2022-07-26T02:00:24.652534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split data into training and testing sets\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.33, random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T02:00:40.706752Z","iopub.execute_input":"2022-07-26T02:00:40.707135Z","iopub.status.idle":"2022-07-26T02:00:41.674889Z","shell.execute_reply.started":"2022-07-26T02:00:40.707101Z","shell.execute_reply":"2022-07-26T02:00:41.673826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import Random Forest classifier\nfrom sklearn.ensemble import RandomForestClassifier\n\n# instantiate the classifier \nrfc = RandomForestClassifier(random_state=0)\n\n# fit the model\nrfc.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T02:00:43.407111Z","iopub.execute_input":"2022-07-26T02:00:43.407529Z","iopub.status.idle":"2022-07-26T02:01:00.106891Z","shell.execute_reply.started":"2022-07-26T02:00:43.407492Z","shell.execute_reply":"2022-07-26T02:01:00.105718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict the Test set results\ny_pred = rfc.predict(X_test)\n\n\n# Check accuracy score \nfrom sklearn.metrics import accuracy_score\nprint('Model accuracy score with 100 decision-trees : {0:0.4f}'. format(accuracy_score(y_test, y_pred)))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T02:01:00.108408Z","iopub.execute_input":"2022-07-26T02:01:00.108963Z","iopub.status.idle":"2022-07-26T02:01:00.689049Z","shell.execute_reply.started":"2022-07-26T02:01:00.108929Z","shell.execute_reply":"2022-07-26T02:01:00.687752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_nodes = rfc.apply(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T02:02:25.624229Z","iopub.execute_input":"2022-07-26T02:02:25.625023Z","iopub.status.idle":"2022-07-26T02:02:26.584472Z","shell.execute_reply.started":"2022-07-26T02:02:25.624983Z","shell.execute_reply":"2022-07-26T02:02:26.583391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_id = 2\nplt.imshow(X_test[sample_id].reshape((28, 28)))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T02:13:25.774433Z","iopub.execute_input":"2022-07-26T02:13:25.774841Z","iopub.status.idle":"2022-07-26T02:13:25.950687Z","shell.execute_reply.started":"2022-07-26T02:13:25.774808Z","shell.execute_reply":"2022-07-26T02:13:25.949442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_nodes = rfc.apply(X_test[sample_id].reshape((1, 784)))[0]\nfriends = np.zeros(X_train.shape[0], dtype=int)\nfor i, p in enumerate(sample_nodes):\n    friends += (train_nodes[:, i] == p)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T02:13:27.236432Z","iopub.execute_input":"2022-07-26T02:13:27.236815Z","iopub.status.idle":"2022-07-26T02:13:27.259146Z","shell.execute_reply.started":"2022-07-26T02:13:27.236784Z","shell.execute_reply":"2022-07-26T02:13:27.258170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ar = np.argsort(-friends)\nfriends[ar][:100]","metadata":{"execution":{"iopub.status.busy":"2022-07-26T02:13:27.921814Z","iopub.execute_input":"2022-07-26T02:13:27.922213Z","iopub.status.idle":"2022-07-26T02:13:27.930475Z","shell.execute_reply.started":"2022-07-26T02:13:27.922177Z","shell.execute_reply":"2022-07-26T02:13:27.929466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.value_counts(friends)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T02:13:28.523846Z","iopub.execute_input":"2022-07-26T02:13:28.524683Z","iopub.status.idle":"2022-07-26T02:13:28.533102Z","shell.execute_reply.started":"2022-07-26T02:13:28.524641Z","shell.execute_reply":"2022-07-26T02:13:28.532162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(50):\n    print(y_train[ar[i]])\n    plt.imshow(X_train[ar[i]].reshape((28, 28)))\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T02:14:10.961500Z","iopub.execute_input":"2022-07-26T02:14:10.961935Z","iopub.status.idle":"2022-07-26T02:14:19.563993Z","shell.execute_reply.started":"2022-07-26T02:14:10.961899Z","shell.execute_reply":"2022-07-26T02:14:19.562764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}