{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-30T17:20:35.664085Z","iopub.execute_input":"2022-07-30T17:20:35.664948Z","iopub.status.idle":"2022-07-30T17:20:35.696159Z","shell.execute_reply.started":"2022-07-30T17:20:35.664849Z","shell.execute_reply":"2022-07-30T17:20:35.695283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nimport re\nimport pandas as pd\nimport sklearn\nimport numpy as np\nfrom pprint import pprint\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.base import BaseEstimator\nfrom datetime import datetime\n\nfrom sklearn.linear_model import LinearRegression, Lasso, Ridge, ElasticNet\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.metrics import (accuracy_score, precision_score, recall_score,\n    f1_score, confusion_matrix, precision_recall_curve, roc_curve, roc_auc_score, auc\n) \nfrom sklearn.model_selection import train_test_split, cross_val_score, cross_val_predict\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import ExtraTreesClassifier, RandomForestClassifier\nfrom sklearn.decomposition import PCA\nfrom sklearn.cluster import KMeans\n\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import StandardScaler\n\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2022-07-30T17:29:23.067600Z","iopub.execute_input":"2022-07-30T17:29:23.067987Z","iopub.status.idle":"2022-07-30T17:29:23.078930Z","shell.execute_reply.started":"2022-07-30T17:29:23.067957Z","shell.execute_reply":"2022-07-30T17:29:23.077733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"############\n# Load data\n############\n\ntrain_df = pd.read_csv(\"/kaggle/input/Kannada-MNIST/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/Kannada-MNIST/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-30T17:20:53.947027Z","iopub.execute_input":"2022-07-30T17:20:53.947817Z","iopub.status.idle":"2022-07-30T17:20:59.191026Z","shell.execute_reply.started":"2022-07-30T17:20:53.947784Z","shell.execute_reply":"2022-07-30T17:20:59.189782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Peek at the train dataset\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T17:21:09.647271Z","iopub.execute_input":"2022-07-30T17:21:09.648118Z","iopub.status.idle":"2022-07-30T17:21:09.669761Z","shell.execute_reply.started":"2022-07-30T17:21:09.648072Z","shell.execute_reply":"2022-07-30T17:21:09.668446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-30T17:21:17.533598Z","iopub.execute_input":"2022-07-30T17:21:17.534002Z","iopub.status.idle":"2022-07-30T17:21:17.540947Z","shell.execute_reply.started":"2022-07-30T17:21:17.533962Z","shell.execute_reply":"2022-07-30T17:21:17.539949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T17:21:23.726711Z","iopub.execute_input":"2022-07-30T17:21:23.727126Z","iopub.status.idle":"2022-07-30T17:21:26.523872Z","shell.execute_reply.started":"2022-07-30T17:21:23.727089Z","shell.execute_reply":"2022-07-30T17:21:26.522840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Peek at the test dataset\ntest_df","metadata":{"execution":{"iopub.status.busy":"2022-07-30T17:21:31.089233Z","iopub.execute_input":"2022-07-30T17:21:31.089636Z","iopub.status.idle":"2022-07-30T17:21:31.117173Z","shell.execute_reply.started":"2022-07-30T17:21:31.089601Z","shell.execute_reply":"2022-07-30T17:21:31.116288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T17:21:50.029111Z","iopub.execute_input":"2022-07-30T17:21:50.029521Z","iopub.status.idle":"2022-07-30T17:21:51.660179Z","shell.execute_reply.started":"2022-07-30T17:21:50.029488Z","shell.execute_reply":"2022-07-30T17:21:51.659136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# combining the train and test dataframes\ndf = pd.concat([train_df, test_df])\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-30T17:52:53.536593Z","iopub.execute_input":"2022-07-30T17:52:53.537003Z","iopub.status.idle":"2022-07-30T17:52:53.721281Z","shell.execute_reply.started":"2022-07-30T17:52:53.536970Z","shell.execute_reply":"2022-07-30T17:52:53.719979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#scale all the data\n\nstdscale = StandardScaler()\nnon_scalar = list(df.drop([\"label\",\"id\"], axis=1))\nfor i in non_scalar:\n       df[i] = stdscale.fit_transform(df[[i]])","metadata":{"execution":{"iopub.status.busy":"2022-07-30T17:53:25.710973Z","iopub.execute_input":"2022-07-30T17:53:25.711396Z","iopub.status.idle":"2022-07-30T17:53:29.232773Z","shell.execute_reply.started":"2022-07-30T17:53:25.711348Z","shell.execute_reply":"2022-07-30T17:53:29.231558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T17:53:39.576873Z","iopub.execute_input":"2022-07-30T17:53:39.577271Z","iopub.status.idle":"2022-07-30T17:53:39.617239Z","shell.execute_reply.started":"2022-07-30T17:53:39.577238Z","shell.execute_reply":"2022-07-30T17:53:39.616074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T17:53:52.899172Z","iopub.execute_input":"2022-07-30T17:53:52.899560Z","iopub.status.idle":"2022-07-30T17:53:56.467318Z","shell.execute_reply.started":"2022-07-30T17:53:52.899528Z","shell.execute_reply":"2022-07-30T17:53:56.466238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reform the training and testing dataframes\n# Training: up to 60,000\n# Testing: 60,000 to 65,000\ntrain_df = df.iloc[:60000,:]\ntest_df = df.iloc[60000:,:]","metadata":{"execution":{"iopub.status.busy":"2022-07-30T17:58:36.263662Z","iopub.execute_input":"2022-07-30T17:58:36.264063Z","iopub.status.idle":"2022-07-30T17:58:36.270178Z","shell.execute_reply.started":"2022-07-30T17:58:36.264031Z","shell.execute_reply":"2022-07-30T17:58:36.269332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.drop(\"id\", axis = 1)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-07-30T17:59:15.060641Z","iopub.execute_input":"2022-07-30T17:59:15.061024Z","iopub.status.idle":"2022-07-30T17:59:15.219696Z","shell.execute_reply.started":"2022-07-30T17:59:15.060985Z","shell.execute_reply":"2022-07-30T17:59:15.218359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_df.drop(\"label\", axis = 1)\ntest_df","metadata":{"execution":{"iopub.status.busy":"2022-07-30T17:58:38.378704Z","iopub.execute_input":"2022-07-30T17:58:38.379698Z","iopub.status.idle":"2022-07-30T17:58:38.431783Z","shell.execute_reply.started":"2022-07-30T17:58:38.379656Z","shell.execute_reply":"2022-07-30T17:58:38.430443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set the values of x and y to train the randomforest model\nx = train_df.drop(\"label\", axis=1)\ny = train_df[\"label\"].values\n\n# split data into training and testing sets\nfrom sklearn.model_selection import train_test_split\nx_train, x_test, y_train, y_test = train_test_split(x, y, test_size = 0.15, random_state = 42)\n\n# instantiate the classifier \nrfc = RandomForestClassifier(random_state=42)\n\n# fit the model and time how long it takes to fit\nstart=datetime.now()\nrfc.fit(x_train, y_train)\nend=datetime.now()\nprint(end-start)\n\ny_pred = rfc.predict(x_test)\n\n\n# Check accuracy score \nfrom sklearn.metrics import accuracy_score\nprint('Model accuracy score with 100 decision-trees : {0:0.4f}'. format(accuracy_score(y_test, y_pred)))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T17:59:55.801781Z","iopub.execute_input":"2022-07-30T17:59:55.802899Z","iopub.status.idle":"2022-07-30T18:00:27.782718Z","shell.execute_reply.started":"2022-07-30T17:59:55.802856Z","shell.execute_reply":"2022-07-30T18:00:27.781437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Inspect the feature importances and print the highest 10\nimportances = rfc.feature_importances_\nindices = np.argsort(importances)[::-1]\n\n\nprint(\"Feature ranking:\")\nfor f in range(0,10):\n    print(\"%d. feature %d (%f)\" % (f + 1, indices[f], importances[indices[f]]))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:31:01.917482Z","iopub.execute_input":"2022-07-30T18:31:01.918011Z","iopub.status.idle":"2022-07-30T18:31:01.953358Z","shell.execute_reply.started":"2022-07-30T18:31:01.917969Z","shell.execute_reply":"2022-07-30T18:31:01.952070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.decomposition import PCA\n\n\n\n#setting number of components\npca_test = PCA(n_components=500, random_state = 42)\n\n# Fitting and timing \nstart=datetime.now()\npca_test.fit(x)\nend=datetime.now()\nprint(end-start)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:31:14.873666Z","iopub.execute_input":"2022-07-30T18:31:14.874157Z","iopub.status.idle":"2022-07-30T18:31:34.184735Z","shell.execute_reply.started":"2022-07-30T18:31:14.874118Z","shell.execute_reply":"2022-07-30T18:31:34.183418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(style='whitegrid')\nplt.plot(np.cumsum(pca_test.explained_variance_ratio_))\nplt.xlabel('number of components')\nplt.ylabel('cumulative explained variance')\nplt.axvline(linewidth=4, color='r', linestyle = '--', x=430, ymin=0, ymax=1)\ndisplay(plt.show())\nevr = pca_test.explained_variance_ratio_\ncvr = np.cumsum(pca_test.explained_variance_ratio_)\npca_df = pd.DataFrame()\npca_df['Cumulative Variance Ratio'] = cvr\npca_df['Explained Variance Ratio'] = evr\ndisplay(pca_df.head(430))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:31:57.545975Z","iopub.execute_input":"2022-07-30T18:31:57.547177Z","iopub.status.idle":"2022-07-30T18:31:57.798326Z","shell.execute_reply.started":"2022-07-30T18:31:57.547131Z","shell.execute_reply":"2022-07-30T18:31:57.797273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit model using PCA, generating principal components that represent 95 percent of the variability in \n# the explanatory features\nstart=datetime.now()\npca = PCA(.95, random_state = 42)\npca.fit(x)\nend=datetime.now()\nprint(end-start)\n\nprint('Principal components count: ', pca.n_components_)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:32:09.504319Z","iopub.execute_input":"2022-07-30T18:32:09.505389Z","iopub.status.idle":"2022-07-30T18:32:17.024866Z","shell.execute_reply.started":"2022-07-30T18:32:09.505326Z","shell.execute_reply":"2022-07-30T18:32:17.023849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pca = pca.transform(x)\npca.explained_variance_ratio_","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:32:38.094296Z","iopub.execute_input":"2022-07-30T18:32:38.094671Z","iopub.status.idle":"2022-07-30T18:32:39.456833Z","shell.execute_reply.started":"2022-07-30T18:32:38.094640Z","shell.execute_reply":"2022-07-30T18:32:39.455652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = train_pca\ny = train_df[\"label\"].values\n\n# split data into training and testing sets\nfrom sklearn.model_selection import train_test_split\nx_train, x_test, y_train, y_test = train_test_split(x, y, test_size = 0.15, random_state = 42)\n\n# instantiate the classifier \npca_rfc = RandomForestClassifier(random_state=42)\n\n# fit the model\nstart=datetime.now()\npca_rfc.fit(x_train, y_train)\nend=datetime.now()\nprint(end-start)\n\ny_pred = pca_rfc.predict(x_test)\n\n\n# Check accuracy score \nfrom sklearn.metrics import accuracy_score\nprint('Model accuracy score with 100 decision-trees : {0:0.4f}'. format(accuracy_score(y_test, y_pred)))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:33:21.513993Z","iopub.execute_input":"2022-07-30T18:33:21.514410Z","iopub.status.idle":"2022-07-30T18:35:59.478205Z","shell.execute_reply.started":"2022-07-30T18:33:21.514353Z","shell.execute_reply":"2022-07-30T18:35:59.476758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#########\n# Random Forest Classifier using PCA test\n#########\n\ntest_x = test_df.drop([\"id\"], axis=1)\npca_test_x = pca.transform(test_x)\ntest_y = test_df[[\"id\"]]\n\ntest_y[\"label\"] = pca_rfc.predict(pca_test_x)\ntest_y.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T18:36:16.953820Z","iopub.execute_input":"2022-07-30T18:36:16.955049Z","iopub.status.idle":"2022-07-30T18:36:17.356076Z","shell.execute_reply.started":"2022-07-30T18:36:16.955006Z","shell.execute_reply":"2022-07-30T18:36:17.354926Z"},"trusted":true},"execution_count":null,"outputs":[]}]}