{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-29T16:29:48.008864Z","iopub.execute_input":"2022-07-29T16:29:48.010052Z","iopub.status.idle":"2022-07-29T16:29:48.039273Z","shell.execute_reply.started":"2022-07-29T16:29:48.009883Z","shell.execute_reply":"2022-07-29T16:29:48.038222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nimport re\nimport pandas as pd\nimport sklearn\nimport numpy as np\nfrom pprint import pprint\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.base import BaseEstimator\nfrom datetime import datetime\n\nfrom sklearn.linear_model import LinearRegression, Lasso, Ridge, ElasticNet\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.metrics import (accuracy_score, precision_score, recall_score,\n    f1_score, confusion_matrix, precision_recall_curve, roc_curve, roc_auc_score, auc\n) \nfrom sklearn.model_selection import train_test_split, cross_val_score, cross_val_predict\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import ExtraTreesClassifier, RandomForestClassifier\nfrom sklearn.decomposition import PCA\nfrom sklearn.cluster import KMeans","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:55:35.728662Z","iopub.execute_input":"2022-07-29T16:55:35.729092Z","iopub.status.idle":"2022-07-29T16:55:35.738078Z","shell.execute_reply.started":"2022-07-29T16:55:35.729057Z","shell.execute_reply":"2022-07-29T16:55:35.736494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"############\n# Load data\n############\n\ntrain_df = pd.read_csv(\"/kaggle/input/Kannada-MNIST/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/Kannada-MNIST/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:34:08.632753Z","iopub.execute_input":"2022-07-29T16:34:08.633204Z","iopub.status.idle":"2022-07-29T16:34:13.435046Z","shell.execute_reply.started":"2022-07-29T16:34:08.633170Z","shell.execute_reply":"2022-07-29T16:34:13.433832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Peek at the train dataset\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:34:22.511794Z","iopub.execute_input":"2022-07-29T16:34:22.512842Z","iopub.status.idle":"2022-07-29T16:34:22.550479Z","shell.execute_reply.started":"2022-07-29T16:34:22.512802Z","shell.execute_reply":"2022-07-29T16:34:22.549429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Peek at the test dataset\ntest_df","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:34:32.526708Z","iopub.execute_input":"2022-07-29T16:34:32.527122Z","iopub.status.idle":"2022-07-29T16:34:32.555895Z","shell.execute_reply.started":"2022-07-29T16:34:32.527088Z","shell.execute_reply":"2022-07-29T16:34:32.554557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set the values of x and y to train the randomforest model\nx = train_df.drop(\"label\", axis=1)\ny = train_df[\"label\"].values\n\n# split data into training and testing sets\nfrom sklearn.model_selection import train_test_split\nx_train, x_test, y_train, y_test = train_test_split(x, y, test_size = 0.15, random_state = 42)\n\n# instantiate the classifier \nrfc = RandomForestClassifier(random_state=42)\n\n# fit the model and time how long it takes to fit\nstart=datetime.now()\nrfc.fit(x_train, y_train)\nend=datetime.now()\nprint(end-start)\n\ny_pred = rfc.predict(x_test)\n\n\n# Check accuracy score \nfrom sklearn.metrics import accuracy_score\nprint('Model accuracy score with 100 decision-trees : {0:0.4f}'. format(accuracy_score(y_test, y_pred)))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:44:56.396193Z","iopub.execute_input":"2022-07-29T16:44:56.396577Z","iopub.status.idle":"2022-07-29T16:45:27.958261Z","shell.execute_reply.started":"2022-07-29T16:44:56.396547Z","shell.execute_reply":"2022-07-29T16:45:27.956419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Export for Kaggle dataset\ntest_x = test_df.drop(['id'], axis = 1)\ntest_y = test_df[['id']]\n\ntest_y['label'] = rfc.predict(test_x)\ntest_y.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:48:03.532613Z","iopub.execute_input":"2022-07-29T16:48:03.533084Z","iopub.status.idle":"2022-07-29T16:48:03.775396Z","shell.execute_reply.started":"2022-07-29T16:48:03.533047Z","shell.execute_reply":"2022-07-29T16:48:03.774245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Inspect the feature importances and print the highest 10\nimportances = rfc.feature_importances_\nindices = np.argsort(importances)[::-1]\n\n\nprint(\"Feature ranking:\")\nfor f in range(0,10):\n    print(\"%d. feature %d (%f)\" % (f + 1, indices[f], importances[indices[f]]))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:55:00.332552Z","iopub.execute_input":"2022-07-29T16:55:00.333010Z","iopub.status.idle":"2022-07-29T16:55:00.362708Z","shell.execute_reply.started":"2022-07-29T16:55:00.332961Z","shell.execute_reply":"2022-07-29T16:55:00.361518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# combining the train and test dataframes\ndf = pd.concat([train_df, test_df])\ndf","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:55:14.069887Z","iopub.execute_input":"2022-07-29T16:55:14.070370Z","iopub.status.idle":"2022-07-29T16:55:14.298057Z","shell.execute_reply.started":"2022-07-29T16:55:14.070328Z","shell.execute_reply":"2022-07-29T16:55:14.296912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.decomposition import PCA\n\n# removing the label\nx = df.drop(['label', 'id'], axis=1)\n\n\n#setting number of components\npca_test = PCA(n_components=500, random_state = 42)\n\n# Fitting and timing \nstart=datetime.now()\npca_test.fit(x)\nend=datetime.now()\nprint(end-start)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:55:39.095046Z","iopub.execute_input":"2022-07-29T16:55:39.095411Z","iopub.status.idle":"2022-07-29T16:55:58.456680Z","shell.execute_reply.started":"2022-07-29T16:55:39.095382Z","shell.execute_reply":"2022-07-29T16:55:58.455506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(style='whitegrid')\nplt.plot(np.cumsum(pca_test.explained_variance_ratio_))\nplt.xlabel('number of components')\nplt.ylabel('cumulative explained variance')\nplt.axvline(linewidth=4, color='r', linestyle = '--', x=153, ymin=0, ymax=1)\ndisplay(plt.show())\nevr = pca_test.explained_variance_ratio_\ncvr = np.cumsum(pca_test.explained_variance_ratio_)\npca_df = pd.DataFrame()\npca_df['Cumulative Variance Ratio'] = cvr\npca_df['Explained Variance Ratio'] = evr\ndisplay(pca_df.head(240))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:56:05.555477Z","iopub.execute_input":"2022-07-29T16:56:05.555882Z","iopub.status.idle":"2022-07-29T16:56:05.612361Z","shell.execute_reply.started":"2022-07-29T16:56:05.555850Z","shell.execute_reply":"2022-07-29T16:56:05.611125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit model using PCA, generating principal components that represent 95 percent of the variability in \n# the explanatory features\nstart=datetime.now()\npca = PCA(.95, random_state = 42)\npca.fit(x)\nend=datetime.now()\nprint(end-start)\n\nprint('Principal components count: ', pca.n_components_)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:56:18.866844Z","iopub.execute_input":"2022-07-29T16:56:18.868109Z","iopub.status.idle":"2022-07-29T16:56:26.454773Z","shell.execute_reply.started":"2022-07-29T16:56:18.868063Z","shell.execute_reply":"2022-07-29T16:56:26.453964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"totimages = pca.transform(x)\npca.explained_variance_ratio_","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:56:32.095019Z","iopub.execute_input":"2022-07-29T16:56:32.095390Z","iopub.status.idle":"2022-07-29T16:56:33.117568Z","shell.execute_reply.started":"2022-07-29T16:56:32.095362Z","shell.execute_reply":"2022-07-29T16:56:33.116199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reform the training and testing dataframes\n# Training: up to 60,000\n# Testing: 60,000 to 65,000\ntrain_pca = pd.DataFrame(totimages[0:60000, :].astype(int))\ntest_pca = pd.DataFrame(totimages[60000:65000, :].astype(int))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:57:36.037408Z","iopub.execute_input":"2022-07-29T16:57:36.037791Z","iopub.status.idle":"2022-07-29T16:57:36.094373Z","shell.execute_reply.started":"2022-07-29T16:57:36.037762Z","shell.execute_reply":"2022-07-29T16:57:36.093045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = train_pca\ny = train_df[\"label\"].values\n\n# split data into training and testing sets\nfrom sklearn.model_selection import train_test_split\nx_train, x_test, y_train, y_test = train_test_split(x, y, test_size = 0.15, random_state = 42)\n\n# instantiate the classifier \npca_rfc = RandomForestClassifier(random_state=42)\n\n# fit the model\nstart=datetime.now()\npca_rfc.fit(x_train, y_train)\nend=datetime.now()\nprint(end-start)\n\ny_pred = pca_rfc.predict(x_test)\n\n\n# Check accuracy score \nfrom sklearn.metrics import accuracy_score\nprint('Model accuracy score with 100 decision-trees : {0:0.4f}'. format(accuracy_score(y_test, y_pred)))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T17:03:18.606525Z","iopub.execute_input":"2022-07-29T17:03:18.606932Z","iopub.status.idle":"2022-07-29T17:04:27.825838Z","shell.execute_reply.started":"2022-07-29T17:03:18.606903Z","shell.execute_reply":"2022-07-29T17:04:27.824729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#########\n# Random Forest Classifier using PCA test\n#########\n\ntest_df = pd.read_csv(\"/kaggle/input/Kannada-MNIST/test.csv\")\n\ntest_x = test_df.drop([\"id\"], axis=1)\npca_test_x = pca.transform(test_x)\ntest_y = test_df[[\"id\"]]\n\ntest_y[\"label\"] = pca_rfc.predict(pca_test_x)\ntest_y.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T17:04:34.668517Z","iopub.execute_input":"2022-07-29T17:04:34.668925Z","iopub.status.idle":"2022-07-29T17:04:35.338271Z","shell.execute_reply.started":"2022-07-29T17:04:34.668892Z","shell.execute_reply":"2022-07-29T17:04:35.337064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def retrieve_info(kmeans, cluster_labels, y_train):\n    # adapted from: https://medium.com/@joel_34096/k-means-clustering-for-image-classification-a648f28bdc47\n    \n    reference_labels = {}\n    \n    for i in range(len(np.unique(kmeans.labels_))):\n        index = np.where(cluster_labels == i,1,0)\n        num = np.bincount(y_train[index==1]).argmax()\n        reference_labels[i] = num\n        \n    return reference_labels","metadata":{"execution":{"iopub.status.busy":"2022-07-29T17:28:01.221811Z","iopub.execute_input":"2022-07-29T17:28:01.222605Z","iopub.status.idle":"2022-07-29T17:28:01.229490Z","shell.execute_reply.started":"2022-07-29T17:28:01.222560Z","shell.execute_reply":"2022-07-29T17:28:01.228341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#########\n# KMeans clustering using PCA test\n#########\n\nx_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.15, random_state=42)\n\n# train our model off of the whole x dataset\nstart = time.time()\nkmeans = KMeans(n_clusters=10, random_state=42).fit(x)\nstop = time.time()\nprint(f\"Training time: {round(stop - start, 4)}s\")\n# given our model and our actual label values...\n# ... determine the labels which correspond to our clusters\n# cluster_labels = infer_cluster_labels(kmeans, y.to_numpy())\n# # to test it out, generate clusters for x_test\n# predict_clusters = kmeans.predict(x)\n# # figure out the labels which correspond to those x_test clusters\n# predicted_labels = infer_data_labels(predict_clusters, cluster_labels)\n# # these should line up\n# print(predicted_labels[:20])\nref_labels = retrieve_info(kmeans, kmeans.labels_, y)\nnumber_labels = []\nfor i in range(len(kmeans.labels_)):\n    number_labels.append(ref_labels[kmeans.labels_[i]])\nprint(number_labels[:20])\nprint(y[:20])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T17:28:11.756987Z","iopub.execute_input":"2022-07-29T17:28:11.757725Z","iopub.status.idle":"2022-07-29T17:28:21.342411Z","shell.execute_reply.started":"2022-07-29T17:28:11.757690Z","shell.execute_reply":"2022-07-29T17:28:21.341070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/Kannada-MNIST/test.csv\")\n\ntest_x = test_df.drop([\"id\"], axis=1)\npca_test_x = pca.transform(test_x)\ntest_y = test_df[[\"id\"]]\n\ntest_predict_clusters = kmeans.predict(pca_test_x)\ntest_labels = []\nfor i in range(len(test_predict_clusters)):\n    test_labels.append(ref_labels[test_predict_clusters[i]])\n#test_y[\"label\"] = infer_data_labels(test_predict_clusters, cluster_labels)\ntest_y[\"label\"] = test_labels\ntest_y.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T17:47:20.288602Z","iopub.execute_input":"2022-07-29T17:47:20.289082Z","iopub.status.idle":"2022-07-29T17:47:20.722677Z","shell.execute_reply.started":"2022-07-29T17:47:20.289048Z","shell.execute_reply":"2022-07-29T17:47:20.718709Z"},"trusted":true},"execution_count":null,"outputs":[]}]}