{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"data = pd.read_csv(\"../input/train.csv\")\ndata.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1d6929a64830de902d2008ef624365df3b630e03"},"cell_type":"code","source":"a = data[\"label\"].unique()\na.sort()\nprint(\"There are {} unique label values present in the given dataset, which are {}\".format(a.shape[0], a))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b03cb757c23d85b5efbc4deb0b69edf0a28b674"},"cell_type":"code","source":"labels = data[\"label\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b311c73d0b81459ea0e28814a0f1a22bfdbb544"},"cell_type":"code","source":"data_final = data.drop(\"label\",axis = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"93da2a59d9430b4805099f7ac897ee6d71e95ade"},"cell_type":"code","source":"print(\"The shape of the input data is {}, which means that there are {} number of entries \".format(data_final.shape, data_final.shape[0]))\nprint (\"The number of the train labels are {}\".format(labels.shape[0]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"36cf13e7659858bd337efc3d3bfc690f31106793"},"cell_type":"code","source":"## Lets plot the image to check the input data (SANITY CHECK)\nplt.figure(figsize =(7,7))\nidx = 121\n\ngrid_data = data_final.iloc[idx].as_matrix().reshape(28,28)\nplt.imshow(grid_data, interpolation = \"none\", cmap = \"gray\")\nplt.show()\n\nprint(labels[idx])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b447a2f4cc9b64628dcfe88b5a0369445540e7ee"},"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nstandardised_data = StandardScaler().fit_transform(data_final)\nprint(standardised_data.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"afab53f521140cf7c63069aa26b8b88321562c84"},"cell_type":"code","source":"covariance_matrix = (1/42000)*(np.matmul(standardised_data.T, standardised_data))\nprint(\"The shape of our Covariance Matrix is {}\".format(covariance_matrix.shape))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"74bc1054c209d15cf8c39dc707c87a55e92b27cc"},"cell_type":"code","source":"from scipy.linalg import eigh\n\nvalues,vectors = eigh(covariance_matrix, eigvals=(782,783))\nprint(vectors.shape)\n\nvectors = vectors.T","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"942b07e15d2dff10fcccbad059f3c6aade694868"},"cell_type":"code","source":"print(vectors.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"180d37823a8cd6e39e88fadb21b33cf3b999f263"},"cell_type":"code","source":"new_coordinates = np.matmul(vectors, standardised_data.T)\nprint( \"new data points are\", vectors.shape, \"X\" ,standardised_data.T.shape, \"is\", new_coordinates.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6ff7820959b1024fe5c08dcb9811b6717f82ba5e"},"cell_type":"code","source":"new_coordinates = np.vstack((new_coordinates, labels)).T\ndataframe = pd.DataFrame(data = new_coordinates, columns = (\"1st_Principal\", \"2nd_Principal\", \"label\"))\nprint(dataframe.head(5))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bb673591cd5aa4dd5fa989b5aa79f28a9b922422"},"cell_type":"code","source":"import seaborn as sns\n\nsns.FacetGrid(dataframe, hue = \"label\", size = 10).map(plt.scatter, \"1st_Principal\", \"2nd_Principal\").add_legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"94ab9482304162e7bf0203636ca895375476d9d5"},"cell_type":"code","source":"from sklearn import decomposition\npca = decomposition.PCA()\npca.n_components = 2 \npca_data = pca.fit_transform(standardised_data)\nprint(pca_data.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cbcacff9ba3c2b865fab4e84ef3282a76cff0b35"},"cell_type":"code","source":"pca_data = np.vstack((pca_data.T, labels)).T\npca_data.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"be04a461319ca909f36e596edc1f255253191b5d"},"cell_type":"code","source":"pca_data_final = pd.DataFrame(data = pca_data, columns = (\"1st_Principal\", \"2nd_Principal\", \"labels\"))\nsns.FacetGrid(pca_data_final, hue = \"labels\", size = 10).map(plt.scatter, \"1st_Principal\", \"2nd_Principal\").add_legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bd9bbbfd5bb7e52d853450adc77dcbe8c336dd88"},"cell_type":"code","source":"from sklearn import decomposition \npca = decomposition.PCA()\npca.ncomponents = 784\npca_data_2 = pca.fit_transform(standardised_data)\npercentage_var_explained = pca.explained_variance_/np.sum(pca.explained_variance_)\ncum_var_explained = np.cumsum(percentage_var_explained)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5bb86e82c23b110bfe0f4d299c8ecf759a972a14"},"cell_type":"code","source":"plt.figure(1, figsize = (10,6))\nplt.clf()\nplt.plot(cum_var_explained, linewidth = 2)\nplt.axis(\"tight\")\nplt.grid()\nplt.xlabel(\"n_components\")\nplt.ylabel(\"Cumulative Explained Variance\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bd97f9a501cfb533eb70ca75e37e3872a7042c10"},"cell_type":"code","source":"from sklearn.manifold import TSNE\n\ndata_5000 = standardised_data[0:5000,:]\nlabels_5000 = labels[0:5000]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"387439b783c850a1bfb543130087238ebd186145"},"cell_type":"code","source":"model = TSNE(n_components = 2, perplexity = 50.0, random_state = 0)\ntsne_data = model.fit_transform(data_5000)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2598ac67698940b31306ff4412109f5a8c1c6f62"},"cell_type":"code","source":"tsne_data = np.vstack((tsne_data.T,labels_5000)).T\ntsne_df = pd.DataFrame(tsne_data, columns = (\"1st_Dim\",\"2nd_Dim\",\"Labels\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"29f62f8194ac25fb6bb2846788cbb82ec1743843"},"cell_type":"code","source":"sns.FacetGrid(tsne_df, hue = \"Labels\", size = 10).map(plt.scatter, \"1st_Dim\", \"2nd_Dim\").add_legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bb8da07080c86b039897bc64cb7efbac32affc30"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}