{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-07T20:50:48.622590Z","iopub.execute_input":"2022-08-07T20:50:48.623173Z","iopub.status.idle":"2022-08-07T20:50:48.820827Z","shell.execute_reply.started":"2022-08-07T20:50:48.623132Z","shell.execute_reply":"2022-08-07T20:50:48.819453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df= pd.read_csv('/kaggle/input/digit-recognizer/train.csv')\ndf.shape \n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:36:38.091313Z","iopub.execute_input":"2022-08-07T20:36:38.092020Z","iopub.status.idle":"2022-08-07T20:36:40.536258Z","shell.execute_reply.started":"2022-08-07T20:36:38.091976Z","shell.execute_reply":"2022-08-07T20:36:40.535013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x=df.iloc[:,1:]\ny=df.iloc[:,0]","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:36:57.145894Z","iopub.execute_input":"2022-08-07T20:36:57.146330Z","iopub.status.idle":"2022-08-07T20:36:57.153883Z","shell.execute_reply.started":"2022-08-07T20:36:57.146294Z","shell.execute_reply":"2022-08-07T20:36:57.152891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## using train_test_split\nfrom sklearn.model_selection import train_test_split \nx_train, x_test, y_train, y_test = train_test_split(x,y,test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:38:00.411515Z","iopub.execute_input":"2022-08-07T20:38:00.412446Z","iopub.status.idle":"2022-08-07T20:38:01.366333Z","shell.execute_reply.started":"2022-08-07T20:38:00.412392Z","shell.execute_reply":"2022-08-07T20:38:01.365031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## standardisation of the data \nfrom sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nx_train1 =scaler.fit_transform(x_train)\nx_test1  = scaler.fit_transform(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:45:47.277249Z","iopub.execute_input":"2022-08-07T20:45:47.277670Z","iopub.status.idle":"2022-08-07T20:45:47.880934Z","shell.execute_reply.started":"2022-08-07T20:45:47.277636Z","shell.execute_reply":"2022-08-07T20:45:47.879629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## PCA\nfrom sklearn.decomposition import PCA\n\npca = PCA(n_components=2)\nx_train1 = pca.fit_transform(x_train1)\nx_test1 = pca.transform(x_test1)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:47:15.204536Z","iopub.execute_input":"2022-08-07T20:47:15.205554Z","iopub.status.idle":"2022-08-07T20:47:16.150840Z","shell.execute_reply.started":"2022-08-07T20:47:15.205510Z","shell.execute_reply":"2022-08-07T20:47:16.148970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train1","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:47:18.724249Z","iopub.execute_input":"2022-08-07T20:47:18.724670Z","iopub.status.idle":"2022-08-07T20:47:18.732653Z","shell.execute_reply.started":"2022-08-07T20:47:18.724631Z","shell.execute_reply":"2022-08-07T20:47:18.731301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## now my data is in two dimension\n## PCA is actually 784 but only taking 2D for visualization","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:48:38.149330Z","iopub.execute_input":"2022-08-07T20:48:38.149937Z","iopub.status.idle":"2022-08-07T20:48:38.155539Z","shell.execute_reply.started":"2022-08-07T20:48:38.149865Z","shell.execute_reply":"2022-08-07T20:48:38.154239Z"}}},{"cell_type":"code","source":"plt.scatter(\n    x=x_train1[:,0],\n    y=x_train1[:,1],\n    c=y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:54:54.063433Z","iopub.execute_input":"2022-08-07T20:54:54.064000Z","iopub.status.idle":"2022-08-07T20:54:54.931077Z","shell.execute_reply.started":"2022-08-07T20:54:54.063954Z","shell.execute_reply":"2022-08-07T20:54:54.930110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" pca.explained_variance_ratio_","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:58:26.876884Z","iopub.execute_input":"2022-08-07T20:58:26.877335Z","iopub.status.idle":"2022-08-07T20:58:26.886681Z","shell.execute_reply.started":"2022-08-07T20:58:26.877292Z","shell.execute_reply":"2022-08-07T20:58:26.885178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca.explained_variance_ ## eigen values","metadata":{"execution":{"iopub.status.busy":"2022-08-07T20:59:42.544769Z","iopub.execute_input":"2022-08-07T20:59:42.545301Z","iopub.status.idle":"2022-08-07T20:59:42.555192Z","shell.execute_reply.started":"2022-08-07T20:59:42.545258Z","shell.execute_reply":"2022-08-07T20:59:42.553804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca.components_ ## eigen vectors \n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:00:05.660146Z","iopub.execute_input":"2022-08-07T21:00:05.661383Z","iopub.status.idle":"2022-08-07T21:00:05.677269Z","shell.execute_reply.started":"2022-08-07T21:00:05.661328Z","shell.execute_reply":"2022-08-07T21:00:05.675928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}