{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-14T14:03:43.605433Z","iopub.execute_input":"2022-07-14T14:03:43.605812Z","iopub.status.idle":"2022-07-14T14:03:43.615642Z","shell.execute_reply.started":"2022-07-14T14:03:43.605780Z","shell.execute_reply":"2022-07-14T14:03:43.614137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# UNSUPERVISED LEARNING USING PCA AND KMEANS CLUSTERING \n*This notebook shows an approach towards unsupervised learning using Principal Component Analysis(PCA) and KMeans Clustering. \nA good source to grasp the concept of PCA is in this [link.](https://youtu.be/FgakZw6K1QQ)*\n\n*I am quite new to unsupervised learning. So, your suggestions to improve it would be of great help.*","metadata":{}},{"cell_type":"markdown","source":"# Loading the data ","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv')\ndf.head().style","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:03:43.620755Z","iopub.execute_input":"2022-07-14T14:03:43.621720Z","iopub.status.idle":"2022-07-14T14:03:44.568080Z","shell.execute_reply.started":"2022-07-14T14:03:43.621678Z","shell.execute_reply":"2022-07-14T14:03:44.566726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = df.drop(['id'], axis=1)\ndf1.head().style","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:03:44.570714Z","iopub.execute_input":"2022-07-14T14:03:44.571069Z","iopub.status.idle":"2022-07-14T14:03:44.598367Z","shell.execute_reply.started":"2022-07-14T14:03:44.571038Z","shell.execute_reply":"2022-07-14T14:03:44.596991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:03:44.599679Z","iopub.execute_input":"2022-07-14T14:03:44.601206Z","iopub.status.idle":"2022-07-14T14:03:44.616477Z","shell.execute_reply.started":"2022-07-14T14:03:44.601159Z","shell.execute_reply":"2022-07-14T14:03:44.614945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**As there are no null values so only scaling is required.**","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\nscaler = MinMaxScaler()\nX_scaled = scaler.fit_transform(df1)\nX_scaled","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:03:44.619850Z","iopub.execute_input":"2022-07-14T14:03:44.620382Z","iopub.status.idle":"2022-07-14T14:03:44.679052Z","shell.execute_reply.started":"2022-07-14T14:03:44.620311Z","shell.execute_reply":"2022-07-14T14:03:44.677390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Using PCA to find principal components**","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA\npca = PCA(n_components=3)\npca.fit(X_scaled)\n\nX_pca = pca.transform(X_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:03:44.681014Z","iopub.execute_input":"2022-07-14T14:03:44.681469Z","iopub.status.idle":"2022-07-14T14:03:45.288309Z","shell.execute_reply.started":"2022-07-14T14:03:44.681429Z","shell.execute_reply":"2022-07-14T14:03:45.287006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_pca.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:03:45.290452Z","iopub.execute_input":"2022-07-14T14:03:45.291308Z","iopub.status.idle":"2022-07-14T14:03:45.299922Z","shell.execute_reply.started":"2022-07-14T14:03:45.291252Z","shell.execute_reply":"2022-07-14T14:03:45.298577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pca.explained_variance_ratio_","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:03:45.301986Z","iopub.execute_input":"2022-07-14T14:03:45.302911Z","iopub.status.idle":"2022-07-14T14:03:45.316865Z","shell.execute_reply.started":"2022-07-14T14:03:45.302852Z","shell.execute_reply":"2022-07-14T14:03:45.315545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport cufflinks as cf\n%matplotlib inline\n\nfrom plotly.offline import download_plotlyjs, init_notebook_mode, plot, iplot\ninit_notebook_mode(connected=True)\n\ncf.go_offline()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:07:18.413103Z","iopub.execute_input":"2022-07-14T14:07:18.413481Z","iopub.status.idle":"2022-07-14T14:07:19.377746Z","shell.execute_reply.started":"2022-07-14T14:07:18.413450Z","shell.execute_reply":"2022-07-14T14:07:19.376784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_new = pd.DataFrame(X_pca)\ndf_new.rename(columns={0:'Component_1', 1:'Component_2', 2:'Component_3'}, inplace=True)\ndf_new.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:07:25.486872Z","iopub.execute_input":"2022-07-14T14:07:25.487737Z","iopub.status.idle":"2022-07-14T14:07:25.502582Z","shell.execute_reply.started":"2022-07-14T14:07:25.487688Z","shell.execute_reply":"2022-07-14T14:07:25.501406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Visualization of the data using principal components**","metadata":{}},{"cell_type":"code","source":"px.scatter_3d(df_new, x='Component_1', y='Component_2', z='Component_3', size_max=12)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:10:04.515578Z","iopub.execute_input":"2022-07-14T14:10:04.515965Z","iopub.status.idle":"2022-07-14T14:10:05.423914Z","shell.execute_reply.started":"2022-07-14T14:10:04.515931Z","shell.execute_reply":"2022-07-14T14:10:05.419569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Using KElbowVisualizer for kinding the optimal value of k\nHere two different scores has been used to find the best value of k.**","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nfrom yellowbrick.cluster import KElbowVisualizer\n\nvisualizer = KElbowVisualizer(\n    KMeans(), k=(2,10)\n)\n\nvisualizer.fit(X_pca)\nvisualizer.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:11:34.016964Z","iopub.execute_input":"2022-07-14T14:11:34.017363Z","iopub.status.idle":"2022-07-14T14:12:00.587384Z","shell.execute_reply.started":"2022-07-14T14:11:34.017305Z","shell.execute_reply":"2022-07-14T14:12:00.586221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualizer = KElbowVisualizer(\n    KMeans(), k=(2,10), metric='calinski_harabasz'\n)\n\nvisualizer.fit(X_pca)\nvisualizer.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:12:10.506814Z","iopub.execute_input":"2022-07-14T14:12:10.507218Z","iopub.status.idle":"2022-07-14T14:12:35.576038Z","shell.execute_reply.started":"2022-07-14T14:12:10.507186Z","shell.execute_reply":"2022-07-14T14:12:35.574821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Modeling using KMeans Clustering**","metadata":{}},{"cell_type":"code","source":"model = KMeans(n_clusters=4, random_state=12).fit(X_pca)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:13:04.206665Z","iopub.execute_input":"2022-07-14T14:13:04.207067Z","iopub.status.idle":"2022-07-14T14:13:06.768298Z","shell.execute_reply.started":"2022-07-14T14:13:04.207034Z","shell.execute_reply":"2022-07-14T14:13:06.767178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.cluster_centers_","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:13:19.426791Z","iopub.execute_input":"2022-07-14T14:13:19.427179Z","iopub.status.idle":"2022-07-14T14:13:19.435384Z","shell.execute_reply.started":"2022-07-14T14:13:19.427143Z","shell.execute_reply":"2022-07-14T14:13:19.433965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.labels_","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:13:24.207140Z","iopub.execute_input":"2022-07-14T14:13:24.207566Z","iopub.status.idle":"2022-07-14T14:13:24.214755Z","shell.execute_reply.started":"2022-07-14T14:13:24.207531Z","shell.execute_reply":"2022-07-14T14:13:24.213611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Final_df = pd.DataFrame(X_pca)\nFinal_df['Predicted_label'] = model.labels_","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:13:48.981244Z","iopub.execute_input":"2022-07-14T14:13:48.981642Z","iopub.status.idle":"2022-07-14T14:13:48.988969Z","shell.execute_reply.started":"2022-07-14T14:13:48.981610Z","shell.execute_reply":"2022-07-14T14:13:48.987894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Final_df","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:16:55.132737Z","iopub.execute_input":"2022-07-14T14:16:55.133150Z","iopub.status.idle":"2022-07-14T14:16:55.149153Z","shell.execute_reply.started":"2022-07-14T14:16:55.133115Z","shell.execute_reply":"2022-07-14T14:16:55.148088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Visualization of Clustered Data**","metadata":{}},{"cell_type":"code","source":"px.scatter_3d(Final_df, x=0, y=1, z=2, color='Predicted_label', size_max=12)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:19:38.293476Z","iopub.execute_input":"2022-07-14T14:19:38.293860Z","iopub.status.idle":"2022-07-14T14:19:39.514084Z","shell.execute_reply.started":"2022-07-14T14:19:38.293827Z","shell.execute_reply":"2022-07-14T14:19:39.513262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = pd.concat([df.id, Final_df.Predicted_label], axis=1)\ndf_submission","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:20:04.238559Z","iopub.execute_input":"2022-07-14T14:20:04.238940Z","iopub.status.idle":"2022-07-14T14:20:04.255644Z","shell.execute_reply.started":"2022-07-14T14:20:04.238910Z","shell.execute_reply":"2022-07-14T14:20:04.254859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.rename(columns={'id':'Id', 'Predicted_label':'Predicted'}, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:20:08.252514Z","iopub.execute_input":"2022-07-14T14:20:08.253071Z","iopub.status.idle":"2022-07-14T14:20:08.258364Z","shell.execute_reply.started":"2022-07-14T14:20:08.253040Z","shell.execute_reply":"2022-07-14T14:20:08.257346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:20:12.622373Z","iopub.execute_input":"2022-07-14T14:20:12.622890Z","iopub.status.idle":"2022-07-14T14:20:12.633903Z","shell.execute_reply.started":"2022-07-14T14:20:12.622847Z","shell.execute_reply":"2022-07-14T14:20:12.633130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Final Submission**","metadata":{}},{"cell_type":"code","source":"df_submission.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T14:22:21.948153Z","iopub.execute_input":"2022-07-14T14:22:21.948572Z","iopub.status.idle":"2022-07-14T14:22:22.111801Z","shell.execute_reply.started":"2022-07-14T14:22:21.948539Z","shell.execute_reply":"2022-07-14T14:22:22.110629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Thank you for going through it and please provide your suggestions and any better approach which could help me improve the score.***","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}