{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Credits\n- https://www.kaggle.com/code/samuelcortinhas/tps-july-22-unsupervised-clustering\n- https://scikit-learn.org/stable/modules/clustering.html","metadata":{}},{"cell_type":"markdown","source":"# Import Modules","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.preprocessing import PowerTransformer\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport seaborn as sn\nfrom sklearn.cluster import KMeans","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-13T20:35:29.404002Z","iopub.execute_input":"2022-07-13T20:35:29.404358Z","iopub.status.idle":"2022-07-13T20:35:29.681315Z","shell.execute_reply.started":"2022-07-13T20:35:29.404329Z","shell.execute_reply":"2022-07-13T20:35:29.680622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv')\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T20:08:19.523329Z","iopub.execute_input":"2022-07-13T20:08:19.523697Z","iopub.status.idle":"2022-07-13T20:08:20.434634Z","shell.execute_reply.started":"2022-07-13T20:08:19.523668Z","shell.execute_reply":"2022-07-13T20:08:20.433155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('SHAPE:', df.shape)\nprint('MISSING VALUES:', df.isna().sum().sum())\nprint('DUPLICATES:', df.duplicated().sum())\ndf.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-13T20:12:51.610824Z","iopub.execute_input":"2022-07-13T20:12:51.611151Z","iopub.status.idle":"2022-07-13T20:12:51.803224Z","shell.execute_reply.started":"2022-07-13T20:12:51.611127Z","shell.execute_reply":"2022-07-13T20:12:51.802329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Correlation","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(7,5))\nsn.heatmap(df.corr().abs(), cmap='Greens', vmin=0, vmax=1)\nplt.title('Absolute correlations')","metadata":{"execution":{"iopub.status.busy":"2022-07-13T20:32:18.105200Z","iopub.execute_input":"2022-07-13T20:32:18.105610Z","iopub.status.idle":"2022-07-13T20:32:18.778565Z","shell.execute_reply.started":"2022-07-13T20:32:18.105576Z","shell.execute_reply":"2022-07-13T20:32:18.777851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Elbow Method (for determining the # of clusters)","metadata":{}},{"cell_type":"code","source":"inertias = []\nfor k in range(1,50):\n    km = KMeans(n_clusters=k)\n    km.fit(df)\n    inertias.append(km.inertia_)\n\n# Plot inertias\nplt.figure(figsize=(16,6))\nplt.plot(range(1,50), inertias, 'bx-')\nplt.xlabel('Number of clusters, k')\nplt.ylabel('Inertia')\nplt.title('Elbow method')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T20:44:43.955699Z","iopub.execute_input":"2022-07-13T20:44:43.956194Z","iopub.status.idle":"2022-07-13T20:51:17.118624Z","shell.execute_reply.started":"2022-07-13T20:44:43.956169Z","shell.execute_reply":"2022-07-13T20:51:17.117603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Scaling","metadata":{}},{"cell_type":"code","source":"scaled_df = pd.DataFrame(PowerTransformer().fit_transform(df))\n\nscaled_df.columns = df.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-13T20:28:51.163997Z","iopub.execute_input":"2022-07-13T20:28:51.164359Z","iopub.status.idle":"2022-07-13T20:28:56.746919Z","shell.execute_reply.started":"2022-07-13T20:28:51.164330Z","shell.execute_reply":"2022-07-13T20:28:56.745753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# k-Means","metadata":{}},{"cell_type":"code","source":"model_km = KMeans(n_clusters=4, random_state=0)\npreds_km = model_km.fit_predict(scaled_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T21:05:01.966064Z","iopub.execute_input":"2022-07-13T21:05:01.966488Z","iopub.status.idle":"2022-07-13T21:05:05.790943Z","shell.execute_reply.started":"2022-07-13T21:05:01.966456Z","shell.execute_reply":"2022-07-13T21:05:05.789783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv('../input/tabular-playground-series-jul-2022/sample_submission.csv')\nsub['Predicted'] = preds_km\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T21:05:15.528935Z","iopub.execute_input":"2022-07-13T21:05:15.529620Z","iopub.status.idle":"2022-07-13T21:05:15.718689Z","shell.execute_reply.started":"2022-07-13T21:05:15.529446Z","shell.execute_reply":"2022-07-13T21:05:15.717557Z"},"trusted":true},"execution_count":null,"outputs":[]}]}