{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport seaborn as sns\nfrom sklearn.cluster import KMeans\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import StandardScaler, RobustScaler, PowerTransformer\nfrom yellowbrick.cluster import KElbowVisualizer","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-23T14:56:08.915849Z","iopub.execute_input":"2022-07-23T14:56:08.916376Z","iopub.status.idle":"2022-07-23T14:56:08.924385Z","shell.execute_reply.started":"2022-07-23T14:56:08.916330Z","shell.execute_reply":"2022-07-23T14:56:08.922449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:56:08.969384Z","iopub.execute_input":"2022-07-23T14:56:08.970363Z","iopub.status.idle":"2022-07-23T14:56:09.717073Z","shell.execute_reply.started":"2022-07-23T14:56:08.970309Z","shell.execute_reply":"2022-07-23T14:56:09.715861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:56:09.719484Z","iopub.execute_input":"2022-07-23T14:56:09.719989Z","iopub.status.idle":"2022-07-23T14:56:09.741189Z","shell.execute_reply.started":"2022-07-23T14:56:09.719945Z","shell.execute_reply":"2022-07-23T14:56:09.739862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:56:09.743697Z","iopub.execute_input":"2022-07-23T14:56:09.744705Z","iopub.status.idle":"2022-07-23T14:56:10.533647Z","shell.execute_reply.started":"2022-07-23T14:56:09.744657Z","shell.execute_reply":"2022-07-23T14:56:10.532523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(15,14))\nfor i, f in enumerate(df.columns):\n    plt.subplot(6, 5, i+1)\n    sns.histplot(x=df[f])\n    plt.title(f'feature: {f}')\n\nfig.suptitle('Feature distributions', size=20)\nfig.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:56:10.536463Z","iopub.execute_input":"2022-07-23T14:56:10.537133Z","iopub.status.idle":"2022-07-23T14:56:23.388333Z","shell.execute_reply.started":"2022-07-23T14:56:10.537086Z","shell.execute_reply":"2022-07-23T14:56:23.387200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,10), dpi= 80)\nsns.heatmap(df.corr(), xticklabels=df.corr().columns, yticklabels=df.corr().columns, cmap='RdYlGn', center=0, annot=False)\n\n# Decorations\nplt.title('Correlogram of clusters', fontsize=22)\nplt.xticks(fontsize=12)\nplt.yticks(fontsize=12)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:56:23.389573Z","iopub.execute_input":"2022-07-23T14:56:23.389937Z","iopub.status.idle":"2022-07-23T14:56:24.786295Z","shell.execute_reply.started":"2022-07-23T14:56:23.389907Z","shell.execute_reply":"2022-07-23T14:56:24.785007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"visualizer = KElbowVisualizer(KMeans(), k=15, timings=False)\n_df = df.drop('id',axis=1)\nvisualizer.fit(_df)\nvisualizer.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:56:24.787848Z","iopub.execute_input":"2022-07-23T14:56:24.788295Z","iopub.status.idle":"2022-07-23T14:57:54.658472Z","shell.execute_reply.started":"2022-07-23T14:56:24.788249Z","shell.execute_reply":"2022-07-23T14:57:54.657019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have established the best amount of clusters are 6 where the modulus value of the gradient of the curve is below 1","metadata":{}},{"cell_type":"code","source":"scaled_data = pd.DataFrame(PowerTransformer().fit_transform(df))\nfeatures= [7, 8, 9, 10, 11, 12, 13, 22, 23, 24, 25, 26, 27, 28]\nscaled_data_crop = scaled_data[features]\nscaled_data_crop\n","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:57:54.660332Z","iopub.execute_input":"2022-07-23T14:57:54.660871Z","iopub.status.idle":"2022-07-23T14:57:58.448427Z","shell.execute_reply.started":"2022-07-23T14:57:54.660820Z","shell.execute_reply":"2022-07-23T14:57:58.446925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kmeans = KMeans(n_clusters=6,algorithm='elkan')\nscaled_data_crop[\"Cluster\"] = kmeans.fit_predict(scaled_data_crop)\nscaled_data_crop[\"Cluster\"] = scaled_data_crop[\"Cluster\"].astype(\"category\")\nscaled_data_crop","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:59:04.071373Z","iopub.execute_input":"2022-07-23T14:59:04.071809Z","iopub.status.idle":"2022-07-23T14:59:08.655437Z","shell.execute_reply.started":"2022-07-23T14:59:04.071770Z","shell.execute_reply":"2022-07-23T14:59:08.654572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.mixture import GaussianMixture, BayesianGaussianMixture\n\nmodel_gmm = GaussianMixture(n_components=6, random_state=0)\npreds_gmm = model_gmm.fit_predict(scaled_data[features])\npreds_gmm","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:28:06.783028Z","iopub.execute_input":"2022-07-23T15:28:06.784001Z","iopub.status.idle":"2022-07-23T15:28:12.957555Z","shell.execute_reply.started":"2022-07-23T15:28:06.783925Z","shell.execute_reply":"2022-07-23T15:28:12.955802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Plot once you have the actual clusters\n\n# plt.figure(figsize=(10,8), dpi= 80)\n# sns.pairplot(df.sample(10).drop('id',axis=1))\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T14:58:02.963196Z","iopub.execute_input":"2022-07-23T14:58:02.963988Z","iopub.status.idle":"2022-07-23T14:58:02.969749Z","shell.execute_reply.started":"2022-07-23T14:58:02.963939Z","shell.execute_reply":"2022-07-23T14:58:02.968423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({'Id' : df['id'],\n                       'Predicted' : preds_gmm})\n\n# output['target'] = output['target'].astype(int).map(labels)\noutput.sort_values(by='Id').to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-23T15:28:45.192074Z","iopub.execute_input":"2022-07-23T15:28:45.192469Z","iopub.status.idle":"2022-07-23T15:28:45.309135Z","shell.execute_reply.started":"2022-07-23T15:28:45.192437Z","shell.execute_reply":"2022-07-23T15:28:45.307881Z"},"trusted":true},"execution_count":null,"outputs":[]}]}