{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-25T12:40:39.680254Z","iopub.execute_input":"2022-07-25T12:40:39.680656Z","iopub.status.idle":"2022-07-25T12:40:39.686164Z","shell.execute_reply.started":"2022-07-25T12:40:39.680623Z","shell.execute_reply":"2022-07-25T12:40:39.684960Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing predicted clusters from local my machine \n> This notebook is regarding observing the distributions of clusters for the 14 features selected from 28 features, to identify what all features are redundant and can be removed.\n\n> For how these 14 features were selected you can refer to my previous [kernel](https://www.kaggle.com/code/ashaykatrojwar/feature-selection-iqr-outliers-eda)\n\n> This data is from my current submission file which achieved 0.81723 lb score for this I trained model on my local machine. ","metadata":{}},{"cell_type":"code","source":"clus=pd.read_csv(\"../input/1aftercob/1aftercob.csv\",index_col='Unnamed: 0')\ndata=pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv',index_col=[0])\nbest_feats=['f_07','f_08', 'f_09', 'f_10','f_11', 'f_12', 'f_13', 'f_22','f_23', 'f_24', 'f_25','f_26','f_27', 'f_28']","metadata":{"execution":{"iopub.status.busy":"2022-07-25T12:40:40.038274Z","iopub.execute_input":"2022-07-25T12:40:40.039087Z","iopub.status.idle":"2022-07-25T12:40:41.430073Z","shell.execute_reply.started":"2022-07-25T12:40:40.039051Z","shell.execute_reply":"2022-07-25T12:40:41.429178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Grouping all clusters and performing mean of the features to get value of cluster means for best features.","metadata":{}},{"cell_type":"code","source":"# adding clusters to data\ndata[\"Clusters\"]=clus.Predicted\n#grouping by data on clusters\nc=np.array(data[best_feats+['Clusters']].groupby('Clusters').mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-25T12:55:43.270619Z","iopub.execute_input":"2022-07-25T12:55:43.270999Z","iopub.status.idle":"2022-07-25T12:55:43.298296Z","shell.execute_reply.started":"2022-07-25T12:55:43.270969Z","shell.execute_reply":"2022-07-25T12:55:43.297407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# The cluster means plot\n1. It can be observed from below scatter plot that features `f_24` and `f_28` have 0 mean and clusters are not distinguishable, but after dropping these features cause the lb score to decrease from **0.8 to 0.6**.\n\n2. While the feature distribution plots for features `f_24` and `f_28` tell the same thing except the distributions are visually different for different clusters.\n\n3. On the other hand including all features i.e all **28 features** didn't affect the lb score by much.\n\n> It appears that integer features are more easily classifying the clusters, while the continuous features are the one having a hard time 😂 ","metadata":{}},{"cell_type":"code","source":"plt.style.use('ggplot')\nplt.figure(figsize=(20,5))\nfor i in range(np.array(c).shape[0]):\n    plt.scatter(np.arange(14), np.array(c)[i])\nplt.xticks(ticks=np.arange(14),labels=best_feats)\nplt.xlabel('Features')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T12:40:46.680934Z","iopub.execute_input":"2022-07-25T12:40:46.681341Z","iopub.status.idle":"2022-07-25T12:40:47.029762Z","shell.execute_reply.started":"2022-07-25T12:40:46.681307Z","shell.execute_reply":"2022-07-25T12:40:47.028545Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig=plt.figure(figsize=(20,22))\nfor i, f in enumerate(best_feats):\n    # New subplot\n    plt.subplot(5,3,i+1)\n    sns.kdeplot(data=data,x=data[f],hue=data['Clusters'],palette=sns.color_palette(\"cool\", as_cmap=True),fill=True)\n    plt.title(f'Feature: {f}')\n    plt.xlabel('')\nfig.tight_layout()  \nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-25T13:00:07.067421Z","iopub.execute_input":"2022-07-25T13:00:07.067839Z","iopub.status.idle":"2022-07-25T13:00:20.413114Z","shell.execute_reply.started":"2022-07-25T13:00:07.067804Z","shell.execute_reply":"2022-07-25T13:00:20.411625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I am carrying out further experiments, I'll keep updating this kernel :)","metadata":{}}]}