{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nfrom time import time\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib.pylab as pylab\n\nfrom sklearn.metrics import silhouette_samples, silhouette_score,davies_bouldin_score\nfrom sklearn.cluster import KMeans,DBSCAN\nfrom sklearn.decomposition import PCA,IncrementalPCA,KernelPCA\nfrom sklearn import metrics\nfrom sklearn.pipeline import Pipeline, make_pipeline\nfrom sklearn.preprocessing import StandardScaler,RobustScaler,MinMaxScaler,PowerTransformer\n\n\nfrom yellowbrick.cluster import KElbowVisualizer\nfrom scipy.stats import mode\n\nfrom sklearn.mixture import GaussianMixture,BayesianGaussianMixture","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-09T10:31:14.022222Z","iopub.execute_input":"2022-07-09T10:31:14.022728Z","iopub.status.idle":"2022-07-09T10:31:14.030859Z","shell.execute_reply.started":"2022-07-09T10:31:14.022697Z","shell.execute_reply":"2022-07-09T10:31:14.029431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read Data\ndata = pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\",index_col='id')\nsubmission = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:31:14.032628Z","iopub.execute_input":"2022-07-09T10:31:14.033063Z","iopub.status.idle":"2022-07-09T10:31:15.445319Z","shell.execute_reply.started":"2022-07-09T10:31:14.032993Z","shell.execute_reply":"2022-07-09T10:31:15.443858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:31:17.543293Z","iopub.execute_input":"2022-07-09T10:31:17.543695Z","iopub.status.idle":"2022-07-09T10:31:17.591841Z","shell.execute_reply.started":"2022-07-09T10:31:17.543662Z","shell.execute_reply":"2022-07-09T10:31:17.590533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T09:09:20.657285Z","iopub.execute_input":"2022-07-04T09:09:20.657701Z","iopub.status.idle":"2022-07-04T09:09:20.685150Z","shell.execute_reply.started":"2022-07-04T09:09:20.657668Z","shell.execute_reply":"2022-07-04T09:09:20.683950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- NO missing values","metadata":{"execution":{"iopub.status.busy":"2022-07-02T12:02:17.606008Z","iopub.execute_input":"2022-07-02T12:02:17.606444Z","iopub.status.idle":"2022-07-02T12:02:17.613519Z","shell.execute_reply.started":"2022-07-02T12:02:17.606409Z","shell.execute_reply":"2022-07-02T12:02:17.611922Z"}}},{"cell_type":"code","source":"\nax = data.dtypes.value_counts().plot.barh(color = ['green','orange'],rot = 0,figsize = (8,3))\nax.set_title('Distribution of Column dtypes')\nax.set_xlabel('Number of columns')\nax.set_ylabel('Column dtype')\n\nfor i in ax.patches:\n    ax.text(i.get_width()+.3, i.get_y()+.38, str(i.get_width()), fontsize=12,\ncolor='dimgrey')\n    \nax.invert_yaxis()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:14:00.584689Z","iopub.execute_input":"2022-07-08T07:14:00.585084Z","iopub.status.idle":"2022-07-08T07:14:00.856647Z","shell.execute_reply.started":"2022-07-08T07:14:00.585052Z","shell.execute_reply":"2022-07-08T07:14:00.855415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Insight**\n- There are only numeric feature\n- There are 22 Continuous and 7 categorical variables","metadata":{}},{"cell_type":"code","source":"cat_cols = data.select_dtypes('int').columns\ncont_cols = data.select_dtypes(exclude = 'int').columns\n\nprint(f'Categorical Variables: \\n{cat_cols} \\n\\nContinuous Variables: \\n{cont_cols}')","metadata":{"execution":{"iopub.status.busy":"2022-07-04T10:48:13.371915Z","iopub.execute_input":"2022-07-04T10:48:13.372321Z","iopub.status.idle":"2022-07-04T10:48:13.393127Z","shell.execute_reply.started":"2022-07-04T10:48:13.372290Z","shell.execute_reply":"2022-07-04T10:48:13.392122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[cont_cols].describe().T","metadata":{"execution":{"iopub.status.busy":"2022-07-04T09:09:30.579987Z","iopub.execute_input":"2022-07-04T09:09:30.580346Z","iopub.status.idle":"2022-07-04T09:09:30.765252Z","shell.execute_reply.started":"2022-07-04T09:09:30.580318Z","shell.execute_reply":"2022-07-04T09:09:30.764003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[cat_cols].describe().T","metadata":{"execution":{"iopub.status.busy":"2022-07-04T09:09:35.021463Z","iopub.execute_input":"2022-07-04T09:09:35.021915Z","iopub.status.idle":"2022-07-04T09:09:35.082590Z","shell.execute_reply.started":"2022-07-04T09:09:35.021856Z","shell.execute_reply":"2022-07-04T09:09:35.081233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(30, 30))\nfor i, col in enumerate(cont_cols):\n    ax = plt.subplot(int(np.ceil(len(cont_cols)/5)),int(np.ceil(len(cont_cols)/5)),i+1)\n    b = sns.histplot(data = data, ax= ax, x=col, kde = True, color = 'red')\n    ax.set_xlabel(b.get_xlabel(),size = 30)\n    ax.set_ylabel(b.get_ylabel(),size = 30)\n\nplt.suptitle('Data distribution of Continuous features',y = 1,fontsize = 30)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T12:20:54.895355Z","iopub.execute_input":"2022-07-03T12:20:54.896216Z","iopub.status.idle":"2022-07-03T12:21:15.940916Z","shell.execute_reply.started":"2022-07-03T12:20:54.896176Z","shell.execute_reply":"2022-07-03T12:21:15.940077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- While the mean and STD are close to 0 and 1, the data ranges almost equally on both sides","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(30, 30),dpi= 120)\nfor i, col in enumerate(cat_cols):\n    ax = plt.subplot(int(len(cat_cols)/2),int(len(cat_cols)/2),i+1)\n    b = sns.histplot(data = data, ax= ax, x=col, kde = True, color = 'red')\n    ax.set_xlabel(b.get_xlabel(),size = 30)\n    ax.set_ylabel(b.get_ylabel(),size = 30)\n\nplt.suptitle('Data distribution of Categorical features',y = 1,fontsize = 30)\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T12:21:15.942594Z","iopub.execute_input":"2022-07-03T12:21:15.943499Z","iopub.status.idle":"2022-07-03T12:21:22.845886Z","shell.execute_reply.started":"2022-07-03T12:21:15.943462Z","shell.execute_reply":"2022-07-03T12:21:22.844667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Insight**\n- The data is left skewed\n- There might be outliers","metadata":{}},{"cell_type":"code","source":"data_scaled = pd.DataFrame(PowerTransformer().fit_transform(data),columns=data.columns)\n\ndata_scaled = pd.DataFrame(RobustScaler().fit_transform(data_scaled), columns=data_scaled.columns)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:31:37.840711Z","iopub.execute_input":"2022-07-09T10:31:37.841146Z","iopub.status.idle":"2022-07-09T10:31:41.808848Z","shell.execute_reply.started":"2022-07-09T10:31:37.841108Z","shell.execute_reply":"2022-07-09T10:31:41.807548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_scaled.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:31:41.810739Z","iopub.execute_input":"2022-07-09T10:31:41.811125Z","iopub.status.idle":"2022-07-09T10:31:42.038144Z","shell.execute_reply.started":"2022-07-09T10:31:41.811088Z","shell.execute_reply":"2022-07-09T10:31:42.036802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of cluster centers for KMeans and BisectingKMeans\nrandom_state = 0\nelbow_values = []\ncluster_labels = []","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:31:42.040657Z","iopub.execute_input":"2022-07-09T10:31:42.041136Z","iopub.status.idle":"2022-07-09T10:31:42.046349Z","shell.execute_reply.started":"2022-07-09T10:31:42.041089Z","shell.execute_reply":"2022-07-09T10:31:42.045225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Finding right component size for PCA","metadata":{}},{"cell_type":"code","source":"pca= PCA()\npca.fit(data)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T12:21:22.908289Z","iopub.execute_input":"2022-07-03T12:21:22.908787Z","iopub.status.idle":"2022-07-03T12:21:23.064633Z","shell.execute_reply.started":"2022-07-03T12:21:22.908743Z","shell.execute_reply":"2022-07-03T12:21:23.063392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Lets check the Variance - Component tradeoff for PCA","metadata":{}},{"cell_type":"code","source":"exp_var = pca.explained_variance_ratio_ * 100\ncum_exp_var = np.cumsum(exp_var)\n\nplt.bar(range(1, 30), exp_var, align='center',\n        label='Individual explained variance')\n\nplt.step(range(1, 30), cum_exp_var, where='mid',\n         label='Cumulative explained variance', color='red')\n\nplt.ylabel('Explained variance percentage')\nplt.xlabel('Principal component index')\nplt.axhline(y=95, color='r', linestyle='-')\nplt.text(0,90, '95% cut-off threshold', color = 'red', fontsize=12)\nplt.legend(loc='best')\nax.grid(axis='x')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T12:21:23.066342Z","iopub.execute_input":"2022-07-03T12:21:23.067515Z","iopub.status.idle":"2022-07-03T12:21:23.490087Z","shell.execute_reply.started":"2022-07-03T12:21:23.067460Z","shell.execute_reply":"2022-07-03T12:21:23.489220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(np.cumsum(pca.explained_variance_ratio_),marker='o')\nplt.axhline(y=0.95, color='r', linestyle='-')\nplt.text(0,0.90, '95% cut-off threshold', color = 'red', fontsize=12)\nplt.xlabel('number of components')\nplt.ylabel('cumulative explained variance')\nax.grid(axis='x')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T12:21:23.491277Z","iopub.execute_input":"2022-07-03T12:21:23.492258Z","iopub.status.idle":"2022-07-03T12:21:24.228882Z","shell.execute_reply.started":"2022-07-03T12:21:23.492224Z","shell.execute_reply":"2022-07-03T12:21:24.227774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(pca.explained_variance_,marker='o')\nplt.xlabel('number of components')\nplt.ylabel('Eigen values')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T12:21:24.230458Z","iopub.execute_input":"2022-07-03T12:21:24.230802Z","iopub.status.idle":"2022-07-03T12:21:24.480133Z","shell.execute_reply.started":"2022-07-03T12:21:24.230773Z","shell.execute_reply":"2022-07-03T12:21:24.479089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fit and Transform data based on identified component size","metadata":{}},{"cell_type":"code","source":"pca= PCA(n_components = 0.97,random_state = random_state)\ndata_scaled_pca = pca.fit_transform(data_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T12:21:24.481371Z","iopub.execute_input":"2022-07-03T12:21:24.482798Z","iopub.status.idle":"2022-07-03T12:21:24.617587Z","shell.execute_reply.started":"2022-07-03T12:21:24.482764Z","shell.execute_reply":"2022-07-03T12:21:24.616272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using non-PCA transformed dataset as it yielded better results\nX = data_scaled\nX.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:31:49.629228Z","iopub.execute_input":"2022-07-09T10:31:49.629611Z","iopub.status.idle":"2022-07-09T10:31:49.636834Z","shell.execute_reply.started":"2022-07-09T10:31:49.629581Z","shell.execute_reply":"2022-07-09T10:31:49.636019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X = X[['f_07','f_08','f_09','f_10','f_11','f_12','f_13','f_22','f_23','f_24','f_25','f_26','f_27','f_28']]","metadata":{"execution":{"iopub.status.busy":"2022-07-07T11:33:58.626026Z","iopub.execute_input":"2022-07-07T11:33:58.627111Z","iopub.status.idle":"2022-07-07T11:33:58.632711Z","shell.execute_reply.started":"2022-07-07T11:33:58.627061Z","shell.execute_reply":"2022-07-07T11:33:58.631466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Find Cluster size","metadata":{}},{"cell_type":"code","source":"clusterer = KMeans(random_state=random_state)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:21:16.072396Z","iopub.execute_input":"2022-07-04T15:21:16.072732Z","iopub.status.idle":"2022-07-04T15:21:16.078484Z","shell.execute_reply.started":"2022-07-04T15:21:16.072703Z","shell.execute_reply":"2022-07-04T15:21:16.077199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Distortion Score for K means\n\nvisualizer = KElbowVisualizer(clusterer, k=(3,30), timings= True)\nvisualizer.fit(X)\nelbow_values.append(visualizer.elbow_value_)\nvisualizer.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T15:21:16.079768Z","iopub.execute_input":"2022-07-04T15:21:16.080082Z","iopub.status.idle":"2022-07-04T15:26:31.141983Z","shell.execute_reply.started":"2022-07-04T15:21:16.080055Z","shell.execute_reply":"2022-07-04T15:26:31.140935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Silhouette Score for K means\n\nvisualizer = KElbowVisualizer(clusterer, k=(3,30),metric='silhouette', timings= True)\nvisualizer.fit(X)\nelbow_values.append(visualizer.elbow_value_)\nvisualizer.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T12:29:14.613153Z","iopub.execute_input":"2022-07-03T12:29:14.613598Z","iopub.status.idle":"2022-07-03T13:23:15.177538Z","shell.execute_reply.started":"2022-07-03T12:29:14.613556Z","shell.execute_reply":"2022-07-03T13:23:15.175739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calinski Harabasz Score for K means\n\nvisualizer = KElbowVisualizer(clusterer, k=(3,30),metric='calinski_harabasz', timings= True)\nvisualizer.fit(X)\nelbow_values.append(visualizer.elbow_value_)\nvisualizer.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T13:23:15.181053Z","iopub.execute_input":"2022-07-03T13:23:15.181615Z","iopub.status.idle":"2022-07-03T13:30:56.612577Z","shell.execute_reply.started":"2022-07-03T13:23:15.181568Z","shell.execute_reply":"2022-07-03T13:30:56.611341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Davies bouldin score for Kmeans\n\ndef get_kmeans_score(data, center):\n    #instantiate kmeans\n    kmeans = KMeans(n_clusters=center,random_state = 42)\n    # Then fit the model to your data using the fit method\n    labels = kmeans.fit_predict(data)\n    \n    # Calculate Davies Bouldin score\n    score = davies_bouldin_score(data, labels)\n    \n    return score\n\n\nscores = []\ncenters = list(range(3,30))\nfor center in centers:\n    scores.append(get_kmeans_score(X, center))\n\ncenter_scores = dict(zip(centers,scores))\nelbow_values.append(min(center_scores, key=center_scores.get))\nplt.plot(centers, scores, linestyle='--', marker='o', color='b');\nplt.xlabel('K');\nplt.ylabel('Davies Bouldin score');\nplt.title('Davies Bouldin score vs. K');\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T13:30:56.614440Z","iopub.execute_input":"2022-07-03T13:30:56.615110Z","iopub.status.idle":"2022-07-03T13:38:47.420076Z","shell.execute_reply.started":"2022-07-03T13:30:56.615066Z","shell.execute_reply":"2022-07-03T13:38:47.418940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fitting Gaussian Mixture","metadata":{}},{"cell_type":"code","source":"n_components = 8 #np.maximum(int(np.ceil(np.mean(elbow_values))),7)\n\n# Fit Gaussian Mixture\nprint('Fitting Gaussian Mixture..')\ngm = GaussianMixture(n_components = n_components,\n                         max_iter = 1000, n_init = 10, \n                     random_state = random_state\n                     )\n\ngm_labels = gm.fit_predict(X)\ncluster_labels = gm_labels","metadata":{"execution":{"iopub.status.busy":"2022-07-08T07:34:33.295656Z","iopub.execute_input":"2022-07-08T07:34:33.296091Z","iopub.status.idle":"2022-07-08T07:38:30.981368Z","shell.execute_reply.started":"2022-07-08T07:34:33.296053Z","shell.execute_reply":"2022-07-08T07:38:30.979948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_components = 9\n\n# Fit Bayesian Gaussian Mixture\nprint('Fitting Bayesian Gaussian Mixture..')\nbgm = BayesianGaussianMixture(n_components = n_components,\n               n_init =10,\n               max_iter = 1000,\n               random_state = random_state,\n             verbose = 1,\n             verbose_interval = 100)\n\nbgm_labels = bgm.fit_predict(X)\ncluster_labels = bgm_labels","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:31:54.053919Z","iopub.execute_input":"2022-07-09T10:31:54.055023Z","iopub.status.idle":"2022-07-09T10:50:26.170450Z","shell.execute_reply.started":"2022-07-09T10:31:54.054978Z","shell.execute_reply":"2022-07-09T10:50:26.168918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Submission","metadata":{}},{"cell_type":"code","source":"submission['Predicted'] = cluster_labels\nsubmission.to_csv('submission.csv',index=False)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:50:26.178156Z","iopub.execute_input":"2022-07-09T10:50:26.179971Z","iopub.status.idle":"2022-07-09T10:50:26.362121Z","shell.execute_reply.started":"2022-07-09T10:50:26.179896Z","shell.execute_reply":"2022-07-09T10:50:26.361003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.Predicted.value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:50:26.363592Z","iopub.execute_input":"2022-07-09T10:50:26.364336Z","iopub.status.idle":"2022-07-09T10:50:26.652542Z","shell.execute_reply.started":"2022-07-09T10:50:26.364289Z","shell.execute_reply":"2022-07-09T10:50:26.651198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Future thoughts:\n- Compare model performace of multiple clustering methods\n- Experiement based on other clustering methods (maybe try ensemble scoring)\n- Study each feature manually\n- Reading few papers for clustering and implement benchmark","metadata":{}}]}