{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, PowerTransformer, RobustScaler\nfrom sklearn.cluster import KMeans\nfrom sklearn.mixture import GaussianMixture\nimport numpy as np","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-15T17:05:59.684518Z","iopub.execute_input":"2022-07-15T17:05:59.684968Z","iopub.status.idle":"2022-07-15T17:05:59.691823Z","shell.execute_reply.started":"2022-07-15T17:05:59.684931Z","shell.execute_reply":"2022-07-15T17:05:59.690359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv', index_col='id')\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:05:59.697403Z","iopub.execute_input":"2022-07-15T17:05:59.697779Z","iopub.status.idle":"2022-07-15T17:06:00.492674Z","shell.execute_reply.started":"2022-07-15T17:05:59.697736Z","shell.execute_reply":"2022-07-15T17:06:00.491537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:06:00.494702Z","iopub.execute_input":"2022-07-15T17:06:00.495018Z","iopub.status.idle":"2022-07-15T17:06:00.516162Z","shell.execute_reply.started":"2022-07-15T17:06:00.494987Z","shell.execute_reply":"2022-07-15T17:06:00.515039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:06:00.517317Z","iopub.execute_input":"2022-07-15T17:06:00.517767Z","iopub.status.idle":"2022-07-15T17:06:00.717084Z","shell.execute_reply.started":"2022-07-15T17:06:00.517720Z","shell.execute_reply":"2022-07-15T17:06:00.715902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(18,14))\nplt.suptitle('Columns Distribution', fontsize=16)\n\nfor i, col in enumerate(data.columns):\n    ax = plt.subplot(5, 6, i+1)\n    sns.histplot(data[col], ax=ax)\n    plt.ylabel(None)\n    plt.xlabel(None)\n    plt.title(col)\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:06:00.718786Z","iopub.execute_input":"2022-07-15T17:06:00.719090Z","iopub.status.idle":"2022-07-15T17:06:14.038687Z","shell.execute_reply.started":"2022-07-15T17:06:00.719062Z","shell.execute_reply":"2022-07-15T17:06:14.037512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* f_01 -> f_06, f_14 -> f_21 have the same distribution\n* f_07 -> f_13 have the same distribution\n* f_22 -> f_28 have the same distribution","metadata":{}},{"cell_type":"code","source":"subset_1 = ['f_{:02d}'.format(i) for i in range(1, 7)] + ['f_{:02d}'.format(i) for i in range(14, 22)]\nsubset_2 = ['f_{:02d}'.format(i) for i in range(7, 14)]\nsubset_3 = ['f_{:02d}'.format(i) for i in range(22, 29)]","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:06:14.040008Z","iopub.execute_input":"2022-07-15T17:06:14.040370Z","iopub.status.idle":"2022-07-15T17:06:14.047136Z","shell.execute_reply.started":"2022-07-15T17:06:14.040336Z","shell.execute_reply":"2022-07-15T17:06:14.045899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(data[subset_1], plot_kws={\"s\": 3})","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:06:14.048305Z","iopub.execute_input":"2022-07-15T17:06:14.049562Z","iopub.status.idle":"2022-07-15T17:07:15.275258Z","shell.execute_reply.started":"2022-07-15T17:06:14.049524Z","shell.execute_reply":"2022-07-15T17:07:15.273480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(data[subset_2], plot_kws={\"s\": 3})","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:07:15.277198Z","iopub.execute_input":"2022-07-15T17:07:15.277608Z","iopub.status.idle":"2022-07-15T17:07:32.122947Z","shell.execute_reply.started":"2022-07-15T17:07:15.277571Z","shell.execute_reply":"2022-07-15T17:07:32.121384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(data[subset_3], plot_kws={\"s\": 3})","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:07:32.124854Z","iopub.execute_input":"2022-07-15T17:07:32.125234Z","iopub.status.idle":"2022-07-15T17:07:49.330899Z","shell.execute_reply.started":"2022-07-15T17:07:32.125197Z","shell.execute_reply":"2022-07-15T17:07:49.329679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Transformation","metadata":{}},{"cell_type":"markdown","source":"Features are close to normlal distribution, although there are two things to notice:\n* Subset 2 features are left skewed\n* Features are not standarized  \nCommon transformations that handels left skewed data are square root, cube root, log, and general power transformation. Standard scaler could be used to standarize the data.  \nVarious transformations are tested out, anyway PowerTransformer is a the proper scaler beacause it makes the data more Gaussian-like and by default, standarizes the transformed data.","metadata":{}},{"cell_type":"code","source":"scalers = {\"standard\": StandardScaler(),\n          \"normal\": MinMaxScaler(),\n          \"power transformer\": PowerTransformer(),\n          \"robust scaler\": RobustScaler()}\ntransformed_data = data.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:07:49.332517Z","iopub.execute_input":"2022-07-15T17:07:49.332851Z","iopub.status.idle":"2022-07-15T17:07:49.346262Z","shell.execute_reply.started":"2022-07-15T17:07:49.332819Z","shell.execute_reply":"2022-07-15T17:07:49.345296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler_name = \"power transformer\"\nif scaler_name in scalers.keys():\n    scaler = scalers[scaler_name]\n    transformed_data = pd.DataFrame(scaler.fit_transform(transformed_data), index=data.index, columns=data.columns)\nelse:\n    None","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:07:49.349969Z","iopub.execute_input":"2022-07-15T17:07:49.350637Z","iopub.status.idle":"2022-07-15T17:07:52.938867Z","shell.execute_reply.started":"2022-07-15T17:07:49.350597Z","shell.execute_reply":"2022-07-15T17:07:52.937754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(18,14))\nplt.suptitle('Columns Distribution, ({} scaling)'.format(scaler_name), fontsize=16)\n\nfor i, col in enumerate(data.columns):\n    ax = plt.subplot(5, 6, i+1)\n    sns.histplot(transformed_data[col], ax=ax)\n    plt.ylabel(None)\n    plt.xlabel(None)\n    plt.title(col)\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:07:52.940347Z","iopub.execute_input":"2022-07-15T17:07:52.940983Z","iopub.status.idle":"2022-07-15T17:08:07.428132Z","shell.execute_reply.started":"2022-07-15T17:07:52.940937Z","shell.execute_reply":"2022-07-15T17:08:07.427320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Clustering Evaluation","metadata":{}},{"cell_type":"markdown","source":"There are various metrics to evaluate a clustering task, some evaluation merics needs a ground truth labels and other metrics does not take ground truth labels in account. since ground truth labels are not provided on this competetion here are some of the metrics that can be used:  ","metadata":{}},{"cell_type":"markdown","source":"* **Silhouette Score**  \nMeasures the separation distance between the clusters formed by the clustring algorithm by calculating average silhouette coefficient that is defined as the mean distance of the intra-cluster and nearest cluster for all the samples.  \nAverage silhouette coefficient ranges from -1 to 1. **The Higher the score, the better separation between the clusters**.  \n\n$$s = \\frac{b - a}{max(a, b)}$$  \n_s: silhouette coefficient for a single sample_  \n_a: mean distance between a sample and all other points in the same class_  \n_b: mean distance between a sample and all other points in the next nearest cluster_  ","metadata":{}},{"cell_type":"markdown","source":"* **Calinski-Harabasz Index**  \nA measure of how similar a sample is to its own cluster compared to other clusters. defined as the ratio of the sum of between-clusters dispersion and of inter-cluster dispersion for all clusters. **Higher score relates to a model with better defined clusters.**  \nMathimatical formulation is a bit complex, refer to it from [here](https://scikit-learn.org/stable/modules/clustering.html#id29).","metadata":{}},{"cell_type":"markdown","source":"* **Davies Bouldin index**  \nThe index signifies the average similarity between clusters.  \nMinimum possible value is 0. **The lower the score, the better separation between the clusters**  \n\n$$D = \\frac{1}{k} \\sum_{i=1}^k \\max_{j \\neq i} \\frac{s_i + s_j}{d_{ij}}$$\n\n$k$: _Number of clusters_  \n$i$: _Outer cluster number_  \n$j$: _Inner cluster number where_ $j \\neq i$  \n$s_x$: _The average distance between each point of cluster $x$ and the centroid of that cluster, known as the within cluster scatter (has to be as low as possible)_  \n$d_{ij}$: _The distance between cluster centroids $i$ and $j$ (has to be as large as possible)_  ","metadata":{}},{"cell_type":"markdown","source":"* **Inertia**  \nInertia vlue gives an indication of how coherent different clusters are, it is calculated as the sum of distances of all the points within a cluster from the centroid of that cluster  \n$$I_k = \\sum_{i=1}^N(x_i - c_k)^2 $$\n$I_k$: _inertia value for a cluster_ $k$  \n$N$: _number of samples_  \n$x_i$: _sample point $i$_  \n$c_k$: _center of cluster k_   \n\nPlotting inertia values for different values of k makes a curve like this  \n![](https://miro.medium.com/max/880/1*xOGY4uu6ng7E8lPLP-onWw.png)  \nIncrasing the number of clusters continuously decreases the inertia value, but the optimal number of clusters is the \"elbow\" point which is marked with the arrow","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import silhouette_score, calinski_harabasz_score, davies_bouldin_score\neval_metrics = {\"silhouette_score\": silhouette_score,\n               \"calinski_harabasz_score\": calinski_harabasz_score,\n               \"davies_bouldin_score\": davies_bouldin_score}","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:08:07.429348Z","iopub.execute_input":"2022-07-15T17:08:07.429800Z","iopub.status.idle":"2022-07-15T17:08:07.434749Z","shell.execute_reply.started":"2022-07-15T17:08:07.429770Z","shell.execute_reply":"2022-07-15T17:08:07.433588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KMeans Modeling","metadata":{}},{"cell_type":"code","source":"clusters_range = range(2, 11)\nn_cols = 3\nn_rows = len(clusters_range) // n_cols + (len(clusters_range) % n_cols > 0)\nplt.figure(figsize=(10, 10))\nplt.suptitle(\"KMeans Clusters\", fontsize=16)\n\nKMeans_predictions = dict()\nKMeans_eval = pd.DataFrame()\nfor i, k in enumerate(clusters_range):\n    model = KMeans(n_clusters=k)\n    model.fit(transformed_data)\n    KMeans_predictions[k] = model.labels_\n    \n    submission = pd.DataFrame({'Id': data.index, 'Predicted': model.labels_})\n    submission.to_csv('KMeans_{}_{}_scaler.csv'.format(k, scaler_name), index=False)\n    \n    KMeans_eval.loc[k, 'inertia'] = model.inertia_\n    for metric in eval_metrics:\n        KMeans_eval.loc[k, metric] = eval_metrics[metric](transformed_data, model.labels_)\n    \n    data_2d = PCA(n_components=2).fit_transform(transformed_data)\n    data_2d_df = pd.DataFrame(data_2d, columns=['C1', 'C2'])\n    ax = plt.subplot(n_rows, n_cols, i+1)\n    plt.scatter(data_2d_df['C1'], data_2d_df['C2'], c=model.labels_, s=3)\n    plt.title(\"K = {}\".format(k))\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:08:07.436425Z","iopub.execute_input":"2022-07-15T17:08:07.436761Z","iopub.status.idle":"2022-07-15T17:24:32.174656Z","shell.execute_reply.started":"2022-07-15T17:08:07.436730Z","shell.execute_reply":"2022-07-15T17:24:32.173492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_cols = 2\nn_rows = len(KMeans_eval.columns) // n_cols + (len(KMeans_eval.columns) % n_cols > 0)\nplt.figure(figsize=(10, 10))\nplt.suptitle(\"KMeans Evaluation\", fontsize=16)\n\nfor i, col in enumerate(KMeans_eval.columns):\n    ax = plt.subplot(n_rows, n_cols, i+1)\n    plt.plot([*clusters_range], KMeans_eval[col])\n    plt.title(col)\n    plt.xlabel('k')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:24:32.176048Z","iopub.execute_input":"2022-07-15T17:24:32.176381Z","iopub.status.idle":"2022-07-15T17:24:32.921303Z","shell.execute_reply.started":"2022-07-15T17:24:32.176350Z","shell.execute_reply":"2022-07-15T17:24:32.920099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Clusteres Pair-plots","metadata":{}},{"cell_type":"code","source":"def clusters_pair_plot(suptitle, subset_features, data, cluster_labels):\n    plt.figure(figsize=(18, 18))\n    plt.suptitle(suptitle, fontsize=16)\n\n    for i, y in enumerate(subset_features):\n        for j, x in enumerate(subset_features):\n            ax = plt.subplot(len(subset_features), len(subset_features), i * len(subset_features) + j + 1)\n\n            if i != j:\n                plt.scatter(data[x], data[y], c=cluster_labels, s=1)\n\n            plt.xticks([])\n            plt.yticks([])\n            if j == 0:\n                plt.ylabel(y)\n            if i == len(subset_features) - 1:\n                plt.xlabel(x)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:24:32.922800Z","iopub.execute_input":"2022-07-15T17:24:32.923118Z","iopub.status.idle":"2022-07-15T17:24:32.930880Z","shell.execute_reply.started":"2022-07-15T17:24:32.923082Z","shell.execute_reply":"2022-07-15T17:24:32.929740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusters_pair_plot(suptitle=\"KMeans Clusters Pair-plot (K=6)\\nSubset 1\",\n                   subset_features=subset_1, data=data, cluster_labels=KMeans_predictions[6])","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:24:32.932676Z","iopub.execute_input":"2022-07-15T17:24:32.933121Z","iopub.status.idle":"2022-07-15T17:27:51.571386Z","shell.execute_reply.started":"2022-07-15T17:24:32.933067Z","shell.execute_reply":"2022-07-15T17:27:51.569958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclusters_pair_plot(suptitle=\"KMeans Clusters Pair-plot (K=6)\\nSubset 2\",\n                   subset_features=subset_2, data=data, cluster_labels=KMeans_predictions[6])","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:27:51.573171Z","iopub.execute_input":"2022-07-15T17:27:51.573589Z","iopub.status.idle":"2022-07-15T17:28:38.093642Z","shell.execute_reply.started":"2022-07-15T17:27:51.573551Z","shell.execute_reply":"2022-07-15T17:28:38.092384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusters_pair_plot(suptitle=\"KMeans Clusters Pair-plot (K=6)\\nSubset 3\",\n                   subset_features=subset_3, data=data, cluster_labels=KMeans_predictions[6])","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:28:38.095240Z","iopub.execute_input":"2022-07-15T17:28:38.095937Z","iopub.status.idle":"2022-07-15T17:29:24.544388Z","shell.execute_reply.started":"2022-07-15T17:28:38.095884Z","shell.execute_reply":"2022-07-15T17:29:24.542547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the plots:\n* Subset 1 features are not good in separating clusteres \n* Subset 2 features are good in separating clusteres \n* Some of subset 3 features are good in separating clusteres ","metadata":{}},{"cell_type":"markdown","source":"# GMM Modeling","metadata":{}},{"cell_type":"code","source":"n_components_range = range(2, 11)\nn_cols = 3\nn_rows = len(n_components_range) // n_cols + (len(n_components_range) % n_cols > 0)\nplt.figure(figsize=(10, 10))\nplt.suptitle(\"Gaussian Mixture Clusters\", fontsize=16)\n\nGMM_predictions = dict()\nGMM_models = dict()\nGMM_eval = pd.DataFrame()\nfor i, k in enumerate(n_components_range):\n    GMM_models[k] = GaussianMixture(n_components=k, random_state=10)\n    GMM_models[k].fit(transformed_data)\n    GMM_predictions[k] = GMM_models[k].predict(transformed_data)\n    \n    GMM_eval.loc[k, \"AIC\"] = GMM_models[k].aic(transformed_data)\n    GMM_eval.loc[k, \"BIC\"] = GMM_models[k].bic(transformed_data)\n    for metric in eval_metrics:\n        GMM_eval.loc[k, metric] = eval_metrics[metric](transformed_data, GMM_predictions[k])\n    \n    data_2d = PCA(n_components=2).fit_transform(transformed_data)\n    data_2d_df = pd.DataFrame(data_2d, columns=['C1', 'C2'])\n    ax = plt.subplot(n_rows, n_cols, i+1)\n    plt.scatter(data_2d_df['C1'], data_2d_df['C2'], c=GMM_predictions[k], s=3)\n    plt.title(\"n = {}\".format(k))\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:29:24.545902Z","iopub.execute_input":"2022-07-15T17:29:24.546240Z","iopub.status.idle":"2022-07-15T17:47:15.365624Z","shell.execute_reply.started":"2022-07-15T17:29:24.546211Z","shell.execute_reply":"2022-07-15T17:47:15.364689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_cols = 3\nn_rows = len(GMM_eval.columns) // n_cols + (len(GMM_eval.columns) % n_cols > 0)\nplt.figure(figsize=(10, 10))\nplt.suptitle(\"GMM Evaluation\", fontsize=16)\n\nfor i, col in enumerate(GMM_eval.columns):\n    ax = plt.subplot(n_rows, n_cols, i+1)\n    plt.plot([*n_components_range], GMM_eval[col])\n    plt.title(col)\n    plt.xlabel('k')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:47:15.366804Z","iopub.execute_input":"2022-07-15T17:47:15.367797Z","iopub.status.idle":"2022-07-15T17:47:16.235038Z","shell.execute_reply.started":"2022-07-15T17:47:15.367757Z","shell.execute_reply":"2022-07-15T17:47:16.233931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"model with lower BIC is generally preferred","metadata":{}},{"cell_type":"code","source":"n_components = 8\nsubmission = pd.DataFrame({'Id': data.index, 'Predicted': GMM_predictions[n_components]})\nsubmission.to_csv('GMM_{}_components_{}_scaler.csv'.format(n_components, scaler_name), index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:47:16.236738Z","iopub.execute_input":"2022-07-15T17:47:16.237119Z","iopub.status.idle":"2022-07-15T17:47:16.337189Z","shell.execute_reply.started":"2022-07-15T17:47:16.237085Z","shell.execute_reply":"2022-07-15T17:47:16.336057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Investigating feature importance","metadata":{}},{"cell_type":"markdown","source":"Noisy features that does not introduce valuable information to the clustering algorithm could lead to bad clusterung. Knowing the important freatures that the algorithm considered in clustering is not possible within GaussianMixture class, one work around is to build a classifier that calculates features importance.","metadata":{}},{"cell_type":"markdown","source":"### Building Clusters Classifier","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBClassifier\n\nmodel = XGBClassifier(random_state=10, objective=\"multi:softmax\")\nmodel.fit(transformed_data, GMM_predictions[8])","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:47:16.338710Z","iopub.execute_input":"2022-07-15T17:47:16.339154Z","iopub.status.idle":"2022-07-15T17:50:59.779236Z","shell.execute_reply.started":"2022-07-15T17:47:16.339118Z","shell.execute_reply":"2022-07-15T17:50:59.778034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 5))\nplt.title(\"Feature Importances (GMM Clustering)\")\nplt.bar(transformed_data.columns, model.feature_importances_)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:50:59.780526Z","iopub.execute_input":"2022-07-15T17:50:59.780840Z","iopub.status.idle":"2022-07-15T17:51:00.122206Z","shell.execute_reply.started":"2022-07-15T17:50:59.780812Z","shell.execute_reply":"2022-07-15T17:51:00.120830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Clusteres Pair-plots","metadata":{}},{"cell_type":"code","source":"clusters_pair_plot(suptitle=\"GMM Clusters Pair-plot (n=8)\\nSubset 1\",\n                   subset_features=subset_1, data=data, cluster_labels=GMM_predictions[8])","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:51:00.123783Z","iopub.execute_input":"2022-07-15T17:51:00.124103Z","iopub.status.idle":"2022-07-15T17:54:19.560866Z","shell.execute_reply.started":"2022-07-15T17:51:00.124074Z","shell.execute_reply":"2022-07-15T17:54:19.559781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusters_pair_plot(suptitle=\"GMM Clusters Pair-plot (n=8)\\nSubset 2\",\n                   subset_features=subset_2, data=data, cluster_labels=GMM_predictions[8])","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:54:19.562378Z","iopub.execute_input":"2022-07-15T17:54:19.563240Z","iopub.status.idle":"2022-07-15T17:55:05.246213Z","shell.execute_reply.started":"2022-07-15T17:54:19.563198Z","shell.execute_reply":"2022-07-15T17:55:05.245134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusters_pair_plot(suptitle=\"GMM Clusters Pair-plot (n=8)\\nSubset 3\",\n                   subset_features=subset_3, data=data, cluster_labels=GMM_predictions[8])","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:55:05.247683Z","iopub.execute_input":"2022-07-15T17:55:05.248078Z","iopub.status.idle":"2022-07-15T17:55:51.179396Z","shell.execute_reply.started":"2022-07-15T17:55:05.248041Z","shell.execute_reply":"2022-07-15T17:55:51.178465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the plots:\n* Subset 1 features are not good in separating clusteres \n* Subset 2 features are good in separating clusteres \n* Subset 3 features are good in separating clusteres ","metadata":{}},{"cell_type":"markdown","source":"### Selecting Important Features","metadata":{}},{"cell_type":"code","source":"selected_data = transformed_data[subset_2 + subset_3]\n\nn_components = 8\nGMM_model_selected_data = GaussianMixture(n_components=n_components, random_state=10)\nGMM_model_selected_data.fit(selected_data)\nGMM_predictions_selected_data = GMM_model_selected_data.predict(selected_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:55:51.180575Z","iopub.execute_input":"2022-07-15T17:55:51.181331Z","iopub.status.idle":"2022-07-15T17:56:01.746776Z","shell.execute_reply.started":"2022-07-15T17:55:51.181273Z","shell.execute_reply":"2022-07-15T17:56:01.745464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GMM_selected_data_eval = pd.DataFrame()\nGMM_selected_data_eval.loc[n_components, \"AIC\"] = GMM_model_selected_data.aic(selected_data)\nGMM_selected_data_eval.loc[n_components, \"BIC\"] = GMM_model_selected_data.bic(selected_data)\nfor metric in eval_metrics:\n    GMM_selected_data_eval.loc[n_components, metric] = eval_metrics[metric](selected_data, GMM_predictions_selected_data)\nGMM_selected_data_eval.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:58:04.012895Z","iopub.execute_input":"2022-07-15T17:58:04.013619Z","iopub.status.idle":"2022-07-15T17:59:39.984990Z","shell.execute_reply.started":"2022-07-15T17:58:04.013580Z","shell.execute_reply":"2022-07-15T17:59:39.983682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'Id': data.index, 'Predicted': GMM_predictions_selected_data})\nsubmission.to_csv('Subsets_2-3_GMM_{}_components_{}_scaler.csv'.format(n_components, scaler_name), index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T17:57:37.572510Z","iopub.execute_input":"2022-07-15T17:57:37.572936Z","iopub.status.idle":"2022-07-15T17:57:37.730683Z","shell.execute_reply.started":"2022-07-15T17:57:37.572888Z","shell.execute_reply":"2022-07-15T17:57:37.729577Z"},"trusted":true},"execution_count":null,"outputs":[]}]}