{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Setup","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:00:37.695850Z","iopub.execute_input":"2022-07-14T08:00:37.700195Z","iopub.status.idle":"2022-07-14T08:00:37.709248Z","shell.execute_reply.started":"2022-07-14T08:00:37.700131Z","shell.execute_reply":"2022-07-14T08:00:37.708016Z"}}},{"cell_type":"code","source":"from IPython.display import clear_output\nfrom IPython.core.interactiveshell import InteractiveShell\nInteractiveShell.ast_node_interactivity = 'all'\n\nRS = 335566\n\nimport os\nfrom pathlib import Path\nimport datetime, time\nimport pickle\n\nimport pandas as pd\npd.options.display.max_columns = None\npd.options.display.max_colwidth = 999\npd.options.display.max_rows = 999\n\nimport numpy as np\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.cluster import KMeans\nfrom sklearn.mixture import GaussianMixture, BayesianGaussianMixture\nfrom sklearn.metrics import calinski_harabasz_score, davies_bouldin_score, silhouette_score\n\nfrom umap import UMAP\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler, Normalizer, PowerTransformer\n\nRS = 335577\ndata_dir = '/kaggle/input/tabular-playground-series-jul-2022/'\n# data_dir = 'data/'\n\ndf_data = pd.read_csv(f'{data_dir}data.csv', index_col='id')#.sample(1000).iloc[:, 6:16]\ncat_features = ['f_09', 'f_10', 'f_11', 'f_12', 'f_13']\ncont_features = ['f_22', 'f_23', 'f_24', 'f_25', 'f_26']\n\nselected_features = ['f_10', 'f_11', 'f_12', 'f_23', 'f_24', 'f_25']","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-14T10:50:57.052528Z","iopub.execute_input":"2022-07-14T10:50:57.053660Z","iopub.status.idle":"2022-07-14T10:51:05.879516Z","shell.execute_reply.started":"2022-07-14T10:50:57.053550Z","shell.execute_reply":"2022-07-14T10:51:05.878365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Functions","metadata":{}},{"cell_type":"code","source":"get_ax = lambda: plt.subplots(figsize=(20,10))[1]\n\ndef add_cluster_lbls(X, n_comp):\n    X = X.copy()\n    X_pt = PowerTransformer().fit_transform(X)\n    X['lbls'] = BayesianGaussianMixture(n_components=n_comp,covariance_type='full', random_state=RS).fit_predict(X_pt)\n    X['lbls'] = X['lbls'].astype('category')\n    return X\n\ndef create_umap_df(X_with_lbls, n_comp):\n    X = X_with_lbls.drop('lbls', axis=1)\n    X_ss = StandardScaler().fit_transform(X)\n    X_umap = UMAP(n_components=n_comp).fit_transform(X_ss)\n    df = pd.DataFrame(data=X_umap)\n    df['lbls'] = X_with_lbls.reset_index()['lbls']\n    df['lbls'] = df['lbls'].astype('category')\n    return df\n\numap_results = {}\n\ndef pairplot_selected_features(n_clusters):\n    df_plt = bgm_clusters[n_clusters][selected_features+['lbls']].sample(2000, random_state=RS)\n    _ = sns.pairplot(data=df_plt, hue='lbls', plot_kws={'s':20}).fig.suptitle(f'BGM {n_clusters} - pairplot selected features', y=1.02)\n    \ndef pairplot_umap(n_clusters, n_comp):\n    if (n_clusters, n_comp) in umap_results:\n        df_umap = umap_results[(n_clusters, n_comp)]\n    else:\n        df_umap = create_umap_df(bgm_clusters[n_clusters], n_comp).sample(2000, random_state=RS)\n        umap_results[(n_clusters, n_comp)] = df_umap\n        \n    _ = sns.pairplot(data=df_umap, hue='lbls', plot_kws={'s':20}).fig.suptitle(f'BGM {n_clusters} - UMAP {n_comp}', y=1.02)\n\ndef plot_umap_2(n_clusters):\n    if (n_clusters) in umap_results:\n        df_umap = umap_results[(n_clusters)]\n    else:\n        df_umap = create_umap_df(bgm_clusters[n_clusters], 2).sample(10000, random_state=RS)\n        umap_results[(n_clusters)] = df_umap\n\n    _ = sns.scatterplot(x=df_umap[0], y=df_umap[1], hue=df_umap['lbls'], ax=get_ax()).set(title=f'BGM {n_clusters} - UMAP 2')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:05.881582Z","iopub.execute_input":"2022-07-14T10:51:05.882148Z","iopub.status.idle":"2022-07-14T10:51:05.898777Z","shell.execute_reply.started":"2022-07-14T10:51:05.882102Z","shell.execute_reply":"2022-07-14T10:51:05.897777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create BGM clusters","metadata":{}},{"cell_type":"code","source":"%%time\ndf_ = df_data\n# df_ = df_data.sample(frac=0.5, random_state=RS).reset_index().copy()\n\nbgm_clusters = {}\nfor n_clusters in [2, 3, 5, 7]:\n    print(n_clusters)\n    bgm_clusters[n_clusters] = add_cluster_lbls(df_, n_clusters)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:05.900196Z","iopub.execute_input":"2022-07-14T10:51:05.900830Z","iopub.status.idle":"2022-07-14T10:51:06.384393Z","shell.execute_reply.started":"2022-07-14T10:51:05.900785Z","shell.execute_reply":"2022-07-14T10:51:06.383641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The clustering 'Pipeline' uses Power Transformer and Bayesian Gaussian Mixture, that with 7 clusters gives the public score of > 0.6.\nNow lets play with the number of clusters, and then visualise it in selected number of continuous and categorical features.\nThen let's reduce the number of dimensions using UMAP to see the clustering results","metadata":{}},{"cell_type":"code","source":"'selected_features', selected_features","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.386289Z","iopub.execute_input":"2022-07-14T10:51:06.386851Z","iopub.status.idle":"2022-07-14T10:51:06.395724Z","shell.execute_reply.started":"2022-07-14T10:51:06.386820Z","shell.execute_reply":"2022-07-14T10:51:06.394630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GBM - 2 Clusters","metadata":{}},{"cell_type":"code","source":"n_clusters=2","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.397038Z","iopub.execute_input":"2022-07-14T10:51:06.397321Z","iopub.status.idle":"2022-07-14T10:51:06.407943Z","shell.execute_reply.started":"2022-07-14T10:51:06.397296Z","shell.execute_reply":"2022-07-14T10:51:06.406991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"GBM with 2 clusters ploted on the original data. First 3 features are categorical, next 3 continuous. No clear separation and a lot of overlap.","metadata":{}},{"cell_type":"code","source":"pairplot_selected_features(n_clusters)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.409046Z","iopub.execute_input":"2022-07-14T10:51:06.409311Z","iopub.status.idle":"2022-07-14T10:51:06.467008Z","shell.execute_reply.started":"2022-07-14T10:51:06.409287Z","shell.execute_reply":"2022-07-14T10:51:06.463595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Using UMAP to reduce from 29 to 2 components. We can clearly see how the 2 clusters are separated, but there is a lot of overlap.","metadata":{}},{"cell_type":"code","source":"plot_umap_2(n_clusters)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.468427Z","iopub.status.idle":"2022-07-14T10:51:06.469433Z","shell.execute_reply.started":"2022-07-14T10:51:06.469135Z","shell.execute_reply":"2022-07-14T10:51:06.469161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Increasing the UMAP components to 3. Still clear seperation, but also overlap.","metadata":{}},{"cell_type":"code","source":"pairplot_umap(n_clusters, 3)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.471228Z","iopub.status.idle":"2022-07-14T10:51:06.472111Z","shell.execute_reply.started":"2022-07-14T10:51:06.471841Z","shell.execute_reply":"2022-07-14T10:51:06.471867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Increasing the UMAP components to 5. More overlap.","metadata":{}},{"cell_type":"code","source":"pairplot_umap(n_clusters, 5)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.473667Z","iopub.status.idle":"2022-07-14T10:51:06.474562Z","shell.execute_reply.started":"2022-07-14T10:51:06.474276Z","shell.execute_reply":"2022-07-14T10:51:06.474302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Increasing the UMAP components to 8. Again additional componenets show even mode overlap.","metadata":{}},{"cell_type":"code","source":"pairplot_umap(n_clusters, 8)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.475652Z","iopub.status.idle":"2022-07-14T10:51:06.476496Z","shell.execute_reply.started":"2022-07-14T10:51:06.476275Z","shell.execute_reply":"2022-07-14T10:51:06.476296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GBM - 3 Clusters","metadata":{}},{"cell_type":"code","source":"n_clusters=3","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.477720Z","iopub.status.idle":"2022-07-14T10:51:06.478096Z","shell.execute_reply.started":"2022-07-14T10:51:06.477913Z","shell.execute_reply":"2022-07-14T10:51:06.477929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"GBM with 3 clusters ploted on the original data. First 3 features are categorical, next 3 continuous. No clear separation and a lot of overlap.","metadata":{}},{"cell_type":"code","source":"pairplot_selected_features(n_clusters)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.479684Z","iopub.status.idle":"2022-07-14T10:51:06.480434Z","shell.execute_reply.started":"2022-07-14T10:51:06.480215Z","shell.execute_reply":"2022-07-14T10:51:06.480238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Using UMAP to reduce from 29 to 2 components. We can clearly see how the 2 clusters are separated, but there is a lot of overlap.","metadata":{}},{"cell_type":"code","source":"plot_umap_2(n_clusters)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.481566Z","iopub.status.idle":"2022-07-14T10:51:06.482171Z","shell.execute_reply.started":"2022-07-14T10:51:06.481863Z","shell.execute_reply":"2022-07-14T10:51:06.481908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Increasing the UMAP components to 3. Still clear seperation, but also overlap.","metadata":{}},{"cell_type":"code","source":"pairplot_umap(n_clusters, 3)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.483595Z","iopub.status.idle":"2022-07-14T10:51:06.483981Z","shell.execute_reply.started":"2022-07-14T10:51:06.483798Z","shell.execute_reply":"2022-07-14T10:51:06.483815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Increasing the UMAP components to 5. More overlap.","metadata":{}},{"cell_type":"code","source":"pairplot_umap(n_clusters, 5)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.485267Z","iopub.status.idle":"2022-07-14T10:51:06.485595Z","shell.execute_reply.started":"2022-07-14T10:51:06.485429Z","shell.execute_reply":"2022-07-14T10:51:06.485443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Increasing the UMAP components to 8. Again additional componenets show even mode overlap.","metadata":{}},{"cell_type":"code","source":"pairplot_umap(n_clusters, 8)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.486831Z","iopub.status.idle":"2022-07-14T10:51:06.487391Z","shell.execute_reply.started":"2022-07-14T10:51:06.487198Z","shell.execute_reply":"2022-07-14T10:51:06.487216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GBM - 5 Clusters","metadata":{}},{"cell_type":"code","source":"n_clusters=5","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.488869Z","iopub.status.idle":"2022-07-14T10:51:06.489391Z","shell.execute_reply.started":"2022-07-14T10:51:06.489201Z","shell.execute_reply":"2022-07-14T10:51:06.489219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"GBM with 2 clusters ploted on the original data. First 5 features are categorical, next 3 continuous. No clear separation and a lot of overlap.","metadata":{}},{"cell_type":"code","source":"pairplot_selected_features(n_clusters)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.490367Z","iopub.status.idle":"2022-07-14T10:51:06.490732Z","shell.execute_reply.started":"2022-07-14T10:51:06.490535Z","shell.execute_reply":"2022-07-14T10:51:06.490551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Using UMAP to reduce from 29 to 2 components. We can clearly see how the 2 clusters are separated, but there is a lot of overlap.","metadata":{}},{"cell_type":"code","source":"plot_umap_2(n_clusters)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.492003Z","iopub.status.idle":"2022-07-14T10:51:06.492321Z","shell.execute_reply.started":"2022-07-14T10:51:06.492163Z","shell.execute_reply":"2022-07-14T10:51:06.492178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Increasing the UMAP components to 3. Still clear seperation, but also overlap.","metadata":{}},{"cell_type":"code","source":"pairplot_umap(n_clusters, 3)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.493306Z","iopub.status.idle":"2022-07-14T10:51:06.493647Z","shell.execute_reply.started":"2022-07-14T10:51:06.493471Z","shell.execute_reply":"2022-07-14T10:51:06.493486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Increasing the UMAP components to 5. More overlap.","metadata":{}},{"cell_type":"code","source":"pairplot_umap(n_clusters, 5)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.494996Z","iopub.status.idle":"2022-07-14T10:51:06.495345Z","shell.execute_reply.started":"2022-07-14T10:51:06.495165Z","shell.execute_reply":"2022-07-14T10:51:06.495180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Increasing the UMAP components to 8. Again additional componenets show even mode overlap.","metadata":{}},{"cell_type":"code","source":"pairplot_umap(n_clusters, 8)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.496541Z","iopub.status.idle":"2022-07-14T10:51:06.496914Z","shell.execute_reply.started":"2022-07-14T10:51:06.496699Z","shell.execute_reply":"2022-07-14T10:51:06.496737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GBM - 7 Clusters","metadata":{}},{"cell_type":"code","source":"n_clusters=7","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.498437Z","iopub.status.idle":"2022-07-14T10:51:06.498829Z","shell.execute_reply.started":"2022-07-14T10:51:06.498610Z","shell.execute_reply":"2022-07-14T10:51:06.498626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"GBM with 7 clusters ploted on the original data. First 3 features are categorical, next 3 continuous. No clear separation and a lot of overlap.","metadata":{}},{"cell_type":"code","source":"pairplot_selected_features(n_clusters)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.499864Z","iopub.status.idle":"2022-07-14T10:51:06.500188Z","shell.execute_reply.started":"2022-07-14T10:51:06.500027Z","shell.execute_reply":"2022-07-14T10:51:06.500041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Using UMAP to reduce from 29 to 2 components. We can clearly see how the 2 clusters are separated, but there is a lot of overlap.","metadata":{}},{"cell_type":"code","source":"plot_umap_2(n_clusters)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.501583Z","iopub.status.idle":"2022-07-14T10:51:06.502233Z","shell.execute_reply.started":"2022-07-14T10:51:06.502035Z","shell.execute_reply":"2022-07-14T10:51:06.502054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Increasing the UMAP components to 3. Still clear seperation, but also overlap.","metadata":{}},{"cell_type":"code","source":"pairplot_umap(n_clusters, 3)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.503437Z","iopub.status.idle":"2022-07-14T10:51:06.504031Z","shell.execute_reply.started":"2022-07-14T10:51:06.503830Z","shell.execute_reply":"2022-07-14T10:51:06.503850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Increasing the UMAP components to 5. More overlap.","metadata":{}},{"cell_type":"code","source":"pairplot_umap(n_clusters, 5)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.505467Z","iopub.status.idle":"2022-07-14T10:51:06.505870Z","shell.execute_reply.started":"2022-07-14T10:51:06.505662Z","shell.execute_reply":"2022-07-14T10:51:06.505677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Increasing the UMAP components to 8. Again additional componenets show even mode overlap.","metadata":{}},{"cell_type":"code","source":"pairplot_umap(n_clusters, 8)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T10:51:06.507053Z","iopub.status.idle":"2022-07-14T10:51:06.507715Z","shell.execute_reply.started":"2022-07-14T10:51:06.507500Z","shell.execute_reply":"2022-07-14T10:51:06.507518Z"},"trusted":true},"execution_count":null,"outputs":[]}]}