{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-18T07:52:22.928495Z","iopub.execute_input":"2022-07-18T07:52:22.928945Z","iopub.status.idle":"2022-07-18T07:52:22.962720Z","shell.execute_reply.started":"2022-07-18T07:52:22.928847Z","shell.execute_reply":"2022-07-18T07:52:22.961905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.decomposition import PCA\nfrom scipy.cluster.hierarchy import dendrogram\nfrom sklearn.cluster import AgglomerativeClustering\nfrom matplotlib import cm\nfrom sklearn.mixture import BayesianGaussianMixture\nfrom sklearn.mixture import GaussianMixture\nfrom sklearn.preprocessing import RobustScaler, PowerTransformer\nfrom sklearn import preprocessing\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:52:22.964474Z","iopub.execute_input":"2022-07-18T07:52:22.964794Z","iopub.status.idle":"2022-07-18T07:52:24.560922Z","shell.execute_reply.started":"2022-07-18T07:52:22.964763Z","shell.execute_reply":"2022-07-18T07:52:24.559960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import data\ndata_path = '../input/tabular-playground-series-jul-2022/data.csv'\ndf = pd.read_csv(data_path)\n\ndata_for_submission=df[['id']]\ndf = df.drop(columns = \"id\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:52:24.564553Z","iopub.execute_input":"2022-07-18T07:52:24.565018Z","iopub.status.idle":"2022-07-18T07:52:26.017692Z","shell.execute_reply.started":"2022-07-18T07:52:24.564971Z","shell.execute_reply":"2022-07-18T07:52:26.016664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# EDA (Pairwised Correlation & PCA)\nstandardized_data = (df-df.mean())/df.std()\n\ncorr_coef =standardized_data.corr()\nplt.rcParams['font.size'] = 12\nfigure=plt.figure(figsize=(40, 40))\nsns.heatmap(corr_coef, vmax=1, vmin=-1, cmap='seismic', square=True, annot=True, xticklabels=1, yticklabels=1)\nplt.xlim([0, corr_coef.shape[0]])\nplt.show()\n\nfor i in range(0,5):\n    sampling_data = standardized_data.sample(n=300, random_state=i)\n    pca=PCA()\n    pca.fit(sampling_data)\n    pca_row = pca.transform(sampling_data)\n\n    plt.figure(figsize=(6, 6))\n    plt.scatter(pca_row[:, 0], pca_row[:, 1], alpha=0.8)\n    plt.grid()\n    plt.xlabel(\"PC1\")\n    plt.ylabel(\"PC2\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:52:26.018906Z","iopub.execute_input":"2022-07-18T07:52:26.019306Z","iopub.status.idle":"2022-07-18T07:52:31.527466Z","shell.execute_reply.started":"2022-07-18T07:52:26.019240Z","shell.execute_reply":"2022-07-18T07:52:31.526256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking characteristic of Cluster by hierarchical clustering\ndef plot_dendrogram(model, **kwargs):\n    counts = np.zeros(model.children_.shape[0])\n    n_samples = len(model.labels_)\n    for i, merge in enumerate(model.children_):\n        current_count = 0\n        for child_idx in merge:\n            if child_idx < n_samples:\n                current_count += 1 # leaf node\n            else:\n                current_count += counts[child_idx - n_samples]\n        counts[i] = current_count\n\n    linkage_matrix = np.column_stack([model.children_, model.distances_, counts]).astype(float)\n    dendrogram(linkage_matrix, **kwargs)\n    \n\nfor k in range(0,9):\n    sampling_data = standardized_data.sample(n=10000, random_state=k)\n    model = AgglomerativeClustering(affinity='euclidean', \n                                linkage='ward', \n                                distance_threshold=0,\n                                n_clusters=None)\n    model = model.fit(sampling_data)\n    plt.figure(figsize=(6, 6))\n    plot_dendrogram(model, truncate_mode='lastp', p=10,leaf_font_size=7)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:52:31.529955Z","iopub.execute_input":"2022-07-18T07:52:31.530343Z","iopub.status.idle":"2022-07-18T07:53:19.760382Z","shell.execute_reply.started":"2022-07-18T07:52:31.530308Z","shell.execute_reply":"2022-07-18T07:53:19.759280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For clustering unknown data, I adopt RobustScaler in this work because there are numerous outlier data point in real world data.  After pre-processing data, I try to clustered the target data by BayesianGaussianMixture and GaussianMixture. In this studies, I calculated silhouette score, calinski harabasz score, and davies bouldin score for evaluating appropreate cluster number.  ","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# BaysianGaussianMixture\n\nscore=[]\n\nscaled_data=preprocessing.RobustScaler().fit_transform(df)\nscaled_data=preprocessing.PowerTransformer().fit_transform(scaled_data)\n\nscaled_data=pd.DataFrame(scaled_data,columns = list(df.columns))\n\npca=PCA()\npca.fit(scaled_data)\npca_row = pca.transform(scaled_data)\npca_df = pd.DataFrame({\"PCA_1\" : pca_row[:,0], \"PCA_2\" : pca_row[:,1]})\n\nfor cluster in range(3,9):\n    prediction = BayesianGaussianMixture(n_components=cluster).fit_predict(scaled_data)\n    pca_df['label']=prediction\n    plt.figure(figsize=(6, 6))\n    sns.scatterplot(data = pca_df, x = \"PCA_1\", y = \"PCA_2\", hue=\"label\", s=3)\n    plt.show()\n    \n    plt.figure(figsize=(6,6))\n    sns.jointplot(data=pca_df, x=\"PCA_1\", y=\"PCA_2\", hue=\"label\", kind=\"kde\", fill=True,levels=3,alpha=0.5)\n    plt.show()\n    \n    shscore=metrics.silhouette_score(scaled_data,prediction, metric='euclidean')\n    chscore=metrics.calinski_harabasz_score(scaled_data,prediction)\n    dbscore=metrics.davies_bouldin_score(scaled_data,prediction)\n    score.append([cluster,shscore,chscore,dbscore])\n\nscore=pd.DataFrame(score)\nscore.columns = ['cluster_num','silhouette','calinski_harabasz','davies_bouldin'] \n\nprint(pd.DataFrame(score))\n\nplt.subplot(3,1,1)\nplt.plot(score.iloc[:,0],score.iloc[:,1],\n        color=cm.Set1.colors[1], label=\"silhouette_score\")\nplt.legend() \nplt.xticks(color='w')\n\nplt.subplot(3,1,2)\nplt.plot(score.iloc[:,0],score.iloc[:,2],\n        color=cm.Set1.colors[0], label=\"calinski_harabasz_score\")\nplt.legend()\nplt.xticks(color='w')\n\nplt.subplot(3,1,3)\nplt.plot(score.iloc[:,0],score.iloc[:,3],\n        color=cm.Set1.colors[2], label=\"davies_bouldin_score\")\nplt.legend()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-18T07:53:19.761712Z","iopub.execute_input":"2022-07-18T07:53:19.762049Z","iopub.status.idle":"2022-07-18T08:15:22.686922Z","shell.execute_reply.started":"2022-07-18T07:53:19.762018Z","shell.execute_reply":"2022-07-18T08:15:22.685790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# GaussianMixture\n\nscore=[]\n\nscaled_data=preprocessing.RobustScaler().fit_transform(df)\nscaled_data=preprocessing.PowerTransformer().fit_transform(scaled_data)\n\nscaled_data=pd.DataFrame(scaled_data,columns = list(df.columns))\n\npca=PCA()\npca.fit(scaled_data)\npca_row = pca.transform(scaled_data)\npca_df = pd.DataFrame({\"PCA_1\" : pca_row[:,0], \"PCA_2\" : pca_row[:,1]})\n\nfor cluster in range(3,9):\n    prediction = GaussianMixture(n_components=cluster, covariance_type = 'full', n_init=3, random_state=2255).fit_predict(scaled_data)\n    pca_df['label']=prediction\n    plt.figure(figsize=(6, 6))\n    sns.scatterplot(data = pca_df, x = \"PCA_1\", y = \"PCA_2\", hue=\"label\", s=3)\n    plt.show()\n    \n    plt.figure(figsize=(6,6))\n    sns.jointplot(data=pca_df, x=\"PCA_1\", y=\"PCA_2\", hue=\"label\", kind=\"kde\", fill=True,levels=3,alpha=0.5)\n    plt.show()\n    \n    shscore=metrics.silhouette_score(scaled_data,prediction, metric='euclidean')\n    chscore=metrics.calinski_harabasz_score(scaled_data,prediction)\n    dbscore=metrics.davies_bouldin_score(scaled_data,prediction)\n    score.append([cluster,shscore,chscore,dbscore])\n\nscore=pd.DataFrame(score)\nscore.columns = ['cluster_num','silhouette','calinski_harabasz','davies_bouldin']\n\nprint(pd.DataFrame(score))\n\n\nplt.subplot(3,1,1)\nplt.plot(score.iloc[:,0],score.iloc[:,1],\n        color=cm.Set1.colors[1], label=\"silhouette_score\")\nplt.legend() \nplt.xticks(color='w')\n\nplt.subplot(3,1,2)\nplt.plot(score.iloc[:,0],score.iloc[:,2],\n        color=cm.Set1.colors[0], label=\"calinski_harabasz_score\")\nplt.legend()\nplt.xticks(color='w')\n\nplt.subplot(3,1,3)\nplt.plot(score.iloc[:,0],score.iloc[:,3],\n        color=cm.Set1.colors[2], label=\"davies_bouldin_score\")\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T08:15:22.688817Z","iopub.execute_input":"2022-07-18T08:15:22.689178Z","iopub.status.idle":"2022-07-18T08:36:45.679025Z","shell.execute_reply.started":"2022-07-18T08:15:22.689145Z","shell.execute_reply":"2022-07-18T08:36:45.678154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"According to above study including dendrogram assay, I finally decided that cluster number is 4, and algorithm is GaussianMixture.  \n\n\nWould you please give me some comments or advises ? \n\nThank you for your reading.\n\n\n","metadata":{}},{"cell_type":"code","source":"# GaussianMixture Cluster_num = 4\n\ncluster=4\n\nscaled_data=preprocessing.RobustScaler().fit_transform(df)\nscaled_data=preprocessing.PowerTransformer().fit_transform(scaled_data)\n\nscaled_data=pd.DataFrame(scaled_data,columns = list(df.columns))\n\npca=PCA()\npca.fit(scaled_data)\npca_row = pca.transform(scaled_data)\npca_df = pd.DataFrame({\"PCA_1\" : pca_row[:,0], \"PCA_2\" : pca_row[:,1]})\n\nprediction = GaussianMixture(n_components=cluster, covariance_type = 'full', n_init=3, random_state=2255).fit_predict(scaled_data)\npca_df['label']=prediction\n\n\nplt.figure(figsize=(6, 6))\nsns.scatterplot(data = pca_df, x = \"PCA_1\", y = \"PCA_2\", hue=\"label\", s=3)\nplt.show()\n\nscore=[]\nshscore=metrics.silhouette_score(scaled_data,prediction, metric='euclidean')\nchscore=metrics.calinski_harabasz_score(scaled_data,prediction)\ndbscore=metrics.davies_bouldin_score(scaled_data,prediction)\nscore.append([cluster,shscore,chscore,dbscore])\nscore=pd.DataFrame(score)\nscore.columns = ['num','silhouette','calinski_harabasz','davies_bouldin'] \n\nprint(pd.DataFrame(score))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-18T08:36:45.680223Z","iopub.execute_input":"2022-07-18T08:36:45.681029Z","iopub.status.idle":"2022-07-18T08:38:59.342350Z","shell.execute_reply.started":"2022-07-18T08:36:45.680987Z","shell.execute_reply":"2022-07-18T08:38:59.341010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission\nsubmission = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")\n\nsubmission['Predicted'] = prediction\nsubmission.to_csv('submission.csv', index = False)\n\n#fin","metadata":{"execution":{"iopub.status.busy":"2022-07-18T08:38:59.344314Z","iopub.execute_input":"2022-07-18T08:38:59.345095Z","iopub.status.idle":"2022-07-18T08:38:59.587022Z","shell.execute_reply.started":"2022-07-18T08:38:59.345027Z","shell.execute_reply":"2022-07-18T08:38:59.585785Z"},"trusted":true},"execution_count":null,"outputs":[]}]}