{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Simple way to assign same label to differents clusters based to clusters centers distance. Calculate Centers, calculate pairwise_distances and reorder cluster proba array.\n\n(With seed 13 just 6 clusters are generated.)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T14:41:22.613477Z","iopub.execute_input":"2022-07-08T14:41:22.613867Z","iopub.status.idle":"2022-07-08T14:41:22.620629Z","shell.execute_reply.started":"2022-07-08T14:41:22.613833Z","shell.execute_reply":"2022-07-08T14:41:22.619428Z"}}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport gc,random,os\nimport scipy.stats\nfrom sklearn.preprocessing import PowerTransformer\nfrom sklearn.mixture import BayesianGaussianMixture\nfrom sklearn.metrics.pairwise import pairwise_distances_argmin\nfrom sklearn.metrics import silhouette_score\n\n# Visualization\nfrom sklearn.decomposition import PCA\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:13:51.767314Z","iopub.execute_input":"2022-07-14T08:13:51.767687Z","iopub.status.idle":"2022-07-14T08:13:51.774124Z","shell.execute_reply.started":"2022-07-14T08:13:51.767658Z","shell.execute_reply":"2022-07-14T08:13:51.773399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=2022):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n\nSEED=2022\nseed_everything(SEED)\n\nN_cluster=7\nN_rows=98000\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:13:51.884543Z","iopub.execute_input":"2022-07-14T08:13:51.885156Z","iopub.status.idle":"2022-07-14T08:13:51.890476Z","shell.execute_reply.started":"2022-07-14T08:13:51.885122Z","shell.execute_reply":"2022-07-14T08:13:51.889439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train=pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\",nrows=N_rows)\ntrain=train.drop(\"id\",axis=1)\nbest_data =['f_07','f_08', 'f_09', 'f_10','f_11', 'f_12', 'f_13', 'f_22','f_23', 'f_24', 'f_25','f_26','f_27', 'f_28']","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:13:52.008263Z","iopub.execute_input":"2022-07-14T08:13:52.008910Z","iopub.status.idle":"2022-07-14T08:13:53.128550Z","shell.execute_reply.started":"2022-07-14T08:13:52.008874Z","shell.execute_reply":"2022-07-14T08:13:53.127253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaled = PowerTransformer().fit_transform(train[best_data])\n\n#For visualization\npca = PCA(n_components=2)\nreduced_data = pca.fit_transform(train[best_data])","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:13:53.130233Z","iopub.execute_input":"2022-07-14T08:13:53.130520Z","iopub.status.idle":"2022-07-14T08:13:55.219692Z","shell.execute_reply.started":"2022-07-14T08:13:53.130495Z","shell.execute_reply":"2022-07-14T08:13:55.218156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_bgm_centers(BGM,X):\n    centers = np.empty(shape=(BGM.n_components, X.shape[1]))\n    for i in range(BGM.n_components):\n        density = scipy.stats.multivariate_normal(cov=BGM.covariances_[i], mean=BGM.means_[i]).logpdf(X)\n        centers[i, :] = X[np.argmax(density)]\n    return centers\n\n\n\ncluster_seeds=[1,13,89,144]\n\nfor seed in cluster_seeds:\n    BGM = BayesianGaussianMixture(n_components=N_cluster, covariance_type='full', random_state=seed, n_init = 5, max_iter=100)\n    BGM.fit(scaled)\n    proba=BGM.predict_proba(scaled)  \n    \n    #Establishing parity between clusters.\n    #https://scikit-learn.org/stable/auto_examples/cluster/plot_mini_batch_kmeans.html#sphx-glr-auto-examples-cluster-plot-mini-batch-kmeans-py\n    \n    #calculate clusters centers.\n    centers=get_bgm_centers(BGM,scaled)\n    \n    #First cluster center to calculate distance.\n    if seed == cluster_seeds[0]:  \n        center_ori=centers \n        preds=np.argmax(proba, axis=1)    \n      \n    #Get matrix columns order and calculate labels with ordered matrix.\n    else:\n        col_order = pairwise_distances_argmin(center_ori,centers)\n        preds=np.argmax(proba[:,col_order], axis=1)        \n        \n        \n    #Plot clusters    \n    df = pd.DataFrame(\n            {\n                \"x\": reduced_data[:,0],\n                \"y\": reduced_data[:,1],\n                \"clusters\" : preds\n            }\n        )\n    plt.figure(figsize=(5, 5))\n    sns.scatterplot(x=df[\"x\"], y=df[\"y\"], hue=df[\"clusters\"],palette=\"deep\")       ","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:13:55.222474Z","iopub.execute_input":"2022-07-14T08:13:55.223038Z","iopub.status.idle":"2022-07-14T08:26:09.711346Z","shell.execute_reply.started":"2022-07-14T08:13:55.222991Z","shell.execute_reply":"2022-07-14T08:26:09.710478Z"},"trusted":true},"execution_count":null,"outputs":[]}]}