{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom tqdm import trange\nfrom sklearn.cluster import KMeans\nfrom sklearn.mixture import GaussianMixture\nfrom scipy.sparse import csr_matrix","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-08T18:06:36.221716Z","iopub.execute_input":"2022-07-08T18:06:36.227045Z","iopub.status.idle":"2022-07-08T18:06:36.235894Z","shell.execute_reply.started":"2022-07-08T18:06:36.226983Z","shell.execute_reply":"2022-07-08T18:06:36.234612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv')\ndf = df.drop(columns = 'id')\n\ndf = df[:10000] # Just for the fast demonstration. ","metadata":{"execution":{"iopub.status.busy":"2022-07-08T18:10:25.842554Z","iopub.execute_input":"2022-07-08T18:10:25.842970Z","iopub.status.idle":"2022-07-08T18:10:26.873397Z","shell.execute_reply.started":"2022-07-08T18:10:25.842938Z","shell.execute_reply":"2022-07-08T18:10:26.872163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusters_k_means = KMeans(n_clusters = 7).fit_predict(df)\nclusters_mixture = GaussianMixture(n_components = 7).fit_predict(df)\n\nprint(clusters_k_means.shape)\nprint(clusters_mixture.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T18:10:27.616131Z","iopub.execute_input":"2022-07-08T18:10:27.616636Z","iopub.status.idle":"2022-07-08T18:10:32.733688Z","shell.execute_reply.started":"2022-07-08T18:10:27.616597Z","shell.execute_reply":"2022-07-08T18:10:32.732101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_sparse_matrix(clusters):\n    n = len(clusters)\n    data = []\n    row = []\n    col = []\n    for i in trange(n):\n        for j in range(i+1, n):\n            if clusters[i] == clusters[j]:\n                data.append(1)\n                row.append(i)\n                col.append(j)\n    return csr_matrix((data, (row, col)), shape=(n, n))\n\nsparse_matrix_k_means = create_sparse_matrix(clusters_k_means)\nsparse_matrix_mixture = create_sparse_matrix(clusters_mixture)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T18:10:34.033664Z","iopub.execute_input":"2022-07-08T18:10:34.034030Z","iopub.status.idle":"2022-07-08T18:11:35.534758Z","shell.execute_reply.started":"2022-07-08T18:10:34.034001Z","shell.execute_reply":"2022-07-08T18:11:35.533534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sparse_matrix_mean = (sparse_matrix_k_means + sparse_matrix_mixture) / 2","metadata":{"execution":{"iopub.status.busy":"2022-07-08T18:11:35.537042Z","iopub.execute_input":"2022-07-08T18:11:35.537456Z","iopub.status.idle":"2022-07-08T18:11:35.815969Z","shell.execute_reply.started":"2022-07-08T18:11:35.537421Z","shell.execute_reply":"2022-07-08T18:11:35.814843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sparse_matrix_mean[sparse_matrix_mean < 0.5] = 0\nsparse_matrix_mean[sparse_matrix_mean >= 0.5] = 1","metadata":{"execution":{"iopub.status.busy":"2022-07-08T18:11:35.817338Z","iopub.execute_input":"2022-07-08T18:11:35.817838Z","iopub.status.idle":"2022-07-08T18:12:05.774833Z","shell.execute_reply.started":"2022-07-08T18:11:35.817798Z","shell.execute_reply":"2022-07-08T18:12:05.773147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sparse_matrix_mean = sparse_matrix_mean.toarray()\n\nclusters_final = np.zeros(len(df))\nclusters_final_next_id = 0\n\nfor i in range(len(df)):\n    if clusters_final[i] == 0:\n        clusters_final_current_id = clusters_final_next_id\n        clusters_final_next_id += 1\n        clusters_final[i] = clusters_final_current_id\n        for j in range(i+1, len(df)):\n            if sparse_matrix_mean[i, j] == 1:\n                clusters_final[j] = clusters_final_current_id\n\nprint(clusters_final)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T18:12:05.777950Z","iopub.execute_input":"2022-07-08T18:12:05.778495Z","iopub.status.idle":"2022-07-08T18:12:07.696551Z","shell.execute_reply.started":"2022-07-08T18:12:05.778448Z","shell.execute_reply":"2022-07-08T18:12:07.695500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}