{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Better Voting\n\nIdea is to better allocate votes, rather than matching clusters by sorting them by their size.\n\nI have chosen to use some good known clustering that isn't doing too badly. at 0.600~ rand, it must be getting a bunch correct already.\n\nWe'll use the allocation of those items to pick out which clusters are which in the clusterings we ensemble.\n\nThere's every chance that I'm overfitting, but I think this is less likely to overfit than training gradient boosting models on labels.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom sklearn.cluster import KMeans, MiniBatchKMeans\nfrom scipy.sparse import csr_matrix\nfrom sklearn.mixture import GaussianMixture, BayesianGaussianMixture\nfrom sklearn.metrics import adjusted_rand_score\nfrom sklearn.preprocessing import RobustScaler, StandardScaler, PowerTransformer, QuantileTransformer\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.metrics import adjusted_rand_score\nfrom sklearn.preprocessing import OneHotEncoder\nimport glob\nfrom collections import Counter, defaultdict\nimport pylab as pl\nimport xxhash","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_rows = 98000\nn_clusters = 7\n\nknown = pd.read_csv(\"/kaggle/input/tps-jul2022-bgmm/submission.csv\", index_col=0).iloc[:, 0].values\nknown2 = pd.read_csv(\"/kaggle/input/tps-jul-22-gmm-baseline/submission.csv\", index_col=0).iloc[:, 0].values\n\n\ndata = pd.read_csv('/kaggle/input/tabular-playground-series-jul-2022/data.csv', index_col=0)\ndata = data[['f_07','f_08','f_09','f_10','f_11','f_12','f_13',\n             'f_22','f_23','f_24','f_25','f_26','f_27','f_28']]\n# data = RobustScaler().fit_transform(data)\ndata = PowerTransformer().fit_transform(data)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Having loaded into two decent solutions, we can see in the plot below that they mostly agree on the clustering.\nI use this to pick out which cluster is which when ensembling clustering by voting.","metadata":{}},{"cell_type":"code","source":"clusters = np.zeros(shape = (7, 7))\nfor n1, n2 in zip(known, known2):\n    clusters[n1, n2] += 1\n\npl.imshow(clusters)\npl.colorbar()\npl.title('clusters by shared membership');","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Build a whole bunch of models to ensemble together","metadata":{}},{"cell_type":"code","source":"all_labels = []\nall_probs = []\n\nbase = 1301\nfor seed in tqdm(range(20)):\n    np.random.seed(seed + base * 20)\n    subspace = data[np.random.random(size=(n_rows,)) < 0.5, :]\n \n    gmm = BayesianGaussianMixture(\n            n_components = n_clusters,\n            random_state = seed + base * 20,\n            tol = 0.001,  # Actually, since we're ensembling we might want a higher tol\n            max_iter = 200, # and a low max_iter! Because noise is our friend.\n            n_init = 4)\n    gmm.fit(subspace)\n\n    all_labels.append(gmm.predict(data))\n    all_probs.append(gmm.predict_proba(data))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we accumulate votes by three methods:\n\n1. Hard voting from the labels - essentially use the cluster matrix to translate labels\n2. Use cluster matrix as a soft membership, score being in a cluster according to how much they \n   match the known clusters.\n3. Matrix multiplication - mix the cluster matrix probabilities with the proba from the model\n   (this one seems to perform the best)","metadata":{}},{"cell_type":"code","source":"hard_accumulator = np.zeros(shape=(n_rows, n_clusters), dtype=np.int32)\nsoft_accumulator = np.zeros(shape=(n_rows, n_clusters), dtype=np.float32)\nsoft_soft_accumulator = np.zeros(shape=(n_rows, n_clusters), dtype=np.float32)\n\nfor model_i, labels in enumerate(all_labels):\n    probs = all_probs[model_i]\n    \n    # We create a small n_cluster x n_cluster matrix to store which\n    # cluster corresponds to which, compared to the known good clustering.\n    clusters = np.zeros(shape = (n_clusters, n_clusters), dtype=int)\n    for n1, n2 in zip(labels, known):\n        clusters[n1, n2] += 1\n\n    hard_clusters = np.argmax(clusters, axis=1)\n    hard_dict = {i: c for i, c in enumerate(hard_clusters)}\n    hard_result = np.vectorize(hard_dict.__getitem__)(labels)\n    hard_accumulator[np.arange(n_rows), hard_result] += 1\n    \n    # soft\n    soft_clusters = clusters / np.sum(clusters, axis=1).reshape(-1, 1)\n    soft_result = soft_clusters[labels]\n    soft_accumulator += soft_clusters[labels]\n\n    # extra soft\n    soft_soft_accumulator += np.matmul(probs, soft_clusters)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# hard_result = np.argmax(hard_accumulator, axis=1)\n# soft_result = np.argmax(soft_accumulator, axis=1)\nsoft_soft_result = np.argmax(soft_soft_accumulator, axis=1)\n\npd.DataFrame(soft_soft_result).to_csv('submission.csv', index_label='Id', header=['Predicted'])","metadata":{},"execution_count":null,"outputs":[]}]}