{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Number of Clusters go Brrrrr\n\n* Yes I thought this was ridiculous when I first tried it, didn't expect the LB to reward the experiment\n* Voting in this way was key to allowing the individual cluster models to have more n_clusters than the final model\n* This means (if we haven't over fit) that there is some more structure to the clusters than just plain gaussians\n* There was a picture somewhere of using GMMs to fit curves of clusters, but I can't be sure I saw it before doing this or after\n* Sorry I couldn't share this insight earlier, I just wanted the t-shirt.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom sklearn.cluster import KMeans, MiniBatchKMeans\nfrom scipy.sparse import csr_matrix\nfrom sklearn.mixture import GaussianMixture, BayesianGaussianMixture\nfrom sklearn.metrics import adjusted_rand_score\nfrom sklearn.preprocessing import RobustScaler, StandardScaler, PowerTransformer, QuantileTransformer\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.metrics import adjusted_rand_score\nfrom sklearn.preprocessing import OneHotEncoder\nimport glob\nfrom collections import Counter, defaultdict\nimport pylab as pl","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-27T11:18:42.676551Z","iopub.execute_input":"2022-07-27T11:18:42.676950Z","iopub.status.idle":"2022-07-27T11:18:43.447090Z","shell.execute_reply.started":"2022-07-27T11:18:42.676918Z","shell.execute_reply":"2022-07-27T11:18:43.445671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\nn_rows = 98000\nn_clusters = 7\nnn_clusters = 49\nn_iter = 35\n\nthreshold_pr = 0.76\nsubsample_pr = 0.23\n\nknown = pd.read_csv(\"/kaggle/input/bgmm-baseline/model_best_bgmm_3.csv\", index_col=0).iloc[:, 0].values\nknown_proba = pd.read_csv('/kaggle/input/bgmm-baseline/model_best_bgmm_3b_2_proba.csv', index_col=0).values\nselection = np.max(known_proba, axis=1) > threshold_pr\n#print(np.sum(selection))\n\nsns.kdeplot(np.max(known_proba, axis=1))\n\n#print(Counter(known[selection]))\n\n#known = pd.read_csv(\"/kaggle/input/tab2022julbest/submission.csv\", index_col=0).iloc[:, 0].values\n#known_proba = pd.read_csv('/kaggle/input/tab2022julbest/submission_proba.csv', index_col=0).values\n\n#known_proba = (known_proba) / np.sum(known_proba, axis=1, keepdims=True)\n#print(known_proba.shape)\n\nsns.kdeplot(np.max(known_proba, axis=1))\n\n# Only use items we are sure about for cluster membership\nselection = np.max(known_proba, axis=1) > threshold_pr\nprint(np.sum(selection))\n\nprint(Counter(known[selection]))\n\ndata = pd.read_csv('/kaggle/input/tabular-playground-series-jul-2022/data.csv', index_col=0)\ndata = data[['f_07','f_08','f_09','f_10','f_11','f_12','f_13',\n             'f_22','f_23','f_24','f_25','f_26','f_27','f_28']]\n\ndata = PowerTransformer().fit_transform(data)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T11:24:50.698527Z","iopub.execute_input":"2022-07-27T11:24:50.699540Z","iopub.status.idle":"2022-07-27T11:24:54.480229Z","shell.execute_reply.started":"2022-07-27T11:24:50.699494Z","shell.execute_reply":"2022-07-27T11:24:54.479081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"soft_soft_accumulator = np.zeros(shape=(n_rows, n_clusters), dtype=np.float32)\n\nbase = 6300\nfor seed in tqdm(range(n_iter)):\n    np.random.seed(seed + base)\n    subsample = data[np.random.random(size=(n_rows,)) < subsample_pr, :]\n \n    gmm = BayesianGaussianMixture(\n            n_components = nn_clusters,\n            random_state = seed + base,\n            tol = 0.001,\n            max_iter = 300,\n            n_init = 4)\n    gmm.fit(subsample)\n\n    result = gmm.predict(data)\n    probs = gmm.predict_proba(data)\n    \n    # Create cluster membership matrix\n    clusters = np.zeros(shape = (nn_clusters, n_clusters), dtype=int)\n    for n1, n2 in zip(result[selection], known[selection]): \n        clusters[n1, n2] += 1\n\n    # hard_clusters = np.argmax(clusters, axis=1)\n    # hard_dict = {i: c for i, c in enumerate(hard_clusters)}\n    # hard_result = np.vectorize(hard_dict.__getitem__)(result)\n    # accumulator[np.arange(98000), hard_result] += 1\n    \n    soft_clusters = clusters / np.sum(clusters, axis=1).reshape(-1, 1)\n    # soft_result = soft_clusters[result]\n    # soft_accumulator += soft_clusters[result]\n\n    # for i in range(n_clusters):\n    #     hard_soft_accumulator[this == i] += soft_clusters[i]\n    soft_soft_accumulator += np.matmul(probs, soft_clusters)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# hard_result = np.argmax(hard_accumulator, axis=1)\n# soft_result = np.argmax(soft_accumulator, axis=1)\nsoft_soft_result = np.argmax(soft_soft_accumulator, axis=1)\n\npd.DataFrame(soft_soft_result).to_csv('submission.csv', index_label='Id', header=['Predicted'])\npd.DataFrame(soft_soft_accumulator).to_csv('submission_proba.csv', index_label='Id')","metadata":{},"execution_count":null,"outputs":[]}]}