{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Improving Meat for Ensemble Voting\n\n* If you're ensembling you want the things you ensemble to be different\n* Some had used low tol and max_iter values to do this, but I think there are better ways\n* So I subsample the dataset, to provide noise (just like you would for gradient boosting)\n* The voting methods used are problematic","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom sklearn.preprocessing import PowerTransformer\nfrom sklearn.mixture import BayesianGaussianMixture\nfrom collections import Counter\nfrom tqdm import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-16T01:47:09.272359Z","iopub.execute_input":"2022-07-16T01:47:09.273071Z","iopub.status.idle":"2022-07-16T01:47:09.279398Z","shell.execute_reply.started":"2022-07-16T01:47:09.273037Z","shell.execute_reply":"2022-07-16T01:47:09.278130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_seed = 42069\nn_clusters = 7\nn_iters = 20\nsubsample_pr = 0.8","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\", index_col=0)\ndata = data[['f_07','f_08','f_09','f_10','f_11','f_12','f_13',\n             'f_22','f_23','f_24','f_25','f_26','f_27','f_28']]\ndata = PowerTransformer().fit_transform(data)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T01:47:12.526932Z","iopub.execute_input":"2022-07-16T01:47:12.528113Z","iopub.status.idle":"2022-07-16T01:47:15.652925Z","shell.execute_reply.started":"2022-07-16T01:47:12.528076Z","shell.execute_reply":"2022-07-16T01:47:15.652079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_votes = np.zeros((data.shape[0], n_clusters))\n\nfor seed in tqdm(range(base_seed, base_seed+n_iters)):\n    np.random.seed(seed)\n    subsample = data[np.random.random(size=(data.shape[0],)) < subsample_pr, :]\n    \n    model = BayesianGaussianMixture(n_components = n_clusters, covariance_type = 'full', random_state = seed, n_init = 3, tol=.001, max_iter=200)\n    model.fit(subsample)\n\n    probs = model.predict_proba(data)\n    labels = np.argmax(probs, axis=1) # this is faster, since we want both\n    \n    # This is just the same merging the other voting notebooks do.\n    # It is still problematic to assume your clusters will consistently be the same size.\n    # especially in this case, where I'm pretty sure some of the clusters are very close in size.\n    cluster_sizes = Counter(labels)\n    translation = {k: i for i, (k, _) in enumerate(cluster_sizes.most_common())}\n\n    for from_i, to_i in translation.items():\n        final_votes[:, to_i] += probs[:, from_i]","metadata":{"execution":{"iopub.status.busy":"2022-07-16T01:47:25.883595Z","iopub.execute_input":"2022-07-16T01:47:25.884039Z","iopub.status.idle":"2022-07-16T02:19:28.084596Z","shell.execute_reply.started":"2022-07-16T01:47:25.884003Z","shell.execute_reply":"2022-07-16T02:19:28.083348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")\n\nsub['Predicted'] = np.argmax(final_votes, axis=1)\nsub.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T02:23:34.899034Z","iopub.execute_input":"2022-07-16T02:23:34.899431Z","iopub.status.idle":"2022-07-16T02:23:35.107096Z","shell.execute_reply.started":"2022-07-16T02:23:34.899400Z","shell.execute_reply":"2022-07-16T02:23:35.105861Z"},"trusted":true},"execution_count":null,"outputs":[]}]}