{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div class=\"list-group\" id=\"list-tab\" role=\"tablist\">\n<p style= \"background-color:#154c79; font-family:fantasy; color:#FFF9ED; font-size:200%; text-align:center; border-radius:10px;\">Tabular Playground Series - Jul 2022</p>  ","metadata":{}},{"cell_type":"markdown","source":"In this notebook, we Ensembling the results of two public notebooks together. We assume that we don't have access to the predict_proba() file, so we first get these files using the BayesianGMMClassifier() library and then simply perform Ensembling. This method is our own initiative.\n\nThanks to: @thedevastator @cabaxiom\n\nhttps://www.kaggle.com/code/thedevastator/the-fine-art-of-fine-tuning\n\nhttps://www.kaggle.com/code/cabaxiom/tps-jul-22-bgmm-semi-supervised\n\n-----\n\n\n0.85 weightage is given to weaker notebook.","metadata":{}},{"cell_type":"code","source":"import warnings # suppress warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:32:59.210570Z","iopub.execute_input":"2022-07-29T16:32:59.210996Z","iopub.status.idle":"2022-07-29T16:32:59.215813Z","shell.execute_reply.started":"2022-07-29T16:32:59.210962Z","shell.execute_reply":"2022-07-29T16:32:59.214585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport random\n\nimport numpy as np \nimport pandas as pd\nimport seaborn as sns\n\nfrom tqdm import tqdm\nfrom scipy import stats\nfrom pathlib import Path\n\nimport matplotlib.pyplot as plt\nimport plotly.figure_factory as ff\nimport plotly.express as px\n%matplotlib inline\n!ls ../input/*","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:32:59.223746Z","iopub.execute_input":"2022-07-29T16:32:59.224483Z","iopub.status.idle":"2022-07-29T16:33:00.058304Z","shell.execute_reply.started":"2022-07-29T16:32:59.224432Z","shell.execute_reply":"2022-07-29T16:33:00.057075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install sklego","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-29T16:33:00.060917Z","iopub.execute_input":"2022-07-29T16:33:00.061299Z","iopub.status.idle":"2022-07-29T16:33:10.992780Z","shell.execute_reply.started":"2022-07-29T16:33:00.061261Z","shell.execute_reply":"2022-07-29T16:33:10.991504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklego.mixture import BayesianGMMClassifier\nfrom sklearn.mixture import BayesianGaussianMixture\n\nfrom sklearn.preprocessing import MinMaxScaler, PowerTransformer, StandardScaler, RobustScaler, LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:33:10.994797Z","iopub.execute_input":"2022-07-29T16:33:10.995727Z","iopub.status.idle":"2022-07-29T16:33:11.002676Z","shell.execute_reply.started":"2022-07-29T16:33:10.995670Z","shell.execute_reply":"2022-07-29T16:33:11.001161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"list-group\" id=\"list-tab\" role=\"tablist\">\n<p style= \"background-color:#154c79; font-family:fantasy; color:#FFF9ED; font-size:200%; text-align:center; border-radius:10px;\">Load Data & Preprocessing</p>  ","metadata":{}},{"cell_type":"code","source":"DATA = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv')\nSAMPLE = pd.read_csv('../input/tabular-playground-series-jul-2022/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:33:11.004190Z","iopub.execute_input":"2022-07-29T16:33:11.004574Z","iopub.status.idle":"2022-07-29T16:33:11.823483Z","shell.execute_reply.started":"2022-07-29T16:33:11.004538Z","shell.execute_reply":"2022-07-29T16:33:11.822273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = DATA.copy()\ndf.drop(\"id\", axis=1, inplace=True)\ncols = list(df.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:33:11.826430Z","iopub.execute_input":"2022-07-29T16:33:11.826918Z","iopub.status.idle":"2022-07-29T16:33:11.844104Z","shell.execute_reply.started":"2022-07-29T16:33:11.826874Z","shell.execute_reply":"2022-07-29T16:33:11.842934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_select = []\nalpha = 0.05\n\nfor col in cols:\n    _, p_value = stats.shapiro(df[col])\n    \n    if (p_value <= alpha): \n        cols_select.append(col)       \nprint(cols_select)  ","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:33:11.845481Z","iopub.execute_input":"2022-07-29T16:33:11.846065Z","iopub.status.idle":"2022-07-29T16:33:12.111105Z","shell.execute_reply.started":"2022-07-29T16:33:11.846029Z","shell.execute_reply":"2022-07-29T16:33:12.109710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"list-group\" id=\"list-tab\" role=\"tablist\">\n<p style= \"background-color:#154c79; font-family:fantasy; color:#FFF9ED; font-size:200%; text-align:center; border-radius:10px;\">Scaling</p>  ","metadata":{}},{"cell_type":"code","source":"dff = DATA[cols_select]\n\ndffs = dff.copy()\ndffs = PowerTransformer().fit_transform(dffs)\ndffs = pd.DataFrame(dffs, columns=cols_select)\ndffs","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:33:12.112435Z","iopub.execute_input":"2022-07-29T16:33:12.112798Z","iopub.status.idle":"2022-07-29T16:33:13.832471Z","shell.execute_reply.started":"2022-07-29T16:33:12.112765Z","shell.execute_reply":"2022-07-29T16:33:13.831443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"list-group\" id=\"list-tab\" role=\"tablist\">\n<p style= \"background-color:#154c79; font-family:fantasy; color:#FFF9ED; font-size:200%; text-align:center; border-radius:10px;\">Ensembling with BayesianGMMClassifier</p>  ","metadata":{}},{"cell_type":"code","source":"sub_prime = pd.read_csv('../input/tps22jul81580/submission.csv', index_col=[0])\n\nsub_prime['Predicted'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:33:13.833474Z","iopub.execute_input":"2022-07-29T16:33:13.834411Z","iopub.status.idle":"2022-07-29T16:33:13.867254Z","shell.execute_reply.started":"2022-07-29T16:33:13.834375Z","shell.execute_reply":"2022-07-29T16:33:13.866377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_prime['Predicted'] += -1\n\nsub_prime['Predicted'].value_counts().plot(kind='bar')\nsub_prime['Predicted'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:33:13.868596Z","iopub.execute_input":"2022-07-29T16:33:13.869167Z","iopub.status.idle":"2022-07-29T16:33:14.077981Z","shell.execute_reply.started":"2022-07-29T16:33:13.869132Z","shell.execute_reply":"2022-07-29T16:33:14.076679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"support = pd.read_csv('../input/tps22jun81232/submission.csv', index_col=[0])\n\nsupport['Predicted'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:33:14.079598Z","iopub.execute_input":"2022-07-29T16:33:14.080745Z","iopub.status.idle":"2022-07-29T16:33:14.115288Z","shell.execute_reply.started":"2022-07-29T16:33:14.080702Z","shell.execute_reply":"2022-07-29T16:33:14.114109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"support['Predicted'] += -1\n\nsupport['Predicted'].value_counts().plot(kind='bar')\nsupport['Predicted'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:33:14.117010Z","iopub.execute_input":"2022-07-29T16:33:14.117691Z","iopub.status.idle":"2022-07-29T16:33:14.317446Z","shell.execute_reply.started":"2022-07-29T16:33:14.117644Z","shell.execute_reply":"2022-07-29T16:33:14.316263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = np.array(dffs)\ny = np.array(sub_prime)\ns = np.array(support)\n\nbgm = BayesianGMMClassifier(n_components=7, random_state=123, tol=0.001, max_iter=200, n_init=3, verbose=0)\n\nbgm.fit(X,y)\nproba = bgm.predict_proba(X)\n\nbgm.fit(X,s)\nprobs = bgm.predict_proba(X)\n\nproba.shape, probs.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:33:14.318796Z","iopub.execute_input":"2022-07-29T16:33:14.319217Z","iopub.status.idle":"2022-07-29T16:38:50.075258Z","shell.execute_reply.started":"2022-07-29T16:33:14.319171Z","shell.execute_reply":"2022-07-29T16:38:50.074019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### <span style=\"color:darkblue;\">We multiply the probabilities of the weaker notebook by 0.95.</span>","metadata":{}},{"cell_type":"code","source":"prob = np.concatenate((proba, probs*0.82), axis=1)\nprob.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:38:50.076720Z","iopub.execute_input":"2022-07-29T16:38:50.077488Z","iopub.status.idle":"2022-07-29T16:38:50.089862Z","shell.execute_reply.started":"2022-07-29T16:38:50.077442Z","shell.execute_reply":"2022-07-29T16:38:50.088498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### <span style=\"color:darkblue;\">The final result is the highest probability (related to each of the algorithms).</span>","metadata":{}},{"cell_type":"code","source":"pred = np.argmax(prob, axis=1)\npred, min(pred), max(pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:38:50.093767Z","iopub.execute_input":"2022-07-29T16:38:50.094330Z","iopub.status.idle":"2022-07-29T16:38:50.118835Z","shell.execute_reply.started":"2022-07-29T16:38:50.094294Z","shell.execute_reply":"2022-07-29T16:38:50.117705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clusters = np.zeros(shape=(7, 7), dtype=int)\nfor n1, n2 in zip(y, s):\n    clusters[n1, n2] += 1\n    \nclusters","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:38:50.120033Z","iopub.execute_input":"2022-07-29T16:38:50.120755Z","iopub.status.idle":"2022-07-29T16:38:50.524525Z","shell.execute_reply.started":"2022-07-29T16:38:50.120719Z","shell.execute_reply":"2022-07-29T16:38:50.523179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_clusters = np.argmax(clusters, axis=0)\nmax_clusters","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:38:50.526014Z","iopub.execute_input":"2022-07-29T16:38:50.526384Z","iopub.status.idle":"2022-07-29T16:38:50.534533Z","shell.execute_reply.started":"2022-07-29T16:38:50.526349Z","shell.execute_reply":"2022-07-29T16:38:50.533247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(pred)):\n    \n    if (pred[i] == 7): \n        pred[i] = max_clusters[0]\n    if (pred[i] == 8): \n        pred[i] = max_clusters[1]\n    if (pred[i] == 9): \n        pred[i] = max_clusters[2]\n    if (pred[i] == 10): \n        pred[i] = max_clusters[3]\n    if (pred[i] == 11): \n        pred[i] = max_clusters[4]\n    if (pred[i] == 12): \n        pred[i] = max_clusters[5]\n    if (pred[i] == 13): \n        pred[i] = max_clusters[6]        \n\npred, min(pred), max(pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:38:50.536328Z","iopub.execute_input":"2022-07-29T16:38:50.536718Z","iopub.status.idle":"2022-07-29T16:38:50.743433Z","shell.execute_reply.started":"2022-07-29T16:38:50.536684Z","shell.execute_reply":"2022-07-29T16:38:50.742271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"list-group\" id=\"list-tab\" role=\"tablist\">\n<p style= \"background-color:#154c79; font-family:fantasy; color:#FFF9ED; font-size:200%; text-align:center; border-radius:10px;\">Submission</p>  ","metadata":{}},{"cell_type":"code","source":"sub = SAMPLE.copy()\nsub['Predicted'] = pred","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:38:50.744832Z","iopub.execute_input":"2022-07-29T16:38:50.745181Z","iopub.status.idle":"2022-07-29T16:38:50.751812Z","shell.execute_reply.started":"2022-07-29T16:38:50.745149Z","shell.execute_reply":"2022-07-29T16:38:50.750534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hist_data = [sub_prime.iloc[:, 0], pred]  \ngroup_labels = ['Sub_Prime', 'Submission']\n  \nfig = ff.create_distplot(hist_data, group_labels, bin_size=.2, show_hist=False, show_rug=False) \nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:38:50.753445Z","iopub.execute_input":"2022-07-29T16:38:50.753809Z","iopub.status.idle":"2022-07-29T16:38:52.244181Z","shell.execute_reply.started":"2022-07-29T16:38:50.753776Z","shell.execute_reply":"2022-07-29T16:38:52.242924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index=False)\nprint(\"submission successful\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T16:38:52.245562Z","iopub.execute_input":"2022-07-29T16:38:52.245911Z","iopub.status.idle":"2022-07-29T16:38:52.355757Z","shell.execute_reply.started":"2022-07-29T16:38:52.245880Z","shell.execute_reply":"2022-07-29T16:38:52.354452Z"},"trusted":true},"execution_count":null,"outputs":[]}]}