{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport matplotlib.pyplot as plt\n\nfrom sklearn import mixture","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-23T04:00:15.445710Z","iopub.execute_input":"2022-07-23T04:00:15.446756Z","iopub.status.idle":"2022-07-23T04:00:15.733080Z","shell.execute_reply.started":"2022-07-23T04:00:15.446717Z","shell.execute_reply":"2022-07-23T04:00:15.731856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/tabular-playground-series-jul-2022/data.csv').drop('id', axis=1)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:58:18.416302Z","iopub.execute_input":"2022-07-23T03:58:18.416771Z","iopub.status.idle":"2022-07-23T03:58:19.724975Z","shell.execute_reply.started":"2022-07-23T03:58:18.416727Z","shell.execute_reply":"2022-07-23T03:58:19.723673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:58:20.368857Z","iopub.execute_input":"2022-07-23T03:58:20.369250Z","iopub.status.idle":"2022-07-23T03:58:20.377377Z","shell.execute_reply.started":"2022-07-23T03:58:20.369218Z","shell.execute_reply":"2022-07-23T03:58:20.375934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Unsupervised Learning\nLets move beyond Kmean clustering.\n* [GMM(Gaussian Mixture Model)](https://michael-fuchs-python.netlify.app/2020/06/24/gaussian-mixture-models/)","metadata":{}},{"cell_type":"markdown","source":"Like Kmean you have to provide number of cluster here. We cannot use inertia or  silhouette score here cus they are not reliable if cluster is not spherical.","metadata":{"execution":{"iopub.status.busy":"2022-07-20T18:54:59.609984Z","iopub.execute_input":"2022-07-20T18:54:59.611901Z","iopub.status.idle":"2022-07-20T18:55:20.529060Z","shell.execute_reply.started":"2022-07-20T18:54:59.611836Z","shell.execute_reply":"2022-07-20T18:55:20.527623Z"}}},{"cell_type":"code","source":"from sklearn import mixture\n\n\nSum_bic = []\nSum_aic = []\n\nK = range(1,15)\nfor k in K:\n    gmm = mixture.GaussianMixture(n_components=k)\n    gmm = gmm.fit(df.values)\n    Sum_bic.append(gmm.bic(df.values))\n    Sum_aic.append(gmm.aic(df.values))\n    \nx1 = K\ny1 = Sum_aic\nplt.plot(x1, y1, label = \"AIC\")\nx2 = K\ny2 = Sum_bic\nplt.plot(x2, y2, label = \"BIC\")\n\nplt.title(\"AIC and BIC for different numbers of k\", fontsize=16, fontweight='bold')\nplt.xlabel(\"k\")\nplt.ylabel(\"Information Criterion\")\nplt.legend(loc='upper right')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T11:30:17.336100Z","iopub.execute_input":"2022-07-21T11:30:17.336464Z","iopub.status.idle":"2022-07-21T11:35:09.100600Z","shell.execute_reply.started":"2022-07-21T11:30:17.336434Z","shell.execute_reply":"2022-07-21T11:35:09.099196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gmm = mixture.GaussianMixture(n_components=8)\ngmm.fit(df.values)\nlabels = gmm.predict(df.values)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:04:42.312990Z","iopub.execute_input":"2022-07-20T19:04:42.313389Z","iopub.status.idle":"2022-07-20T19:04:59.672659Z","shell.execute_reply.started":"2022-07-20T19:04:42.313357Z","shell.execute_reply":"2022-07-20T19:04:59.671023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"score - 0.465","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:06:28.412810Z","iopub.execute_input":"2022-07-20T19:06:28.413203Z","iopub.status.idle":"2022-07-20T19:06:28.595038Z","shell.execute_reply.started":"2022-07-20T19:06:28.413172Z","shell.execute_reply":"2022-07-20T19:06:28.593956Z"}}},{"cell_type":"markdown","source":"* lets make data more gaussian like - for this we have power transformation","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import PowerTransformer\nimport random\nimport os\nimport numpy as np\n\ndef seed_everything(seed=42):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n\nseed_everything()\ntransform_data = PowerTransformer().fit_transform(df.values)\ndf_scaled = pd.DataFrame(transform_data, columns=df.columns)\ndf_scaled.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:58:32.067215Z","iopub.execute_input":"2022-07-23T03:58:32.067641Z","iopub.status.idle":"2022-07-23T03:58:36.230433Z","shell.execute_reply.started":"2022-07-23T03:58:32.067607Z","shell.execute_reply":"2022-07-23T03:58:36.229190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Sum_bic = []\nSum_aic = []\n\nK = range(1,15)\nfor k in K:\n    gmm = mixture.GaussianMixture(n_components=k)\n    gmm = gmm.fit(df_scaled.values)\n    Sum_bic.append(gmm.bic(df_scaled.values))\n    Sum_aic.append(gmm.aic(df_scaled.values))\n    \nx1 = K\ny1 = Sum_aic\nplt.plot(x1, y1, label = \"AIC\")\nx2 = K\ny2 = Sum_bic\nplt.plot(x2, y2, label = \"BIC\")\n\nplt.title(\"AIC and BIC for different numbers of k\", fontsize=16, fontweight='bold')\nplt.xlabel(\"k\")\nplt.ylabel(\"Information Criterion\")\nplt.legend(loc='upper right')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T11:35:34.252068Z","iopub.execute_input":"2022-07-21T11:35:34.252477Z","iopub.status.idle":"2022-07-21T11:45:17.195923Z","shell.execute_reply.started":"2022-07-21T11:35:34.252446Z","shell.execute_reply":"2022-07-21T11:45:17.195079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gmm = mixture.GaussianMixture(n_components=8)\ngmm.fit(df_scaled.values)\nlabels = gmm.predict(df_scaled.values)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T19:27:22.690732Z","iopub.execute_input":"2022-07-20T19:27:22.691137Z","iopub.status.idle":"2022-07-20T19:27:43.659430Z","shell.execute_reply.started":"2022-07-20T19:27:22.691104Z","shell.execute_reply":"2022-07-20T19:27:43.657840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"score - 0.540","metadata":{}},{"cell_type":"markdown","source":"impressive increase","metadata":{}},{"cell_type":"markdown","source":"* [Bayesian Gaussian Mixture Model](https://michael-fuchs-python.netlify.app/2020/06/26/bayesian-gaussian-mixture-models/)","metadata":{}},{"cell_type":"code","source":"from sklearn import mixture\nbgm_model = mixture.BayesianGaussianMixture(n_components=7, n_init=1, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:06:49.300611Z","iopub.execute_input":"2022-07-21T17:06:49.301224Z","iopub.status.idle":"2022-07-21T17:06:49.658255Z","shell.execute_reply.started":"2022-07-21T17:06:49.301174Z","shell.execute_reply":"2022-07-21T17:06:49.656412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bgm_model.fit(df_scaled)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:06:51.390796Z","iopub.execute_input":"2022-07-21T17:06:51.392111Z","iopub.status.idle":"2022-07-21T17:08:02.980895Z","shell.execute_reply.started":"2022-07-21T17:06:51.392043Z","shell.execute_reply":"2022-07-21T17:08:02.979479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bgm_predict_proba = bgm_model.predict_proba(df_scaled)  #shape - (98000, 7)\nbgm_predict = np.argmax(bgm_predict_proba, axis=1)  #shape - (98000,)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T17:08:06.765540Z","iopub.execute_input":"2022-07-21T17:08:06.766976Z","iopub.status.idle":"2022-07-21T17:08:07.147157Z","shell.execute_reply.started":"2022-07-21T17:08:06.766910Z","shell.execute_reply":"2022-07-21T17:08:07.145443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"score - 0.544","metadata":{}},{"cell_type":"markdown","source":"if you check predict proba, sometime highest probability is 0.54 . So applying probability threshold. \n\nfor proba value higher than threshold, will be a treated as training data with corresponding cluster as label. and will create supervised model on it, to predict rest.","metadata":{}},{"cell_type":"code","source":"proba_threshold = 0.7\n\n#getting corresponding probability value of prediction \ndf_scaled['predict'] = bgm_predict\n\nfor n in range(7):\n    df_scaled[f'predict_proba_{n}'] = bgm_predict_proba[:, n]\n    df_scaled.loc[df_scaled['predict'] == n, 'predict_proba'] = df_scaled[f'predict_proba_{n}']","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:00:13.179819Z","iopub.execute_input":"2022-07-21T10:00:13.180223Z","iopub.status.idle":"2022-07-21T10:00:13.210161Z","shell.execute_reply.started":"2022-07-21T10:00:13.180190Z","shell.execute_reply":"2022-07-21T10:00:13.209213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#applying threshold\n\nidxs = np.array([])\nfor n in range(7):\n    idx = df_scaled[(df_scaled.predict==n) & (df_scaled.predict_proba > proba_threshold)].index\n    idxs = np.concatenate((idxs, idx))\n    print(f'Class n{n}  | Training data : {len(idx)/len(df_scaled[(df_scaled.predict==n)]):.1%}')\n\n# creating training data\nX = df_scaled.loc[idxs].reset_index(drop=True)\ny = df_scaled.loc[idxs]['predict'].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:00:19.862687Z","iopub.execute_input":"2022-07-21T10:00:19.863095Z","iopub.status.idle":"2022-07-21T10:00:20.001109Z","shell.execute_reply.started":"2022-07-21T10:00:19.863060Z","shell.execute_reply":"2022-07-21T10:00:19.999940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:00:20.463629Z","iopub.execute_input":"2022-07-21T10:00:20.464053Z","iopub.status.idle":"2022-07-21T10:00:20.470413Z","shell.execute_reply.started":"2022-07-21T10:00:20.464018Z","shell.execute_reply":"2022-07-21T10:00:20.469661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = X.iloc[:, :29]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T10:01:22.410156Z","iopub.execute_input":"2022-07-21T10:01:22.410550Z","iopub.status.idle":"2022-07-21T10:01:22.419128Z","shell.execute_reply.started":"2022-07-21T10:01:22.410510Z","shell.execute_reply":"2022-07-21T10:01:22.418279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params_lgb = {\n    'objective': 'multiclass',\n    'boosting': 'gbdt',\n    'learning_rate': 4e-2,\n    'verbosity': -1,\n    'n_jobs': -1,\n    'num_classes': 7,\n    'random_state': 42\n}\n\nfrom sklearn.metrics import balanced_accuracy_score, roc_auc_score\n\ndef get_score(labels, preds, probas):\n    s = (balanced_accuracy_score(labels, preds),\n        roc_auc_score(labels, probas, average=\"weighted\", multi_class=\"ovo\"))\n    return s","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:11:45.254740Z","iopub.execute_input":"2022-07-22T05:11:45.255669Z","iopub.status.idle":"2022-07-22T05:11:45.263991Z","shell.execute_reply.started":"2022-07-22T05:11:45.255626Z","shell.execute_reply":"2022-07-22T05:11:45.262399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nimport lightgbm as lgb\n\nclassif_scores = []\nlgb_predict_proba = 0\n\nsk = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\nfor fold, (train_idx, val_idx) in enumerate(sk.split(X, y)):\n    print(f'====== Fold - {fold} ======')\n    \n    X_train, y_train = X.loc[train_idx], y.loc[train_idx]\n    X_val, y_val = X.loc[val_idx], y.loc[val_idx]\n    \n    lgbm_train = lgb.Dataset(X_train, y_train)\n    lgbm_valid = lgb.Dataset(X_val, y_val)\n    \n    ES = lgb.early_stopping(stopping_rounds=500, verbose=False)\n    \n    model = lgb.train(params=params_lgb,\n                      train_set=lgbm_train,\n                      valid_sets=lgbm_valid,\n                      num_boost_round = 20000,\n                      callbacks=[ES]\n                     )\n    \n    y_prob = model.predict(X_val)\n    y_pred = np.argmax(y_prob, axis=1)\n    \n    s = get_score(y_val, y_pred, y_prob)\n    print(f\"LightGBM   AUC : {s[1]:.3f} | Accuracy : {s[0]:.1%}\")\n    classif_scores.append(s)\n\n    lgb_predict_proba += model.predict(df_scaled.iloc[:, :29]) / 5\n    \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-21T11:16:53.284943Z","iopub.execute_input":"2022-07-21T11:16:53.285356Z","iopub.status.idle":"2022-07-21T11:27:28.489906Z","shell.execute_reply.started":"2022-07-21T11:16:53.285322Z","shell.execute_reply":"2022-07-21T11:27:28.488830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = np.argmax(lgb_predict_proba, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T11:28:10.423109Z","iopub.execute_input":"2022-07-21T11:28:10.423906Z","iopub.status.idle":"2022-07-21T11:28:10.431371Z","shell.execute_reply.started":"2022-07-21T11:28:10.423871Z","shell.execute_reply":"2022-07-21T11:28:10.430093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"score = 0.60","metadata":{}},{"cell_type":"markdown","source":"Nice improvement","metadata":{}},{"cell_type":"markdown","source":"According to this awesome discussion - [Visualizing the seven clusters](https://www.kaggle.com/competitions/tabular-playground-series-jul-2022/discussion/334808)\n\nUseful feature are:-\n ['f_07', 'f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_22', 'f_23', 'f_24', 'f_25', 'f_26', 'f_27', 'f_28']\n \n Let's change data in previous solution \n","metadata":{}},{"cell_type":"code","source":"feats=['f_07', 'f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_22', 'f_23', 'f_24', \n       'f_25', 'f_26', 'f_27', 'f_28']","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:59:47.562906Z","iopub.execute_input":"2022-07-23T03:59:47.563311Z","iopub.status.idle":"2022-07-23T03:59:47.568575Z","shell.execute_reply.started":"2022-07-23T03:59:47.563276Z","shell.execute_reply":"2022-07-23T03:59:47.567752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_comp = 7","metadata":{"execution":{"iopub.status.busy":"2022-07-23T03:59:53.216429Z","iopub.execute_input":"2022-07-23T03:59:53.216848Z","iopub.status.idle":"2022-07-23T03:59:53.221332Z","shell.execute_reply.started":"2022-07-23T03:59:53.216813Z","shell.execute_reply":"2022-07-23T03:59:53.220291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bgm_model = mixture.BayesianGaussianMixture(n_components=n_comp,covariance_type='full',random_state=1,n_init=5,tol=0.001,max_iter=400)\nbgm_model.fit(df_scaled[feats])\nbgm_predict_proba = bgm_model.predict_proba(df_scaled[feats])\nbgm_predict = np.argmax(bgm_predict_proba, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T04:00:25.637266Z","iopub.execute_input":"2022-07-23T04:00:25.637801Z","iopub.status.idle":"2022-07-23T04:04:45.378511Z","shell.execute_reply.started":"2022-07-23T04:00:25.637741Z","shell.execute_reply":"2022-07-23T04:04:45.377093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.round(bgm_model.weights_, 2)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T04:04:45.385449Z","iopub.execute_input":"2022-07-23T04:04:45.389164Z","iopub.status.idle":"2022-07-23T04:04:45.400802Z","shell.execute_reply.started":"2022-07-23T04:04:45.389102Z","shell.execute_reply":"2022-07-23T04:04:45.399571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"proba_threshold = 0.7\n\n#getting corresponding probability value of prediction \ndf_scaled['predict'] = bgm_predict\n\nfor n in range(n_comp):\n    df_scaled[f'predict_proba_{n}'] = bgm_predict_proba[:, n]\n    df_scaled.loc[df_scaled['predict'] == n, 'predict_proba'] = df_scaled[f'predict_proba_{n}']\n    \n\nidxs = np.array([])\nfor n in range(n_comp):\n    idx = df_scaled[(df_scaled.predict==n) & (df_scaled.predict_proba > proba_threshold)].index\n    idxs = np.concatenate((idxs, idx))\n    print(f'Class n{n}  | Training data : {len(idx)/len(df_scaled[(df_scaled.predict==n)]):.1%}')\n\n# creating training data\nX = df_scaled.loc[idxs, feats].reset_index(drop=True)\ny = df_scaled.loc[idxs,'predict'].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T04:05:00.024509Z","iopub.execute_input":"2022-07-23T04:05:00.024945Z","iopub.status.idle":"2022-07-23T04:05:00.172455Z","shell.execute_reply.started":"2022-07-23T04:05:00.024910Z","shell.execute_reply":"2022-07-23T04:05:00.171090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\nimport lightgbm as lgb\n\nparams_lgb = {'learning_rate': 0.07,'objective': 'multiclass','boosting': 'gbdt','verbosity': -1,'n_jobs': -1, 'num_classes':n_comp} \n\nclassif_scores = []\nlgb_predict_proba = 0\n\nsk = StratifiedKFold(n_splits=10, shuffle=True, random_state=42)\n\nfor fold, (train_idx, val_idx) in enumerate(sk.split(X, y)):\n    print(f'====== Fold - {fold} ======')\n    \n    X_train, y_train = X.loc[train_idx], y.loc[train_idx]\n    X_val, y_val = X.loc[val_idx], y.loc[val_idx]\n    \n    lgbm_train = lgb.Dataset(X_train, y_train)\n    lgbm_valid = lgb.Dataset(X_val, y_val)\n    \n    ES = lgb.early_stopping(stopping_rounds=500, verbose=False)\n    LE = lgb.log_evaluation(period=200)\n    \n    model = lgb.train(params=params_lgb,\n                      train_set=lgbm_train,\n                      valid_sets=lgbm_valid,\n                      num_boost_round = 5000,\n                      callbacks=[ES, LE]\n                     )\n    \n    y_prob = model.predict(X_val)\n    y_pred = np.argmax(y_prob, axis=1)\n    \n    s = get_score(y_val, y_pred, y_prob)\n    print(f\"LightGBM   AUC : {s[1]:.3f} | Accuracy : {s[0]:.1%}\")\n    classif_scores.append(s)\n\n    lgb_predict_proba += model.predict(df_scaled[feats]) / 10\n    \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-22T09:38:10.350570Z","iopub.execute_input":"2022-07-22T09:38:10.350974Z","iopub.status.idle":"2022-07-22T10:00:36.973009Z","shell.execute_reply.started":"2022-07-22T09:38:10.350944Z","shell.execute_reply":"2022-07-22T10:00:36.971895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict = np.argmax(lgb_predict_proba, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T10:15:11.258325Z","iopub.execute_input":"2022-07-22T10:15:11.258786Z","iopub.status.idle":"2022-07-22T10:15:11.269231Z","shell.execute_reply.started":"2022-07-22T10:15:11.258750Z","shell.execute_reply":"2022-07-22T10:15:11.268333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Score:- 0.62","metadata":{}},{"cell_type":"markdown","source":"Some Improvement Nice!!","metadata":{}},{"cell_type":"markdown","source":"According to this amazing Notebook by [Outattime](https://www.kaggle.com/code/karlcini/bayesiangmmclassifier/notebook)\nBayesianGMMClassifier is all you need to excel.","metadata":{}},{"cell_type":"code","source":"feats=['f_07', 'f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_22', 'f_23', 'f_24', \n       'f_25', 'f_26', 'f_27', 'f_28']\n\nn_comp=7\n\nbgm_model = mixture.BayesianGaussianMixture(n_components=n_comp,covariance_type='full',random_state=1,n_init=5,tol=0.001,max_iter=400)\nbgm_model.fit(df_scaled[feats])\nbgm_predict_proba = bgm_model.predict_proba(df_scaled[feats])\nbgm_predict = np.argmax(bgm_predict_proba, axis=1)\n\nproba_threshold = 0.7\n\n#getting corresponding probability value of prediction \ndf_scaled['predict'] = bgm_predict\n\nfor n in range(n_comp):\n    df_scaled[f'predict_proba_{n}'] = bgm_predict_proba[:, n]\n    df_scaled.loc[df_scaled['predict'] == n, 'predict_proba'] = df_scaled[f'predict_proba_{n}']\n    \n\nidxs = np.array([])\nfor n in range(n_comp):\n    idx = df_scaled[(df_scaled.predict==n) & (df_scaled.predict_proba > proba_threshold)].index\n    idxs = np.concatenate((idxs, idx))\n    print(f'Class n{n}  | Training data : {len(idx)/len(df_scaled[(df_scaled.predict==n)]):.1%}')\n\n# creating training data\nX = df_scaled.loc[idxs, feats].reset_index(drop=True)\ny = df_scaled.loc[idxs,'predict'].reset_index(drop=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install scikit-lego\nfrom sklego.mixture import BayesianGMMClassifier\nbgm = BayesianGMMClassifier(\n            n_components=7,\n            random_state = 1,\n            tol =1e-3,\n            covariance_type = 'full',\n            max_iter = 200,\n            n_init=3,\n#             init_params='k-means++'\n                     )\n\nbgm.fit(X,y)\npredict = bgm.predict(df_scaled[feats])\npred_seed = bgm.predict_proba(df_scaled[feats])","metadata":{"execution":{"iopub.status.busy":"2022-07-23T04:32:39.694585Z","iopub.execute_input":"2022-07-23T04:32:39.695462Z","iopub.status.idle":"2022-07-23T04:35:07.126270Z","shell.execute_reply.started":"2022-07-23T04:32:39.695414Z","shell.execute_reply":"2022-07-23T04:35:07.124835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Score :- 0.735","metadata":{}},{"cell_type":"markdown","source":"Extraordinary improvement!!\n\nOutattime also suggested for improvement one can integration and/or ensembling with other classifiers.","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:13:12.148738Z","iopub.execute_input":"2022-07-23T05:13:12.149143Z","iopub.status.idle":"2022-07-23T05:13:12.154953Z","shell.execute_reply.started":"2022-07-23T05:13:12.149113Z","shell.execute_reply":"2022-07-23T05:13:12.153870Z"}}},{"cell_type":"markdown","source":"Next stop will be ensembeling!!!","metadata":{"execution":{"iopub.status.busy":"2022-07-23T05:15:20.344544Z","iopub.execute_input":"2022-07-23T05:15:20.345016Z","iopub.status.idle":"2022-07-23T05:15:20.352233Z","shell.execute_reply.started":"2022-07-23T05:15:20.344977Z","shell.execute_reply":"2022-07-23T05:15:20.351085Z"}}},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")\nsub['Predicted'] = predict\nsub.to_csv(\"submission12.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T04:38:03.051474Z","iopub.execute_input":"2022-07-23T04:38:03.051928Z","iopub.status.idle":"2022-07-23T04:38:03.241967Z","shell.execute_reply.started":"2022-07-23T04:38:03.051881Z","shell.execute_reply":"2022-07-23T04:38:03.240961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Helpful Notebooks:-\n* [Simple soft voting](https://www.kaggle.com/code/pourchot/simple-soft-voting)\n* [TPS Jul 2022 unsupervised and supervised learning](https://www.kaggle.com/code/hiro5299834/tps-jul-2022-unsupervised-and-supervised-learning)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T07:21:46.956867Z","iopub.execute_input":"2022-07-21T07:21:46.957575Z","iopub.status.idle":"2022-07-21T07:21:46.966380Z","shell.execute_reply.started":"2022-07-21T07:21:46.957537Z","shell.execute_reply":"2022-07-21T07:21:46.965513Z"}}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}