{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### **Credits and References:**\n\nThis is an update of the work made by Bizen :\nhttps://www.kaggle.com/code/hiro5299834/tps-jul-2022-unsupervised-and-supervised-learning\n\nwhich was also based upon :\n* https://www.kaggle.com/code/adaubas/tps-jul22-lgbm-extratree-qda-soft-voting  \n* https://www.kaggle.com/code/pourchot/simple-soft-voting  \n* https://www.kaggle.com/code/ricopue/tps-jul22-clusters-and-lgb  \n* https://www.kaggle.com/code/ambrosm/tpsjul22-gaussian-mixture-cluster-analysis  \n* https://www.kaggle.com/code/thedevastator/how-to-ensemble-clustering-algorithms-updated  \n* https://www.kaggle.com/code/eduus710/getting-cluster-ensembles-to-work  \n* https://www.kaggle.com/code/plarmuseau/bruteforce-clustering  \n* https://www.kaggle.com/code/thedevastator/bruteforce-clustering  \n\n# updates :\n* Neural Network added\n* Optimization function for the weighted soft voting\n\n","metadata":{"papermill":{"duration":0.00598,"end_time":"2022-07-17T01:28:31.03507","exception":false,"start_time":"2022-07-17T01:28:31.02909","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Librairies","metadata":{"papermill":{"duration":0.004566,"end_time":"2022-07-17T01:28:31.044567","exception":false,"start_time":"2022-07-17T01:28:31.040001","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport random\nimport os\nimport gc\n\n\nimport xgboost as xgb\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import PowerTransformer\nfrom sklearn.metrics import silhouette_score, calinski_harabasz_score, davies_bouldin_score,balanced_accuracy_score, roc_auc_score\nfrom sklearn.mixture import BayesianGaussianMixture\nfrom sklearn.ensemble import ExtraTreesClassifier \nfrom sklearn.discriminant_analysis import QuadraticDiscriminantAnalysis\nfrom sklearn.svm import SVC\nfrom sklearn.neighbors import KNeighborsClassifier\n\n\nfrom scipy.optimize import dual_annealing\nfrom numpy.random import rand\n \n\npd.options.display.max_rows = 100\npd.options.display.max_columns = 100","metadata":{"papermill":{"duration":4.622558,"end_time":"2022-07-17T01:28:35.671943","exception":false,"start_time":"2022-07-17T01:28:31.049385","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-17T12:47:05.497166Z","iopub.execute_input":"2022-07-17T12:47:05.497633Z","iopub.status.idle":"2022-07-17T12:47:05.508395Z","shell.execute_reply.started":"2022-07-17T12:47:05.497593Z","shell.execute_reply":"2022-07-17T12:47:05.506999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Parameters","metadata":{"papermill":{"duration":0.004769,"end_time":"2022-07-17T01:28:35.682617","exception":false,"start_time":"2022-07-17T01:28:35.677848","status":"completed"},"tags":[]}},{"cell_type":"code","source":"class CFG:\n    seed_bgm = 1\n    seed = 42\n    n_splits = 10\n    n_clusters = 7\n    threshold = 0.7","metadata":{"papermill":{"duration":0.013624,"end_time":"2022-07-17T01:28:35.701009","exception":false,"start_time":"2022-07-17T01:28:35.687385","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-17T12:43:41.148398Z","iopub.execute_input":"2022-07-17T12:43:41.148879Z","iopub.status.idle":"2022-07-17T12:43:41.155760Z","shell.execute_reply.started":"2022-07-17T12:43:41.148844Z","shell.execute_reply":"2022-07-17T12:43:41.154364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=42):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)","metadata":{"papermill":{"duration":0.013081,"end_time":"2022-07-17T01:28:35.718972","exception":false,"start_time":"2022-07-17T01:28:35.705891","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-17T12:43:41.593936Z","iopub.execute_input":"2022-07-17T12:43:41.594421Z","iopub.status.idle":"2022-07-17T12:43:41.601288Z","shell.execute_reply.started":"2022-07-17T12:43:41.594380Z","shell.execute_reply":"2022-07-17T12:43:41.599841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{"papermill":{"duration":0.005286,"end_time":"2022-07-17T01:28:35.729548","exception":false,"start_time":"2022-07-17T01:28:35.724262","status":"completed"},"tags":[]}},{"cell_type":"code","source":"df = pd.read_csv(\"../input/tabular-playground-series-jul-2022/data.csv\").drop('id', axis=1)\ndf.head()","metadata":{"papermill":{"duration":1.082404,"end_time":"2022-07-17T01:28:36.816805","exception":false,"start_time":"2022-07-17T01:28:35.734401","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-17T12:43:44.025408Z","iopub.execute_input":"2022-07-17T12:43:44.025864Z","iopub.status.idle":"2022-07-17T12:43:45.494993Z","shell.execute_reply.started":"2022-07-17T12:43:44.025828Z","shell.execute_reply":"2022-07-17T12:43:45.493670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_scores = []\nbest_features = [f\"f_{i:02d}\" for i in list(range(7, 14)) + list(range(22, 29))]\nfeatures = df.columns\n\ndef scores(preds, lib, df=df[best_features], verbose=True, compute_silhouette=None): \n    # Silhouette is very slow\n    sil = 0\n    if compute_silhouette:\n        sil = silhouette_score(df, preds, metric='euclidean')\n    \n    s = (lib,\n         sil, \n         calinski_harabasz_score(df, preds), \n         davies_bouldin_score(df, preds))\n    \n    if verbose:\n        print(f\"{s[0]} : Silhouette : {s[1]:.1%} | Calinski Harabasz : {s[2]:.1f} | Davis Bouldin : {s[3]:.3f}\")\n        \n    return s","metadata":{"papermill":{"duration":0.017779,"end_time":"2022-07-17T01:28:36.840036","exception":false,"start_time":"2022-07-17T01:28:36.822257","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-17T12:43:47.315838Z","iopub.execute_input":"2022-07-17T12:43:47.317162Z","iopub.status.idle":"2022-07-17T12:43:47.330927Z","shell.execute_reply.started":"2022-07-17T12:43:47.317072Z","shell.execute_reply":"2022-07-17T12:43:47.329388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bayesian Gaussian Mixture","metadata":{"papermill":{"duration":0.005041,"end_time":"2022-07-17T01:28:36.850182","exception":false,"start_time":"2022-07-17T01:28:36.845141","status":"completed"},"tags":[]}},{"cell_type":"code","source":"seed_everything(CFG.seed_bgm)\ndf_scaled = pd.DataFrame(PowerTransformer().fit_transform(df[features]), columns=features)\n\nBGM = BayesianGaussianMixture(n_components=CFG.n_clusters, covariance_type='full', random_state=CFG.seed_bgm, max_iter=300, n_init=1, tol=1e-3)\nBGM.fit(df_scaled[best_features])\n\nBGM_predict_proba = BGM.predict_proba(df_scaled[best_features])\nBGM_predict = np.argmax(BGM_predict_proba, axis=1)\n\nall_scores.append(scores(BGM_predict, lib=\"BayesianGaussianMixture after powertransformer\"))","metadata":{"papermill":{"duration":139.612692,"end_time":"2022-07-17T01:30:56.468036","exception":false,"start_time":"2022-07-17T01:28:36.855344","status":"completed"},"scrolled":true,"tags":[],"execution":{"iopub.status.busy":"2022-07-17T12:43:50.870754Z","iopub.execute_input":"2022-07-17T12:43:50.871391Z","iopub.status.idle":"2022-07-17T12:45:04.593159Z","shell.execute_reply.started":"2022-07-17T12:43:50.871348Z","shell.execute_reply":"2022-07-17T12:45:04.588772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Trusted Data","metadata":{"papermill":{"duration":0.019955,"end_time":"2022-07-17T01:30:56.511106","exception":false,"start_time":"2022-07-17T01:30:56.491151","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# get trusted data to train LGB model.\nproba_threshold = CFG.threshold\n\ndf_scaled['predict'] = BGM_predict\ndf_scaled['predict_proba'] = 0\nfor n in range(CFG.n_clusters):\n    df_scaled[f'predict_proba_{n}'] = BGM_predict_proba[:, n]\n    df_scaled.loc[df_scaled['predict']==n, 'predict_proba'] = df_scaled[f'predict_proba_{n}']\n    \n    \nidxs = np.array([])\nfor n in range(CFG.n_clusters):\n    median = df_scaled[df_scaled.predict==n]['predict_proba'].median()\n    idx = df_scaled[(df_scaled.predict==n) & (df_scaled.predict_proba > proba_threshold)].index\n    idxs = np.concatenate((idxs, idx))\n    print(f'Class n{n}  |  Median : {median:.4f}  |  Training data : {len(idx)/len(df_scaled[(df_scaled.predict==n)]):.1%}')\n    \nX = df_scaled.loc[idxs][best_features].reset_index(drop=True)\ny = df_scaled.loc[idxs]['predict'].reset_index(drop=True)","metadata":{"papermill":{"duration":0.213867,"end_time":"2022-07-17T01:30:56.743299","exception":false,"start_time":"2022-07-17T01:30:56.529432","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-17T12:45:04.596154Z","iopub.execute_input":"2022-07-17T12:45:04.597152Z","iopub.status.idle":"2022-07-17T12:45:04.838510Z","shell.execute_reply.started":"2022-07-17T12:45:04.597081Z","shell.execute_reply":"2022-07-17T12:45:04.837208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Supervised learning\n### Neural Network / XGBoost / ExtraTreesClassifier / QuadraticDiscriminantAnalysis / SVC / KNeighborsClassifier","metadata":{"papermill":{"duration":0.009837,"end_time":"2022-07-17T01:30:56.821233","exception":false,"start_time":"2022-07-17T01:30:56.811396","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def get_score(labels, preds, probas):\n    s = (balanced_accuracy_score(labels, preds),\n        roc_auc_score(labels, probas, average=\"weighted\", multi_class=\"ovo\"))\n    return s","metadata":{"papermill":{"duration":0.016814,"end_time":"2022-07-17T01:30:56.845643","exception":false,"start_time":"2022-07-17T01:30:56.828829","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-17T12:45:04.871655Z","iopub.execute_input":"2022-07-17T12:45:04.872005Z","iopub.status.idle":"2022-07-17T12:45:04.878193Z","shell.execute_reply.started":"2022-07-17T12:45:04.871971Z","shell.execute_reply":"2022-07-17T12:45:04.877243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Parameters","metadata":{}},{"cell_type":"code","source":"params_xgb = {\n    'booster': 'gbtree',\n    'objective': 'multi:softprob',\n    'learning_rate': 4e-2,\n    'num_class': CFG.n_clusters,\n    'seed': CFG.seed,\n    'gpu_id': 0,\n    'tree_method': 'gpu_hist',\n    'predictor': 'gpu_predictor'\n    }\n\n\n","metadata":{"papermill":{"duration":0.013247,"end_time":"2022-07-17T01:30:56.86496","exception":false,"start_time":"2022-07-17T01:30:56.851713","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-17T12:45:04.879723Z","iopub.execute_input":"2022-07-17T12:45:04.880673Z","iopub.status.idle":"2022-07-17T12:45:04.889285Z","shell.execute_reply.started":"2022-07-17T12:45:04.880623Z","shell.execute_reply":"2022-07-17T12:45:04.888269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seed_everything(CFG.seed)\n\nxgb_predict_proba = 0\netc_predict_proba = 0\nqda_predict_proba = 0\nsvc_predict_proba = 0\nknc_predict_proba = 0\n\nclassif_scores = []\n\nskf = StratifiedKFold(CFG.n_splits, shuffle=True, random_state=CFG.seed)\n\nfor fold, (trn_idx, val_idx) in enumerate(skf.split(X, y)):\n    print(f\"===== fold{fold} =====\")\n    X_train, y_train = X.iloc[trn_idx], y.iloc[trn_idx]\n    X_valid, y_valid = X.iloc[val_idx], y.iloc[val_idx]\n    \n        \n    # XGBoost\n    xgb_train = xgb.DMatrix(X_train, label=y_train)\n    xgb_valid = xgb.DMatrix(X_valid, label=y_valid)\n\n    model = xgb.train(params_xgb,\n                      dtrain=xgb_train,\n                      evals=[(xgb_train, 'train'),(xgb_valid, 'eval')],\n                      verbose_eval=False,\n                      num_boost_round=20000,\n                      early_stopping_rounds=200,\n                     )\n    \n    y_pred_proba = model.predict(xgb_valid, iteration_range=(0, model.best_ntree_limit))\n    y_pred = np.argmax(y_pred_proba, axis=1)\n    \n    s = get_score(y_valid, y_pred, y_pred_proba)\n    print(f\"XGBoost    AUC : {s[1]:.3f} | Accuracy : {s[0]:.1%}\")\n    classif_scores.append(s)\n\n    xgb_predict_proba += model.predict(\n        xgb.DMatrix(df_scaled[best_features]),\n        iteration_range=(0, model.best_ntree_limit)\n    ) / CFG.n_splits\n    \n    del xgb_train, xgb_valid, model, s, y_pred, y_pred_proba\n    gc.collect()\n    \n    # ExtraTreesClassifier\n    model = ExtraTreesClassifier(n_estimators=1000, random_state=CFG.seed)\n    model.fit(X_train, y_train)\n    \n    y_pred = model.predict(X_valid)\n    y_pred_proba = model.predict_proba(X_valid)\n    \n    s = get_score(y_valid, y_pred, y_pred_proba)\n    print(f\"ExtraTree  AUC : {s[1]:.3f} | Accuracy : {s[0]:.1%}\")\n    classif_scores.append(s)\n\n    etc_predict_proba += model.predict_proba(df_scaled[best_features]) / CFG.n_splits\n\n    del model, s, y_pred, y_pred_proba\n    gc.collect()\n    \n    # QuadraticDiscriminantAnalysis\n    model = QuadraticDiscriminantAnalysis(priors=CFG.n_clusters)\n    model.fit(X_train, y_train) # on trusted data only\n    \n    y_pred = model.predict(X_valid)\n    y_pred_proba = model.predict_proba(X_valid)\n    \n    s = get_score(y_valid, y_pred, y_pred_proba)\n    print(f\"QDA        AUC : {s[1]:.3f} | Accuracy : {s[0]:.1%}\")\n    classif_scores.append(s)\n\n    qda_predict_proba += model.predict_proba(df_scaled[best_features]) / CFG.n_splits\n\n    del model, s, y_pred, y_pred_proba\n    gc.collect()\n\n    # SVC\n    model = SVC(probability=True)\n    model.fit(X_train, y_train) # on trusted data only\n    \n    y_pred = model.predict(X_valid)\n    y_pred_proba = model.predict_proba(X_valid)\n    \n    s = get_score(y_valid, y_pred, y_pred_proba)\n    print(f\"SVC        AUC : {s[1]:.3f} | Accuracy : {s[0]:.1%}\")\n    classif_scores.append(s)\n\n    svc_predict_proba += model.predict_proba(df_scaled[best_features]) / CFG.n_splits\n\n    del model, s, y_pred, y_pred_proba\n    gc.collect()\n\n    # KNeighborsClassifier\n    model = KNeighborsClassifier(n_neighbors=20)\n    model.fit(X_train, y_train) # on trusted data only\n    \n    y_pred = model.predict(X_valid)\n    y_pred_proba = model.predict_proba(X_valid)\n    \n    s = get_score(y_valid, y_pred, y_pred_proba)\n    print(f\"KNeighbors AUC : {s[1]:.3f} | Accuracy : {s[0]:.1%}\")\n    classif_scores.append(s)\n\n    knc_predict_proba += model.predict_proba(df_scaled[best_features]) / CFG.n_splits\n\n    del model, s, y_pred, y_pred_proba\n    gc.collect()","metadata":{"papermill":{"duration":1465.264169,"end_time":"2022-07-17T01:55:22.135252","exception":false,"start_time":"2022-07-17T01:30:56.871083","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-17T12:47:10.508877Z","iopub.execute_input":"2022-07-17T12:47:10.509492Z","iopub.status.idle":"2022-07-17T12:47:10.589569Z","shell.execute_reply.started":"2022-07-17T12:47:10.509440Z","shell.execute_reply":"2022-07-17T12:47:10.588517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Soft voting","metadata":{"papermill":{"duration":0.017768,"end_time":"2022-07-17T01:55:22.17324","exception":false,"start_time":"2022-07-17T01:55:22.155472","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def soft_voting(preds_probas, weights):\n    pred_test = np.zeros((df.shape[0], CFG.n_clusters))\n    \n    for i, (p, w) in enumerate(zip(preds_probas, weights)):\n        preds = np.argmax(p, axis=1)\n        pred_idx = pd.Series(preds).value_counts().index.tolist()\n        pred_test += p[:, pred_idx] * w\n    \n    return np.argmax(pred_test, axis=1)","metadata":{"papermill":{"duration":0.03384,"end_time":"2022-07-17T01:55:22.225915","exception":false,"start_time":"2022-07-17T01:55:22.192075","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-17T10:53:09.964327Z","iopub.execute_input":"2022-07-17T10:53:09.965391Z","iopub.status.idle":"2022-07-17T10:53:09.972817Z","shell.execute_reply.started":"2022-07-17T10:53:09.965351Z","shell.execute_reply":"2022-07-17T10:53:09.971394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Optimization","metadata":{"papermill":{"duration":0.019794,"end_time":"2022-07-17T01:56:57.036763","exception":false,"start_time":"2022-07-17T01:56:57.016969","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"We look for an optimization of the blending. The metric chosen is davies_bouldin to decide what are the best blending (maybe there is a better one...)","metadata":{}},{"cell_type":"code","source":"for ITER in [25,50,75] :\n\n    # the weight evaluation function for weighting classifiers (blending coef) :\n    def objective(w):\n        w1, w2, w3, w4, w5 = w\n\n        sv_predict = soft_voting(\n        [\n         xgb_predict_proba, \n         etc_predict_proba, \n         qda_predict_proba, \n         svc_predict_proba, \n         knc_predict_proba],\n        [w1, w2, w3, w4, w5])\n        \n        score = davies_bouldin_score(df, sv_predict)\n\n        return score\n\n    # the set of possible values of the variables (definition domain) :\n    bounds = [[1, 100], \n              [1, 100],  \n              [1, 100], \n              [1, 100], \n              [1, 100]]\n\n    # the Dual Annealing optimization :\n    result = dual_annealing(objective, bounds, maxiter = ITER, seed = 42 )\n\n    # results:\n    print(f'\\n------------- maxiter:{ITER} -------------\\n')\n    print('Success :', result['success'])\n    print('Total Evaluations: %d' % result['nfev'])\n\n    solution = result['x']\n    evaluation = objective(solution)\n    print('Solution Weights : {} for minimum davies_bouldin_score = {}'.format((result['x'].astype(int)), evaluation))\n\n    sv_predict = soft_voting(\n    [\n     xgb_predict_proba, \n     etc_predict_proba, \n     qda_predict_proba, \n     svc_predict_proba, \n     knc_predict_proba],\n    solution.tolist())\n\n    sub = pd.read_csv(\"../input/tabular-playground-series-jul-2022/sample_submission.csv\")\n    sub['Predicted'] = sv_predict\n    sub.to_csv(f\"submission_iter-{ITER}.csv\", index=False)","metadata":{"papermill":{"duration":0.010068,"end_time":"2022-07-17T01:56:57.261464","exception":false,"start_time":"2022-07-17T01:56:57.251396","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-17T10:57:15.553942Z","iopub.execute_input":"2022-07-17T10:57:15.554751Z","iopub.status.idle":"2022-07-17T10:59:28.018193Z","shell.execute_reply.started":"2022-07-17T10:57:15.554717Z","shell.execute_reply":"2022-07-17T10:59:28.017233Z"},"trusted":true},"execution_count":null,"outputs":[]}]}