{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom skopt import gp_minimize\nfrom skopt.space import Categorical, Integer\nfrom sklearn.metrics import f1_score\nimport matplotlib.pyplot as plt\n\nSEED = 2023\n\ndef get_best_threshold_and_score(ts, ps, start=0.2, end=0.8, step=0.01, n_calls=100):\n\n    def f(t):\n        _ps = (ps > t).astype(\"int\")\n        return -f1_score(ts, _ps, average=\"macro\")\n\n    res = gp_minimize(\n      f,                  # the function to minimize\n      [(.2, .8)],      # the bounds on each dimension of x\n      acq_func=\"EI\",      # the acquisition function\n      n_calls=n_calls,         # the number of evaluations of f\n      n_random_starts=3,  # the number of random initialization points\n      random_state=SEED,\n     )   # the random seed\n\n    bt = res.x[0]\n    s = res.fun\n    ts = [xs[0] for xs in res.x_iters]\n    ss = [-xs for xs in res.func_vals]\n    return bt, -s, ts, ss\n\n\ndef get_thres_and_plot(y_true, y_pred):\n    best_threshold, best_score, thresholds, scores = get_best_threshold_and_score(\n        y_true, y_pred, n_calls=50,\n    )\n\n    print(best_threshold, best_score)\n\n    plt.figure(figsize=(20, 5))\n    plt.scatter(thresholds, scores, color=\"blue\")\n    plt.scatter([best_threshold], [best_score], color=\"red\")\n    plt.xlabel(\"Threshold\", size=14)\n    plt.ylabel(\"Validation F1 Score\", size=14)\n    plt.title(\n        f\"Threshold vs. F1_Score with Best F1_Score={best_score:.5f} at Best Threshold = {best_threshold:.5f}\",\n        size=18,\n    )\n    plt.show()\n    return best_threshold, best_score\n\ndef time_feature(train):\n    train[\"year\"] = (\n        train[\"session_id\"].apply(lambda x: int(str(x)[:2])).astype(np.uint8)\n    )\n    train[\"month\"] = (\n        train[\"session_id\"].apply(lambda x: int(str(x)[2:4]) + 1).astype(np.uint8)\n    )\n    train[\"day\"] = (\n        train[\"session_id\"].apply(lambda x: int(str(x)[4:6])).astype(np.uint8)\n    )\n    train[\"hour\"] = (\n        train[\"session_id\"].apply(lambda x: int(str(x)[6:8])).astype(np.uint8)\n    )\n    train[\"minute\"] = (\n        train[\"session_id\"].apply(lambda x: int(str(x)[8:10])).astype(np.uint8)\n    )\n    train[\"second\"] = (\n        train[\"session_id\"].apply(lambda x: int(str(x)[10:12])).astype(np.uint8)\n    )\n\n    return train","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-28T09:37:40.138291Z","iopub.execute_input":"2023-06-28T09:37:40.138722Z","iopub.status.idle":"2023-06-28T09:37:42.180708Z","shell.execute_reply.started":"2023-06-28T09:37:40.138692Z","shell.execute_reply":"2023-06-28T09:37:42.179732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"catb1_base = \"/kaggle/input/psp-catboost-sort-by-time-only\"\ncatb2_base = \"/kaggle/input/xgboost-sort-by-time\"\n\ncatb3_base = \"/kaggle/input/psp-catb-1-model-18-qs/catb 17-06-2023/catb 17-06-2023\"\ncatb4_base = \"/kaggle/input/psp-catb-1-model-18-qs/xgb 17-06-2023/xgb 17-06-2023\"\n\ncatb5_base = \"/kaggle/input/psp-catb-1-model-18-qs/xgb 20-06-2023/xgb 20-06-2023\" # model tuning from 17-06\ncatb6_base = \"/kaggle/input/psp-catb-1-model-18-qs/xgb 23-06-2023/xgb 23-06-2023\" # 10-fold\ncatb7_base = \"/kaggle/input/psp-catb-1-model-18-qs/xgb 24-06-2023/xgb 24-06-2023\" # 10-fold continue training\n\n\nstd_grp_catb = \"/kaggle/input/psp-catboost-std-retrain\"\n\nstd_18in1_catb = \"/kaggle/input/psp-catb-1-model-18-qs/std-catb-26-06-2023/std-catb-26-06-2023\"\nstd_18in1_xgb = \"/kaggle/input/psp-catb-1-model-18-qs/std-xgb-26-06-2023/std-xgb-26-06-2023\"\nstd_18in1_xgb_2 = \"/kaggle/input/psp-catb-1-model-18-qs/std-xgb-28-06-2023/std-xgb-28-06-2023\"\n\n\nyyykrk_xgb1 = \"/kaggle/input/20230610-011236\"\nyyykrk_lgb1 = \"/kaggle/input/20230610-021710\"","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:37:42.182634Z","iopub.execute_input":"2023-06-28T09:37:42.183010Z","iopub.status.idle":"2023-06-28T09:37:42.189182Z","shell.execute_reply.started":"2023-06-28T09:37:42.182981Z","shell.execute_reply":"2023-06-28T09:37:42.188099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NN_df1 = pd.read_csv(\"/kaggle/input/public-nn-v32/oof.csv\",index_col=0)\nNN_df1 = NN_df1.stack().reset_index()\nNN_df1.columns = ['session_id','q','NN_pred1']\nNN_df1['q'] = NN_df1['q'].astype(int) + 1","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:37:42.191207Z","iopub.execute_input":"2023-06-28T09:37:42.193558Z","iopub.status.idle":"2023-06-28T09:37:42.600824Z","shell.execute_reply.started":"2023-06-28T09:37:42.193518Z","shell.execute_reply":"2023-06-28T09:37:42.599839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NN_df2 = pd.read_csv(\"/kaggle/input/public-nn-v46/oof.csv\",index_col=0)\nNN_df2 = NN_df2.stack().reset_index()\nNN_df2.columns = ['session_id','q','NN_pred2']\nNN_df2['q'] = NN_df2['q'].astype(int) + 1","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:37:42.603353Z","iopub.execute_input":"2023-06-28T09:37:42.604020Z","iopub.status.idle":"2023-06-28T09:37:42.949708Z","shell.execute_reply.started":"2023-06-28T09:37:42.603986Z","shell.execute_reply":"2023-06-28T09:37:42.948691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NN_df3 = pd.read_csv(\"/kaggle/input/public-nn-v47/oof.csv\",index_col=0)\nNN_df3 = NN_df3.stack().reset_index()\nNN_df3.columns = ['session_id','q','NN_pred3']\nNN_df3['q'] = NN_df3['q'].astype(int) + 1","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:37:42.951136Z","iopub.execute_input":"2023-06-28T09:37:42.951469Z","iopub.status.idle":"2023-06-28T09:37:43.290173Z","shell.execute_reply.started":"2023-06-28T09:37:42.951441Z","shell.execute_reply":"2023-06-28T09:37:43.288991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NN_df4 = pd.read_csv(\"/kaggle/input/public-nn-v48/oof.csv\",index_col=0)\nNN_df4 = NN_df4.stack().reset_index()\nNN_df4.columns = ['session_id','q','NN_pred4']\nNN_df4['q'] = NN_df4['q'].astype(int) + 1","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:37:43.291501Z","iopub.execute_input":"2023-06-28T09:37:43.291825Z","iopub.status.idle":"2023-06-28T09:37:43.618830Z","shell.execute_reply.started":"2023-06-28T09:37:43.291798Z","shell.execute_reply":"2023-06-28T09:37:43.617518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NN_df5 = pd.read_csv(\"/kaggle/input/public-nn-v51/oof.csv\",index_col=0)\nNN_df5 = NN_df5.stack().reset_index()\nNN_df5.columns = ['session_id','q','NN_pred5']\nNN_df5['q'] = NN_df5['q'].astype(int) + 1","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:37:43.620055Z","iopub.execute_input":"2023-06-28T09:37:43.620375Z","iopub.status.idle":"2023-06-28T09:37:43.917135Z","shell.execute_reply.started":"2023-06-28T09:37:43.620347Z","shell.execute_reply":"2023-06-28T09:37:43.916040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NN_df6 = pd.read_csv(\"/kaggle/input/public-nn-v52/oof.csv\",index_col=0)\nNN_df6 = NN_df6.stack().reset_index()\nNN_df6.columns = ['session_id','q','NN_pred6']\nNN_df6['q'] = NN_df6['q'].astype(int) + 1","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:37:43.918536Z","iopub.execute_input":"2023-06-28T09:37:43.918862Z","iopub.status.idle":"2023-06-28T09:37:44.214817Z","shell.execute_reply.started":"2023-06-28T09:37:43.918833Z","shell.execute_reply":"2023-06-28T09:37:44.213800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NN_df7 = pd.read_csv(\"/kaggle/input/public-nn-v53/oof.csv\",index_col=0)\nNN_df7 = NN_df7.stack().reset_index()\nNN_df7.columns = ['session_id','q','NN_pred7']\nNN_df7['q'] = NN_df7['q'].astype(int) + 1","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:37:44.216186Z","iopub.execute_input":"2023-06-28T09:37:44.216477Z","iopub.status.idle":"2023-06-28T09:37:44.515443Z","shell.execute_reply.started":"2023-06-28T09:37:44.216452Z","shell.execute_reply":"2023-06-28T09:37:44.514241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_qs = [\n    ('0-4', 1, 4),\n    ('5-12', 4, 14),\n    ('13-22', 14, 19),\n]\n\nval_dfs = []\nfor grp_idx, (grp, a, b) in enumerate(trn_qs):\n    for fold in range(5):\n        val_dfs.append(\n            pd.read_csv(f\"{catb1_base}/val_preds_fold_{fold}_grp_{grp}_final.csv\")\n        )\ncatb1_df = pd.concat(val_dfs)\n\nval_dfs = []\nfor grp_idx, (grp, a, b) in enumerate(trn_qs):\n    for fold in range(5):\n        val_dfs.append(\n            pd.read_csv(f\"{catb2_base}/val_preds_fold_{fold}_grp_{grp}_final.csv\")\n        )\ncatb2_df = pd.concat(val_dfs)\n\n\nval_dfs = []\nfor fold in range(5):\n    val_dfs.append(\n        pd.read_csv(f\"{catb3_base}/val_preds_fold_{fold}_final.csv\")\n    )\ncatb3_df = pd.concat(val_dfs)\n\nval_dfs = []\nfor fold in range(5):\n    val_dfs.append(\n        pd.read_csv(f\"{catb4_base}/val_preds_fold_{fold}_final.csv\")\n    )\ncatb4_df = pd.concat(val_dfs)\n\nval_dfs = []\nfor fold in range(5):\n    val_dfs.append(\n        pd.read_csv(f\"{catb5_base}/val_preds_fold_{fold}_final.csv\")\n    )\ncatb5_df = pd.concat(val_dfs)\n\nval_dfs = []\nfor fold in range(10):\n    val_dfs.append(\n        pd.read_csv(f\"{catb6_base}/val_preds_fold_{fold}_final.csv\")\n    )\ncatb6_df = pd.concat(val_dfs)\n\nval_dfs = []\nfor fold in range(10):\n    val_dfs.append(\n        pd.read_csv(f\"{catb7_base}/val_preds_fold_{fold}_final.csv\")\n    )\ncatb7_df = pd.concat(val_dfs)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:37:44.518396Z","iopub.execute_input":"2023-06-28T09:37:44.518725Z","iopub.status.idle":"2023-06-28T09:37:49.847799Z","shell.execute_reply.started":"2023-06-28T09:37:44.518697Z","shell.execute_reply":"2023-06-28T09:37:49.846725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_dfs = []\nfor grp_idx, (grp, a, b) in enumerate(trn_qs):\n    for fold in range(5):\n        val_dfs.append(\n            pd.read_csv(f\"{std_grp_catb}/val_preds_fold_{fold}_grp_{grp}_final.csv\")\n        )\nstd_grp_catb_df = pd.concat(val_dfs)\n\n\nval_dfs = []\nfor fold in range(5):\n    val_dfs.append(\n        pd.read_csv(f\"{std_18in1_catb}/val_preds_fold_{fold}_final.csv\")\n    )\nstd_18in1_catb_df = pd.concat(val_dfs)\n\n\nval_dfs = []\nfor fold in range(5):\n    val_dfs.append(\n        pd.read_csv(f\"{std_18in1_xgb}/val_preds_fold_{fold}_final.csv\")\n    )\nstd_18in1_xgb_df = pd.concat(val_dfs)\n\nval_dfs = []\nfor fold in range(5):\n    val_dfs.append(\n        pd.read_csv(f\"{std_18in1_xgb_2}/val_preds_fold_{fold}_final.csv\")\n    )\nstd_18in1_xgb_2_df = pd.concat(val_dfs)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:37:49.852464Z","iopub.execute_input":"2023-06-28T09:37:49.852806Z","iopub.status.idle":"2023-06-28T09:37:53.015875Z","shell.execute_reply.started":"2023-06-28T09:37:49.852778Z","shell.execute_reply":"2023-06-28T09:37:53.014947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yyykrk_xgb1_df = pd.read_csv(f\"{yyykrk_xgb1}/oof.csv\")\nyyykrk_lgb1_df = pd.read_csv(f\"{yyykrk_lgb1}/oof.csv\")\n\n\ndef cleaning_yyykrk_df(df, pred_name):\n    cols_rename = {\"Unnamed: 0\": \"session_id\"}\n    for q in range(0,18):\n        cols_rename[str(q)] = str(q+1)\n    df = df.rename(columns=cols_rename)\n    df = df.set_index(\"session_id\")\n    df = df.stack().reset_index()\n    df = df.rename(columns={\"level_1\": \"q\", 0: f\"{pred_name}_pred\"})\n    df[\"q\"] = df[\"q\"].astype(int)\n    return df\n    df.head()\n    \nyyykrk_xgb1_df = cleaning_yyykrk_df(yyykrk_xgb1_df, \"yyykrk_xgb1\")\nyyykrk_lgb1_df = cleaning_yyykrk_df(yyykrk_lgb1_df, \"yyykrk_lgb1\")","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:37:53.017691Z","iopub.execute_input":"2023-06-28T09:37:53.018027Z","iopub.status.idle":"2023-06-28T09:37:53.650965Z","shell.execute_reply.started":"2023-06-28T09:37:53.017999Z","shell.execute_reply":"2023-06-28T09:37:53.649798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"catb1_df[\"cat1_pred\"] = catb1_df[\"pred\"]\ncatb2_df[\"cat2_pred\"] = catb2_df[\"pred\"]\ncatb3_df[\"cat3_pred\"] = catb3_df[\"pred\"]\ncatb4_df[\"cat4_pred\"] = catb4_df[\"pred\"]\ncatb5_df[\"cat5_pred\"] = catb5_df[\"pred\"]\ncatb6_df[\"cat6_pred\"] = catb6_df[\"pred\"]\ncatb7_df[\"cat7_pred\"] = catb7_df[\"pred\"]","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:37:53.652231Z","iopub.execute_input":"2023-06-28T09:37:53.652554Z","iopub.status.idle":"2023-06-28T09:37:53.676411Z","shell.execute_reply.started":"2023-06-28T09:37:53.652526Z","shell.execute_reply":"2023-06-28T09:37:53.675190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"std_grp_catb_df[\"std_grp_catb_pred\"] = std_grp_catb_df[\"pred\"]\nstd_18in1_catb_df[\"std_18in1_catb_pred\"] = std_18in1_catb_df[\"pred\"]\nstd_18in1_xgb_df[\"std_18in1_xgb_pred\"] = std_18in1_xgb_df[\"pred\"]\nstd_18in1_xgb_2_df[\"std_18in1_xgb_2_pred\"] = std_18in1_xgb_2_df[\"pred\"]","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:37:53.677718Z","iopub.execute_input":"2023-06-28T09:37:53.678055Z","iopub.status.idle":"2023-06-28T09:37:53.701379Z","shell.execute_reply.started":"2023-06-28T09:37:53.678027Z","shell.execute_reply":"2023-06-28T09:37:53.699614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_qs = [\n    (\"0-4\", 1, 4),\n    (\"5-12\", 4, 14),\n    (\"13-22\", 14, 19),\n]\nlevel_group_map = {}\nfor grp, a, b in trn_qs:\n    for q in range(a, b):\n        level_group_map[q] = grp\nlevel_group_map","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:37:53.706782Z","iopub.execute_input":"2023-06-28T09:37:53.707214Z","iopub.status.idle":"2023-06-28T09:37:53.718753Z","shell.execute_reply.started":"2023-06-28T09:37:53.707182Z","shell.execute_reply":"2023-06-28T09:37:53.717295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_df = pd.merge(\n    catb1_df[[\"session_id\", \"q\", \"true\", \"cat1_pred\"]], \n    catb2_df[[\"session_id\", \"q\", \"cat2_pred\"]], \n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df = pd.merge(\n    merged_df, \n    catb3_df[[\"session_id\", \"q\", \"cat3_pred\"]],\n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df = pd.merge(\n    merged_df, \n    catb4_df[[\"session_id\", \"q\", \"cat4_pred\"]],\n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df = pd.merge(\n    merged_df, \n    catb5_df[[\"session_id\", \"q\", \"cat5_pred\"]],\n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df = pd.merge(\n    merged_df, \n    catb6_df[[\"session_id\", \"q\", \"cat6_pred\"]],\n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df = pd.merge(\n    merged_df, \n    catb7_df[[\"session_id\", \"q\", \"cat7_pred\"]],\n    how=\"left\", on=[\"session_id\", \"q\"])\n\n\nmerged_df = pd.merge(\n    merged_df, \n    std_grp_catb_df[[\"session_id\", \"q\", \"std_grp_catb_pred\"]],\n    how=\"left\", on=[\"session_id\", \"q\"])\n\n\nmerged_df = pd.merge(\n    merged_df, \n    std_18in1_catb_df[[\"session_id\", \"q\", \"std_18in1_catb_pred\"]],\n    how=\"left\", on=[\"session_id\", \"q\"])\n\n\nmerged_df = pd.merge(\n    merged_df, \n    std_18in1_xgb_df[[\"session_id\", \"q\", \"std_18in1_xgb_pred\"]],\n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df = pd.merge(\n    merged_df, \n    std_18in1_xgb_2_df[[\"session_id\", \"q\", \"std_18in1_xgb_2_pred\"]],\n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df = pd.merge(\n    merged_df, \n    yyykrk_xgb1_df,\n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df = pd.merge(\n    merged_df, \n    yyykrk_lgb1_df,\n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df = pd.merge(\n    merged_df, \n    NN_df1,\n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df = pd.merge(\n    merged_df, \n    NN_df2,\n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df = pd.merge(\n    merged_df, \n    NN_df3,\n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df = pd.merge(\n    merged_df, \n    NN_df4,\n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df = pd.merge(\n    merged_df, \n    NN_df5,\n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df = pd.merge(\n    merged_df, \n    NN_df6,\n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df = pd.merge(\n    merged_df, \n    NN_df7,\n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df[\"grp\"] = merged_df[\"q\"].map(level_group_map)\nmerged_df","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:37:53.720262Z","iopub.execute_input":"2023-06-28T09:37:53.720752Z","iopub.status.idle":"2023-06-28T09:37:57.114276Z","shell.execute_reply.started":"2023-06-28T09:37:53.720719Z","shell.execute_reply":"2023-06-28T09:37:57.111963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_df = time_feature(merged_df)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:37:57.115478Z","iopub.execute_input":"2023-06-28T09:37:57.115825Z","iopub.status.idle":"2023-06-28T09:38:00.328694Z","shell.execute_reply.started":"2023-06-28T09:37:57.115795Z","shell.execute_reply":"2023-06-28T09:38:00.327807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression, LinearRegression\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.preprocessing import StandardScaler\n\nimport pickle\n\ndef try_blend(df,cols, name=\"full\"):\n    blender = LogisticRegression()\n#     blender = LinearRegression(fit_intercept=False, positive=True)\n    x = df[cols].reset_index(drop=True)\n    \n    print(x.describe())\n\n    blender.fit(x.values, df[\"true\"].values)\n    blended_preds = blender.predict_proba(x.values)[:,1]\n#     blended_preds = blender.predict(x.values)\n    \n    print(\n        get_thres_and_plot(df.true.values, blended_preds)\n    )\n    pickle.dump(blender, open(f\"./blend_{name}.bin\", 'wb'))\n    return blended_preds","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:38:00.329776Z","iopub.execute_input":"2023-06-28T09:38:00.330278Z","iopub.status.idle":"2023-06-28T09:38:00.354957Z","shell.execute_reply.started":"2023-06-28T09:38:00.330248Z","shell.execute_reply":"2023-06-28T09:38:00.353710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def inverse_sigmoid(x):\n    x = x.clip(0.01,0.99)\n    return np.log(x/(1-x))\n\n\ndef try_blend2(df,cols, name=\"full\"):\n    blender = LogisticRegression()\n    x = df[cols].reset_index(drop=True)\n    x = inverse_sigmoid(x)\n    \n    print(x.describe())\n\n    blender.fit(x.values, df[\"true\"].values)\n    blended_preds = blender.predict_proba(x.values)[:,1]\n    \n    \n    print(\n        get_thres_and_plot(df.true.values, blended_preds)\n    )\n    pickle.dump(blender, open(f\"./blend_NN_{name}.bin\", 'wb'))\n    return blended_preds","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:38:00.356570Z","iopub.execute_input":"2023-06-28T09:38:00.356924Z","iopub.status.idle":"2023-06-28T09:38:00.367942Z","shell.execute_reply.started":"2023-06-28T09:38:00.356894Z","shell.execute_reply":"2023-06-28T09:38:00.366706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_df.month.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:38:00.369228Z","iopub.execute_input":"2023-06-28T09:38:00.369810Z","iopub.status.idle":"2023-06-28T09:38:00.392868Z","shell.execute_reply.started":"2023-06-28T09:38:00.369776Z","shell.execute_reply":"2023-06-28T09:38:00.391598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_df[\n    [\"cat1_pred\", \"cat2_pred\",\"cat3_pred\", \"cat4_pred\", \n     \"std_grp_catb_pred\",\n     \"std_18in1_catb_pred\", \"std_18in1_xgb_pred\", \"std_18in1_xgb_2_pred\",\n     \n     'NN_pred1',\n    'NN_pred2',\n    'NN_pred3',\n    'NN_pred4',\n    'NN_pred5',\n    'NN_pred6',\n    'NN_pred7',\n\n    ]\n].corr()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:38:00.398137Z","iopub.execute_input":"2023-06-28T09:38:00.401984Z","iopub.status.idle":"2023-06-28T09:38:00.864519Z","shell.execute_reply.started":"2023-06-28T09:38:00.401938Z","shell.execute_reply":"2023-06-28T09:38:00.863632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = [\"std_grp_catb_pred\",\n        \"std_18in1_catb_pred\", \"std_18in1_xgb_pred\", \"std_18in1_xgb_2_pred\"]\ngbt_preds = try_blend(merged_df, cols, name=\"GBT\")","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:18:59.581322Z","iopub.execute_input":"2023-06-28T10:18:59.581741Z","iopub.status.idle":"2023-06-28T10:19:18.969740Z","shell.execute_reply.started":"2023-06-28T10:18:59.581710Z","shell.execute_reply":"2023-06-28T10:19:18.968914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = ['NN_pred1', \"NN_pred3\",\"NN_pred5\"]\nnn_preds = try_blend(merged_df, cols, name=\"NN\")","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:56:15.585265Z","iopub.execute_input":"2023-06-28T09:56:15.585673Z","iopub.status.idle":"2023-06-28T09:56:34.405562Z","shell.execute_reply.started":"2023-06-28T09:56:15.585625Z","shell.execute_reply":"2023-06-28T09:56:34.400916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cols = [\"std_grp_catb_pred\",\n#         \"std_18in1_catb_pred\", \"std_18in1_xgb_pred\",\n#         'NN_pred1', \"NN_pred2\", \"NN_pred3\",\n#         \"NN_pred5\", 'NN_pred7'\n#        ]\n# blend_preds = try_blend2(merged_df, cols, name=\"full\")","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:56:34.407682Z","iopub.execute_input":"2023-06-28T09:56:34.408143Z","iopub.status.idle":"2023-06-28T09:56:34.413857Z","shell.execute_reply.started":"2023-06-28T09:56:34.408097Z","shell.execute_reply":"2023-06-28T09:56:34.412691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask = (merged_df['year'] == 22)|((merged_df['year'] == 21)&(merged_df['month'] >= 12))\ndf = merged_df[mask]\nget_thres_and_plot(df.true.values, gbt_preds[mask])\nget_thres_and_plot(df.true.values, nn_preds[mask])\n# get_thres_and_plot(df.true.values, blend_preds[mask])","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:21:03.173215Z","iopub.execute_input":"2023-06-28T10:21:03.173616Z","iopub.status.idle":"2023-06-28T10:21:29.132070Z","shell.execute_reply.started":"2023-06-28T10:21:03.173586Z","shell.execute_reply":"2023-06-28T10:21:29.130942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cols = [\"std_grp_catb_pred\",\n#         \"std_18in1_catb_pred\", \"std_18in1_xgb_pred\"\n# #         , \"std_18in1_xgb_2_pred\"\n#        ]\n# gbt_preds = try_blend2(merged_df[(merged_df['year'] == 22)|((merged_df['year'] == 21)&(merged_df['month'] >= 12))], cols, name=\"years\")","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:59:15.412328Z","iopub.execute_input":"2023-06-28T09:59:15.413443Z","iopub.status.idle":"2023-06-28T09:59:15.417449Z","shell.execute_reply.started":"2023-06-28T09:59:15.413399Z","shell.execute_reply":"2023-06-28T09:59:15.416593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cols = ['NN_pred1', \"NN_pred2\", \"NN_pred3\",\n#         \"NN_pred5\", 'NN_pred7'\n#        ]\n# nn_preds = try_blend2(merged_df[(merged_df['year'] == 22)|((merged_df['year'] == 21)&(merged_df['month'] >= 12))], cols, name=\"years\")","metadata":{"execution":{"iopub.status.busy":"2023-06-28T09:59:15.419701Z","iopub.execute_input":"2023-06-28T09:59:15.420615Z","iopub.status.idle":"2023-06-28T09:59:15.433036Z","shell.execute_reply.started":"2023-06-28T09:59:15.420580Z","shell.execute_reply":"2023-06-28T09:59:15.431727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cols = [\"std_grp_catb_pred\",\n#         \"std_18in1_catb_pred\", \"std_18in1_xgb_pred\",\n#         'NN_pred1', \"NN_pred2\", \"NN_pred3\",\n#         \"NN_pred5\", 'NN_pred7'\n#        ]\n# blend_preds = try_blend2(merged_df[(merged_df['year'] == 22)|((merged_df['year'] == 21)&(merged_df['month'] >= 12))], cols, name=\"years\")","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:21:29.134184Z","iopub.execute_input":"2023-06-28T10:21:29.134521Z","iopub.status.idle":"2023-06-28T10:21:29.140539Z","shell.execute_reply.started":"2023-06-28T10:21:29.134493Z","shell.execute_reply":"2023-06-28T10:21:29.139457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = merged_df\nfor k in [0.8,0.7,0.6,0.55,0.5,0.45,0.4,0.3,0.2]:\n    print(k)\n    get_thres_and_plot(df.true.values, k*gbt_preds+(1-k)*nn_preds)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:21:29.141863Z","iopub.execute_input":"2023-06-28T10:21:29.142286Z","iopub.status.idle":"2023-06-28T10:24:18.892897Z","shell.execute_reply.started":"2023-06-28T10:21:29.142243Z","shell.execute_reply":"2023-06-28T10:24:18.891721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask = (merged_df['year'] == 22)|((merged_df['year'] == 21)&(merged_df['month'] >= 12))\ndf = merged_df[mask]\nfor k in [0.8,0.7,0.6,0.55,0.5,0.45,0.4,0.3,0.2]:\n    print(k)\n    get_thres_and_plot(df.true.values, k*gbt_preds[mask]+(1-k)*nn_preds[mask])","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:24:18.894968Z","iopub.execute_input":"2023-06-28T10:24:18.895317Z","iopub.status.idle":"2023-06-28T10:26:14.874831Z","shell.execute_reply.started":"2023-06-28T10:24:18.895288Z","shell.execute_reply":"2023-06-28T10:26:14.873769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(gbt_preds).hist(bins=100, label='GBT', alpha=0.3)\npd.Series(nn_preds).hist(bins=100, label='NN', alpha=0.3)\npd.Series((gbt_preds+nn_preds)/2).hist(bins=100, label='Blend', alpha=0.3)\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:26:14.876041Z","iopub.execute_input":"2023-06-28T10:26:14.876347Z","iopub.status.idle":"2023-06-28T10:26:15.938447Z","shell.execute_reply.started":"2023-06-28T10:26:14.876319Z","shell.execute_reply":"2023-06-28T10:26:15.937323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(inverse_sigmoid(gbt_preds)).hist(bins=100, label='GBT', alpha=0.3)\npd.Series(inverse_sigmoid(nn_preds)).hist(bins=100, label='NN', alpha=0.3)\npd.Series(inverse_sigmoid((gbt_preds+nn_preds)/2)).hist(bins=100, label='Blend', alpha=0.3)\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:26:15.939714Z","iopub.execute_input":"2023-06-28T10:26:15.940027Z","iopub.status.idle":"2023-06-28T10:26:17.006979Z","shell.execute_reply.started":"2023-06-28T10:26:15.940001Z","shell.execute_reply":"2023-06-28T10:26:17.006022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p = 0.6*gbt_preds+0.4*nn_preds","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:27:46.254495Z","iopub.execute_input":"2023-06-28T10:27:46.254988Z","iopub.status.idle":"2023-06-28T10:27:46.263265Z","shell.execute_reply.started":"2023-06-28T10:27:46.254951Z","shell.execute_reply":"2023-06-28T10:27:46.261898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_score(merged_df.true.values, p>0.654668797257206 , average=\"macro\")","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:27:48.150441Z","iopub.execute_input":"2023-06-28T10:27:48.150926Z","iopub.status.idle":"2023-06-28T10:27:48.331910Z","shell.execute_reply.started":"2023-06-28T10:27:48.150891Z","shell.execute_reply":"2023-06-28T10:27:48.330550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_score(merged_df.true.values, p>0.6511487416987165, average=\"macro\")","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:27:48.410383Z","iopub.execute_input":"2023-06-28T10:27:48.410898Z","iopub.status.idle":"2023-06-28T10:27:48.587318Z","shell.execute_reply.started":"2023-06-28T10:27:48.410864Z","shell.execute_reply":"2023-06-28T10:27:48.585732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask = (merged_df['year'] == 22)|((merged_df['year'] == 21)&(merged_df['month'] >= 12))","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:27:48.631111Z","iopub.execute_input":"2023-06-28T10:27:48.631590Z","iopub.status.idle":"2023-06-28T10:27:48.640355Z","shell.execute_reply.started":"2023-06-28T10:27:48.631555Z","shell.execute_reply":"2023-06-28T10:27:48.638960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_score(merged_df[mask].true.values, p[mask]>0.654668797257206 , average=\"macro\")","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:27:48.849655Z","iopub.execute_input":"2023-06-28T10:27:48.850083Z","iopub.status.idle":"2023-06-28T10:27:48.912800Z","shell.execute_reply.started":"2023-06-28T10:27:48.850049Z","shell.execute_reply":"2023-06-28T10:27:48.911821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_score(merged_df[mask].true.values, p[mask]>0.6511487416987165, average=\"macro\")","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:27:49.037828Z","iopub.execute_input":"2023-06-28T10:27:49.038400Z","iopub.status.idle":"2023-06-28T10:27:49.095144Z","shell.execute_reply.started":"2023-06-28T10:27:49.038367Z","shell.execute_reply":"2023-06-28T10:27:49.094149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}