{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom skopt import gp_minimize\nfrom skopt.space import Categorical, Integer\nfrom sklearn.metrics import f1_score\nimport matplotlib.pyplot as plt\n\nSEED = 2023\n\ndef get_best_threshold_and_score(ts, ps, start=0.2, end=0.8, step=0.01, n_calls=100):\n\n    def f(t):\n        _ps = (ps > t).astype(\"int\")\n        return -f1_score(ts, _ps, average=\"macro\")\n\n    res = gp_minimize(\n      f,                  # the function to minimize\n      [(.2, .8)],      # the bounds on each dimension of x\n      acq_func=\"EI\",      # the acquisition function\n      n_calls=n_calls,         # the number of evaluations of f\n      n_random_starts=3,  # the number of random initialization points\n      random_state=SEED,\n     )   # the random seed\n\n    bt = res.x[0]\n    s = res.fun\n    ts = [xs[0] for xs in res.x_iters]\n    ss = [-xs for xs in res.func_vals]\n    return bt, -s, ts, ss\n\n\ndef get_thres_and_plot(y_true, y_pred):\n    best_threshold, best_score, thresholds, scores = get_best_threshold_and_score(\n        y_true, y_pred, n_calls=50,\n    )\n\n    print(best_threshold, best_score)\n\n    plt.figure(figsize=(20, 5))\n    plt.scatter(thresholds, scores, color=\"blue\")\n    plt.scatter([best_threshold], [best_score], color=\"red\")\n    plt.xlabel(\"Threshold\", size=14)\n    plt.ylabel(\"Validation F1 Score\", size=14)\n    plt.title(\n        f\"Threshold vs. F1_Score with Best F1_Score={best_score:.5f} at Best Threshold = {best_threshold:.5f}\",\n        size=18,\n    )\n    plt.show()\n    return best_threshold, best_score\n\ndef time_feature(train):\n    train[\"year\"] = (\n        train[\"session_id\"].apply(lambda x: int(str(x)[:2])).astype(np.uint8)\n    )\n    train[\"month\"] = (\n        train[\"session_id\"].apply(lambda x: int(str(x)[2:4]) + 1).astype(np.uint8)\n    )\n    train[\"day\"] = (\n        train[\"session_id\"].apply(lambda x: int(str(x)[4:6])).astype(np.uint8)\n    )\n    train[\"hour\"] = (\n        train[\"session_id\"].apply(lambda x: int(str(x)[6:8])).astype(np.uint8)\n    )\n    train[\"minute\"] = (\n        train[\"session_id\"].apply(lambda x: int(str(x)[8:10])).astype(np.uint8)\n    )\n    train[\"second\"] = (\n        train[\"session_id\"].apply(lambda x: int(str(x)[10:12])).astype(np.uint8)\n    )\n\n    return train","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-09T14:18:44.145159Z","iopub.execute_input":"2023-06-09T14:18:44.145849Z","iopub.status.idle":"2023-06-09T14:18:46.364606Z","shell.execute_reply.started":"2023-06-09T14:18:44.145812Z","shell.execute_reply":"2023-06-09T14:18:46.363366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"catb1_base = \"/kaggle/input/psp-catboost-sort-by-time-only\"\ncatb2_base = \"/kaggle/input/xgboost-sort-by-time\"\nyyykrk_xgb1 = \"/kaggle/input/20230610-011236\"\nyyykrk_lgb1 = \"/kaggle/input/20230610-021710\"","metadata":{"execution":{"iopub.status.busy":"2023-06-09T14:22:50.501181Z","iopub.execute_input":"2023-06-09T14:22:50.501772Z","iopub.status.idle":"2023-06-09T14:22:50.507422Z","shell.execute_reply.started":"2023-06-09T14:22:50.50173Z","shell.execute_reply":"2023-06-09T14:22:50.505736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_qs = [\n    ('0-4', 1, 4),\n    ('5-12', 4, 14),\n    ('13-22', 14, 19),\n]\n\nval_dfs = []\nfor grp_idx, (grp, a, b) in enumerate(trn_qs):\n    for fold in range(5):\n        val_dfs.append(\n            pd.read_csv(f\"{catb1_base}/val_preds_fold_{fold}_grp_{grp}_final.csv\")\n        )\ncatb1_df = pd.concat(val_dfs)\n\nval_dfs = []\nfor grp_idx, (grp, a, b) in enumerate(trn_qs):\n    for fold in range(5):\n        val_dfs.append(\n            pd.read_csv(f\"{catb2_base}/val_preds_fold_{fold}_grp_{grp}_final.csv\")\n        )\ncatb2_df = pd.concat(val_dfs)\n\n\n# val_dfs = []\n# for grp_idx, (grp, a, b) in enumerate(trn_qs):\n#     for fold in range(5):\n#         val_dfs.append(\n#             pd.read_csv(f\"{xgb1_base}/val_preds_fold_{fold}_grp_{grp}_final.csv\")\n#         )\n# xgb1_df = pd.concat(val_dfs)","metadata":{"execution":{"iopub.status.busy":"2023-06-09T14:22:50.714644Z","iopub.execute_input":"2023-06-09T14:22:50.715188Z","iopub.status.idle":"2023-06-09T14:22:51.563853Z","shell.execute_reply.started":"2023-06-09T14:22:50.715142Z","shell.execute_reply":"2023-06-09T14:22:51.561972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"yyykrk_xgb1_df = pd.read_csv(f\"{yyykrk_xgb1}/oof.csv\")\nyyykrk_lgb1_df = pd.read_csv(f\"{yyykrk_lgb1}/oof.csv\")\n\n\ndef cleaning_yyykrk_df(df, pred_name):\n    cols_rename = {\"Unnamed: 0\": \"session_id\"}\n    for q in range(0,18):\n        cols_rename[str(q)] = str(q+1)\n    df = df.rename(columns=cols_rename)\n    df = df.set_index(\"session_id\")\n    df = df.stack().reset_index()\n    df = df.rename(columns={\"level_1\": \"q\", 0: f\"{pred_name}_pred\"})\n    df[\"q\"] = df[\"q\"].astype(int)\n    return df\n    df.head()\n    \nyyykrk_xgb1_df = cleaning_yyykrk_df(yyykrk_xgb1_df, \"yyykrk_xgb1\")\nyyykrk_lgb1_df = cleaning_yyykrk_df(yyykrk_lgb1_df, \"yyykrk_lgb1\")","metadata":{"execution":{"iopub.status.busy":"2023-06-09T14:30:46.465361Z","iopub.execute_input":"2023-06-09T14:30:46.465804Z","iopub.status.idle":"2023-06-09T14:30:46.653712Z","shell.execute_reply.started":"2023-06-09T14:30:46.465767Z","shell.execute_reply":"2023-06-09T14:30:46.652682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"catb1_df[\"cat1_pred\"] = catb1_df[\"pred\"]\ncatb2_df[\"cat2_pred\"] = catb2_df[\"pred\"]\n# xgb1_df[\"xgb1_pred\"] = xgb1_df[\"pred\"]","metadata":{"execution":{"iopub.status.busy":"2023-06-09T14:30:50.2019Z","iopub.execute_input":"2023-06-09T14:30:50.202301Z","iopub.status.idle":"2023-06-09T14:30:50.212559Z","shell.execute_reply.started":"2023-06-09T14:30:50.202268Z","shell.execute_reply":"2023-06-09T14:30:50.211427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert catb1_df.shape[0] == catb2_df.shape[0] == yyykrk_xgb1_df.shape[0] == yyykrk_lgb1_df.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-06-09T14:31:01.742862Z","iopub.execute_input":"2023-06-09T14:31:01.745854Z","iopub.status.idle":"2023-06-09T14:31:01.752839Z","shell.execute_reply.started":"2023-06-09T14:31:01.745808Z","shell.execute_reply":"2023-06-09T14:31:01.751829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"catb1_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T14:30:50.66469Z","iopub.execute_input":"2023-06-09T14:30:50.665756Z","iopub.status.idle":"2023-06-09T14:30:50.67879Z","shell.execute_reply.started":"2023-06-09T14:30:50.665721Z","shell.execute_reply":"2023-06-09T14:30:50.677818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"catb2_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T14:30:51.055455Z","iopub.execute_input":"2023-06-09T14:30:51.05586Z","iopub.status.idle":"2023-06-09T14:30:51.069261Z","shell.execute_reply.started":"2023-06-09T14:30:51.055828Z","shell.execute_reply":"2023-06-09T14:30:51.068127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_qs = [\n    (\"0-4\", 1, 4),\n    (\"5-12\", 4, 14),\n    (\"13-22\", 14, 19),\n]\nlevel_group_map = {}\nfor grp, a, b in trn_qs:\n    for q in range(a, b):\n        level_group_map[q] = grp\nlevel_group_map","metadata":{"execution":{"iopub.status.busy":"2023-06-09T14:30:51.681303Z","iopub.execute_input":"2023-06-09T14:30:51.682236Z","iopub.status.idle":"2023-06-09T14:30:51.691048Z","shell.execute_reply.started":"2023-06-09T14:30:51.68219Z","shell.execute_reply":"2023-06-09T14:30:51.690011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_df = pd.merge(\n    catb1_df[[\"session_id\", \"q\", \"true\", \"cat1_pred\"]], \n    catb2_df[[\"session_id\", \"q\", \"cat2_pred\"]], \n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df = pd.merge(\n    merged_df, \n    yyykrk_xgb1_df,\n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df = pd.merge(\n    merged_df, \n    yyykrk_lgb1_df,\n    how=\"left\", on=[\"session_id\", \"q\"])\n\nmerged_df[\"grp\"] = merged_df[\"q\"].map(level_group_map)\nmerged_df","metadata":{"execution":{"iopub.status.busy":"2023-06-09T14:30:52.205264Z","iopub.execute_input":"2023-06-09T14:30:52.205665Z","iopub.status.idle":"2023-06-09T14:30:52.554508Z","shell.execute_reply.started":"2023-06-09T14:30:52.205633Z","shell.execute_reply":"2023-06-09T14:30:52.553496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_df = time_feature(merged_df)","metadata":{"execution":{"iopub.status.busy":"2023-06-09T14:31:25.915474Z","iopub.execute_input":"2023-06-09T14:31:25.915898Z","iopub.status.idle":"2023-06-09T14:31:29.199137Z","shell.execute_reply.started":"2023-06-09T14:31:25.915867Z","shell.execute_reply":"2023-06-09T14:31:29.197987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.naive_bayes import GaussianNB\nimport pickle\n\ndef try_blend(df, name=\"full\"):\n    print(\n        get_thres_and_plot(df.true.values, df.cat1_pred.values)\n    )\n    print(\n        get_thres_and_plot(df.true.values, df.cat2_pred.values)\n    )\n    print(\n        get_thres_and_plot(df.true.values, df.yyykrk_xgb1_pred.values)\n    )\n    print(\n        get_thres_and_plot(df.true.values, df.yyykrk_lgb1_pred.values)\n    )\n#     print(\n#         get_thres_and_plot(df.true.values, df.xgb1_pred.values)\n#     )\n    blender = LogisticRegression()\n    x = df[[\"cat1_pred\", \"cat2_pred\", \"yyykrk_xgb1_pred\", \"yyykrk_lgb1_pred\"]].reset_index(drop=True)\n    print(x.describe())\n\n    blender.fit(x.values, df[\"true\"].values)\n    blended_preds = blender.predict_proba(x.values)[:,1]\n    print(\n        get_thres_and_plot(df.true.values, blended_preds)\n    )\n    pickle.dump(blender, open(f\"./blend_{name}.bin\", 'wb'))","metadata":{"execution":{"iopub.status.busy":"2023-06-09T14:31:55.125794Z","iopub.execute_input":"2023-06-09T14:31:55.126187Z","iopub.status.idle":"2023-06-09T14:31:55.138322Z","shell.execute_reply.started":"2023-06-09T14:31:55.126157Z","shell.execute_reply":"2023-06-09T14:31:55.137487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_df.month.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T14:31:56.224716Z","iopub.execute_input":"2023-06-09T14:31:56.225097Z","iopub.status.idle":"2023-06-09T14:31:56.237708Z","shell.execute_reply.started":"2023-06-09T14:31:56.225066Z","shell.execute_reply":"2023-06-09T14:31:56.236641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_df[merged_df.year==22].month.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-09T14:31:57.098875Z","iopub.execute_input":"2023-06-09T14:31:57.099881Z","iopub.status.idle":"2023-06-09T14:31:57.124539Z","shell.execute_reply.started":"2023-06-09T14:31:57.099839Z","shell.execute_reply":"2023-06-09T14:31:57.12357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try_blend(merged_df, name=\"full\")","metadata":{"execution":{"iopub.status.busy":"2023-06-09T14:31:58.214997Z","iopub.execute_input":"2023-06-09T14:31:58.215691Z","iopub.status.idle":"2023-06-09T14:32:56.240435Z","shell.execute_reply.started":"2023-06-09T14:31:58.215657Z","shell.execute_reply":"2023-06-09T14:32:56.239536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try_blend(merged_df[((merged_df.month <= 5) | (merged_df.month >= 12))], name=\"months\")","metadata":{"execution":{"iopub.status.busy":"2023-06-09T14:33:39.692963Z","iopub.execute_input":"2023-06-09T14:33:39.693778Z","iopub.status.idle":"2023-06-09T14:34:24.923396Z","shell.execute_reply.started":"2023-06-09T14:33:39.693743Z","shell.execute_reply":"2023-06-09T14:34:24.922235Z"},"trusted":true},"execution_count":null,"outputs":[]}]}