{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np\nimport os, sys, pickle, glob, gc, itertools, math, json\nimport cudf\nfrom datetime import datetime as dt\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import StratifiedKFold as skfold\n# from collections import Counter\nimport xgboost as xgb\n\nprint('We will use RAPIDS version',cudf.__version__)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-31T11:01:43.297887Z","iopub.execute_input":"2023-01-31T11:01:43.298470Z","iopub.status.idle":"2023-01-31T11:01:46.319105Z","shell.execute_reply.started":"2023-01-31T11:01:43.298352Z","shell.execute_reply":"2023-01-31T11:01:46.317926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parameters\nVER = 2\nTEST = False\nTRAINING = True\nTEST_FRAC = 0.1\nNEG_FRAC = 0.85\nMIN_CANDIDATES = 30\nRS = 719\n\n# Data paths\nVAL_FT = '/kaggle/input/otto-tr-cand40-v2-tail40-top404050/'\nINFER_FT = ''\nVAL_LB = '/kaggle/input/otto-train-and-test-data-for-local-validation/test_labels.parquet'\nFT_DS = VAL_FT if TRAINING else INFER_FT\nCANDIDATES = 40\n\n# Data preparation\nUSE_WEIGHT = True # Whether to use event weight (other wise use event count)\nUSE_FREQ_ONEHOT = False # Whether to use one-hot encoding of frequency features\n\n# Training\nXGB_OBJ = 'rank:pairwise'\nFOLDS = 5\nTARGET_COL = 'click_target'\nTARGET_TYPE = 0\nMODEL_SAVE_NAME = f'XGBRanker_Click_V{VER}'\n\ntype_weight = {0:1, 1:3, 2:6}\ntype_label = {'clicks':0, 'carts':1, 'orders':2}","metadata":{"execution":{"iopub.status.busy":"2023-01-31T11:01:46.321247Z","iopub.execute_input":"2023-01-31T11:01:46.321694Z","iopub.status.idle":"2023-01-31T11:01:46.328938Z","shell.execute_reply.started":"2023-01-31T11:01:46.321656Z","shell.execute_reply":"2023-01-31T11:01:46.327902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utility Functions","metadata":{}},{"cell_type":"code","source":"def timer(sta):\n    return round((dt.now() - sta).seconds, 3)\n\ndef load_pqt(path):\n    return pd.read_parquet(path)\n\ndef load2cudf(path):\n    return cudf.from_pandas(load_pqt(path))\n\ndef select_frac_session(df, frac=TEST_FRAC):\n    sessions = df.session.unique().sample(frac=TEST_FRAC, random_state=RS)\n    return sessions.to_array()","metadata":{"execution":{"iopub.status.busy":"2023-01-31T11:01:46.330296Z","iopub.execute_input":"2023-01-31T11:01:46.331316Z","iopub.status.idle":"2023-01-31T11:01:46.340310Z","shell.execute_reply.started":"2023-01-31T11:01:46.331279Z","shell.execute_reply":"2023-01-31T11:01:46.339440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_candidates(file, target_col, dtype_map, path=FT_DS, test=TEST):\n    def random_arrays():\n        tmp = list(range(CANDIDATES))\n        np.random.shuffle(tmp)\n        return tmp\n    # Load data\n    sta = dt.now()\n    candidates = load2cudf(path + file)\n    print('Candidates loaded', end='')\n    # Truncate data by session\n    if test:\n        test_sessions = select_frac_session(candidates)\n        candidates = candidates.loc[candidates.session.isin(test_sessions)].reset_index(drop=True)\n        print(' and truncated', end='')\n    else:\n        test_sessions = None\n        \n    # Truncate data by random sampling negatives\n    if (not test) & TRAINING:\n        # Find number of negatives to retain in each session\n        tmp = candidates.groupby('session')[target_col].sum().reset_index()\n        tmp = tmp.rename(columns={target_col:'negatives'})\n        tmp['negatives'] = CANDIDATES - tmp['negatives'].astype('int32')\n        tmp['negatives'] = np.ceil(tmp['negatives']*NEG_FRAC).astype('int32')\n        tmp['negatives'] = tmp['negatives'].clip(lower=np.int32(MIN_CANDIDATES))\n        # Assign a random rank for each aid\n        candidates = candidates.sort_values('session').reset_index(drop=True)\n        candidates['random_rank'] = np.concatenate([random_arrays() for _ in range(len(tmp))])\n        # Split into negatives and positives\n        pos = candidates.loc[candidates[target_col] == 1]\n        neg = candidates.loc[candidates[target_col] != 1]\n        # Retain NEG_FRAC of negative set\n        neg = neg.merge(tmp, on='session', how='left')\n        neg = neg.sort_values(['session', 'random_rank']).reset_index(drop=True)\n        neg['n'] = neg.groupby('session').cumcount()\n        neg = neg.loc[neg.n < neg.negatives]\n        neg = neg.drop(columns=['negatives', 'n'])        \n        # Concat negatives and positives\n        candidates = cudf.concat([neg, pos], axis=0, ignore_index=True)\n        candidates = candidates.drop(columns=['random_rank'])\n        candidates = candidates.reset_index(drop=True)\n        del pos, neg, tmp\n        gc.collect()\n        print(f' and negative-sampled by {NEG_FRAC}', end='')\n    print(f' ({timer(sta)} sec)!')\n    \n    # Convert dtype\n    columns = candidates.columns\n    new_dtype_map = {f: dtype_map[f] for f in dtype_map if f in columns}\n    candidates = candidates.astype(new_dtype_map)\n    len_cand = len(candidates)\n    print(f'{len_cand} candidates, {round(len(candidates.session.unique()))} sessions')\n    print(candidates.dtypes)\n    return candidates, test_sessions\n\ndef process_candidates(candidates, features, dtype_map, path, verbose=0):\n    # Process features\n    verbsoe = (verbose > 0)\n    if verbose: print('Processing...', end='')\n    sta = dt.now()\n    for f in features:\n        ft = features[f]\n        tmp = load2cudf(path + f)\n        if ft['columns'] is not None:\n            tmp = tmp[ft['on'] + ft['columns']]\n        candidates = candidates.merge(tmp, on=ft['on'], how='left')\n        candidates = candidates.fillna(0)\n        if verbose: print(f'{f}...', end='')\n    if verbose: print(f'done ({timer(sta)} sec)!')\n\n    # Convert dtype\n    columns = candidates.columns\n    new_dtype_map = {f: dtype_map[f] for f in dtype_map if f in columns}\n    candidates = candidates.astype(new_dtype_map)\n    return candidates\n    \ndef load_process_candidates(file, target_col, features, dtype_map, path=FT_DS, test=TEST, verbose=1):\n    candidates, test_sessions = load_candidates(file, target_col, dtype_map, path=path, test=test)\n    candidates = process_candidates(candidates, features, dtype_map, path=path, verbose=verbose)\n    return candidates, test_sessions\n\ndef load_labels(test_sessions=None):\n    sta = dt.now()\n    labels = load2cudf(VAL_LB)\n    labels['type'] = labels['type'].map(type_label)\n    labels = labels.explode('ground_truth')\n    statement = 'Labels loaded'\n    if test_sessions is not None:\n        labels = labels.loc[labels.session.isin(test_sessions)]\n        labels = labels.reset_index(drop=True)\n        statement += ' and truncated'\n    print(f'{statement} ({timer(sta)} sec)!')\n    return labels\n\nft_meta = load2cudf(FT_DS + f'feature_metadata_v{VER}.pqt')\nprint(ft_meta.name.to_arrow().to_pylist())","metadata":{"execution":{"iopub.status.busy":"2023-01-31T11:15:31.398504Z","iopub.execute_input":"2023-01-31T11:15:31.398943Z","iopub.status.idle":"2023-01-31T11:15:31.431878Z","shell.execute_reply.started":"2023-01-31T11:15:31.398889Z","shell.execute_reply":"2023-01-31T11:15:31.430653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Click\n## Data Load and Process","metadata":{}},{"cell_type":"code","source":"%%time\nFEATURES = {\n    f'i_all_v{VER}.pqt': {\n        'on': ['aid']\n        , 'columns': [\n            'i_clk_wgt'\n            , 'i_order_wgt'\n            , 'i_freq_quarter_day'\n            , 'i_freq_dow'\n        ]\n    }\n    , f'u_all_v{VER}.pqt': {\n        'on': ['session']\n        , 'columns': [\n            'u_clk_wgt'\n            , 'u_order_wgt'\n            , 'u_freq_quarter_day'\n            , 'u_freq_dow'\n            , 'u_activ_period'\n        ]\n    }\n    , f'ui_all_v{VER}.pqt': {\n        'on': ['session', 'aid']\n        , 'columns': [\n            'ui_clk_wgt'\n            , 'ui_order_wgt'\n            , 'ui_freq_quarter_day'\n            , 'ui_freq_dow'\n            , 'ui_last_ts'\n            , 'ui_last_n'\n            , 'ui_clicks'\n            , 'ui_carts'\n            , 'ui_orders'\n        ]\n    }\n}\nDTYPE_MAP = {\n    TARGET_COL: 'uint16'\n    , 'top_clk_wk': 'uint16'\n}\nfile = f'click_candidates_top{CANDIDATES}_v{VER}.pqt'\n# click_cand, TEST_SESSIONS = load_process_candidates(\n#     file, target, CLICK_FEATURES\n#     , dtype_map={\n#         TARGET_COL: 'uint16'\n# #         , 'top_clk_wk': 'uint16'\n#     })\nclick_cand, TEST_SESSIONS = load_candidates(\n    file, TARGET_COL, dtype_map=DTYPE_MAP\n)\nclick_cand.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-31T11:01:48.375950Z","iopub.execute_input":"2023-01-31T11:01:48.376238Z","iopub.status.idle":"2023-01-31T11:02:20.923032Z","shell.execute_reply.started":"2023-01-31T11:01:48.376200Z","shell.execute_reply":"2023-01-31T11:02:20.922060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAINING:\n    labels = load_labels(TEST_SESSIONS)\n    # Count number of positives for stratified sampling\n    labels_pivot = labels.groupby(['session','type']).size().reset_index()\n    labels_pivot = labels_pivot.pivot(index='session', columns='type').reset_index()\n    labels_pivot = labels_pivot.fillna(0)\n    labels_pivot.columns = ['session', 'clicks', 'carts', 'orders']","metadata":{"execution":{"iopub.status.busy":"2023-01-31T11:02:20.924645Z","iopub.execute_input":"2023-01-31T11:02:20.925322Z","iopub.status.idle":"2023-01-31T11:02:23.059788Z","shell.execute_reply.started":"2023-01-31T11:02:20.925285Z","shell.execute_reply":"2023-01-31T11:02:23.058703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Training\n### Train/Val Split","metadata":{}},{"cell_type":"code","source":"if TRAINING:    \n    # Stratified split\n    def split_train_val(event, label_df, n_splits=FOLDS):\n        X = label_df['session'].copy()\n        y = label_df[event].copy()\n        folder = skfold(n_splits=n_splits, shuffle=True, random_state=RS)\n        splits = []\n        for f, (tr, val) in enumerate(folder.split(X.to_array(), y.to_array())):\n            sessions = X.loc[val]\n            splits.append(cudf.DataFrame({'session': sessions, 'fold': [f]*len(sessions)}))\n        splits = cudf.concat(splits, ignore_index=True)\n        splits.fold = splits.fold.astype('uint16')\n        return splits\n    \n    click_splits = split_train_val('clicks', labels_pivot)\n    click_cand = click_cand.merge(click_splits, on='session', how='left')\n    del click_splits, labels_pivot\n    gc.collect()\n    print(click_cand.dtypes)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T11:02:23.061281Z","iopub.execute_input":"2023-01-31T11:02:23.061849Z","iopub.status.idle":"2023-01-31T11:02:23.749704Z","shell.execute_reply.started":"2023-01-31T11:02:23.061808Z","shell.execute_reply":"2023-01-31T11:02:23.747947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train Model","metadata":{"execution":{"iopub.status.busy":"2023-01-03T15:13:52.250553Z","iopub.execute_input":"2023-01-03T15:13:52.250926Z","iopub.status.idle":"2023-01-03T15:13:52.574011Z","shell.execute_reply.started":"2023-01-03T15:13:52.25089Z","shell.execute_reply":"2023-01-03T15:13:52.573031Z"}}},{"cell_type":"code","source":"# Get Model\ndef get_xgb(params):\n#     params['eval_metric'] = recall_internal\n    return xgb.sklearn.XGBRanker(**params)\n\n# Prepare a DF into X and y format\ndef get_Xy(df, y_col, qid_col='session', aid_col='aid'):\n    df = df.sort_values('session', ascending=True).reset_index(drop=True)\n    qid = df[qid_col]\n    y = df[y_col]\n    aids = df[aid_col]\n    dropped = [y_col, qid_col, aid_col]\n    if 'fold' in df.columns: dropped.append('fold')\n    if 'random_rank' in df.columns: dropped.append('random_rank')\n    df = df.drop(columns=dropped)\n    return df, y, qid, aids\n\n# CALCULATE RECALL\n\ndef score_recall(predicts, population_targets):\n    # Calculate recall on candidate dataset\n    true_pos = len(predicts.loc[(predicts.predict == 1) & (predicts.target == 1)])\n    pos = predicts.target.astype('int32').sum()\n    candidate_recall = true_pos*1.0/pos\n    \n    # Calculate recall on population dataset\n    predicts = population_targets.merge(predicts, on=['session', 'aid'], how='left').fillna(0)\n    predicts = predicts.groupby('session').agg({'predict': ['count', 'sum']}).reset_index()\n    predicts.columns = ['session', 'actual_pos', 'true_pos']\n    predicts['actual_pos_trunc'] = predicts['actual_pos'].clip(upper=np.int32(20)) # error if not converted\n    population_recall = predicts['true_pos'].sum() / predicts['actual_pos_trunc'].sum()\n        \n    return candidate_recall, population_recall\n\ndef score_recall_by_round(\n    X_valid, qid_valid, aid_valid, session_valid, target_valid, target_type\n    , model, n_intervals=20\n    , labels=labels\n):\n    # Get predictions\n    def get_predictions(predicts, iteration_range=None):\n        if iteration_range is None:\n            pred = model.predict(X_valid)\n        else:\n            pred = model.predict(X_valid, iteration_range=iteration_range)\n        predicts['predict'] = pred\n        predicts = predicts.sort_values(['session', 'predict'], ascending=False).reset_index(drop=True)\n        predicts['n'] = predicts.groupby('session').cumcount()\n        predicts['predict'] = (predicts.n < 20).astype('int32')\n        return predicts.drop(columns=['n'])\n        \n    targets = labels.loc[(labels['type']==target_type) & (labels.session.isin(session_valid))]\n    targets = targets.rename(columns={'ground_truth':'aid'})\n    \n    trees = len(model.get_booster().get_dump())\n    if n_intervals > trees: n_intervals=trees\n    candidate_recall = {}\n    population_recall = {}\n    \n    print(f'Scoring {n_intervals} boosting round intervals...')\n    for intvl in np.array_split(np.array(range(trees)), n_intervals):\n        sta = 0\n        en = int(intvl[-1]+1)\n        print(f'{en}...', end='')\n        \n        predicts = cudf.DataFrame({\n            'session': qid_valid\n            , 'aid': aid_valid\n            , 'target': target_valid\n        })\n        predicts = get_predictions(predicts, iteration_range=[sta, en])\n        candidate_recall[en], population_recall[en] = score_recall(predicts, targets)\n    return candidate_recall, population_recall, candidate_recall[en], population_recall[en]\n\ndef plot(X, y, title, figsize=(10,7)):\n    plt.figure(figsize=(10,5))\n    plt.plot(X, y)\n    plt.title(title)\n    plt.show()\n\ndef train_model(\n    model_gen, params, model_save_name\n    , df, target_col, target_type, folds=FOLDS\n    , batches=None, plot_intervals=20\n):\n    results = {}\n    for f in range(FOLDS):\n\n        # Get data\n        s = dt.now()\n        X_val = df.loc[df['fold'] == f].reset_index(drop=True)\n        X_val = process_candidates(X_val, FEATURES, DTYPE_MAP, FT_DS)\n        if f==0: print(f'DATA TYPES\\n{X_val.dtypes}')\n        sess_val = X_val.session.unique().to_arrow().to_pylist()\n        \n        fid = f'FOLD {f}'\n        print('='*15 + '\\n' + fid + '...')\n        print(f'Data loaded ({timer(s)} secs)!')\n        \n        # Prepare data\n        s = dt.now()\n        model = model_gen(params)\n        X_val, y_val, qid_val, aid_val = get_Xy(X_val, target_col)\n        print(f'Data prepared ({timer(s)} secs)!')\n\n        # Train model\n        s = dt.now()\n        if batches: # Training model in multiple batches\n            print(f'Training in {batches} batches...', end='')\n            folds = list(range(FOLDS))\n            folds.remove(f)\n            for b_idx, b in enumerate(np.array_split(np.array(folds), batches)):\n                X_tr = df.loc[df.fold.isin(b)].reset_index(drop=True)\n                X_tr = process_candidates(X_tr, FEATURES, DTYPE_MAP, FT_DS)\n                X_tr, y_tr, qid_tr, _ = get_Xy(X_tr, target_col)\n                \n                if b_idx!=0:\n                    model_prev = model.get_booster()\n                    model = model_gen(params)\n                else:\n                    model_prev = None\n                model.fit(\n                    X=X_tr, y=y_tr, qid=qid_tr\n                    , eval_set=[(X_val, y_val)], eval_qid=[qid_val]\n                    , xgb_model=model_prev\n                    , verbose=0\n                )\n                print(f'{\"...\".join(map(str, b))}...', end='')\n        else:\n            print('Training entire set...', end='')\n            X_tr = df.loc[df['fold'] != f].reset_index(drop=True)\n            X_tr, y_tr, qid_tr, _ = get_Xy(X_tr, target_col)\n            model.fit(\n                X=X_tr, y=y_tr, qid=qid_tr\n                , eval_set=[(X_val, y_val)], eval_qid=[qid_val]\n                , verbose=0\n            )\n        print(f'finished ({timer(s)} secs)!')\n        # Score\n        s = dt.now()\n        cand, pop, f_cand, f_pop = score_recall_by_round(\n            X_val, qid_val, aid_val, sess_val, y_val\n            , target_type, model, n_intervals=plot_intervals)\n        print(f'Scoring finished ({timer(s)} secs)!')\n        # Save data\n        trees = len(model.get_booster().get_dump())\n        n_estimators = model.get_num_boosting_rounds()\n        print(f'{trees}/{n_estimators} trained')\n        if trees < n_estimators:\n            print(f'Best iteration: {model.best_iteration}')\n        results[fid] = {\n#             'model': model\n            'candidate_recall': cand\n            , 'final_candidate_recall': f_cand\n            , 'population_recall': pop\n            , 'final_population_recall': f_pop\n            , 'ft_importances': {}\n        }\n        for i_, name in enumerate(model.feature_names_in_):\n            results[fid]['ft_importances'][name] = str(model.feature_importances_[i_])\n        model.save_model(model_save_name + f'_Fold-{f}.json')\n        # Plot score\n        plt.figure(figsize=(14,5))\n        ax1 = plt.subplot(1,2,1)\n        ax2 = plt.subplot(1,2,2)\n        ax1.plot(list(cand.keys()), list(cand.values()))\n        ax1.title.set_text(f'FOLD {f} - Candidate recall: {round(f_cand, 5)}')\n        ax2.plot(list(pop.keys()), list(pop.values()))\n        ax2.title.set_text(f'FOLD {f} - Population recall: {round(f_pop, 5)}')\n        plt.show()\n#         if f==3: break\n    \n    # Averaging score\n    cand_rec = np.array([results[f'FOLD {f}']['final_candidate_recall'] for f in range(len(results))])\n    cand_avg = cand_rec.mean()\n    cand_std = cand_rec.std()\n    pop_rec = np.array([results[f'FOLD {f}']['final_population_recall'] for f in range(len(results))])\n    pop_avg = pop_rec.mean()\n    pop_std = pop_rec.std()\n    print('='*20)\n    print('Model hyperparams:')\n    for k in params:\n        print(f'{k}: {params[k]}')\n    print('='*20)\n    print(f'Candidate recall: mean={round(cand_avg,5)}, std={round(cand_std,5)}')\n    print(f'Population recall: mean={round(pop_avg,5)}, std={round(pop_std,5)}')\n    return results","metadata":{"execution":{"iopub.status.busy":"2023-01-31T11:15:31.433546Z","iopub.execute_input":"2023-01-31T11:15:31.434106Z","iopub.status.idle":"2023-01-31T11:15:31.469141Z","shell.execute_reply.started":"2023-01-31T11:15:31.434069Z","shell.execute_reply":"2023-01-31T11:15:31.468015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # CALCULATE BASELINE\n# # Using only handcrafted rules\n# tmp = click_cand[['session', 'aid', 'clk_cov_wgt', 'ui_clk_wgt', 'top_clk_wk', 'click_target']]\n# tmp = tmp.sort_values(['session', 'ui_clk_wgt', 'clk_cov_wgt', 'top_clk_wk'], ascending=False)\n# tmp = tmp.reset_index(drop=True)\n# tmp['n'] = tmp.groupby('session').cumcount()\n# tmp['predict'] = (tmp.n < 20).astype('uint32')\n# tmp = tmp[['session', 'aid', 'click_target', 'predict']]\n# tmp = tmp.rename(columns={'click_target': 'target'})\n\n# tmp2 = labels.loc[labels['type']==0]\n# tmp2 = tmp2.rename(columns={'ground_truth':'aid'})\n# if TEST:\n#     tmp2 = tmp2.loc[tmp2.session.isin(TEST_SESSIONS)]\n    \n# cand_baseline, pop_baseline = score_recall(tmp, tmp2)\n# del tmp, tmp2\n# gc.collect()\n\n# print('CLICK - Recall baseline')\n# print(f'Candidate recall: {round(cand_baseline,5)}')\n# print(f'Population recall: {round(pop_baseline,5)}')\n\n# CLICK - Recall baseline\n# Candidate recall: 0.92985\n# Population recall: 0.52786\n","metadata":{"execution":{"iopub.status.busy":"2023-01-31T11:02:23.787551Z","iopub.execute_input":"2023-01-31T11:02:23.788611Z","iopub.status.idle":"2023-01-31T11:02:23.802776Z","shell.execute_reply.started":"2023-01-31T11:02:23.788575Z","shell.execute_reply":"2023-01-31T11:02:23.801863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_DEPTH = 13\nN_ESTIMATORS = 33\nXGB_PARAMS = {\n    'tree_method': 'gpu_hist'\n    , 'objective': XGB_OBJ\n    , 'n_estimators': N_ESTIMATORS\n    , 'max_depth': MAX_DEPTH\n    , 'max_leaves': int(2**(MAX_DEPTH-2))\n    , 'reg_alpha': 200\n    , 'reg_lambda': 5\n    , 'random_state': RS\n    , 'learning_rate': 0.0065\n    , 'colsample_bytree': 0.8\n    , 'colsample_bylevel': 0.8\n    , 'colsample_bynode': 0.8\n#     , 'gamma': 10\n    , 'early_stopping_rounds': max(int(N_ESTIMATORS/3), 10)\n}\n\ntraining_result = train_model(\n    get_xgb\n    , XGB_PARAMS\n    , MODEL_SAVE_NAME\n    , click_cand\n    , target_col=TARGET_COL\n    , target_type=TARGET_TYPE\n    , folds=FOLDS\n    , batches=FOLDS-1\n    , plot_intervals=20\n)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-31T11:31:55.549744Z","iopub.execute_input":"2023-01-31T11:31:55.550104Z","iopub.status.idle":"2023-01-31T11:36:00.148022Z","shell.execute_reply.started":"2023-01-31T11:31:55.550071Z","shell.execute_reply":"2023-01-31T11:36:00.147100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CLICK - Recall baseline\n# Candidate recall: 0.92985\n# Population recall: 0.52786\n\n# ====================\n# Candidate recall: mean=0.9248, std=0.0003\n# Population recall: mean=0.525, std=0.00095\n\n# HIGHEST SCORE\n# ====================\n# Model hyperparams:\n# tree_method: gpu_hist\n# objective: rank:pairwise\n# n_estimators: 60\n# max_depth: 12\n# max_leaves: 1024\n# reg_alpha: 230\n# reg_lambda: 8\n# random_state: 719\n# learning_rate: 0.003\n# colsample_bytree: 0.75\n# colsample_bylevel: 0.75\n# colsample_bynode: 0.75\n# ====================\n# Candidate recall: mean=0.92495, std=0.0003\n# Population recall: mean=0.52508, std=0.00083\n\n# MOST RECENT\n# ====================\n# Model hyperparams:\n# tree_method: gpu_hist\n# objective: rank:pairwise\n# n_estimators: 33\n# max_depth: 13\n# max_leaves: 2048\n# reg_alpha: 200\n# reg_lambda: 5\n# random_state: 719\n# learning_rate: 0.0065\n# colsample_bytree: 0.8\n# colsample_bylevel: 0.8\n# colsample_bynode: 0.8\n# early_stopping_rounds: 11\n# ====================\n# Candidate recall: mean=0.92451, std=0.00011\n# Population recall: mean=0.52483, std=0.00101","metadata":{"execution":{"iopub.status.busy":"2023-01-31T11:08:00.274042Z","iopub.execute_input":"2023-01-31T11:08:00.275493Z","iopub.status.idle":"2023-01-31T11:08:00.280449Z","shell.execute_reply.started":"2023-01-31T11:08:00.275452Z","shell.execute_reply":"2023-01-31T11:08:00.279287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_model = get_xgb({})\nmodel_pth = glob.glob(MODEL_SAVE_NAME + '*.json')\ntmp_model.load_model(model_pth[0])\nprint(tmp_model.feature_names_in_)\n\ntraining_result['feature_names'] = list(tmp_model.feature_names_in_)\nwith open('training_result.txt', 'w') as file:\n     file.write(json.dumps(training_result))","metadata":{"execution":{"iopub.status.busy":"2023-01-31T11:38:41.684374Z","iopub.execute_input":"2023-01-31T11:38:41.684750Z","iopub.status.idle":"2023-01-31T11:38:41.854392Z","shell.execute_reply.started":"2023-01-31T11:38:41.684718Z","shell.execute_reply":"2023-01-31T11:38:41.853395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FEATURE IMPORTANCES\nfi = training_result['FOLD 0']['ft_importances'].copy()\nfor f in range(1, FOLDS):\n    fold_fi = training_result[f'FOLD {f}']['ft_importances']\n    for ft in fold_fi:\n        fi[ft] = float(fi[ft]) + float(fold_fi[ft])\n        \nfi = pd.DataFrame({'features': fi.keys(), 'score': fi.values()})\nfi['score'] = fi['score'] / FOLDS\nfi = fi.sort_values('score', ascending=False).reset_index(drop=True)\nfi","metadata":{},"execution_count":null,"outputs":[]}]}