{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np\nimport os, sys, pickle, glob, gc, itertools, math, json\nimport cudf\nfrom datetime import datetime as dt\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import StratifiedKFold as skfold\n# from collections import Counter\nimport xgboost as xgb\n\nprint('We will use RAPIDS version',cudf.__version__)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-31T18:18:48.553312Z","iopub.execute_input":"2023-01-31T18:18:48.554029Z","iopub.status.idle":"2023-01-31T18:18:52.571696Z","shell.execute_reply.started":"2023-01-31T18:18:48.553915Z","shell.execute_reply":"2023-01-31T18:18:52.570533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parameters\nVER = 2\nTEST = False\nTRAINING = True\nTEST_FRAC = 0.1\nNEG_FRAC = 0.75\nMIN_CANDIDATES = 30\nRS = 719\n\n# Data paths\nVAL_FT = '/kaggle/input/otto-tr-cand40-v2-tail40-top404050/'\nINFER_FT = ''\nVAL_LB = '/kaggle/input/otto-train-and-test-data-for-local-validation/test_labels.parquet'\nFT_DS = VAL_FT if TRAINING else INFER_FT\nCANDIDATES = 40\n\n# Training\nXGB_OBJ = 'rank:pairwise'\nFOLDS = 5\nTARGET_COL = 'cart_target'\nTARGET_DROP = 'order_target'\nTARGET_TYPE = 1\nMODEL_SAVE_NAME = f'XGBRanker_Cart_V{VER}'\n\ntype_weight = {0:1, 1:3, 2:6}\ntype_label = {'clicks':0, 'carts':1, 'orders':2}","metadata":{"execution":{"iopub.status.busy":"2023-01-31T18:18:52.573854Z","iopub.execute_input":"2023-01-31T18:18:52.574977Z","iopub.status.idle":"2023-01-31T18:18:52.584808Z","shell.execute_reply.started":"2023-01-31T18:18:52.574939Z","shell.execute_reply":"2023-01-31T18:18:52.582717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utility Functions","metadata":{}},{"cell_type":"code","source":"def timer(sta):\n    return round((dt.now() - sta).seconds, 3)\n\ndef load_pqt(path):\n    return pd.read_parquet(path)\n\ndef load2cudf(path):\n    return cudf.from_pandas(load_pqt(path))\n\ndef select_frac_session(df, frac=TEST_FRAC):\n    sessions = df.session.unique().sample(frac=TEST_FRAC, random_state=RS)\n    return sessions.to_array()","metadata":{"execution":{"iopub.status.busy":"2023-01-31T18:18:52.587008Z","iopub.execute_input":"2023-01-31T18:18:52.588014Z","iopub.status.idle":"2023-01-31T18:18:52.602898Z","shell.execute_reply.started":"2023-01-31T18:18:52.587764Z","shell.execute_reply":"2023-01-31T18:18:52.601616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Load and Process","metadata":{}},{"cell_type":"code","source":"ft_meta = load2cudf(FT_DS + f'feature_metadata_v{VER}.pqt')\nprint(ft_meta.name.to_arrow().to_pylist())","metadata":{"execution":{"iopub.status.busy":"2023-01-31T18:18:52.606202Z","iopub.execute_input":"2023-01-31T18:18:52.607119Z","iopub.status.idle":"2023-01-31T18:18:55.058398Z","shell.execute_reply.started":"2023-01-31T18:18:52.607067Z","shell.execute_reply":"2023-01-31T18:18:55.056486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_process_candidates(file, target_col, features, dtype_map, path=FT_DS, test=TEST, drop_cols=None):\n    def random_arrays():\n        tmp = list(range(CANDIDATES))\n        np.random.shuffle(tmp)\n        return tmp\n    # Load data\n    sta = dt.now()\n    candidates = load2cudf(path + file)\n    print('Candidates loaded', end='')\n    # Truncate data by session\n    if test:\n        test_sessions = select_frac_session(candidates)\n        candidates = candidates.loc[candidates.session.isin(test_sessions)].reset_index(drop=True)\n        print(' and truncated', end='')\n    else:\n        test_sessions = None\n        \n    # Truncate data by random sampling negatives\n    if (not test) & TRAINING:\n        # Find number of negatives to retain in each session\n        tmp = candidates.groupby('session')[target_col].sum().reset_index()\n        tmp = tmp.rename(columns={target_col:'negatives'})\n        tmp['negatives'] = CANDIDATES - tmp['negatives'].astype('int32')\n        tmp['negatives'] = np.ceil(tmp['negatives']*NEG_FRAC).astype('int32')\n        tmp['negatives'] = tmp['negatives'].clip(lower=np.int32(MIN_CANDIDATES))\n        # Assign a random rank for each aid\n        candidates = candidates.sort_values('session').reset_index(drop=True)\n        candidates['random_rank'] = np.concatenate([random_arrays() for _ in range(len(tmp))])\n        # Split into negatives and positives\n        pos = candidates.loc[candidates[target_col] == 1]\n        neg = candidates.loc[candidates[target_col] != 1]\n        # Retain NEG_FRAC of negative set\n        neg = neg.merge(tmp, on='session', how='left')\n        neg = neg.sort_values(['session', 'random_rank']).reset_index(drop=True)\n        neg['n'] = neg.groupby('session').cumcount()\n        neg = neg.loc[neg.n < neg.negatives]\n        neg = neg.drop(columns=['negatives', 'n'])        \n        # Concat negatives and positives\n        candidates = cudf.concat([neg, pos], axis=0, ignore_index=True)\n        candidates = candidates.reset_index(drop=True)\n        del pos, neg, tmp\n        gc.collect()\n        print(f' and negative-sampled by {NEG_FRAC}', end='')\n    print(f' ({timer(sta)} sec)!')\n    \n    # Process features\n    print('Processing...', end='')\n    sta = dt.now()\n    for f in features:\n        ft = features[f]\n        tmp = load2cudf(path + f)\n        if ft['columns'] is not None:\n            tmp = tmp[ft['on'] + ft['columns']]\n        candidates = candidates.merge(tmp, on=ft['on'], how='left')\n        candidates = candidates.fillna(0)\n        print(f'{f}...', end='')\n    print(f'done ({timer(sta)} sec)!')\n    len_cand = len(candidates)\n    print(f'{len_cand} candidates, {round(len(candidates.session.unique()))} sessions')\n    \n    # Convert dtype\n    candidates = candidates.astype(dtype_map)\n    \n    # Drop unused columnbs\n    if drop_cols is not None:\n        candidates = candidates.drop(columns=drop_cols)\n    \n    print(candidates.dtypes)\n    return candidates, test_sessions\n\ndef load_labels(test_sessions=None):\n    sta = dt.now()\n    labels = load2cudf(VAL_LB)\n    labels['type'] = labels['type'].map(type_label)\n    labels = labels.explode('ground_truth')\n    statement = 'Labels loaded'\n    if test_sessions is not None:\n        labels = labels.loc[labels.session.isin(test_sessions)]\n        labels = labels.reset_index(drop=True)\n        statement += ' and truncated'\n    print(f'{statement} ({timer(sta)} sec)!')\n    return labels\n\nft_meta = load2cudf(FT_DS + f'feature_metadata_v{VER}.pqt')\nprint(ft_meta.name.to_arrow().to_pylist())","metadata":{"execution":{"iopub.status.busy":"2023-01-31T18:18:55.060090Z","iopub.execute_input":"2023-01-31T18:18:55.060745Z","iopub.status.idle":"2023-01-31T18:18:55.091449Z","shell.execute_reply.started":"2023-01-31T18:18:55.060704Z","shell.execute_reply":"2023-01-31T18:18:55.090242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nCART_FEATURES = {\n    f'i_all_v{VER}.pqt': {\n        'on': ['aid']\n        , 'columns': [\n            'i_clk_wgt'\n            , 'i_order_wgt'\n            , 'i_freq_quarter_day'\n            , 'i_freq_dow'\n        ]\n    }\n    , f'u_all_v{VER}.pqt': {\n        'on': ['session']\n        , 'columns': [\n            'u_clk_wgt'\n            , 'u_order_wgt'\n            , 'u_freq_quarter_day'\n            , 'u_freq_dow'\n            , 'u_activ_period'\n        ]\n    }\n    , f'ui_all_v{VER}.pqt': {\n        'on': ['session', 'aid']\n        , 'columns': [\n            'ui_clk_wgt'\n            , 'ui_order_wgt'\n            , 'ui_freq_quarter_day'\n            , 'ui_freq_dow'\n            , 'ui_last_ts'\n        ]\n    }\n}\nfile = f'order_carts_candidates_top{CANDIDATES}_v{VER}.pqt'\n# target = 'cart_target'\ncart_cand, TEST_SESSIONS = load_process_candidates(\n    file, TARGET_COL, CART_FEATURES\n    , dtype_map={\n        TARGET_COL: 'uint16'\n        , 'top_buy_wk': 'uint16'\n    }\n    , drop_cols = TARGET_DROP\n)\ncart_cand.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-31T18:19:29.869439Z","iopub.execute_input":"2023-01-31T18:19:29.869898Z","iopub.status.idle":"2023-01-31T18:20:08.315501Z","shell.execute_reply.started":"2023-01-31T18:19:29.869859Z","shell.execute_reply":"2023-01-31T18:20:08.314302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAINING:\n    labels = load_labels(TEST_SESSIONS)\n    # Count number of positives for stratified sampling\n    labels_pivot = labels.groupby(['session','type']).size().reset_index()\n    labels_pivot = labels_pivot.pivot(index='session', columns='type').reset_index()\n    labels_pivot = labels_pivot.fillna(0)\n    labels_pivot.columns = ['session', 'clicks', 'carts', 'orders']","metadata":{"execution":{"iopub.status.busy":"2023-01-31T18:20:08.318150Z","iopub.execute_input":"2023-01-31T18:20:08.318992Z","iopub.status.idle":"2023-01-31T18:20:10.173297Z","shell.execute_reply.started":"2023-01-31T18:20:08.318946Z","shell.execute_reply":"2023-01-31T18:20:10.172058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Training\n## Train/Val Split","metadata":{}},{"cell_type":"code","source":"def merge_cudf(df1, df2, on, how):\n    return df1.merge(df2, on=on, how=how)\nif TRAINING:    \n    # Stratified split\n    def split_train_val(event, label_df, n_splits=FOLDS):\n        X = label_df['session'].copy()\n        y = label_df[event].copy()\n        folder = skfold(n_splits=n_splits, shuffle=True, random_state=RS)\n        splits = []\n        for f, (tr, val) in enumerate(folder.split(X.to_array(), y.to_array())):\n            sessions = X.loc[val]\n            splits.append(cudf.DataFrame({'session': sessions, 'fold': [f]*len(sessions)}))\n        splits = cudf.concat(splits, ignore_index=True)\n        splits.fold = splits.fold.astype('uint16')\n        return splits\n    \n    cart_splits = split_train_val('carts', labels_pivot)\n#     cart_cand = cart_cand.merge(cart_splits, on='session', how='left')\n    cart_cand = merge_cudf(cart_cand, cart_splits, 'session', 'left')\n    del cart_splits, labels_pivot\n    gc.collect()\n    print(cart_cand.dtypes)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T18:20:10.175102Z","iopub.execute_input":"2023-01-31T18:20:10.175804Z","iopub.status.idle":"2023-01-31T18:20:11.093583Z","shell.execute_reply.started":"2023-01-31T18:20:10.175753Z","shell.execute_reply":"2023-01-31T18:20:11.092474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Model","metadata":{"execution":{"iopub.status.busy":"2023-01-03T15:13:52.250553Z","iopub.execute_input":"2023-01-03T15:13:52.250926Z","iopub.status.idle":"2023-01-03T15:13:52.574011Z","shell.execute_reply.started":"2023-01-03T15:13:52.25089Z","shell.execute_reply":"2023-01-03T15:13:52.573031Z"}}},{"cell_type":"code","source":"# Get Model\ndef get_xgb(params):\n#     params['eval_metric'] = recall_internal\n    return xgb.sklearn.XGBRanker(**params)\n\n# Prepare a DF into X and y format\ndef get_Xy(df, y_col, qid_col='session', aid_col='aid'):\n    df = df.sort_values('session', ascending=True).reset_index(drop=True)\n    qid = df[qid_col]\n    y = df[y_col]\n    aids = df[aid_col]\n    dropped = [y_col, qid_col, aid_col]\n    if 'fold' in df.columns: dropped.append('fold')\n    if 'random_rank' in df.columns: dropped.append('random_rank')\n    df = df.drop(columns=dropped)\n    return df, y, qid, aids\n\n# CALCULATE RECALL\n\ndef score_recall(predicts, population_targets):\n    # Calculate recall on candidate dataset\n    true_pos = len(predicts.loc[(predicts.predict == 1) & (predicts.target == 1)])\n    pos = predicts.target.astype('int32').sum()\n    candidate_recall = true_pos*1.0/pos\n    \n    # Calculate recall on population dataset\n    predicts = population_targets.merge(predicts, on=['session', 'aid'], how='left').fillna(0)\n    predicts = predicts.groupby('session').agg({'predict': ['count', 'sum']}).reset_index()\n    predicts.columns = ['session', 'actual_pos', 'true_pos']\n    predicts['actual_pos_trunc'] = predicts['actual_pos'].clip(upper=np.int32(20)) # error if not converted\n    population_recall = predicts['true_pos'].sum() / predicts['actual_pos_trunc'].sum()\n        \n    return candidate_recall, population_recall\n\ndef score_recall_by_round(\n    X_valid, qid_valid, aid_valid, session_valid, target_valid, target_type\n    , model, n_intervals=20\n    , labels=labels\n):\n    # Get predictions\n    def get_predictions(predicts, iteration_range=None):\n        if iteration_range is None:\n            pred = model.predict(X_valid)\n        else:\n            pred = model.predict(X_valid, iteration_range=iteration_range)\n        predicts['predict'] = pred\n        predicts = predicts.sort_values(['session', 'predict'], ascending=False).reset_index(drop=True)\n        predicts['n'] = predicts.groupby('session').cumcount()\n        predicts['predict'] = (predicts.n < 20).astype('int32')\n        return predicts.drop(columns=['n'])\n        \n    targets = labels.loc[(labels['type']==target_type) & (labels.session.isin(session_valid))]\n    targets = targets.rename(columns={'ground_truth':'aid'})\n    \n    trees = len(model.get_booster().get_dump())\n    if n_intervals > trees: n_intervals=trees\n    candidate_recall = {}\n    population_recall = {}\n    \n    print(f'Scoring {n_intervals} boosting round intervals...')\n    for intvl in np.array_split(np.array(range(trees)), n_intervals):\n        sta = 0\n        en = int(intvl[-1]+1)\n        print(f'{en}...', end='')\n        \n        predicts = cudf.DataFrame({\n            'session': qid_valid\n            , 'aid': aid_valid\n            , 'target': target_valid\n        })\n        predicts = get_predictions(predicts, iteration_range=[sta, en])\n        candidate_recall[en], population_recall[en] = score_recall(predicts, targets)\n    return candidate_recall, population_recall, candidate_recall[en], population_recall[en]\n\ndef plot(X, y, title, figsize=(10,7)):\n    plt.figure(figsize=(10,5))\n    plt.plot(X, y)\n    plt.title(title)\n    plt.show()\n\ndef train_model(\n    model_gen, params, model_save_name\n    , df, target_col, target_type, folds=FOLDS\n    , batches=None, plot_intervals=20\n):\n    results = {}\n    print('Model hyperparams:')\n    for k in params:\n        print(f'{k}: {params[k]}')\n    for f in range(FOLDS):\n        fid = f'FOLD {f}'\n        print('='*15 + '\\n' + fid + '...')\n\n        # Get data\n        s = dt.now()\n        X_val = df.loc[df['fold'] == f].reset_index(drop=True)\n        sess_val = X_val.session.unique().to_arrow().to_pylist()\n        print(f'Data loaded ({timer(s)} secs)!')\n        \n        # Prepare data\n        s = dt.now()\n        model = model_gen(params)\n        X_val, y_val, qid_val, aid_val = get_Xy(X_val, target_col)\n        print(f'Data prepared ({timer(s)} secs)!')\n\n        # Train model\n        s = dt.now()\n        if batches: # Training model in multiple batches\n            # Partly train - Train partly on each batch --> Lower scall than whole-training\n#             print(f'Training in {batches} batches...')\n#             folds = list(range(FOLDS))\n#             batch_estimators = np.array_split(list(range(params['n_estimators'])), batches)\n#             batch_estimators = np.array([len(ls) for ls in batch_estimators])\n#             folds.remove(f)\n#             batch_indexes = np.array_split(np.array(folds), batches)\n#             for i_e, est in enumerate(batch_estimators):\n#                 print(f'{est} estimators...', end='')\n#                 new_params = params.copy()\n#                 new_params['n_estimators'] = est\n#                 for i, idx in enumerate(batch_indexes):\n#                     X_tr = df.loc[df.fold.isin(idx)].reset_index(drop=True)\n#                     X_tr, y_tr, qid_tr, _ = get_Xy(X_tr, target_col)\n#                     if ((i_e==0) & (i==0)):\n#                         model_prev = None\n#                     else:\n#                         model_prev = model.get_booster()\n#                     model = model_gen(new_params)\n#                     model.fit(\n#                         X=X_tr, y=y_tr, qid=qid_tr\n#                         , eval_set=[(X_val, y_val)], eval_qid=[qid_val]\n#                         , xgb_model=model_prev\n#                         , verbose=0\n#                     )\n#                     print(f'{\"...\".join(map(str, idx))}...', end='')\n#                 print()\n            # Whole training - Train entire `n_estimators` on each batch\n            print(f'Training in {batches} batches...', end='')\n            folds = list(range(FOLDS))\n            folds.remove(f)\n            for b_idx, b in enumerate(np.array_split(np.array(folds), batches)):\n                X_tr = df.loc[df.fold.isin(b)].reset_index(drop=True)\n                X_tr, y_tr, qid_tr, _ = get_Xy(X_tr, target_col)\n                \n                if b_idx!=0:\n                    model_prev = model.get_booster()\n                    model = model_gen(params)\n                else:\n                    model_prev = None\n                model.fit(\n                    X=X_tr, y=y_tr, qid=qid_tr\n                    , eval_set=[(X_val, y_val)], eval_qid=[qid_val]\n                    , xgb_model=model_prev\n                    , verbose=0\n                )\n                print(f'{\"...\".join(map(str, b))}...', end='')\n        else:\n            print('Training entire set...', end='')\n            X_tr = df.loc[df['fold'] != f].reset_index(drop=True)\n            X_tr, y_tr, qid_tr, _ = get_Xy(X_tr, target_col)\n            model.fit(\n                X=X_tr, y=y_tr, qid=qid_tr\n                , eval_set=[(X_val, y_val)], eval_qid=[qid_val]\n                , verbose=0\n            )\n        print(f'finished ({timer(s)} secs)!')\n        # Score\n        s = dt.now()\n        cand, pop, f_cand, f_pop = score_recall_by_round(\n            X_val, qid_val, aid_val, sess_val, y_val\n            , target_type, model, n_intervals=plot_intervals)\n        print(f'Scoring finished ({timer(s)} secs)!')\n        # Save data\n        trees = len(model.get_booster().get_dump())\n        n_estimators = model.get_num_boosting_rounds()\n        print(f'{trees}/{n_estimators} trained')\n        if trees < n_estimators:\n            print(f'Best iteration: {model.best_iteration}')\n        results[fid] = {\n#             'model': model\n            'candidate_recall': cand\n            , 'final_candidate_recall': f_cand\n            , 'population_recall': pop\n            , 'final_population_recall': f_pop\n            , 'ft_importances': {}\n        }\n        for i_, name in enumerate(model.feature_names_in_):\n            results[fid]['ft_importances'][name] = str(model.feature_importances_[i_])\n        model.save_model(model_save_name + f'_Fold-{f}.json')\n        # Plot score\n        plt.figure(figsize=(14,5))\n        ax1 = plt.subplot(1,2,1)\n        ax2 = plt.subplot(1,2,2)\n        ax1.plot(list(cand.keys()), list(cand.values()))\n        ax1.title.set_text(f'FOLD {f} - Candidate recall: {round(f_cand, 5)}')\n        ax2.plot(list(pop.keys()), list(pop.values()))\n        ax2.title.set_text(f'FOLD {f} - Population recall: {round(f_pop, 5)}')\n        plt.show()\n#         if f==(FOLDS-2): break\n    \n    # Averaging score\n    cand_rec = np.array([results[f'FOLD {f}']['final_candidate_recall'] for f in range(len(results))])\n    cand_avg = cand_rec.mean()\n    cand_std = cand_rec.std()\n    pop_rec = np.array([results[f'FOLD {f}']['final_population_recall'] for f in range(len(results))])\n    pop_avg = pop_rec.mean()\n    pop_std = pop_rec.std()\n    print('='*20)\n    print(f'Candidate recall: mean={round(cand_avg,5)}, std={round(cand_std,5)}')\n    print(f'Population recall: mean={round(pop_avg,5)}, std={round(pop_std,5)}')\n    return results","metadata":{"execution":{"iopub.status.busy":"2023-01-31T18:20:11.097149Z","iopub.execute_input":"2023-01-31T18:20:11.097439Z","iopub.status.idle":"2023-01-31T18:20:11.133038Z","shell.execute_reply.started":"2023-01-31T18:20:11.097410Z","shell.execute_reply":"2023-01-31T18:20:11.131983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CALCULATE BASELINE\n# Using only handcrafted rules\ntmp_ft_cols = ['ui_order_wgt', 'b2b_cov_wgt'\n               , 'buy_cov_wgt', 'top_buy_wk']\ntmp_target_col = 'cart_target'\ntmp_target_type = 1\n\ntmp = cart_cand[['session', 'aid'] + tmp_ft_cols + [tmp_target_col]]\ntmp = tmp.sort_values(['session'] + tmp_ft_cols, ascending=False)\ntmp = tmp.reset_index(drop=True)\ntmp['n'] = tmp.groupby('session').cumcount()\ntmp['predict'] = (tmp.n < 20).astype('uint32')\ntmp = tmp[['session', 'aid', tmp_target_col, 'predict']]\ntmp = tmp.rename(columns={tmp_target_col: 'target'})\n\ntmp2 = labels.loc[labels['type']==tmp_target_type]\ntmp2 = tmp2.rename(columns={'ground_truth':'aid'})\nif TEST:\n    tmp2 = tmp2.loc[tmp2.session.isin(TEST_SESSIONS)]\n    \ncand_baseline, pop_baseline = score_recall(tmp, tmp2)\ndel tmp, tmp2\ngc.collect()\n\nprint('CART - Recall baseline')\nprint(f'Candidate recall: {round(cand_baseline,5)}')\nprint(f'Population recall: {round(pop_baseline,5)}')","metadata":{"execution":{"iopub.status.busy":"2023-01-31T18:20:11.134928Z","iopub.execute_input":"2023-01-31T18:20:11.135292Z","iopub.status.idle":"2023-01-31T18:20:12.266637Z","shell.execute_reply.started":"2023-01-31T18:20:11.135252Z","shell.execute_reply":"2023-01-31T18:20:12.265473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_DEPTH = 10\nXGB_PARAMS = {\n    'tree_method': 'gpu_hist'\n    , 'objective': XGB_OBJ\n    , 'n_estimators': 25\n    , 'max_depth': MAX_DEPTH\n    , 'max_leaves': int(2**(MAX_DEPTH-2))\n    , 'reg_alpha': 200\n    , 'reg_lambda': 5\n    , 'random_state': RS\n    , 'learning_rate': 0.005\n    , 'colsample_bytree': 0.8\n    , 'colsample_bylevel': 0.8\n    , 'colsample_bynode': 0.8\n#     , 'gamma': 10\n#     , 'early_stopping_rounds': 20\n}\n\ntraining_result = train_model(\n    get_xgb\n    , XGB_PARAMS\n    , MODEL_SAVE_NAME\n    , cart_cand\n    , target_col=TARGET_COL\n    , target_type=TARGET_TYPE\n    , batches=4\n    , plot_intervals=20\n)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-31T18:20:12.271329Z","iopub.execute_input":"2023-01-31T18:20:12.271813Z","iopub.status.idle":"2023-01-31T18:22:16.416817Z","shell.execute_reply.started":"2023-01-31T18:20:12.271765Z","shell.execute_reply":"2023-01-31T18:22:16.415646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_model = get_xgb({})\nmodel_pth = glob.glob(MODEL_SAVE_NAME + '*.json')\ntmp_model.load_model(model_pth[0])\nprint(tmp_model.feature_names_in_)\n\ntraining_result['feature_names'] = list(tmp_model.feature_names_in_)\nwith open('training_result.txt', 'w') as file:\n     file.write(json.dumps(training_result))","metadata":{"execution":{"iopub.status.busy":"2023-01-31T18:23:09.336714Z","iopub.execute_input":"2023-01-31T18:23:09.337431Z","iopub.status.idle":"2023-01-31T18:23:09.396453Z","shell.execute_reply.started":"2023-01-31T18:23:09.337394Z","shell.execute_reply":"2023-01-31T18:23:09.395536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FEATURE IMPORTANCES\nfi = training_result['FOLD 0']['ft_importances'].copy()\nfor f in range(1, FOLDS):\n    fold_fi = training_result[f'FOLD {f}']['ft_importances']\n    for ft in fold_fi:\n        fi[ft] = float(fi[ft]) + float(fold_fi[ft])\n        \nfi = pd.DataFrame({'features': fi.keys(), 'score': fi.values()})\nfi['score'] = fi['score'] / FOLDS\nfi = fi.sort_values('score', ascending=False).reset_index(drop=True)\nfi","metadata":{"execution":{"iopub.status.busy":"2023-01-31T18:33:49.691920Z","iopub.execute_input":"2023-01-31T18:33:49.692341Z","iopub.status.idle":"2023-01-31T18:33:49.698943Z","shell.execute_reply.started":"2023-01-31T18:33:49.692304Z","shell.execute_reply":"2023-01-31T18:33:49.697817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}