{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This kernel for Catboost training cycle with cross-validation, based on example from competition and some other popular notebooks.","metadata":{}},{"cell_type":"code","source":"import random\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import KFold, StratifiedKFold, train_test_split\nfrom sklearn.metrics import roc_auc_score\n\nimport vaex\nvaex.multithreading.thread_count_default = 8\nimport vaex.ml\n\nfrom cycler import cycler\nfrom IPython.display import display\nimport datetime\nimport scipy.stats\nimport warnings\nfrom colorama import Fore, Back, Style\nimport gc\n\nfrom catboost import CatBoostClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:53:28.502424Z","iopub.execute_input":"2022-07-19T10:53:28.502777Z","iopub.status.idle":"2022-07-19T11:41:07.645678Z","shell.execute_reply.started":"2022-07-19T10:53:28.502748Z","shell.execute_reply":"2022-07-19T11:41:07.644653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_data(read_from_cache=True):\n    train = pd.read_parquet(TRAIN_FEAT_PATH)\n    test = pd.read_parquet(TEST_FEAT_PATH)\n    return train, test\n\ndef plot_feature_importance(importance,names,model_type, n=50, figsize=(16,10)):\n\n    #Create arrays from feature importance and feature names\n    feature_importance = np.array(importance)\n    feature_names = np.array(names)\n\n    #Create a DataFrame using a Dictionary\n    data={'feature_names':feature_names,'feature_importance':feature_importance}\n    fi_df = pd.DataFrame(data)\n\n    #Sort the DataFrame in order decreasing feature importance\n    fi_df.sort_values(by=['feature_importance'], ascending=False,inplace=True)\n    fi_df = fi_df[:n]\n\n    #Define size of bar plot\n    plt.figure(figsize=figsize)\n    #Plot Searborn bar chart\n    sns.barplot(x=fi_df['feature_importance'], y=fi_df['feature_names'])\n    #Add chart labels\n    plt.title(model_type + ' FEATURE IMPORTANCE')\n    plt.xlabel('FEATURE IMPORTANCE')\n    plt.ylabel('FEATURE NAMES')\n    plt.tight_layout()\n    plt.show()\n    return fi_df.feature_names\n\ndef amex_metric(y_true: np.array, y_pred: np.array) -> float:\n\n    # count of positives and negatives\n    n_pos = y_true.sum()\n    n_neg = y_true.shape[0] - n_pos\n\n    # sorting by describing prediction values\n    indices = np.argsort(y_pred)[::-1]\n    preds, target = y_pred[indices], y_true[indices]\n\n    # filter the top 4% by cumulative row weights\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_filter = cum_norm_weight <= 0.04\n\n    # default rate captured at 4%\n    d = target[four_pct_filter].sum() / n_pos\n\n    # weighted Gini coefficient\n    lorentz = (target / n_pos).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    # max weighted Gini coefficient\n    gini_max = 10 * n_neg * (1 - 19 / (n_pos + 20 * n_neg))\n\n    # normalized weighted Gini coefficient\n    g = gini / gini_max\n\n    return 0.5 * (g + d)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# config\n\nMAX_DEPTH = 10\nITERATIONS = 28000\n\nDATA_PATH = '../input/amex-data-integer-dtypes-parquet-format/'   # denoised data from raddar\nLABELS_PATH = '../input/amex-default-prediction/train_labels.csv' # original data sources\n\nTEST_FEAT_PATH = '../input/amex-denoised-aggregated-features/test_feat.parquet' # aggregated features I shared\nTRAIN_FEAT_PATH = '../input/amex-denoised-aggregated-features/train_feat.parquet'","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param = {\n    \"objective\": \"CrossEntropy\",\n    \"learning_rate\": 0.01,\n    \"n_estimators\": ITERATIONS,\n    \"max_depth\": MAX_DEPTH,\n    \"l2_leaf_reg\": 25,\n    \"od_type\": \"Iter\",\n    \"od_wait\": 700,\n    \"task_type\":\"GPU\",\n    \"use_best_model\": True,\n    \"random_strength\": 2.1,\n    'bagging_temperature': 0.4,\n\n}\n\ntarget = pd.read_csv(LABELS_PATH).target.values\ntrain, test = get_data(read_from_cache=True)\nprint(f\"target shape: {target.shape}, train shape: {train.shape}, test shape: {test.shape}\")\nprint(f\"target shape: {target.shape}, train shape: {train.shape}\")\nfeatures = [f for f in train.columns if f != 'customer_ID' and f != 'target']\n\ncv_folds = 5\nONLY_FIRST_FOLD = False\nscore_list, y_pred_list = [], []\nkf = StratifiedKFold(n_splits=cv_folds)\n\nfor fold, (idx_tr, idx_va) in enumerate(kf.split(train, target)):\n    X_tr, X_va, y_tr, y_va, model = None, None, None, None, None\n    start_time = datetime.datetime.now()\n    X_tr = train.iloc[idx_tr][features]\n    X_va = train.iloc[idx_va][features]\n    y_tr = target[idx_tr]\n    y_va = target[idx_va]\n\n    with warnings.catch_warnings():\n        warnings.filterwarnings('ignore', category=UserWarning)\n        model = CatBoostClassifier(**param)\n        model.fit(X_tr, y_tr,\n                  eval_set = [(X_va, y_va)], \n                  metric_period=100\n                 )\n    X_tr, y_tr = None, None\n    y_va_pred = model.predict_proba(X_va)[:,1]\n    score = amex_metric(y_va, y_va_pred)\n    n_trees = model.best_iteration_\n    if n_trees is None: n_trees = model.n_estimators\n    print(f\"{Fore.GREEN}{Style.BRIGHT}Fold {fold} | {str(datetime.datetime.now() - start_time)[-12:-7]} |\"\n          f\" {n_trees:5} trees |\"\n          f\"                Score = {score:.5f}{Style.RESET_ALL}\")\n\n    score_list.append(score)\n    y_pred_list.append(model.predict_proba(test[features])[:,1])\n    gc.collect()\n    if ONLY_FIRST_FOLD: break # we only want the first fold\n\nprint(f\"{Fore.GREEN}{Style.BRIGHT}OOF Score:                       {np.mean(score_list):.5f}{Style.RESET_ALL}\")\ngc.collect()\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.get_all_params()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:41:08.936914Z","iopub.execute_input":"2022-07-19T11:41:08.938532Z","iopub.status.idle":"2022-07-19T11:41:08.948079Z","shell.execute_reply.started":"2022-07-19T11:41:08.938492Z","shell.execute_reply":"2022-07-19T11:41:08.946950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top_n = plot_feature_importance(model.feature_importances_, features,'CatBoost', n=80, figsize=(20,12))\n\nprint(f'Mean CV score for {MAX_DEPTH} max_depth: {np.mean(score_list)}')","metadata":{"execution":{"iopub.status.busy":"2022-07-19T13:08:17.095525Z","iopub.execute_input":"2022-07-19T13:08:17.095901Z","iopub.status.idle":"2022-07-19T13:08:17.171858Z","shell.execute_reply.started":"2022-07-19T13:08:17.095822Z","shell.execute_reply":"2022-07-19T13:08:17.170506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'LGBM baseline: {np.mean([0.79605, 0.79674, 0.79362, 0.79283, 0.79442])}')","metadata":{"execution":{"iopub.status.busy":"2022-07-19T07:51:59.674184Z","iopub.execute_input":"2022-07-19T07:51:59.674529Z","iopub.status.idle":"2022-07-19T07:51:59.679803Z","shell.execute_reply.started":"2022-07-19T07:51:59.6745Z","shell.execute_reply":"2022-07-19T07:51:59.67887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame({'customer_ID': test.index,\n                    'prediction':  np.exp((np.log(y_pred_list[0])+np.log(y_pred_list[1])+np.log(y_pred_list[2])+np.log(y_pred_list[3])+np.log(y_pred_list[4]))/5)})\nsub.to_csv('./submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T10:16:49.504656Z","iopub.execute_input":"2022-07-19T10:16:49.505003Z","iopub.status.idle":"2022-07-19T10:16:54.258397Z","shell.execute_reply.started":"2022-07-19T10:16:49.504976Z","shell.execute_reply":"2022-07-19T10:16:54.257437Z"},"trusted":true},"execution_count":null,"outputs":[]}]}