{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import gc\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport category_encoders as ce\nimport lightgbm as lgb\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn import metrics\n\npd.set_option('display.max_columns', None)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-06T18:18:32.845984Z","iopub.execute_input":"2022-06-06T18:18:32.846758Z","iopub.status.idle":"2022-06-06T18:18:35.384169Z","shell.execute_reply.started":"2022-06-06T18:18:32.846609Z","shell.execute_reply":"2022-06-06T18:18:35.383377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def compute_recall_at4(y_true: np.array, y_pred: np.array) -> float:\n    \n    # count of positives and negatives\n    n_pos = y_true.sum()\n    n_neg = y_true.shape[0] - n_pos\n    \n    # desc sorting by prediction values\n    indices = np.argsort(y_pred)[::-1]\n    target = y_true[indices]\n    \n    # filter the top 4% by cumulative row weights\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_mask = cum_norm_weight <= 0.04\n    \n    # default rate captured at 4%\n    d = target[four_pct_mask].sum() / n_pos\n    \n    return d\n\ndef compute_normalized_gini(y_true: np.array, y_pred: np.array) -> float:\n    \n    # count of positives and negatives\n    n_pos = y_true.sum()\n    n_neg = y_true.shape[0] - n_pos\n\n    # sorting desc by prediction values\n    indices = np.argsort(y_pred)[::-1]\n    target = y_true[indices]\n\n    # weighted gini coefficient\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n\n    lorentz = (target / n_pos).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    # max weighted gini coefficient\n    gini_max = 10 * n_neg * (1 - 19 / (n_pos + 20 * n_neg))\n\n    # normalized weighted gini coefficient\n    g = gini / gini_max\n    \n    return g\n    \ndef compute_amex_metric(y_true: np.array, y_pred: np.array) -> float:\n\n    # count of positives and negatives\n    n_pos = y_true.sum()\n    n_neg = y_true.shape[0] - n_pos\n\n    # sorting desc by prediction values\n    indices = np.argsort(y_pred)[::-1]\n    target = y_true[indices]\n\n    # filter the top 4% by cumulative row weights\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_filter = cum_norm_weight <= 0.04\n\n    # default rate captured at 4%\n    d = target[four_pct_filter].sum() / n_pos\n\n    # weighted gini coefficient\n    lorentz = (target / n_pos).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    # max weighted gini coefficient\n    gini_max = 10 * n_neg * (1 - 19 / (n_pos + 20 * n_neg))\n\n    # normalized weighted gini coefficient\n    g = gini / gini_max\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T18:18:40.548241Z","iopub.execute_input":"2022-06-06T18:18:40.548818Z","iopub.status.idle":"2022-06-06T18:18:40.565581Z","shell.execute_reply.started":"2022-06-06T18:18:40.548766Z","shell.execute_reply":"2022-06-06T18:18:40.564562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# metrics in lgbm format\n\ndef metric_recall_at4(y_pred: np.ndarray, data: lgb.Dataset):\n    y_true = data.get_label()\n    return 'recall_at4', compute_recall_at4(y_true, y_pred), True\n\ndef metric_normalized_gini(y_pred: np.ndarray, data: lgb.Dataset):\n    y_true = data.get_label()\n    return 'norm_gini', compute_normalized_gini(y_true, y_pred), True\n\ndef metric_amex(y_pred: np.ndarray, data: lgb.Dataset):\n    y_true = data.get_label()\n    return 'amex_metric', compute_amex_metric(y_true, y_pred), True","metadata":{"execution":{"iopub.status.busy":"2022-06-06T18:18:47.168622Z","iopub.execute_input":"2022-06-06T18:18:47.169132Z","iopub.status.idle":"2022-06-06T18:18:47.177884Z","shell.execute_reply.started":"2022-06-06T18:18:47.169092Z","shell.execute_reply":"2022-06-06T18:18:47.176409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***\n## load and prepare data","metadata":{}},{"cell_type":"code","source":"train = pd.read_parquet(\"../input/amex-data-integer-dtypes-parquet-format/train.parquet\")\ntrain_labels = pd.read_csv(\"../input/amex-default-prediction/train_labels.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-06-06T18:18:48.817679Z","iopub.execute_input":"2022-06-06T18:18:48.818156Z","iopub.status.idle":"2022-06-06T18:19:10.063739Z","shell.execute_reply.started":"2022-06-06T18:18:48.818122Z","shell.execute_reply":"2022-06-06T18:19:10.062460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_agg = (\n    train\n    .sort_values([\"customer_ID\",\"S_2\"], ascending=[True,False])\n    .drop_duplicates(subset=[\"customer_ID\"], keep=\"first\", ignore_index=True)\n)\n\ndel train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T18:19:12.951464Z","iopub.execute_input":"2022-06-06T18:19:12.951879Z","iopub.status.idle":"2022-06-06T18:19:21.341310Z","shell.execute_reply.started":"2022-06-06T18:19:12.951844Z","shell.execute_reply":"2022-06-06T18:19:21.340539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_feats = train_agg.columns[2:].tolist()\ncateg_feats = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']","metadata":{"execution":{"iopub.status.busy":"2022-06-06T18:19:21.342841Z","iopub.execute_input":"2022-06-06T18:19:21.343901Z","iopub.status.idle":"2022-06-06T18:19:21.349366Z","shell.execute_reply.started":"2022-06-06T18:19:21.343864Z","shell.execute_reply":"2022-06-06T18:19:21.348280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_agg = pd.merge(train_agg, train_labels, how=\"inner\", on=\"customer_ID\")","metadata":{"execution":{"iopub.status.busy":"2022-06-06T18:19:21.350755Z","iopub.execute_input":"2022-06-06T18:19:21.351623Z","iopub.status.idle":"2022-06-06T18:19:28.411027Z","shell.execute_reply.started":"2022-06-06T18:19:21.351587Z","shell.execute_reply":"2022-06-06T18:19:28.410059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in input_feats:\n    print(col, train_agg[col].dtype)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T18:19:31.097731Z","iopub.execute_input":"2022-06-06T18:19:31.098263Z","iopub.status.idle":"2022-06-06T18:19:31.122580Z","shell.execute_reply.started":"2022-06-06T18:19:31.098226Z","shell.execute_reply":"2022-06-06T18:19:31.121761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***\n## model training","metadata":{}},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=2112)\nskf_split = list(skf.split(train_agg, train_agg[\"target\"].values))","metadata":{"execution":{"iopub.status.busy":"2022-06-06T18:19:34.867937Z","iopub.execute_input":"2022-06-06T18:19:34.868370Z","iopub.status.idle":"2022-06-06T18:19:34.974840Z","shell.execute_reply.started":"2022-06-06T18:19:34.868332Z","shell.execute_reply":"2022-06-06T18:19:34.973512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_params = {\n    'objective': 'binary',\n    'metric': 'None',\n    'learning_rate': 0.05,\n    'num_leaves': 64,\n    'force_col_wise': True,\n    'bagging_freq': 1,\n    'seed': 2112,\n    'verbosity': 0,\n    'first_metric_only': True,\n    'bin_construct_sample_cnt': 100000000,\n    'feature_pre_filter': False,\n    'bagging_fraction': 0.9,\n    'feature_fraction': 0.2,\n    'lambda_l1': 0.1,\n    'lambda_l2': 0.1,\n    'min_data_in_leaf': 1000,\n    'path_smooth': 10,\n    'max_bin': 255,\n}","metadata":{"execution":{"iopub.status.busy":"2022-06-06T18:19:36.798769Z","iopub.execute_input":"2022-06-06T18:19:36.799802Z","iopub.status.idle":"2022-06-06T18:19:36.806719Z","shell.execute_reply.started":"2022-06-06T18:19:36.799745Z","shell.execute_reply":"2022-06-06T18:19:36.805860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodels = list()\n\n# dataframe to store the oof predictions\noof = train_agg[[\"target\"]].copy()\noof[\"pred\"] = -1\n\nfor fold,(train_idx,valid_idx) in enumerate(skf_split):\n    \n    print(f\" training model {fold+1}/{len(skf_split)} \".center(100, \"#\"))\n    \n    train_dset = lgb.Dataset(\n        data=train_agg.loc[train_idx,input_feats],\n        label=train_agg.loc[train_idx,\"target\"].values,\n        categorical_feature=categ_feats,\n        free_raw_data=True\n    )\n    valid_dset = lgb.Dataset(\n        data=train_agg.loc[valid_idx,input_feats],\n        label=train_agg.loc[valid_idx,\"target\"].values,\n        categorical_feature=categ_feats,\n        free_raw_data=True\n    )\n    \n    model = lgb.train(\n        params=model_params,\n        train_set=train_dset,\n        valid_sets=[valid_dset,],\n        feval=[metric_amex, metric_recall_at4, metric_normalized_gini],\n        num_boost_round=3000,\n        callbacks=[lgb.log_evaluation(period=50), lgb.early_stopping(50)],\n    )\n    \n    lgb.plot_importance(model, figsize=(8,15), importance_type=\"split\", max_num_features=30)\n    lgb.plot_importance(model, figsize=(8,15), importance_type=\"gain\", max_num_features=30)\n    plt.show()\n    \n    oof.loc[valid_idx,\"pred\"] = model.predict(train_agg.loc[valid_idx,input_feats])\n    \n    models.append(model)\n    del train_dset,valid_dset\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T18:19:38.412787Z","iopub.execute_input":"2022-06-06T18:19:38.413376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# oof metrics\nprint(\"OOF recall_at4:\", compute_recall_at4(oof.target.values, oof.pred.values))\nprint(\"OOF normalized_gini:\", compute_normalized_gini(oof.target.values, oof.pred.values))\nprint(\"OOF competition metric:\", compute_amex_metric(oof.target.values, oof.pred.values))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_agg\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***\n## make predictions and submit","metadata":{}},{"cell_type":"code","source":"test = pd.read_parquet(\"../input/amex-data-integer-dtypes-parquet-format/test.parquet\")\nsample_sub = pd.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")\nsample_sub","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_agg = (\n    test\n    .sort_values([\"customer_ID\",\"S_2\"], ascending=[True,False])\n    .drop_duplicates(subset=[\"customer_ID\"], keep=\"first\", ignore_index=True)\n)\n\ndel test\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\npreds = [model.predict(test_agg[input_feats]) for model in models]\ntest_agg[\"prediction\"] = np.mean(preds, axis=0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.merge(sample_sub[[\"customer_ID\"]], test_agg[[\"customer_ID\",\"prediction\"]])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"assert sub.prediction.isna().sum() == 0","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***","metadata":{}}]}