{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# American Express - Default Prediction\n> Calculated by observing 18 months performance window after the latest credit card statement, <br>\nand if the customer does not pay due amount in 120 days after their latest statement date it is considered a default event.","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5"}},{"cell_type":"markdown","source":"It is based on the <a href=\"https://www.kaggle.com/code/ambrosm/amex-eda-which-makes-sense\" target=\"_blank\">EDA which makes sense ⭐️⭐️⭐️⭐️⭐️</a>.","metadata":{}},{"cell_type":"markdown","source":"## 1. Check Data\n\nD_* : Delinquency variables<br>\nS_* : Spend variables <br>\nP_* : Payment variables <br>\nB_* : Balance variables <br>\nR_* : Risk variables <br>","metadata":{}},{"cell_type":"markdown","source":"### Labels\nFor trainnig data, We can know there aren't duplicated customer_ID. <br>\nBecause of imbalancing classes, It needs cross-validation.","metadata":{}},{"cell_type":"code","source":"# import pandas as pd\n# import numpy as np\n# import pickle, gc\n# from matplotlib import pyplot as plt","metadata":{"execution":{"iopub.execute_input":"2022-07-12T05:24:29.026709Z","iopub.status.busy":"2022-07-12T05:24:29.026255Z","iopub.status.idle":"2022-07-12T05:24:29.055760Z","shell.execute_reply":"2022-07-12T05:24:29.054090Z","shell.execute_reply.started":"2022-07-12T05:24:29.026622Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_labels = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\n# train_labels.head()","metadata":{"execution":{"iopub.execute_input":"2022-07-12T05:24:59.210465Z","iopub.status.busy":"2022-07-12T05:24:59.210079Z","iopub.status.idle":"2022-07-12T05:25:00.217271Z","shell.execute_reply":"2022-07-12T05:25:00.216097Z","shell.execute_reply.started":"2022-07-12T05:24:59.210432Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # check null data\n# train_labels.isnull().sum()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# label_stats = pd.DataFrame({'absolute': train_labels.target.value_counts(),\n#               'relative': train_labels.target.value_counts() / len(train_labels)})\n# label_stats['absolute upsampled'] =  label_stats.absolute * np.array([20, 1])\n# label_stats['relative upsampled'] = label_stats['absolute upsampled'] / label_stats['absolute upsampled'].sum()\n# label_stats","metadata":{"scrolled":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # import data\n# train = pd.read_feather('../input/amex-default-prediction/test_data.csv')\n# test = pd.read_feather('../input/amex-default-prediction/train_data.csv')\n# with pd.option_context(\"display.min_rows\", 6):\n#     display(train)\n#     display(test)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train.info(max_cols=200, show_counts=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> There are many columns and many missing data -> Don't have to drop columns which have null data. <br>\nThere are many 16-bit float features -> Don't have to round up.","metadata":{}},{"cell_type":"code","source":"# # Check the statements per customer\n# fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 5))\n# train_sc = train.customer_ID.value_counts().value_counts().sort_index(ascending=False).rename('Train statements per customer')\n# ax1.pie(train_sc, labels=train_sc.index)\n# ax1.set_title(train_sc.name)\n\n# test_sc = test.customer_ID.value_counts().value_counts().sort_index(ascending=False).rename('Test statements per customer')\n# ax2.pie(test_sc, labels=train_sc.index)\n# ax2.set_title(test_sc.name)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Find customers last stement\n# temp = train['S_2'].groupby(train.customer_ID).max()\n# plt.figure(figsize=(16, 4))\n# plt.hist(temp, bins=pd.date_range(\"2018-03-01\", \"2018-04-01\", freq=\"d\"), rwidth=0.8, color='#ffd700')\n# plt.title('When did the train customers get their last statements?', fontsize=20)\n# plt.xlabel('Last statement date per customer')\n# plt.ylabel('Count')\n# plt.gca().set_facecolor('#0057b8')\n# plt.show()\n# del temp\n\n# temp = test['S_2'].groupby(test.customer_ID).max()\n# plt.figure(figsize=(16, 4))\n# plt.hist(temp, bins=pd.date_range(\"2019-04-01\", \"2019-11-01\", freq=\"d\"), rwidth=0.74, color='#ffd700')\n# plt.title('When did the test customers get their last statements?', fontsize=20)\n# plt.xlabel('Last statement date per customer')\n# plt.ylabel('Count')\n# plt.gca().set_facecolor('#0057b8')\n# plt.show()\n# del temp","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Train data placed between \"2018-03-01\" and \"2018-04-01\", <br>\nTest data placed between \"2019-04-01\"and \"2019-11-01\"","metadata":{}},{"cell_type":"code","source":"# temp =  pd.concat([train[['customer_ID', 'S_2']], test[['customer_ID', 'S_2']]], axis=0)\n# temp.set_index('customer_ID', inplace=True)\n# temp['last_month'] = temp.groupby('customer_ID')['S_2'].max().dt.month\n# last_month = temp['last_month'].values\n\n# plt.figure(figsize=(16, 4))\n# plt.hist([temp.S_2[temp.last_month == 3],\n#           temp.S_2[temp.last_month == 4],\n#           temp.S_2[temp.last_month == 10]],\n#          bins=pd.date_range(\"2017-03-01\", \"2019-11-01\", freq=\"MS\"),\n#          label=['Training', 'Public leaderboard', 'Private leaderboard'],\n#          stacked=True)\n# plt.xticks(pd.date_range(\"2017-03-01\", \"2019-11-01\", freq=\"QS\"))\n# plt.xlabel('Statement date')\n# plt.ylabel('Count')\n# plt.title('The three datasets over time', fontsize=20)\n# plt.legend()\n# plt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for f in [ 'B_29', 'S_9','D_87']:\n#     temp = pd.concat([train[[f, 'S_2']], test[[f, 'S_2']]], axis=0)\n#     temp['last_month'] = last_month\n#     temp['has_f'] = ~temp[f].isna() \n\n#     plt.figure(figsize=(16, 4))\n#     plt.hist([temp.S_2[temp.has_f & (temp.last_month == 3)],   # ending 03/18 -> training\n#               temp.S_2[temp.has_f & (temp.last_month == 4)],   # ending 04/19 -> public lb\n#               temp.S_2[temp.has_f & (temp.last_month == 10)]], # ending 10/19 -> private lb\n#              bins=pd.date_range(\"2017-03-01\", \"2019-11-01\", freq=\"MS\"),\n#              label=['Training', 'Public leaderboard', 'Private leaderboard'],\n#              stacked=True)\n#     plt.xticks(pd.date_range(\"2017-03-01\", \"2019-11-01\", freq=\"QS\"))\n#     plt.xlabel('Statement date')\n#     plt.ylabel(f'Count of {f} non-null values')\n#     plt.title(f'{f} non-null values over time', fontsize=20)\n#     plt.legend()\n#     plt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # The categorical features - histograms for target = 0,1\n# cat_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n# plt.figure(figsize=(16, 16))\n# for i, f in enumerate(cat_features): # 0, 1, ... B_30, B_38, D_114, ...\n#     plt.subplot(4, 3, i+1)\n#     temp = pd.DataFrame(train[f][train.target == 0].value_counts(dropna=False, normalize=True).sort_index().rename('count'))\n#     temp.index.name = 'value'\n#     temp.reset_index(inplace=True)\n#     plt.bar(temp.index, temp['count'], alpha=0.5, label='target=0')\n#     temp = pd.DataFrame(train[f][train.target == 1].value_counts(dropna=False, normalize=True).sort_index().rename('count'))\n#     temp.index.name = 'value'\n#     temp.reset_index(inplace=True)\n#     plt.bar(temp.index, temp['count'], alpha=0.5, label='target=1')\n#     plt.xlabel(f)\n#     plt.ylabel('frequency')\n#     plt.legend()\n#     plt.xticks(temp.index, temp.value)\n# plt.suptitle('Categorical features', fontsize=20, y=0.93)\n# plt.show()\n# del temp","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # The binary features\n# bin_features = ['B_31', 'D_87']\n# plt.figure(figsize=(16, 4))\n# for i, f in enumerate(bin_features):\n#     plt.subplot(1, 2, i+1)\n#     temp = pd.DataFrame(train[f][train.target == 0].value_counts(dropna=False, normalize=True).sort_index().rename('count'))\n#     temp.index.name = 'value'\n#     temp.reset_index(inplace=True)\n#     plt.bar(temp.index, temp['count'], alpha=0.5, label='target=0')\n#     temp = pd.DataFrame(train[f][train.target == 1].value_counts(dropna=False, normalize=True).sort_index().rename('count'))\n#     temp.index.name = 'value'\n#     temp.reset_index(inplace=True)\n#     plt.bar(temp.index, temp['count'], alpha=0.5, label='target=1')\n#     plt.xlabel(f)\n#     plt.ylabel('frequency')\n#     plt.legend()\n#     plt.xticks(temp.index, temp.value)\n# plt.suptitle('Binary features', fontsize=20)\n# plt.show()\n# del temp","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # The numerical features - without categorical, binary, id, target and S_2\n# cont_features = sorted([f for f in train.columns if f not in cat_features + bin_features + ['customer_ID', 'target', 'S_2']])\n# print(len(cont_features))\n# # print(cont_features)\n# ncols = 4\n# for i, f in enumerate(cont_features):\n#     if i % ncols == 0: \n#         if i > 0: plt.show()\n#         plt.figure(figsize=(16, 3))\n#         if i == 0: plt.suptitle('Continuous features', fontsize=20, y=1.02)\n#     plt.subplot(1, ncols, i % ncols + 1)\n#     plt.hist(train[f], bins=200)\n#     plt.xlabel(f)\n# plt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Search data with noise data\n# def read_columns(name, features):\n#     \"\"\"Read the specified columns of the train/test csv at full precision\"\"\"\n#     chunksize = 1000000\n#     chunklist = []\n#     with pd.read_csv(f\"./{name}_data.csv\", chunksize=chunksize) as reader:\n#         for i, chunk in enumerate(reader):\n#             chunk = chunk[features] # keep only selected columns\n#             chunklist.append(chunk)\n#             print(i, end=' ')\n#             if i == 5: break\n#         print()\n#     df = pd.concat(chunklist, axis=0)\n#     return df\n\n# df = read_columns('train', ['B_19', 'S_13'])\n# df.info()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # B_19 data histograme\n# y = df['B_19']\n# for i in np.linspace(0, 1, 11):\n#     plt.figure(figsize=(16, 3))\n#     plt.hist(y, bins=np.linspace(i, i+0.1, 101), rwidth=0.8, color='m')\n#     plt.xticks(np.linspace(i, i+0.1, 11))\n#     plt.title(f\"B_19 histogram, part {int(i*10+1)}\")\n#     plt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Don't have to scale data but have to remove noise data","metadata":{}},{"cell_type":"code","source":"# # S_13 data histograme\n# y = df['S_13']\n# for i in np.linspace(0, 1, 11):\n#     plt.figure(figsize=(16, 3))\n#     plt.hist(y, bins=np.linspace(i, i+0.1, 101), rwidth=0.8, color='c')\n#     plt.xticks(np.linspace(i, i+0.1, 11))\n#     plt.title(f\"S_13 histogram, part {int(i*10+1)}\")\n#     plt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Because of overlap, It cannot remove data between 0.68 and 0.7","metadata":{}},{"cell_type":"markdown","source":"## LightGBM Prediction","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator\nfrom matplotlib.colors import ListedColormap\nfrom cycler import cycler\nfrom IPython.display import display\nimport datetime\nimport scipy.stats\nimport warnings\nfrom colorama import Fore, Back, Style\nimport gc\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.calibration import CalibrationDisplay\nfrom lightgbm import LGBMClassifier, log_evaluation\n\nplt.rcParams['axes.facecolor'] = '#0057b8' # blue\nplt.rcParams['axes.prop_cycle'] = cycler(color=['#ffd700'] +\n                                         plt.rcParams['axes.prop_cycle'].by_key()['color'][1:])\nplt.rcParams['text.color'] = 'w'\n\nINFERENCE = True # set to False if you only want to cross-validate","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# @yunchonggan's fast metric implementation\n# From https://www.kaggle.com/competitions/amex-default-prediction/discussion/328020\ndef amex_metric(y_true: np.array, y_pred: np.array) -> float:\n\n    # count of positives and negatives\n    n_pos = y_true.sum()\n    n_neg = y_true.shape[0] - n_pos\n\n    # sorting by descring prediction values\n    indices = np.argsort(y_pred)[::-1]\n    preds, target = y_pred[indices], y_true[indices]\n\n    # filter the top 4% by cumulative row weights\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_filter = cum_norm_weight <= 0.04\n\n    # default rate captured at 4%\n    d = target[four_pct_filter].sum() / n_pos\n\n    # weighted gini coefficient\n    lorentz = (target / n_pos).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n\n    # max weighted gini coefficient\n    gini_max = 10 * n_neg * (1 - 19 / (n_pos + 20 * n_neg))\n\n    # normalized weighted gini coefficient\n    g = gini / gini_max\n\n    return 0.5 * (g + d)\n\ndef lgb_amex_metric(y_true, y_pred):\n    \"\"\"The competition metric with lightgbm's calling convention\"\"\"\n    return ('amex',\n            amex_metric(y_true, y_pred),\n            True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing the Data\n- LightGBM handles missing values automatically.\n- LightGBM handles Categorical features to be one-hot encoding automatically.\n- Tree-based algorithms deal with outliers easily.\n- Neural networks need scaled inputs; tree-based algorithms don't depend on scaling.","metadata":{}},{"cell_type":"code","source":"%%time\nfeatures_avg = ['B_1', 'B_2', 'B_3', 'B_4', 'B_5', 'B_6', 'B_8', 'B_9', 'B_10', 'B_11', 'B_12', 'B_13', 'B_14', 'B_15', 'B_16', 'B_17', 'B_18', 'B_19', 'B_20', 'B_21', 'B_22', 'B_23', 'B_24', 'B_25', 'B_28', 'B_29', 'B_30', 'B_32', 'B_33', 'B_37', 'B_38', 'B_39', 'B_40', 'B_41', 'B_42', 'D_39', 'D_41', 'D_42', 'D_43', 'D_44', 'D_45', 'D_46', 'D_47', 'D_48', 'D_50', 'D_51', 'D_53', 'D_54', 'D_55', 'D_58', 'D_59', 'D_60', 'D_61', 'D_62', 'D_65', 'D_66', 'D_69', 'D_70', 'D_71', 'D_72', 'D_73', 'D_74', 'D_75', 'D_76', 'D_77', 'D_78', 'D_80', 'D_82', 'D_84', 'D_86', 'D_91', 'D_92', 'D_94', 'D_96', 'D_103', 'D_104', 'D_108', 'D_112', 'D_113', 'D_114', 'D_115', 'D_117', 'D_118', 'D_119', 'D_120', 'D_121', 'D_122', 'D_123', 'D_124', 'D_125', 'D_126', 'D_128', 'D_129', 'D_131', 'D_132', 'D_133', 'D_134', 'D_135', 'D_136', 'D_140', 'D_141', 'D_142', 'D_144', 'D_145', 'P_2', 'P_3', 'P_4', 'R_1', 'R_2', 'R_3', 'R_7', 'R_8', 'R_9', 'R_10', 'R_11', 'R_14', 'R_15', 'R_16', 'R_17', 'R_20', 'R_21', 'R_22', 'R_24', 'R_26', 'R_27', 'S_3', 'S_5', 'S_6', 'S_7', 'S_9', 'S_11', 'S_12', 'S_13', 'S_15', 'S_16', 'S_18', 'S_22', 'S_23', 'S_25', 'S_26']\nfeatures_min = ['B_2', 'B_4', 'B_5', 'B_9', 'B_13', 'B_14', 'B_15', 'B_16', 'B_17', 'B_19', 'B_20', 'B_28', 'B_29', 'B_33', 'B_36', 'B_42', 'D_39', 'D_41', 'D_42', 'D_45', 'D_46', 'D_48', 'D_50', 'D_51', 'D_53', 'D_55', 'D_56', 'D_58', 'D_59', 'D_60', 'D_62', 'D_70', 'D_71', 'D_74', 'D_75', 'D_78', 'D_83', 'D_102', 'D_112', 'D_113', 'D_115', 'D_118', 'D_119', 'D_121', 'D_122', 'D_128', 'D_132', 'D_140', 'D_141', 'D_144', 'D_145', 'P_2', 'P_3', 'R_1', 'R_27', 'S_3', 'S_5', 'S_7', 'S_9', 'S_11', 'S_12', 'S_23', 'S_25']\nfeatures_max = ['B_1', 'B_2', 'B_3', 'B_4', 'B_5', 'B_6', 'B_7', 'B_8', 'B_9', 'B_10', 'B_12', 'B_13', 'B_14', 'B_15', 'B_16', 'B_17', 'B_18', 'B_19', 'B_21', 'B_23', 'B_24', 'B_25', 'B_29', 'B_30', 'B_33', 'B_37', 'B_38', 'B_39', 'B_40', 'B_42', 'D_39', 'D_41', 'D_42', 'D_43', 'D_44', 'D_45', 'D_46', 'D_47', 'D_48', 'D_49', 'D_50', 'D_52', 'D_55', 'D_56', 'D_58', 'D_59', 'D_60', 'D_61', 'D_63', 'D_64', 'D_65', 'D_70', 'D_71', 'D_72', 'D_73', 'D_74', 'D_76', 'D_77', 'D_78', 'D_80', 'D_82', 'D_84', 'D_91', 'D_102', 'D_105', 'D_107', 'D_110', 'D_111', 'D_112', 'D_115', 'D_116', 'D_117', 'D_118', 'D_119', 'D_121', 'D_122', 'D_123', 'D_124', 'D_125', 'D_126', 'D_128', 'D_131', 'D_132', 'D_133', 'D_134', 'D_135', 'D_136', 'D_138', 'D_140', 'D_141', 'D_142', 'D_144', 'D_145', 'P_2', 'P_3', 'P_4', 'R_1', 'R_3', 'R_5', 'R_6', 'R_7', 'R_8', 'R_10', 'R_11', 'R_14', 'R_17', 'R_20', 'R_26', 'R_27', 'S_3', 'S_5', 'S_7', 'S_8', 'S_11', 'S_12', 'S_13', 'S_15', 'S_16', 'S_22', 'S_23', 'S_24', 'S_25', 'S_26', 'S_27']\nfeatures_last = ['B_1', 'B_2', 'B_3', 'B_4', 'B_5', 'B_6', 'B_7', 'B_8', 'B_9', 'B_10', 'B_11', 'B_12', 'B_13', 'B_14', 'B_15', 'B_16', 'B_17', 'B_18', 'B_19', 'B_20', 'B_21', 'B_22', 'B_23', 'B_24', 'B_25', 'B_26', 'B_28', 'B_29', 'B_30', 'B_32', 'B_33', 'B_36', 'B_37', 'B_38', 'B_39', 'B_40', 'B_41', 'B_42', 'D_39', 'D_41', 'D_42', 'D_43', 'D_44', 'D_45', 'D_46', 'D_47', 'D_48', 'D_49', 'D_50', 'D_51', 'D_52', 'D_53', 'D_54', 'D_55', 'D_56', 'D_58', 'D_59', 'D_60', 'D_61', 'D_62', 'D_63', 'D_64', 'D_65', 'D_69', 'D_70', 'D_71', 'D_72', 'D_73', 'D_75', 'D_76', 'D_77', 'D_78', 'D_79', 'D_80', 'D_81', 'D_82', 'D_83', 'D_86', 'D_91', 'D_96', 'D_105', 'D_106', 'D_112', 'D_114', 'D_119', 'D_120', 'D_121', 'D_122', 'D_124', 'D_125', 'D_126', 'D_127', 'D_130', 'D_131', 'D_132', 'D_133', 'D_134', 'D_138', 'D_140', 'D_141', 'D_142', 'D_145', 'P_2', 'P_3', 'P_4', 'R_1', 'R_2', 'R_3', 'R_4', 'R_5', 'R_6', 'R_7', 'R_8', 'R_9', 'R_10', 'R_11', 'R_12', 'R_13', 'R_14', 'R_15', 'R_19', 'R_20', 'R_26', 'R_27', 'S_3', 'S_5', 'S_6', 'S_7', 'S_8', 'S_9', 'S_11', 'S_12', 'S_13', 'S_16', 'S_19', 'S_20', 'S_22', 'S_23', 'S_24', 'S_25', 'S_26', 'S_27']\nfor i in ['test', 'train'] if INFERENCE else ['train']:\n    df = pd.read_parquet(f'../input/amex-data-integer-dtypes-parquet-format/{i}.parquet')\n    cid = pd.Categorical(df.pop('customer_ID'), ordered=True)\n    last = (cid != np.roll(cid, -1)) # mask for last statement of every customer\n    if 'target' in df.columns:\n        df.drop(columns=['target'], inplace=True)\n    gc.collect()\n    print('Read', i)\n    df_avg = (df\n              .groupby(cid)\n              .mean()[features_avg]\n              .rename(columns={f: f\"{f}_avg\" for f in features_avg})\n             )\n    gc.collect()\n    print('Computed avg', i)\n    df_min = (df\n              .groupby(cid)\n              .min()[features_min]\n              .rename(columns={f: f\"{f}_min\" for f in features_min})\n             )\n    gc.collect()\n    print('Computed min', i)\n    df_max = (df\n              .groupby(cid)\n              .max()[features_max]\n              .rename(columns={f: f\"{f}_max\" for f in features_max})\n             )\n    gc.collect()\n    print('Computed max', i)\n    df = (df.loc[last, features_last]\n          .rename(columns={f: f\"{f}_last\" for f in features_last})\n          .set_index(np.asarray(cid[last]))\n         )\n    gc.collect()\n    print('Computed last', i)\n    df = pd.concat([df, df_min, df_max, df_avg], axis=1)\n    if i == 'train': train = df\n    else: test = df\n    print(f\"{i} shape: {df.shape}\")\n    del df, df_avg, df_min, df_max, cid, last\n\ntarget = pd.read_csv('../input/amex-default-prediction/train_labels.csv').target.values\nprint(f\"target shape: {target.shape}\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cross-validation\nCross-validate with five-fold StratifiedKFold because the classes are imbalanced. <br>\nLightGBM logs the validation score when every hundred iteration.","metadata":{}},{"cell_type":"code","source":"%%time\n# Cross-validation of the classifier\n\nONLY_FIRST_FOLD = False\n\nfeatures = [f for f in train.columns if f != 'customer_ID' and f != 'target']\n\ndef my_booster(random_state=1, n_estimators=1200):\n    return LGBMClassifier(n_estimators=n_estimators,\n                          learning_rate=0.03, reg_lambda=50,\n                          min_child_samples=2400,\n                          num_leaves=95,\n                          colsample_bytree=0.19,\n                          max_bins=511, random_state=random_state)\n      \nprint(f\"{len(features)} features\")\nscore_list = []\ny_pred_list = []\nkf = StratifiedKFold(n_splits=5)\nfor fold, (idx_tr, idx_va) in enumerate(kf.split(train, target)):\n    X_tr, X_va, y_tr, y_va, model = None, None, None, None, None\n    start_time = datetime.datetime.now()\n    X_tr = train.iloc[idx_tr][features]\n    X_va = train.iloc[idx_va][features]\n    y_tr = target[idx_tr]\n    y_va = target[idx_va]\n    \n    model = my_booster()\n    with warnings.catch_warnings():\n        warnings.filterwarnings('ignore', category=UserWarning)\n        model.fit(X_tr, y_tr,\n                  eval_set = [(X_va, y_va)], \n                  eval_metric=[lgb_amex_metric],\n                  callbacks=[log_evaluation(100)])\n    X_tr, y_tr = None, None\n    y_va_pred = model.predict_proba(X_va, raw_score=True)\n    score = amex_metric(y_va, y_va_pred)\n    n_trees = model.best_iteration_\n    if n_trees is None: n_trees = model.n_estimators\n    print(f\"{Fore.GREEN}{Style.BRIGHT}Fold {fold} | {str(datetime.datetime.now() - start_time)[-12:-7]} |\"\n          f\" {n_trees:5} trees |\"\n          f\"                Score = {score:.5f}{Style.RESET_ALL}\")\n    score_list.append(score)\n    \n    if INFERENCE:\n        y_pred_list.append(model.predict_proba(test[features], raw_score=True))\n        \n    if ONLY_FIRST_FOLD: break # we only want the first fold\n    \nprint(f\"{Fore.GREEN}{Style.BRIGHT}OOF Score: {np.mean(score_list):.5f}{Style.RESET_ALL}\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prediction histogram","metadata":{}},{"cell_type":"code","source":"def sigmoid(log_odds):\n    return 1 / (1 + np.exp(-log_odds))\n\nplt.figure(figsize=(10, 4))\nplt.hist(sigmoid(y_va_pred[y_va == 0]), bins=np.linspace(0, 1, 101),\n         alpha=0.5, density=True, label='0')\nplt.hist(sigmoid(y_va_pred[y_va == 1]), bins=np.linspace(0, 1, 101),\n         alpha=0.5, density=True, label='1')\nplt.xlabel('y_pred')\nplt.ylabel('density')\nplt.title('OOF Prediction histogram', color='k')\nplt.legend()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 4))\nCalibrationDisplay.from_predictions(y_va, sigmoid(y_va_pred), n_bins=50,\n                                    strategy='quantile', ax=plt.gca())\nplt.title('Probability calibration')\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission\nmean of the five predictions","metadata":{}},{"cell_type":"code","source":"if INFERENCE:\n    sub = pd.DataFrame({'customer_ID': test.index,\n                        'prediction': np.mean(y_pred_list, axis=0)})\n    sub.to_csv('submission.csv', index=False)\n    display(sub)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check\nplt.figure(figsize=(12, 4))\nplt.hist(sigmoid(sub.prediction), bins=np.linspace(0, 1, 101), density=True)\nplt.hist(sigmoid(y_va_pred), bins=np.linspace(0, 1, 101), rwidth=0.5, color='orange', density=True)\nplt.show()","metadata":{},"execution_count":null,"outputs":[]}]}