{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# American Express Default Loan Predictor","metadata":{"papermill":{"duration":0.008552,"end_time":"2022-06-24T11:56:48.692388","exception":false,"start_time":"2022-06-24T11:56:48.683836","status":"completed"},"tags":[]}},{"cell_type":"code","source":"#!pip install polars","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:09:37.309284Z","iopub.execute_input":"2022-08-12T12:09:37.310688Z","iopub.status.idle":"2022-08-12T12:09:37.339456Z","shell.execute_reply.started":"2022-08-12T12:09:37.310541Z","shell.execute_reply":"2022-08-12T12:09:37.338094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \n#import polars as pl\nimport scipy as sp\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom matplotlib import pyplot as plt\nfrom sklearn.model_selection import train_test_split, StratifiedShuffleSplit, StratifiedKFold\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom itertools import combinations\nfrom sklearn.cluster import KMeans\nfrom sklearn.preprocessing import  OrdinalEncoder\nimport warnings\nwarnings.filterwarnings('ignore')\n%config InLineBackend.figure_format = 'png'\nimport os\nfrom tqdm import tqdm\nimport lightgbm as lgb\nfrom itertools import combinations\nimport gc\n#os.mkdir('./')","metadata":{"papermill":{"duration":3.132189,"end_time":"2022-06-24T11:56:51.838484","exception":false,"start_time":"2022-06-24T11:56:48.706295","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-12T12:09:37.341578Z","iopub.execute_input":"2022-08-12T12:09:37.342199Z","iopub.status.idle":"2022-08-12T12:09:39.897502Z","shell.execute_reply.started":"2022-08-12T12:09:37.342150Z","shell.execute_reply":"2022-08-12T12:09:39.896355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class config:\n    seed = 420\n    n_folds = 200\n    target = 'target'\n    input_dir = '../input/amexfeather'","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:09:39.899413Z","iopub.execute_input":"2022-08-12T12:09:39.899792Z","iopub.status.idle":"2022-08-12T12:09:39.904090Z","shell.execute_reply.started":"2022-08-12T12:09:39.899764Z","shell.execute_reply":"2022-08-12T12:09:39.903053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preprocessing","metadata":{"papermill":{"duration":0.007378,"end_time":"2022-06-24T11:56:51.875552","exception":false,"start_time":"2022-06-24T11:56:51.868174","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# %%time\n\n# def get_difference(data, num_features):\n#     df1 = []\n#     customer_ids = []\n#     for customer_id, df in tqdm(data.groupby(['customer_ID'])):\n#         diff_df1 = df[num_features].diff(1).iloc[[-1]].values.astype(np.float32)\n#         df1.append(diff_df1)\n#         customer_ids.append(customer_id)\n#     df1 = np.concatenate(df1, axis = 0)\n#     df1 = pd.DataFrame(df1, columns = [col + '_diff1' for col in df[num_features].columns])\n#     df1['customer_ID'] = customer_ids\n#     return df1\n\n# def read_train_data():\n#     train = pd.read_feather('../input/amexfeather/train_data.ftr')\n#     features = train.drop(['customer_ID', 'S_2'], axis = 1).columns.to_list()\n#     cat_features = [\n#         \"B_30\",\n#         \"B_38\",\n#         \"D_114\",\n#         \"D_116\",\n#         \"D_117\",\n#         \"D_120\",\n#         \"D_126\",\n#         \"D_63\",\n#         \"D_64\",\n#         \"D_66\",\n#         \"D_68\",\n#     ]\n    \n#     num_features = [col for col in features if col not in cat_features]\n    \n#     print('Starting training feature engineer...')\n    \n#     train_num_agg = train.groupby(\"customer_ID\")[num_features].agg(['first', 'mean', 'std', 'min', 'max', 'last'])\n#     train_num_agg.columns = ['_'.join(x) for x in train_num_agg.columns]\n#     train_num_agg.reset_index(inplace = True)\n    \n#     print('Creating lag features for train set')\n    \n#     # Lag Features\n#     for col in train_num_agg:\n#         for col_2 in ['first', 'mean', 'std', 'min', 'max']:\n#             if 'last' in col and col.replace('last', col_2) in train_num_agg:\n#                 train_num_agg[col + '_lag_sub'] = train_num_agg[col] - train_num_agg[col.replace('last', col_2)]\n#                 train_num_agg[col + '_lag_div'] = train_num_agg[col] / train_num_agg[col.replace('last', col_2)]\n\n#     train_cat_agg = train.groupby(\"customer_ID\")[cat_features].agg(['count', 'first', 'last', 'nunique'])\n#     train_cat_agg.columns = ['_'.join(x) for x in train_cat_agg.columns]\n#     train_cat_agg.reset_index(inplace = True)\n#     train_labels = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\n    \n#     # Transform float64 columns to float32\n#     cols = list(train_num_agg.dtypes[train_num_agg.dtypes == 'float64'].index)\n#     for col in tqdm(cols):\n#         train_num_agg[col] = train_num_agg[col].astype(np.float32)\n#     # Transform int64 columns to int32\n#     cols = list(train_cat_agg.dtypes[train_cat_agg.dtypes == 'int64'].index)\n#     for col in tqdm(cols):\n#         train_cat_agg[col] = train_cat_agg[col].astype(np.int32)    \n#     # Get the difference\n#     train_diff = get_difference(train, num_features)\n    \n    \n#     train = train_num_agg.merge(train_cat_agg, how = 'inner', on = 'customer_ID').merge(train_diff, how = 'inner', on = 'customer_ID').merge(train_labels, how = 'inner', on = 'customer_ID')\n    \n#     del train_num_agg, train_cat_agg, train_diff\n#     gc.collect()\n#     #train.to_parquet('../input/feature_eng_data/train_loaded.parquet')\n#     print('Training data done')\n#     return train\n\n# def read_test_data():     \n#     test = pd.read_feather('../input/amexfeather/test_data.ftr')\n#     features = test.drop(['customer_ID', 'S_2'], axis = 1).columns.to_list()\n#     cat_features = [\n#         \"B_30\",\n#         \"B_38\",\n#         \"D_114\",\n#         \"D_116\",\n#         \"D_117\",\n#         \"D_120\",\n#         \"D_126\",\n#         \"D_63\",\n#         \"D_64\",\n#         \"D_66\",\n#         \"D_68\",\n#     ]\n    \n#     num_features = [col for col in features if col not in cat_features]\n#     num = test.shape[0]\n#     n1 = num//3\n#     n2 = 2*num//3\n    \n#     print('First Slice')\n#     test = test.iloc[0:n1,:]\n#     gc.collect()\n#     print('Starting test feature engineer...')\n#     test_num_agg = test.groupby(\"customer_ID\")[num_features].agg(['first', 'mean', 'std', 'min', 'max', 'last'])\n#     gc.collect()\n#     test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n#     test_num_agg.reset_index(inplace = True)\n    \n#     print('Creating lag features for test set')\n    \n#     # Lag Features\n#     for col in test_num_agg:\n#         for col_2 in ['first', 'mean', 'std', 'min', 'max']:\n#             if 'last' in col and col.replace('last', col_2) in test_num_agg:\n#                 test_num_agg[col + '_lag_sub'] = test_num_agg[col] - test_num_agg[col.replace('last', col_2)]\n#                 test_num_agg[col + '_lag_div'] = test_num_agg[col] / test_num_agg[col.replace('last', col_2)]\n    \n#     test_cat_agg = test.groupby(\"customer_ID\")[cat_features].agg(['count', 'first', 'last', 'nunique'])\n#     test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n#     test_cat_agg.reset_index(inplace = True)\n    \n#     # Transform float64 columns to float32\n#     cols = list(test_num_agg.dtypes[test_num_agg.dtypes == 'float64'].index)\n#     for col in tqdm(cols):\n#         test_num_agg[col] = test_num_agg[col].astype(np.float32)\n#     # Transform int64 columns to int32\n#     cols = list(test_cat_agg.dtypes[test_cat_agg.dtypes == 'int64'].index)\n#     for col in tqdm(cols):\n#         test_cat_agg[col] = test_cat_agg[col].astype(np.int32)\n#     # Get the difference\n#     test_diff = get_difference(test, num_features)\n#     gc.collect()\n#     test = test_num_agg.merge(test_cat_agg, how = 'inner', on = 'customer_ID').merge(test_diff, how = 'inner', on = 'customer_ID')\n#     del test_num_agg, test_cat_agg, test_diff\n#     gc.collect()\n    \n#     print('Test Slice 1 done')\n#     test.to_feather('test_data_preprocessed_slice_1.ftr')\n#     del test\n#     gc.collect()\n    \n#     test = pd.read_feather('../input/amexfeather/test_data.ftr')\n#     print('Second Slice')\n#     test = test.iloc[n1+1:n2,:]\n#     gc.collect()\n#     print('Starting test feature engineer...')\n#     test_num_agg = test.groupby(\"customer_ID\")[num_features].agg(['first', 'mean', 'std', 'min', 'max', 'last'])\n#     gc.collect()\n#     test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n#     test_num_agg.reset_index(inplace = True)\n    \n#     print('Creating lag features for test set')\n    \n#     # Lag Features\n#     for col in test_num_agg:\n#         for col_2 in ['first', 'mean', 'std', 'min', 'max']:\n#             if 'last' in col and col.replace('last', col_2) in test_num_agg:\n#                 test_num_agg[col + '_lag_sub'] = test_num_agg[col] - test_num_agg[col.replace('last', col_2)]\n#                 test_num_agg[col + '_lag_div'] = test_num_agg[col] / test_num_agg[col.replace('last', col_2)]\n    \n#     test_cat_agg = test.groupby(\"customer_ID\")[cat_features].agg(['count', 'first', 'last', 'nunique'])\n#     test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n#     test_cat_agg.reset_index(inplace = True)\n    \n#     # Transform float64 columns to float32\n#     cols = list(test_num_agg.dtypes[test_num_agg.dtypes == 'float64'].index)\n#     for col in tqdm(cols):\n#         test_num_agg[col] = test_num_agg[col].astype(np.float32)\n#     # Transform int64 columns to int32\n#     cols = list(test_cat_agg.dtypes[test_cat_agg.dtypes == 'int64'].index)\n#     for col in tqdm(cols):\n#         test_cat_agg[col] = test_cat_agg[col].astype(np.int32)\n#     # Get the difference\n#     test_diff = get_difference(test, num_features)\n#     gc.collect()\n#     test = test_num_agg.merge(test_cat_agg, how = 'inner', on = 'customer_ID').merge(test_diff, how = 'inner', on = 'customer_ID')\n#     del test_num_agg, test_cat_agg, test_diff\n#     gc.collect()\n    \n#     print('Test Slice 2 done')\n#     test.to_feather('test_data_preprocessed_slice_2.ftr')\n#     del test\n#     gc.collect()\n    \n#     test = pd.read_feather('../input/amexfeather/test_data.ftr')\n#     print('Third Slice')\n#     test = test.iloc[n2+1:,:]\n#     gc.collect()\n#     print('Starting test feature engineer...')\n#     test_num_agg = test.groupby(\"customer_ID\")[num_features].agg(['first', 'mean', 'std', 'min', 'max', 'last'])\n#     gc.collect()\n#     test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n#     test_num_agg.reset_index(inplace = True)\n    \n#     print('Creating lag features for test set')\n    \n#     # Lag Features\n#     for col in test_num_agg:\n#         for col_2 in ['first', 'mean', 'std', 'min', 'max']:\n#             if 'last' in col and col.replace('last', col_2) in test_num_agg:\n#                 test_num_agg[col + '_lag_sub'] = test_num_agg[col] - test_num_agg[col.replace('last', col_2)]\n#                 test_num_agg[col + '_lag_div'] = test_num_agg[col] / test_num_agg[col.replace('last', col_2)]\n    \n#     test_cat_agg = test.groupby(\"customer_ID\")[cat_features].agg(['count', 'first', 'last', 'nunique'])\n#     test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n#     test_cat_agg.reset_index(inplace = True)\n    \n#     # Transform float64 columns to float32\n#     cols = list(test_num_agg.dtypes[test_num_agg.dtypes == 'float64'].index)\n#     for col in tqdm(cols):\n#         test_num_agg[col] = test_num_agg[col].astype(np.float32)\n#     # Transform int64 columns to int32\n#     cols = list(test_cat_agg.dtypes[test_cat_agg.dtypes == 'int64'].index)\n#     for col in tqdm(cols):\n#         test_cat_agg[col] = test_cat_agg[col].astype(np.int32)\n#     # Get the difference\n#     test_diff = get_difference(test, num_features)\n#     gc.collect()\n#     test = test_num_agg.merge(test_cat_agg, how = 'inner', on = 'customer_ID').merge(test_diff, how = 'inner', on = 'customer_ID')\n#     del test_num_agg, test_cat_agg, test_diff\n#     gc.collect()\n    \n#     print('Test Slice 3 done')\n#     test.to_feather('test_data_preprocessed_slice_3.ftr')\n#     del test\n#     gc.collect()\n    \n    \n\n","metadata":{"papermill":{"duration":21.069342,"end_time":"2022-06-24T11:57:12.952425","exception":false,"start_time":"2022-06-24T11:56:51.883083","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-12T12:09:39.905363Z","iopub.execute_input":"2022-08-12T12:09:39.905647Z","iopub.status.idle":"2022-08-12T12:09:39.920233Z","shell.execute_reply.started":"2022-08-12T12:09:39.905624Z","shell.execute_reply":"2022-08-12T12:09:39.918984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# train = read_train_data()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:09:39.922440Z","iopub.execute_input":"2022-08-12T12:09:39.922750Z","iopub.status.idle":"2022-08-12T12:09:39.936680Z","shell.execute_reply.started":"2022-08-12T12:09:39.922724Z","shell.execute_reply":"2022-08-12T12:09:39.935420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train.to_feather('train_data_preprocessed.ftr')","metadata":{"papermill":{"duration":2.575275,"end_time":"2022-06-24T11:57:15.588644","exception":false,"start_time":"2022-06-24T11:57:13.013369","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-12T12:09:39.939635Z","iopub.execute_input":"2022-08-12T12:09:39.940076Z","iopub.status.idle":"2022-08-12T12:09:39.950078Z","shell.execute_reply.started":"2022-08-12T12:09:39.940040Z","shell.execute_reply":"2022-08-12T12:09:39.948906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# read_test_data()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:09:39.951936Z","iopub.execute_input":"2022-08-12T12:09:39.952325Z","iopub.status.idle":"2022-08-12T12:09:39.962398Z","shell.execute_reply.started":"2022-08-12T12:09:39.952298Z","shell.execute_reply":"2022-08-12T12:09:39.961144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## import the datasets \n# run model \n# make preds \n# get a list of imp features \n# LGBM \n# Try ensemble LGBM + AdaBoost + Catboost + XGBoost - KNN - Naive Bayes - Logistic Regression \n# Try NN ","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:09:39.963328Z","iopub.execute_input":"2022-08-12T12:09:39.963660Z","iopub.status.idle":"2022-08-12T12:09:39.973908Z","shell.execute_reply.started":"2022-08-12T12:09:39.963634Z","shell.execute_reply":"2022-08-12T12:09:39.972773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain = pd.read_feather('../input/amex-preprocesses-feather/train_data_preprocessed.ftr')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:09:39.974762Z","iopub.execute_input":"2022-08-12T12:09:39.975016Z","iopub.status.idle":"2022-08-12T12:10:05.303360Z","shell.execute_reply.started":"2022-08-12T12:09:39.974991Z","shell.execute_reply":"2022-08-12T12:10:05.302360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntest1 = pd.read_feather('../input/amex-preprocesses-feather/test_data_preprocessed_slice_1.ftr') \ntest2 = pd.read_feather('../input/amex-preprocesses-feather/test_data_preprocessed_slice_2.ftr')\ntest3 = pd.read_feather('../input/amex-preprocesses-feather/test_data_preprocessed_slice_3.ftr')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:10:05.304914Z","iopub.execute_input":"2022-08-12T12:10:05.305300Z","iopub.status.idle":"2022-08-12T12:10:36.121309Z","shell.execute_reply.started":"2022-08-12T12:10:05.305265Z","shell.execute_reply":"2022-08-12T12:10:36.120188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.concat([test1, test2, test3])\n\ndel test1, test2, test3\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:10:36.124757Z","iopub.execute_input":"2022-08-12T12:10:36.125081Z","iopub.status.idle":"2022-08-12T12:10:51.691507Z","shell.execute_reply.started":"2022-08-12T12:10:36.125055Z","shell.execute_reply":"2022-08-12T12:10:51.690414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_features = set(train.columns.to_list()) - set(test.columns.to_list()) - {'target'}","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:10:51.693017Z","iopub.execute_input":"2022-08-12T12:10:51.693324Z","iopub.status.idle":"2022-08-12T12:10:51.698941Z","shell.execute_reply.started":"2022-08-12T12:10:51.693299Z","shell.execute_reply":"2022-08-12T12:10:51.697697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop(drop_features, axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:10:51.700335Z","iopub.execute_input":"2022-08-12T12:10:51.700748Z","iopub.status.idle":"2022-08-12T12:10:52.898857Z","shell.execute_reply.started":"2022-08-12T12:10:51.700712Z","shell.execute_reply":"2022-08-12T12:10:52.897271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true, y_pred):\n    labels = np.transpose(np.array([y_true, y_pred]))\n    labels = labels[labels[:, 1].argsort()[::-1]]\n    weights = np.where(labels[:,0]==0, 20, 1)\n    cut_vals = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n    gini = [0,0]\n    for i in [1,0]:\n        labels = np.transpose(np.array([y_true, y_pred]))\n        labels = labels[labels[:, i].argsort()[::-1]]\n        weight = np.where(labels[:,0]==0, 20, 1)\n        weight_random = np.cumsum(weight / np.sum(weight))\n        total_pos = np.sum(labels[:, 0] *  weight)\n        cum_pos_found = np.cumsum(labels[:, 0] * weight)\n        lorentz = cum_pos_found / total_pos\n        gini[i] = np.sum((lorentz - weight_random) * weight)\n    return 0.5 * (gini[1]/gini[0] + top_four)\n\ndef amex_metric_np(preds, target):\n    indices = np.argsort(preds)[::-1]\n    preds, target = preds[indices], target[indices]\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_mask = cum_norm_weight <= 0.04\n    d = np.sum(target[four_pct_mask]) / np.sum(target)\n    weighted_target = target * weight\n    lorentz = (weighted_target / weighted_target.sum()).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n    n_pos = np.sum(target)\n    n_neg = target.shape[0] - n_pos\n    gini_max = 10 * n_neg * (n_pos + 20 * n_neg - 19) / (n_pos + 20 * n_neg)\n    g = gini / gini_max\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:10:52.900498Z","iopub.execute_input":"2022-08-12T12:10:52.900816Z","iopub.status.idle":"2022-08-12T12:10:52.914645Z","shell.execute_reply.started":"2022-08-12T12:10:52.900789Z","shell.execute_reply":"2022-08-12T12:10:52.913435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lgb_amex_metric(y_pred, y_true):\n    y_true = y_true.get_label()\n    return 'amex_metric', amex_metric(y_true, y_pred), True\n\ndef train_and_evaluate(train, test):\n    # Label encode categorical features\n    cat_features = [\n        \"B_30\",\n        \"B_38\",\n        \"D_114\",\n        \"D_116\",\n        \"D_117\",\n        \"D_120\",\n        \"D_126\",\n        \"D_63\",\n        \"D_64\",\n        \"D_66\",\n        \"D_68\"\n    ]\n    cat_features = [f\"{cf}_last\" for cf in cat_features]\n    for cat_col in cat_features:\n        encoder = LabelEncoder()\n        train[cat_col] = encoder.fit_transform(train[cat_col])\n        test[cat_col] = encoder.transform(test[cat_col])\n   \n    # Round last float features to 2 decimal place\n    num_cols = list(train.dtypes[(train.dtypes == 'float32') | (train.dtypes == 'float64')].index)\n    num_cols = [col for col in num_cols if 'last' in col]\n    \n    for col in num_cols:\n        train[col + '_round2'] = train[col].round(2)\n        test[col + '_round2'] = test[col].round(2)\n    \n    # Get the difference between last and mean\n    num_cols = [col for col in train.columns if 'last' in col]\n    num_cols = [col[:-5] for col in num_cols if 'round' not in col]\n    \n    for col in num_cols:\n        try:\n            train[f'{col}_last_mean_diff'] = train[f'{col}_last'] - train[f'{col}_mean']\n            test[f'{col}_last_mean_diff'] = test[f'{col}_last'] - test[f'{col}_mean']\n        except:\n            pass\n    \n    # Transform float64 and float32 to float16\n    num_cols = list(train.dtypes[(train.dtypes == 'float32') | (train.dtypes == 'float64')].index)\n    for col in tqdm(num_cols):\n        train[col] = train[col].astype(np.float16)\n        test[col] = test[col].astype(np.float16)\n    \n    # Get feature list\n    features = [col for col in train.columns if col not in ['customer_ID', config.target]]\n    params = {\n        'objective': 'binary',\n        'metric': \"binary_logloss\",\n        'boosting': 'dart',\n        'seed': config.seed,\n        'num_leaves': 100,\n        'learning_rate': 0.01,\n        'feature_fraction': 0.20,\n        'bagging_freq': 10,\n        'bagging_fraction': 0.50,\n        #'n_jobs': -1,\n        'lambda_l2': 2,\n        'min_data_in_leaf': 40\n        }\n    \n    # Create a numpy array to store test predictions\n    test_predictions = np.zeros(len(test))\n    \n    # Create a numpy array to store out of folds predictions\n    oof_predictions = np.zeros(len(train))\n    kfold = StratifiedKFold(n_splits = config.n_folds, shuffle = True, random_state = config.seed)\n    \n    for fold, (trn_ind, val_ind) in enumerate(kfold.split(train, train[config.target])):\n        print(' ')\n        print('-'*50)\n        print(f'Training fold {fold} with {len(features)} features...')\n        x_train, x_val = train[features].iloc[trn_ind], train[features].iloc[val_ind]\n        y_train, y_val = train[config.target].iloc[trn_ind], train[config.target].iloc[val_ind]\n        lgb_train = lgb.Dataset(x_train, y_train, categorical_feature = cat_features)\n        lgb_valid = lgb.Dataset(x_val, y_val, categorical_feature = cat_features)\n        model = lgb.train(\n            params = params,\n            train_set = lgb_train,\n            num_boost_round = 10500,\n            valid_sets = [lgb_train, lgb_valid],\n            early_stopping_rounds = 100,\n            verbose_eval = 500,\n            feval = lgb_amex_metric\n            )\n        # Save best model\n        joblib.dump(model, f'lgbm_fold{fold}_seed{config.seed}.pkl')\n\n        # Predict validation\n        val_pred = model.predict(x_val)\n        # Add to out of folds array\n        oof_predictions[val_ind] = val_pred\n        # Predict the test set\n        test_pred = model.predict(test[features])\n        test_predictions += test_pred / config.n_folds\n        # Compute fold metric\n        score = amex_metric(y_val, val_pred)\n        print(f'Our fold {fold} CV score is {score}')\n        del x_train, x_val, y_train, y_val, lgb_train, lgb_valid\n        gc.collect()\n\n    # Compute out of folds metric\n    score = amex_metric(train[config.target], oof_predictions)\n    print(f'Our out of folds CV score is {score}')\n    # Create a dataframe to store out of folds predictions\n    oof_df = pd.DataFrame({'customer_ID': train['customer_ID'], 'target': train[config.target], 'prediction': oof_predictions})\n    oof_df.to_csv(f'oof_lgbm_baseline_{config.n_folds}fold_seed{config.seed}.csv', index = False)\n\n    # Create a dataframe to store test prediction\n    test_df = pd.DataFrame({'customer_ID': test['customer_ID'], 'prediction': test_predictions})\n    test_df.to_csv(f'test_lgbm_baseline_{config.n_folds}fold_seed{config.seed}.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:10:52.915820Z","iopub.execute_input":"2022-08-12T12:10:52.916122Z","iopub.status.idle":"2022-08-12T12:10:52.938409Z","shell.execute_reply.started":"2022-08-12T12:10:52.916090Z","shell.execute_reply":"2022-08-12T12:10:52.937661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_and_evaluate(train, test)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T12:10:52.940381Z","iopub.execute_input":"2022-08-12T12:10:52.940829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train, X_test, y_train, y_test = train_test_split(x, y, test_size=0.3, stratify=y)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\n\ndf_1 = pd.read_csv('../input/amex-best-of-both-v2/test_lgbm_v3_loaded_5fold_seed42.csv')\ndf_1.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}