{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-29T20:51:21.827451Z","iopub.execute_input":"2022-08-29T20:51:21.827904Z","iopub.status.idle":"2022-08-29T20:51:21.841667Z","shell.execute_reply.started":"2022-08-29T20:51:21.827842Z","shell.execute_reply":"2022-08-29T20:51:21.840289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(check.isna().sum()/check.shape[0]).nlargest(20).plot.bar(ti)","metadata":{"execution":{"iopub.status.busy":"2022-08-28T21:24:18.554673Z","iopub.execute_input":"2022-08-28T21:24:18.555132Z","iopub.status.idle":"2022-08-28T21:24:21.158737Z","shell.execute_reply.started":"2022-08-28T21:24:18.555096Z","shell.execute_reply":"2022-08-28T21:24:21.157714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplot_features = [feature for feature in check.columns if feature not in cat_features + ['customer_ID']]\nncols = 4\nnrows = len(plot_features)//ncols + 1\nplt.figure(figsize = (20, 130))\nfor i, feature in enumerate(plot_features):\n    \n    \n        \n    plt.subplot(nrows, ncols, i+1)\n    plt.hist(check[feature], bins = 200)\n    plt.title(feature)\n    plt.subplots_adjust(hspace = 0.4)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/train.parquet')\ncheck.fillna(0, inplace = True)\ncat_features = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n\ndelinquency = [column for column in check.columns if column.startswith('D_')]\nrisk = [column for column in check.columns if column.startswith('R_')]\npayment = [column for column in check.columns if column.startswith('P_')]\nbalance = [column for column in check.columns if column.startswith('B_')]\nspend = [column for column in check.columns if column.startswith('S_')]\n\ndef get_n2_last(series):\n    if len(series) > 1:\n        return series.iloc[-2] \n    else:\n        return series.iloc[-1]\n    return \ndef get_n3_last(series):\n    if len(series) > 2:\n        return series.iloc[-3]\n    else:\n        return get_n2_last(series)\n \n \ndef get_train_data(df, cat_f = None, num_f = None):\n    print('starting')\n    if cat_f == None:\n        num_f = [column for column in check.columns if column != 'customer_ID']\n    else:\n        num_f = [column for column in check.columns if column not in cat_features + ['customer_ID']]\n    \n    train_num = df.groupby(['customer_ID'])[num_f].agg(['mean', 'min', 'max', 'last', get_n2_last, get_n3_last])\n    train_num.columns = ['_'.join(x) for x in train_num.columns]\n    print('finished with numeric')\n    train_cat = df.groupby(['customer_ID'])[cat_f].agg(['count', 'last', 'nunique',get_n2_last, get_n3_last])\n    train_cat.columns = ['_'.join(x) for x in train_cat.columns]\n    print('finished with categorical')\n    result = pd.concat([train_cat, train_num], axis = 1)\n    del train_cat\n    del train_num\n    return result.reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-08-29T20:51:25.054848Z","iopub.execute_input":"2022-08-29T20:51:25.056221Z","iopub.status.idle":"2022-08-29T20:51:50.145279Z","shell.execute_reply.started":"2022-08-29T20:51:25.056163Z","shell.execute_reply":"2022-08-29T20:51:50.143662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check = get_train_data(check, cat_f = cat_features)","metadata":{"execution":{"iopub.status.busy":"2022-08-29T20:51:50.148050Z","iopub.execute_input":"2022-08-29T20:51:50.148818Z","iopub.status.idle":"2022-08-29T21:29:52.449839Z","shell.execute_reply.started":"2022-08-29T20:51:50.148769Z","shell.execute_reply":"2022-08-29T21:29:52.447314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\nmerged = pd.merge(check, targets, on= 'customer_ID')\nmerged.to_csv('data_merged.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-29T21:29:52.453707Z","iopub.execute_input":"2022-08-29T21:29:52.455435Z","iopub.status.idle":"2022-08-29T21:37:57.537845Z","shell.execute_reply.started":"2022-08-29T21:29:52.455366Z","shell.execute_reply":"2022-08-29T21:37:57.536411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split, StratifiedKFold, KFold\nfrom sklearn.inspection import permutation_importance\nfrom sklearn.metrics import roc_auc_score\nimport catboost as cb\nimport lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2022-08-29T20:30:33.944528Z","iopub.execute_input":"2022-08-29T20:30:33.944907Z","iopub.status.idle":"2022-08-29T20:30:33.963765Z","shell.execute_reply.started":"2022-08-29T20:30:33.944876Z","shell.execute_reply":"2022-08-29T20:30:33.962625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets","metadata":{"execution":{"iopub.status.busy":"2022-08-29T20:30:42.802954Z","iopub.execute_input":"2022-08-29T20:30:42.803350Z","iopub.status.idle":"2022-08-29T20:30:42.824839Z","shell.execute_reply.started":"2022-08-29T20:30:42.803318Z","shell.execute_reply":"2022-08-29T20:30:42.823920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = pd.read_csv('../input/amex-default-prediction/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-29T20:30:34.646570Z","iopub.execute_input":"2022-08-29T20:30:34.647299Z","iopub.status.idle":"2022-08-29T20:30:35.467217Z","shell.execute_reply.started":"2022-08-29T20:30:34.647260Z","shell.execute_reply":"2022-08-29T20:30:35.466298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged = pd.merge(check, targets, on= 'customer_ID')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-29T20:34:53.472016Z","iopub.execute_input":"2022-08-29T20:34:53.472359Z","iopub.status.idle":"2022-08-29T20:36:44.610648Z","shell.execute_reply.started":"2022-08-29T20:34:53.472331Z","shell.execute_reply":"2022-08-29T20:36:44.609825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged","metadata":{"execution":{"iopub.status.busy":"2022-08-29T20:38:10.162386Z","iopub.execute_input":"2022-08-29T20:38:10.163280Z","iopub.status.idle":"2022-08-29T20:38:10.234225Z","shell.execute_reply.started":"2022-08-29T20:38:10.163183Z","shell.execute_reply":"2022-08-29T20:38:10.232963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = merged.target.values\nfeatures = [x for x in merged.columns if x not in ['customer_ID', 'target']]\ncv = KFold(n_splits=5, random_state=100, shuffle=True)\n\noof = np.zeros(len(merged))\ntrain_preds = np.zeros(len(merged))\n\nmodels = []\n\ntree_params = {\n    'objective': 'binary',\n    'metric': 'auc',\n    'learning_rate': 0.05,\n    'max_depth': 3,\n    'reg_lambda': 1,\n    'num_leaves': 64,\n    'n_jobs': 5,\n    'n_estimators': 1000\n}\n\nfor fold_, (train_idx, val_idx) in enumerate(cv.split(merged, targets), 1):\n    print(f'Training with fold {fold_} started.')\n    lgb_model = lgb.LGBMClassifier(**tree_params)\n    train, val = merged.iloc[train_idx], merged.iloc[val_idx]\n    \n    lgb_model.fit(train[features], train.target.values, eval_set=[(val[features], val.target.values)],\n              early_stopping_rounds=50, verbose=50)\n\n    \n    oof[val_idx] = lgb_model.predict_proba(val[features])[:, 1]\n    train_preds[train_idx] += lgb_model.predict_proba(train[features])[:, 1] / (cv.n_splits-1)\n    models.append(lgb_model)\n    print(f'Training with fold {fold_} completed.')","metadata":{"execution":{"iopub.status.busy":"2022-08-29T20:37:44.722937Z","iopub.execute_input":"2022-08-29T20:37:44.723338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}