{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"CV = 0.7916799436731615\n\nhttps://www.kaggle.com/code/stgkrtua/amex-howtopreprocess-onkaggle\n\nhttps://www.kaggle.com/code/stgkrtua/amex-train-lgbm-baseline-oof-onkaggle\n\nhttps://www.kaggle.com/code/stgkrtua/amex-howtopreprocess-testdata-onkaggle","metadata":{"execution":{"iopub.status.busy":"2022-08-12T06:50:09.991542Z","iopub.execute_input":"2022-08-12T06:50:09.992252Z","iopub.status.idle":"2022-08-12T06:50:12.447529Z","shell.execute_reply.started":"2022-08-12T06:50:09.992068Z","shell.execute_reply":"2022-08-12T06:50:12.446420Z"}}},{"cell_type":"code","source":"import os\nimport pickle\nimport random\nimport joblib\nimport gc\nimport itertools\nfrom itertools import combinations\n\nimport scipy as sp\nimport numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\n\nimport lightgbm as lgb\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import roc_auc_score, roc_curve, auc\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\n# from hyperopt import STATUS_OK, Trials, fmin, hp, tpe\n\npd.set_option('display.width', 1000)\npd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\nimport warnings; warnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-19T12:21:05.267887Z","iopub.execute_input":"2022-08-19T12:21:05.268342Z","iopub.status.idle":"2022-08-19T12:21:06.934467Z","shell.execute_reply.started":"2022-08-19T12:21:05.268246Z","shell.execute_reply":"2022-08-19T12:21:06.933233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEBUG = False","metadata":{"execution":{"iopub.status.busy":"2022-08-19T12:21:06.937433Z","iopub.execute_input":"2022-08-19T12:21:06.937951Z","iopub.status.idle":"2022-08-19T12:21:06.943926Z","shell.execute_reply.started":"2022-08-19T12:21:06.937903Z","shell.execute_reply":"2022-08-19T12:21:06.942594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    seed = 42\n    n_folds = 5\n    target = 'target'\n    boosting_type = 'gbdt'\n    metric = 'binary'\n    oof_columns = [\"customer_ID\", \"target\"]\n    MODELDIR = \"../input/amex-train-lgbm-baseline-oof-onkaggle\"\n    enc = '../input/amex-train-lgbm-baseline-oof-onkaggle/label_encoder.pickle'","metadata":{"execution":{"iopub.status.busy":"2022-08-19T12:21:06.945951Z","iopub.execute_input":"2022-08-19T12:21:06.946407Z","iopub.status.idle":"2022-08-19T12:21:06.954830Z","shell.execute_reply.started":"2022-08-19T12:21:06.946363Z","shell.execute_reply":"2022-08-19T12:21:06.953816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n\nseed_everything(CFG.seed)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T12:21:06.956595Z","iopub.execute_input":"2022-08-19T12:21:06.957076Z","iopub.status.idle":"2022-08-19T12:21:06.965810Z","shell.execute_reply.started":"2022-08-19T12:21:06.957032Z","shell.execute_reply":"2022-08-19T12:21:06.964627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocess","metadata":{}},{"cell_type":"code","source":"def preprocess_to_testLGBM(test_):\n    cat_features = [\n        \"B_30\",\n        \"B_38\",\n        \"D_114\",\n        \"D_116\",\n        \"D_117\",\n        \"D_120\",\n        \"D_126\",\n        \"D_63\",\n        \"D_64\",\n        \"D_66\",\n        \"D_68\"\n    ]\n    cat_features = [f\"{cf}_last\" for cf in cat_features]\n    \n    for cat_col in cat_features:\n        encoder = LabelEncoder()\n        test_[cat_col] = encoder.fit_transform(test_[cat_col])\n    \n    num_cols = list(test_.dtypes[(test_.dtypes == 'float32') | (test_.dtypes == 'float64')].index)\n    num_cols = [col for col in num_cols if 'last' in col]\n    \n    for col in num_cols:\n        test_[col + '_round2'] = test_[col].round(2)\n    num_cols = [col for col in test_.columns if 'last' in col]\n    num_cols = [col[:-5] for col in num_cols if 'round' not in col]\n    \n    for col in num_cols:\n        try:\n            test_[f'{col}_last_mean_diff'] = test_[f'{col}_last'] - test_[f'{col}_mean']\n        except: pass\n    \n    num_cols = list(test_.dtypes[(test_.dtypes == 'float32') | (test_.dtypes == 'float64')].index)\n    for col in tqdm(num_cols):\n        test_[col] = test_[col].astype(np.float16)\n    \n    features = [col for col in test_.columns if col not in ['customer_ID', CFG.target]]\n    \n    return cat_features, features","metadata":{"execution":{"iopub.status.busy":"2022-08-19T12:21:06.968839Z","iopub.execute_input":"2022-08-19T12:21:06.969185Z","iopub.status.idle":"2022-08-19T12:21:06.982691Z","shell.execute_reply.started":"2022-08-19T12:21:06.969152Z","shell.execute_reply":"2022-08-19T12:21:06.981577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Main","metadata":{}},{"cell_type":"code","source":"sub_df = pd.DataFrame()\n\nfor fold in range(CFG.n_folds):\n    print(\"\\nFold {}\".format(fold+1))\n    # preparing input data\n    test = pd.read_feather(f\"../input/amex-howtopreprocess-testdata-onkaggle/test_fe_{fold}.ftr\")\n    if DEBUG:\n        test = test.sample(1000).reset_index(drop=True)\n    pred_df = test[\"customer_ID\"]\n\n    cat_features, features = preprocess_to_testLGBM(test)\n    test_features = features\n\n    enc = pickle.load(open(CFG.enc, 'rb'))\n    for col in cat_features[:-1]:\n        test[col] = enc.fit_transform(test[col])\n    X = test[test_features]\n    for fold in range(CFG.n_folds):\n        gbm = pickle.load(open(f\"{CFG.MODELDIR}/trained_model_{fold}.pkl\", 'rb'))\n        gbm_prob = gbm.predict_proba(X)[:,1]\n\n        pred_df = pd.concat([pred_df, pd.DataFrame(gbm_prob, columns=[f\"pred_{fold}\"])], axis=1)\n    sub_df = pd.concat([sub_df, pred_df], axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T12:21:06.984416Z","iopub.execute_input":"2022-08-19T12:21:06.984742Z","iopub.status.idle":"2022-08-19T12:21:51.961876Z","shell.execute_reply.started":"2022-08-19T12:21:06.984714Z","shell.execute_reply":"2022-08-19T12:21:51.960739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = sub_df.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T12:23:39.598098Z","iopub.execute_input":"2022-08-19T12:23:39.598525Z","iopub.status.idle":"2022-08-19T12:23:39.605254Z","shell.execute_reply.started":"2022-08-19T12:23:39.598490Z","shell.execute_reply":"2022-08-19T12:23:39.603852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")\ndisplay(sub)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T12:21:51.991608Z","iopub.execute_input":"2022-08-19T12:21:51.992337Z","iopub.status.idle":"2022-08-19T12:21:53.769028Z","shell.execute_reply.started":"2022-08-19T12:21:51.992296Z","shell.execute_reply":"2022-08-19T12:21:53.767854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_list = list([f\"pred_{fold}\" for fold in range(CFG.n_folds)])\nsub_df[\"prediction\"] = sub_df[pred_list].mean(axis=1)\nsub = pd.merge(sub[\"customer_ID\"], sub_df[[\"customer_ID\", \"prediction\"]], on=\"customer_ID\")","metadata":{"execution":{"iopub.status.busy":"2022-08-19T12:25:27.573713Z","iopub.execute_input":"2022-08-19T12:25:27.574819Z","iopub.status.idle":"2022-08-19T12:25:27.596933Z","shell.execute_reply.started":"2022-08-19T12:25:27.574745Z","shell.execute_reply":"2022-08-19T12:25:27.595914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub","metadata":{"execution":{"iopub.status.busy":"2022-08-19T12:25:29.012848Z","iopub.execute_input":"2022-08-19T12:25:29.013303Z","iopub.status.idle":"2022-08-19T12:25:29.028168Z","shell.execute_reply.started":"2022-08-19T12:25:29.013265Z","shell.execute_reply":"2022-08-19T12:25:29.026916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(sub), len(sub_df))","metadata":{"execution":{"iopub.status.busy":"2022-08-19T12:25:32.454322Z","iopub.execute_input":"2022-08-19T12:25:32.454717Z","iopub.status.idle":"2022-08-19T12:25:32.460688Z","shell.execute_reply.started":"2022-08-19T12:25:32.454675Z","shell.execute_reply":"2022-08-19T12:25:32.459389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-19T12:21:53.918071Z","iopub.status.idle":"2022-08-19T12:21:53.918459Z","shell.execute_reply.started":"2022-08-19T12:21:53.918276Z","shell.execute_reply":"2022-08-19T12:21:53.918294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}