{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pickle\nimport random\nimport joblib\nimport gc\nimport itertools\nfrom itertools import combinations\n\nimport scipy as sp\nimport numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\n\nimport lightgbm as lgb\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import roc_auc_score, roc_curve, auc\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\n# from hyperopt import STATUS_OK, Trials, fmin, hp, tpe\n\npd.set_option('display.width', 1000)\npd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\nimport warnings; warnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T06:40:48.209021Z","iopub.execute_input":"2022-07-25T06:40:48.210089Z","iopub.status.idle":"2022-07-25T06:40:48.219909Z","shell.execute_reply.started":"2022-07-25T06:40:48.210034Z","shell.execute_reply":"2022-07-25T06:40:48.218924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    seed = 42\n    n_folds = 5\n    target = 'target'\n    boosting_type = 'gbdt'\n    metric = 'binary'","metadata":{"execution":{"iopub.status.busy":"2022-07-25T06:40:48.222285Z","iopub.execute_input":"2022-07-25T06:40:48.223268Z","iopub.status.idle":"2022-07-25T06:40:48.237927Z","shell.execute_reply.started":"2022-07-25T06:40:48.223231Z","shell.execute_reply":"2022-07-25T06:40:48.236276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n\nseed_everything(CFG.seed)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T06:40:48.239775Z","iopub.execute_input":"2022-07-25T06:40:48.240439Z","iopub.status.idle":"2022-07-25T06:40:48.253491Z","shell.execute_reply.started":"2022-07-25T06:40:48.240399Z","shell.execute_reply":"2022-07-25T06:40:48.252121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocess","metadata":{}},{"cell_type":"code","source":"def preprocess_to_trainLGBM(train_):\n    cat_features = [\n        \"B_30\",\n        \"B_38\",\n        \"D_114\",\n        \"D_116\",\n        \"D_117\",\n        \"D_120\",\n        \"D_126\",\n        \"D_63\",\n        \"D_64\",\n        \"D_66\",\n        \"D_68\"\n    ]\n    cat_features = [f\"{cf}_last\" for cf in cat_features]\n    \n    for cat_col in cat_features:\n        encoder = LabelEncoder()\n        train_[cat_col] = encoder.fit_transform(train_[cat_col])\n    \n    num_cols = list(train_.dtypes[(train_.dtypes == 'float32') | (train_.dtypes == 'float64')].index)\n    num_cols = [col for col in num_cols if 'last' in col]\n    \n    for col in num_cols:\n        train_[col + '_round2'] = train_[col].round(2)\n    num_cols = [col for col in train_.columns if 'last' in col]\n    num_cols = [col[:-5] for col in num_cols if 'round' not in col]\n    \n    for col in num_cols:\n        try:\n            train_[f'{col}_last_mean_diff'] = train_[f'{col}_last'] - train_[f'{col}_mean']\n        except: pass\n    \n    num_cols = list(train_.dtypes[(train_.dtypes == 'float32') | (train_.dtypes == 'float64')].index)\n    for col in tqdm(num_cols):\n        train_[col] = train_[col].astype(np.float16)\n    \n    features = [col for col in train_.columns if col not in ['customer_ID', CFG.target]]\n    \n    return cat_features, features","metadata":{"execution":{"iopub.status.busy":"2022-07-25T06:40:48.255298Z","iopub.execute_input":"2022-07-25T06:40:48.255689Z","iopub.status.idle":"2022-07-25T06:40:48.273037Z","shell.execute_reply.started":"2022-07-25T06:40:48.255657Z","shell.execute_reply":"2022-07-25T06:40:48.271760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Competition's metric","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)\n\n\ndef plot_roc(y_val,y_prob):\n    colors=px.colors.qualitative.Prism\n    fig=go.Figure()\n    fig.add_trace(go.Scatter(x=np.linspace(0,1,11), y=np.linspace(0,1,11), \n                             name='Random Chance',mode='lines', showlegend=False,\n                             line=dict(color=\"Black\", width=1, dash=\"dot\")))\n    for i in range(len(y_val)):\n        y=y_val[i]\n        prob=y_prob[i]\n        fpr, tpr, _ = roc_curve(y, prob)\n        roc_auc = auc(fpr,tpr)\n        fig.add_trace(go.Scatter(x=fpr, y=tpr, line=dict(color=colors[::-1][i+1], width=3), \n                                 hovertemplate = 'True positive rate = %{y:.3f}<br>False positive rate = %{x:.3f}',\n                                 name='Fold {}:  Gini = {:.3f}, AUC = {:.3f}'.format(i+1, gini[i],roc_auc)))\n    fig.update_layout(template=temp, title=\"Cross-Validation ROC Curves\", \n                      hovermode=\"x unified\", width=700,height=600,\n                      xaxis_title='False Positive Rate (1 - Specificity)',\n                      yaxis_title='True Positive Rate (Sensitivity)',\n                      legend=dict(orientation='v', y=.07, x=1, xanchor=\"right\",\n                                  bordercolor=\"black\", borderwidth=.5))\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T06:40:48.276539Z","iopub.execute_input":"2022-07-25T06:40:48.277539Z","iopub.status.idle":"2022-07-25T06:40:48.302542Z","shell.execute_reply.started":"2022-07-25T06:40:48.277491Z","shell.execute_reply":"2022-07-25T06:40:48.301252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# train lgbm functions","metadata":{}},{"cell_type":"markdown","source":"# Main","metadata":{}},{"cell_type":"code","source":"train = pd.read_parquet('../input/amex-traindataread-and-preprocess/train_fe.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T06:40:48.305075Z","iopub.execute_input":"2022-07-25T06:40:48.306007Z","iopub.status.idle":"2022-07-25T06:40:56.644486Z","shell.execute_reply.started":"2022-07-25T06:40:48.305955Z","shell.execute_reply":"2022-07-25T06:40:56.643545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_features, features = preprocess_to_trainLGBM(train)\ntrain_features = features\n# train_features.extend(cat_features)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T06:40:56.646140Z","iopub.execute_input":"2022-07-25T06:40:56.646839Z","iopub.status.idle":"2022-07-25T06:46:48.936603Z","shell.execute_reply.started":"2022-07-25T06:40:56.646802Z","shell.execute_reply":"2022-07-25T06:46:48.934998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"enc = LabelEncoder()\nfor col in cat_features[:-1]:\n    train[col] = enc.fit_transform(train[col])\npickle.dump(enc, open('label_encoder.pickle', 'wb'))\n\n# X=train.drop(train_features, axis=1)\nX=train[train_features]\ny=train[CFG.target]\n\ndel train, enc\ngc.collect()\n# y_valid, gbm_val_probs, gbm_test_preds, gini=[],[],[],[]\ny_valid, gbm_val_probs, gini=[],[],[]\nft_importance=pd.DataFrame(index=X.columns)\n\nsk_fold = StratifiedKFold(n_splits=CFG.n_folds, shuffle=True)\n\nfor fold, (train_idx, val_idx) in enumerate(sk_fold.split(X, y)):\n    \n    print(\"\\nFold {}\".format(fold+1))\n    X_train, y_train = X.iloc[train_idx,:], y[train_idx]\n    X_val, y_val = X.iloc[val_idx,:], y[val_idx]\n    print(\"Train shape: {}, {}, Valid shape: {}, {}\\n\".format(\n        X_train.shape, y_train.shape, X_val.shape, y_val.shape))\n    \n    params = {'boosting_type': 'gbdt',\n              'n_estimators': 1000,\n              'num_leaves': 50,\n              'learning_rate': 0.05,\n              'colsample_bytree': 0.9,\n              'min_child_samples': 2000,\n              'max_bins': 500,\n              'reg_alpha': 2,\n              'objective': 'binary',\n              'random_state': 21}\n    \n    gbm = LGBMClassifier(**params).fit(X_train, y_train, \n                                       eval_set=[(X_train, y_train), (X_val, y_val)],\n                                       callbacks=[early_stopping(200), log_evaluation(500)],\n                                       eval_metric=['auc','binary_logloss'])\n    gbm_prob = gbm.predict_proba(X_val)[:,1]\n    gbm_val_probs.append(gbm_prob)\n    y_valid.append(y_val)\n    \n    y_pred=pd.DataFrame(data={'prediction':gbm_prob})\n    y_true=pd.DataFrame(data={'target':y_val.reset_index(drop=True)})\n    gini_score=amex_metric(y_true = y_true, y_pred = y_pred)\n    gini.append(gini_score)\n    \n    auc_score=roc_auc_score(y_val, gbm_prob)\n#     gbm_test_preds.append(gbm.predict_proba(test)[:,1])    \n    ft_importance[\"Importance_Fold\"+str(fold)]=gbm.feature_importances_    \n    print(\"Validation Gini: {:.5f}, AUC: {:.4f}\".format(gini_score,auc_score))\n    \n    file = f'trained_model_{fold}.pkl'\n    pickle.dump(gbm, open(file, 'wb'))\n    \n    del X_train, y_train, X_val, y_val\n    _ = gc.collect()\n    \ndel X, y","metadata":{"execution":{"iopub.status.busy":"2022-07-25T06:46:48.940014Z","iopub.execute_input":"2022-07-25T06:46:48.940591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}