{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## About this NB\n\nThis notebook contains ideas taken from the following great NBs:\n\n1 From https://www.kaggle.com/code/ambrosm/amex-lightgbm-quickstart I took the idea of masking, which I used to build the feature last / next-to-last\n\n2 From https://www.kaggle.com/code/kunheekimkr/amex-lgbm-gpu-starter-0-795/notebook I took the idea of reading the test set by chunks.I also took the idea that predict(df, raw_score = True) gives the log-odds. Amex metric is invariant to log-odds.\n\n3 From https://www.kaggle.com/code/thedevastator/lag-features-are-all-you-need/ I took ideas for new features (last/mean, last - first, etc.)\n\n4 https://www.kaggle.com/competitions/amex-default-prediction/discussion/335892 gives a great overview of topics and tricks on tabular classification.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\nfrom sklearn.model_selection import train_test_split\n\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-20T16:28:31.410442Z","iopub.execute_input":"2022-08-20T16:28:31.411026Z","iopub.status.idle":"2022-08-20T16:28:33.567411Z","shell.execute_reply.started":"2022-08-20T16:28:31.410912Z","shell.execute_reply":"2022-08-20T16:28:33.566162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class config:\n    random_state = 4222\n    #kaggle = True\n    #path = '../input/amexfeather'\n    #local_path = ''","metadata":{"execution":{"iopub.status.busy":"2022-08-20T16:28:41.795193Z","iopub.execute_input":"2022-08-20T16:28:41.795751Z","iopub.status.idle":"2022-08-20T16:28:41.802579Z","shell.execute_reply.started":"2022-08-20T16:28:41.795705Z","shell.execute_reply":"2022-08-20T16:28:41.801048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Data Preprocessing**","metadata":{}},{"cell_type":"code","source":"def read_file(path = '', usecols = None):\n    \n    if usecols is not None: df = pd.read_parquet(path,columns = usecols)\n    else: df = pd.read_parquet(path)\n   \n    print('ajá:')\n    df.S_2 = pd.to_datetime( df.S_2 )\n    #df = df.fillna(NaN_value) \n    print('shape of data:', df.shape)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-20T16:28:44.770194Z","iopub.execute_input":"2022-08-20T16:28:44.770718Z","iopub.status.idle":"2022-08-20T16:28:44.778900Z","shell.execute_reply.started":"2022-08-20T16:28:44.770663Z","shell.execute_reply":"2022-08-20T16:28:44.777137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocessing(df, cat_features, num_features, i = 'train'):\n    \n    cid = pd.Categorical(df.pop('customer_ID'), ordered = True)\n    last = (cid != np.roll(cid, -1))\n    penul = np.roll(last, -1)\n    \n    if 'target' in df.columns:\n        df.drop(columns=['target'], inplace=True)\n    gc.collect()\n    print('Read', i)\n    \n    df_num = (df.groupby(cid)[num_features]\n              .agg(['first','mean','last'])\n             )\n    df_num.columns = ['_'.join(x) for x in df_num.columns]\n    print('Computed df_num', i)\n    \n    df_penul = (df.loc[penul,num_features]\n              .rename(columns={f: f\"{f}_pl\" for f in num_features})\n              .set_index(np.asarray(cid[last]))\n             )\n    print('Computed penul', i)\n    \n    df_num = pd.concat([df_num, df_penul], axis=1)\n    print('Computed concat penul', i)\n         \n    for col in df_num:\n        if 'last' in col and col.replace('last', 'pl') in df_num:\n                df_num[col + '_lg'] = df_num[col] / df_num[col.replace('last', 'pl')]         \n    print('Computed lg', i)\n    \n    new_cols = [col for col in df_num.columns if '_pl' not in col]\n    df_num = df_num[new_cols]  \n    \n    for col in df_num:\n        if 'last' in col and col.replace('last', 'mean') in df_num:\n                df_num[col + '_lm'] = df_num[col] / df_num[col.replace('last', 'mean')]     \n    print('Computed lm', i)\n    \n    for col in df_num:\n        if 'last' in col and col.replace('last', 'first') in df_num:\n                df_num[col + '_lf'] = df_num[col] - df_num[col.replace('last', 'first')]     \n    print('Computed lf', i)\n                  \n    df_cat = (df.groupby(cid)[cat_features]\n              .agg(['first','last'])\n             )\n    df_cat.columns = ['_'.join(x) for x in df_cat.columns]\n    \n    df = pd.concat([df_num, df_cat], axis=1)\n    \n    del df_num, df_cat, df_penul,cid, last, penul, new_cols\n    \n    for col in df.columns:\n        if df[col].dtype=='float64': df[col] = df[col].astype('float16')\n        if df[col].dtype=='int64': df[col] = df[col].astype('int16')\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-20T16:28:48.034867Z","iopub.execute_input":"2022-08-20T16:28:48.036845Z","iopub.status.idle":"2022-08-20T16:28:48.058436Z","shell.execute_reply.started":"2022-08-20T16:28:48.036780Z","shell.execute_reply":"2022-08-20T16:28:48.057043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Reading train data...')\ntrain_path = '../input/amex-data-integer-dtypes-parquet-format/train.parquet'\ntrain = read_file(path = train_path)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T16:28:52.669271Z","iopub.execute_input":"2022-08-20T16:28:52.669782Z","iopub.status.idle":"2022-08-20T16:29:17.038746Z","shell.execute_reply.started":"2022-08-20T16:28:52.669742Z","shell.execute_reply":"2022-08-20T16:29:17.037369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = train.drop(['customer_ID','S_2'], axis = 1).columns.to_list()\n#cat_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n#cat_features =[\"B_4\",\"S_11\",\"S_13\",\"S_15\",\"D_39\",\"D_51\",\"D_59\",\"D_74\",\"D_75\",\"D_80\",\"D_91\",\"D_92\"]\n\ncat_features =[\"B_4\",'B_30','B_38',\"S_11\",\"S_13\",\"S_15\",\"D_39\",\"D_51\",\"D_59\",'D_63','D_64','D_66','D_68',\"D_74\",\"D_75\",\"D_80\",\"D_91\",\"D_92\",'D_114','D_116','D_117','D_120','D_126']\n\nnum_features = [col for col in features if col not in cat_features]","metadata":{"execution":{"iopub.status.busy":"2022-08-20T16:29:24.837832Z","iopub.execute_input":"2022-08-20T16:29:24.838374Z","iopub.status.idle":"2022-08-20T16:29:27.098592Z","shell.execute_reply.started":"2022-08-20T16:29:24.838328Z","shell.execute_reply":"2022-08-20T16:29:27.097196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain = preprocessing(train,cat_features,num_features)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T16:29:29.813940Z","iopub.execute_input":"2022-08-20T16:29:29.814454Z","iopub.status.idle":"2022-08-20T16:30:41.807741Z","shell.execute_reply.started":"2022-08-20T16:29:29.814416Z","shell.execute_reply":"2022-08-20T16:30:41.806203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = [feat for feat in train.columns if feat != 'customer_ID' and feat != 'target' and feat != \"S_2\"]\nlen(features)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T16:30:53.242179Z","iopub.execute_input":"2022-08-20T16:30:53.242729Z","iopub.status.idle":"2022-08-20T16:30:53.251754Z","shell.execute_reply.started":"2022-08-20T16:30:53.242684Z","shell.execute_reply":"2022-08-20T16:30:53.250561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = pd.read_csv('../input/amex-default-prediction/train_labels.csv').target.values\nprint(f\"target shape: {target.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-20T16:30:55.980642Z","iopub.execute_input":"2022-08-20T16:30:55.981596Z","iopub.status.idle":"2022-08-20T16:30:57.207194Z","shell.execute_reply.started":"2022-08-20T16:30:55.981524Z","shell.execute_reply":"2022-08-20T16:30:57.205709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Model Training**","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    if isinstance(y_true, np.ndarray):\n            y_true = pd.DataFrame(y_true, columns = [\"target\"])\n    \n    if isinstance(y_pred, np.ndarray):\n            y_pred = pd.DataFrame(y_pred, columns = [\"prediction\"])\n            #y_pred[\"prediction\"] = y_pred\n    \n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n      \n        df['weight'] = df[\"target\"].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df[\"target\"] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df[\"target\"].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df[\"target\"] * df['weight']).sum()\n        df['cum_pos_found'] = (df[\"target\"] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    d = top_four_percent_captured(y_true, y_pred)\n    g = normalized_weighted_gini(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T16:31:00.324166Z","iopub.execute_input":"2022-08-20T16:31:00.324736Z","iopub.status.idle":"2022-08-20T16:31:00.341982Z","shell.execute_reply.started":"2022-08-20T16:31:00.324693Z","shell.execute_reply":"2022-08-20T16:31:00.340209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lgb_amex_metric(y_true, y_pred):\n    \"\"\"The competition metric with lightgbm's calling convention\"\"\"\n    return ('amex',\n            amex_metric(y_true, y_pred),\n            True)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T16:31:03.951512Z","iopub.execute_input":"2022-08-20T16:31:03.952540Z","iopub.status.idle":"2022-08-20T16:31:03.958157Z","shell.execute_reply.started":"2022-08-20T16:31:03.952491Z","shell.execute_reply":"2022-08-20T16:31:03.956742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"search_params = { \n    'learning_rate' : 0.03, #0.065,\n    #'lambda_l1': 7.200056653766078,\n    'lambda_l2': 50, #9.35685026658397 \n    'num_leaves': 100, #55, \n    'feature_fraction': 0.4, #0.19,\n    'bagging_fraction': 0.9, #1.0, \n    'bagging_freq': 0,\n    'min_child_samples': 2400, #100\n}\n\nfixed_params={\n    'objective': 'binary',\n    'metric': 'custom', \n    'boosting_type' : 'gbdt',\n    'random_state' : config.random_state,\n    #'n_jobs': -1,\n    #'extra_trees' : True,\n    #'feature_pre_filter': False,\n    'n_estimators': 1200, \n    'early_stopping_round': 100\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-20T16:31:30.010518Z","iopub.execute_input":"2022-08-20T16:31:30.011713Z","iopub.status.idle":"2022-08-20T16:31:30.018753Z","shell.execute_reply.started":"2022-08-20T16:31:30.011655Z","shell.execute_reply":"2022-08-20T16:31:30.017761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NaN_value = -127","metadata":{"execution":{"iopub.status.busy":"2022-08-20T16:31:34.253747Z","iopub.execute_input":"2022-08-20T16:31:34.254673Z","iopub.status.idle":"2022-08-20T16:31:34.260037Z","shell.execute_reply.started":"2022-08-20T16:31:34.254612Z","shell.execute_reply":"2022-08-20T16:31:34.258856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_modelo(df,target,features):\n    \n    df = df.fillna(NaN_value)\n    x = df[features]\n    y = pd.Series(target)\n    \n    X_train, X_test, y_train, y_test = train_test_split(x,y,test_size = 0.3,\n                                random_state = 4222, stratify = y)\n    \n    model = LGBMClassifier(**fixed_params, **search_params)\n    \n    model.fit(\n        X_train, y_train, \n        eval_set=[(X_test,y_test)],\n        eval_metric= lgb_amex_metric,\n        callbacks=[log_evaluation(100)]\n    )\n    \n    del x,y, X_train, y_train\n    \n    return model, X_test, y_test","metadata":{"execution":{"iopub.status.busy":"2022-08-20T16:31:37.075058Z","iopub.execute_input":"2022-08-20T16:31:37.076203Z","iopub.status.idle":"2022-08-20T16:31:37.086642Z","shell.execute_reply.started":"2022-08-20T16:31:37.076084Z","shell.execute_reply":"2022-08-20T16:31:37.085152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel, X_test, y_test = train_modelo(train,target,features)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T16:31:43.051027Z","iopub.execute_input":"2022-08-20T16:31:43.052466Z","iopub.status.idle":"2022-08-20T18:21:44.409664Z","shell.execute_reply.started":"2022-08-20T16:31:43.052406Z","shell.execute_reply":"2022-08-20T18:21:44.407970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = pd.DataFrame(y_test, columns = [\"target\"])\ny_pred = pd.DataFrame(y_test.copy(), columns = [\"prediction\"])\n\ny_pred[\"prediction\"] = model.predict_proba(X_test)[:,1]\namex_metric(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T18:24:37.942020Z","iopub.execute_input":"2022-08-20T18:24:37.942642Z","iopub.status.idle":"2022-08-20T18:25:28.162259Z","shell.execute_reply.started":"2022-08-20T18:24:37.942587Z","shell.execute_reply":"2022-08-20T18:25:28.160900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train, target,features, X_test, y_test, y_pred\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-17T23:30:53.848749Z","iopub.execute_input":"2022-08-17T23:30:53.849210Z","iopub.status.idle":"2022-08-17T23:30:54.013826Z","shell.execute_reply.started":"2022-08-17T23:30:53.849173Z","shell.execute_reply":"2022-08-17T23:30:54.012486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.booster_.save_model(\"./amex-model.txt\")","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:39:38.623031Z","iopub.execute_input":"2022-08-16T15:39:38.623724Z","iopub.status.idle":"2022-08-16T15:39:38.788388Z","shell.execute_reply.started":"2022-08-16T15:39:38.623689Z","shell.execute_reply":"2022-08-16T15:39:38.787320Z"},"trusted":true},"execution_count":null,"outputs":[]}]}