{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Built from this great notebook\n\nPreprocessing and training parts based on this great notebook:\nhttps://www.kaggle.com/code/ambrosm/amex-lightgbm-quickstart","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\nfrom sklearn.model_selection import train_test_split\n\n#import matplotlib.pyplot as plt, gc, os\n\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:02:01.117094Z","iopub.execute_input":"2022-08-16T15:02:01.118135Z","iopub.status.idle":"2022-08-16T15:02:01.122894Z","shell.execute_reply.started":"2022-08-16T15:02:01.118083Z","shell.execute_reply":"2022-08-16T15:02:01.122109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class config:\n    random_state = 4222\n    #kaggle = True\n    #path = '../input/amexfeather'\n    #local_path = ''","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:02:04.699954Z","iopub.execute_input":"2022-08-16T15:02:04.700362Z","iopub.status.idle":"2022-08-16T15:02:04.704907Z","shell.execute_reply.started":"2022-08-16T15:02:04.700313Z","shell.execute_reply":"2022-08-16T15:02:04.703999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Data Preprocessing**","metadata":{}},{"cell_type":"code","source":"def preprocessing(df, cat_features, num_features, i = 'train'):\n    \n    cid = pd.Categorical(df.pop('customer_ID'), ordered = True)\n    last = (cid != np.roll(cid, -1))\n    penul = np.roll(last, -1)\n    \n    if 'target' in df.columns:\n        df.drop(columns=['target'], inplace=True)\n    gc.collect()\n    print('Read', i)\n    \n    df_num = (df.groupby(cid)[num_features]\n              .agg(['std','mean','last'])\n             )\n    df_num.columns = ['_'.join(x) for x in df_num.columns]\n    print('Computed df_num', i)\n    \n    df_penul = (df.loc[penul,num_features]\n              .rename(columns={f: f\"{f}_pl\" for f in num_features})\n              .set_index(np.asarray(cid[last]))\n             )\n    print('Computed penul', i)\n    \n    df_num = pd.concat([df_num, df_penul], axis=1)\n    print('Computed concat penul', i)\n         \n    for col in df_num:\n        if 'last' in col and col.replace('last', 'pl') in df_num:\n                df_num[col + '_dv'] = df_num[col] / df_num[col.replace('last', 'pl')]         \n    print('Computed div', i)\n    \n    new_cols = [col for col in df_num.columns if '_pl' not in col]\n    df_num = df_num[new_cols]  \n              \n    df_cat = (df.groupby(cid)[cat_features]\n              .agg(['first','last', 'nunique'])\n             )\n    df_cat.columns = ['_'.join(x) for x in df_cat.columns]\n    \n    df = pd.concat([df_num, df_cat], axis=1)\n    \n    del df_num, df_cat, df_penul,cid, last, penul, new_cols\n    \n    for col in df.columns:\n        if df[col].dtype=='float64': df[col] = df[col].astype('float16')\n        if df[col].dtype=='int64': df[col] = df[col].astype('int16')\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:02:08.035832Z","iopub.execute_input":"2022-08-16T15:02:08.036241Z","iopub.status.idle":"2022-08-16T15:02:08.050267Z","shell.execute_reply.started":"2022-08-16T15:02:08.036208Z","shell.execute_reply":"2022-08-16T15:02:08.049443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_parquet(f'../input/amex-data-integer-dtypes-parquet-format/train.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:02:23.616998Z","iopub.execute_input":"2022-08-16T15:02:23.617474Z","iopub.status.idle":"2022-08-16T15:02:42.934064Z","shell.execute_reply.started":"2022-08-16T15:02:23.617436Z","shell.execute_reply":"2022-08-16T15:02:42.933262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df):\n    \"\"\" \n    iterate through all the columns of a dataframe and \n    modify the data type to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print(('Memory usage of dataframe is {:.2f}' \n                     'MB').format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max <\\\n                  np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max <\\\n                   np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max <\\\n                   np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max <\\\n                   np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max <\\\n                   np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max <\\\n                   np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n    end_mem = df.memory_usage().sum() / 1024**2\n    print(('Memory usage after optimization is: {:.2f}' \n                              'MB').format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) \n                                             / start_mem))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:02:50.323248Z","iopub.execute_input":"2022-08-16T15:02:50.323650Z","iopub.status.idle":"2022-08-16T15:02:50.337830Z","shell.execute_reply.started":"2022-08-16T15:02:50.323618Z","shell.execute_reply":"2022-08-16T15:02:50.336786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = train.drop(['customer_ID','S_2'], axis = 1).columns.to_list()\n#cat_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n#cat_features =[\"B_4\",\"S_11\",\"S_13\",\"S_15\",\"D_39\",\"D_51\",\"D_59\",\"D_74\",\"D_75\",\"D_80\",\"D_91\",\"D_92\"]\n\ncat_features =[\"B_4\",'B_30','B_38',\"S_11\",\"S_13\",\"S_15\",\"D_39\",\"D_51\",\"D_59\",'D_63','D_64','D_66','D_68',\"D_74\",\"D_75\",\"D_80\",\"D_91\",\"D_92\",'D_114','D_116','D_117','D_120','D_126']\n\nnum_features = [col for col in features if col not in cat_features]","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:02:55.544479Z","iopub.execute_input":"2022-08-16T15:02:55.544924Z","iopub.status.idle":"2022-08-16T15:02:56.819198Z","shell.execute_reply.started":"2022-08-16T15:02:55.544891Z","shell.execute_reply":"2022-08-16T15:02:56.817962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain = preprocessing(train,cat_features,num_features)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:03:04.323963Z","iopub.execute_input":"2022-08-16T15:03:04.324937Z","iopub.status.idle":"2022-08-16T15:04:12.091314Z","shell.execute_reply.started":"2022-08-16T15:03:04.324895Z","shell.execute_reply":"2022-08-16T15:04:12.090194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = reduce_mem_usage(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:04:26.123233Z","iopub.execute_input":"2022-08-16T15:04:26.124549Z","iopub.status.idle":"2022-08-16T15:04:35.253984Z","shell.execute_reply.started":"2022-08-16T15:04:26.124491Z","shell.execute_reply":"2022-08-16T15:04:35.252908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = [feat for feat in train.columns if feat != 'customer_ID' and feat != 'target' and feat != \"S_2\"]\nlen(features)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:04:47.388132Z","iopub.execute_input":"2022-08-16T15:04:47.388555Z","iopub.status.idle":"2022-08-16T15:04:47.395509Z","shell.execute_reply.started":"2022-08-16T15:04:47.388518Z","shell.execute_reply":"2022-08-16T15:04:47.394779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = pd.read_csv('../input/amex-default-prediction/train_labels.csv').target.values\nprint(f\"target shape: {target.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:04:49.666851Z","iopub.execute_input":"2022-08-16T15:04:49.667502Z","iopub.status.idle":"2022-08-16T15:04:50.776783Z","shell.execute_reply.started":"2022-08-16T15:04:49.667462Z","shell.execute_reply":"2022-08-16T15:04:50.775689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Model Training**","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    if isinstance(y_true, np.ndarray):\n            y_true = pd.DataFrame(y_true, columns = [\"target\"])\n    \n    if isinstance(y_pred, np.ndarray):\n            y_pred = pd.DataFrame(y_pred, columns = [\"prediction\"])\n            #y_pred[\"prediction\"] = y_pred\n    \n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n      \n        df['weight'] = df[\"target\"].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df[\"target\"] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df[\"target\"].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df[\"target\"] * df['weight']).sum()\n        df['cum_pos_found'] = (df[\"target\"] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    d = top_four_percent_captured(y_true, y_pred)\n    g = normalized_weighted_gini(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:04:58.131671Z","iopub.execute_input":"2022-08-16T15:04:58.132085Z","iopub.status.idle":"2022-08-16T15:04:58.147860Z","shell.execute_reply.started":"2022-08-16T15:04:58.132053Z","shell.execute_reply":"2022-08-16T15:04:58.146818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lgb_amex_metric(y_true, y_pred):\n    \"\"\"The competition metric with lightgbm's calling convention\"\"\"\n    return ('amex',\n            amex_metric(y_true, y_pred),\n            True)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:05:19.853686Z","iopub.execute_input":"2022-08-16T15:05:19.854514Z","iopub.status.idle":"2022-08-16T15:05:19.859188Z","shell.execute_reply.started":"2022-08-16T15:05:19.854471Z","shell.execute_reply":"2022-08-16T15:05:19.858416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"search_params = { \n    'learning_rate' : 0.03, #0.065,\n    #'lambda_l1': 8.481514197781607,\n    'lambda_l2': 50, #0.0004266339834880936,\n    'num_leaves': 100, #27,\n    'feature_fraction': 0.19, #0.484,\n    #'bagging_fraction': 0.8477112190030014,\n    #'bagging_freq': 2,\n    'min_child_samples': 2400 #40 #20\n}\n\nfixed_params={\n    'objective': 'binary',\n    'metric': 'custom', \n    'boosting_type' : 'gbdt',\n    #'force_row_wise' : True,\n    #'device': 'gpu',\n    'random_state' : config.random_state,\n    #'n_jobs': -1,\n    #'extra_trees' : True,\n    #'feature_pre_filter': False,\n    'n_estimators': 1200, \n    'early_stopping_round': 100\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:05:22.889459Z","iopub.execute_input":"2022-08-16T15:05:22.889867Z","iopub.status.idle":"2022-08-16T15:05:22.896204Z","shell.execute_reply.started":"2022-08-16T15:05:22.889835Z","shell.execute_reply":"2022-08-16T15:05:22.895421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NaN_value = -127","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:05:36.501206Z","iopub.execute_input":"2022-08-16T15:05:36.501672Z","iopub.status.idle":"2022-08-16T15:05:36.506525Z","shell.execute_reply.started":"2022-08-16T15:05:36.501635Z","shell.execute_reply":"2022-08-16T15:05:36.505420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_modelo(df,target,features):\n    \n    df = df.fillna(NaN_value)\n    x = df[features]\n    y = pd.Series(target)\n    \n    X_train, X_test, y_train, y_test = train_test_split(x,y,test_size = 0.3,\n                                random_state = 4222, stratify = y)\n    \n    model = LGBMClassifier(**fixed_params, **search_params)\n    \n    model.fit(\n        X_train, y_train, \n        eval_set=[(X_test,y_test)],\n        eval_metric= lgb_amex_metric,\n        callbacks=[log_evaluation(100)]\n    )\n    \n    del x,y, X_train, y_train\n    \n    return model, X_test, y_test","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:05:38.743072Z","iopub.execute_input":"2022-08-16T15:05:38.743476Z","iopub.status.idle":"2022-08-16T15:05:38.750717Z","shell.execute_reply.started":"2022-08-16T15:05:38.743443Z","shell.execute_reply":"2022-08-16T15:05:38.749404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel, X_test, y_test = train_modelo(train,target,features)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:05:42.767297Z","iopub.execute_input":"2022-08-16T15:05:42.767721Z","iopub.status.idle":"2022-08-16T15:15:26.259086Z","shell.execute_reply.started":"2022-08-16T15:05:42.767686Z","shell.execute_reply":"2022-08-16T15:15:26.258003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = pd.DataFrame(y_test, columns = [\"target\"])\ny_pred = pd.DataFrame(y_test.copy(), columns = [\"prediction\"])\n\ny_pred[\"prediction\"] = model.predict_proba(X_test)[:,1]\namex_metric(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:15:32.104128Z","iopub.execute_input":"2022-08-16T15:15:32.104563Z","iopub.status.idle":"2022-08-16T15:15:40.170563Z","shell.execute_reply.started":"2022-08-16T15:15:32.104525Z","shell.execute_reply":"2022-08-16T15:15:40.169369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train, target,features, X_test, y_test, y_pred\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T15:16:07.078713Z","iopub.execute_input":"2022-08-16T15:16:07.079098Z","iopub.status.idle":"2022-08-16T15:16:07.278172Z","shell.execute_reply.started":"2022-08-16T15:16:07.079067Z","shell.execute_reply":"2022-08-16T15:16:07.277149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Submission**\n\nRead the test file in chunks. Idea from this great notebook:\nhttps://www.kaggle.com/code/kunheekimkr/amex-lgbm-gpu-starter-0-795/comments","metadata":{}},{"cell_type":"code","source":"def read_file(path = '', usecols = None):\n    \n    if usecols is not None: df = pd.read_parquet(path,columns = usecols)\n    else: df = pd.read_parquet(path)\n   \n    print('ajá:')\n    #df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = pd.to_datetime( df.S_2 )\n    #df = df.fillna(NaN_value) \n    print('shape of data:', df.shape)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-16T04:34:53.276733Z","iopub.execute_input":"2022-08-16T04:34:53.277093Z","iopub.status.idle":"2022-08-16T04:34:53.282336Z","shell.execute_reply.started":"2022-08-16T04:34:53.277069Z","shell.execute_reply":"2022-08-16T04:34:53.281460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate size of each separate test part\n\ndef get_rows(customers, test, NUM_PARTS = 4, verbose = ''):\n    chunk = len(customers)//NUM_PARTS\n    if verbose != '':\n        print(f'We will process {verbose} data as {NUM_PARTS} separate parts.')\n        print(f'There will be {chunk} customers in each part (except the last part).')\n        print('Below are number of rows in each part:')\n    rows = []\n\n    for k in range(NUM_PARTS):\n        if k == NUM_PARTS-1: cc = customers[k*chunk:]\n        else: cc = customers[k*chunk:(k+1)*chunk]\n        s = test.loc[test.customer_ID.isin(cc)].shape[0]\n        rows.append(s)\n    \n    if verbose != '': print( rows )\n    \n    return rows,chunk","metadata":{"execution":{"iopub.status.busy":"2022-08-16T04:35:00.986478Z","iopub.execute_input":"2022-08-16T04:35:00.986824Z","iopub.status.idle":"2022-08-16T04:35:00.993848Z","shell.execute_reply.started":"2022-08-16T04:35:00.986800Z","shell.execute_reply":"2022-08-16T04:35:00.993237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute size of parts for test data\nNUM_PARTS = 4\nTEST_PATH =  '../input/amex-data-integer-dtypes-parquet-format/test.parquet'\n\nprint(f'Reading test data...')\ntest = read_file(path = TEST_PATH, usecols = ['customer_ID','S_2'])\n\ncustomers = test[['customer_ID']].drop_duplicates().sort_index().values.flatten()\n\nrows,num_cust = get_rows(customers,test[['customer_ID']], NUM_PARTS = NUM_PARTS, verbose = 'test')","metadata":{"execution":{"iopub.status.busy":"2022-08-16T04:35:05.265941Z","iopub.execute_input":"2022-08-16T04:35:05.266257Z","iopub.status.idle":"2022-08-16T04:35:14.049470Z","shell.execute_reply.started":"2022-08-16T04:35:05.266233Z","shell.execute_reply":"2022-08-16T04:35:14.048479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INFER TEST DATA IN PARTS\nskip_rows = 0\nskip_cust = 0\ntest_preds = []\n\nfor k in range(NUM_PARTS):\n    print(f'\\nReading test data...')\n    test = read_file(path = TEST_PATH)\n    test = test.iloc[skip_rows:skip_rows + rows[k]]\n    skip_rows += rows[k]\n    print(f'=> Test part {k+1} has shape', test.shape )\n          \n    test = preprocessing(test, cat_features, num_features, i = 'test')\n    if k == 0: \n        features = [feat for feat in test.columns if feat != 'customer_ID' and feat != 'target' and feat != \"S_2\"]\n \n    if k == NUM_PARTS - 1: test = test.loc[customers[skip_cust:]]\n    else: test = test.loc[customers[skip_cust:skip_cust+num_cust]]\n    skip_cust += num_cust\n    \n    preds = model.predict_proba(test[features])[:,1]\n    print(\"1=\",preds[:3])\n    test_preds.append(preds)\n\n# Clean Memory\ndel test, model\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T04:36:18.196805Z","iopub.execute_input":"2022-08-16T04:36:18.197149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predictions = np.concatenate(test_preds)\n\nsubmission = pd.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")\nsubmission.loc[:, \"prediction\"] = test_predictions\n\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{},"execution_count":null,"outputs":[]}]}