{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Built from this great notebook\n\nPreprocessing and training parts based on this great notebook:\nhttps://www.kaggle.com/code/ambrosm/amex-lightgbm-quickstart","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\n\nfrom sklearn.model_selection import train_test_split\n\nimport gc","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:12:04.157513Z","iopub.execute_input":"2022-07-30T05:12:04.157955Z","iopub.status.idle":"2022-07-30T05:12:06.049599Z","shell.execute_reply.started":"2022-07-30T05:12:04.157868Z","shell.execute_reply":"2022-07-30T05:12:06.049011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class config:\n    random_state = 4222\n    kaggle = True\n    #path = '../input/amexfeather'\n    #local_path = ''","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:12:09.049038Z","iopub.execute_input":"2022-07-30T05:12:09.049333Z","iopub.status.idle":"2022-07-30T05:12:09.053363Z","shell.execute_reply.started":"2022-07-30T05:12:09.049310Z","shell.execute_reply":"2022-07-30T05:12:09.052549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Data Preprocessing**","metadata":{}},{"cell_type":"code","source":"def preprocessing(df, cat_features,num_features, i = 'train'):\n    \n    cid = pd.Categorical(df.pop('customer_ID'), ordered = True)\n    last = (cid != np.roll(cid, -1)) # mask for last statement of every customer\n    penul = np.roll(last, -1)\n    lt2 = (cid != np.roll(cid, -2))\n    features = cat_features + num_features\n    \n    if 'target' in df.columns:\n        df.drop(columns=['target'], inplace=True)\n    gc.collect()\n    print('Read', i)\n        \n    df_last = (df.loc[last,features]\n              .rename(columns={f: f\"{f}_lt\" for f in features})\n              .set_index(np.asarray(cid[last]))\n             )\n    gc.collect()\n    print('Computed last', i)\n    \n    df_std = (df\n              .groupby(cid)\n              .std()[num_features]\n              .rename(columns={f: f\"{f}_std\" for f in num_features})\n            )\n    gc.collect()\n    print(\"computed std\", i)\n\n    df_avg = (df\n              .groupby(cid)\n              .mean()[num_features]\n              .rename(columns={f: f\"{f}_avg\" for f in num_features})\n            )\n    gc.collect()\n    print(\"computed avg\", i)\n        \n    df = pd.concat([df_last,df_std, df_avg], axis=1)\n    \n    del df_last, df_std, df_avg,cid, last, penul, lt2, features\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:12:29.951318Z","iopub.execute_input":"2022-07-30T05:12:29.951658Z","iopub.status.idle":"2022-07-30T05:12:29.963353Z","shell.execute_reply.started":"2022-07-30T05:12:29.951635Z","shell.execute_reply":"2022-07-30T05:12:29.961439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#    df_p = (df.loc[lt2,num_features]\n#           .rename(columns={f: f\"{f}_pct\" for f in num_features})\n#           .set_index(np.asarray(cid[lt2]))\n#         )\n#    gc.collect()\n#    print('Computed p', i)\n    \n#    df_pct = df_p.groupby(cid[lt2]).pct_change()\n#    mask = (cid[lt2] != np.roll(cid[lt2], -1))\n#    df_pct = df_pct.loc[mask,]            \n#    gc.collect()\n#    print('Computed pct', i)   ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_parquet(f'../input/amex-data-integer-dtypes-parquet-format/train.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:12:35.187407Z","iopub.execute_input":"2022-07-30T05:12:35.187759Z","iopub.status.idle":"2022-07-30T05:12:52.206501Z","shell.execute_reply.started":"2022-07-30T05:12:35.187731Z","shell.execute_reply":"2022-07-30T05:12:52.205836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nnum_features = [col for col in train.columns if col not in cat_features + [\"target\", \"customer_ID\", \"S_2\"] ]","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:12:54.942954Z","iopub.execute_input":"2022-07-30T05:12:54.943261Z","iopub.status.idle":"2022-07-30T05:12:54.948953Z","shell.execute_reply.started":"2022-07-30T05:12:54.943238Z","shell.execute_reply":"2022-07-30T05:12:54.947834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = preprocessing(train, cat_features, num_features)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:12:57.886170Z","iopub.execute_input":"2022-07-30T05:12:57.886553Z","iopub.status.idle":"2022-07-30T05:13:29.642859Z","shell.execute_reply.started":"2022-07-30T05:12:57.886524Z","shell.execute_reply":"2022-07-30T05:13:29.642057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = [feat for feat in train.columns if feat != 'customer_ID' and feat != 'target' and feat != \"S_2\"]\nlen(features)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:13:36.508022Z","iopub.execute_input":"2022-07-30T05:13:36.508324Z","iopub.status.idle":"2022-07-30T05:13:36.514508Z","shell.execute_reply.started":"2022-07-30T05:13:36.508302Z","shell.execute_reply":"2022-07-30T05:13:36.513556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = pd.read_csv('../input/amex-default-prediction/train_labels.csv').target.values\nprint(f\"target shape: {target.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:13:39.498172Z","iopub.execute_input":"2022-07-30T05:13:39.498498Z","iopub.status.idle":"2022-07-30T05:13:40.382971Z","shell.execute_reply.started":"2022-07-30T05:13:39.498469Z","shell.execute_reply":"2022-07-30T05:13:40.381943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Model Training**","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    if isinstance(y_true, np.ndarray):\n            y_true = pd.DataFrame(y_true, columns = [\"target\"])\n    \n    if isinstance(y_pred, np.ndarray):\n            y_pred = pd.DataFrame(y_pred, columns = [\"prediction\"])\n            #y_pred[\"prediction\"] = y_pred\n    \n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n      \n        df['weight'] = df[\"target\"].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df[\"target\"] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df[\"target\"].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df[\"target\"] * df['weight']).sum()\n        df['cum_pos_found'] = (df[\"target\"] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    d = top_four_percent_captured(y_true, y_pred)\n    g = normalized_weighted_gini(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:13:43.322638Z","iopub.execute_input":"2022-07-30T05:13:43.323518Z","iopub.status.idle":"2022-07-30T05:13:43.336060Z","shell.execute_reply.started":"2022-07-30T05:13:43.323489Z","shell.execute_reply":"2022-07-30T05:13:43.335268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lgb_amex_metric(y_true, y_pred):\n    \"\"\"The competition metric with lightgbm's calling convention\"\"\"\n    return ('amex',\n            amex_metric(y_true, y_pred),\n            True)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:13:46.813196Z","iopub.execute_input":"2022-07-30T05:13:46.813616Z","iopub.status.idle":"2022-07-30T05:13:46.819214Z","shell.execute_reply.started":"2022-07-30T05:13:46.813588Z","shell.execute_reply":"2022-07-30T05:13:46.818148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"search_params = { \n    'learning_rate' : 0.065,\n    'lambda_l1': 8.481514197781607,\n    'lambda_l2': 0, #0.0004266339834880936,\n    'num_leaves': 27,\n    'feature_fraction': 0.484,\n    'bagging_fraction': 0.8477112190030014,\n    'bagging_freq': 2,\n    'min_child_samples': 20\n}\n\nfixed_params={\n    'objective': 'binary',\n    'metric': 'custom', #'binay_logloss',\n    'boosting_type' : 'gbdt',\n    #'force_row_wise' : True,\n    #'device': 'gpu',\n    'random_state' : config.random_state,\n    #'extra_trees' : True,\n    #'feature_pre_filter': False,\n    'n_estimators': 600,\n    'early_stopping_round': 50\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:14:03.121685Z","iopub.execute_input":"2022-07-30T05:14:03.122025Z","iopub.status.idle":"2022-07-30T05:14:03.128004Z","shell.execute_reply.started":"2022-07-30T05:14:03.122001Z","shell.execute_reply":"2022-07-30T05:14:03.127249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_modelo(df,target,features):\n    \n    x = df[features]\n    y = pd.Series(target)\n    \n    #enc = OrdinalEncoder()\n    #x[cat_features] = enc.fit_transform(x[cat_features])\n\n    X_train, X_test, y_train, y_test = train_test_split(x,y,test_size = 0.3,\n                                random_state = config.random_state, stratify = y)\n    \n    model = LGBMClassifier(**fixed_params, **search_params)\n    \n    model.fit(\n        X_train, y_train, \n        eval_set=[(X_test,y_test)],\n        eval_metric= lgb_amex_metric,\n        callbacks=[log_evaluation(50)]\n    )\n    \n    del x,y,X_train, y_train\n    \n    return model, X_test, y_test","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:13:50.399183Z","iopub.execute_input":"2022-07-30T05:13:50.399592Z","iopub.status.idle":"2022-07-30T05:13:50.410057Z","shell.execute_reply.started":"2022-07-30T05:13:50.399555Z","shell.execute_reply":"2022-07-30T05:13:50.409108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel, X_test, y_test = train_modelo(train,target,features)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:14:07.414746Z","iopub.execute_input":"2022-07-30T05:14:07.415109Z","iopub.status.idle":"2022-07-30T05:18:56.877724Z","shell.execute_reply.started":"2022-07-30T05:14:07.415081Z","shell.execute_reply":"2022-07-30T05:18:56.876788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = pd.DataFrame(y_test, columns = [\"target\"])\ny_pred = pd.DataFrame(y_test.copy(), columns = [\"prediction\"])\n\ny_pred[\"prediction\"] = model.predict_proba(X_test)[:,1]\namex_metric(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T05:19:16.851499Z","iopub.execute_input":"2022-07-30T05:19:16.851855Z","iopub.status.idle":"2022-07-30T05:19:19.483555Z","shell.execute_reply.started":"2022-07-30T05:19:16.851829Z","shell.execute_reply":"2022-07-30T05:19:19.482849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train, target, X_test, y_test, y_pred\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:40:02.272718Z","iopub.execute_input":"2022-07-28T15:40:02.273697Z","iopub.status.idle":"2022-07-28T15:40:02.503928Z","shell.execute_reply.started":"2022-07-28T15:40:02.273657Z","shell.execute_reply":"2022-07-28T15:40:02.502949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Submission**\n\nRead the test file in chunks. Idea from this great notebook:\nhttps://www.kaggle.com/code/kunheekimkr/amex-lgbm-gpu-starter-0-795/comments","metadata":{}},{"cell_type":"code","source":"def read_file(path = '', usecols = None):\n    if usecols is not None: df = pd.read_parquet(path,columns = usecols)\n    else: df = pd.read_parquet(path)\n   \n    #df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = pd.to_datetime( df.S_2 )\n    #df = df.fillna(NAN_VALUE) \n    print('shape of data:', df.shape)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:40:18.084802Z","iopub.execute_input":"2022-07-28T15:40:18.085250Z","iopub.status.idle":"2022-07-28T15:40:18.091672Z","shell.execute_reply.started":"2022-07-28T15:40:18.085214Z","shell.execute_reply":"2022-07-28T15:40:18.090444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate size of each separate test part\n\ndef get_rows(customers, test, NUM_PARTS = 4, verbose = ''):\n    chunk = len(customers)//NUM_PARTS\n    if verbose != '':\n        print(f'We will process {verbose} data as {NUM_PARTS} separate parts.')\n        print(f'There will be {chunk} customers in each part (except the last part).')\n        print('Below are number of rows in each part:')\n    rows = []\n\n    for k in range(NUM_PARTS):\n        if k == NUM_PARTS-1: cc = customers[k*chunk:]\n        else: cc = customers[k*chunk:(k+1)*chunk]\n        s = test.loc[test.customer_ID.isin(cc)].shape[0]\n        rows.append(s)\n    \n    if verbose != '': print( rows )\n    \n    return rows,chunk","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:40:23.979343Z","iopub.execute_input":"2022-07-28T15:40:23.979736Z","iopub.status.idle":"2022-07-28T15:40:23.986560Z","shell.execute_reply.started":"2022-07-28T15:40:23.979703Z","shell.execute_reply":"2022-07-28T15:40:23.985586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute size of parts for test data\nNUM_PARTS = 4\nTEST_PATH =  '../input/amex-data-integer-dtypes-parquet-format/test.parquet'\n\nprint(f'Reading test data...')\ntest = read_file(path = TEST_PATH, usecols = ['customer_ID','S_2'])\n\ncustomers = test[['customer_ID']].drop_duplicates().sort_index().values.flatten()\n\nrows,num_cust = get_rows(customers,test[['customer_ID']], NUM_PARTS = NUM_PARTS, verbose = 'test')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:40:31.428396Z","iopub.execute_input":"2022-07-28T15:40:31.429070Z","iopub.status.idle":"2022-07-28T15:40:42.085479Z","shell.execute_reply.started":"2022-07-28T15:40:31.429033Z","shell.execute_reply":"2022-07-28T15:40:42.083829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T15:40:47.887523Z","iopub.execute_input":"2022-07-28T15:40:47.887926Z","iopub.status.idle":"2022-07-28T15:40:48.053210Z","shell.execute_reply.started":"2022-07-28T15:40:47.887880Z","shell.execute_reply":"2022-07-28T15:40:48.052392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INFER TEST DATA IN PARTS\nskip_rows = 0\nskip_cust = 0\ntest_preds = []\n\nfor k in range(NUM_PARTS):\n    print(f'\\nReading test data...')\n    test = read_file(path = TEST_PATH)\n    test = test.iloc[skip_rows:skip_rows + rows[k]]\n    skip_rows += rows[k]\n    print(f'=> Test part {k+1} has shape', test.shape )\n          \n    test = preprocessing(test, cat_features, num_features, i = 'test')\n    if k == 0: \n        features = [feat for feat in test.columns if feat != 'customer_ID' and feat != 'target' and feat != \"S_2\"]\n \n    if k == NUM_PARTS - 1: test = test.loc[customers[skip_cust:]]\n    else: test = test.loc[customers[skip_cust:skip_cust+num_cust]]\n    skip_cust += num_cust\n    \n    preds = model.predict_proba(test[features])[:,1]\n    print(\"1=\",preds[:3])\n    test_preds.append(preds)\n\n# Clean Memory\ndel test, model\n_ = gc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predictions = np.concatenate(test_preds)\n\nsubmission = pd.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")\nsubmission.loc[:, \"prediction\"] = test_predictions\n\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{},"execution_count":null,"outputs":[]}]}