{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Built from this great notebook\n\nPreprocessing and training parts based on this great notebook:\nhttps://www.kaggle.com/code/ambrosm/amex-lightgbm-quickstart","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom lightgbm import Booster, LGBMRegressor\n\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:38:45.487465Z","iopub.execute_input":"2022-08-16T16:38:45.487973Z","iopub.status.idle":"2022-08-16T16:38:45.492900Z","shell.execute_reply.started":"2022-08-16T16:38:45.487936Z","shell.execute_reply":"2022-08-16T16:38:45.491964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class config:\n    random_state = 4222\n    #kaggle = True\n    #path = '../input/amexfeather'\n    #local_path = ''","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:38:48.287324Z","iopub.execute_input":"2022-08-16T16:38:48.288246Z","iopub.status.idle":"2022-08-16T16:38:48.292292Z","shell.execute_reply.started":"2022-08-16T16:38:48.288208Z","shell.execute_reply":"2022-08-16T16:38:48.291561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Data Preprocessing**","metadata":{}},{"cell_type":"code","source":"def read_file(path = '', usecols = None):\n    \n    if usecols is not None: df = pd.read_parquet(path,columns = usecols)\n    else: df = pd.read_parquet(path)\n   \n    print('ajá:')\n    df.S_2 = pd.to_datetime( df.S_2 )\n    print('shape of data:', df.shape)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:38:51.291996Z","iopub.execute_input":"2022-08-16T16:38:51.292685Z","iopub.status.idle":"2022-08-16T16:38:51.297746Z","shell.execute_reply.started":"2022-08-16T16:38:51.292650Z","shell.execute_reply":"2022-08-16T16:38:51.297012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocessing(df, cat_features, num_features, i = 'train'):\n    \n    cid = pd.Categorical(df.pop('customer_ID'), ordered = True)\n    last = (cid != np.roll(cid, -1))\n    penul = np.roll(last, -1)\n    \n    if 'target' in df.columns:\n        df.drop(columns=['target'], inplace=True)\n    gc.collect()\n    print('Read', i)\n    \n    df_num = (df.groupby(cid)[num_features]\n              .agg(['first','std','mean','last'])\n             )\n    df_num.columns = ['_'.join(x) for x in df_num.columns]\n    print('Computed df_num', i)\n    \n    df_penul = (df.loc[penul,num_features]\n              .rename(columns={f: f\"{f}_pl\" for f in num_features})\n              .set_index(np.asarray(cid[last]))\n             )\n    print('Computed penul', i)\n    \n    df_num = pd.concat([df_num, df_penul], axis=1)\n    print('Computed concat penul', i)\n         \n    for col in df_num:\n        if 'last' in col and col.replace('last', 'pl') in df_num:\n                df_num[col + '_dv'] = df_num[col] / df_num[col.replace('last', 'pl')]         \n    print('Computed div', i)\n    \n    new_cols = [col for col in df_num.columns if '_pl' not in col]\n    df_num = df_num[new_cols]  \n    \n    for col in df_num:\n        if 'last' in col and col.replace('last', 'mean') in df_num:\n                df_num[col + '_lm'] = df_num[col] / df_num[col.replace('last', 'mean')]     \n    print('Computed lm', i)\n    \n    for col in df_num:\n        if 'last' in col and col.replace('last', 'first') in df_num:\n                df_num[col + '_lf'] = df_num[col] - df_num[col.replace('last', 'first')]     \n    print('Computed lf', i)\n              \n    df_cat = (df.groupby(cid)[cat_features]\n              .agg(['first','last', 'nunique'])\n             )\n    df_cat.columns = ['_'.join(x) for x in df_cat.columns]\n    \n    df = pd.concat([df_num, df_cat], axis=1)\n    \n    del df_num, df_cat, df_penul,cid, last, penul, new_cols\n    \n    for col in df.columns:\n        if df[col].dtype=='float64': df[col] = df[col].astype('float16')\n        if df[col].dtype=='int64': df[col] = df[col].astype('int16')\n    \n    return df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Submission**\n\nRead the test file in chunks. Idea from this great notebook:\nhttps://www.kaggle.com/code/kunheekimkr/amex-lgbm-gpu-starter-0-795/comments","metadata":{}},{"cell_type":"code","source":"# Calculate size of each separate test part\n\ndef get_rows(customers, test, NUM_PARTS = 4, verbose = ''):\n    chunk = len(customers)//NUM_PARTS\n    if verbose != '':\n        print(f'We will process {verbose} data as {NUM_PARTS} separate parts.')\n        print(f'There will be {chunk} customers in each part (except the last part).')\n        print('Below are number of rows in each part:')\n    rows = []\n\n    for k in range(NUM_PARTS):\n        if k == NUM_PARTS-1: cc = customers[k*chunk:]\n        else: cc = customers[k*chunk:(k+1)*chunk]\n        s = test.loc[test.customer_ID.isin(cc)].shape[0]\n        rows.append(s)\n    \n    if verbose != '': print( rows )\n    \n    return rows,chunk","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:39:02.135496Z","iopub.execute_input":"2022-08-16T16:39:02.136369Z","iopub.status.idle":"2022-08-16T16:39:02.144293Z","shell.execute_reply.started":"2022-08-16T16:39:02.136332Z","shell.execute_reply":"2022-08-16T16:39:02.143435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute size of parts for test data\nNUM_PARTS = 4\ntest_path =  '../input/amex-data-integer-dtypes-parquet-format/test.parquet'\n\nprint(f'Reading test data...')\ntest = read_file(path = test_path, usecols = ['customer_ID','S_2'])\n\ncustomers = test[['customer_ID']].drop_duplicates().sort_index().values.flatten()\n\nrows,num_cust = get_rows(customers,test[['customer_ID']], NUM_PARTS = NUM_PARTS, verbose = 'test')","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:39:05.365491Z","iopub.execute_input":"2022-08-16T16:39:05.366440Z","iopub.status.idle":"2022-08-16T16:39:18.422536Z","shell.execute_reply.started":"2022-08-16T16:39:05.366402Z","shell.execute_reply":"2022-08-16T16:39:18.421578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:39:51.048813Z","iopub.execute_input":"2022-08-16T16:39:51.049746Z","iopub.status.idle":"2022-08-16T16:39:51.437806Z","shell.execute_reply.started":"2022-08-16T16:39:51.049702Z","shell.execute_reply":"2022-08-16T16:39:51.436879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_file = \"../input/amexmodel/amex-model.txt\"\nmodel = Booster(model_file = model_file)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:39:55.394528Z","iopub.execute_input":"2022-08-16T16:39:55.394981Z","iopub.status.idle":"2022-08-16T16:39:55.681315Z","shell.execute_reply.started":"2022-08-16T16:39:55.394943Z","shell.execute_reply":"2022-08-16T16:39:55.680408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NaN_value = -127","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_features =[\"B_4\",'B_30','B_38',\"S_11\",\"S_13\",\"S_15\",\"D_39\",\"D_51\",\"D_59\",'D_63','D_64','D_66','D_68',\"D_74\",\"D_75\",\"D_80\",\"D_91\",\"D_92\",'D_114','D_116','D_117','D_120','D_126']","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:50:06.995349Z","iopub.execute_input":"2022-08-16T16:50:06.997120Z","iopub.status.idle":"2022-08-16T16:50:07.004449Z","shell.execute_reply.started":"2022-08-16T16:50:06.997051Z","shell.execute_reply":"2022-08-16T16:50:07.003381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INFER TEST DATA IN PARTS\nskip_rows = 0\nskip_cust = 0\ntest_preds = []\n\nfor k in range(NUM_PARTS):\n    print(f'\\nReading test data...')\n    test = read_file(path = test_path)\n    test = test.iloc[skip_rows:skip_rows + rows[k]]\n    skip_rows += rows[k]\n    print(f'=> Test part {k+1} has shape', test.shape )\n\n    features = test.drop(['customer_ID','S_2'], axis = 1).columns.to_list()\n    num_features = [col for col in features if col not in cat_features]\n    \n    test = preprocessing(test, cat_features, num_features, i = 'test')\n    if k == NUM_PARTS - 1: test = test.loc[customers[skip_cust:]]\n    else: test = test.loc[customers[skip_cust:skip_cust+num_cust]]\n    skip_cust += num_cust\n    \n    test = test.fillna(NaN_value)\n    features = [feat for feat in test.columns if feat != 'customer_ID' and feat != 'target' and feat != \"S_2\"]\n    preds = model.predict(test[features],raw_score=True)\n    \n    print(\"1=\",preds[:3])\n    test_preds.append(preds)\n\n# Clean Memory\ndel test, model\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:50:25.997426Z","iopub.execute_input":"2022-08-16T16:50:25.998102Z","iopub.status.idle":"2022-08-16T16:56:13.402123Z","shell.execute_reply.started":"2022-08-16T16:50:25.998057Z","shell.execute_reply":"2022-08-16T16:56:13.399332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predictions = np.concatenate(test_preds)\n\nsubmission = pd.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")\nsubmission.loc[:, \"prediction\"] = test_predictions\n\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{},"execution_count":null,"outputs":[]}]}