{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cupy, cudf # GPU libraries\nimport matplotlib.pyplot as plt, gc, os\n\nprint('RAPIDS version',cudf.__version__)\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-29T12:21:28.532178Z","iopub.execute_input":"2022-06-29T12:21:28.533410Z","iopub.status.idle":"2022-06-29T12:21:33.685896Z","shell.execute_reply.started":"2022-06-29T12:21:28.533293Z","shell.execute_reply":"2022-06-29T12:21:33.684844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FILL NAN VALUE\nNAN_VALUE = -127 # will fit in int8\n\ndef read_file(path = '', usecols = None):\n    # LOAD DATAFRAME\n    if usecols is not None: df = cudf.read_parquet(path, columns=usecols)\n    else: df = cudf.read_parquet(path)\n    # REDUCE DTYPE FOR CUSTOMER AND DATE\n    df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = cudf.to_datetime( df.S_2 )\n    # SORT BY CUSTOMER AND DATE (so agg('last') works correctly)\n    #df = df.sort_values(['customer_ID','S_2'])\n    #df = df.reset_index(drop=True)\n    # FILL NAN\n    df = df.fillna(NAN_VALUE)\n    print('shape of data:', df.shape)\n    \n    return df\n\nprint('Reading train data...')\nTRAIN_PATH = '../input/amex-data-integer-dtypes-parquet-format/train.parquet'\nX_train = read_file(path = TRAIN_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T12:21:33.688294Z","iopub.execute_input":"2022-06-29T12:21:33.689025Z","iopub.status.idle":"2022-06-29T12:21:55.897108Z","shell.execute_reply.started":"2022-06-29T12:21:33.688980Z","shell.execute_reply":"2022-06-29T12:21:55.895956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = cudf.read_csv('../input/amex-default-prediction/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-29T12:21:55.898920Z","iopub.execute_input":"2022-06-29T12:21:55.899518Z","iopub.status.idle":"2022-06-29T12:21:56.301651Z","shell.execute_reply.started":"2022-06-29T12:21:55.899475Z","shell.execute_reply":"2022-06-29T12:21:56.300658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T12:21:56.305251Z","iopub.execute_input":"2022-06-29T12:21:56.306523Z","iopub.status.idle":"2022-06-29T12:21:56.556382Z","shell.execute_reply.started":"2022-06-29T12:21:56.306479Z","shell.execute_reply":"2022-06-29T12:21:56.554175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_cols = [c for c in list(X_train.columns) if c not in ['customer_ID','S_2']]\ncat_cols = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\nnum_cols = [col for col in all_cols if col not in cat_cols]","metadata":{"execution":{"iopub.status.busy":"2022-06-29T12:21:56.558063Z","iopub.execute_input":"2022-06-29T12:21:56.559583Z","iopub.status.idle":"2022-06-29T12:21:56.566387Z","shell.execute_reply.started":"2022-06-29T12:21:56.559541Z","shell.execute_reply":"2022-06-29T12:21:56.565308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"delinquency_cols = [col for col in list(X_train.columns) if 'D_' in col]\nspend_cols = [col for col in list(X_train.columns) if 'S_' in col]\npayment_cols = [col for col in list(X_train.columns) if 'P_' in col]\nbalance_cols = [col for col in list(X_train.columns) if 'B_' in col]\nrisk_cols = [col for col in list(X_train.columns) if 'R_' in col]","metadata":{"execution":{"iopub.status.busy":"2022-06-29T12:21:56.568421Z","iopub.execute_input":"2022-06-29T12:21:56.568822Z","iopub.status.idle":"2022-06-29T12:21:56.583694Z","shell.execute_reply.started":"2022-06-29T12:21:56.568784Z","shell.execute_reply":"2022-06-29T12:21:56.582533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d_cat_matches = [x for x in delinquency_cols if x in cat_cols]\n# X_train[d_cat_matches]\n\nb_cat_matches = [x for x in balance_cols if x in cat_cols]\n# X_train[b_cat_matches]","metadata":{"execution":{"iopub.status.busy":"2022-06-29T12:21:56.585775Z","iopub.execute_input":"2022-06-29T12:21:56.586408Z","iopub.status.idle":"2022-06-29T12:21:56.594412Z","shell.execute_reply.started":"2022-06-29T12:21:56.586368Z","shell.execute_reply":"2022-06-29T12:21:56.593536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nSince CUDF DataFrame doesn't seem to have a `value_counts()` attribute (even though the documentation says\notherwise..), we'll have to implement the algorithm on our own.\n\nCreate a value counts based on customer id for the delinquency categorical variables.\n\n'''\nval_cnt = {}\ntmp = list(X_train.groupby('customer_ID')[d_cat_matches[0]].agg('unique').index.to_pandas())\nfor cst_id in tmp:\n    val_cnt[str(int(cst_id))] = {}\n    if j != 0 and j % 100000 == 0:\n        print(f'column {d_col} processed, {j+1} of {len(d_cat_matches)} columns processed.')\n\nfor (j, d_col) in enumerate(d_cat_matches):\n    tmp = X_train.groupby('customer_ID')[d_col].agg('unique')\n    for i in range(len(tmp)):\n        cst_id = tmp.index[i]\n        cnt = Counter(tmp.iloc[i])\n        val_cnt[str(int(cst_id))][str(d_col)] = dict(cnt)\n        if i % 100000 == 0 and i != 0:\n            print(f'row {i} in column name {d_col}, {j+1} of {len(d_cat_matches)} columns processed.')\n        ","metadata":{"execution":{"iopub.status.busy":"2022-06-29T13:17:18.415964Z","iopub.execute_input":"2022-06-29T13:17:18.416346Z","iopub.status.idle":"2022-06-29T13:51:16.100663Z","shell.execute_reply.started":"2022-06-29T13:17:18.416315Z","shell.execute_reply":"2022-06-29T13:51:16.099522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\n# ...\nwith open('/kaggle/working/d_cat_val_cnts.pickle', 'wb') as f:\n    # Pickle the 'data' dictionary using the highest protocol available.\n    pickle.dump(val_cnt, f, pickle.HIGHEST_PROTOCOL)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T13:54:35.016009Z","iopub.execute_input":"2022-06-29T13:54:35.016388Z","iopub.status.idle":"2022-06-29T13:54:38.378767Z","shell.execute_reply.started":"2022-06-29T13:54:35.016357Z","shell.execute_reply":"2022-06-29T13:54:38.377744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nSince CUDF DataFrame doesn't seem to have a `value_counts()` attribute (even though the documentation says\notherwise..), we'll have to implement the algorithm on our own.\n\nCreate a value counts based on customer id for the balance categorical variables.\n\n'''\nval_cnt = {}\ntmp = list(X_train.groupby('customer_ID')[b_cat_matches[0]].agg('unique').index.to_pandas())\nfor cst_id in tmp:\n    val_cnt[str(int(cst_id))] = {}\n    if j != 0 and j % 100000 == 0:\n        print(f'column {d_col} processed, {j+1} of {len(d_cat_matches)} columns processed.')\n\nfor (j, b_col) in enumerate(b_cat_matches):\n    tmp = X_train.groupby('customer_ID')[b_col].agg('unique')\n    for i in range(len(tmp)):\n        cst_id = tmp.index[i]\n        cnt = Counter(tmp.iloc[i])\n        val_cnt[str(int(cst_id))][str(b_col)] = dict(cnt)\n        if i % 100000 == 0 and i != 0:\n            print(f'row {i} in column name {b_col}, {j+1} of {len(b_cat_matches)} columns processed.')\n        ","metadata":{"execution":{"iopub.status.busy":"2022-06-29T13:57:24.415513Z","iopub.execute_input":"2022-06-29T13:57:24.415872Z","iopub.status.idle":"2022-06-29T14:04:55.529995Z","shell.execute_reply.started":"2022-06-29T13:57:24.415843Z","shell.execute_reply":"2022-06-29T14:04:55.529001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\n# ...\nwith open('/kaggle/working/b_cat_val_cnts.pickle', 'wb') as f:\n    # Pickle the 'data' dictionary using the highest protocol available.\n    pickle.dump(val_cnt, f, pickle.HIGHEST_PROTOCOL)","metadata":{"execution":{"iopub.status.busy":"2022-06-29T14:27:38.333341Z","iopub.execute_input":"2022-06-29T14:27:38.334412Z","iopub.status.idle":"2022-06-29T14:27:39.406610Z","shell.execute_reply.started":"2022-06-29T14:27:38.334365Z","shell.execute_reply":"2022-06-29T14:27:39.405324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_cnt","metadata":{"execution":{"iopub.status.busy":"2022-06-29T14:27:49.591096Z","iopub.execute_input":"2022-06-29T14:27:49.591779Z","iopub.status.idle":"2022-06-29T14:27:49.796018Z","shell.execute_reply.started":"2022-06-29T14:27:49.591739Z","shell.execute_reply":"2022-06-29T14:27:49.794884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_cnt[str(int(cst_id))][0]","metadata":{"execution":{"iopub.status.busy":"2022-06-29T13:11:09.816046Z","iopub.execute_input":"2022-06-29T13:11:09.816482Z","iopub.status.idle":"2022-06-29T13:11:09.823914Z","shell.execute_reply.started":"2022-06-29T13:11:09.816444Z","shell.execute_reply":"2022-06-29T13:11:09.822759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}