{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.simplefilter(action='ignore', category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T10:49:23.992165Z","iopub.execute_input":"2022-07-03T10:49:23.992609Z","iopub.status.idle":"2022-07-03T10:49:24.024668Z","shell.execute_reply.started":"2022-07-03T10:49:23.992529Z","shell.execute_reply":"2022-07-03T10:49:24.023784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nimport cudf\nimport pickle\nimport numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2022-07-03T10:49:24.520828Z","iopub.execute_input":"2022-07-03T10:49:24.521478Z","iopub.status.idle":"2022-07-03T10:49:28.909140Z","shell.execute_reply.started":"2022-07-03T10:49:24.521432Z","shell.execute_reply":"2022-07-03T10:49:28.908111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_data(df, is_train=True):\n    \n    all_cols = [c for c in list(df.columns) if c not in ['customer_ID','S_2']]\n    cat_features = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n    num_features = [col for col in all_cols if col not in cat_features]\n\n    test_num_agg = df.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n    num_feat = ['_'.join(x) for x in test_num_agg.columns]\n    test_num_agg.columns = num_feat\n\n    test_cat_agg = df.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\n    cat_feat = ['_'.join(x) for x in test_cat_agg.columns]\n    test_cat_agg.columns = cat_feat\n    \n    df = cudf.concat([test_num_agg, test_cat_agg], axis=1)\n    \n    del test_num_agg, test_cat_agg\n    _ = gc.collect()\n    \n    if is_train:\n        targets_df = cudf.read_csv(f'../input/amex-default-prediction/train_labels.csv')\n        targets_df = targets_df.set_index('customer_ID')\n\n        df = df.merge(targets_df, left_index=True, right_index=True, how='left')\n\n        df.target = df.target.astype('int8')\n        \n        del targets_df\n        _ = gc.collect()\n\n    df = df.sort_index().reset_index()\n    \n    print('shape after engineering', df.shape )\n    \n    return df, num_feat, cat_feat","metadata":{"execution":{"iopub.status.busy":"2022-07-03T10:49:28.911368Z","iopub.execute_input":"2022-07-03T10:49:28.911859Z","iopub.status.idle":"2022-07-03T10:49:28.924434Z","shell.execute_reply.started":"2022-07-03T10:49:28.911815Z","shell.execute_reply":"2022-07-03T10:49:28.922828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH_TO_DATA='../input/amex-data-integer-dtypes-parquet-format'","metadata":{"execution":{"iopub.status.busy":"2022-07-03T10:49:28.926269Z","iopub.execute_input":"2022-07-03T10:49:28.927182Z","iopub.status.idle":"2022-07-03T10:49:28.939027Z","shell.execute_reply.started":"2022-07-03T10:49:28.927121Z","shell.execute_reply":"2022-07-03T10:49:28.937985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df, num_feat, cat_feat = load_data(cudf.read_parquet(f'{PATH_TO_DATA}/train.parquet'))","metadata":{"execution":{"iopub.status.busy":"2022-07-03T10:49:28.941534Z","iopub.execute_input":"2022-07-03T10:49:28.942169Z","iopub.status.idle":"2022-07-03T10:49:55.260284Z","shell.execute_reply.started":"2022-07-03T10:49:28.942108Z","shell.execute_reply":"2022-07-03T10:49:55.258930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Save to disk\ntrain_df.to_parquet('train_fe.parquet', index=False)\n\nwith open('num_feat.pkl', 'wb') as f:\n    pickle.dump(num_feat, f)\n\nwith open('cat_feat.pkl', 'wb') as f:\n    pickle.dump(cat_feat, f)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T10:50:06.157292Z","iopub.execute_input":"2022-07-03T10:50:06.157699Z","iopub.status.idle":"2022-07-03T10:50:14.209230Z","shell.execute_reply.started":"2022-07-03T10:50:06.157668Z","shell.execute_reply":"2022-07-03T10:50:14.207726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  CALCULATE SIZE OF EACH SEPARATE TEST PART\ndef get_rows(customers, test, NUM_PARTS = 4, verbose = ''):\n    chunk = len(customers)//NUM_PARTS\n    if verbose != '':\n        print(f'We will process {verbose} data as {NUM_PARTS} separate parts.')\n        print(f'There will be {chunk} customers in each part (except the last part).')\n        print('Below are number of rows in each part:')\n    rows = []\n\n    for k in range(NUM_PARTS):\n        if k==NUM_PARTS-1: cc = customers[k*chunk:]\n        else: cc = customers[k*chunk:(k+1)*chunk]\n        s = test.loc[test.customer_ID.isin(cc)].shape[0]\n        rows.append(s)\n    if verbose != '': print( rows )\n    return rows,chunk","metadata":{"execution":{"iopub.status.busy":"2022-07-03T10:50:18.459132Z","iopub.execute_input":"2022-07-03T10:50:18.459576Z","iopub.status.idle":"2022-07-03T10:50:18.468581Z","shell.execute_reply.started":"2022-07-03T10:50:18.459545Z","shell.execute_reply":"2022-07-03T10:50:18.467404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# COMPUTE SIZE OF 4 PARTS FOR TEST DATA\nNUM_PARTS = 8\nTEST_PATH = f'{PATH_TO_DATA}/test.parquet'\n\nprint(f'Reading test data...')\ntest = pd.read_parquet(TEST_PATH, columns = ['customer_ID','S_2'])\ncustomers = test['customer_ID'].unique().tolist()\nrows, num_cust = get_rows(customers, test[['customer_ID']], NUM_PARTS = NUM_PARTS, verbose = 'test')\ndel test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T10:50:23.770547Z","iopub.execute_input":"2022-07-03T10:50:23.771060Z","iopub.status.idle":"2022-07-03T10:50:37.732684Z","shell.execute_reply.started":"2022-07-03T10:50:23.771030Z","shell.execute_reply":"2022-07-03T10:50:37.731569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skip_rows = 0\n# skip_cust = 0\n\nfor k in range(NUM_PARTS):\n    \n    # READ PART OF TEST DATA\n    print(f'\\nReading test data...')\n    test = cudf.read_parquet(TEST_PATH)\n    test = test.iloc[skip_rows:skip_rows+rows[k]]\n    skip_rows += rows[k]\n    print(f'=> Test part {k+1} has shape', test.shape )\n    \n    # PROCESS AND FEATURE ENGINEER PART OF TEST DATA\n    test, _, _ = load_data(test)\n#     if k==NUM_PARTS-1: test = test.loc[customers[skip_cust:]]\n#     else: test = test.loc[customers[skip_cust:skip_cust+num_cust]]\n#     skip_cust += num_cust\n    \n    test.to_parquet(f'test_fe_{k+1}.parquet', index=False)\n\n    # CLEAN MEMORY\n    del test\n    _ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T10:55:43.752020Z","iopub.execute_input":"2022-07-03T10:55:43.752435Z","iopub.status.idle":"2022-07-03T10:56:44.126775Z","shell.execute_reply.started":"2022-07-03T10:55:43.752404Z","shell.execute_reply":"2022-07-03T10:56:44.125740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(num_feat)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T10:56:44.129581Z","iopub.execute_input":"2022-07-03T10:56:44.130054Z","iopub.status.idle":"2022-07-03T10:56:44.135396Z","shell.execute_reply.started":"2022-07-03T10:56:44.129997Z","shell.execute_reply":"2022-07-03T10:56:44.134391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(cat_feat)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T10:56:44.137020Z","iopub.execute_input":"2022-07-03T10:56:44.137922Z","iopub.status.idle":"2022-07-03T10:56:44.151312Z","shell.execute_reply.started":"2022-07-03T10:56:44.137834Z","shell.execute_reply":"2022-07-03T10:56:44.150132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}