{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# LOAD LIBRARIES\nimport pandas as pd, numpy as np # CPU libraries\nimport cupy, cudf # GPU libraries\nimport matplotlib.pyplot as plt, gc, os\n\nfrom tsfresh.feature_extraction import feature_calculators as fc\n\nprint('RAPIDS version',cudf.__version__)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-04T14:14:56.983495Z","iopub.execute_input":"2023-05-04T14:14:56.98454Z","iopub.status.idle":"2023-05-04T14:15:04.313756Z","shell.execute_reply.started":"2023-05-04T14:14:56.984492Z","shell.execute_reply":"2023-05-04T14:15:04.312699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# VERSION NAME FOR SAVED MODEL FILES\nVER = 1\n\n# TRAIN RANDOM SEED\nSEED = 42\n\n# FILL NAN VALUE\nNAN_VALUE = -127 # will fit in int8\n\n# FOLDS PER MODEL\nFOLDS = 5","metadata":{"execution":{"iopub.status.busy":"2023-05-04T14:15:04.315603Z","iopub.execute_input":"2023-05-04T14:15:04.31592Z","iopub.status.idle":"2023-05-04T14:15:04.320394Z","shell.execute_reply.started":"2023-05-04T14:15:04.315887Z","shell.execute_reply":"2023-05-04T14:15:04.319376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path = '', usecols = None):\n    # LOAD DATAFRAME\n    if usecols is not None: df = cudf.read_parquet(path, columns=usecols)\n    else: df = cudf.read_parquet(path)\n    # REDUCE DTYPE FOR CUSTOMER AND DATE\n    df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = cudf.to_datetime( df.S_2 )\n    # SORT BY CUSTOMER AND DATE (so agg('last') works correctly)\n    #df = df.sort_values(['customer_ID','S_2'])\n    #df = df.reset_index(drop=True)\n    # FILL NAN\n    df = df.fillna(NAN_VALUE) \n    print('shape of data:', df.shape)\n    \n    return df\n\nprint('Reading train data...')\nTRAIN_PATH = '../input/amex-data-integer-dtypes-parquet-format/train.parquet'\ntrain = read_file(path = TRAIN_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-05-04T14:15:04.322112Z","iopub.execute_input":"2023-05-04T14:15:04.322573Z","iopub.status.idle":"2023-05-04T14:15:22.969492Z","shell.execute_reply.started":"2023-05-04T14:15:04.322536Z","shell.execute_reply":"2023-05-04T14:15:22.968402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train = train[:100]","metadata":{"execution":{"iopub.status.busy":"2023-05-04T14:15:22.972566Z","iopub.execute_input":"2023-05-04T14:15:22.972919Z","iopub.status.idle":"2023-05-04T14:15:22.979031Z","shell.execute_reply.started":"2023-05-04T14:15:22.972885Z","shell.execute_reply":"2023-05-04T14:15:22.978117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_cols0 = [c for c in list(train.columns)]\nall_cols = [c for c in all_cols0 if c not in ['customer_ID','S_2']]\ncat_features = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\nnum_features = [col for col in all_cols if col not in cat_features]\nid_num_features = [col for col in all_cols0 if col not in cat_features]\n\ntsfresh_train = train[id_num_features]\ntsfresh_train = tsfresh_train.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2023-05-04T14:15:22.981123Z","iopub.execute_input":"2023-05-04T14:15:22.982162Z","iopub.status.idle":"2023-05-04T14:15:29.633186Z","shell.execute_reply.started":"2023-05-04T14:15:22.982132Z","shell.execute_reply":"2023-05-04T14:15:29.63219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tsfresh import extract_features\n\nfc_parameters = {\n    \"abs_energy\": None,\n    \"count_above_mean\": None,\n    \"count_below_mean\": None,\n    \"mean_abs_change\": None,\n    \"mean_change\": None\n}\n\n# X = extract_features(tsfresh_train, column_id='customer_ID', column_sort='S_2', default_fc_parameters = fc_parameters)\n# X.index.name = 'customer_ID'","metadata":{"execution":{"iopub.status.busy":"2023-05-04T14:15:29.634413Z","iopub.execute_input":"2023-05-04T14:15:29.634969Z","iopub.status.idle":"2023-05-04T14:15:29.641036Z","shell.execute_reply.started":"2023-05-04T14:15:29.634933Z","shell.execute_reply":"2023-05-04T14:15:29.640094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X = cudf.from_pandas(X)","metadata":{"execution":{"iopub.status.busy":"2023-05-04T14:15:29.642941Z","iopub.execute_input":"2023-05-04T14:15:29.643633Z","iopub.status.idle":"2023-05-04T14:15:29.650404Z","shell.execute_reply.started":"2023-05-04T14:15:29.643601Z","shell.execute_reply":"2023-05-04T14:15:29.649358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_and_feature_engineer(df):\n    # FEATURE ENGINEERING FROM \n    # https://www.kaggle.com/code/huseyincot/amex-agg-data-how-it-created\n    all_cols = [c for c in list(df.columns) if c not in ['customer_ID','S_2']]\n    cat_features = [\"B_30\",\"B_38\",\"D_114\",\"D_116\",\"D_117\",\"D_120\",\"D_126\",\"D_63\",\"D_64\",\"D_66\",\"D_68\"]\n    num_features = [col for col in all_cols if col not in cat_features]\n\n    test_num_agg = df.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n    test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n\n    test_cat_agg = df.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\n    test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n\n    df = cudf.concat([test_num_agg, test_cat_agg], axis=1)\n    del test_num_agg, test_cat_agg\n    print('shape after engineering', df.shape )\n    \n    return df\n\n# train = process_and_feature_engineer(train)","metadata":{"execution":{"iopub.status.busy":"2023-05-04T14:15:29.652241Z","iopub.execute_input":"2023-05-04T14:15:29.652664Z","iopub.status.idle":"2023-05-04T14:15:29.662138Z","shell.execute_reply.started":"2023-05-04T14:15:29.652633Z","shell.execute_reply":"2023-05-04T14:15:29.661048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # ADD TARGETS\n# targets = cudf.read_csv('../input/amex-default-prediction/train_labels.csv')\n# targets['customer_ID'] = targets['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n# targets = targets.set_index('customer_ID')\n# train = train.merge(X, left_index=True, right_index=True, how='left')\n# train = train.merge(targets, left_index=True, right_index=True, how='left')\n# train.target = train.target.astype('int8')\n# del X, targets\n\n# # NEEDED TO MAKE CV DETERMINISTIC (cudf merge above randomly shuffles rows)\n# train = train.sort_index().reset_index()\n# train.to_parquet('train.parquet')\n\n# FEATURES\nFEATURES = train.columns[1:-1]\nprint(f'There are {len(FEATURES)} features!')","metadata":{"execution":{"iopub.status.busy":"2023-05-04T14:15:29.664954Z","iopub.execute_input":"2023-05-04T14:15:29.665896Z","iopub.status.idle":"2023-05-04T14:15:29.677469Z","shell.execute_reply.started":"2023-05-04T14:15:29.665867Z","shell.execute_reply":"2023-05-04T14:15:29.676485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CALCULATE SIZE OF EACH SEPARATE TEST PART\ndef get_rows(customers, test, NUM_PARTS = 4, verbose = ''):\n    chunk = len(customers)//NUM_PARTS\n    if verbose != '':\n        print(f'We will process {verbose} data as {NUM_PARTS} separate parts.')\n        print(f'There will be {chunk} customers in each part (except the last part).')\n        print('Below are number of rows in each part:')\n    rows = []\n\n    for k in range(NUM_PARTS):\n        if k==NUM_PARTS-1: cc = customers[k*chunk:]\n        else: cc = customers[k*chunk:(k+1)*chunk]\n        s = test.loc[test.customer_ID.isin(cc)].shape[0]\n        rows.append(s)\n    if verbose != '': print( rows )\n    return rows,chunk\n\n# COMPUTE SIZE OF 4 PARTS FOR TEST DATA\nNUM_PARTS = 4\nTEST_PATH = '../input/amex-data-integer-dtypes-parquet-format/test.parquet'\n\nprint(f'Reading test data...')\ntest = read_file(path = TEST_PATH, usecols = ['customer_ID','S_2'])\ncustomers = test[['customer_ID']].drop_duplicates().sort_index().values.flatten()\nrows,num_cust = get_rows(customers, test[['customer_ID']], NUM_PARTS = NUM_PARTS, verbose = 'test')","metadata":{"execution":{"iopub.status.busy":"2023-05-04T14:15:29.680343Z","iopub.execute_input":"2023-05-04T14:15:29.680687Z","iopub.status.idle":"2023-05-04T14:15:33.059945Z","shell.execute_reply.started":"2023-05-04T14:15:29.680658Z","shell.execute_reply":"2023-05-04T14:15:33.058988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INFER TEST DATA IN PARTS\nskip_rows = 0\nskip_cust = 0\ntest_preds = []\n\nfor k in range(NUM_PARTS - 2):\n    \n    # READ PART OF TEST DATA\n    print(f'\\nReading test data...')\n    test = read_file(path = TEST_PATH)\n    test = test.iloc[skip_rows:skip_rows+rows[k]]\n    skip_rows += rows[k]\n    print(f'=> Test part {k+1} has shape', test.shape )\n    \n#     test = test[:100]\n    \n    tsfresh_test = test[id_num_features]\n    tsfresh_test = tsfresh_test.to_pandas()\n    X = extract_features(tsfresh_test, column_id='customer_ID', column_sort='S_2', default_fc_parameters = fc_parameters)\n    X.index.name = 'customer_ID'\n    \n    X = cudf.from_pandas(X)\n    \n    # PROCESS AND FEATURE ENGINEER PART OF TEST DATA\n    test = process_and_feature_engineer(test)\n    \n    test = test.merge(X, left_index=True, right_index=True, how='left')\n    del X\n    \n    test = test.sort_index().reset_index()\n    \n    train.to_parquet(f'test{k}.parquet')\n\n    # CLEAN MEMORY\n    del lgtest, model\n    _ = gc.collect()\n    \nprint(skip_rows)\nprint(skip_cust)","metadata":{"execution":{"iopub.status.busy":"2023-05-04T14:15:33.062233Z","iopub.execute_input":"2023-05-04T14:15:33.063157Z","iopub.status.idle":"2023-05-04T14:16:18.344861Z","shell.execute_reply.started":"2023-05-04T14:15:33.063125Z","shell.execute_reply":"2023-05-04T14:16:18.341885Z"},"trusted":true},"execution_count":null,"outputs":[]}]}