{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**By the end of this notebook, you should be able to read the train and test data in ~45 seconds**\n\nThe various steps followed in this kernel are given below:\n\n    1. Read the csv files by explicitly specifying the required column datatype.\n    2. The `customer_ID` column takes a lot of memory. So a `new_customer_id` column is created to reduce memory usage.\n    3. The processed data is saved in feather format.\n    4. For quick future usage, the data is uploaded as private kaggle dataset using the kaggle API","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom pathlib import Path\nimport gc\n\npd.set_option(\"max_rows\", 500)\npd.set_option(\"max_columns\", None)\n\ninput_path = Path('/kaggle/input/amex-default-prediction/')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-23T00:49:31.387028Z","iopub.execute_input":"2022-06-23T00:49:31.387442Z","iopub.status.idle":"2022-06-23T00:49:31.393709Z","shell.execute_reply.started":"2022-06-23T00:49:31.387412Z","shell.execute_reply":"2022-06-23T00:49:31.392801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_DIR = '../working/data/'\n!mkdir -p {DATA_DIR}","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:31.497293Z","iopub.execute_input":"2022-06-23T00:49:31.497994Z","iopub.status.idle":"2022-06-23T00:49:32.307777Z","shell.execute_reply.started":"2022-06-23T00:49:31.497958Z","shell.execute_reply":"2022-06-23T00:49:32.306483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\ncol_dtypes = {'P_2': np.dtype('float16'), 'D_39': np.dtype('float16'), 'B_1': np.dtype('float16'), 'B_2': np.dtype('float16'), 'R_1': np.dtype('float16'), 'S_3': np.dtype('float16'), 'D_41': np.dtype('float16'), 'B_3': np.dtype('float16'), 'D_42': np.dtype('float16'), 'D_43': np.dtype('float16'), 'D_44': np.dtype('float16'), 'B_4': np.dtype('float16'), 'D_45': np.dtype('float16'), 'B_5': np.dtype('float16'), 'R_2': np.dtype('float16'), 'D_46': np.dtype('float16'), 'D_47': np.dtype('float16'), 'D_48': np.dtype('float16'), 'D_49': np.dtype('float16'), 'B_6': np.dtype('float16'), 'B_7': np.dtype('float16'), 'B_8': np.dtype('float16'), 'D_50': np.dtype('float16'), 'D_51': np.dtype('float16'), 'B_9': np.dtype('float16'), 'R_3': np.dtype('float16'), 'D_52': np.dtype('float16'), 'P_3': np.dtype('float16'), 'B_10': np.dtype('float16'), 'D_53': np.dtype('float16'), 'S_5': np.dtype('float16'), 'B_11': np.dtype('float16'), 'S_6': np.dtype('float16'), 'D_54': np.dtype('float16'), 'R_4': np.dtype('float16'), 'S_7': np.dtype('float16'), 'B_12': np.dtype('float16'), 'S_8': np.dtype('float16'), 'D_55': np.dtype('float16'), 'D_56': np.dtype('float16'), 'B_13': np.dtype('float16'), 'R_5': np.dtype('float16'), 'D_58': np.dtype('float16'), 'S_9': np.dtype('float16'), 'B_14': np.dtype('float16'), 'D_59': np.dtype('float16'), 'D_60': np.dtype('float16'), 'D_61': np.dtype('float16'), 'B_15': np.dtype('float16'), 'S_11': np.dtype('float16'), 'D_62': np.dtype('float16'), 'D_65': np.dtype('float16'), 'B_16': np.dtype('float16'), 'B_17': np.dtype('float16'), 'B_18': np.dtype('float16'), 'B_19': np.dtype('float16'), 'B_20': np.dtype('float16'), 'S_12': np.dtype('float16'), 'R_6': np.dtype('float16'), 'S_13': np.dtype('float16'), 'B_21': np.dtype('float16'), 'D_69': np.dtype('float16'), 'B_22': np.dtype('float16'), 'D_70': np.dtype('float16'), 'D_71': np.dtype('float16'), 'D_72': np.dtype('float16'), 'S_15': np.dtype('float16'), 'B_23': np.dtype('float16'), 'D_73': np.dtype('float16'), 'P_4': np.dtype('float16'), 'D_74': np.dtype('float16'), 'D_75': np.dtype('float16'), 'D_76': np.dtype('float16'), 'B_24': np.dtype('float16'), 'R_7': np.dtype('float16'), 'D_77': np.dtype('float16'), 'B_25': np.dtype('float16'), 'B_26': np.dtype('float16'), 'D_78': np.dtype('float16'), 'D_79': np.dtype('float16'), 'R_8': np.dtype('float16'), 'R_9': np.dtype('float16'), 'S_16': np.dtype('float16'), 'D_80': np.dtype('float16'), 'R_10': np.dtype('float16'), 'R_11': np.dtype('float16'), 'B_27': np.dtype('float16'), 'D_81': np.dtype('float16'), 'D_82': np.dtype('float16'), 'S_17': np.dtype('float16'), 'R_12': np.dtype('float16'), 'B_28': np.dtype('float16'), 'R_13': np.dtype('float16'), 'D_83': np.dtype('float16'), 'R_14': np.dtype('float16'), 'R_15': np.dtype('float16'), 'D_84': np.dtype('float16'), 'R_16': np.dtype('float16'), 'B_29': np.dtype('float16'), 'S_18': np.dtype('float16'), 'D_86': np.dtype('float16'), 'D_87': np.dtype('float16'), 'R_17': np.dtype('float16'), 'R_18': np.dtype('float16'), 'D_88': np.dtype('float16'), 'B_31': np.dtype('int8'), 'S_19': np.dtype('float16'), 'R_19': np.dtype('float16'), 'B_32': np.dtype('float16'), 'S_20': np.dtype('float16'), 'R_20': np.dtype('float16'), 'R_21': np.dtype('float16'), 'B_33': np.dtype('float16'), 'D_89': np.dtype('float16'), 'R_22': np.dtype('float16'), 'R_23': np.dtype('float16'), 'D_91': np.dtype('float16'), 'D_92': np.dtype('float16'), 'D_93': np.dtype('float16'), 'D_94': np.dtype('float16'), 'R_24': np.dtype('float16'), 'R_25': np.dtype('float16'), 'D_96': np.dtype('float16'), 'S_22': np.dtype('float16'), 'S_23': np.dtype('float16'), 'S_24': np.dtype('float16'), 'S_25': np.dtype('float16'), 'S_26': np.dtype('float16'), 'D_102': np.dtype('float16'), 'D_103': np.dtype('float16'), 'D_104': np.dtype('float16'), 'D_105': np.dtype('float16'), 'D_106': np.dtype('float16'), 'D_107': np.dtype('float16'), 'B_36': np.dtype('float16'), 'B_37': np.dtype('float16'), 'R_26': np.dtype('float16'), 'R_27': np.dtype('float16'), 'D_108': np.dtype('float16'), 'D_109': np.dtype('float16'), 'D_110': np.dtype('float16'), 'D_111': np.dtype('float16'), 'B_39': np.dtype('float16'), 'D_112': np.dtype('float16'), 'B_40': np.dtype('float16'), 'S_27': np.dtype('float16'), 'D_113': np.dtype('float16'), 'D_115': np.dtype('float16'), 'D_118': np.dtype('float16'), 'D_119': np.dtype('float16'), 'D_121': np.dtype('float16'), 'D_122': np.dtype('float16'), 'D_123': np.dtype('float16'), 'D_124': np.dtype('float16'), 'D_125': np.dtype('float16'), 'D_127': np.dtype('float16'), 'D_128': np.dtype('float16'), 'D_129': np.dtype('float16'), 'B_41': np.dtype('float16'), 'B_42': np.dtype('float16'), 'D_130': np.dtype('float16'), 'D_131': np.dtype('float16'), 'D_132': np.dtype('float16'), 'D_133': np.dtype('float16'), 'R_28': np.dtype('float16'), 'D_134': np.dtype('float16'), 'D_135': np.dtype('float16'), 'D_136': np.dtype('float16'), 'D_137': np.dtype('float16'), 'D_138': np.dtype('float16'), 'D_139': np.dtype('float16'), 'D_140': np.dtype('float16'), 'D_141': np.dtype('float16'), 'D_142': np.dtype('float16'), 'D_143': np.dtype('float16'), 'D_144': np.dtype('float16'), 'D_145': np.dtype('float16')}","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:32.310195Z","iopub.execute_input":"2022-06-23T00:49:32.310850Z","iopub.status.idle":"2022-06-23T00:49:32.353746Z","shell.execute_reply.started":"2022-06-23T00:49:32.310811Z","shell.execute_reply":"2022-06-23T00:49:32.352432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# processing test data","metadata":{"execution":{"iopub.status.busy":"2022-06-22T23:28:01.042999Z","iopub.execute_input":"2022-06-22T23:28:01.043414Z","iopub.status.idle":"2022-06-22T23:28:01.04877Z","shell.execute_reply.started":"2022-06-22T23:28:01.043372Z","shell.execute_reply":"2022-06-22T23:28:01.047813Z"}}},{"cell_type":"code","source":"%%time\n\ntest_submission = pd.read_csv(\n    input_path / 'sample_submission.csv'\n)\nTOTAL_TEST_CUSTOMERS = len(test_submission)","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:32.355850Z","iopub.execute_input":"2022-06-23T00:49:32.356593Z","iopub.status.idle":"2022-06-23T00:49:33.620793Z","shell.execute_reply.started":"2022-06-23T00:49:32.356541Z","shell.execute_reply":"2022-06-23T00:49:33.619488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_customer_id_map = pd.DataFrame({\n    'customer_ID': test_submission.customer_ID.unique().tolist()\n})\ntest_customer_id_map = test_customer_id_map.reset_index().rename(columns={'index':'new_customer_id'})","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:33.623385Z","iopub.execute_input":"2022-06-23T00:49:33.623748Z","iopub.status.idle":"2022-06-23T00:49:34.173569Z","shell.execute_reply.started":"2022-06-23T00:49:33.623717Z","shell.execute_reply":"2022-06-23T00:49:34.172360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_submission = test_customer_id_map.merge(test_submission)\ntest_submission.to_feather(DATA_DIR+'sample_submission.feather')\ntest_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:34.174827Z","iopub.execute_input":"2022-06-23T00:49:34.175205Z","iopub.status.idle":"2022-06-23T00:49:35.445018Z","shell.execute_reply.started":"2022-06-23T00:49:34.175174Z","shell.execute_reply":"2022-06-23T00:49:35.444103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_submission\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:35.446800Z","iopub.execute_input":"2022-06-23T00:49:35.447610Z","iopub.status.idle":"2022-06-23T00:49:36.643506Z","shell.execute_reply.started":"2022-06-23T00:49:35.447566Z","shell.execute_reply":"2022-06-23T00:49:36.642086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntest_tmp = pd.read_csv(\n    input_path / 'test_data.csv',\n    dtype=col_dtypes,\n    chunksize=100000\n)\n\ntest_data = pd.DataFrame()\nfor itr, row in enumerate(test_tmp):\n    row = test_customer_id_map.merge(row)\n    row.drop('customer_ID', axis=1, inplace=True)\n    \n    test_data = pd.concat((test_data, row), axis=0, sort=False, ignore_index=True)\n    print(f\"completed itr # {itr}\")\n\ndel test_tmp\ngc.collect()\n\nprint(test_data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:36.645264Z","iopub.execute_input":"2022-06-23T00:49:36.646127Z","iopub.status.idle":"2022-06-23T00:49:43.245845Z","shell.execute_reply.started":"2022-06-23T00:49:36.646081Z","shell.execute_reply":"2022-06-23T00:49:43.244688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data[cat_cols] = test_data[cat_cols].astype('object')\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:43.247759Z","iopub.execute_input":"2022-06-23T00:49:43.248129Z","iopub.status.idle":"2022-06-23T00:49:43.487574Z","shell.execute_reply.started":"2022-06-23T00:49:43.248098Z","shell.execute_reply":"2022-06-23T00:49:43.486370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# saving all the test data\ntest_customer_id_map.to_feather(DATA_DIR+'test_customer_id_map.feather')\ntest_data.to_feather(DATA_DIR+\"test_data.feather\")","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:43.489617Z","iopub.execute_input":"2022-06-23T00:49:43.490019Z","iopub.status.idle":"2022-06-23T00:49:44.076813Z","shell.execute_reply.started":"2022-06-23T00:49:43.489971Z","shell.execute_reply":"2022-06-23T00:49:44.075469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_data, test_customer_id_map\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:44.082559Z","iopub.execute_input":"2022-06-23T00:49:44.082949Z","iopub.status.idle":"2022-06-23T00:49:44.396958Z","shell.execute_reply.started":"2022-06-23T00:49:44.082916Z","shell.execute_reply":"2022-06-23T00:49:44.395395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# processing train data","metadata":{}},{"cell_type":"code","source":"train_labels = pd.read_csv(input_path / 'train_labels.csv', dtype={'target': np.dtype('int8')})","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:44.398739Z","iopub.execute_input":"2022-06-23T00:49:44.399150Z","iopub.status.idle":"2022-06-23T00:49:44.963722Z","shell.execute_reply.started":"2022-06-23T00:49:44.399106Z","shell.execute_reply":"2022-06-23T00:49:44.962405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_customer_id_map = pd.DataFrame({\n    'customer_ID': train_labels.customer_ID.unique().tolist()\n})\ntrain_customer_id_map = train_customer_id_map.reset_index().rename(columns={'index':'new_customer_id'})\ntrain_customer_id_map['new_customer_id'] = TOTAL_TEST_CUSTOMERS + train_customer_id_map['new_customer_id']","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:44.965602Z","iopub.execute_input":"2022-06-23T00:49:44.965926Z","iopub.status.idle":"2022-06-23T00:49:45.272569Z","shell.execute_reply.started":"2022-06-23T00:49:44.965897Z","shell.execute_reply":"2022-06-23T00:49:45.271312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = train_customer_id_map.merge(train_labels)\ntrain_labels.drop('customer_ID', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:45.274259Z","iopub.execute_input":"2022-06-23T00:49:45.275293Z","iopub.status.idle":"2022-06-23T00:49:45.680692Z","shell.execute_reply.started":"2022-06-23T00:49:45.275257Z","shell.execute_reply":"2022-06-23T00:49:45.679598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntrain_tmp = pd.read_csv(\n    input_path / 'train_data.csv',\n    dtype=col_dtypes,\n    chunksize=100000\n)\n\ntrain_data = pd.DataFrame()\nfor itr, row in enumerate(train_tmp):\n    row = train_customer_id_map.merge(row)\n    row.drop('customer_ID', axis=1, inplace=True)\n    \n    train_data = pd.concat((train_data, row), axis=0, sort=False, ignore_index=True)\n    print(f\"completed itr # {itr}\")\n\ndel train_tmp\ngc.collect()\n\nprint(train_data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:45.682140Z","iopub.execute_input":"2022-06-23T00:49:45.682535Z","iopub.status.idle":"2022-06-23T00:49:51.910929Z","shell.execute_reply.started":"2022-06-23T00:49:45.682503Z","shell.execute_reply":"2022-06-23T00:49:51.909794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[cat_cols] = train_data[cat_cols].astype('object')\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:51.912588Z","iopub.execute_input":"2022-06-23T00:49:51.912934Z","iopub.status.idle":"2022-06-23T00:49:52.165487Z","shell.execute_reply.started":"2022-06-23T00:49:51.912901Z","shell.execute_reply":"2022-06-23T00:49:52.164133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# saving all the training data\n\ntrain_data.to_feather(DATA_DIR+\"train_data.feather\")\ntrain_customer_id_map.to_feather(DATA_DIR+'train_customer_id_map.feather')\ntrain_labels.to_feather(DATA_DIR+\"train_labels.feather\")","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:52.167607Z","iopub.execute_input":"2022-06-23T00:49:52.168105Z","iopub.status.idle":"2022-06-23T00:49:52.568691Z","shell.execute_reply.started":"2022-06-23T00:49:52.168069Z","shell.execute_reply":"2022-06-23T00:49:52.567791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_data, train_customer_id_map, train_labels\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:52.570288Z","iopub.execute_input":"2022-06-23T00:49:52.571425Z","iopub.status.idle":"2022-06-23T00:49:52.894953Z","shell.execute_reply.started":"2022-06-23T00:49:52.571381Z","shell.execute_reply":"2022-06-23T00:49:52.894065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### saving the dataset as private kaggle dataset for quickly loading them next time","metadata":{}},{"cell_type":"code","source":"# In order to use the Kaggle’s public API, you must first authenticate using an API token. From the site header, click on your user profile picture, then on “Account” from the dropdown menu. \n# This will take you to your account settings at https://www.kaggle.com/account. Scroll down to the section of the page labelled API:\n# To create a new token, click on the “Create New API Token” button. This will download a fresh authentication token onto your machine.\n\n# Open the kaggle.json file and replace the USER_ID and USER_SECRET key accordingly\n# Create a JSON file containing user-specific metadata. \nUSER_ID = 'xxxxx' # REPLACE WITH YOUR OWN USER NAME\nUSER_SECRET = 'xxxxxxxxxxxxxx' # REPLACE WITH YOUR OWN PRIVATE API TOKEN \n\nimport os, json, nbformat, pandas as pd\nKAGGLE_CONFIG_DIR = os.path.join(os.path.expandvars('$HOME'), '.kaggle')\nos.makedirs(KAGGLE_CONFIG_DIR, exist_ok = True)\nwith open(os.path.join(KAGGLE_CONFIG_DIR, 'kaggle.json'), 'w') as f:\n    json.dump({'username': USER_ID, 'key': USER_SECRET}, f)\n!chmod 600 {KAGGLE_CONFIG_DIR}/kaggle.json","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:52.896470Z","iopub.execute_input":"2022-06-23T00:49:52.897507Z","iopub.status.idle":"2022-06-23T00:49:53.719254Z","shell.execute_reply.started":"2022-06-23T00:49:52.897471Z","shell.execute_reply":"2022-06-23T00:49:53.717545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### uncomment all the below cells and run them successfully to create your own private dataset","metadata":{}},{"cell_type":"code","source":"# !kaggle datasets init -p {DATA_DIR}","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:53.721370Z","iopub.execute_input":"2022-06-23T00:49:53.721780Z","iopub.status.idle":"2022-06-23T00:49:53.727329Z","shell.execute_reply.started":"2022-06-23T00:49:53.721736Z","shell.execute_reply":"2022-06-23T00:49:53.725856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Updating the metadata file to have a proper `title` and `id` for future reference","metadata":{}},{"cell_type":"code","source":"# with open('../working/data/dataset-metadata.json', 'r') as f:\n#     metadata = json.load(f)\n#     metadata['title'] = 'Amex Competition 2022 Dataset'\n#     metadata['id'] = f'{USER_ID}/amex-comp-2022'\n\n# with open('../working/data/dataset-metadata.json', 'w') as f:\n#     json.dump(metadata, f)","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:53.729462Z","iopub.execute_input":"2022-06-23T00:49:53.729917Z","iopub.status.idle":"2022-06-23T00:49:53.738849Z","shell.execute_reply.started":"2022-06-23T00:49:53.729873Z","shell.execute_reply":"2022-06-23T00:49:53.737868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !kaggle datasets create -p {DATA_DIR}","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:53.740928Z","iopub.execute_input":"2022-06-23T00:49:53.741897Z","iopub.status.idle":"2022-06-23T00:49:53.752788Z","shell.execute_reply.started":"2022-06-23T00:49:53.741844Z","shell.execute_reply":"2022-06-23T00:49:53.751557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"NOTE: \n* We can always create multiple intermediate processed files and upload them to the same dataset. This will avoid the pain of unnecessary long runs. \n* We can directly start consuming the datasets on a newly launched kernel by adding the dataset on the top right corner of the kernel page.\n* If this is followed effectively, we can successful create a pipeline of connected notebooks","metadata":{}},{"cell_type":"markdown","source":"### Add the above created data to the existing kernel and run the below cell. The cell will be executed in ~45 seconds!!!","metadata":{}},{"cell_type":"code","source":"# %%time\n\n# train_data = pd.read_feather('../input/amex-comp-2022/train_data.feather')\n# train_labels = pd.read_feather('../input/amex-comp-2022/train_labels.feather')\n# test_data = pd.read_feather('../input/amex-comp-2022/test_data.feather')\n\n# print(train_data.shape, test_data.shape)\n\n##############################\n## Output\n##############################\n## (5531451, 190) (11263762, 190)\n## CPU times: user 26.3 s, sys: 14.6 s, total: 40.9 s\n## Wall time: 45.9 s\"","metadata":{"execution":{"iopub.status.busy":"2022-06-23T00:49:53.754439Z","iopub.execute_input":"2022-06-23T00:49:53.755150Z","iopub.status.idle":"2022-06-23T00:49:53.763980Z","shell.execute_reply.started":"2022-06-23T00:49:53.755118Z","shell.execute_reply":"2022-06-23T00:49:53.763098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}