{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook describes how to read data from a dataset that is stored in parquet format with data frames generated for each chunk.  \nParquet files here: https://www.kaggle.com/datasets/columbia2131/otto-chunk-data-inparquet-format","metadata":{}},{"cell_type":"markdown","source":"# Data set storage and loading speed\n\nThe following shows how to save paqruet files from jsonl　files and how fast it can be loaded in paqquet files.","metadata":{}},{"cell_type":"markdown","source":"### Save patquet files","metadata":{}},{"cell_type":"code","source":"import os \n\nimport numpy as np\nimport pandas as pd\n\nfrom pathlib import Path\nfrom glob import glob\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-11-02T04:40:17.429972Z","iopub.execute_input":"2022-11-02T04:40:17.430457Z","iopub.status.idle":"2022-11-02T04:40:17.436470Z","shell.execute_reply.started":"2022-11-02T04:40:17.430400Z","shell.execute_reply":"2022-11-02T04:40:17.435154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndata_path = Path('/kaggle/input/otto-recommender-system/')\nchunksize = 100_000\nsave = False\n\nchunks = pd.read_json(data_path / 'train.jsonl', lines=True, chunksize=chunksize)\nos.mkdir('train_parquet')\n\nfor e, chunk in enumerate(tqdm(chunks, total=129)):\n    event_dict = {\n        'session': [],\n        'aid': [],\n        'ts': [],\n        'type': [],\n    }\n    \n    for session, events in zip(chunk['session'].tolist(), chunk['events'].tolist()):\n        for event in events:\n            event_dict['session'].append(session)\n            event_dict['aid'].append(event['aid'])\n            event_dict['ts'].append(event['ts'])\n            event_dict['type'].append(event['type'])\n    \n    # save DataFrame\n    start = str(e*chunksize).zfill(9)\n    end = str(e*chunksize+chunksize).zfill(9)\n    event_df = pd.DataFrame(event_dict)\n    if save == True:\n        event_df.to_parquet(f\"train_parquet/{start}_{end}.parquet\")\n    \n    break","metadata":{"execution":{"iopub.status.busy":"2022-11-02T04:40:22.212249Z","iopub.execute_input":"2022-11-02T04:40:22.212666Z","iopub.status.idle":"2022-11-02T04:40:42.941751Z","shell.execute_reply.started":"2022-11-02T04:40:22.212632Z","shell.execute_reply":"2022-11-02T04:40:42.940631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load parquet files\n\nIt takes about 20 seconds per chunk to create a DataFrame from jsonl, while loadling a parquet file takes 0.5 to 2.0 seconds! ","metadata":{}},{"cell_type":"code","source":"%%time\nevent_df = pd.read_parquet('../input/otto-chunk-data-inparquet-format/train_parquet/000000000_000100000.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T04:41:17.626713Z","iopub.execute_input":"2022-11-02T04:41:17.627245Z","iopub.status.idle":"2022-11-02T04:41:18.162152Z","shell.execute_reply.started":"2022-11-02T04:41:17.627198Z","shell.execute_reply":"2022-11-02T04:41:18.161321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And, we can load the targeted files.","metadata":{}},{"cell_type":"code","source":"for path in sorted(glob('../input/otto-chunk-data-inparquet-format/train_parquet/*')):\n    print(os.path.basename(path))","metadata":{"execution":{"iopub.status.busy":"2022-11-02T04:38:31.685293Z","iopub.execute_input":"2022-11-02T04:38:31.686412Z","iopub.status.idle":"2022-11-02T04:38:31.694200Z","shell.execute_reply.started":"2022-11-02T04:38:31.686367Z","shell.execute_reply":"2022-11-02T04:38:31.693061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfiles = sorted(glob('../input/otto-chunk-data-inparquet-format/train_parquet/*'))[0:5]\ndfs = []\n\nfor path in files:\n    dfs.append(pd.read_parquet(path))\n\ndfs = pd.concat(dfs).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T04:52:14.341241Z","iopub.execute_input":"2022-11-02T04:52:14.342485Z","iopub.status.idle":"2022-11-02T04:52:19.061018Z","shell.execute_reply.started":"2022-11-02T04:52:14.342441Z","shell.execute_reply":"2022-11-02T04:52:19.060134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(dfs)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T04:52:20.337363Z","iopub.execute_input":"2022-11-02T04:52:20.337971Z","iopub.status.idle":"2022-11-02T04:52:20.357027Z","shell.execute_reply.started":"2022-11-02T04:52:20.337936Z","shell.execute_reply":"2022-11-02T04:52:20.356063Z"},"trusted":true},"execution_count":null,"outputs":[]}]}