{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div align=\"center\"><p style=\"font-family: 'Mochiy Pop P One';font-size:32px; font-weight: bold; color:black\">How to read a big JSON file with Python? </p></div>\n\n\n<div align=\"left\" style=\"font-family: 'Mochiy Pop P One';font-size:16px; font-weight: normal; \">\n    \n\nThis notebook compares the two methods of reading large JSON files in Python:    \n<br>     \n\nMethod 1: Using huggingface <a href=\"https://pypi.org/project/datasets/\">datasets package</a> (already in Kaggle enviroment and no need to install). This method brought up by <a href=\"https://www.kaggle.com/daquarti\">Daquarti Gustavo</a> in <a href=\"https://www.kaggle.com/code/daquarti/reading-jsonl-data-using-huggingface-datasets\">Reading jsonl data using HuggingFace Datasets</a>\n<br>   \nMethod 2: Using pandas read_json function and read file into chunks. This method is detailed in <a href=\"https://www.kaggle.com/edwardcrookenden\">Edward Crookenden</a>'s notebook <a href=\"https://www.kaggle.com/code/edwardcrookenden/otto-getting-started-eda-baseline\">OTTO - Getting Started (EDA + Baseline)</a>  \n<br>    \n<br>     \n\n<mark>Method 1 is superior to Method 2 in terms of conveniences in accessing anything sample and use of memory.  </mark>\n<br>    \n\n</div>   \n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport gc\nimport copy\nimport json\nimport sys\nfrom pathlib import Path\nfrom datetime import datetime, timedelta, date\nimport time\nfrom dateutil.relativedelta import relativedelta \n\nimport pyarrow.parquet as pq\nimport pyarrow as pa\n\nfrom tqdm import tqdm\n\npd.options.display.max_rows = 100\npd.options.display.max_columns = 100\n\npd.options.display.float_format = '{:,.2f}'.format\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport pytorch_lightning as pl\nrandom_seed=1\npl.seed_everything(random_seed)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T10:17:58.224204Z","iopub.execute_input":"2022-11-02T10:17:58.225106Z","iopub.status.idle":"2022-11-02T10:18:03.593506Z","shell.execute_reply.started":"2022-11-02T10:17:58.225009Z","shell.execute_reply":"2022-11-02T10:18:03.591995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div align=\"center\"><p style=\"font-family: 'Mochiy Pop P One';font-size:20px; font-weight: bold; color:black\">Method 1: use datasets package </p></div>\n\n","metadata":{}},{"cell_type":"code","source":"from datasets import load_dataset","metadata":{"execution":{"iopub.status.busy":"2022-11-02T10:18:05.966166Z","iopub.execute_input":"2022-11-02T10:18:05.967472Z","iopub.status.idle":"2022-11-02T10:18:06.302061Z","shell.execute_reply.started":"2022-11-02T10:18:05.967404Z","shell.execute_reply":"2022-11-02T10:18:06.300899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_file = \"../input/otto-recommender-system/train.jsonl\"","metadata":{"execution":{"iopub.status.busy":"2022-11-02T10:18:07.224783Z","iopub.execute_input":"2022-11-02T10:18:07.225925Z","iopub.status.idle":"2022-11-02T10:18:07.230108Z","shell.execute_reply.started":"2022-11-02T10:18:07.225877Z","shell.execute_reply":"2022-11-02T10:18:07.229261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndataset = load_dataset(\"json\", data_files= str(train_file))","metadata":{"execution":{"iopub.status.busy":"2022-11-02T10:18:08.246683Z","iopub.execute_input":"2022-11-02T10:18:08.247954Z","iopub.status.idle":"2022-11-02T10:21:44.597319Z","shell.execute_reply.started":"2022-11-02T10:18:08.247899Z","shell.execute_reply":"2022-11-02T10:21:44.595943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.shape","metadata":{"execution":{"iopub.status.busy":"2022-11-02T10:21:44.599839Z","iopub.execute_input":"2022-11-02T10:21:44.600898Z","iopub.status.idle":"2022-11-02T10:21:44.607615Z","shell.execute_reply.started":"2022-11-02T10:21:44.600855Z","shell.execute_reply":"2022-11-02T10:21:44.606651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_size = dataset.shape['train'][0]\nprint(f'Num. of samples in train dataset: {train_size:,}')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T10:21:44.608655Z","iopub.execute_input":"2022-11-02T10:21:44.609369Z","iopub.status.idle":"2022-11-02T10:21:44.618981Z","shell.execute_reply.started":"2022-11-02T10:21:44.609333Z","shell.execute_reply":"2022-11-02T10:21:44.617542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsession_stats = []\nfor i in range(100000):\n    cnt = len(dataset['train'][i]['events'])\n    session_stats.append([dataset['train'][i]['session'], cnt])","metadata":{"execution":{"iopub.status.busy":"2022-11-02T10:21:44.621430Z","iopub.execute_input":"2022-11-02T10:21:44.621860Z","iopub.status.idle":"2022-11-02T10:23:34.283834Z","shell.execute_reply.started":"2022-11-02T10:21:44.621824Z","shell.execute_reply":"2022-11-02T10:23:34.282376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div align=\"center\"><p style=\"font-family: 'Mochiy Pop P One';font-size:20px; font-weight: bold; color:black\">Method 2: use pandas and read file by chunks </p></div>\n\n","metadata":{}},{"cell_type":"code","source":"%%time\nnum_lines = sum(1 for line in open(train_file))\nprint(f'Num. of samples in train dataset: {num_lines:,}')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T10:23:34.285232Z","iopub.execute_input":"2022-11-02T10:23:34.285624Z","iopub.status.idle":"2022-11-02T10:23:45.732210Z","shell.execute_reply.started":"2022-11-02T10:23:34.285590Z","shell.execute_reply":"2022-11-02T10:23:45.730854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nchunksize = 100000\nchunks = pd.read_json(train_file, lines=True, chunksize=chunksize)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T10:23:45.733722Z","iopub.execute_input":"2022-11-02T10:23:45.734267Z","iopub.status.idle":"2022-11-02T10:23:45.743726Z","shell.execute_reply.started":"2022-11-02T10:23:45.734234Z","shell.execute_reply":"2022-11-02T10:23:45.742467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsession_stats2 = []\nfor chunk in tqdm(chunks):\n    #chunk.set_index(keys=['session'], inplace=True )\n    chunk['cnt'] = chunk['events'].apply(lambda x: len(x))\n    session_stats.append(chunk[['session', 'cnt',]])\n    break","metadata":{"execution":{"iopub.status.busy":"2022-11-02T10:23:45.745149Z","iopub.execute_input":"2022-11-02T10:23:45.745995Z","iopub.status.idle":"2022-11-02T10:23:54.134540Z","shell.execute_reply.started":"2022-11-02T10:23:45.745959Z","shell.execute_reply":"2022-11-02T10:23:54.133372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div align=\"center\"><p style=\"font-family: 'Mochiy Pop P One';font-size:20px; font-weight: bold; color:black\">References: </p></div>\n\n\n1.  <a href=\"https://www.kaggle.com/daquarti\">Daquarti Gustavo</a>'s notebook <a href=\"https://www.kaggle.com/code/daquarti/reading-jsonl-data-using-huggingface-datasets\">Reading jsonl data using HuggingFace Datasets</a>\n<br>   \n2.  <a href=\"https://www.kaggle.com/edwardcrookenden\">Edward Crookenden</a>'s notebook <a href=\"https://www.kaggle.com/code/edwardcrookenden/otto-getting-started-eda-baseline\">OTTO - Getting Started (EDA + Baseline)</a>  ","metadata":{}}]}