{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom pathlib import Path\n\ndata_path = Path('/kaggle/input/otto-recommender-system/')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-01T22:00:24.991044Z","iopub.execute_input":"2022-11-01T22:00:24.992213Z","iopub.status.idle":"2022-11-01T22:00:24.996928Z","shell.execute_reply.started":"2022-11-01T22:00:24.992167Z","shell.execute_reply":"2022-11-01T22:00:24.996018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Reading the data in pandas\n\nThe training data use a lot of memory in pandas. It might not be the most efficient way to process the data, but you can certainly load data into pandas in chunks.","metadata":{}},{"cell_type":"code","source":"num_lines = sum(1 for line in open(data_path / 'train.jsonl'))\nprint(f'number of lines in train: {num_lines:_}')\n\nchunksize = 100_000\nnum_chunks = int(np.ceil(num_lines / 100_000))\nprint(f'number of chunks: {num_chunks:_}')","metadata":{"execution":{"iopub.status.busy":"2022-11-01T22:01:35.032699Z","iopub.execute_input":"2022-11-01T22:01:35.033140Z","iopub.status.idle":"2022-11-01T22:03:42.985568Z","shell.execute_reply.started":"2022-11-01T22:01:35.033106Z","shell.execute_reply":"2022-11-01T22:03:42.984026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Read the first two chunks","metadata":{}},{"cell_type":"code","source":"n = 2\ntrain_sessions = pd.DataFrame()\nchunks = pd.read_json(data_path / 'train.jsonl', lines=True, chunksize=chunksize)\n\nfor e, chunk in enumerate(chunks):\n    if e < 2:\n        train_sessions = pd.concat([train_sessions, chunk])\n    else:\n        break\ntrain_sessions = train_sessions.set_index('session', drop=True).sort_index()","metadata":{"execution":{"iopub.status.busy":"2022-11-01T22:11:24.529469Z","iopub.execute_input":"2022-11-01T22:11:24.529937Z","iopub.status.idle":"2022-11-01T22:12:02.529819Z","shell.execute_reply.started":"2022-11-01T22:11:24.529890Z","shell.execute_reply":"2022-11-01T22:12:02.528565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sessions","metadata":{"execution":{"iopub.status.busy":"2022-11-01T22:12:04.341296Z","iopub.execute_input":"2022-11-01T22:12:04.341735Z","iopub.status.idle":"2022-11-01T22:12:04.496189Z","shell.execute_reply.started":"2022-11-01T22:12:04.341701Z","shell.execute_reply":"2022-11-01T22:12:04.495053Z"},"trusted":true},"execution_count":null,"outputs":[]}]}