{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Reading the data using HuggingFace Datasets\n\n🤗 Datasets uses Arrow for its local caching system. It allows datasets to be backed by an on-disk cache, which is memory-mapped for fast lookup. This architecture allows for large datasets to be used on machines with relatively small device memory.","metadata":{}},{"cell_type":"code","source":"!pip install -q datasets","metadata":{"execution":{"iopub.status.busy":"2022-11-02T02:39:26.437400Z","iopub.execute_input":"2022-11-02T02:39:26.438593Z","iopub.status.idle":"2022-11-02T02:39:37.446638Z","shell.execute_reply.started":"2022-11-02T02:39:26.438547Z","shell.execute_reply":"2022-11-02T02:39:37.445350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom pathlib import Path\n\ndata_path = Path('/kaggle/input/otto-recommender-system/')\n\ntest = data_path / 'test.jsonl'\n\ntrain = data_path / 'train.jsonl'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-02T02:30:21.858141Z","iopub.execute_input":"2022-11-02T02:30:21.859244Z","iopub.status.idle":"2022-11-02T02:30:21.865724Z","shell.execute_reply.started":"2022-11-02T02:30:21.859198Z","shell.execute_reply":"2022-11-02T02:30:21.864342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datasets import load_dataset\ndataset = load_dataset(\"json\", data_files= str(train))","metadata":{"execution":{"iopub.status.busy":"2022-11-02T02:30:22.824253Z","iopub.execute_input":"2022-11-02T02:30:22.825600Z","iopub.status.idle":"2022-11-02T02:30:52.840350Z","shell.execute_reply.started":"2022-11-02T02:30:22.825548Z","shell.execute_reply":"2022-11-02T02:30:52.839263Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Indexing\nA Dataset contains columns of data, and each column can be a different type of data. The index, or axis label, is used to access examples from the dataset. For example, indexing by the row returns a dictionary of an example from the dataset:","metadata":{}},{"cell_type":"code","source":"dataset['train'][3]","metadata":{"execution":{"iopub.status.busy":"2022-11-02T02:37:51.062400Z","iopub.execute_input":"2022-11-02T02:37:51.063550Z","iopub.status.idle":"2022-11-02T02:37:51.099435Z","shell.execute_reply.started":"2022-11-02T02:37:51.063505Z","shell.execute_reply":"2022-11-02T02:37:51.098354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}