{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.graph_objs as go\n\nimport plotly\nplotly.offline.init_notebook_mode(connected=True)\n\n#Ignore warnings\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-02T01:23:22.542253Z","iopub.execute_input":"2022-11-02T01:23:22.542796Z","iopub.status.idle":"2022-11-02T01:23:25.327013Z","shell.execute_reply.started":"2022-11-02T01:23:22.542705Z","shell.execute_reply":"2022-11-02T01:23:25.325106Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"![](https://bunnyacademy.b-cdn.net/6xdc9-What-Is-HTTP-Chunked-Encoding-How-and-When-Is-It-Used.png)bunny.net","metadata":{}},{"cell_type":"code","source":"#Code by Mohamed Aesawy https://www.kaggle.com/competitions/LANL-Earthquake-Prediction/discussion/77292\n\nimport dask.dataframe as dd\n\ndf = dd.read_json(\"../input/otto-recommender-system/train.jsonl\", dtype={'acoustic_data': np.int16, 'time_to_failure': np.float64})\n\n# returns the first \"partition\" of the dataframe\npart_acoustic_data = np.array(df.acoustic_data.partitions[0]) \npart_time = np.array(df.time_to_failure.partitions[0])\n\n# print total number of partitions\nprint(df.npartitions)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T00:50:57.647782Z","iopub.execute_input":"2022-11-02T00:50:57.648200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Your notebook tried to allocate more memory than is available. It has restarted.","metadata":{}},{"cell_type":"code","source":"train= pd.read_json(path_or_buf=\"../input/otto-recommender-system/train.jsonl\",lines=True, chunksize=5000)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T01:29:30.878992Z","iopub.execute_input":"2022-11-02T01:29:30.879385Z","iopub.status.idle":"2022-11-02T01:29:30.888314Z","shell.execute_reply.started":"2022-11-02T01:29:30.879347Z","shell.execute_reply":"2022-11-02T01:29:30.886065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Remember chunks are True","metadata":{}},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2022-11-02T01:29:35.557166Z","iopub.execute_input":"2022-11-02T01:29:35.558009Z","iopub.status.idle":"2022-11-02T01:29:35.564731Z","shell.execute_reply.started":"2022-11-02T01:29:35.557976Z","shell.execute_reply":"2022-11-02T01:29:35.563600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Didn't help too.","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv(\"/kaggle/input/otto-recommender-system/sample_submission.csv\", delimiter=',', encoding='ISO-8859-2')\n\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-02T01:43:52.110245Z","iopub.execute_input":"2022-11-02T01:43:52.111872Z","iopub.status.idle":"2022-11-02T01:43:59.760806Z","shell.execute_reply.started":"2022-11-02T01:43:52.111793Z","shell.execute_reply":"2022-11-02T01:43:59.758904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Only with Inversion's lines I could read one column","metadata":{}},{"cell_type":"code","source":"#code by Inversion https://www.kaggle.com/code/inversion/read-a-chunk-of-jsonl\n\nfrom pathlib import Path\n\ndata_path = Path('/kaggle/input/otto-recommender-system/')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T00:33:48.316744Z","iopub.execute_input":"2022-11-02T00:33:48.317146Z","iopub.status.idle":"2022-11-02T00:33:48.324179Z","shell.execute_reply.started":"2022-11-02T00:33:48.317095Z","shell.execute_reply":"2022-11-02T00:33:48.321650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#code by Inversion https://www.kaggle.com/code/inversion/read-a-chunk-of-jsonl\n\nchunksize = 100_000","metadata":{"execution":{"iopub.status.busy":"2022-11-02T00:34:17.087910Z","iopub.execute_input":"2022-11-02T00:34:17.088281Z","iopub.status.idle":"2022-11-02T00:34:17.094309Z","shell.execute_reply.started":"2022-11-02T00:34:17.088256Z","shell.execute_reply":"2022-11-02T00:34:17.092632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#code by Inversion https://www.kaggle.com/code/inversion/read-a-chunk-of-jsonl\n\nn = 2\ntrain_sessions = pd.DataFrame()\nchunks = pd.read_json(data_path / 'train.jsonl', lines=True, chunksize=chunksize)\n\nfor e, chunk in enumerate(chunks):\n    if e < 2:\n        train_sessions = pd.concat([train_sessions, chunk])\n    else:\n        break\ntrain_sessions = train_sessions.set_index('session', drop=True).sort_index()","metadata":{"execution":{"iopub.status.busy":"2022-11-02T00:34:22.990256Z","iopub.execute_input":"2022-11-02T00:34:22.990619Z","iopub.status.idle":"2022-11-02T00:34:47.358870Z","shell.execute_reply.started":"2022-11-02T00:34:22.990594Z","shell.execute_reply":"2022-11-02T00:34:47.358198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#code by Inversion https://www.kaggle.com/code/inversion/read-a-chunk-of-jsonl\n\ntrain_sessions","metadata":{"execution":{"iopub.status.busy":"2022-11-02T00:40:29.442656Z","iopub.execute_input":"2022-11-02T00:40:29.443232Z","iopub.status.idle":"2022-11-02T00:40:29.578893Z","shell.execute_reply.started":"2022-11-02T00:40:29.443182Z","shell.execute_reply":"2022-11-02T00:40:29.577447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#I don't even know what to make with 200000 rows and one single column","metadata":{}},{"cell_type":"markdown","source":"Amazing work by Edward Crookenden https://www.kaggle.com/code/edwardcrookenden/otto-getting-started","metadata":{}},{"cell_type":"markdown","source":"![](https://encrypted-tbn0.gstatic.com/images?q=tbn:ANd9GcQwGyj8Venz2OUiKoWgkhkfIxihqURm7OvW8A&usqp=CAU)azquotes.com","metadata":{}},{"cell_type":"markdown","source":"#I couldn't chunk anything here. My bad. Chunking it's not an easy task.","metadata":{}}]}