{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**This work is inspired by this source -- https://www.kaggle.com/code/columbia2131/otto-read-a-chunk-of-jsonl-to-manageable-df/**","metadata":{}},{"cell_type":"code","source":"\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking runor pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-08T15:52:08.631517Z","iopub.execute_input":"2022-11-08T15:52:08.63202Z","iopub.status.idle":"2022-11-08T15:52:08.669885Z","shell.execute_reply.started":"2022-11-08T15:52:08.631915Z","shell.execute_reply":"2022-11-08T15:52:08.66879Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = Path('/kaggle/input/otto-recommender-system/')\ndata_train_file = os.path.join(data_dir,'train.jsonl')\nprint(data_train_file)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T16:59:27.365538Z","iopub.execute_input":"2022-11-08T16:59:27.366187Z","iopub.status.idle":"2022-11-08T16:59:27.374832Z","shell.execute_reply.started":"2022-11-08T16:59:27.366136Z","shell.execute_reply":"2022-11-08T16:59:27.373008Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_sessions = pd.DataFrame()\nchunks = pd.read_json(data_train_file, lines=True, chunksize=100000)\n\nfor e, chunk in enumerate(chunks):\n    event_dict = {\n        'session': [],\n        'aid': [],\n        'ts': [],\n        'type': [],\n    }\n    if e < 2:\n        # train_sessions = pd.concat([train_sessions, chunk])\n        for session, events in zip(chunk['session'].tolist(), chunk['events'].tolist()):\n            for event in events:\n                event_dict['session'].append(session)\n                event_dict['aid'].append(event['aid'])\n                event_dict['ts'].append(event['ts'])\n                event_dict['type'].append(event['type'])\n        chunk_session = pd.DataFrame(event_dict)\n        train_sessions = pd.concat([train_sessions, chunk_session])\n    else:\n        break\n        \ntrain_sessions = train_sessions.reset_index(drop=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-08T17:00:29.168486Z","iopub.execute_input":"2022-11-08T17:00:29.169249Z","iopub.status.idle":"2022-11-08T17:01:34.044707Z","shell.execute_reply.started":"2022-11-08T17:00:29.169201Z","shell.execute_reply":"2022-11-08T17:01:34.043238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**A look at re-arranged data**","metadata":{}},{"cell_type":"code","source":"print(train_sessions)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T17:01:34.04682Z","iopub.execute_input":"2022-11-08T17:01:34.047396Z","iopub.status.idle":"2022-11-08T17:01:34.061724Z","shell.execute_reply.started":"2022-11-08T17:01:34.047334Z","shell.execute_reply":"2022-11-08T17:01:34.060641Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_sessions[train_sessions['session']==747])\n      \n#np.dtype(train_sessions.aid)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T17:49:05.158179Z","iopub.execute_input":"2022-11-08T17:49:05.158746Z","iopub.status.idle":"2022-11-08T17:49:05.183729Z","shell.execute_reply.started":"2022-11-08T17:49:05.158708Z","shell.execute_reply":"2022-11-08T17:49:05.182812Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_sessions[train_sessions['type']=='clicks'].count())","metadata":{"_kg_hide-output":true},"execution_count":null,"outputs":[]}]}