{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport jsonlines\nfrom tqdm import tqdm\nimport datetime\nfrom glob import glob\n\n%load_ext autoreload\n%autoreload 2","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:36:09.314443Z","iopub.execute_input":"2022-11-15T10:36:09.314884Z","iopub.status.idle":"2022-11-15T10:36:09.377152Z","shell.execute_reply.started":"2022-11-15T10:36:09.314789Z","shell.execute_reply":"2022-11-15T10:36:09.376452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Just use the parquet files from this notebooks in data tab. ","metadata":{}},{"cell_type":"markdown","source":"# Test data","metadata":{}},{"cell_type":"code","source":"data_dict = {'session': [], 'aid': [], 'ts': [], 'type': []}\n\nwith jsonlines.open('/kaggle/input/otto-recommender-system/test.jsonl') as reader:\n    #Iterate over the each line on the reader\n    for result in tqdm(reader):   \n        for event in result['events']:     \n            data_dict['session'].append(result['session'])\n            data_dict['aid'].append(event['aid'])\n            data_dict['ts'].append(event['ts'])\n            data_dict['type'].append(event['type'])\n            break\n        break","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:36:09.378464Z","iopub.execute_input":"2022-11-15T10:36:09.378827Z","iopub.status.idle":"2022-11-15T10:36:09.423723Z","shell.execute_reply.started":"2022-11-15T10:36:09.378805Z","shell.execute_reply":"2022-11-15T10:36:09.422356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.DataFrame.from_dict(data_dict)","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:36:09.425028Z","iopub.execute_input":"2022-11-15T10:36:09.425726Z","iopub.status.idle":"2022-11-15T10:36:09.456920Z","shell.execute_reply.started":"2022-11-15T10:36:09.425700Z","shell.execute_reply":"2022-11-15T10:36:09.455576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train data","metadata":{}},{"cell_type":"markdown","source":"With train data it is not so easy, becouse of the memmory issiues. We will use the loop for saving by chunks and will concatinate that.","metadata":{}},{"cell_type":"code","source":"chunksize = 100_000\nchunks = pd.read_json('/kaggle/input/otto-recommender-system/train.jsonl', lines=True, chunksize=chunksize)\n\nfor e, chunk in enumerate(tqdm(chunks)):\n    data_dict = {\n        'session': [],\n        'aid': [],\n        'ts': [],\n        'type': [],\n    }\n    \n    for session, events in zip(chunk['session'].tolist(), chunk['events'].tolist()):\n        for event in events:\n            data_dict['session'].append(session)\n            data_dict['aid'].append(event['aid'])\n            data_dict['ts'].append(event['ts'])\n            data_dict['type'].append(event['type'])\n            break\n        break\n    df = pd.DataFrame(data_dict)\n#     df.to_parquet(f\"../data/interim/train/train_{datetime.datetime.now().strftime('%y-%m-%d_%H-%M-%S')}.parquet\")\n    break","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:36:09.459508Z","iopub.execute_input":"2022-11-15T10:36:09.460062Z","iopub.status.idle":"2022-11-15T10:36:16.722716Z","shell.execute_reply.started":"2022-11-15T10:36:09.460029Z","shell.execute_reply":"2022-11-15T10:36:16.721267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files = sorted(glob('/kaggle/input/ottorecsystrain/*'))[:1]\ntrain_data = []\n\nfor path in files:\n    train_data.append(pd.read_parquet(path))\n\ntrain_data = pd.concat(train_data).reset_index(drop=True)\n# train_data.to_parquet('/kaggle/input/ottorecsys/train.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:36:16.724010Z","iopub.execute_input":"2022-11-15T10:36:16.724368Z","iopub.status.idle":"2022-11-15T10:36:35.309827Z","shell.execute_reply.started":"2022-11-15T10:36:16.724336Z","shell.execute_reply":"2022-11-15T10:36:35.308803Z"},"trusted":true},"execution_count":null,"outputs":[]}]}