{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from tqdm import tqdm\nimport pathlib\nimport json\nimport numpy as np\nimport pandas as pd","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-05T13:44:45.678423Z","iopub.execute_input":"2022-11-05T13:44:45.679402Z","iopub.status.idle":"2022-11-05T13:44:45.688082Z","shell.execute_reply.started":"2022-11-05T13:44:45.679319Z","shell.execute_reply":"2022-11-05T13:44:45.686683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_directory = pathlib.Path('../input/otto-recommender-system')","metadata":{"execution":{"iopub.status.busy":"2022-11-05T13:44:45.913123Z","iopub.execute_input":"2022-11-05T13:44:45.913504Z","iopub.status.idle":"2022-11-05T13:44:45.918793Z","shell.execute_reply.started":"2022-11-05T13:44:45.913475Z","shell.execute_reply":"2022-11-05T13:44:45.917413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type_dict = {\n    'clicks': 0,\n    'carts': 1,\n    'orders': 2\n}\n\ndef create_ground_truth(events):\n    \n    \"\"\"\n    Create ground-truth labels from given list of event dictionaries\n    \n    Parameters\n    ----------\n    events: list of shape (n_events)\n        List of event dictionaries\n \n    Returns\n    -------\n    events: list of shape (n_events)\n        List of event dictionaries with labels\n    \"\"\"\n    \n    previous_labels = {'clicks': None, 'carts': set(), 'orders': set()}\n\n    for event in reversed(events):\n        \n        event['labels'] = {}\n\n        for label in ['clicks', 'carts', 'orders']:\n            if previous_labels[label]:\n                if label != 'clicks':\n                    event['labels'][type_dict[label]] = previous_labels[label].copy()\n                else:\n                    event['labels'][type_dict[label]] = previous_labels[label]\n\n        if event['type'] == 'clicks':\n            previous_labels['clicks'] = event['aid']\n        if event['type'] == 'carts':\n            previous_labels['carts'].add(event['aid'])\n        elif event['type'] == 'orders':\n            previous_labels['orders'].add(event['aid'])\n\n    return events\n\n\ndef create_dataframe(json_file_path, chunk_size):\n    \n    \"\"\"\n    Create pandas.DataFrame from given json_file_path\n    \n    Parameters\n    ----------\n    json_file_path: path-like str\n        Path of the json file\n        \n    chunk_size: int\n        Size of chunks while reading the json file\n        \n    Returns\n    -------\n    df: pandas.DataFrame of shape (n_samples, 4)\n    \"\"\"\n    \n    chunks = pd.read_json(json_file_path, lines=True, chunksize=chunk_size)\n    df = pd.DataFrame()\n    \n    for chunk_idx, chunk in enumerate(chunks):\n        \n        print(f'Reading Chunk {chunk_idx} ({chunk_size * chunk_idx}-{(chunk_size * (chunk_idx + 1))})')\n        \n        event_dict = {\n            'session': [],\n            'aid': [],\n            'ts': [],\n            'type': []\n        }\n        \n        for session, events in tqdm(zip(chunk['session'].tolist(), chunk['events'].tolist()), total=len(chunk['session'].tolist())):\n            \n            for event in events:\n                \n                event_dict['session'].append(session)\n                event_dict['aid'].append(event['aid'])\n                event_dict['ts'].append(event['ts'])\n                event_dict['type'].append(type_dict[event['type']])\n\n        chunk_session = pd.DataFrame(event_dict)\n        chunk_session['session'] = chunk_session['session'].astype(np.uint32)\n        chunk_session['aid'] = chunk_session['aid'].astype(np.uint32)\n        chunk_session['ts'] = pd.to_datetime(chunk_session['ts'], unit='ms')\n        chunk_session['type'] = chunk_session['type'].astype(np.uint8)\n        df = pd.concat([df, chunk_session])\n        \n    df.reset_index(drop=True, inplace=True)\n        \n    return df\n","metadata":{"execution":{"iopub.status.busy":"2022-11-05T13:49:49.396069Z","iopub.execute_input":"2022-11-05T13:49:49.396496Z","iopub.status.idle":"2022-11-05T13:49:49.413251Z","shell.execute_reply.started":"2022-11-05T13:49:49.396459Z","shell.execute_reply":"2022-11-05T13:49:49.412267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = create_dataframe(json_file_path=str(dataset_directory / 'train.jsonl'), chunk_size=100000)\ndf_train","metadata":{"execution":{"iopub.status.busy":"2022-11-05T13:49:49.958461Z","iopub.execute_input":"2022-11-05T13:49:49.959354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = create_dataframe(json_file_path=str(dataset_directory / 'test.jsonl'), chunk_size=100000)\ndf_test","metadata":{"execution":{"iopub.status.busy":"2022-11-05T13:12:18.599457Z","iopub.execute_input":"2022-11-05T13:12:18.599899Z","iopub.status.idle":"2022-11-05T13:13:55.923021Z","shell.execute_reply.started":"2022-11-05T13:12:18.599864Z","shell.execute_reply":"2022-11-05T13:13:55.921754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Training Shape: {df_train.shape} - Memory Usage: {df_train.memory_usage().sum() / 1024 ** 2:.2f} MB')\nprint(f'Test Shape: {df_test.shape} - Memory Usage: {df_test.memory_usage().sum() / 1024 ** 2:.2f} MB')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ndf_train.to_pickle('train.pkl')\ndf_test.to_pickle('test.pkl')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T10:00:22.700561Z","iopub.execute_input":"2022-11-02T10:00:22.700962Z","iopub.status.idle":"2022-11-02T10:00:44.675481Z","shell.execute_reply.started":"2022-11-02T10:00:22.700936Z","shell.execute_reply":"2022-11-02T10:00:44.673928Z"},"trusted":true},"execution_count":null,"outputs":[]}]}