{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-12T12:22:28.617123Z","iopub.execute_input":"2023-03-12T12:22:28.617531Z","iopub.status.idle":"2023-03-12T12:22:28.659558Z","shell.execute_reply.started":"2023-03-12T12:22:28.617495Z","shell.execute_reply":"2023-03-12T12:22:28.658059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\nimport pathlib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport json\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-03-12T12:40:01.574057Z","iopub.execute_input":"2023-03-12T12:40:01.574515Z","iopub.status.idle":"2023-03-12T12:40:01.581490Z","shell.execute_reply.started":"2023-03-12T12:40:01.574477Z","shell.execute_reply":"2023-03-12T12:40:01.579969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_PATH = pathlib.Path('../input/otto-recommender-system')\nTRAIN_PATH = DATA_PATH/'train.jsonl'\nTEST_PATH = DATA_PATH/'test.jsonl'\nSAMPLE_SUB_PATH = pathlib.Path('../input/otto-recommender-system/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-03-12T12:22:30.103488Z","iopub.execute_input":"2023-03-12T12:22:30.103965Z","iopub.status.idle":"2023-03-12T12:22:30.111094Z","shell.execute_reply.started":"2023-03-12T12:22:30.103916Z","shell.execute_reply":"2023-03-12T12:22:30.109160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets check how many lines the training data has!\n# with open(TRAIN_PATH, 'r') as f:\n#     print(f\"We have {len(f.readlines()):,} lines in the training data\")\n\n# Load in a sample to a pandas df\nsample_size = 100\nchunks = pd.read_json(TRAIN_PATH, lines=True, chunksize = sample_size)\nfor c in chunks:\n    sample_train_df = c\n    break    \nsample_train_df.set_index('session', drop=True, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T12:22:30.116757Z","iopub.execute_input":"2023-03-12T12:22:30.117285Z","iopub.status.idle":"2023-03-12T12:22:30.185697Z","shell.execute_reply.started":"2023-03-12T12:22:30.117234Z","shell.execute_reply":"2023-03-12T12:22:30.184061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.options.display.max_colwidth = 100\nsample_train_df['no_events'] = sample_train_df[\"events\"].apply(lambda x: len(x))\nsample_train_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T12:22:30.187382Z","iopub.execute_input":"2023-03-12T12:22:30.188016Z","iopub.status.idle":"2023-03-12T12:22:30.305039Z","shell.execute_reply.started":"2023-03-12T12:22:30.187977Z","shell.execute_reply":"2023-03-12T12:22:30.303974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with open(TEST_PATH, 'r') as f:\n#     print(f\"We have {len(f.readlines()):,} lines in the test data\")\n\n# Load in a sample to a pandas df\nsample_size = 150\nchunks = pd.read_json(TEST_PATH, lines=True, chunksize = sample_size)\nfor c in chunks:\n    sample_test_df = c\n    break","metadata":{"execution":{"iopub.status.busy":"2023-03-12T12:22:30.306450Z","iopub.execute_input":"2023-03-12T12:22:30.307096Z","iopub.status.idle":"2023-03-12T12:22:30.335372Z","shell.execute_reply.started":"2023-03-12T12:22:30.307055Z","shell.execute_reply":"2023-03-12T12:22:30.333520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_test_df[\"no_events\"] = sample_test_df[\"events\"].apply(lambda x: len(x))\n\npd.options.display.max_colwidth = 120\nsample_test_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T12:22:30.337050Z","iopub.execute_input":"2023-03-12T12:22:30.337512Z","iopub.status.idle":"2023-03-12T12:22:30.400253Z","shell.execute_reply.started":"2023-03-12T12:22:30.337469Z","shell.execute_reply":"2023-03-12T12:22:30.398093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_test_df.iloc[1].events","metadata":{"execution":{"iopub.status.busy":"2023-03-12T12:22:30.407067Z","iopub.execute_input":"2023-03-12T12:22:30.408134Z","iopub.status.idle":"2023-03-12T12:22:30.418780Z","shell.execute_reply.started":"2023-03-12T12:22:30.408069Z","shell.execute_reply":"2023-03-12T12:22:30.417036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(SAMPLE_SUB_PATH)\nsample_submission[\"count\"] = sample_submission[\"labels\"].apply(lambda x: len(x.split(\" \")))\n\npd.options.display.max_colwidth = 100\nsample_submission.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T12:22:30.422174Z","iopub.execute_input":"2023-03-12T12:22:30.422722Z","iopub.status.idle":"2023-03-12T12:22:41.878413Z","shell.execute_reply.started":"2023-03-12T12:22:30.422664Z","shell.execute_reply":"2023-03-12T12:22:41.876895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T12:22:41.882792Z","iopub.execute_input":"2023-03-12T12:22:41.883663Z","iopub.status.idle":"2023-03-12T12:22:41.893856Z","shell.execute_reply.started":"2023-03-12T12:22:41.883607Z","shell.execute_reply":"2023-03-12T12:22:41.892544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_directory = pathlib.Path('../input/otto-recommender-system')","metadata":{"execution":{"iopub.status.busy":"2023-03-12T12:34:46.471016Z","iopub.execute_input":"2023-03-12T12:34:46.471598Z","iopub.status.idle":"2023-03-12T12:34:46.477558Z","shell.execute_reply.started":"2023-03-12T12:34:46.471545Z","shell.execute_reply":"2023-03-12T12:34:46.476283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type_dict = {\n    'clicks': 0,\n    'carts': 1,\n    'orders': 2\n}\n\ndef create_ground_truth(events):\n    \n    \"\"\"\n    Create ground-truth labels from given list of event dictionaries\n    \n    Parameters\n    ----------\n    events: list of shape (n_events)\n        List of event dictionaries\n \n    Returns\n    -------\n    events: list of shape (n_events)\n        List of event dictionaries with labels\n    \"\"\"\n    \n    previous_labels = {'clicks': None, 'carts': set(), 'orders': set()}\n\n    for event in reversed(events):\n        \n        event['labels'] = {}\n\n        for label in ['clicks', 'carts', 'orders']:\n            if previous_labels[label]:\n                if label != 'clicks':\n                    event['labels'][type_dict[label]] = previous_labels[label].copy()\n                else:\n                    event['labels'][type_dict[label]] = previous_labels[label]\n\n        if event['type'] == 'clicks':\n            previous_labels['clicks'] = event['aid']\n        if event['type'] == 'carts':\n            previous_labels['carts'].add(event['aid'])\n        elif event['type'] == 'orders':\n            previous_labels['orders'].add(event['aid'])\n\n    return events\n\n\ndef create_dataframe(json_file_path, chunk_size):\n    \n    \"\"\"\n    Create pandas.DataFrame from given json_file_path\n    \n    Parameters\n    ----------\n    json_file_path: path-like str\n        Path of the json file\n        \n    chunk_size: int\n        Size of chunks while reading the json file\n        \n    Returns\n    -------\n    df: pandas.DataFrame of shape (n_samples, 4)\n    \"\"\"\n    \n    chunks = pd.read_json(json_file_path, lines=True, chunksize=chunk_size)\n    df = pd.DataFrame()\n    \n    for chunk_idx, chunk in enumerate(chunks):\n        \n        print(f'Reading Chunk {chunk_idx} ({chunk_size * chunk_idx}-{(chunk_size * (chunk_idx + 1))})')\n        \n        event_dict = {\n            'session': [],\n            'aid': [],\n            'ts': [],\n            'type': []\n        }\n        \n        for session, events in tqdm(zip(chunk['session'].tolist(), chunk['events'].tolist()), total=len(chunk['session'].tolist())):\n            \n            for event in events:\n                \n                event_dict['session'].append(session)\n                event_dict['aid'].append(event['aid'])\n                event_dict['ts'].append(event['ts'])\n                event_dict['type'].append(type_dict[event['type']])\n\n        chunk_session = pd.DataFrame(event_dict)\n        chunk_session['session'] = chunk_session['session'].astype(np.uint32)\n        chunk_session['aid'] = chunk_session['aid'].astype(np.uint32)\n        chunk_session['ts'] = pd.to_datetime(chunk_session['ts'], unit='ms')\n        chunk_session['type'] = chunk_session['type'].astype(np.uint8)\n        df = pd.concat([df, chunk_session])\n        \n    df.reset_index(drop=True, inplace=True)\n        \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-03-12T12:40:07.072960Z","iopub.execute_input":"2023-03-12T12:40:07.074497Z","iopub.status.idle":"2023-03-12T12:40:07.092982Z","shell.execute_reply.started":"2023-03-12T12:40:07.074445Z","shell.execute_reply":"2023-03-12T12:40:07.091251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = create_dataframe(json_file_path=str(dataset_directory / 'train.jsonl'), chunk_size=100000)\ndf_train","metadata":{"execution":{"iopub.status.busy":"2023-03-12T12:40:10.339758Z","iopub.execute_input":"2023-03-12T12:40:10.340226Z"},"trusted":true},"execution_count":null,"outputs":[]}]}