{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install beartype -q","metadata":{"execution":{"iopub.status.busy":"2022-12-21T08:48:54.156704Z","iopub.execute_input":"2022-12-21T08:48:54.157337Z","iopub.status.idle":"2022-12-21T08:49:08.176213Z","shell.execute_reply.started":"2022-12-21T08:48:54.157202Z","shell.execute_reply":"2022-12-21T08:49:08.174603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport argparse\nimport json\nimport random\nfrom copy import deepcopy\nfrom pathlib import Path\nimport pandas as pd, numpy as np\nfrom tqdm.notebook import tqdm\nimport os, sys, pickle, glob, gc\nfrom collections import Counter,defaultdict\n#import cudf, itertools\nimport pyarrow.parquet as pq\nimport json\nimport pandas as pd\nfrom beartype import beartype\nfrom pandas.io.json._json import JsonReader\nfrom tqdm.auto import tqdm\nfrom collections import defaultdict\nimport os\nimport glob\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-21T08:49:08.179278Z","iopub.execute_input":"2022-12-21T08:49:08.179688Z","iopub.status.idle":"2022-12-21T08:49:08.435394Z","shell.execute_reply.started":"2022-12-21T08:49:08.179650Z","shell.execute_reply":"2022-12-21T08:49:08.434407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Count clicks&carts&orders for each session","metadata":{}},{"cell_type":"code","source":"# CACHE FUNCTIONS\ndef read_file(f):\n    return cudf.DataFrame( data_cache[f] )\ndef read_file_to_cache(f):\n    df = pd.read_parquet(f)\n    df.ts = (df.ts/1000).astype('int32')\n    df['type'] = df['type'].map(type_labels).astype('int8')\n    return df\n\n# CACHE THE DATA ON CPU BEFORE PROCESSING ON GPU\ndata_cache = {}\ntype_labels = {'clicks':0, 'carts':1, 'orders':2}\nfiles = glob.glob('/kaggle/input/otto-chunk-data-inparquet-format/*_parquet/*')\nfor f in files: data_cache[f] = read_file_to_cache(f)","metadata":{"execution":{"iopub.status.busy":"2022-12-21T08:49:08.436691Z","iopub.execute_input":"2022-12-21T08:49:08.437773Z","iopub.status.idle":"2022-12-21T08:50:11.705397Z","shell.execute_reply.started":"2022-12-21T08:49:08.437733Z","shell.execute_reply":"2022-12-21T08:50:11.703919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = data_cache[files[0]]\ndf","metadata":{"execution":{"iopub.status.busy":"2022-12-21T08:50:11.709851Z","iopub.execute_input":"2022-12-21T08:50:11.710371Z","iopub.status.idle":"2022-12-21T08:50:11.743236Z","shell.execute_reply.started":"2022-12-21T08:50:11.710329Z","shell.execute_reply":"2022-12-21T08:50:11.741647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nres = []\nfor df in data_cache.values():\n    res.append(df.groupby(['session', 'aid', 'type'], as_index=False)['type'].agg({'item_count_per_session':'count'}).sort_values('session').reset_index(drop=True))\n    \nres = pd.concat(res)\n\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T08:50:11.744729Z","iopub.execute_input":"2022-12-21T08:50:11.745145Z","iopub.status.idle":"2022-12-21T08:52:18.549281Z","shell.execute_reply.started":"2022-12-21T08:50:11.745112Z","shell.execute_reply":"2022-12-21T08:52:18.547903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-21T08:52:18.550697Z","iopub.execute_input":"2022-12-21T08:52:18.551082Z","iopub.status.idle":"2022-12-21T08:52:18.559553Z","shell.execute_reply.started":"2022-12-21T08:52:18.551048Z","shell.execute_reply":"2022-12-21T08:52:18.557831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res","metadata":{"execution":{"iopub.status.busy":"2022-12-21T08:44:41.584169Z","iopub.execute_input":"2022-12-21T08:44:41.585227Z","iopub.status.idle":"2022-12-21T08:44:41.605941Z","shell.execute_reply.started":"2022-12-21T08:44:41.585187Z","shell.execute_reply":"2022-12-21T08:44:41.604488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"most_clicks_per_users = res[res['type']==0][['session', 'aid', 'item_count_per_session']]\nmost_carts_per_users = res[res['type']==1][['session', 'aid', 'item_count_per_session']]\nmost_orders_per_users = res[res['type']==2][['session', 'aid', 'item_count_per_session']]\nmost_clicks_per_users.to_parquet(\"most_clicks_per_users.parquet\")\nmost_carts_per_users.to_parquet(\"most_carts_per_users.parquet\")\nmost_orders_per_users.to_parquet(\"most_orders_per_users.parquet\")","metadata":{"execution":{"iopub.status.busy":"2022-12-21T08:53:09.839873Z","iopub.execute_input":"2022-12-21T08:53:09.841253Z","iopub.status.idle":"2022-12-21T08:53:34.404731Z","shell.execute_reply.started":"2022-12-21T08:53:09.841195Z","shell.execute_reply":"2022-12-21T08:53:34.403215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndel data_cache, most_clicks_per_users, most_carts_per_users, most_orders_per_users\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T08:56:10.160377Z","iopub.execute_input":"2022-12-21T08:56:10.160997Z","iopub.status.idle":"2022-12-21T08:56:10.325859Z","shell.execute_reply.started":"2022-12-21T08:56:10.160940Z","shell.execute_reply":"2022-12-21T08:56:10.324285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Count clicks&carts&orders global","metadata":{}},{"cell_type":"code","source":"res[(res['aid'] == 3) & (res['type'] == 0)]","metadata":{"execution":{"iopub.status.busy":"2022-12-21T08:58:23.009433Z","iopub.execute_input":"2022-12-21T08:58:23.011127Z","iopub.status.idle":"2022-12-21T08:58:23.410514Z","shell.execute_reply.started":"2022-12-21T08:58:23.011044Z","shell.execute_reply":"2022-12-21T08:58:23.408312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_total = res.groupby(['aid', 'type'], as_index=False)['type'].agg({'item_count_per_session':'count'})\nmost_clicks_per_users = df_total[df_total['type']==0][['aid', 'item_count_per_session']]\nmost_carts_per_users = df_total[df_total['type']==1][['aid', 'item_count_per_session']]\nmost_orders_per_users = df_total[df_total['type']==2][['aid', 'item_count_per_session']]\nmost_clicks_per_users.to_parquet(\"most_clicks_for_all.parquet\")\nmost_carts_per_users.to_parquet(\"most_carts_for_all.parquet\")\nmost_orders_per_users.to_parquet(\"most_orders_for_all.parquet\")","metadata":{"execution":{"iopub.status.busy":"2022-12-21T09:01:29.691644Z","iopub.execute_input":"2022-12-21T09:01:29.692195Z","iopub.status.idle":"2022-12-21T09:02:00.348098Z","shell.execute_reply.started":"2022-12-21T09:01:29.692154Z","shell.execute_reply":"2022-12-21T09:02:00.346616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_total","metadata":{"execution":{"iopub.status.busy":"2022-12-21T09:07:41.822903Z","iopub.execute_input":"2022-12-21T09:07:41.823505Z","iopub.status.idle":"2022-12-21T09:07:41.841802Z","shell.execute_reply.started":"2022-12-21T09:07:41.823464Z","shell.execute_reply":"2022-12-21T09:07:41.840289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## User cart & buy ratio","metadata":{}},{"cell_type":"code","source":"df_clicks = res[res['type']==0]\ndf_carts = res[res['type']==1]\ndf_orders = res[res['type']==2]\ndf_orders","metadata":{"execution":{"iopub.status.busy":"2022-12-21T09:23:03.730387Z","iopub.execute_input":"2022-12-21T09:23:03.731383Z","iopub.status.idle":"2022-12-21T09:23:13.936456Z","shell.execute_reply.started":"2022-12-21T09:23:03.731323Z","shell.execute_reply":"2022-12-21T09:23:13.935147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_clicks = df_clicks.groupby(['session'], as_index=False)['session'].agg({'total_click':'count'})\ndf_carts = df_carts.groupby(['session'], as_index=False)['session'].agg({'total_carts':'count'})\ndf_orders = df_orders.groupby(['session'], as_index=False)['session'].agg({'total_order':'count'})\ndf_clicks","metadata":{"execution":{"iopub.status.busy":"2022-12-21T09:25:07.459572Z","iopub.execute_input":"2022-12-21T09:25:07.460732Z","iopub.status.idle":"2022-12-21T09:25:16.612522Z","shell.execute_reply.started":"2022-12-21T09:25:07.460665Z","shell.execute_reply":"2022-12-21T09:25:16.611093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_carts_orders_ratio = df_clicks.merge(df_carts, on='session', how='left').merge(df_orders, on='session', how='left').fillna(0)\ndf_carts_orders_ratio['cart_2_order_ratio'] = df_carts_orders_ratio['total_order'] / df_carts_orders_ratio['total_carts']\ndf_carts_orders_ratio['click_2_order_ratio'] = df_carts_orders_ratio['total_order'] / df_carts_orders_ratio['total_click']\ndf_carts_orders_ratio['click_2_cart_ratio'] = df_carts_orders_ratio['total_carts'] / df_carts_orders_ratio['total_click']\ndf_carts_orders_ratio = df_carts_orders_ratio.fillna(0)\ndf_carts_orders_ratio.to_parquet(\"df_carts_orders_ratio.parquet\")","metadata":{"execution":{"iopub.status.busy":"2022-12-21T09:41:59.471333Z","iopub.execute_input":"2022-12-21T09:41:59.471820Z","iopub.status.idle":"2022-12-21T09:42:23.560825Z","shell.execute_reply.started":"2022-12-21T09:41:59.471762Z","shell.execute_reply":"2022-12-21T09:42:23.558953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_carts_orders_ratio","metadata":{"execution":{"iopub.status.busy":"2022-12-21T09:42:23.562572Z","iopub.execute_input":"2022-12-21T09:42:23.564743Z","iopub.status.idle":"2022-12-21T09:42:23.590492Z","shell.execute_reply.started":"2022-12-21T09:42:23.564639Z","shell.execute_reply":"2022-12-21T09:42:23.589382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_carts_orders_ratio,df_clicks,df_carts,df_orders\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-12-21T09:47:11.131039Z","iopub.execute_input":"2022-12-21T09:47:11.131527Z","iopub.status.idle":"2022-12-21T09:47:11.450441Z","shell.execute_reply.started":"2022-12-21T09:47:11.131491Z","shell.execute_reply":"2022-12-21T09:47:11.449105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"de","metadata":{}},{"cell_type":"code","source":"@beartype\ndef ground_truth(events):\n    prev_labels = {\"clicks\": None, \"carts\": set(), \"orders\": set()}\n\n    for event in reversed(events):\n        event[\"labels\"] = {}\n\n        for label in ['clicks', 'carts', 'orders']:\n            if prev_labels[label]:\n                if label != 'clicks':\n                    event[\"labels\"][label] = prev_labels[label].copy()\n                else:\n                    event[\"labels\"][label] = prev_labels[label]\n\n        if event[\"type\"] == \"clicks\":\n            prev_labels['clicks'] = event[\"aid\"]\n        if event[\"type\"] == \"carts\":\n            prev_labels['carts'].add(event[\"aid\"])\n        elif event[\"type\"] == \"orders\":\n            prev_labels['orders'].add(event[\"aid\"])\n\n    return events[:-1]","metadata":{"execution":{"iopub.status.busy":"2022-12-20T12:22:46.409945Z","iopub.execute_input":"2022-12-20T12:22:46.410364Z","iopub.status.idle":"2022-12-20T12:22:46.419387Z","shell.execute_reply.started":"2022-12-20T12:22:46.410321Z","shell.execute_reply":"2022-12-20T12:22:46.418342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class setEncoder(json.JSONEncoder):\n\n    def default(self, obj):\n        return list(obj)\n\n\n@beartype\ndef split_events(events, split_idx=None):\n    test_events = ground_truth(deepcopy(events))\n    if not split_idx:\n        split_idx = random.randint(1, len(test_events))\n    test_events = test_events[:split_idx]\n    labels = test_events[-1]['labels']\n    for event in test_events:\n        del event['labels']\n    return test_events, labels\n\n\n@beartype\ndef create_kaggle_testset(sessions: pd.DataFrame, sessions_output, labels_output):\n    last_labels = []\n    splitted_sessions = []\n\n    for _, session in tqdm(sessions.iterrows(), desc=\"Creating trimmed testset\", total=len(sessions)):\n        session = session.to_dict()\n        splitted_events, labels = split_events(session['events'])\n        last_labels.append({'session': session['session'], 'labels': labels})\n        splitted_sessions.append({'session': session['session'], 'events': splitted_events})\n\n    with open(sessions_output, 'w') as f:\n        for session in splitted_sessions:\n            f.write(json.dumps(session) + '\\n')\n\n    with open(labels_output, 'w') as f:\n        for label in last_labels:\n            f.write(json.dumps(label, cls=setEncoder) + '\\n')\n\n\n@beartype\ndef trim_session(session: dict, max_ts: int) -> dict:\n    session['events'] = [event for event in session['events'] if event['ts'] < max_ts]\n    return session\n\n\n@beartype\ndef get_max_ts(sessions_path) -> int:\n    max_ts = float('-inf')\n    with open(sessions_path) as f:\n        for line in tqdm(f, desc=\"Finding max timestamp\"):\n            session = json.loads(line)\n            max_ts = max(max_ts, session['events'][-1]['ts'])\n    return max_ts\n\n\n@beartype\ndef filter_unknown_items(session_path, known_items):\n    filtered_sessions = []\n    with open(session_path) as f:\n        for line in tqdm(f, desc=\"Filtering unknown items\"):\n            session = json.loads(line)\n            session['events'] = [event for event in session['events'] if event['aid'] in known_items]\n            if len(session['events']) >= 2:\n                filtered_sessions.append(session)\n    with open(session_path, 'w') as f:\n        for session in filtered_sessions:\n            f.write(json.dumps(session) + '\\n')\n\n\n@beartype\ndef train_test_split(session_chunks: JsonReader, train_path, test_path, max_ts: int, test_days: int):\n    split_millis = test_days * 24 * 60 * 60 * 1000\n    split_ts = max_ts - split_millis\n    train_items = set()\n    Path(train_path).parent.mkdir(parents=True, exist_ok=True)\n    train_file = open(train_path, \"w\")\n    Path(test_path).parent.mkdir(parents=True, exist_ok=True)\n    test_file = open(test_path, \"w\")\n    for chunk in tqdm(session_chunks, desc=\"Splitting sessions\"):\n        for _, session in chunk.iterrows():\n            session = session.to_dict()\n            if session['events'][0]['ts'] > split_ts:\n                test_file.write(json.dumps(session) + \"\\n\")\n            else:\n                session = trim_session(session, split_ts)\n                if len(session['events']) >= 2:\n                    train_items.update([event['aid'] for event in session['events']])\n                    train_file.write(json.dumps(session) + \"\\n\")\n    train_file.close()\n    test_file.close()\n    filter_unknown_items(test_path, train_items)\n\n\n@beartype\ndef main(train_set, output_path, days: int, seed: int):\n    random.seed(seed)\n    max_ts = get_max_ts(train_set)\n\n    session_chunks = pd.read_json(train_set, lines=True, chunksize=100000)\n    train_file = os.path.join(output_path, 'train_sessions.jsonl')\n    test_file_full = os.path.join(output_path , 'test_sessions_full.jsonl')\n    train_test_split(session_chunks, train_file, test_file_full, max_ts, days)\n\n    test_sessions = pd.read_json(test_file_full, lines=True)\n    test_sessions_file = os.path.join( output_path, 'test_sessions.jsonl')\n    test_labels_file = os.path.join( output_path, 'test_labels.jsonl')\n    create_kaggle_testset(test_sessions, test_sessions_file, test_labels_file)","metadata":{"execution":{"iopub.status.busy":"2022-12-20T12:22:46.422297Z","iopub.execute_input":"2022-12-20T12:22:46.422715Z","iopub.status.idle":"2022-12-20T12:22:46.450572Z","shell.execute_reply.started":"2022-12-20T12:22:46.422673Z","shell.execute_reply":"2022-12-20T12:22:46.449531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"main('/kaggle/input/otto-recommender-system/train.jsonl', '/kaggle/working/', 2, 64)","metadata":{"execution":{"iopub.status.busy":"2022-12-20T12:22:46.452781Z","iopub.execute_input":"2022-12-20T12:22:46.453963Z","iopub.status.idle":"2022-12-20T13:05:23.272128Z","shell.execute_reply.started":"2022-12-20T12:22:46.453916Z","shell.execute_reply":"2022-12-20T13:05:23.270588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2022-12-20T13:25:11.253202Z","iopub.execute_input":"2022-12-20T13:25:11.253737Z","iopub.status.idle":"2022-12-20T13:25:12.288924Z","shell.execute_reply.started":"2022-12-20T13:25:11.253691Z","shell.execute_reply":"2022-12-20T13:25:12.287346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom pathlib import Path\ndata_path = Path('/kaggle/working/')\nchunksize = 100000\nsave = True\n\ndef save_2_parquet(input_json, output_name):\n    chunks = pd.read_json(data_path / f'{input_json}.jsonl', lines=True, chunksize=chunksize)\n    os.mkdir(f'{output_name}')\n\n    for e, chunk in enumerate(tqdm(chunks)):\n        event_dict = {\n            'session': [],\n            'aid': [],\n            'ts': [],\n            'type': [],\n        }\n\n        for session, events in zip(chunk['session'].tolist(), chunk['events'].tolist()):\n            for event in events:\n                event_dict['session'].append(session)\n                event_dict['aid'].append(event['aid'])\n                event_dict['ts'].append(event['ts'])\n                event_dict['type'].append(event['type'])\n\n        # save DataFrame\n        start = str(e*chunksize).zfill(9)\n        end = str(e*chunksize+chunksize).zfill(9)\n        event_df = pd.DataFrame(event_dict)\n        if save == True:\n            event_df.to_parquet(f\"{output_name}/{start}_{end}.parquet\")","metadata":{"execution":{"iopub.status.busy":"2022-12-20T13:37:51.640771Z","iopub.execute_input":"2022-12-20T13:37:51.642290Z","iopub.status.idle":"2022-12-20T13:37:51.652527Z","shell.execute_reply.started":"2022-12-20T13:37:51.642239Z","shell.execute_reply":"2022-12-20T13:37:51.651097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_2_parquet('train_sessions', 'train')","metadata":{"execution":{"iopub.status.busy":"2022-12-20T13:37:49.274718Z","iopub.execute_input":"2022-12-20T13:37:49.276048Z","iopub.status.idle":"2022-12-20T13:37:49.398007Z","shell.execute_reply.started":"2022-12-20T13:37:49.275991Z","shell.execute_reply":"2022-12-20T13:37:49.396248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nsave_2_parquet('test_sessions_full', 'test')","metadata":{"execution":{"iopub.status.busy":"2022-12-20T13:36:30.902081Z","iopub.execute_input":"2022-12-20T13:36:30.902446Z","iopub.status.idle":"2022-12-20T13:36:35.255222Z","shell.execute_reply.started":"2022-12-20T13:36:30.902413Z","shell.execute_reply":"2022-12-20T13:36:35.253712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}