{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"I would like to understand how the competition organizer split the training set and how to create the ground truth. But I don't have the resources to run the [scripts](https://github.com/otto-de/recsys-dataset/blob/main/src/testset.py) locally nor I think Kaggle could finish running the script online (but I should try). So, I just copy and paste the scripts here and try to read it without running the code.\n\nI would like to share what I understand and found, and hopefully get my newbie-questions answered by amazing kagglers \n\nNOTE: for beginners like me, to read script like this one, we should probably start from the main function near the bottom, and then read the related functions as you go along.","metadata":{}},{"cell_type":"code","source":"!pip install polars","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:25:35.772098Z","iopub.execute_input":"2022-12-27T07:25:35.772692Z","iopub.status.idle":"2022-12-27T07:25:52.923152Z","shell.execute_reply.started":"2022-12-27T07:25:35.772583Z","shell.execute_reply":"2022-12-27T07:25:52.921455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"pipenv run python -m src.testset --train-set train.jsonl --days 2 --output-path 'out/' --seed 42 ","metadata":{}},{"cell_type":"code","source":"import polars as pl\nimport pandas as pd\nimport random","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:25:52.925591Z","iopub.execute_input":"2022-12-27T07:25:52.925960Z","iopub.status.idle":"2022-12-27T07:25:52.991784Z","shell.execute_reply.started":"2022-12-27T07:25:52.925925Z","shell.execute_reply":"2022-12-27T07:25:52.990572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.core.interactiveshell import InteractiveShell\nInteractiveShell.ast_node_interactivity = \"all\"","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:25:52.993428Z","iopub.execute_input":"2022-12-27T07:25:52.994461Z","iopub.status.idle":"2022-12-27T07:25:52.999834Z","shell.execute_reply.started":"2022-12-27T07:25:52.994419Z","shell.execute_reply":"2022-12-27T07:25:52.998656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # train = pl.scan_parquet('/kaggle/input/otto-radek-style-polars/train.parquet')\n\n\ntrain_v = pl.scan_parquet('/kaggle/input/otto-train-and-test-data-for-local-validation/train.parquet')\ntest_v = pl.scan_parquet('/kaggle/input/otto-train-and-test-data-for-local-validation/test.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:25:53.002133Z","iopub.execute_input":"2022-12-27T07:25:53.002554Z","iopub.status.idle":"2022-12-27T07:25:53.049441Z","shell.execute_reply.started":"2022-12-27T07:25:53.002517Z","shell.execute_reply":"2022-12-27T07:25:53.048329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_v7 = pl.scan_parquet('/kaggle/input/ottovalidation7days/train_sessions.parquet')\ntest_v7 = pl.scan_parquet('/kaggle/input/ottovalidation7days/test_sessions.parquet')\ntest_v7_full = pl.scan_parquet('/kaggle/input/ottovalidation7days/test_sessions_full.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:26:20.698296Z","iopub.execute_input":"2022-12-27T07:26:20.698692Z","iopub.status.idle":"2022-12-27T07:26:20.731922Z","shell.execute_reply.started":"2022-12-27T07:26:20.698660Z","shell.execute_reply":"2022-12-27T07:26:20.730666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_v7_2nd = pl.scan_parquet('/kaggle/input/ottovalidation7days2nd/train_sessions.parquet')\ntest_v7_2nd = pl.scan_parquet('/kaggle/input/ottovalidation7days2nd/test_sessions.parquet')\ntest_v7_full_2nd = pl.scan_parquet('/kaggle/input/ottovalidation7days2nd/test_sessions_full.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:26:21.214141Z","iopub.execute_input":"2022-12-27T07:26:21.214573Z","iopub.status.idle":"2022-12-27T07:26:21.244585Z","shell.execute_reply.started":"2022-12-27T07:26:21.214539Z","shell.execute_reply":"2022-12-27T07:26:21.243565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_v.select([\n    pl.col('session').n_unique().alias('total_sessions'),\n    pl.col('aid').n_unique().alias('total_aid'),    \n    pl.col('session').count().alias('total_rows'),    \n    (pl.col('ts').cast(pl.Int64)*1000).min().cast(pl.Datetime).dt.with_time_unit('ms').alias(\"min_datetime\"), \n    (pl.col('ts').cast(pl.Int64)*1000).max().cast(pl.Datetime).dt.with_time_unit('ms').alias(\"max_datetime\"), \n    (pl.col('ts').cast(pl.Int64)*1000).first().cast(pl.Datetime).dt.with_time_unit('ms').alias(\"first_datetime\"),     \n    (pl.col('ts').cast(pl.Int64)*1000).last().cast(pl.Datetime).dt.with_time_unit('ms').alias(\"last_datetime\"),     \n]).collect()","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:26:24.192188Z","iopub.execute_input":"2022-12-27T07:26:24.193559Z","iopub.status.idle":"2022-12-27T07:27:00.206608Z","shell.execute_reply.started":"2022-12-27T07:26:24.193504Z","shell.execute_reply":"2022-12-27T07:27:00.205289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_v7.select([\n    pl.col('session').n_unique().alias('total_sessions'),\n    pl.col('aid').n_unique().alias('total_aid'),    \n    pl.col('session').count().alias('total_rows'),      \n    pl.col('ts').min().cast(pl.Datetime).dt.with_time_unit('ms').alias(\"min_datetime\"), \n    pl.col('ts').max().cast(pl.Datetime).dt.with_time_unit('ms').alias(\"max_datetime\"), \n    pl.col('ts').first().cast(pl.Datetime).dt.with_time_unit('ms').alias(\"first_datetime\"),     \n    pl.col('ts').last().cast(pl.Datetime).dt.with_time_unit('ms').alias(\"last_datetime\"),     \n]).collect()","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:27:00.208616Z","iopub.execute_input":"2022-12-27T07:27:00.208952Z","iopub.status.idle":"2022-12-27T07:27:31.879098Z","shell.execute_reply.started":"2022-12-27T07:27:00.208922Z","shell.execute_reply":"2022-12-27T07:27:31.878185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_v7_2nd.select([\n    pl.col('session').n_unique().alias('total_sessions'),\n    pl.col('aid').n_unique().alias('total_aid'),    \n    pl.col('session').count().alias('total_rows'),      \n    pl.col('ts').min().cast(pl.Datetime).dt.with_time_unit('ms').alias(\"min_datetime\"), \n    pl.col('ts').max().cast(pl.Datetime).dt.with_time_unit('ms').alias(\"max_datetime\"), \n    pl.col('ts').first().cast(pl.Datetime).dt.with_time_unit('ms').alias(\"first_datetime\"),     \n    pl.col('ts').last().cast(pl.Datetime).dt.with_time_unit('ms').alias(\"last_datetime\"),     \n]).collect()","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:27:31.880588Z","iopub.execute_input":"2022-12-27T07:27:31.881437Z","iopub.status.idle":"2022-12-27T07:28:02.468425Z","shell.execute_reply.started":"2022-12-27T07:27:31.881397Z","shell.execute_reply":"2022-12-27T07:28:02.467253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# create validation 7 days from train set (4 weeks)","metadata":{}},{"cell_type":"code","source":"train_ms = pl.scan_parquet('/kaggle/input/otto-radek-style-polars/train_ms.parquet')\ntest_ms = pl.scan_parquet('/kaggle/input/otto-radek-style-polars/test_ms.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:28:17.436855Z","iopub.execute_input":"2022-12-27T07:28:17.437321Z","iopub.status.idle":"2022-12-27T07:28:17.460469Z","shell.execute_reply.started":"2022-12-27T07:28:17.437280Z","shell.execute_reply":"2022-12-27T07:28:17.459283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_radek = pl.scan_parquet('/kaggle/input/otto-full-optimized-memory-footprint/train.parquet')\ntest_radek = pl.scan_parquet('/kaggle/input/otto-full-optimized-memory-footprint/test.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:28:18.690423Z","iopub.execute_input":"2022-12-27T07:28:18.690836Z","iopub.status.idle":"2022-12-27T07:28:18.708475Z","shell.execute_reply.started":"2022-12-27T07:28:18.690802Z","shell.execute_reply":"2022-12-27T07:28:18.707400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"split_ts_ms = train_ms.select([\n    (pl.col('ts').max() - 7*24*60*60*1000).alias('split_ts')\n]).collect().to_series().to_list()[0]","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:28:25.129495Z","iopub.execute_input":"2022-12-27T07:28:25.129912Z","iopub.status.idle":"2022-12-27T07:28:42.601832Z","shell.execute_reply.started":"2022-12-27T07:28:25.129875Z","shell.execute_reply":"2022-12-27T07:28:42.600480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ms.select([\n    pl.col('ts').max().cast(pl.Datetime).dt.with_time_unit('ms').alias('max_datetime'),\n    pl.col('ts').min().cast(pl.Datetime).dt.with_time_unit('ms').alias('min_datetime'),\n    pl.col('ts').first().cast(pl.Datetime).dt.with_time_unit('ms').alias('first_datetime'),\n    pl.col('ts').last().cast(pl.Datetime).dt.with_time_unit('ms').alias('last_datetime'),\n    pl.lit(split_ts_ms).cast(pl.Datetime).dt.with_time_unit('ms').alias('split_datetime'),\n]).collect()","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:28:42.603672Z","iopub.execute_input":"2022-12-27T07:28:42.604026Z","iopub.status.idle":"2022-12-27T07:28:50.460932Z","shell.execute_reply.started":"2022-12-27T07:28:42.603995Z","shell.execute_reply":"2022-12-27T07:28:50.459614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ms.select([\n    pl.col('session').n_unique().alias('total_sessions'),\n    pl.col('aid').n_unique().alias('total_aid'),    \n    pl.col('session').count().alias('total_rows'),       \n]).collect()","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:28:50.463545Z","iopub.execute_input":"2022-12-27T07:28:50.464028Z","iopub.status.idle":"2022-12-27T07:29:25.359069Z","shell.execute_reply.started":"2022-12-27T07:28:50.463981Z","shell.execute_reply":"2022-12-27T07:29:25.357848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ms.filter(~(pl.col('ts').first() > split_ts_ms).over('session')).select([\n    pl.col('session').n_unique().alias('total_sessions'),\n    pl.col('aid').n_unique().alias('total_aid'),    \n    pl.col('session').count().alias('total_rows'),       \n]).collect()","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:29:25.361683Z","iopub.execute_input":"2022-12-27T07:29:25.362041Z","iopub.status.idle":"2022-12-27T07:30:23.354528Z","shell.execute_reply.started":"2022-12-27T07:29:25.362010Z","shell.execute_reply":"2022-12-27T07:30:23.353316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ms.filter(~(pl.col('ts').first() > split_ts_ms).over('session')).filter(pl.col('ts') < split_ts_ms).select([\n    pl.col('session').n_unique().alias('total_sessions'),\n    pl.col('aid').n_unique().alias('total_aid'),    \n    pl.col('session').count().alias('total_rows'),       \n]).collect()","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:30:23.356109Z","iopub.execute_input":"2022-12-27T07:30:23.357342Z","iopub.status.idle":"2022-12-27T07:31:10.342350Z","shell.execute_reply.started":"2022-12-27T07:30:23.357292Z","shell.execute_reply":"2022-12-27T07:31:10.341184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ms.filter(~(pl.col('ts').first() > split_ts_ms).over('session')).filter(pl.col('ts') < split_ts_ms).filter((pl.col('aid').count()>=2).over('session')).select([\n    pl.col('session').n_unique().alias('total_sessions'),\n    pl.col('aid').n_unique().alias('total_aid'),    \n    pl.col('session').count().alias('total_rows'),       \n]).collect()","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:31:10.344051Z","iopub.execute_input":"2022-12-27T07:31:10.344908Z","iopub.status.idle":"2022-12-27T07:32:15.539677Z","shell.execute_reply.started":"2022-12-27T07:31:10.344862Z","shell.execute_reply":"2022-12-27T07:32:15.537900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataframe below confirms my implmenetation above is correct.","metadata":{}},{"cell_type":"code","source":"train_v7.select([\n    pl.col('session').n_unique().alias('total_sessions'),\n    pl.col('aid').n_unique().alias('total_aid'),    \n    pl.col('session').count().alias('total_rows'),      \n    pl.col('ts').min().cast(pl.Datetime).dt.with_time_unit('ms').alias(\"min_datetime\"), \n    pl.col('ts').max().cast(pl.Datetime).dt.with_time_unit('ms').alias(\"max_datetime\"), \n    pl.col('ts').first().cast(pl.Datetime).dt.with_time_unit('ms').alias(\"first_datetime\"),     \n    pl.col('ts').last().cast(pl.Datetime).dt.with_time_unit('ms').alias(\"last_datetime\"),     \n]).collect()","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:32:15.542235Z","iopub.execute_input":"2022-12-27T07:32:15.543798Z","iopub.status.idle":"2022-12-27T07:32:45.953118Z","shell.execute_reply.started":"2022-12-27T07:32:15.543746Z","shell.execute_reply":"2022-12-27T07:32:45.951957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_radek.filter(~(pl.col('ts').first() > (split_ts_ms/1000)).over('session')).filter(pl.col('ts') < (split_ts_ms/1000)).filter((pl.col('aid').count()>=2).over('session')).select([\n    pl.col('session').n_unique().alias('total_sessions'),\n    pl.col('aid').n_unique().alias('total_aid'),    \n    pl.col('session').count().alias('total_rows'),       \n]).collect()","metadata":{"execution":{"iopub.status.busy":"2022-12-27T07:32:45.956164Z","iopub.execute_input":"2022-12-27T07:32:45.956508Z","iopub.status.idle":"2022-12-27T07:33:49.117791Z","shell.execute_reply.started":"2022-12-27T07:32:45.956478Z","shell.execute_reply":"2022-12-27T07:33:49.116494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## How the ground truth is created\n\nBy reading the src.script below, we can confirm that \n\n- only the earliest click aid is kept as ground truth for 'clicks'\n- only unique aids of 'carts' and 'orders' of the session are kept as ground truth for 'carts' and 'orders'\n\n**QUESTION**: \n\n- why remove the earliest event or row of each session? \n- As @radek1 has not implemented this in his CV [notebook](https://www.kaggle.com/code/radek1/a-robust-local-validation-framework), so probably not important?","metadata":{}},{"cell_type":"code","source":"# # https://github.com/otto-de/recsys-dataset/blob/main/src/labels.py\n# # from src.labels import ground_truth\n# @beartype\n# # def ground_truth(events: list[dict]):\n# def ground_truth(events):\n#     print(\"starting on ground_truth --------------------------\")\n#     prev_labels = {\"clicks\": None, \"carts\": set(), \"orders\": set()} # clicks: has only one aid, carts and orders have unique aids\n\n#     for event in reversed(events): # loop every event (include session, aid, type and ts) and in descending order \n#         event[\"labels\"] = {} # add a key \"labels\" and a value {} to store ground truth for this event\n\n#         for label in ['clicks', 'carts', 'orders']:\n#             if prev_labels[label]: # None and set() are both False, so it means if prev_labels[label] is not empty\n#                 if label != 'clicks':\n#                     event[\"labels\"][label] = prev_labels[label].copy() # update the latest prev_labels on carts and orders to the current event\n#                 else:\n#                     event[\"labels\"][label] = prev_labels[label] # update the latest prev_labels on clicks to the current event\n\n#         if event[\"type\"] == \"clicks\": # if this row's type is 'clicks'\n#             prev_labels['clicks'] = event[\"aid\"] # update the current click aid in prev_labels['clicks'], only the earliest click aid is kept\n#         if event[\"type\"] == \"carts\": # if this row's type is 'carts'\n#             prev_labels['carts'].add(event[\"aid\"]) # update the current cart aid in prev_labels['carts']\n#         elif event[\"type\"] == \"orders\":\n#             prev_labels['orders'].add(event[\"aid\"])# update the current oder aid in prev_labels['orders']\n\n#     return events[:-1] # question: remove the last event or row (the earliest event), but why? ","metadata":{"execution":{"iopub.status.busy":"2022-12-03T11:37:33.154010Z","iopub.execute_input":"2022-12-03T11:37:33.155249Z","iopub.status.idle":"2022-12-03T11:37:33.166168Z","shell.execute_reply.started":"2022-12-03T11:37:33.155186Z","shell.execute_reply":"2022-12-03T11:37:33.164561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class setEncoder(json.JSONEncoder):\n\n#     def default(self, obj):\n#         return list(obj)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T11:37:33.334299Z","iopub.execute_input":"2022-12-03T11:37:33.334770Z","iopub.status.idle":"2022-12-03T11:37:33.341412Z","shell.execute_reply.started":"2022-12-03T11:37:33.334733Z","shell.execute_reply":"2022-12-03T11:37:33.339429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## How the train_4th_week is split\n\nReading the code below, we can confirm that \n\n- it's a random split of train_4th_week into test and valid sets\n- the last event's 'labels' is used as ground truth or labels","metadata":{}},{"cell_type":"code","source":"# @beartype\n# # def split_events(events: list[dict], split_idx=None):\n# def split_events(events, split_idx=None):    \n#     print(\"starting on split_events --------------------------\")\n#     test_events = ground_truth(deepcopy(events))\n#     if not split_idx:\n#         split_idx = random.randint(1, len(test_events)) # random split on a session\n#     test_events = test_events[:split_idx] # get the test part\n#     labels = test_events[-1]['labels'] # get the ground truth part\n#     for event in test_events: # remove the accumulated ground truth in each earlier event or row\n#         del event['labels']\n#     return test_events, labels","metadata":{"execution":{"iopub.status.busy":"2022-12-03T11:37:33.497569Z","iopub.execute_input":"2022-12-03T11:37:33.498039Z","iopub.status.idle":"2022-12-03T11:37:33.506657Z","shell.execute_reply.started":"2022-12-03T11:37:33.498001Z","shell.execute_reply":"2022-12-03T11:37:33.504539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## How to format and save test and valid sets","metadata":{}},{"cell_type":"code","source":"# @beartype\n# def create_kaggle_testset(sessions: pd.DataFrame, sessions_output: Path, labels_output: Path):\n#     print(\"starting create_kaggle_testset -------------------------\")\n#     last_labels = []\n#     splitted_sessions = []\n\n#     for _, session in tqdm(sessions.iterrows(), desc=\"Creating trimmed testset\", total=len(sessions)):\n#         session = session.to_dict()\n#         splitted_events, labels = split_events(session['events']) # do random split on a session of rows or events\n#         last_labels.append({'session': session['session'], 'labels': labels})\n#         splitted_sessions.append({'session': session['session'], 'events': splitted_events})\n\n#     with open(sessions_output, 'w') as f:\n#         for session in splitted_sessions:\n#             f.write(json.dumps(session) + '\\n')\n\n#     with open(labels_output, 'w') as f:\n#         for label in last_labels:\n#             f.write(json.dumps(label, cls=setEncoder) + '\\n')","metadata":{"execution":{"iopub.status.busy":"2022-12-03T11:37:33.678726Z","iopub.execute_input":"2022-12-03T11:37:33.680166Z","iopub.status.idle":"2022-12-03T11:37:33.691703Z","shell.execute_reply.started":"2022-12-03T11:37:33.680102Z","shell.execute_reply":"2022-12-03T11:37:33.690258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# @beartype\n# def trim_session(session: dict, max_ts: int) -> dict:\n#     session['events'] = [event for event in session['events'] if event['ts'] < max_ts]\n#     return session","metadata":{"execution":{"iopub.status.busy":"2022-12-03T11:37:33.839774Z","iopub.execute_input":"2022-12-03T11:37:33.840982Z","iopub.status.idle":"2022-12-03T11:37:33.847946Z","shell.execute_reply.started":"2022-12-03T11:37:33.840910Z","shell.execute_reply":"2022-12-03T11:37:33.846728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# @beartype\n# def get_max_ts(sessions_path: Path) -> int:\n#     max_ts = float('-inf')\n#     with open(sessions_path) as f:\n#         for line in tqdm(f, desc=\"Finding max timestamp\"):\n#             session = json.loads(line)\n#             max_ts = max(max_ts, session['events'][-1]['ts'])\n#     return max_ts","metadata":{"execution":{"iopub.status.busy":"2022-12-03T11:37:34.045963Z","iopub.execute_input":"2022-12-03T11:37:34.046457Z","iopub.status.idle":"2022-12-03T11:37:34.055182Z","shell.execute_reply.started":"2022-12-03T11:37:34.046417Z","shell.execute_reply":"2022-12-03T11:37:34.053830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# @beartype\n# # def filter_unknown_items(session_path: Path, known_items: set[int]):\n# def filter_unknown_items(session_path: Path, known_items):    \n#     print(\"starting on filter_unknown_items --------------\")\n#     filtered_sessions = []\n#     with open(session_path) as f: # open the test_file for train_4th_week\n#         for line in tqdm(f, desc=\"Filtering unknown items\"): \n#             session = json.loads(line) # every line of the file is a session\n#             # only accept events whose aids are known from the train_3_weeks\n#             session['events'] = [event for event in session['events'] if event['aid'] in known_items] \n#             if len(session['events']) >= 2: # only accept sessions which have more than 1 rows\n#                 filtered_sessions.append(session)\n#     with open(session_path, 'w') as f:\n#         for session in filtered_sessions:\n#             f.write(json.dumps(session) + '\\n')","metadata":{"execution":{"iopub.status.busy":"2022-12-03T11:37:34.200519Z","iopub.execute_input":"2022-12-03T11:37:34.200998Z","iopub.status.idle":"2022-12-03T11:37:34.210118Z","shell.execute_reply.started":"2022-12-03T11:37:34.200960Z","shell.execute_reply":"2022-12-03T11:37:34.208825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## How to split train_3_weeks and train_4th_week\n\n- create a `split_ts` (split time), 2 days in organizer's script or 7 days in Radek's CV notebook. In this notebook I will refer the part before split time as `train_3_weeks` and the part after split time as `train_4th_week`.\n\nReading the code below, we can confirm that \n\n- `train_4th_week` has only sessions which start after the split time\n- `train_3_weeks` take all events whose time before the split time, but drop sessions which have only one event.\n\n**QUESTION**:  \n\n- why remove sessions which have only one event/row? (maybe, they don't give us much information on the aids relationship or user info)\n- why remove the rows or events (of `train_4th_week`) whose aids which are not found in `train_3_weeks`? (maybe, if we didn't learn about some aids in `train_3_weeks`, then we should not make prediction on them?)\n- I don't remember @radek1 has implemented these two minor things above in his CV framework notebook? not important?","metadata":{}},{"cell_type":"code","source":"# @beartype\n# def train_test_split(session_chunks: JsonReader, train_path: Path, test_path: Path, max_ts: int, test_days: int):\n#     print(\"starting on train_test_split ------------------\")\n#     split_millis = test_days * 24 * 60 * 60 * 1000\n#     split_ts = max_ts - split_millis\n#     train_items = set()\n#     Path(train_path).parent.mkdir(parents=True, exist_ok=True)\n#     train_file = open(train_path, \"w\")\n#     Path(test_path).parent.mkdir(parents=True, exist_ok=True)\n#     test_file = open(test_path, \"w\")\n#     for chunk in tqdm(session_chunks, desc=\"Splitting sessions\"):\n#         for _, session in chunk.iterrows():\n#             session = session.to_dict()\n#             if session['events'][0]['ts'] > split_ts: # test_file only accept sessions starting after the split_ts\n#                 test_file.write(json.dumps(session) + \"\\n\")\n#             else:\n#                 session = trim_session(session, split_ts) # train_file keep all sessions and events whose time is before the split_ts\n#                 if len(session['events']) >= 2:\n#                     train_items.update([event['aid'] for event in session['events']])\n#                     train_file.write(json.dumps(session) + \"\\n\") # train_3_weeks removed all sessions which has just 1 row (probably such session is no importance?)\n#     train_file.close()\n#     test_file.close()\n#     # make sure train_4th_week has only sessions which has more than 1 rows and whose aids are known from the train_3_weeks\n#     filter_unknown_items(test_path, train_items) ","metadata":{"execution":{"iopub.status.busy":"2022-12-03T11:37:34.376024Z","iopub.execute_input":"2022-12-03T11:37:34.376505Z","iopub.status.idle":"2022-12-03T11:37:34.391208Z","shell.execute_reply.started":"2022-12-03T11:37:34.376466Z","shell.execute_reply":"2022-12-03T11:37:34.389431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# @beartype\n# def main(train_set: Path, output_path: Path, days: int, seed: int):\n#     random.seed(seed) # ensure the reproducibility\n# #     max_ts = get_max_ts(train_set) # use Radek's dataset can get it with ease\n#     max_ts = 1661723999*1000\n#     session_chunks = pd.read_json(train_set, lines=True, chunksize=100000) # split the jsonl into multiple chunks iterable\n#     train_file = output_path / 'train_sessions.jsonl' # train_3_weeks\n#     test_file_full = output_path / 'test_sessions_full.jsonl' # train_4th_week\n#     train_test_split(session_chunks, train_file, test_file_full, max_ts, days)\n\n#     test_sessions = pd.read_json(test_file_full, lines=True) # train_4th_week\n#     test_sessions_file = output_path / 'test_sessions.jsonl' # test_4th_week\n#     test_labels_file = output_path / 'test_labels.jsonl' # valid_4th_week\n#     create_kaggle_testset(test_sessions, test_sessions_file, test_labels_file)","metadata":{"execution":{"iopub.status.busy":"2022-12-03T11:37:34.540784Z","iopub.execute_input":"2022-12-03T11:37:34.541987Z","iopub.status.idle":"2022-12-03T11:37:34.551038Z","shell.execute_reply.started":"2022-12-03T11:37:34.541902Z","shell.execute_reply":"2022-12-03T11:37:34.549871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# main(Path('/kaggle/input/otto-recommender-system/train.jsonl'), Path('.'), 2, 42) ","metadata":{"execution":{"iopub.status.busy":"2022-12-03T11:37:34.715608Z","iopub.execute_input":"2022-12-03T11:37:34.716087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if __name__ == '__main__':\n#     parser = argparse.ArgumentParser()\n#     parser.add_argument('--train-set', type=Path, required=True)\n#     parser.add_argument('--output-path', type=Path, required=True)\n#     parser.add_argument('--days', type=int, default=2)\n#     parser.add_argument('--seed', type=int, default=42)\n#     args = parser.parse_args()\n#     main(args.train_set, args.output_path, args.days, args.seed)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"pipenv run python -m src.testset --train-set train.jsonl --days 2 --output-path 'out/' --seed 42 ","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}