{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":38760,"databundleVersionId":4493939}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:03:30.179569Z","iopub.execute_input":"2026-03-16T22:03:30.179887Z","iopub.status.idle":"2026-03-16T22:03:31.365027Z","shell.execute_reply.started":"2026-03-16T22:03:30.179861Z","shell.execute_reply":"2026-03-16T22:03:31.364027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nimport csv\nimport json\nfrom math import log2\nfrom pathlib import Path\nfrom collections import Counter, defaultdict\nfrom tqdm.auto import tqdm\n\nDATA_DIR = Path(\"/kaggle/input/competitions/otto-recommender-system\")\nTRAIN_PATH = DATA_DIR / \"train.jsonl\"\nTEST_PATH = DATA_DIR / \"test.jsonl\"\nOUT_PATH = Path(\"/kaggle/working/submission.csv\")\n\nMAX_TRAIN_SESSIONS_FOR_COVISIT = 300_000\nMAX_SESSION_AIDS = 20\nCLICK_WINDOW = 5\nBUY_WINDOW = 5\nTOPN_NEIGHBORS = 40\nANCHORS_PER_SESSION = 5\nBUY_ANCHORS_PER_SESSION = 5\nTOP_POPULAR = 50\n\nTYPE_STR_TO_INT = {\"clicks\": 0, \"carts\": 1, \"orders\": 2}\n\nprint(TRAIN_PATH.exists(), TEST_PATH.exists())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:07:05.059175Z","iopub.execute_input":"2026-03-16T22:07:05.059493Z","iopub.status.idle":"2026-03-16T22:07:05.236707Z","shell.execute_reply.started":"2026-03-16T22:07:05.059467Z","shell.execute_reply":"2026-03-16T22:07:05.235862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def event_type_to_int(t):\n    if isinstance(t, str):\n        return TYPE_STR_TO_INT[t]\n    return int(t)\n\n\ndef count_lines(path):\n    cnt = 0\n    with open(path, \"r\") as f:\n        for _ in f:\n            cnt += 1\n    return cnt\n\n\ndef get_recent_unique(aids, types=None, allowed_types=None, max_len=20):\n    seen = set()\n    result = []\n\n    if types is None:\n        iterator = zip(reversed(aids), [None] * len(aids))\n    else:\n        iterator = zip(reversed(aids), reversed(types))\n\n    for aid, t in iterator:\n        if allowed_types is not None and t not in allowed_types:\n            continue\n        if aid in seen:\n            continue\n        seen.add(aid)\n        result.append(aid)\n        if len(result) >= max_len:\n            break\n\n    return result\n\n\ndef add_pairs(counter_dict, items, window):\n    n = len(items)\n    for i in range(n):\n        a = items[i]\n        for j in range(i + 1, min(n, i + 1 + window)):\n            b = items[j]\n            if a == b:\n                continue\n            counter_dict[a][b] += 1\n            counter_dict[b][a] += 1\n\n\ndef counterdict_to_topneighbors(counter_dict, topk=40):\n    out = {}\n    for aid, ctr in tqdm(counter_dict.items(), desc=\"Trim neighbors\"):\n        out[aid] = [x for x, _ in ctr.most_common(topk)]\n    return out\n\n\ndef add_self_scores(scores, aids, types, type_weights):\n    for pos, (aid, t) in enumerate(zip(reversed(aids), reversed(types))):\n        recency = 1.0 / log2(pos + 2.0)\n        scores[aid] += recency * type_weights[t]\n\n\ndef add_neighbor_scores(scores, anchors, neighbor_map, mult):\n    for a_pos, aid in enumerate(anchors):\n        nbrs = neighbor_map.get(aid)\n        if not nbrs:\n            continue\n        for n_pos, nbr in enumerate(nbrs):\n            if nbr == aid:\n                continue\n            scores[nbr] += mult / ((a_pos + 1) * (n_pos + 1))\n\n\ndef topk_from_scores(scores, fallback, k=20):\n    result = []\n    seen = set()\n\n    for aid, _ in sorted(scores.items(), key=lambda x: (-x[1], x[0])):\n        if aid in seen:\n            continue\n        seen.add(aid)\n        result.append(aid)\n        if len(result) == k:\n            return result\n\n    for aid in fallback:\n        if aid in seen:\n            continue\n        seen.add(aid)\n        result.append(aid)\n        if len(result) == k:\n            return result\n\n    return result[:k]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:07:30.477951Z","iopub.execute_input":"2026-03-16T22:07:30.478732Z","iopub.status.idle":"2026-03-16T22:07:30.567065Z","shell.execute_reply.started":"2026-03-16T22:07:30.478699Z","shell.execute_reply":"2026-03-16T22:07:30.566008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_train_sessions = count_lines(TRAIN_PATH)\nstart_idx = max(0, n_train_sessions - MAX_TRAIN_SESSIONS_FOR_COVISIT)\n\nprint(\"Total train sessions:\", n_train_sessions)\nprint(\"Using last sessions for covisit:\", MAX_TRAIN_SESSIONS_FOR_COVISIT)\nprint(\"Start line index:\", start_idx)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:07:42.992430Z","iopub.execute_input":"2026-03-16T22:07:42.992942Z","iopub.status.idle":"2026-03-16T22:10:36.881182Z","shell.execute_reply.started":"2026-03-16T22:07:42.992909Z","shell.execute_reply":"2026-03-16T22:10:36.880192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"popular_clicks = Counter()\npopular_buys = Counter()\n\nclick_neighbors_counter = defaultdict(Counter)\nbuy_neighbors_counter = defaultdict(Counter)\n\nwith open(TRAIN_PATH, \"r\") as f:\n    for idx, line in enumerate(tqdm(f, total=n_train_sessions, desc=\"Parse train\")):\n        if idx < start_idx:\n            continue\n\n        obj = json.loads(line)\n        events = obj[\"events\"]\n\n        events = sorted(events, key=lambda x: x[\"ts\"])\n\n        aids = [int(ev[\"aid\"]) for ev in events]\n        types = [event_type_to_int(ev[\"type\"]) for ev in events]\n\n        # Popularity\n        for aid, t in zip(aids, types):\n            if t == 0:\n                popular_clicks[aid] += 1\n            else:\n                popular_buys[aid] += 1\n\n        # Последние уникальные айтемы по всей сессии\n        recent_all = get_recent_unique(\n            aids=aids,\n            types=types,\n            allowed_types=None,\n            max_len=MAX_SESSION_AIDS\n        )\n        add_pairs(click_neighbors_counter, recent_all, CLICK_WINDOW)\n\n        # Последние уникальные buy-oriented айтемы\n        recent_buy = get_recent_unique(\n            aids=aids,\n            types=types,\n            allowed_types={1, 2},\n            max_len=MAX_SESSION_AIDS\n        )\n        if len(recent_buy) >= 2:\n            add_pairs(buy_neighbors_counter, recent_buy, BUY_WINDOW)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:10:53.472043Z","iopub.execute_input":"2026-03-16T22:10:53.473372Z","iopub.status.idle":"2026-03-16T22:11:56.702984Z","shell.execute_reply.started":"2026-03-16T22:10:53.473320Z","shell.execute_reply":"2026-03-16T22:11:56.701951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"top_click_neighbors = counterdict_to_topneighbors(\n    click_neighbors_counter,\n    topk=TOPN_NEIGHBORS\n)\n\ntop_buy_neighbors = counterdict_to_topneighbors(\n    buy_neighbors_counter,\n    topk=TOPN_NEIGHBORS\n)\n\npopular_clicks_list = [aid for aid, _ in popular_clicks.most_common(TOP_POPULAR)]\n\nif len(popular_buys) > 0:\n    popular_buys_list = [aid for aid, _ in popular_buys.most_common(TOP_POPULAR)]\nelse:\n    popular_buys_list = popular_clicks_list[:]\n\nprint(\"click neighbor anchors:\", len(top_click_neighbors))\nprint(\"buy neighbor anchors:\", len(top_buy_neighbors))\nprint(\"popular_clicks sample:\", popular_clicks_list[:10])\nprint(\"popular_buys sample:\", popular_buys_list[:10])\n\ndel click_neighbors_counter, buy_neighbors_counter, popular_clicks, popular_buys\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:12:19.330028Z","iopub.execute_input":"2026-03-16T22:12:19.331093Z","iopub.status.idle":"2026-03-16T22:12:25.482772Z","shell.execute_reply.started":"2026-03-16T22:12:19.331051Z","shell.execute_reply":"2026-03-16T22:12:25.481691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict_one_session(events):\n    events = sorted(events, key=lambda x: x[\"ts\"])\n    aids = [int(ev[\"aid\"]) for ev in events]\n    types = [event_type_to_int(ev[\"type\"]) for ev in events]\n\n    recent_all = get_recent_unique(\n        aids=aids,\n        types=types,\n        allowed_types=None,\n        max_len=MAX_SESSION_AIDS\n    )\n\n    recent_buy = get_recent_unique(\n        aids=aids,\n        types=types,\n        allowed_types={1, 2},\n        max_len=MAX_SESSION_AIDS\n    )\n\n    click_scores = defaultdict(float)\n    cart_scores = defaultdict(float)\n    order_scores = defaultdict(float)\n\n    add_self_scores(click_scores, aids, types, {0: 1.0, 1: 4.0, 2: 2.0})\n    add_self_scores(cart_scores, aids, types, {0: 1.0, 1: 5.0, 2: 4.0})\n    add_self_scores(order_scores, aids, types, {0: 1.0, 1: 6.0, 2: 6.0})\n\n    add_neighbor_scores(\n        click_scores,\n        recent_all[:ANCHORS_PER_SESSION],\n        top_click_neighbors,\n        mult=0.7,\n    )\n    add_neighbor_scores(\n        cart_scores,\n        recent_all[:ANCHORS_PER_SESSION],\n        top_click_neighbors,\n        mult=0.25,\n    )\n    add_neighbor_scores(\n        order_scores,\n        recent_all[:ANCHORS_PER_SESSION],\n        top_click_neighbors,\n        mult=0.15,\n    )\n\n    buy_anchors = recent_buy if len(recent_buy) > 0 else recent_all\n    add_neighbor_scores(\n        cart_scores,\n        buy_anchors[:BUY_ANCHORS_PER_SESSION],\n        top_buy_neighbors,\n        mult=0.8,\n    )\n    add_neighbor_scores(\n        order_scores,\n        buy_anchors[:BUY_ANCHORS_PER_SESSION],\n        top_buy_neighbors,\n        mult=1.0,\n    )\n\n    click_preds = topk_from_scores(click_scores, popular_clicks_list, k=20)\n    cart_preds = topk_from_scores(cart_scores, popular_buys_list, k=20)\n    order_preds = topk_from_scores(order_scores, popular_buys_list, k=20)\n\n    return click_preds, cart_preds, order_preds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:12:25.484519Z","iopub.execute_input":"2026-03-16T22:12:25.484852Z","iopub.status.idle":"2026-03-16T22:12:25.495052Z","shell.execute_reply.started":"2026-03-16T22:12:25.484825Z","shell.execute_reply":"2026-03-16T22:12:25.494107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_test_sessions = count_lines(TEST_PATH)\nprint(\"Total test sessions:\", n_test_sessions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:12:27.888585Z","iopub.execute_input":"2026-03-16T22:12:27.889492Z","iopub.status.idle":"2026-03-16T22:12:32.785976Z","shell.execute_reply.started":"2026-03-16T22:12:27.889458Z","shell.execute_reply":"2026-03-16T22:12:32.785107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(OUT_PATH, \"w\", newline=\"\") as fout:\n    writer = csv.writer(fout)\n    writer.writerow([\"session_type\", \"labels\"])\n\n    with open(TEST_PATH, \"r\") as f:\n        for line in tqdm(f, total=n_test_sessions, desc=\"Predict test\"):\n            obj = json.loads(line)\n            session = int(obj[\"session\"])\n            events = obj[\"events\"]\n\n            pred_clicks, pred_carts, pred_orders = predict_one_session(events)\n\n            writer.writerow([f\"{session}_clicks\", \" \".join(map(str, pred_clicks))])\n            writer.writerow([f\"{session}_carts\", \" \".join(map(str, pred_carts))])\n            writer.writerow([f\"{session}_orders\", \" \".join(map(str, pred_orders))])\n\nprint(\"Saved to:\", OUT_PATH)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:24:41.230030Z","iopub.execute_input":"2026-03-16T22:24:41.230396Z","iopub.status.idle":"2026-03-16T22:29:46.685240Z","shell.execute_reply.started":"2026-03-16T22:24:41.230366Z","shell.execute_reply":"2026-03-16T22:29:46.683546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_preview = pd.read_csv(OUT_PATH, nrows=10)\nsubmission_preview","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:29:46.687182Z","iopub.execute_input":"2026-03-16T22:29:46.687652Z","iopub.status.idle":"2026-03-16T22:29:46.733504Z","shell.execute_reply.started":"2026-03-16T22:29:46.687588Z","shell.execute_reply":"2026-03-16T22:29:46.732489Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Тестирование","metadata":{}},{"cell_type":"code","source":"import json\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nfrom collections import Counter, defaultdict\nfrom math import log2\nimport gc\n\nDATA_DIR = Path(\"/kaggle/input/competitions/otto-recommender-system\")\nTRAIN_PATH = DATA_DIR / \"train.jsonl\"\n\nTYPE_STR_TO_INT = {\"clicks\": 0, \"carts\": 1, \"orders\": 2}\n\ndef event_type_to_int(t):\n    if isinstance(t, str):\n        return TYPE_STR_TO_INT[t]\n    return int(t)\n\ndef count_lines(path):\n    cnt = 0\n    with open(path, \"r\") as f:\n        for _ in f:\n            cnt += 1\n    return cnt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:29:46.734966Z","iopub.execute_input":"2026-03-16T22:29:46.735259Z","iopub.status.idle":"2026-03-16T22:29:46.741876Z","shell.execute_reply.started":"2026-03-16T22:29:46.735235Z","shell.execute_reply":"2026-03-16T22:29:46.740996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TOTAL_TRAIN = count_lines(TRAIN_PATH)\nTOTAL_TRAIN","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:29:46.743262Z","iopub.execute_input":"2026-03-16T22:29:46.743714Z","iopub.status.idle":"2026-03-16T22:29:58.876454Z","shell.execute_reply.started":"2026-03-16T22:29:46.743677Z","shell.execute_reply":"2026-03-16T22:29:58.875513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"VAL_SESSIONS = 50_000\nTRAIN_SESSIONS_FOR_STATS = 200_000\n\nstart_idx = max(0, TOTAL_TRAIN - (VAL_SESSIONS + TRAIN_SESSIONS_FOR_STATS))\nsplit_idx = max(0, TOTAL_TRAIN - VAL_SESSIONS)\n\nprint(\"start_idx:\", start_idx)\nprint(\"split_idx:\", split_idx)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:29:58.877882Z","iopub.execute_input":"2026-03-16T22:29:58.878237Z","iopub.status.idle":"2026-03-16T22:29:58.883802Z","shell.execute_reply.started":"2026-03-16T22:29:58.878201Z","shell.execute_reply":"2026-03-16T22:29:58.882803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_recent_unique(aids, types=None, allowed_types=None, max_len=20):\n    seen = set()\n    result = []\n\n    if types is None:\n        iterator = zip(reversed(aids), [None] * len(aids))\n    else:\n        iterator = zip(reversed(aids), reversed(types))\n\n    for aid, t in iterator:\n        if allowed_types is not None and t not in allowed_types:\n            continue\n        if aid in seen:\n            continue\n        seen.add(aid)\n        result.append(aid)\n        if len(result) >= max_len:\n            break\n\n    return result\n\n\ndef add_pairs(counter_dict, items, window):\n    n = len(items)\n    for i in range(n):\n        a = items[i]\n        for j in range(i + 1, min(n, i + 1 + window)):\n            b = items[j]\n            if a == b:\n                continue\n            counter_dict[a][b] += 1\n            counter_dict[b][a] += 1\n\n\ndef counterdict_to_topneighbors(counter_dict, topk=40):\n    out = {}\n    for aid, ctr in tqdm(counter_dict.items(), desc=\"Trim neighbors\"):\n        out[aid] = [x for x, _ in ctr.most_common(topk)]\n    return out\n\n\ndef add_self_scores(scores, aids, types, type_weights):\n    for pos, (aid, t) in enumerate(zip(reversed(aids), reversed(types))):\n        recency = 1.0 / log2(pos + 2.0)\n        scores[aid] += recency * type_weights[t]\n\n\ndef add_neighbor_scores(scores, anchors, neighbor_map, mult):\n    for a_pos, aid in enumerate(anchors):\n        nbrs = neighbor_map.get(aid)\n        if not nbrs:\n            continue\n        for n_pos, nbr in enumerate(nbrs):\n            if nbr == aid:\n                continue\n            scores[nbr] += mult / ((a_pos + 1) * (n_pos + 1))\n\n\ndef topk_from_scores(scores, fallback, k=20):\n    result = []\n    seen = set()\n\n    for aid, _ in sorted(scores.items(), key=lambda x: (-x[1], x[0])):\n        if aid in seen:\n            continue\n        seen.add(aid)\n        result.append(aid)\n        if len(result) == k:\n            return result\n\n    for aid in fallback:\n        if aid in seen:\n            continue\n        seen.add(aid)\n        result.append(aid)\n        if len(result) == k:\n            return result\n\n    return result[:k]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:29:58.884977Z","iopub.execute_input":"2026-03-16T22:29:58.885284Z","iopub.status.idle":"2026-03-16T22:29:58.910735Z","shell.execute_reply.started":"2026-03-16T22:29:58.885259Z","shell.execute_reply":"2026-03-16T22:29:58.909543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"MAX_SESSION_AIDS = 20\nCLICK_WINDOW = 5\nBUY_WINDOW = 5\nTOPN_NEIGHBORS = 40\nTOP_POPULAR = 50\n\npopular_clicks = Counter()\npopular_buys = Counter()\nclick_neighbors_counter = defaultdict(Counter)\nbuy_neighbors_counter = defaultdict(Counter)\n\nwith open(TRAIN_PATH, \"r\") as f:\n    for idx, line in enumerate(tqdm(f, total=TOTAL_TRAIN, desc=\"Build train stats\")):\n        if idx < start_idx:\n            continue\n        if idx >= split_idx:\n            break\n\n        obj = json.loads(line)\n        events = sorted(obj[\"events\"], key=lambda x: x[\"ts\"])\n\n        aids = [int(ev[\"aid\"]) for ev in events]\n        types = [event_type_to_int(ev[\"type\"]) for ev in events]\n\n        for aid, t in zip(aids, types):\n            if t == 0:\n                popular_clicks[aid] += 1\n            else:\n                popular_buys[aid] += 1\n\n        recent_all = get_recent_unique(aids, types, allowed_types=None, max_len=MAX_SESSION_AIDS)\n        add_pairs(click_neighbors_counter, recent_all, CLICK_WINDOW)\n\n        recent_buy = get_recent_unique(aids, types, allowed_types={1, 2}, max_len=MAX_SESSION_AIDS)\n        if len(recent_buy) >= 2:\n            add_pairs(buy_neighbors_counter, recent_buy, BUY_WINDOW)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:29:58.912203Z","iopub.execute_input":"2026-03-16T22:29:58.912579Z","iopub.status.idle":"2026-03-16T22:30:24.349001Z","shell.execute_reply.started":"2026-03-16T22:29:58.912537Z","shell.execute_reply":"2026-03-16T22:30:24.347927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"top_click_neighbors = counterdict_to_topneighbors(click_neighbors_counter, topk=TOPN_NEIGHBORS)\ntop_buy_neighbors = counterdict_to_topneighbors(buy_neighbors_counter, topk=TOPN_NEIGHBORS)\n\npopular_clicks_list = [aid for aid, _ in popular_clicks.most_common(TOP_POPULAR)]\npopular_buys_list = [aid for aid, _ in popular_buys.most_common(TOP_POPULAR)]\nif len(popular_buys_list) == 0:\n    popular_buys_list = popular_clicks_list[:]\n\ndel click_neighbors_counter, buy_neighbors_counter\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:30:24.350417Z","iopub.execute_input":"2026-03-16T22:30:24.350965Z","iopub.status.idle":"2026-03-16T22:30:28.578228Z","shell.execute_reply.started":"2026-03-16T22:30:24.350934Z","shell.execute_reply":"2026-03-16T22:30:28.577331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ANCHORS_PER_SESSION = 5\nBUY_ANCHORS_PER_SESSION = 5\n\ndef predict_one_session(events):\n    events = sorted(events, key=lambda x: x[\"ts\"])\n    aids = [int(ev[\"aid\"]) for ev in events]\n    types = [event_type_to_int(ev[\"type\"]) for ev in events]\n\n    recent_all = get_recent_unique(aids, types, allowed_types=None, max_len=MAX_SESSION_AIDS)\n    recent_buy = get_recent_unique(aids, types, allowed_types={1, 2}, max_len=MAX_SESSION_AIDS)\n\n    click_scores = defaultdict(float)\n    cart_scores = defaultdict(float)\n    order_scores = defaultdict(float)\n\n    add_self_scores(click_scores, aids, types, {0: 1.0, 1: 4.0, 2: 2.0})\n    add_self_scores(cart_scores, aids, types, {0: 1.0, 1: 5.0, 2: 4.0})\n    add_self_scores(order_scores, aids, types, {0: 1.0, 1: 6.0, 2: 6.0})\n\n    add_neighbor_scores(click_scores, recent_all[:ANCHORS_PER_SESSION], top_click_neighbors, mult=0.7)\n    add_neighbor_scores(cart_scores, recent_all[:ANCHORS_PER_SESSION], top_click_neighbors, mult=0.25)\n    add_neighbor_scores(order_scores, recent_all[:ANCHORS_PER_SESSION], top_click_neighbors, mult=0.15)\n\n    buy_anchors = recent_buy if len(recent_buy) > 0 else recent_all\n    add_neighbor_scores(cart_scores, buy_anchors[:BUY_ANCHORS_PER_SESSION], top_buy_neighbors, mult=0.8)\n    add_neighbor_scores(order_scores, buy_anchors[:BUY_ANCHORS_PER_SESSION], top_buy_neighbors, mult=1.0)\n\n    click_preds = topk_from_scores(click_scores, popular_clicks_list, k=20)\n    cart_preds = topk_from_scores(cart_scores, popular_buys_list, k=20)\n    order_preds = topk_from_scores(order_scores, popular_buys_list, k=20)\n\n    return click_preds, cart_preds, order_preds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:30:28.579357Z","iopub.execute_input":"2026-03-16T22:30:28.579719Z","iopub.status.idle":"2026-03-16T22:30:28.589675Z","shell.execute_reply.started":"2026-03-16T22:30:28.579681Z","shell.execute_reply":"2026-03-16T22:30:28.588702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\n\nrandom.seed(42)\n\ndef make_prefix_and_labels(events, min_prefix_len=2):\n    events = sorted(events, key=lambda x: x[\"ts\"])\n    n = len(events)\n\n    if n < min_prefix_len + 1:\n        return None\n\n    cut = random.randint(min_prefix_len, n - 1)\n    prefix = events[:cut]\n    suffix = events[cut:]\n\n    click_label = [int(suffix[0][\"aid\"])] if len(suffix) > 0 else []\n    carts = []\n    orders = []\n\n    for ev in suffix:\n        aid = int(ev[\"aid\"])\n        t = event_type_to_int(ev[\"type\"])\n        if t == 1:\n            carts.append(aid)\n        elif t == 2:\n            orders.append(aid)\n\n    carts = list(dict.fromkeys(carts))\n    orders = list(dict.fromkeys(orders))\n\n    return prefix, {\"clicks\": click_label, \"carts\": carts, \"orders\": orders}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:30:28.590848Z","iopub.execute_input":"2026-03-16T22:30:28.591215Z","iopub.status.idle":"2026-03-16T22:30:28.610569Z","shell.execute_reply.started":"2026-03-16T22:30:28.591176Z","shell.execute_reply":"2026-03-16T22:30:28.609629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hits_clicks = 0\ntrue_clicks = 0\n\nhits_carts = 0\ntrue_carts = 0\n\nhits_orders = 0\ntrue_orders = 0\n\nn_eval = 0\n\nwith open(TRAIN_PATH, \"r\") as f:\n    for idx, line in enumerate(tqdm(f, total=TOTAL_TRAIN, desc=\"Validate\")):\n        if idx < split_idx:\n            continue\n\n        obj = json.loads(line)\n        session = int(obj[\"session\"])\n        events = obj[\"events\"]\n\n        sample = make_prefix_and_labels(events, min_prefix_len=2)\n        if sample is None:\n            continue\n\n        prefix, labels = sample\n        pred_clicks, pred_carts, pred_orders = predict_one_session(prefix)\n\n        # clicks\n        if len(labels[\"clicks\"]) > 0:\n            hits_clicks += len(set(pred_clicks[:20]) & set(labels[\"clicks\"]))\n            true_clicks += min(20, len(labels[\"clicks\"]))\n\n        # carts\n        if len(labels[\"carts\"]) > 0:\n            hits_carts += len(set(pred_carts[:20]) & set(labels[\"carts\"]))\n            true_carts += min(20, len(labels[\"carts\"]))\n\n        # orders\n        if len(labels[\"orders\"]) > 0:\n            hits_orders += len(set(pred_orders[:20]) & set(labels[\"orders\"]))\n            true_orders += min(20, len(labels[\"orders\"]))\n\n        n_eval += 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:30:28.611900Z","iopub.execute_input":"2026-03-16T22:30:28.612303Z","iopub.status.idle":"2026-03-16T22:30:49.999334Z","shell.execute_reply.started":"2026-03-16T22:30:28.612268Z","shell.execute_reply":"2026-03-16T22:30:49.998339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"recall_clicks = hits_clicks / true_clicks if true_clicks > 0 else 0.0\nrecall_carts = hits_carts / true_carts if true_carts > 0 else 0.0\nrecall_orders = hits_orders / true_orders if true_orders > 0 else 0.0\nweighted_recall = 0.10 * recall_clicks + 0.30 * recall_carts + 0.60 * recall_orders\n\nprint(\"sessions evaluated:\", n_eval)\nprint(\"recall_clicks@20 :\", round(recall_clicks, 6))\nprint(\"recall_carts@20  :\", round(recall_carts, 6))\nprint(\"recall_orders@20 :\", round(recall_orders, 6))\nprint(\"weighted_recall@20:\", round(weighted_recall, 6))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-16T22:30:50.001455Z","iopub.execute_input":"2026-03-16T22:30:50.001800Z","iopub.status.idle":"2026-03-16T22:30:50.009531Z","shell.execute_reply.started":"2026-03-16T22:30:50.001773Z","shell.execute_reply":"2026-03-16T22:30:50.008652Z"}},"outputs":[],"execution_count":null}]}