{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":38760,"databundleVersionId":4493939}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import json\nfrom collections import defaultdict, Counter\nfrom tqdm import tqdm\n\nTYPE_MAP = {\"clicks\": 0, \"carts\": 1, \"orders\": 2}\n\nnext_AIDs = defaultdict(Counter)\n\ntrain_path = '/kaggle/input/competitions/otto-recommender-system/train.jsonl'\n\nwith open(train_path, 'r') as f:\n    for line in tqdm(f):\n        obj = json.loads(line)\n        events = obj[\"events\"]\n\n        # сортируем по времени, если надо\n        events = sorted(events, key=lambda x: x[\"ts\"])\n\n        # берём только aid из последних действий\n        aids = [ev[\"aid\"] for ev in events][-30:]\n\n        # убираем подряд идущие дубли не обязательно, но полезно\n        aids = list(dict.fromkeys(aids))\n\n        for i in range(len(aids)):\n            for j in range(i + 1, min(i + 6, len(aids))):\n                if aids[i] != aids[j]:\n                    next_AIDs[aids[i]][aids[j]] += 1\n                    next_AIDs[aids[j]][aids[i]] += 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-17T21:17:47.042328Z","iopub.execute_input":"2026-03-17T21:17:47.042611Z","iopub.status.idle":"2026-03-17T21:34:32.611513Z","shell.execute_reply.started":"2026-03-17T21:17:47.042586Z","shell.execute_reply":"2026-03-17T21:34:32.609806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"top_next_AIDs = {}\n\nfor aid, counter in tqdm(next_AIDs.items()):\n    top_next_AIDs[aid] = [x for x, _ in counter.most_common(40)]\n\ndel next_AIDs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-17T21:43:46.263239Z","iopub.execute_input":"2026-03-17T21:43:46.264021Z","iopub.status.idle":"2026-03-17T21:47:29.657815Z","shell.execute_reply.started":"2026-03-17T21:43:46.263940Z","shell.execute_reply":"2026-03-17T21:47:29.656665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import Counter\n\ntop_clicks_counter = Counter()\ntop_orders_counter = Counter()\n\nwith open(train_path, 'r') as f:\n    for line in tqdm(f):\n        obj = json.loads(line)\n        for ev in obj[\"events\"]:\n            aid = ev[\"aid\"]\n            ev_type = ev[\"type\"]\n\n            if ev_type == \"clicks\":\n                top_clicks_counter[aid] += 1\n            else:\n                top_orders_counter[aid] += 1\n\ntop_clicks = [aid for aid, _ in top_clicks_counter.most_common(20)]\ntop_orders = [aid for aid, _ in top_orders_counter.most_common(20)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-17T21:47:29.659744Z","iopub.execute_input":"2026-03-17T21:47:29.660186Z","iopub.status.idle":"2026-03-17T21:55:45.967483Z","shell.execute_reply.started":"2026-03-17T21:47:29.660158Z","shell.execute_reply":"2026-03-17T21:55:45.966264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ntest_path = '/kaggle/input/competitions/otto-recommender-system/test.jsonl'\n\npred_rows = []\n\nwith open(test_path, 'r') as f:\n    for line in tqdm(f):\n        obj = json.loads(line)\n        session = obj[\"session\"]\n        events = sorted(obj[\"events\"], key=lambda x: x[\"ts\"])\n\n        aids = [ev[\"aid\"] for ev in events]\n        types = [ev[\"type\"] for ev in events]\n\n        # reverse unique recent items\n        unique_aids = list(dict.fromkeys(aids[::-1]))\n\n        # clicks\n        click_candidates = unique_aids.copy()\n        for aid in unique_aids[:10]:\n            click_candidates += top_next_AIDs.get(aid, [])\n\n        click_preds = list(dict.fromkeys(click_candidates))\n        click_preds = click_preds[:20]\n        if len(click_preds) < 20:\n            click_preds += [x for x in top_clicks if x not in click_preds][:20 - len(click_preds)]\n\n        # carts/orders\n        buy_candidates = unique_aids.copy()\n        for aid in unique_aids[:10]:\n            buy_candidates += top_next_AIDs.get(aid, [])\n\n        buy_preds = list(dict.fromkeys(buy_candidates))\n        buy_preds = buy_preds[:20]\n        if len(buy_preds) < 20:\n            buy_preds += [x for x in top_orders if x not in buy_preds][:20 - len(buy_preds)]\n\n        pred_rows.append([f\"{session}_clicks\", \" \".join(map(str, click_preds))])\n        pred_rows.append([f\"{session}_carts\", \" \".join(map(str, buy_preds))])\n        pred_rows.append([f\"{session}_orders\", \" \".join(map(str, buy_preds))])\n\nsubmission = pd.DataFrame(pred_rows, columns=[\"session_type\", \"labels\"])\nsubmission.to_csv(\"submission.csv\", index=False)\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-17T21:55:45.969173Z","iopub.execute_input":"2026-03-17T21:55:45.969616Z","iopub.status.idle":"2026-03-17T21:58:41.461604Z","shell.execute_reply.started":"2026-03-17T21:55:45.969576Z","shell.execute_reply":"2026-03-17T21:58:41.460687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}