{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":38760,"databundleVersionId":4493939}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-03T12:05:25.007687Z","iopub.execute_input":"2026-03-03T12:05:25.008059Z","iopub.status.idle":"2026-03-03T12:05:26.240905Z","shell.execute_reply.started":"2026-03-03T12:05:25.008031Z","shell.execute_reply":"2026-03-03T12:05:26.240061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport csv\nimport json\nfrom collections import defaultdict\nfrom pathlib import Path\nfrom typing import Dict, List, Tuple, Optional\n\nimport numpy as np\nfrom tqdm.auto import tqdm\n\n# Быстрый json-парсер (если доступен)\ntry:\n    import orjson  # type: ignore\nexcept Exception:\n    orjson = None\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-03T12:05:26.242794Z","iopub.execute_input":"2026-03-03T12:05:26.243256Z","iopub.status.idle":"2026-03-03T12:05:26.415147Z","shell.execute_reply.started":"2026-03-03T12:05:26.243229Z","shell.execute_reply":"2026-03-03T12:05:26.414318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ----------------------------\n# Объём данных (trade-off quality vs speed)\n# ----------------------------\n# None -> использовать весь train (дольше)\nMAX_SESSIONS_POP = 5_000_000\nMAX_SESSIONS_COVISIT = 5_000_000\n\n# ----------------------------\n# Параметры сессии / пар\n# ----------------------------\nMAX_EVENTS_IN_SESSION = 30            # берем только хвост сессии\nNEXT_PAIR_WINDOW = 20                # пары i -> i+1..i+NEXT_PAIR_WINDOW (ускоряет)\nTIME_WINDOW_MS = 24 * 60 * 60 * 1000 # 1 день для click2click\n\n# ----------------------------\n# Размер item-item соседей\n# ----------------------------\nMAX_NEIGHBORS_CLICK = 120\nMAX_NEIGHBORS_BUY = 200\nPRUNE_AT_CLICK = MAX_NEIGHBORS_CLICK * 4\nPRUNE_AT_BUY = MAX_NEIGHBORS_BUY * 4\n\n# ----------------------------\n# Типы событий (как в OTTO)\n# ----------------------------\nTYPE_MAP = {\"clicks\": 0, \"carts\": 1, \"orders\": 2}\nINV_TYPE = {0: \"clicks\", 1: \"carts\", 2: \"orders\"}\n\n# ----------------------------\n# Веса для микса соседей по типу задачи\n# ----------------------------\n# target: clicks / carts / orders\n# внутри: (w_click2click, w_buy2buy)\nMIX_WEIGHTS = {\n    0: (1.0, 0.2),  # clicks\n    1: (0.6, 1.0),  # carts\n    2: (0.4, 1.2),  # orders\n}\n\n# Усиление \"последних\" айтемов в сессии (recency)\nRECENCY_SELF_BOOST = 2.0\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-03T12:05:26.416414Z","iopub.execute_input":"2026-03-03T12:05:26.416746Z","iopub.status.idle":"2026-03-03T12:05:26.42306Z","shell.execute_reply.started":"2026-03-03T12:05:26.416721Z","shell.execute_reply":"2026-03-03T12:05:26.422198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def find_otto_data_dir() -> Path:\n    candidates = [\n        Path(\"/kaggle/input/competitions/otto-recommender-system\"),\n        Path(\"/kaggle/input/otto-recommender-system\"),\n        Path(\"/kaggle/input/otto-recsys-dataset\"),\n        Path(\"/kaggle/input/recsys-dataset\"),\n    ]\n    for c in candidates:\n        if (c / \"train.jsonl\").exists() and (c / \"test.jsonl\").exists():\n            return c\n\n    # fallback: рекурсивный поиск\n    root = Path(\"/kaggle/input\")\n    if root.exists():\n        for tr in root.rglob(\"train.jsonl\"):\n            d = tr.parent\n            if (d / \"test.jsonl\").exists():\n                return d\n\n    raise FileNotFoundError(\"Не найдено train.jsonl/test.jsonl в /kaggle/input. Проверьте Input в Kaggle.\")\n\n\nDATA_DIR = find_otto_data_dir()\nTRAIN_PATH = DATA_DIR / \"train.jsonl\"\nTEST_PATH = DATA_DIR / \"test.jsonl\"\nSAMPLE_SUB_PATH = DATA_DIR / \"sample_submission.csv\"\n\nprint(\"DATA_DIR:\", DATA_DIR)\nprint(\"TRAIN:\", TRAIN_PATH.exists(), TRAIN_PATH)\nprint(\"TEST :\", TEST_PATH.exists(), TEST_PATH)\nprint(\"SAMPLE:\", SAMPLE_SUB_PATH.exists(), SAMPLE_SUB_PATH)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-03T12:05:26.424263Z","iopub.execute_input":"2026-03-03T12:05:26.424573Z","iopub.status.idle":"2026-03-03T12:05:26.447806Z","shell.execute_reply.started":"2026-03-03T12:05:26.424541Z","shell.execute_reply":"2026-03-03T12:05:26.447006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def _loads(line: bytes):\n    if orjson is not None:\n        return orjson.loads(line)\n    return json.loads(line)\n\n\ndef prune_row_inplace(row: Dict[int, float], keep_n: int) -> None:\n    # Оставляет top-N соседей по весу (in-place).\n    if len(row) <= keep_n:\n        return\n    top = sorted(row.items(), key=lambda x: x[1], reverse=True)[:keep_n]\n    row.clear()\n    row.update(top)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-03T12:05:26.449142Z","iopub.execute_input":"2026-03-03T12:05:26.449444Z","iopub.status.idle":"2026-03-03T12:05:26.45738Z","shell.execute_reply.started":"2026-03-03T12:05:26.449418Z","shell.execute_reply":"2026-03-03T12:05:26.456424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_pop_and_covisitation(\n    train_path: Path,\n    *,\n    max_sessions_pop: Optional[int],\n    max_sessions_covis: Optional[int],\n) -> Tuple[Dict[int, List[int]], Dict[int, Dict[int, float]], Dict[int, Dict[int, float]]]:\n    # popularity отдельно по типам\n    pop_counts = {\n        0: defaultdict(int),\n        1: defaultdict(int),\n        2: defaultdict(int),\n    }\n\n    covis_click: Dict[int, Dict[int, float]] = defaultdict(lambda: defaultdict(float))\n    covis_buy: Dict[int, Dict[int, float]] = defaultdict(lambda: defaultdict(float))\n\n    n_pop = 0\n    n_cov = 0\n\n    with open(train_path, \"rb\") as f:\n        for line in tqdm(f, desc=\"Build pop + covis from train\"):\n            obj = _loads(line)\n            events_raw = obj[\"events\"]\n\n            # компактный формат: [(aid, ts, type_id), ...]\n            events: List[Tuple[int, int, int]] = [\n                (int(e[\"aid\"]), int(e[\"ts\"]), TYPE_MAP[e[\"type\"]]) for e in events_raw\n            ]\n\n            if not events:\n                continue\n\n            # ограничиваем хвост сессии\n            events = events[-MAX_EVENTS_IN_SESSION:]\n\n            # ----------------------------\n            # Popularity (по типам)\n            # ----------------------------\n            if (max_sessions_pop is None) or (n_pop < max_sessions_pop):\n                for aid, ts, t in events:\n                    pop_counts[t][aid] += 1\n                n_pop += 1\n\n            # ----------------------------\n            # Co-visitation (две матрицы)\n            # ----------------------------\n            if (max_sessions_covis is None) or (n_cov < max_sessions_covis):\n                # click2click: учитываем только клики\n                click_events = [(aid, ts) for aid, ts, t in events if t == 0]\n\n                # Для скорости считаем пары только в пределах NEXT_PAIR_WINDOW\n                for i, (ai, ti) in enumerate(click_events):\n                    # пары с ближайшими следующими событиями\n                    for aj, tj in click_events[i + 1 : i + 1 + NEXT_PAIR_WINDOW]:\n                        # окно по времени (если события отсортированы по ts, можно делать break)\n                        if (tj - ti) > TIME_WINDOW_MS:\n                            break\n                        covis_click[ai][aj] += 1.0\n                        covis_click[aj][ai] += 1.0\n\n                    if len(covis_click[ai]) > PRUNE_AT_CLICK:\n                        prune_row_inplace(covis_click[ai], MAX_NEIGHBORS_CLICK)\n\n                # buy2buy: carts + orders, берем уникальные aid (обычно их мало)\n                seen = set()\n                buy_aids: List[int] = []\n                for aid, ts, t in events:\n                    if t in (1, 2) and aid not in seen:\n                        buy_aids.append(aid)\n                        seen.add(aid)\n\n                # пары всех buy-айтемов (маленький список)\n                for i, ai in enumerate(buy_aids):\n                    for aj in buy_aids[i + 1 :]:\n                        covis_buy[ai][aj] += 1.0\n                        covis_buy[aj][ai] += 1.0\n\n                    if len(covis_buy[ai]) > PRUNE_AT_BUY:\n                        prune_row_inplace(covis_buy[ai], MAX_NEIGHBORS_BUY)\n\n                n_cov += 1\n\n            # (опционально) ранний выход, если всё добрали\n            if (max_sessions_pop is not None) and (max_sessions_covis is not None):\n                if (n_pop >= max_sessions_pop) and (n_cov >= max_sessions_covis):\n                    break\n\n    # Финальная обрезка соседей\n    for a in list(covis_click.keys()):\n        prune_row_inplace(covis_click[a], MAX_NEIGHBORS_CLICK)\n    for a in list(covis_buy.keys()):\n        prune_row_inplace(covis_buy[a], MAX_NEIGHBORS_BUY)\n\n    # Top popular (по типам)\n    top_pop_by_type: Dict[int, List[int]] = {}\n    for t in (0, 1, 2):\n        top_pop_by_type[t] = [\n            a for a, _ in sorted(pop_counts[t].items(), key=lambda x: x[1], reverse=True)[:500]\n        ]\n\n    return top_pop_by_type, covis_click, covis_buy\n\n\ntop_pop_by_type, covis_click, covis_buy = build_pop_and_covisitation(\n    TRAIN_PATH,\n    max_sessions_pop=MAX_SESSIONS_POP,\n    max_sessions_covis=MAX_SESSIONS_COVISIT,\n)\n\nprint(\"Top popular sizes:\", {k: len(v) for k, v in top_pop_by_type.items()})\nprint(\"covis_click keys:\", len(covis_click))\nprint(\"covis_buy   keys:\", len(covis_buy))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-03T12:05:26.458579Z","iopub.execute_input":"2026-03-03T12:05:26.459135Z","iopub.status.idle":"2026-03-03T12:15:21.999158Z","shell.execute_reply.started":"2026-03-03T12:05:26.459097Z","shell.execute_reply":"2026-03-03T12:15:21.998306Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def recommend_for_prefix(\n    prefix_events: List[Tuple[int, int, int]],\n    *,\n    target_type_id: int,\n    covis_click: Dict[int, Dict[int, float]],\n    covis_buy: Dict[int, Dict[int, float]],\n    top_pop_by_type: Dict[int, List[int]],\n    k: int = 20,\n) -> List[int]:\n    # последние aid в порядке \"с конца\", но уникальные\n    aids = [a for a, ts, t in prefix_events]\n    seen = set()\n    recent: List[int] = []\n    for a in reversed(aids):\n        if a not in seen:\n            recent.append(a)\n            seen.add(a)\n        if len(recent) == 20:\n            break\n\n    # Начнем список рекомендаций с последних айтемов (часто помогает)\n    recs: List[int] = []\n    used = set()\n\n    for a in recent:\n        if a not in used:\n            recs.append(a)\n            used.add(a)\n        if len(recs) == k:\n            return recs\n\n    # Счётчик баллов для кандидатов\n    scores = defaultdict(float)\n\n    w_click, w_buy = MIX_WEIGHTS[target_type_id]\n\n    for r, a in enumerate(recent):\n        # boost для самого айтема (recency)\n        scores[a] += RECENCY_SELF_BOOST / (r + 1)\n\n        # соседи click2click\n        for nb, w in covis_click.get(a, {}).items():\n            scores[nb] += w_click * w\n\n        # соседи buy2buy\n        for nb, w in covis_buy.get(a, {}).items():\n            scores[nb] += w_buy * w\n\n    # Ранжируем кандидатов, исключая уже добавленные\n    ranked = [a for a, _ in sorted(scores.items(), key=lambda x: x[1], reverse=True) if a not in used]\n\n    for a in ranked:\n        recs.append(a)\n        used.add(a)\n        if len(recs) == k:\n            return recs\n\n    # Добиваем популярными по типу\n    for a in top_pop_by_type[target_type_id]:\n        if a not in used:\n            recs.append(a)\n            used.add(a)\n            if len(recs) == k:\n                return recs\n\n    # Фоллбек (на всякий)\n    if len(recs) < k:\n        for a in top_pop_by_type[target_type_id]:\n            recs.append(a)\n            if len(recs) == k:\n                break\n\n    return recs[:k]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-03T12:15:22.001487Z","iopub.execute_input":"2026-03-03T12:15:22.001783Z","iopub.status.idle":"2026-03-03T12:15:22.013025Z","shell.execute_reply.started":"2026-03-03T12:15:22.001757Z","shell.execute_reply":"2026-03-03T12:15:22.012146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SUBMISSION_PATH = Path(\"/kaggle/working/submission.csv\")\n\ndef write_submission(\n    test_path: Path,\n    out_path: Path,\n    *,\n    covis_click: Dict[int, Dict[int, float]],\n    covis_buy: Dict[int, Dict[int, float]],\n    top_pop_by_type: Dict[int, List[int]],\n):\n    with open(out_path, \"w\", newline=\"\") as f_out:\n        writer = csv.writer(f_out)\n        writer.writerow([\"session_type\", \"labels\"])\n\n        with open(test_path, \"rb\") as f_in:\n            for line in tqdm(f_in, desc=\"Write submission\"):\n                obj = _loads(line)\n                session = int(obj[\"session\"])\n                events_raw = obj[\"events\"]\n\n                prefix_events: List[Tuple[int, int, int]] = [\n                    (int(e[\"aid\"]), int(e[\"ts\"]), TYPE_MAP[e[\"type\"]]) for e in events_raw\n                ]\n                prefix_events = prefix_events[-MAX_EVENTS_IN_SESSION:]\n\n                for target_type_id in (0, 1, 2):\n                    recs = recommend_for_prefix(\n                        prefix_events,\n                        target_type_id=target_type_id,\n                        covis_click=covis_click,\n                        covis_buy=covis_buy,\n                        top_pop_by_type=top_pop_by_type,\n                        k=20,\n                    )\n                    labels = \" \".join(str(a) for a in recs)\n                    writer.writerow([f\"{session}_{INV_TYPE[target_type_id]}\", labels])\n\nwrite_submission(\n    TEST_PATH,\n    SUBMISSION_PATH,\n    covis_click=covis_click,\n    covis_buy=covis_buy,\n    top_pop_by_type=top_pop_by_type,\n)\n\nprint(\"Saved:\", SUBMISSION_PATH, \"exists:\", SUBMISSION_PATH.exists())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-03T12:15:22.014087Z","iopub.execute_input":"2026-03-03T12:15:22.01438Z","iopub.status.idle":"2026-03-03T12:40:12.375508Z","shell.execute_reply.started":"2026-03-03T12:15:22.014354Z","shell.execute_reply":"2026-03-03T12:40:12.374708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nsub = pd.read_csv(SUBMISSION_PATH)\nprint(\"submission rows:\", len(sub))\n\nsample = pd.read_csv(SAMPLE_SUB_PATH) if SAMPLE_SUB_PATH.exists() else None\nif sample is not None:\n    print(\"sample rows    :\", len(sample))\n\ndisplay(sub.head(3))\n\n# 20 лейблов в каждой строке\nlens = sub[\"labels\"].astype(str).str.split().str.len()\nprint(\"min labels:\", lens.min(), \"max labels:\", lens.max())\nassert (lens == 20).all()\n\n# нет пустых строк / NaN\nassert sub[\"labels\"].isna().sum() == 0\nassert (sub[\"labels\"].str.len() > 0).all()\n\n# строгий паттерн: 20 целых чисел через пробел\nassert sub[\"labels\"].str.fullmatch(r\"(?:\\d+\\s){19}\\d+\").notna().all()\n\n# уникальность ключа\nassert sub[\"session_type\"].is_unique\n\nif sample is not None:\n    assert len(sub) == len(sample)\n\nprint(\"OK: submission format looks valid.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-03T12:40:12.376597Z","iopub.execute_input":"2026-03-03T12:40:12.376934Z","iopub.status.idle":"2026-03-03T12:42:04.076166Z","shell.execute_reply.started":"2026-03-03T12:40:12.376908Z","shell.execute_reply":"2026-03-03T12:42:04.075396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}