{"nbformat": 4, "nbformat_minor": 5, "metadata": {"kernelspec": {"display_name": "Python 3", "language": "python", "name": "python3"}, "language_info": {"name": "python", "version": "3.10.0"}}, "cells": [{"cell_type": "markdown", "metadata": {}, "source": "# OTTO \u2013 baseline recommender\n\nCo-visitation matrix, memory-efficient version.\nOnly track co-visits for aids that actually appear in test sessions."}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "import pandas as pd\nimport numpy as np\nimport pyarrow.parquet as pq\nfrom collections import defaultdict\nimport gc\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\n\nDATA = Path(\"/kaggle/input/otto-full-optimized-memory-footprint\")\nCOMP = Path(\"/kaggle/input/otto-recommender-system\")\nprint(\"ok\")"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "# load test first \u2014 we only need covisit for test aids\nprint(\"loading test...\")\ntest = pd.read_parquet(DATA / \"test.parquet\")\nprint(f\"test shape: {test.shape}\")\n\ntest_aids = set(test[\"aid\"].unique().tolist())\nprint(f\"unique aids in test: {len(test_aids):,}\")\n\n# prepare test for fast iteration\ntest = test.sort_values([\"session\", \"ts\"]).reset_index(drop=True)\nsess_arr = test[\"session\"].values\naid_arr  = test[\"aid\"].values\ntype_arr = test[\"type\"].values\nboundaries = np.concatenate([[0], np.where(np.diff(sess_arr) != 0)[0] + 1, [len(sess_arr)]])\nunique_sess = sess_arr[boundaries[:-1]]\nprint(f\"test sessions: {len(unique_sess):,}\")"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "SEED_N   = 20   # reduced from 30 to save RAM\nTIME_WIN = 24 * 3600\nMAX_KEEP = 15   # reduced from 40\n\n# key optimization: only track covisit where SOURCE is a test aid\ncovisit_click = defaultdict(lambda: defaultdict(int))\ncovisit_buy   = defaultdict(lambda: defaultdict(int))\n\nclick_pop  = defaultdict(int)\nbuy_pop    = defaultdict(int)\norder_pop  = defaultdict(int)\n\nTRIM_EVERY = 2   # trim every N batches to prevent RAM explosion\nbatch_size = 2_000_000\n\npf = pq.ParquetFile(DATA / \"train.parquet\")\nn_sessions_total = 0\nprev_session = None\nbuf_aids, buf_tss, buf_types = [], [], []\n\ndef process_session(aids, tss, types):\n    global n_sessions_total\n    n_sessions_total += 1\n    for t, aid in zip(types, aids):\n        if t == 0:     click_pop[aid] += 1\n        if t in (1,2): buy_pop[aid]   += 1\n        if t == 2:     order_pop[aid] += 1\n\n    aids  = aids[-SEED_N:]\n    tss   = tss[-SEED_N:]\n    types = types[-SEED_N:]\n\n    for i in range(len(aids)):\n        if aids[i] not in test_aids:   # only track test aids as sources\n            continue\n        for j in range(len(aids)):\n            if i == j: continue\n            if abs(tss[i] - tss[j]) > TIME_WIN: continue\n            d_c = covisit_click[aids[i]]; d_c[aids[j]] = d_c.get(aids[j], 0) + 1\n            if types[j] in (1, 2):\n                d_b = covisit_buy[aids[i]]; d_b[aids[j]] = d_b.get(aids[j], 0) + 1\n\ndef trim_covisit():\n    for aid in list(covisit_click.keys()):\n        top = sorted(covisit_click[aid].items(), key=lambda x: -x[1])[:MAX_KEEP]\n        covisit_click[aid] = dict(top)\n    for aid in list(covisit_buy.keys()):\n        top = sorted(covisit_buy[aid].items(), key=lambda x: -x[1])[:MAX_KEEP]\n        covisit_buy[aid] = dict(top)\n    gc.collect()\n\nfor batch_idx, batch in enumerate(tqdm(pf.iter_batches(batch_size=batch_size), desc=\"building covisit\")):\n    df = batch.to_pandas()\n    for sess, grp in df.groupby(\"session\", sort=False):\n        if sess != prev_session and prev_session is not None:\n            process_session(buf_aids, buf_tss, buf_types)\n            buf_aids, buf_tss, buf_types = [], [], []\n        buf_aids.extend(grp[\"aid\"].tolist())\n        buf_tss.extend(grp[\"ts\"].tolist())\n        buf_types.extend(grp[\"type\"].tolist())\n        prev_session = sess\n    del df\n    # trim every TRIM_EVERY batches\n    if (batch_idx + 1) % TRIM_EVERY == 0:\n        trim_covisit()\n    gc.collect()\n\nif buf_aids:\n    process_session(buf_aids, buf_tss, buf_types)\n\ntrim_covisit()\nprint(f\"sessions: {n_sessions_total:,}\")\nprint(f\"covisit_click sources: {len(covisit_click):,}\")\nprint(f\"covisit_buy sources:   {len(covisit_buy):,}\")"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "top20_clicks = [a for a,_ in sorted(click_pop.items(), key=lambda x:-x[1])[:20]]\ntop20_buys   = [a for a,_ in sorted(buy_pop.items(),   key=lambda x:-x[1])[:20]]\ntop20_orders = [a for a,_ in sorted(order_pop.items(), key=lambda x:-x[1])[:20]]\nprint(\"top5 clicks:\", top20_clicks[:5])\nprint(\"top5 orders:\", top20_orders[:5])\n# free popularity dicts\ndel click_pop, buy_pop, order_pop\ngc.collect()"}, {"cell_type": "markdown", "metadata": {}, "source": "## Predictions"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "def recommend(session_aids, covisit, fallback, topk=20):\n    seeds  = session_aids[-SEED_N:]\n    seen   = set(seeds)\n    scores = defaultdict(float)\n    for rank, aid in enumerate(reversed(seeds)):\n        w = 1.0 / (rank + 1)\n        for nbr, cnt in covisit.get(aid, {}).items():\n            if nbr not in seen:\n                scores[nbr] += w * cnt\n    recs = [a for a,_ in sorted(scores.items(), key=lambda x:-x[1])[:topk]]\n    for pop in fallback:\n        if len(recs) >= topk: break\n        if pop not in seen and pop not in recs:\n            recs.append(pop)\n    return recs"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "print(f\"predicting {len(unique_sess):,} sessions...\")\nrows = []\n\nfor i in tqdm(range(len(unique_sess))):\n    s, e   = int(boundaries[i]), int(boundaries[i+1])\n    sess   = int(unique_sess[i])\n    aids   = aid_arr[s:e].tolist()\n    types  = type_arr[s:e].tolist()\n\n    buy_aids   = [a for a, t in zip(aids, types) if t in (1, 2)]\n    order_aids = [a for a, t in zip(aids, types) if t == 2]\n\n    recs_c    = recommend(aids, covisit_click, top20_clicks)\n    seed_cart = buy_aids  if len(buy_aids)   >= 3 else aids\n    recs_cart = recommend(seed_cart, covisit_buy, top20_buys)\n    seed_ord  = order_aids if len(order_aids) >= 2 else (buy_aids if len(buy_aids) >= 2 else aids)\n    recs_ord  = recommend(seed_ord, covisit_buy, top20_orders)\n\n    rows.append({\"session_type\": f\"{sess}_clicks\", \"labels\": \" \".join(map(str, recs_c))})\n    rows.append({\"session_type\": f\"{sess}_carts\",  \"labels\": \" \".join(map(str, recs_cart))})\n    rows.append({\"session_type\": f\"{sess}_orders\", \"labels\": \" \".join(map(str, recs_ord))})\n\nprint(f\"{len(rows):,} rows\")"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "sub = pd.DataFrame(rows)\nlc = sub[\"labels\"].str.split().str.len()\nprint(lc.describe())\nprint(sub.head(4))"}, {"cell_type": "code", "execution_count": null, "metadata": {}, "outputs": [], "source": "sub.to_csv(\"submission.csv\", index=False)\nprint(\"saved submission.csv\")\nprint(sub.shape)"}]}