{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":38760,"databundleVersionId":4493939},{"sourceType":"datasetVersion","sourceId":4499280,"datasetId":2631076,"databundleVersionId":4559652}],"dockerImageVersionId":30301,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nimport heapq\nimport pickle\nimport numba as nb\nimport numpy as np\nimport pandas as pd\nfrom tqdm.auto import tqdm\n\ntail = 30\nparallel = 1024\ntopn = 20\nops_weights = np.array([1.0, 6.0, 3.0])\nOP_WEIGHT = 0; TIME_WEIGHT = 1\nparallel = 1024\ntest_ops_weights = np.array([1.0, 6.0, 3.0])","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2026-03-11T19:55:58.849893Z","iopub.execute_input":"2026-03-11T19:55:58.850171Z","iopub.status.idle":"2026-03-11T19:55:59.620817Z","shell.execute_reply.started":"2026-03-11T19:55:58.850102Z","shell.execute_reply":"2026-03-11T19:55:59.619607Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Data\n\n#### Meta DataFrame\n\n`train.csv/test.csv` columns:\n\n- `session`: ordered session ids\n- `length`: length of each session\n- `start_time`: start time stamp (unit=1sec) of this session\n\n#### Data Array\n\n`train.npz/test.npz`:\n\nAn npz contains three 1-dim arrays: \n\n- `aids`: aid array\n- `ts`: time stamp array (minused by start_time of corresponding session)\n- `ops`: type array, `clicks=0, carts=1, orders=2`\n\nAn array is the concatenation of each session, ordered by session id\n\ne.g.\n\n```python\naids = np.concatenate([\n        [1, 2, 3], # session 0 time-ordered aids\n        [4, 5, 6], # session 1 time-ordered aids\n        ...])\n```\n","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"../input/otto-data/train.csv\")\ndf_test = pd.read_csv(\"../input/otto-data/test.csv\")\ndf = pd.concat([df, df_test]).reset_index(drop = True)\nnpz = np.load(\"../input/otto-data/train.npz\")\nnpz_test = np.load(\"../input/otto-data/test.npz\")\naids = np.concatenate([npz['aids'], npz_test['aids']])\nts = np.concatenate([npz['ts'], npz_test['ts']])\nops = np.concatenate([npz['ops'], npz_test['ops']])\n\ndf[\"idx\"] = np.cumsum(df.length) - df.length\ndf[\"end_time\"] = df.start_time + ts[df.idx + df.length - 1]","metadata":{"execution":{"iopub.status.busy":"2026-03-11T19:55:59.623084Z","iopub.execute_input":"2026-03-11T19:55:59.623355Z","iopub.status.idle":"2026-03-11T19:56:20.373621Z","shell.execute_reply.started":"2026-03-11T19:55:59.623331Z","shell.execute_reply":"2026-03-11T19:56:20.372298Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Numba functions\n\nTo fully utilize the ability of numba's jit, and to avoid memory errors, we process `parallel=1024` sessions in one jit function call.\n\n- Firstly, we count the `(aid1, aid2)` pairs in each session in parallel. \n\n- Then, we merge these pairs into a nested counter `{aid1: {aid2: n}, ...}`\n\n- Finally, we find the top-k `aid2` from `aid1`'s counter","metadata":{}},{"cell_type":"code","source":"# get pair dict {(aid1, aid2): weight} for each session\n# The maximum time span between two points is 1 day = 24 * 60 * 60 sec\n@nb.jit(nopython = True, cache = True)\ndef get_single_pairs(pairs, aids, ts, ops, idx, length, start_time, ops_weights, mode):\n    max_idx = idx + length\n    min_idx = max(max_idx - tail, idx)\n    for i in range(min_idx, max_idx):\n        for j in range(i + 1, max_idx):\n            if ts[j] - ts[i] >= 24 * 60 * 60: break\n            if aids[i] == aids[j]: continue\n            if mode == OP_WEIGHT:\n                w1 = ops_weights[ops[j]]\n                w2 = ops_weights[ops[i]]\n            elif mode == TIME_WEIGHT:\n                w1 = 1 + 3 * (ts[i] + start_time - 1659304800) / (1662328791 - 1659304800)\n                w2 = 1 + 3 * (ts[j] + start_time - 1659304800) / (1662328791 - 1659304800)\n            pairs[(aids[i], aids[j])] = w1\n            pairs[(aids[j], aids[i])] = w2\n\n# get pair dict of each session in parallel\n# merge pairs into a nested dict format (cnt)\n@nb.jit(nopython = True, parallel = True, cache = True)\ndef get_pairs(aids, ts, ops, row, cnts, ops_weights, mode):\n    par_n = len(row)\n    pairs = [{(0, 0): 0.0 for _ in range(0)} for _ in range(par_n)]\n    for par_i in nb.prange(par_n):\n        _, idx, length, start_time = row[par_i]\n        get_single_pairs(pairs[par_i], aids, ts, ops, idx, length, start_time, ops_weights, mode)\n    for par_i in range(par_n):\n        for (aid1, aid2), w in pairs[par_i].items():\n            if aid1 not in cnts: cnts[aid1] = {0: 0.0 for _ in range(0)}\n            cnt = cnts[aid1]\n            if aid2 not in cnt: cnt[aid2] = 0.0\n            cnt[aid2] += w\n    \n# util function to get most common keys from a counter dict using min-heap\n# overwrite == 1 means the later item with equal weight is more important\n# otherwise, means the former item with equal weight is more important\n# the result is ordered from higher weight to lower weight\n@nb.jit(nopython = True, cache = True)\ndef heap_topk(cnt, overwrite, cap):\n    q = [(0.0, 0, 0) for _ in range(0)]\n    for i, (k, n) in enumerate(cnt.items()):\n        if overwrite == 1:\n            heapq.heappush(q, (n, i, k))\n        else:\n            heapq.heappush(q, (n, -i, k))\n        if len(q) > cap:\n            heapq.heappop(q)\n    return [heapq.heappop(q)[2] for _ in range(len(q))][::-1]\n   \n# save top-k aid2 for each aid1's cnt\n@nb.jit(nopython = True, cache = True)\ndef get_topk(cnts, topk, k):\n    for aid1, cnt in cnts.items():\n        topk[aid1] = np.array(heap_topk(cnt, 1, k))","metadata":{"execution":{"iopub.status.busy":"2026-03-11T19:56:20.374700Z","iopub.execute_input":"2026-03-11T19:56:20.374928Z","iopub.status.idle":"2026-03-11T19:56:20.600321Z","shell.execute_reply.started":"2026-03-11T19:56:20.374906Z","shell.execute_reply":"2026-03-11T19:56:20.599482Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Train\n\nHere we use topk counting from [Deotte's notebooks](https://www.kaggle.com/code/cdeotte/candidate-rerank-model-lb-0-573). I dropped some of his ideas to make the whole pipeline simple but still strong.\n\n- **mode 0**: Counters are weighted by operation types, with `op_weights = [1.0, 6.0, 3.0]`. This will be used for carts and orders prediction.\n\n- **mode 1**: Counters are weighted by operation time.  This will be used for clicks prediction.\n\nFor each mode, the eastimated running time is **7min~8min** in kaggle notebook","metadata":{}},{"cell_type":"code","source":"topks = {}\n\n# for two modes\nfor mode in [OP_WEIGHT, TIME_WEIGHT]:\n    # get nested counter\n    cnts = nb.typed.Dict.empty(\n        key_type = nb.types.int64,\n        value_type = nb.typeof(nb.typed.Dict.empty(key_type = nb.types.int64, value_type = nb.types.float64)))\n    max_idx = len(df)\n    for idx in tqdm(range(0, max_idx, parallel)):\n        row = df.iloc[idx:min(idx + parallel, max_idx)][['session', 'idx', 'length', 'start_time']].values\n        get_pairs(aids, ts, ops, row, cnts, ops_weights, mode)\n\n    # get topk from counter\n    topk = nb.typed.Dict.empty(\n            key_type = nb.types.int64,\n            value_type = nb.types.int64[:])\n    get_topk(cnts, topk, topn)\n\n    del cnts; gc.collect()\n    topks[mode] = topk","metadata":{"execution":{"iopub.status.busy":"2026-03-11T19:56:20.601636Z","iopub.execute_input":"2026-03-11T19:56:20.601933Z","iopub.status.idle":"2026-03-11T20:10:27.144047Z","shell.execute_reply.started":"2026-03-11T19:56:20.601908Z","shell.execute_reply":"2026-03-11T20:10:27.143032Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Inference Numba Functions\n\nThe logic is a simpler version of [Deotte's notebook](https://www.kaggle.com/code/cdeotte/candidate-rerank-model-lb-0-573).\n\nSame as training functions, we process `parallel=1024` sessions in one jit function call. For each session, if the unique aids is more than 20, we rerank them by a time-decay weight. Otherwise, we use current aids to recall new aids from their topk co-visitation candidates, weighted by the query's operation type (`test_ops_weights=[1.0, 6.0, 3.0]`). \n\nWe used **mode0** topk to generate predictions for carts and orders, and **mode1** topk to generate predictions for clicks. ","metadata":{}},{"cell_type":"code","source":"@nb.jit(nopython = True, cache = True)\ndef inference_(aids, ops, row, result, topk, test_ops_weights, seq_weight):\n    for session, idx, length in row:\n        unique_aids = nb.typed.Dict.empty(key_type = nb.types.int64, value_type = nb.types.float64)\n        cnt = nb.typed.Dict.empty(key_type = nb.types.int64, value_type = nb.types.float64)\n        \n        candidates = aids[idx:idx + length][::-1]\n        candidates_ops = ops[idx:idx + length][::-1]\n        for a in candidates:\n            unique_aids[a] = 0\n                \n        if len(unique_aids) >= 20:\n            sequence_weight = np.power(2, np.linspace(seq_weight, 1, len(candidates)))[::-1] - 1\n            for a, op, w in zip(candidates, candidates_ops, sequence_weight):\n                if a not in cnt: cnt[a] = 0\n                cnt[a] += w * test_ops_weights[op]\n            result_candidates = heap_topk(cnt, 0, 20)\n        else:\n            result_candidates = list(unique_aids)\n            for a in result_candidates:\n                if a not in topk: continue\n                for b in topk[a]:\n                    if b in unique_aids: continue\n                    if b not in cnt: cnt[b] = 0\n                    cnt[b] += 1\n            result_candidates.extend(heap_topk(cnt, 0, 20 - len(result_candidates)))\n        result[session] = np.array(result_candidates)\n        \n@nb.jit(nopython = True)\ndef inference(aids, ops, row, \n              result_clicks, result_buy,\n              topk_clicks, topk_buy,\n              test_ops_weights):\n    inference_(aids, ops, row, result_clicks, topk_clicks, test_ops_weights, 0.1)\n    inference_(aids, ops, row, result_buy, topk_buy, test_ops_weights, 0.5)","metadata":{"execution":{"iopub.status.busy":"2026-03-11T20:10:27.183305Z","iopub.execute_input":"2026-03-11T20:10:27.183706Z","iopub.status.idle":"2026-03-11T20:10:30.196693Z","shell.execute_reply.started":"2026-03-11T20:10:27.183677Z","shell.execute_reply":"2026-03-11T20:10:30.195784Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# result place holder\nresult_clicks = nb.typed.Dict.empty(\n    key_type = nb.types.int64,\n    value_type = nb.types.int64[:])\nresult_buy = nb.typed.Dict.empty(\n    key_type = nb.types.int64,\n    value_type = nb.types.int64[:])\nfor idx in tqdm(range(len(df) - len(df_test), len(df), parallel)):\n    row = df.iloc[idx:min(idx + parallel, len(df))][['session', 'idx', 'length']].values\n    inference(aids, ops, row, result_clicks, result_buy, topks[TIME_WEIGHT], topks[OP_WEIGHT], test_ops_weights)","metadata":{"execution":{"iopub.status.busy":"2026-03-11T20:10:30.197838Z","iopub.execute_input":"2026-03-11T20:10:30.198114Z","iopub.status.idle":"2026-03-11T20:11:16.737890Z","shell.execute_reply.started":"2026-03-11T20:10:30.198091Z","shell.execute_reply":"2026-03-11T20:11:16.737097Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"subs = []\nop_names = [\"clicks\", \"carts\", \"orders\"]\nfor result, op in zip([result_clicks, result_buy, result_buy], op_names):\n\n    sub = pd.DataFrame({\"session_type\": result.keys(), \"labels\": result.values()})\n    sub.session_type = sub.session_type.astype(str) + f\"_{op}\"\n    sub.labels = sub.labels.apply(lambda x: \" \".join(x.astype(str)))\n    subs.append(sub)\n    \nsub = pd.concat(subs).reset_index(drop = True)\nsub.to_csv('submission.csv', index = False)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2026-03-11T20:11:23.948765Z","iopub.execute_input":"2026-03-11T20:11:23.949122Z","iopub.status.idle":"2026-03-11T20:13:04.446057Z","shell.execute_reply.started":"2026-03-11T20:11:23.949093Z","shell.execute_reply":"2026-03-11T20:13:04.445156Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!ls","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T20:14:56.837171Z","iopub.execute_input":"2026-03-11T20:14:56.837536Z","iopub.status.idle":"2026-03-11T20:14:58.349839Z","shell.execute_reply.started":"2026-03-11T20:14:56.837512Z","shell.execute_reply":"2026-03-11T20:14:58.348763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}