{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"raw","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import os\nimport gc\nimport heapq\nimport pickle\nimport numba as nb\nimport numpy as np\nimport pandas as pd\nfrom tqdm.auto import tqdm\n\ntail = 30\nparallel = 1024\ntopn = 20\nops_weights = np.array([1.0, 6.0, 3.0])\nOP_WEIGHT = 0; TIME_WEIGHT = 1\nparallel = 1024\ntest_ops_weights = np.array([1.0, 6.0, 3.0])","metadata":{"execution":{"iopub.status.busy":"2022-12-18T10:24:15.546064Z","iopub.execute_input":"2022-12-18T10:24:15.546887Z","iopub.status.idle":"2022-12-18T10:24:16.358716Z","shell.execute_reply.started":"2022-12-18T10:24:15.546782Z","shell.execute_reply":"2022-12-18T10:24:16.357770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"../input/otto-data/train.csv\")\ndf_test = pd.read_csv(\"../input/otto-data/test.csv\")\ndf = pd.concat([df, df_test]).reset_index(drop = True)\nnpz = np.load(\"../input/otto-data/train.npz\")\nnpz_test = np.load(\"../input/otto-data/test.npz\")\naids = np.concatenate([npz['aids'], npz_test['aids']])\nts = np.concatenate([npz['ts'], npz_test['ts']])\nops = np.concatenate([npz['ops'], npz_test['ops']])\n\ndf[\"idx\"] = np.cumsum(df.length) - df.length\ndf[\"end_time\"] = df.start_time + ts[df.idx + df.length - 1]","metadata":{"execution":{"iopub.status.busy":"2022-12-18T10:24:16.360493Z","iopub.execute_input":"2022-12-18T10:24:16.360798Z","iopub.status.idle":"2022-12-18T10:24:55.855923Z","shell.execute_reply.started":"2022-12-18T10:24:16.360770Z","shell.execute_reply":"2022-12-18T10:24:55.854792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get pair dict {(aid1, aid2): weight} for each session\n# The maximum time span between two points is 1 day = 24 * 60 * 60 sec\n@nb.jit(nopython = True, cache = True)\ndef get_single_pairs(pairs, aids, ts, ops, idx, length, start_time, ops_weights, mode):\n    max_idx = idx + length\n    min_idx = max(max_idx - tail, idx)\n    for i in range(min_idx, max_idx):\n        for j in range(i + 1, max_idx):\n            if ts[j] - ts[i] >= 24 * 60 * 60: break\n            if aids[i] == aids[j]: continue\n            if mode == OP_WEIGHT:\n                w1 = ops_weights[ops[j]]\n                w2 = ops_weights[ops[i]]\n            elif mode == TIME_WEIGHT:\n                w1 = 1 + 3 * (ts[i] + start_time - 1659304800) / (1662328791 - 1659304800)\n                w2 = 1 + 3 * (ts[j] + start_time - 1659304800) / (1662328791 - 1659304800)\n            pairs[(aids[i], aids[j])] = w1\n            pairs[(aids[j], aids[i])] = w2\n\n# get pair dict of each session in parallel\n# merge pairs into a nested dict format (cnt)\n@nb.jit(nopython = True, parallel = True, cache = True)\ndef get_pairs(aids, ts, ops, row, cnts, ops_weights, mode):\n    par_n = len(row)\n    pairs = [{(0, 0): 0.0 for _ in range(0)} for _ in range(par_n)]\n    for par_i in nb.prange(par_n):\n        _, idx, length, start_time = row[par_i]\n        get_single_pairs(pairs[par_i], aids, ts, ops, idx, length, start_time, ops_weights, mode)\n    for par_i in range(par_n):\n        for (aid1, aid2), w in pairs[par_i].items():\n            if aid1 not in cnts: cnts[aid1] = {0: 0.0 for _ in range(0)}\n            cnt = cnts[aid1]\n            if aid2 not in cnt: cnt[aid2] = 0.0\n            cnt[aid2] += w\n    \n# util function to get most common keys from a counter dict using min-heap\n# overwrite == 1 means the later item with equal weight is more important\n# otherwise, means the former item with equal weight is more important\n# the result is ordered from higher weight to lower weight\n@nb.jit(nopython = True, cache = True)\ndef heap_topk(cnt, overwrite, cap):\n    q = [(0.0, 0, 0) for _ in range(0)]\n    for i, (k, n) in enumerate(cnt.items()):\n        if overwrite == 1:\n            heapq.heappush(q, (n, i, k))\n        else:\n            heapq.heappush(q, (n, -i, k))\n        if len(q) > cap:\n            heapq.heappop(q)\n    return [heapq.heappop(q)[2] for _ in range(len(q))][::-1]\n   \n# save top-k aid2 for each aid1's cnt\n@nb.jit(nopython = True, cache = True)\ndef get_topk(cnts, topk, k):\n    for aid1, cnt in cnts.items():\n        topk[aid1] = np.array(heap_topk(cnt, 1, k))","metadata":{"execution":{"iopub.status.busy":"2022-12-18T10:24:55.857507Z","iopub.execute_input":"2022-12-18T10:24:55.857997Z","iopub.status.idle":"2022-12-18T10:24:56.133186Z","shell.execute_reply.started":"2022-12-18T10:24:55.857955Z","shell.execute_reply":"2022-12-18T10:24:56.131936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"topks = {}\n\n# for two modes\nfor mode in [OP_WEIGHT, TIME_WEIGHT]:\n    # get nested counter\n    cnts = nb.typed.Dict.empty(\n        key_type = nb.types.int64,\n        value_type = nb.typeof(nb.typed.Dict.empty(key_type = nb.types.int64, value_type = nb.types.float64)))\n    max_idx = len(df)\n    for idx in tqdm(range(0, max_idx, parallel)):\n        row = df.iloc[idx:min(idx + parallel, max_idx)][['session', 'idx', 'length', 'start_time']].values\n        get_pairs(aids, ts, ops, row, cnts, ops_weights, mode)\n\n    # get topk from counter\n    topk = nb.typed.Dict.empty(\n            key_type = nb.types.int64,\n            value_type = nb.types.int64[:])\n    get_topk(cnts, topk, topn)\n\n    del cnts; gc.collect()\n    topks[mode] = topk","metadata":{"execution":{"iopub.status.busy":"2022-12-18T10:24:56.136352Z","iopub.execute_input":"2022-12-18T10:24:56.136744Z","iopub.status.idle":"2022-12-18T10:41:21.162911Z","shell.execute_reply.started":"2022-12-18T10:24:56.136707Z","shell.execute_reply":"2022-12-18T10:41:21.161608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@nb.jit(nopython = True, cache = True)\ndef inference_(aids, ops, row, result, topk, test_ops_weights, seq_weight):\n    for session, idx, length in row:\n        unique_aids = nb.typed.Dict.empty(key_type = nb.types.int64, value_type = nb.types.float64)\n        cnt = nb.typed.Dict.empty(key_type = nb.types.int64, value_type = nb.types.float64)\n        \n        candidates = aids[idx:idx + length][::-1]\n        candidates_ops = ops[idx:idx + length][::-1]\n        for a in candidates:\n            unique_aids[a] = 0\n                \n        if len(unique_aids) >= 20:\n            sequence_weight = np.power(2, np.linspace(seq_weight, 1, len(candidates)))[::-1] - 1\n            for a, op, w in zip(candidates, candidates_ops, sequence_weight):\n                if a not in cnt: cnt[a] = 0\n                cnt[a] += w * test_ops_weights[op]\n            result_candidates = heap_topk(cnt, 0, 20)\n        else:\n            result_candidates = list(unique_aids)\n            for a in result_candidates:\n                if a not in topk: continue\n                for b in topk[a]:\n                    if b in unique_aids: continue\n                    if b not in cnt: cnt[b] = 0\n                    cnt[b] += 1\n            result_candidates.extend(heap_topk(cnt, 0, 20 - len(result_candidates)))\n        result[session] = np.array(result_candidates)\n        \n@nb.jit(nopython = True)\ndef inference(aids, ops, row, \n              result_clicks, result_buy,\n              topk_clicks, topk_buy,\n              test_ops_weights):\n    inference_(aids, ops, row, result_clicks, topk_clicks, test_ops_weights, 0.1)\n    inference_(aids, ops, row, result_buy, topk_buy, test_ops_weights, 0.5)","metadata":{"execution":{"iopub.status.busy":"2022-12-18T10:41:21.195303Z","iopub.execute_input":"2022-12-18T10:41:21.195956Z","iopub.status.idle":"2022-12-18T10:41:24.097808Z","shell.execute_reply.started":"2022-12-18T10:41:21.195916Z","shell.execute_reply":"2022-12-18T10:41:24.096874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# result place holder\nresult_clicks = nb.typed.Dict.empty(\n    key_type = nb.types.int64,\n    value_type = nb.types.int64[:])\nresult_buy = nb.typed.Dict.empty(\n    key_type = nb.types.int64,\n    value_type = nb.types.int64[:])\nfor idx in tqdm(range(len(df) - len(df_test), len(df), parallel)):\n    row = df.iloc[idx:min(idx + parallel, len(df))][['session', 'idx', 'length']].values\n    inference(aids, ops, row, result_clicks, result_buy, topks[TIME_WEIGHT], topks[OP_WEIGHT], test_ops_weights)","metadata":{"execution":{"iopub.status.busy":"2022-12-18T10:41:24.099264Z","iopub.execute_input":"2022-12-18T10:41:24.099932Z","iopub.status.idle":"2022-12-18T10:42:32.909426Z","shell.execute_reply.started":"2022-12-18T10:41:24.099901Z","shell.execute_reply":"2022-12-18T10:42:32.908326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subs = []\nop_names = [\"clicks\", \"carts\", \"orders\"]\nfor result, op in zip([result_clicks, result_buy, result_buy], op_names):\n\n    sub = pd.DataFrame({\"session_type\": result.keys(), \"labels\": result.values()})\n    sub.session_type = sub.session_type.astype(str) + f\"_{op}\"\n    sub.labels = sub.labels.apply(lambda x: \" \".join(x.astype(str)))\n    subs.append(sub)\n    \nsub = pd.concat(subs).reset_index(drop = True)\nsub.to_csv('submission.csv', index = False)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-18T10:42:32.910726Z","iopub.execute_input":"2022-12-18T10:42:32.911050Z","iopub.status.idle":"2022-12-18T10:44:54.308251Z","shell.execute_reply.started":"2022-12-18T10:42:32.911022Z","shell.execute_reply":"2022-12-18T10:44:54.307228Z"},"trusted":true},"execution_count":null,"outputs":[]}]}