{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# OTTO: Co-visitation Matrix\n\nThere exist products that are frequently viewed and bought together. Here we leverage this idea by computing a co-visitation matrix of products. It's done in the following way:\n\n1. First we look at all pairs of events within the same session that are close to each other in time (< 1 day). We compute co-visitation matrix $M_{aid1,aid2}$ by counting global number of event pairs for each pair across all sessions.\n2. For each $aid1$ we find top 20 most frequent aid2:  `aid2=argsort(M[aid])[-20:]`\n3. We produce test results by concatenating `tail(20)` of test session events (see https://www.kaggle.com/code/simamumu/old-test-data-last-20-aid-get-lb0-947) with the most likely recommendations from co-visitation matrix. These recommendations are generated from session AIDs and `aid2` from the step 2\n\n\n**Please, smash that thumbs up button and subscribe if you like this notebook!**","metadata":{}},{"cell_type":"markdown","source":"## Utils, imports","metadata":{}},{"cell_type":"code","source":"### import numpy as np\nfrom collections import defaultdict\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nimport glob\nimport numpy as np\nimport multiprocessing\n\n\nimport glob\nfrom collections import Counter\n\nDEBUG=False   \nSAMPLING = 9  # Reduce it to improve performance","metadata":{"execution":{"iopub.status.busy":"2022-11-02T19:34:06.455802Z","iopub.execute_input":"2022-11-02T19:34:06.456226Z","iopub.status.idle":"2022-11-02T19:34:06.462217Z","shell.execute_reply.started":"2022-11-02T19:34:06.456179Z","shell.execute_reply":"2022-11-02T19:34:06.46091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generate AID pairs","metadata":{}},{"cell_type":"code","source":"def gen_pairs(df):\n    df = df.query('session % @SAMPLING == 0').groupby('session', as_index=False).apply(lambda g: g.tail(100)).reset_index(drop=True)\n    df = pd.merge(df, df, on='session')\n    df['time_delta'] = (df.ts_x - df.ts_y).abs()\n    pairs = df.query('time_delta < 24 * 60 * 60 * 1000 and aid_x != aid_y')[['session', 'aid_x', 'aid_y']].drop_duplicates()\n    return pairs[['aid_x', 'aid_y']].values\n    \n\ndef gen_aid_pairs():\n    all_pairs = defaultdict(lambda: Counter())\n    \n    for idx, chunk_file in enumerate(tqdm(glob.glob('../input/otto-chunk-data-inparquet-format/train_parquet/*'), desc='Chunks')):\n        chunk = pd.read_parquet(chunk_file)\n        \n        with multiprocessing.Pool(4) as p:\n            pair_chunks = p.map(gen_pairs, np.array_split(chunk.head(100000000 if not DEBUG else 10000), 12))            \n            for pairs in pair_chunks:\n                for aid1, aid2 in pairs:\n                    all_pairs[aid1][aid2] +=1 \n        if DEBUG and idx >= 2:\n            break\n    return all_pairs\n        \nall_pairs = gen_aid_pairs()","metadata":{"execution":{"iopub.status.busy":"2022-11-02T19:34:06.688049Z","iopub.execute_input":"2022-11-02T19:34:06.688461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_top_20 = []\nfor aid, cnt in tqdm(all_pairs.items()):\n    df_top_20.append({'aid1': aid, 'aid2': [aid2 for aid2, freq in cnt.most_common(20)]})\n    \ndf_top_20 = pd.DataFrame(df_top_20).set_index('aid1')\ndf_top_20","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top_20 = df_top_20.aid2.to_dict()","metadata":{"execution":{"iopub.status.busy":"2022-11-02T19:12:39.486132Z","iopub.execute_input":"2022-11-02T19:12:39.486477Z","iopub.status.idle":"2022-11-02T19:12:39.96533Z","shell.execute_reply.started":"2022-11-02T19:12:39.486444Z","shell.execute_reply":"2022-11-02T19:12:39.964013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test set inference","metadata":{}},{"cell_type":"code","source":"def load_test():    \n    dfs = []\n    for e, chunk_file in enumerate(tqdm(glob.glob('../input/otto-chunk-data-inparquet-format/test_parquet/*'))):\n        chunk = pd.read_parquet(chunk_file)\n        dfs.append(chunk)\n\n    return pd.concat(dfs).reset_index(drop=True).astype({\"ts\": \"datetime64[ms]\"})","metadata":{"execution":{"iopub.status.busy":"2022-11-02T19:12:39.968394Z","iopub.execute_input":"2022-11-02T19:12:39.969462Z","iopub.status.idle":"2022-11-02T19:12:39.976688Z","shell.execute_reply.started":"2022-11-02T19:12:39.969397Z","shell.execute_reply":"2022-11-02T19:12:39.975536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = load_test()","metadata":{"execution":{"iopub.status.busy":"2022-11-02T19:12:39.978145Z","iopub.execute_input":"2022-11-02T19:12:39.978553Z","iopub.status.idle":"2022-11-02T19:12:41.953072Z","shell.execute_reply.started":"2022-11-02T19:12:39.978509Z","shell.execute_reply":"2022-11-02T19:12:41.952089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import itertools\n\ndef suggest_aids(df):\n    aids = df.tail(20).aid.tolist()\n    \n    if len(aids) >= 20:\n        # We have enough events in the test session\n        return aids\n    \n    # Append it with AIDs from the co-visitation matrix. \n    aids = set(aids)\n    aids2 = list(itertools.chain(*[top_20[aid] for aid in aids if aid in top_20]))\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids2).most_common(20) if aid2 not in aids]        \n    return list(aids) + top_aids2[:20 - len(aids)]\n\n        ","metadata":{"execution":{"iopub.status.busy":"2022-11-02T19:25:46.722213Z","iopub.execute_input":"2022-11-02T19:25:46.723397Z","iopub.status.idle":"2022-11-02T19:25:46.731308Z","shell.execute_reply.started":"2022-11-02T19:25:46.723353Z","shell.execute_reply":"2022-11-02T19:25:46.730106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = test_df.sort_values([\"session\", \"type\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_aids(x)\n)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-02T19:26:21.72945Z","iopub.execute_input":"2022-11-02T19:26:21.730245Z","iopub.status.idle":"2022-11-02T19:32:45.141259Z","shell.execute_reply.started":"2022-11-02T19:26:21.730204Z","shell.execute_reply":"2022-11-02T19:32:45.140117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clicks_pred_df = pd.DataFrame(pred_df.add_suffix(\"_clicks\"), columns=[\"labels\"]).reset_index()\norders_pred_df = pd.DataFrame(pred_df.add_suffix(\"_orders\"), columns=[\"labels\"]).reset_index()\ncarts_pred_df = pd.DataFrame(pred_df.add_suffix(\"_carts\"), columns=[\"labels\"]).reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-11-02T19:32:45.143455Z","iopub.execute_input":"2022-11-02T19:32:45.143923Z","iopub.status.idle":"2022-11-02T19:32:48.719967Z","shell.execute_reply.started":"2022-11-02T19:32:45.143881Z","shell.execute_reply":"2022-11-02T19:32:48.718744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df","metadata":{"execution":{"iopub.status.busy":"2022-11-02T19:32:48.721625Z","iopub.execute_input":"2022-11-02T19:32:48.722094Z","iopub.status.idle":"2022-11-02T19:32:48.733485Z","shell.execute_reply.started":"2022-11-02T19:32:48.722046Z","shell.execute_reply":"2022-11-02T19:32:48.732642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.concat(\n    [clicks_pred_df, orders_pred_df, carts_pred_df]\n)\npred_df.columns = [\"session_type\", \"labels\"]\npred_df[\"labels\"] = pred_df.labels.apply(lambda x: \" \".join(map(str,x)))\npred_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T19:32:48.735782Z","iopub.execute_input":"2022-11-02T19:32:48.736158Z","iopub.status.idle":"2022-11-02T19:33:59.339514Z","shell.execute_reply.started":"2022-11-02T19:32:48.736117Z","shell.execute_reply":"2022-11-02T19:33:59.337922Z"},"trusted":true},"execution_count":null,"outputs":[]}]}