{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook builds on all the wonderful work on the co-visitation matrix, most recently by [@radek1](https://www.kaggle.com/radek1) [here](https://www.kaggle.com/code/radek1/co-visitation-matrix-simplified-imprvd-logic).\n\nIt produces the same result in much less time due to only a few optimizations. Apart from a more modular refactor, I've kept the original lines near the new lines so the improvements are easier to spot.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\ntrain = pd.read_parquet('../input/otto-full-optimized-memory-footprint/train.parquet')\ntest = pd.read_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')\n\n!pip install pickle5\nimport pickle5 as pickle\n\nwith open('../input/otto-full-optimized-memory-footprint/id2type.pkl', \"rb\") as fh:\n    id2type = pickle.load(fh)\nwith open('../input/otto-full-optimized-memory-footprint/type2id.pkl', \"rb\") as fh:\n    type2id = pickle.load(fh)\n    \nsample_sub = pd.read_csv('../input/otto-recommender-system/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-11-08T04:36:25.019614Z","iopub.execute_input":"2022-11-08T04:36:25.020027Z","iopub.status.idle":"2022-11-08T04:37:04.167848Z","shell.execute_reply.started":"2022-11-08T04:36:25.019936Z","shell.execute_reply":"2022-11-08T04:37:04.166697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fraction_of_sessions_to_use = 1","metadata":{"execution":{"iopub.status.busy":"2022-11-08T04:37:04.170055Z","iopub.execute_input":"2022-11-08T04:37:04.170433Z","iopub.status.idle":"2022-11-08T04:37:04.175413Z","shell.execute_reply.started":"2022-11-08T04:37:04.170390Z","shell.execute_reply":"2022-11-08T04:37:04.174371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Just a bit more of memory footprint optimization. We will be pushing the Kaggle VM to the limit in this notebook!","metadata":{}},{"cell_type":"code","source":"train.session = train.session.astype(np.int32)\ntrain.aid = train.aid.astype(np.int32)\ntrain.ts /= 1000\ntrain.ts = train.ts.astype(np.int32)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T04:37:04.176380Z","iopub.execute_input":"2022-11-08T04:37:04.176667Z","iopub.status.idle":"2022-11-08T04:37:07.032641Z","shell.execute_reply.started":"2022-11-08T04:37:04.176646Z","shell.execute_reply":"2022-11-08T04:37:07.031562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-08T04:37:07.034378Z","iopub.execute_input":"2022-11-08T04:37:07.034791Z","iopub.status.idle":"2022-11-08T04:37:07.051673Z","shell.execute_reply.started":"2022-11-08T04:37:07.034764Z","shell.execute_reply":"2022-11-08T04:37:07.050943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nlucky_sessions_train = train.drop_duplicates(['session']).sample(frac=fraction_of_sessions_to_use)['session']\nsubset_of_train = train[train.session.isin(lucky_sessions_train)]\n\nlucky_sessions_test = test.drop_duplicates(['session']).sample(frac=fraction_of_sessions_to_use)['session']\nsubset_of_test = test[test.session.isin(lucky_sessions_test)]\n\ndel train","metadata":{"execution":{"iopub.status.busy":"2022-11-08T04:37:07.052769Z","iopub.execute_input":"2022-11-08T04:37:07.053191Z","iopub.status.idle":"2022-11-08T04:37:27.284187Z","shell.execute_reply.started":"2022-11-08T04:37:07.053167Z","shell.execute_reply":"2022-11-08T04:37:27.282659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subset_of_train.index = pd.MultiIndex.from_frame(subset_of_train[['session']])\nsubset_of_test.index = pd.MultiIndex.from_frame(subset_of_test[['session']])","metadata":{"execution":{"iopub.status.busy":"2022-11-08T04:37:27.286165Z","iopub.execute_input":"2022-11-08T04:37:27.286513Z","iopub.status.idle":"2022-11-08T04:37:32.537248Z","shell.execute_reply.started":"2022-11-08T04:37:27.286484Z","shell.execute_reply":"2022-11-08T04:37:32.536134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import defaultdict, Counter\nfrom tqdm import tqdm\n\nnext_AIDs = defaultdict(Counter)\n\n\ndef update_covisitation_counter(train_or_test, chunk_size=30_000):\n    sessions = train_or_test.session.unique()\n    \n    # Loop through chunks of sessions\n    for i in tqdm(range(0, sessions.shape[0], chunk_size)):\n        # Get current chunk of sessions\n        consecutive_AIDs = train_or_test.loc[sessions[i]:sessions[min(sessions.shape[0]-1, i+chunk_size-1)]].reset_index(drop=True)\n\n        # Get the 30 most recent products per session\n        # consecutive_AIDs = consecutive_AIDs.groupby('session').apply(lambda g: g.tail(30)).reset_index(drop=True)\n        consecutive_AIDs = consecutive_AIDs.groupby('session', as_index=False).nth(list(range(-30,0))).reset_index(drop=True)\n\n        # Merge sessions onto themselves so we can find pairs of products that are actioned on together\n        consecutive_AIDs = consecutive_AIDs.merge(consecutive_AIDs, on='session')\n\n        # Remove entries that are themselves\n        consecutive_AIDs = consecutive_AIDs[consecutive_AIDs.aid_x != consecutive_AIDs.aid_y]\n\n        # Calculate how many days between products\n        consecutive_AIDs['days_elapsed'] = (consecutive_AIDs.ts_y - consecutive_AIDs.ts_x) / (24 * 60 * 60)\n\n        # Only keep products that are within a day of each other\n        consecutive_AIDs = consecutive_AIDs[(consecutive_AIDs.days_elapsed > 0) & (consecutive_AIDs.days_elapsed <= 1)]\n\n        # Iterate through all sessions and count pairs of co-visited products\n        # for row in consecutive_AIDs.drop_duplicates(['session', 'aid_x', 'aid_y']).itertuples():\n        #     next_AIDs[row.aid_x][row.aid_y] += 1\n        consecutive_AIDs.drop_duplicates(['session', 'aid_x', 'aid_y'], inplace=True)\n        for aid_x, aid_y in zip(consecutive_AIDs['aid_x'], consecutive_AIDs['aid_y']):\n            next_AIDs[aid_x][aid_y] += 1\n        \n\nupdate_covisitation_counter(subset_of_train)\nupdate_covisitation_counter(subset_of_test)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T04:37:46.144940Z","iopub.execute_input":"2022-11-08T04:37:46.145477Z","iopub.status.idle":"2022-11-08T04:54:10.198506Z","shell.execute_reply.started":"2022-11-08T04:37:46.145437Z","shell.execute_reply":"2022-11-08T04:54:10.196832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(next_AIDs)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T04:54:10.201408Z","iopub.execute_input":"2022-11-08T04:54:10.201810Z","iopub.status.idle":"2022-11-08T04:54:10.208236Z","shell.execute_reply.started":"2022-11-08T04:54:10.201774Z","shell.execute_reply":"2022-11-08T04:54:10.207234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now let's generate the predictions.","metadata":{"execution":{"iopub.status.busy":"2022-11-04T00:06:55.37083Z","iopub.execute_input":"2022-11-04T00:06:55.371203Z","iopub.status.idle":"2022-11-04T00:06:55.389977Z","shell.execute_reply.started":"2022-11-04T00:06:55.371178Z","shell.execute_reply":"2022-11-04T00:06:55.388167Z"}}},{"cell_type":"code","source":"%%time\n\ntest_session_AIDs = test.groupby('session')['aid'].apply(list)\nsession_types = ['clicks', 'carts', 'orders']\n\ntest_session_AIDs.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-08T04:54:10.209708Z","iopub.execute_input":"2022-11-08T04:54:10.210032Z","iopub.status.idle":"2022-11-08T04:55:23.377950Z","shell.execute_reply.started":"2022-11-08T04:54:10.210000Z","shell.execute_reply":"2022-11-08T04:55:23.377032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = []\n\n# Loop through all test session product IDs (AIDs)\nfor AIDs in tqdm(test_session_AIDs):\n    # Reverse the set of AIDs\n    AIDs = list(dict.fromkeys(AIDs[::-1]))\n\n    # Take the most recent 20 if there are more than 20\n    if len(AIDs) >= 20:\n        labels.append(AIDs[:20])\n\n    # Otherwise, get the most likely products to see next to fill up to the 20th spot\n    else:\n        # counter = Counter()\n        # for AID in AIDs:\n        #     subsequent_AID_counter = next_AIDs.get(AID)\n        #     if subsequent_AID_counter:\n        #         counter += subsequent_AID_counter\n        counter = Counter()\n        for AID in AIDs:\n            counter.update(next_AIDs.get(AID, {}))\n        \n        AIDs += [AID for AID, cnt in counter.most_common(40) if AID not in AIDs]\n\n        labels.append(AIDs[:20])","metadata":{"execution":{"iopub.status.busy":"2022-11-08T04:55:50.247598Z","iopub.execute_input":"2022-11-08T04:55:50.247972Z","iopub.status.idle":"2022-11-08T06:19:08.792015Z","shell.execute_reply.started":"2022-11-08T04:55:50.247940Z","shell.execute_reply":"2022-11-08T06:19:08.790913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib import pyplot as plt\n\nplt.hist([len(l) for l in labels]);","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:23:52.874396Z","iopub.execute_input":"2022-11-08T06:23:52.874795Z","iopub.status.idle":"2022-11-08T06:23:58.299887Z","shell.execute_reply.started":"2022-11-08T06:23:52.874767Z","shell.execute_reply":"2022-11-08T06:23:58.299182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_as_strings = [' '.join([str(l) for l in lls]) for lls in labels]\n\npredictions = pd.DataFrame(data={'session_type': test_session_AIDs.index, 'labels': labels_as_strings})","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:23:58.301196Z","iopub.execute_input":"2022-11-08T06:23:58.302238Z","iopub.status.idle":"2022-11-08T06:24:13.541858Z","shell.execute_reply.started":"2022-11-08T06:23:58.302207Z","shell.execute_reply":"2022-11-08T06:24:13.540436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_dfs = []\n\nfor st in session_types:\n    modified_predictions = predictions.copy()\n    modified_predictions.session_type = modified_predictions.session_type.astype('str') + f'_{st}'\n    prediction_dfs.append(modified_predictions)\n\nsubmission = pd.concat(prediction_dfs).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:24:13.543178Z","iopub.execute_input":"2022-11-08T06:24:13.543481Z","iopub.status.idle":"2022-11-08T06:24:17.021443Z","shell.execute_reply.started":"2022-11-08T06:24:13.543454Z","shell.execute_reply":"2022-11-08T06:24:17.020535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:24:17.023144Z","iopub.execute_input":"2022-11-08T06:24:17.023429Z","iopub.status.idle":"2022-11-08T06:24:17.033915Z","shell.execute_reply.started":"2022-11-08T06:24:17.023379Z","shell.execute_reply":"2022-11-08T06:24:17.032816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T06:24:17.034920Z","iopub.execute_input":"2022-11-08T06:24:17.035189Z","iopub.status.idle":"2022-11-08T06:24:31.706287Z","shell.execute_reply.started":"2022-11-08T06:24:17.035167Z","shell.execute_reply":"2022-11-08T06:24:31.705143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}