{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook builds on all the wonderful work on the co-visitation matrix, most recently by [@radek1](https://www.kaggle.com/radek1) [here](https://www.kaggle.com/code/radek1/co-visitation-matrix-simplified-imprvd-logic).\n\nIt produces the same result in much less time due to only a few optimizations. Apart from a more modular refactor, I've kept the original lines near the new lines so the improvements are easier to spot.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\ntrain = pd.read_parquet('../input/otto-full-optimized-memory-footprint/train.parquet')\ntest = pd.read_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')\n\n!pip install pickle5\nimport pickle5 as pickle\n\nwith open('../input/otto-full-optimized-memory-footprint/id2type.pkl', \"rb\") as fh:\n    id2type = pickle.load(fh)\nwith open('../input/otto-full-optimized-memory-footprint/type2id.pkl', \"rb\") as fh:\n    type2id = pickle.load(fh)\n    \nsample_sub = pd.read_csv('../input/otto-recommender-system/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-11-16T15:00:53.165816Z","iopub.execute_input":"2022-11-16T15:00:53.166532Z","iopub.status.idle":"2022-11-16T15:01:35.887509Z","shell.execute_reply.started":"2022-11-16T15:00:53.166438Z","shell.execute_reply":"2022-11-16T15:01:35.886048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fraction_of_sessions_to_use = 0.5","metadata":{"execution":{"iopub.status.busy":"2022-11-16T15:01:57.150425Z","iopub.execute_input":"2022-11-16T15:01:57.150895Z","iopub.status.idle":"2022-11-16T15:01:57.158391Z","shell.execute_reply.started":"2022-11-16T15:01:57.150854Z","shell.execute_reply":"2022-11-16T15:01:57.156689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Just a bit more of memory footprint optimization. We will be pushing the Kaggle VM to the limit in this notebook!","metadata":{}},{"cell_type":"code","source":"train.session = train.session.astype(np.int32)\ntrain.aid = train.aid.astype(np.int32)\ntrain.ts /= 1000\ntrain.ts = train.ts.astype(np.int32)","metadata":{"execution":{"iopub.status.busy":"2022-11-16T15:02:00.774908Z","iopub.execute_input":"2022-11-16T15:02:00.775441Z","iopub.status.idle":"2022-11-16T15:02:06.616137Z","shell.execute_reply.started":"2022-11-16T15:02:00.775399Z","shell.execute_reply":"2022-11-16T15:02:06.614922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-16T15:02:11.754949Z","iopub.execute_input":"2022-11-16T15:02:11.756600Z","iopub.status.idle":"2022-11-16T15:02:11.780563Z","shell.execute_reply.started":"2022-11-16T15:02:11.756554Z","shell.execute_reply":"2022-11-16T15:02:11.779155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nlucky_sessions_train = train.drop_duplicates(['session']).sample(frac=fraction_of_sessions_to_use)['session']\nsubset_of_train = train[train.session.isin(lucky_sessions_train)]\n\nlucky_sessions_test = test.drop_duplicates(['session']).sample(frac=fraction_of_sessions_to_use)['session']\nsubset_of_test = test[test.session.isin(lucky_sessions_test)]\n\ndel train","metadata":{"execution":{"iopub.status.busy":"2022-11-16T15:02:17.983513Z","iopub.execute_input":"2022-11-16T15:02:17.984025Z","iopub.status.idle":"2022-11-16T15:02:40.283262Z","shell.execute_reply.started":"2022-11-16T15:02:17.983961Z","shell.execute_reply":"2022-11-16T15:02:40.281765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subset_of_train.index = pd.MultiIndex.from_frame(subset_of_train[['session']])\nsubset_of_test.index = pd.MultiIndex.from_frame(subset_of_test[['session']])","metadata":{"execution":{"iopub.status.busy":"2022-11-16T15:02:46.983613Z","iopub.execute_input":"2022-11-16T15:02:46.984045Z","iopub.status.idle":"2022-11-16T15:02:47.737160Z","shell.execute_reply.started":"2022-11-16T15:02:46.984009Z","shell.execute_reply":"2022-11-16T15:02:47.735916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import defaultdict, Counter\nfrom tqdm import tqdm\n\nnext_AIDs = defaultdict(Counter)\n\n\ndef update_covisitation_counter(train_or_test, chunk_size=30_000):\n    sessions = train_or_test.session.unique()\n    \n    # Loop through chunks of sessions\n    for i in tqdm(range(0, sessions.shape[0], chunk_size)):\n        # Get current chunk of sessions\n        consecutive_AIDs = train_or_test.loc[sessions[i]:sessions[min(sessions.shape[0]-1, i+chunk_size-1)]].reset_index(drop=True)\n\n        # Get the 30 most recent products per session\n        # consecutive_AIDs = consecutive_AIDs.groupby('session').apply(lambda g: g.tail(30)).reset_index(drop=True)\n        consecutive_AIDs = consecutive_AIDs.groupby('session', as_index=False).nth(list(range(-30,0))).reset_index(drop=True)\n\n        # Merge sessions onto themselves so we can find pairs of products that are actioned on together\n        consecutive_AIDs = consecutive_AIDs.merge(consecutive_AIDs, on='session')\n\n        # Remove entries that are themselves\n        consecutive_AIDs = consecutive_AIDs[consecutive_AIDs.aid_x != consecutive_AIDs.aid_y]\n\n        # Calculate how many days between products\n        consecutive_AIDs['days_elapsed'] = (consecutive_AIDs.ts_y - consecutive_AIDs.ts_x) / (24 * 60 * 60)\n\n        # Only keep products that are within a day of each other\n        consecutive_AIDs = consecutive_AIDs[(consecutive_AIDs.days_elapsed > 0) & (consecutive_AIDs.days_elapsed <= 1)]\n\n        # Iterate through all sessions and count pairs of co-visited products\n        # for row in consecutive_AIDs.drop_duplicates(['session', 'aid_x', 'aid_y']).itertuples():\n        #     next_AIDs[row.aid_x][row.aid_y] += 1\n        consecutive_AIDs.drop_duplicates(['session', 'aid_x', 'aid_y'], inplace=True)\n        for aid_x, aid_y in zip(consecutive_AIDs['aid_x'], consecutive_AIDs['aid_y']):\n            next_AIDs[aid_x][aid_y] += 1\n        \n\nupdate_covisitation_counter(subset_of_train)\nupdate_covisitation_counter(subset_of_test)","metadata":{"execution":{"iopub.status.busy":"2022-11-16T15:02:51.728910Z","iopub.execute_input":"2022-11-16T15:02:51.729412Z","iopub.status.idle":"2022-11-16T15:04:51.617908Z","shell.execute_reply.started":"2022-11-16T15:02:51.729346Z","shell.execute_reply":"2022-11-16T15:04:51.616460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(next_AIDs)","metadata":{"execution":{"iopub.status.busy":"2022-11-16T15:06:11.948652Z","iopub.execute_input":"2022-11-16T15:06:11.949820Z","iopub.status.idle":"2022-11-16T15:06:11.956160Z","shell.execute_reply.started":"2022-11-16T15:06:11.949775Z","shell.execute_reply":"2022-11-16T15:06:11.955182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now let's generate the predictions.","metadata":{"execution":{"iopub.status.busy":"2022-11-04T00:06:55.37083Z","iopub.execute_input":"2022-11-04T00:06:55.371203Z","iopub.status.idle":"2022-11-04T00:06:55.389977Z","shell.execute_reply.started":"2022-11-04T00:06:55.371178Z","shell.execute_reply":"2022-11-04T00:06:55.388167Z"}}},{"cell_type":"code","source":"%%time\n\ntest_session_AIDs = test.groupby('session')['aid'].apply(list)\nsession_types = ['clicks', 'carts', 'orders']\n\ntest_session_AIDs.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-16T15:06:20.916700Z","iopub.execute_input":"2022-11-16T15:06:20.917194Z","iopub.status.idle":"2022-11-16T15:07:04.420132Z","shell.execute_reply.started":"2022-11-16T15:06:20.917155Z","shell.execute_reply":"2022-11-16T15:07:04.418898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = []\n\n# Loop through all test session product IDs (AIDs)\nfor AIDs in tqdm(test_session_AIDs):\n    # Reverse the set of AIDs\n    AIDs = list(dict.fromkeys(AIDs[::-1]))\n\n    # Take the most recent 20 if there are more than 20\n    if len(AIDs) >= 20:\n        labels.append(AIDs[:20])\n\n    # Otherwise, get the most likely products to see next to fill up to the 20th spot\n    else:\n        # counter = Counter()\n        # for AID in AIDs:\n        #     subsequent_AID_counter = next_AIDs.get(AID)\n        #     if subsequent_AID_counter:\n        #         counter += subsequent_AID_counter\n        counter = Counter()\n        for AID in AIDs:\n            counter.update(next_AIDs.get(AID, {}))\n        \n        AIDs += [AID for AID, cnt in counter.most_common(40) if AID not in AIDs]\n\n        labels.append(AIDs[:20])","metadata":{"execution":{"iopub.status.busy":"2022-11-16T15:07:17.713774Z","iopub.execute_input":"2022-11-16T15:07:17.714294Z","iopub.status.idle":"2022-11-16T15:29:47.088112Z","shell.execute_reply.started":"2022-11-16T15:07:17.714249Z","shell.execute_reply":"2022-11-16T15:29:47.086721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib import pyplot as plt\n\nplt.hist([len(l) for l in labels]);","metadata":{"execution":{"iopub.status.busy":"2022-11-16T16:02:02.251574Z","iopub.execute_input":"2022-11-16T16:02:02.253273Z","iopub.status.idle":"2022-11-16T16:02:09.576516Z","shell.execute_reply.started":"2022-11-16T16:02:02.253218Z","shell.execute_reply":"2022-11-16T16:02:09.575508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_as_strings = [' '.join([str(l) for l in lls]) for lls in labels]\n\npredictions = pd.DataFrame(data={'session_type': test_session_AIDs.index, 'labels': labels_as_strings})","metadata":{"execution":{"iopub.status.busy":"2022-11-16T16:02:23.523819Z","iopub.execute_input":"2022-11-16T16:02:23.524259Z","iopub.status.idle":"2022-11-16T16:02:39.096563Z","shell.execute_reply.started":"2022-11-16T16:02:23.524227Z","shell.execute_reply":"2022-11-16T16:02:39.095219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_dfs = []\n\nfor st in session_types:\n    modified_predictions = predictions.copy()\n    modified_predictions.session_type = modified_predictions.session_type.astype('str') + f'_{st}'\n    prediction_dfs.append(modified_predictions)\n\nsubmission = pd.concat(prediction_dfs).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-11-16T16:02:43.571089Z","iopub.execute_input":"2022-11-16T16:02:43.571554Z","iopub.status.idle":"2022-11-16T16:02:49.287995Z","shell.execute_reply.started":"2022-11-16T16:02:43.571515Z","shell.execute_reply":"2022-11-16T16:02:49.286602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-16T16:02:54.025437Z","iopub.execute_input":"2022-11-16T16:02:54.025994Z","iopub.status.idle":"2022-11-16T16:02:54.040974Z","shell.execute_reply.started":"2022-11-16T16:02:54.025944Z","shell.execute_reply":"2022-11-16T16:02:54.039135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-16T16:03:11.464164Z","iopub.execute_input":"2022-11-16T16:03:11.464622Z","iopub.status.idle":"2022-11-16T16:03:32.267328Z","shell.execute_reply.started":"2022-11-16T16:03:11.464586Z","shell.execute_reply":"2022-11-16T16:03:32.266360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}