{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Beginners Guide to Aid\n\nThank you for the great notebook by @vslaykovsky [here](https://www.kaggle.com/code/vslaykovsky/co-visitation-matrix) and @radek1 [here](https://www.kaggle.com/code/radek1/co-visitation-matrix-simplified-imprvd-logic/). I don't reproduce what they did just make it easier to understand for beginners and also to help you can make your own variations as needed! I was inspired by a comment by @danielliao which commented on the code. I tweaked that and put it into a notebook for everyone to use. \n\nSo, Aid. What does it mean?! And how do you get it?\n\n![What is this?](https://imgur.com/t/jack_skellington/nGO7VyD)\n\n<img src= \"https://imgur.com/t/jack_skellington/nGO7VyD\" alt =\"Jack Skellington\" style='width: 200px;'>","metadata":{}},{"cell_type":"markdown","source":"First step is to add data:\nhttps://www.kaggle.com/datasets/radek1/otto-full-optimized-memory-footprint","metadata":{}},{"cell_type":"code","source":"fraction_of_sessions_to_use = 1\n\nimport pandas as pd\nimport numpy as np\n\n# read the OTTO parquet \n\ntrain = pd.read_parquet('../input/otto-full-optimized-memory-footprint//train.parquet')\ntest = pd.read_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')\n\n# read submission csv file\nsample_sub = pd.read_csv('../input/otto-recommender-system/sample_submission.csv')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-23T13:47:27.412990Z","iopub.execute_input":"2022-11-23T13:47:27.413354Z","iopub.status.idle":"2022-11-23T13:47:34.055761Z","shell.execute_reply.started":"2022-11-23T13:47:27.413323Z","shell.execute_reply":"2022-11-23T13:47:34.054786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# taking subset of the sessions data to make this faster\nfraction_of_sessions_to_use = 1\n\n# take subset\n# what does frac do? Interesting discussion and example here:\n# https://stackoverflow.com/questions/71758460/effect-of-pandas-dataframe-sample-with-frac-set-to-1\n\nif fraction_of_sessions_to_use != 1:\n    lucky_sessions_train = train.drop_duplicates(['session']).sample(frac=fraction_of_sessions_to_use, random_state=42)['session']\n    subset_of_train = train[train.session.isin(lucky_sessions_train)]\n    \n    lucky_sessions_test = test.drop_duplicates(['session']).sample(frac=fraction_of_sessions_to_use, random_state=42)['session']\n    subset_of_test = test[test.session.isin(lucky_sessions_test)]\nelse:\n    subset_of_train = train\n    subset_of_test = test","metadata":{"execution":{"iopub.status.busy":"2022-11-23T13:47:34.057405Z","iopub.execute_input":"2022-11-23T13:47:34.057690Z","iopub.status.idle":"2022-11-23T13:47:34.111666Z","shell.execute_reply.started":"2022-11-23T13:47:34.057662Z","shell.execute_reply":"2022-11-23T13:47:34.110511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# session column set as index\n\nsubset_of_train.index = pd.MultiIndex.from_frame(subset_of_train[['session']])\nsubset_of_test.index = pd.MultiIndex.from_frame(subset_of_test[['session']])\n\n# pecify chunksize to read chunks of 30,000 rows of data at a time\nchunk_size = 30_000\n### get starting and end timestamp for all sessions\nmin_ts = train.ts.min()\nmax_ts = test.ts.max()\n\n### use defaultdict + Counter to count occurrences of each\nfrom collections import defaultdict, Counter\nnext_AIDs = defaultdict(Counter)\n\n### use test for training\nsubsets = pd.concat([subset_of_train, subset_of_test])\nsessions = subsets.session.unique()","metadata":{"execution":{"iopub.status.busy":"2022-11-23T13:47:34.113133Z","iopub.execute_input":"2022-11-23T13:47:34.113459Z","iopub.status.idle":"2022-11-23T13:47:49.557698Z","shell.execute_reply.started":"2022-11-23T13:47:34.113431Z","shell.execute_reply":"2022-11-23T13:47:49.552772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# view the subsets for a sanity check\nsubsets","metadata":{"execution":{"iopub.status.busy":"2022-11-23T13:47:49.563068Z","iopub.execute_input":"2022-11-23T13:47:49.563926Z","iopub.status.idle":"2022-11-23T13:47:49.613754Z","shell.execute_reply.started":"2022-11-23T13:47:49.563870Z","shell.execute_reply":"2022-11-23T13:47:49.611551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"#Training: when one aid occurred, keep track of what other aid occurred, how often they occurred. \n#Do this for every aid across both train and test sessions.\n\n# loop every chunk_size number unique sessions\nfor i in range(0, sessions.shape[0], chunk_size):\n    # For number of sessions from subsets as current_chunk\n    current_chunk = subsets.loc[sessions[i]:sessions[min(sessions.shape[0]-1, i+chunk_size-1)]].reset_index(drop=True)\n    # from each session, take the last 30 events and combine them to update current_chunk\n    current_chunk = current_chunk.groupby('session', as_index=False).nth(list(range(-30,0))).reset_index(drop=True)\n    # merge session on itself to understand the relationship between aids within each session\n    consecutive_AIDs = current_chunk.merge(current_chunk, on='session')\n    # remove all the rows which aid_x == aid_y (remove the row when the two articles are the same) as they are meaningless\n    consecutive_AIDs = consecutive_AIDs[consecutive_AIDs.aid_x != consecutive_AIDs.aid_y]\n    # add a column named 'days_elapsed' which shows how many days passed between the two aids in a session\n    consecutive_AIDs['days_elapsed'] = (consecutive_AIDs.ts_y - consecutive_AIDs.ts_x) / (24 * 60 * 60)\n    # keep the rows if the two aids of a session are occurred within the same day on the right order (one after the other)\n    consecutive_AIDs = consecutive_AIDs[(consecutive_AIDs.days_elapsed >= 0) & (consecutive_AIDs.days_elapsed <= 1)]\n\n    # count how often the other aids are occurred\n    for aid_x, aid_y in zip(consecutive_AIDs['aid_x'], consecutive_AIDs['aid_y']):\n        next_AIDs[aid_x][aid_y] += 1\n\n# remove some data objects to save RAM\ndel train, subset_of_train, subsets\n\n## make predictions\n\nsession_types = ['clicks', 'carts', 'orders']\n###  group the test set by session, under each session, put all aids into a list, and put all action types into another list\ntest_session_AIDs = test.reset_index(drop=True).groupby('session')['aid'].apply(list)\n####  put all article ids or type into a list under the same session, test.reset_index(drop=True).groupby('session')['aid'].apply(list)\n\ntest_session_types = test.reset_index(drop=True).groupby('session')['type'].apply(list)\n\n###  loop every session, access all of its aids and types\n###  setup, create some containers, such as labels, no_data, no_data_all_aids\n\nlabels = []\n\nno_data = 0\nno_data_all_aids = 0\ntype_weight_multipliers = {0: 1, 1: 6, 2: 3}\n\n# when there are >= 20 aids in a session: \n# assign logspace weight (https://numpy.org/doc/stable/reference/generated/numpy.logspace.html) to each aid for each session\n# as the latter aids should have higher weight/probability to occur than the earlier aids\n\nfor AIDs, types in zip(test_session_AIDs, test_session_types):\n    if len(AIDs) >= 20:\n        # exponential weight decay, or even using df.ts and time-dependent weight decay\n        weights=np.logspace(0.1,1,len(AIDs),base=2, endpoint=True)-1\n        # create a defaultdict (if no value to the key, set value to 0)\n        aids_temp=defaultdict(lambda: 0)\n        # loop each aid, weight, event_type of a session:\n        # Within each session, accumulate the weight for each aid based on its occurences, event_type and logspaced weight; \n        # save the accumulated weight as value and each aid as key into a defaultdict (aids_temp)\n        # no duplicated aid here in this dict, and every session has its own aid_temp\n        \n        # to each aids give the sum of the weights of its occurences within the session \n        for aid,w,t in zip(AIDs,weights,types): \n            aids_temp[aid]+= w * type_weight_multipliers[t]\n        # order in descending order the aids depending on its weight\n        # put its keys into a list named sorted_aids\n        # store the first 20 aids (the most weighted or most likely aids to be acted upon within a session) into the list 'labels'\n        sorted_aids=[k for k, v in sorted(aids_temp.items(), key=lambda item: -item[1])]\n        labels.append(sorted_aids[:20])\n        \n# when there are < 20 aids in a session: \n# within each test session, reverse the order of AIDs, \n# remove the duplicated, put into a list, reassign it to AIDs\n       \n    else:\n        AIDs = list(dict.fromkeys(AIDs[::-1]))\n        AIDs_len_start = len(AIDs)\n        # keep track the length of AIDs and create an empty list named candidates\n        candidates = []\n        for AID in AIDs:\n            # (within a session) for each AID inside AIDs: if AID is in the keys of next_AIDs (from training), \n            # then take the 20 most common other aids occurred (from next_AIDs) when AID occurred, \n            # into a list and add this list into the list named candidate (not a list of list, just a merged list). \n            # Each candidate in its full size has len(AIDs) * 20 number of other aids, which can have duplicated ids.\n            if AID in next_AIDs: candidates += [aid for aid, count in next_AIDs[AID].most_common(20)]\n        # find the first 40 most common aids in a candidate (for a session); and if they (these aids) are not found in AIDs then merge them into AIDs list, so that a session has a updated AIDs list (which most likely to occur)\n        AIDs += [AID for AID, cnt in Counter(candidates).most_common(40) if AID not in AIDs]\n        # give the first 20 aids from AIDs to labels (a list); count how many test sessions whose aids are not seen in next_AIDs from training; count how many test sessions don't receive additional aids from next_AIDs\n        labels.append(AIDs[:20])\n        if candidates == []: no_data += 1\n        if AIDs_len_start == len(AIDs): no_data_all_aids += 1\n\n##  prepare results to CSV\n\n###  make a list of lists (labels) into a list of strings (labels_as_strings)\n\n\nlabels_as_strings = [' '.join([str(l) for l in lls]) for lls in labels]\n###  give each list of label strings a session number\npredictions = pd.DataFrame(data={'session_type': test_session_AIDs.index, 'labels': labels_as_strings})\n\nprediction_dfs = []\n###  multi-objective means 'clicks', 'carts', and 'orders'; and we make the same predictions on them \nfor st in session_types:\n    modified_predictions = predictions.copy()\n    modified_predictions.session_type = modified_predictions.session_type.astype('str') + f'_{st}'\n    prediction_dfs.append(modified_predictions)\n###  get the csv file ready, stack on each other.\nsubmission = pd.concat(prediction_dfs).reset_index(drop=True)\nsubmission.to_csv('submission.csv', index=False)\n\nprint(f'Test sessions that we did not manage to extend based on the co-visitation matrix: {no_data_all_aids}')","metadata":{"execution":{"iopub.status.busy":"2022-11-23T13:47:49.616742Z","iopub.execute_input":"2022-11-23T13:47:49.617295Z"},"trusted":true},"execution_count":null,"outputs":[]}]}