{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-28T01:29:23.284156Z","iopub.execute_input":"2023-01-28T01:29:23.285206Z","iopub.status.idle":"2023-01-28T01:29:23.312333Z","shell.execute_reply.started":"2023-01-28T01:29:23.285106Z","shell.execute_reply":"2023-01-28T01:29:23.311012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"USER_ID = 'session'\nITEM_ID = 'aid'\n\nCLICK = 'clicks'\nCART = 'carts'\nORDER = 'orders'","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# calc bigram count in all data","metadata":{"execution":{"iopub.status.busy":"2023-01-28T01:33:44.422042Z","iopub.execute_input":"2023-01-28T01:33:44.422489Z","iopub.status.idle":"2023-01-28T01:33:44.427545Z","shell.execute_reply.started":"2023-01-28T01:33:44.422451Z","shell.execute_reply":"2023-01-28T01:33:44.426537Z"}}},{"cell_type":"code","source":"# calc item frequency, used to normalize the covisit （very important）\ndf = pd.concat([train_actions, test_actions])\nitem_cnt = {'all': df.groupby(ITEM_ID).size().to_dict()}\nfor t in [CLICK, CART, ORDER]:\n    item_cnt[t] = df.loc[df['type'] == t].groupby(ITEM_ID).size().to_dict()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pairs = pd.concat([train_actions, test_actions])\ntrain_pairs['aid_next'] = train_pairs.groupby('sid')[ITEM_ID].shift(-1)\ntrain_pairs = train_pairs.dropna()\ntrain_pairs['aid_next'] = train_pairs['aid_next'].astype('int32')\ntrain_pairs = train_pairs[['aid', 'aid_next']]\nbigram_counter = train_pairs.groupby(['aid', 'aid_next']).size().to_dict()\nnormed_bigram_counter = {}\nfor (a1, a2), cnt in bigram_counter.items():\n    normed_bigram_counter[(a1, a2)] = cnt/np.sqrt(item_cnt['all'].get(a1, 1) * item_cnt['all'].get(a2, 1))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# make features","metadata":{}},{"cell_type":"code","source":"def get_bigram_fea(train_sample, user_2_test_action_with_type, bigram_count, name):\n    feas = {}\n    for src in [CLICK, CART]:\n        sum_sim, mean_sim, max_sim, min_sim, last_sim = [], [], [], [], []\n        last_sim2, last_sim3 = [], []\n        for idx, u, i in tqdm(train_sample[[USER_ID, ITEM_ID]].itertuples(), total=len(train_sample)):\n            acts = user_2_test_action_with_type[src].get(u, [])\n            sims = []\n            if len(acts)>0:\n                for a in acts:\n                    sims.append(bigram_count.get((a, i), 0))\n                sum_sim.append(np.sum(sims))\n                mean_sim.append(np.mean(sims))\n                max_sim.append(np.max(sims))\n                min_sim.append(np.min(sims))\n                last_sim.append(sims[-1])\n            else:\n                sum_sim.append(-1)\n                mean_sim.append(-1)\n                max_sim.append(-1)\n                min_sim.append(-1)\n                last_sim.append(-1)\n        feas.update({\n            name + f'_{src}_sum': sum_sim, name + f'_{src}_mean': mean_sim, name + f'_{src}_max': max_sim, name + f'_{src}_min': min_sim,\n            name + f'_{src}_last': last_sim\n        })\n    return pd.DataFrame(feas)","metadata":{"execution":{"iopub.status.busy":"2023-01-28T02:03:45.376512Z","iopub.execute_input":"2023-01-28T02:03:45.376930Z","iopub.status.idle":"2023-01-28T02:03:45.390890Z","shell.execute_reply.started":"2023-01-28T02:03:45.376896Z","shell.execute_reply":"2023-01-28T02:03:45.389804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"user_2_test_action_with_type = {t: test_actions.loc[test_actions['type'] == t, :].groupby(USER_ID)[ITEM_ID].agg(list) for t in [CLICK, CART, ORDER]}\n# train_sample is a dataframe of candidates, whose columns are [session, aid]\nnaive_bigram = get_bigram_fea(train_sample, user_2_test_action_with_type, bigram_counter, 'bigram')\nnormed_bigram = get_bigram_fea(train_sample, user_2_test_action_with_type, normed_bigram_counter, 'bigram_normed')","metadata":{},"execution_count":null,"outputs":[]}]}