{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"VER = 6\n\nimport pandas as pd, numpy as np\nfrom tqdm.notebook import tqdm\nimport os, sys, pickle, glob, gc\nfrom collections import Counter\nimport itertools\n# import cudf, itertools\n# print('We will use RAPIDS version',cudf.__version__)","metadata":{"papermill":{"duration":2.845087,"end_time":"2022-11-10T16:03:27.484916","exception":false,"start_time":"2022-11-10T16:03:24.639829","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-01-25T02:52:09.118676Z","iopub.execute_input":"2023-01-25T02:52:09.119142Z","iopub.status.idle":"2023-01-25T02:52:09.125709Z","shell.execute_reply.started":"2023-01-25T02:52:09.119106Z","shell.execute_reply":"2023-01-25T02:52:09.124509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time\n# # CACHE FUNCTIONS\n# def read_file(f):\n#     return cudf.DataFrame( data_cache[f] )\n# def read_file_to_cache(f):\n#     df = pd.read_parquet(f)\n#     df.ts = (df.ts/1000).astype('int32')\n#     df['type'] = df['type'].map(type_labels).astype('int8')\n#     return df\n\n# # CACHE THE DATA ON CPU BEFORE PROCESSING ON GPU\n# data_cache = {}\n# type_labels = {'clicks':0, 'carts':1, 'orders':2}\n# files = glob.glob('../input/otto-validation/*_parquet/*')\n# for f in files: data_cache[f] = read_file_to_cache(f)\n\n# # CHUNK PARAMETERS\n# READ_CT = 5\n# CHUNK = int( np.ceil( len(files)/6 ))\n# print(f'We will process {len(files)} files, in groups of {READ_CT} and chunks of {CHUNK}.')","metadata":{"papermill":{"duration":0.044165,"end_time":"2022-11-10T16:03:27.542811","exception":false,"start_time":"2022-11-10T16:03:27.498646","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-01-25T02:12:35.182291Z","iopub.execute_input":"2023-01-25T02:12:35.182689Z","iopub.status.idle":"2023-01-25T02:12:35.188825Z","shell.execute_reply.started":"2023-01-25T02:12:35.182647Z","shell.execute_reply":"2023-01-25T02:12:35.187842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type_weight = {0:0.5,\n               1:9,\n               2:0.5}\ntype_weight_multipliers = type_weight\n\n# Use top X for clicks, carts and orders Top何位までを使うか\nclicks_th = 15 # クリック数\ncarts_th  = 20 # カート数\norders_th = 20 # 購入数\n\nTOP_OUTPUT = 100","metadata":{"execution":{"iopub.status.busy":"2023-01-25T02:43:06.348912Z","iopub.execute_input":"2023-01-25T02:43:06.349359Z","iopub.status.idle":"2023-01-25T02:43:06.356470Z","shell.execute_reply.started":"2023-01-25T02:43:06.349326Z","shell.execute_reply":"2023-01-25T02:43:06.354956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1) \"Carts Orders\" Co-visitation Matrix - Type Weighted","metadata":{"papermill":{"duration":0.004349,"end_time":"2022-11-10T16:03:27.551761","exception":false,"start_time":"2022-11-10T16:03:27.547412","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# %%time\n\n# # USE SMALLEST DISK_PIECES POSSIBLE WITHOUT MEMORY ERROR\n# DISK_PIECES = 4\n# SIZE = 1.86e6/DISK_PIECES\n\n# # COMPUTE IN PARTS FOR MEMORY MANGEMENT\n# for PART in range(DISK_PIECES):\n#     print()\n#     print('### DISK PART',PART+1)\n    \n#     # MERGE IS FASTEST PROCESSING CHUNKS WITHIN CHUNKS\n#     # => OUTER CHUNKS\n#     for j in range(6):\n#         a = j*CHUNK\n#         b = min( (j+1)*CHUNK, len(files) )\n#         print(f'Processing files {a} thru {b-1} in groups of {READ_CT}...')\n        \n#         # => INNER CHUNKS\n#         for k in range(a,b,READ_CT):\n#             # READ FILE\n#             df = [read_file(files[k])]\n#             for i in range(1,READ_CT): \n#                 if k+i<b: df.append( read_file(files[k+i]) )\n#             df = cudf.concat(df,ignore_index=True,axis=0)\n#             df = df.sort_values(['session','ts'],ascending=[True,False])\n            \n#             # USE TAIL OF SESSION\n#             df = df.reset_index(drop=True)\n#             df['n'] = df.groupby('session').cumcount()\n#             df = df.loc[df.n<30].drop('n',axis=1)\n            \n#             # CREATE PAIRS\n#             df = df.merge(df,on='session')\n#             df = df.loc[ ((df.ts_x - df.ts_y).abs()< 24 * 60 * 60) & (df.aid_x != df.aid_y) ]\n            \n#             # MEMORY MANAGEMENT COMPUTE IN PARTS\n#             df = df.loc[(df.aid_x >= PART*SIZE)&(df.aid_x < (PART+1)*SIZE)]\n            \n#             # ASSIGN WEIGHTS\n#             df = df[['session', 'aid_x', 'aid_y','type_y']].drop_duplicates(['session', 'aid_x', 'aid_y', 'type_y'])\n#             df['wgt'] = df.type_y.map(type_weight)\n#             df = df[['aid_x','aid_y','wgt']]\n#             df.wgt = df.wgt.astype('float32')\n#             df = df.groupby(['aid_x','aid_y']).wgt.sum()\n            \n#             # COMBINE INNER CHUNKS\n#             if k==a: tmp2 = df\n#             else: tmp2 = tmp2.add(df, fill_value=0)\n#             print(k,', ',end='')\n        \n#         print()\n        \n#         # COMBINE OUTER CHUNKS\n#         if a==0: tmp = tmp2\n#         else: tmp = tmp.add(tmp2, fill_value=0)\n#         del tmp2, df\n#         gc.collect()\n\n#     # CONVERT MATRIX TO DICTIONARY\n#     tmp = tmp.reset_index()\n#     tmp = tmp.sort_values(['aid_x','wgt'],ascending=[True,False])\n    \n#     # SAVE TOP 40\n#     tmp = tmp.reset_index(drop=True)\n#     tmp['n'] = tmp.groupby('aid_x').aid_y.cumcount()\n#     tmp = tmp.loc[tmp.n<carts_th].drop('n',axis=1)\n    \n#     # SAVE PART TO DISK (convert to pandas first uses less memory)\n#     tmp.to_pandas().to_parquet(f'top_15_carts_orders_v{VER}_{PART}.pqt')","metadata":{"papermill":{"duration":403.713782,"end_time":"2022-11-10T16:10:11.270161","exception":false,"start_time":"2022-11-10T16:03:27.556379","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-01-25T02:12:35.201908Z","iopub.execute_input":"2023-01-25T02:12:35.202613Z","iopub.status.idle":"2023-01-25T02:12:35.211753Z","shell.execute_reply.started":"2023-01-25T02:12:35.202577Z","shell.execute_reply":"2023-01-25T02:12:35.210899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2) \"Buy2Buy\" Co-visitation Matrix","metadata":{"papermill":{"duration":0.030783,"end_time":"2022-11-10T16:10:11.331839","exception":false,"start_time":"2022-11-10T16:10:11.301056","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# %%time\n# # USE SMALLEST DISK_PIECES POSSIBLE WITHOUT MEMORY ERROR\n# DISK_PIECES = 1\n# SIZE = 1.86e6/DISK_PIECES\n\n# # COMPUTE IN PARTS FOR MEMORY MANGEMENT\n# for PART in range(DISK_PIECES):\n#     print()\n#     print('### DISK PART',PART+1)\n    \n#     # MERGE IS FASTEST PROCESSING CHUNKS WITHIN CHUNKS\n#     # => OUTER CHUNKS\n#     for j in range(6):\n#         a = j*CHUNK\n#         b = min( (j+1)*CHUNK, len(files) )\n#         print(f'Processing files {a} thru {b-1} in groups of {READ_CT}...')\n        \n#         # => INNER CHUNKS\n#         for k in range(a,b,READ_CT):\n            \n#             # READ FILE\n#             df = [read_file(files[k])]\n#             for i in range(1,READ_CT): \n#                 if k+i<b: df.append( read_file(files[k+i]) )\n#             df = cudf.concat(df,ignore_index=True,axis=0)\n#             df = df.loc[df['type'].isin([1,2])] # ONLY WANT CARTS AND ORDERS\n#             df = df.sort_values(['session','ts'],ascending=[True,False])\n            \n#             # USE TAIL OF SESSION\n#             df = df.reset_index(drop=True)\n#             df['n'] = df.groupby('session').cumcount()\n#             df = df.loc[df.n<30].drop('n',axis=1)\n            \n#             # CREATE PAIRS\n#             df = df.merge(df,on='session')\n#             df = df.loc[ ((df.ts_x - df.ts_y).abs()< 14 * 24 * 60 * 60) & (df.aid_x != df.aid_y) ] # 14 DAYS\n            \n#             # MEMORY MANAGEMENT COMPUTE IN PARTS\n#             df = df.loc[(df.aid_x >= PART*SIZE)&(df.aid_x < (PART+1)*SIZE)]\n            \n#             # ASSIGN WEIGHTS\n#             df = df[['session', 'aid_x', 'aid_y','type_y']].drop_duplicates(['session', 'aid_x', 'aid_y', 'type_y'])\n#             df['wgt'] = 1\n#             df = df[['aid_x','aid_y','wgt']]\n#             df.wgt = df.wgt.astype('float32')\n#             df = df.groupby(['aid_x','aid_y']).wgt.sum()\n            \n#             # COMBINE INNER CHUNKS\n#             if k==a: tmp2 = df\n#             else: tmp2 = tmp2.add(df, fill_value=0)\n#             print(k,', ',end='')\n\n#         print()\n        \n#         # COMBINE OUTER CHUNKS\n#         if a==0: tmp = tmp2\n#         else: tmp = tmp.add(tmp2, fill_value=0)\n#         del tmp2, df\n#         gc.collect()\n\n#     # CONVERT MATRIX TO DICTIONARY\n#     tmp = tmp.reset_index()\n#     tmp = tmp.sort_values(['aid_x','wgt'],ascending=[True,False])\n    \n#     # SAVE TOP 40\n#     tmp = tmp.reset_index(drop=True)\n#     tmp['n'] = tmp.groupby('aid_x').aid_y.cumcount()\n#     tmp = tmp.loc[tmp.n<orders_th].drop('n',axis=1)\n    \n#     # SAVE PART TO DISK (convert to pandas first uses less memory)\n#     tmp.to_pandas().to_parquet(f'top_15_buy2buy_v{VER}_{PART}.pqt')","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"papermill":{"duration":84.918311,"end_time":"2022-11-10T16:11:36.280983","exception":false,"start_time":"2022-11-10T16:10:11.362672","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-01-25T02:12:35.213293Z","iopub.execute_input":"2023-01-25T02:12:35.213816Z","iopub.status.idle":"2023-01-25T02:12:35.226735Z","shell.execute_reply.started":"2023-01-25T02:12:35.213776Z","shell.execute_reply":"2023-01-25T02:12:35.225738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3) \"Clicks\" Co-visitation Matrix - Time Weighted","metadata":{"papermill":{"duration":0.036979,"end_time":"2022-11-10T16:11:36.356076","exception":false,"start_time":"2022-11-10T16:11:36.319097","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# %%time\n# # USE SMALLEST DISK_PIECES POSSIBLE WITHOUT MEMORY ERROR\n# DISK_PIECES = 4\n# SIZE = 1.86e6/DISK_PIECES\n\n# # COMPUTE IN PARTS FOR MEMORY MANGEMENT\n# for PART in range(DISK_PIECES):\n#     print()\n#     print('### DISK PART',PART+1)\n    \n#     # MERGE IS FASTEST PROCESSING CHUNKS WITHIN CHUNKS\n#     # => OUTER CHUNKS\n#     for j in range(6):\n#         a = j*CHUNK\n#         b = min( (j+1)*CHUNK, len(files) )\n#         print(f'Processing files {a} thru {b-1} in groups of {READ_CT}...')\n        \n#         # => INNER CHUNKS\n#         for k in range(a,b,READ_CT):\n#             # READ FILE\n#             df = [read_file(files[k])]\n#             for i in range(1,READ_CT): \n#                 if k+i<b: df.append( read_file(files[k+i]) )\n#             df = cudf.concat(df,ignore_index=True,axis=0)\n#             df = df.sort_values(['session','ts'],ascending=[True,False])\n            \n#             # USE TAIL OF SESSION\n#             df = df.reset_index(drop=True)\n#             df['n'] = df.groupby('session').cumcount()\n#             df = df.loc[df.n<30].drop('n',axis=1)\n            \n#             # CREATE PAIRS\n#             df = df.merge(df,on='session')\n#             df = df.loc[ ((df.ts_x - df.ts_y).abs()< 24 * 60 * 60) & (df.aid_x != df.aid_y) ]\n            \n#             # MEMORY MANAGEMENT COMPUTE IN PARTS\n#             df = df.loc[(df.aid_x >= PART*SIZE)&(df.aid_x < (PART+1)*SIZE)]\n            \n#             # ASSIGN WEIGHTS\n#             df = df[['session', 'aid_x', 'aid_y','ts_x']].drop_duplicates(['session', 'aid_x', 'aid_y'])\n#             df['wgt'] = 1 + 3*(df.ts_x - 1659304800)/(1662328791-1659304800)\n#             # 1659304800 : minimum timestamp\n#             # 1662328791 : maximum timestamp\n#             df = df[['aid_x','aid_y','wgt']]\n#             df.wgt = df.wgt.astype('float32')\n#             df = df.groupby(['aid_x','aid_y']).wgt.sum()\n            \n#             # COMBINE INNER CHUNKS\n#             if k==a: tmp2 = df\n#             else: tmp2 = tmp2.add(df, fill_value=0)\n#             print(k,', ',end='')\n#         print()\n        \n#         # COMBINE OUTER CHUNKS\n#         if a==0: tmp = tmp2\n#         else: tmp = tmp.add(tmp2, fill_value=0)\n#         del tmp2, df\n#         gc.collect()\n\n#     # CONVERT MATRIX TO DICTIONARY\n#     tmp = tmp.reset_index()\n#     tmp = tmp.sort_values(['aid_x','wgt'],ascending=[True,False])\n    \n#     # SAVE TOP 40\n#     tmp = tmp.reset_index(drop=True)\n#     tmp['n'] = tmp.groupby('aid_x').aid_y.cumcount()\n#     tmp = tmp.loc[tmp.n<clicks_th].drop('n',axis=1)\n    \n#     # SAVE PART TO DISK (convert to pandas first uses less memory)\n#     tmp.to_pandas().to_parquet(f'top_20_clicks_v{VER}_{PART}.pqt')","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"papermill":{"duration":null,"end_time":null,"exception":false,"start_time":"2022-11-10T16:11:36.393657","status":"running"},"tags":[],"execution":{"iopub.status.busy":"2023-01-25T02:12:35.228163Z","iopub.execute_input":"2023-01-25T02:12:35.228736Z","iopub.status.idle":"2023-01-25T02:12:35.242174Z","shell.execute_reply.started":"2023-01-25T02:12:35.228702Z","shell.execute_reply":"2023-01-25T02:12:35.241200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # FREE MEMORY\n# del data_cache, tmp\n# _ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-25T02:12:35.243713Z","iopub.execute_input":"2023-01-25T02:12:35.244059Z","iopub.status.idle":"2023-01-25T02:12:35.255878Z","shell.execute_reply.started":"2023-01-25T02:12:35.244025Z","shell.execute_reply":"2023-01-25T02:12:35.254887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 2 - ReRank (choose 20) using handcrafted rules\nFor description of the handcrafted rules, read this notebook's intro.","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]}},{"cell_type":"code","source":"type_labels = {'clicks':0, 'carts':1, 'orders':2}\ndef load_test():    \n    dfs = []\n    for e, chunk_file in enumerate(glob.glob('../input/otto-validation/test_parquet/*')):\n        chunk = pd.read_parquet(chunk_file)\n        chunk.ts = (chunk.ts/1000).astype('int32')\n        chunk['type'] = chunk['type'].map(type_labels).astype('int8')\n        dfs.append(chunk)\n    return pd.concat(dfs).reset_index(drop=True) #.astype({\"ts\": \"datetime64[ms]\"})\n\ntest_df = load_test()\nprint('Test data has shape',test_df.shape)\ntest_df.head()","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-01-25T02:40:11.953960Z","iopub.execute_input":"2023-01-25T02:40:11.954405Z","iopub.status.idle":"2023-01-25T02:40:15.264621Z","shell.execute_reply.started":"2023-01-25T02:40:11.954368Z","shell.execute_reply":"2023-01-25T02:40:15.263220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nDISK_PIECES = 4\n# LOAD THREE CO-VISITATION MATRICES\ndef pqt_to_dict(df):\n    return df.groupby('aid_x').aid_y.apply(list).to_dict()\n\ntop_20_clicks = pqt_to_dict( pd.read_parquet(f'/kaggle/input/otto-public-rerank-valid/top_20_clicks_v{VER}_0.pqt') )\nfor k in range(1,DISK_PIECES): \n    top_20_clicks.update( pqt_to_dict( pd.read_parquet(f'/kaggle/input/otto-public-rerank-valid/top_20_clicks_v{VER}_{k}.pqt') ) )\ntop_20_buys = pqt_to_dict( pd.read_parquet(f'/kaggle/input/otto-public-rerank-valid/top_15_carts_orders_v{VER}_0.pqt') )\nfor k in range(1,DISK_PIECES): \n    top_20_buys.update( pqt_to_dict( pd.read_parquet(f'/kaggle/input/otto-public-rerank-valid/top_15_carts_orders_v{VER}_{k}.pqt') ) )\ntop_20_buy2buy = pqt_to_dict( pd.read_parquet(f'/kaggle/input/otto-public-rerank-valid/top_15_buy2buy_v{VER}_0.pqt') )\n\n# TOP CLICKS AND ORDERS IN TEST\ntop_clicks = test_df.loc[test_df['type']=='clicks','aid'].value_counts().index.values[:TOP_OUTPUT]\ntop_carts = test_df.loc[test_df['type']=='carts','aid'].value_counts().index.values[:TOP_OUTPUT]\ntop_orders = test_df.loc[test_df['type']=='orders','aid'].value_counts().index.values[:TOP_OUTPUT]\n\nprint('Here are size of our 3 co-visitation matrices:')\nprint( len( top_20_clicks ), len( top_20_buy2buy ), len( top_20_buys ) )","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-01-25T02:43:14.236340Z","iopub.execute_input":"2023-01-25T02:43:14.236785Z","iopub.status.idle":"2023-01-25T02:45:08.531341Z","shell.execute_reply.started":"2023-01-25T02:43:14.236742Z","shell.execute_reply":"2023-01-25T02:45:08.530105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def suggest_clicks(df):\n    # USE USER HISTORY AIDS AND TYPES\n    aids=df.aid.tolist()\n    types = df.type.tolist()\n    unique_aids = list(dict.fromkeys(aids[::-1] ))\n    # RERANK CANDIDATES USING WEIGHTS\n    if len(unique_aids)>=TOP_OUTPUT:\n        weights=np.logspace(0.1,1,len(aids),base=2, endpoint=True)-1\n        aids_temp = Counter() \n        # RERANK BASED ON REPEAT ITEMS AND TYPE OF ITEMS\n        for aid,w,t in zip(aids,weights,types): \n            aids_temp[aid] += w * type_weight_multipliers[t]\n        sorted_aids = [k for k,v in aids_temp.most_common(TOP_OUTPUT)]\n        return sorted_aids\n    # USE \"CLICKS\" CO-VISITATION MATRIX\n    aids2 = list(itertools.chain(*[top_20_clicks[aid] for aid in unique_aids if aid in top_20_clicks]))\n    # RERANK CANDIDATES\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids2).most_common(TOP_OUTPUT) if aid2 not in unique_aids]    \n    result = unique_aids + top_aids2[:TOP_OUTPUT - len(unique_aids)]\n    # USE TOP20 TEST CLICKS\n    return result + list(top_clicks)[:TOP_OUTPUT-len(result)]","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-01-25T02:42:20.435313Z","iopub.execute_input":"2023-01-25T02:42:20.436214Z","iopub.status.idle":"2023-01-25T02:42:20.481492Z","shell.execute_reply.started":"2023-01-25T02:42:20.436161Z","shell.execute_reply":"2023-01-25T02:42:20.479484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def suggest_carts(df):\n    # User history aids and types\n    aids = df.aid.tolist()\n    types = df.type.tolist()\n    \n    # UNIQUE AIDS AND UNIQUE BUYS\n    unique_aids = list(dict.fromkeys(aids[::-1] ))\n    df = df.loc[(df['type'] == 0)|(df['type'] == 1)]\n    unique_buys = list(dict.fromkeys(df.aid.tolist()[::-1]))\n    \n    # Rerank candidates using weights\n    if len(unique_aids) >= TOP_OUTPUT:\n        weights=np.logspace(0.5,1,len(aids),base=2, endpoint=True)-1\n        aids_temp = Counter() \n        \n        # Rerank based on repeat items and types of items\n        for aid,w,t in zip(aids,weights,types): \n            aids_temp[aid] += w * type_weight_multipliers[t]\n        \n        # Rerank candidates using\"top_20_carts\" co-visitation matrix\n        aids2 = list(itertools.chain(*[top_20_buys[aid] for aid in unique_buys if aid in top_20_buys]))\n        for aid in aids2: aids_temp[aid] += 0.1\n        sorted_aids = [k for k,v in aids_temp.most_common(TOP_OUTPUT)]\n        return sorted_aids\n    \n    # Use \"cart order\" and \"clicks\" co-visitation matrices\n    aids1 = list(itertools.chain(*[top_20_clicks[aid] for aid in unique_aids if aid in top_20_clicks]))\n    aids2 = list(itertools.chain(*[top_20_buys[aid] for aid in unique_aids if aid in top_20_buys]))\n    \n    # RERANK CANDIDATES\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids1+aids2).most_common(TOP_OUTPUT) if aid2 not in unique_aids] \n    result = unique_aids + top_aids2[:TOP_OUTPUT - len(unique_aids)]\n    \n    # USE TOP20 TEST ORDERS\n    return result + list(top_carts)[:TOP_OUTPUT-len(result)]","metadata":{"execution":{"iopub.status.busy":"2023-01-25T02:42:20.484695Z","iopub.execute_input":"2023-01-25T02:42:20.486177Z","iopub.status.idle":"2023-01-25T02:42:20.508441Z","shell.execute_reply.started":"2023-01-25T02:42:20.486120Z","shell.execute_reply":"2023-01-25T02:42:20.506288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def suggest_buys(df):\n    # USER HISTORY AIDS AND TYPES\n    aids=df.aid.tolist()\n    types = df.type.tolist()\n    # UNIQUE AIDS AND UNIQUE BUYS\n    unique_aids = list(dict.fromkeys(aids[::-1] ))\n    df = df.loc[(df['type']==1)|(df['type']==2)]\n    unique_buys = list(dict.fromkeys( df.aid.tolist()[::-1] ))\n    # RERANK CANDIDATES USING WEIGHTS\n    if len(unique_aids)>=TOP_OUTPUT:\n        weights=np.logspace(0.5,1,len(aids),base=2, endpoint=True)-1\n        aids_temp = Counter() \n        # RERANK BASED ON REPEAT ITEMS AND TYPE OF ITEMS\n        for aid,w,t in zip(aids,weights,types): \n            aids_temp[aid] += w * type_weight_multipliers[t]\n        # RERANK CANDIDATES USING \"BUY2BUY\" CO-VISITATION MATRIX\n        aids3 = list(itertools.chain(*[top_20_buy2buy[aid] for aid in unique_buys if aid in top_20_buy2buy]))\n        for aid in aids3: aids_temp[aid] += 0.1\n        sorted_aids = [k for k,v in aids_temp.most_common(TOP_OUTPUT)]\n        return sorted_aids\n    # USE \"CART ORDER\" CO-VISITATION MATRIX\n    aids2 = list(itertools.chain(*[top_20_buys[aid] for aid in unique_aids if aid in top_20_buys]))\n    # USE \"BUY2BUY\" CO-VISITATION MATRIX\n    aids3 = list(itertools.chain(*[top_20_buy2buy[aid] for aid in unique_buys if aid in top_20_buy2buy]))\n    # RERANK CANDIDATES\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids2+aids3).most_common(TOP_OUTPUT) if aid2 not in unique_aids] \n    result = unique_aids + top_aids2[:TOP_OUTPUT - len(unique_aids)]\n    # USE TOP20 TEST ORDERS\n    return result + list(top_orders)[:TOP_OUTPUT-len(result)]","metadata":{"execution":{"iopub.status.busy":"2023-01-25T02:42:20.510825Z","iopub.execute_input":"2023-01-25T02:42:20.511470Z","iopub.status.idle":"2023-01-25T02:42:20.529569Z","shell.execute_reply.started":"2023-01-25T02:42:20.511413Z","shell.execute_reply":"2023-01-25T02:42:20.527884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Submission CSV\nInferring test data with Pandas groupby is slow. We need to accelerate the following code.","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]}},{"cell_type":"code","source":"%%time\n\npred_df_clicks = test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_clicks(x)\n)\n\npred_df_carts = test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_carts(x)\n)\n\npred_df_buys = test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_buys(x)\n)","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-01-25T02:52:51.517775Z","iopub.execute_input":"2023-01-25T02:52:51.518192Z","iopub.status.idle":"2023-01-25T02:52:51.558740Z","shell.execute_reply.started":"2023-01-25T02:52:51.518160Z","shell.execute_reply":"2023-01-25T02:52:51.557257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clicks_pred_df = pd.DataFrame(pred_df_clicks.add_suffix(\"_clicks\"), columns=[\"labels\"]).reset_index()\norders_pred_df = pd.DataFrame(pred_df_buys.add_suffix(\"_orders\"), columns=[\"labels\"]).reset_index()\ncarts_pred_df = pd.DataFrame(pred_df_carts.add_suffix(\"_carts\"), columns=[\"labels\"]).reset_index()","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-01-25T02:52:59.574854Z","iopub.execute_input":"2023-01-25T02:52:59.575274Z","iopub.status.idle":"2023-01-25T02:52:59.586012Z","shell.execute_reply.started":"2023-01-25T02:52:59.575240Z","shell.execute_reply":"2023-01-25T02:52:59.585026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.concat([clicks_pred_df, orders_pred_df, carts_pred_df])\npred_df.columns = [\"session_type\", \"labels\"]\npred_df[\"labels\"] = pred_df.labels.apply(lambda x: \" \".join(map(str,x)))\npred_df.to_csv(\"validation_preds.csv\", index=False)\npred_df.head()","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-01-25T02:53:00.834711Z","iopub.execute_input":"2023-01-25T02:53:00.835493Z","iopub.status.idle":"2023-01-25T02:53:00.855082Z","shell.execute_reply.started":"2023-01-25T02:53:00.835455Z","shell.execute_reply":"2023-01-25T02:53:00.853999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compute Validation Score\nThis code is from Radek [here][1]. It has been modified to use less memory.\n\n[1]: https://www.kaggle.com/competitions/otto-recommender-system/discussion/364991","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]}},{"cell_type":"code","source":"# FREE MEMORY\ndel pred_df_clicks, pred_df_buys, clicks_pred_df, orders_pred_df, carts_pred_df\ndel top_20_clicks, top_20_buy2buy, top_20_buys, top_clicks, top_carts, top_orders, test_df\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-25T02:53:02.619219Z","iopub.execute_input":"2023-01-25T02:53:02.619718Z","iopub.status.idle":"2023-01-25T02:53:06.935583Z","shell.execute_reply.started":"2023-01-25T02:53:02.619681Z","shell.execute_reply":"2023-01-25T02:53:06.934169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# COMPUTE METRIC\nscore = 0\nweights = {'clicks': 0.10, 'carts': 0.30, 'orders': 0.60}\nfor t in ['clicks','carts','orders']:\n    sub = pred_df.loc[pred_df.session_type.str.contains(t)].copy()\n    sub['session'] = sub.session_type.apply(lambda x: int(x.split('_')[0]))\n    sub.labels = sub.labels.apply(lambda x: [int(i) for i in x.split(' ')])\n    test_labels = pd.read_parquet('../input/otto-validation/test_labels.parquet')\n    test_labels = test_labels.loc[test_labels['type']==t]\n    test_labels = test_labels.merge(sub, how='left', on=['session'])\n    test_labels['hits'] = test_labels.apply(lambda df: len(set(df.ground_truth).intersection(set(df.labels))), axis=1)\n    test_labels['gt_count'] = test_labels.ground_truth.str.len().clip(0,20)\n    recall = test_labels['hits'].sum() / test_labels['gt_count'].sum()\n    score += weights[t]*recall\n    print(f'{t} recall =',recall)\n    \nprint('=============')\nprint('Overall Recall =',score)\nprint('=============')","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-01-25T02:53:06.937595Z","iopub.execute_input":"2023-01-25T02:53:06.938078Z","iopub.status.idle":"2023-01-25T02:53:10.518016Z","shell.execute_reply.started":"2023-01-25T02:53:06.938041Z","shell.execute_reply":"2023-01-25T02:53:10.516123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}