{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"VER = 9\nimport pandas as pd, numpy as np\nfrom tqdm.notebook import tqdm\nimport os, sys, pickle, glob, gc\nfrom collections import Counter\nimport cudf, itertools\nprint('We will use RAPIDS version',cudf.__version__)","metadata":{"papermill":{"duration":3.036143,"end_time":"2022-11-10T16:03:24.014816","exception":false,"start_time":"2022-11-10T16:03:20.978673","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-01-28T12:48:37.287575Z","iopub.execute_input":"2023-01-28T12:48:37.288374Z","iopub.status.idle":"2023-01-28T12:48:40.022935Z","shell.execute_reply.started":"2023-01-28T12:48:37.288275Z","shell.execute_reply":"2023-01-28T12:48:40.021412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# CACHE FUNCTIONS\ndef read_file(f):\n    return cudf.DataFrame( data_cache[f] )\ndef read_file_to_cache(f):\n    df = pd.read_parquet(f)\n    df.ts = (df.ts/1000).astype('int32')\n    df['type'] = df['type'].map(type_labels).astype('int8')\n    return df\n\n# CACHE THE DATA ON CPU BEFORE PROCESSING ON GPU\ndata_cache = {}\ntype_labels = {'clicks':0, 'carts':1, 'orders':2}\nfiles = glob.glob('../input/otto-chunk-data-inparquet-format/*_parquet/*')\nfor f in files: \n    data_cache[f] = read_file_to_cache(f)\n\n# CHUNK PARAMETERS\nREAD_CT = 5\nCHUNK = int( np.ceil( len(files)/6 ))\nprint(f'We will process {len(files)} files, in groups of {READ_CT} and chunks of {CHUNK}.')","metadata":{"papermill":{"duration":0.063943,"end_time":"2022-11-10T16:03:24.091816","exception":false,"start_time":"2022-11-10T16:03:24.027873","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-01-28T12:48:42.655747Z","iopub.execute_input":"2023-01-28T12:48:42.656106Z","iopub.status.idle":"2023-01-28T12:49:42.430648Z","shell.execute_reply.started":"2023-01-28T12:48:42.656070Z","shell.execute_reply":"2023-01-28T12:49:42.429482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#data_cache['../input/otto-chunk-data-inparquet-format/test_parquet/000200000_000300000.parquet']","metadata":{"execution":{"iopub.status.busy":"2023-01-23T08:50:51.886921Z","iopub.execute_input":"2023-01-23T08:50:51.887624Z","iopub.status.idle":"2023-01-23T08:50:51.904017Z","shell.execute_reply.started":"2023-01-23T08:50:51.887586Z","shell.execute_reply":"2023-01-23T08:50:51.903111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(data_cache)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T08:51:10.614066Z","iopub.execute_input":"2023-01-23T08:51:10.614459Z","iopub.status.idle":"2023-01-23T08:51:10.621417Z","shell.execute_reply.started":"2023-01-23T08:51:10.614430Z","shell.execute_reply":"2023-01-23T08:51:10.620257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type_weight = {0:1, 1:9, 2:3}\n\n# USE SMALLEST DISK_PIECES POSSIBLE WITHOUT MEMORY ERROR\nDISK_PIECES = 4","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:50:53.252788Z","iopub.execute_input":"2023-01-28T12:50:53.253160Z","iopub.status.idle":"2023-01-28T12:50:53.258026Z","shell.execute_reply.started":"2023-01-28T12:50:53.253125Z","shell.execute_reply":"2023-01-28T12:50:53.256878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DISK_PIECES = 4","metadata":{"execution":{"iopub.status.busy":"2023-01-28T13:04:10.840073Z","iopub.execute_input":"2023-01-28T13:04:10.840572Z","iopub.status.idle":"2023-01-28T13:04:10.846483Z","shell.execute_reply.started":"2023-01-28T13:04:10.840527Z","shell.execute_reply":"2023-01-28T13:04:10.845488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SIZE = 1.86e6/DISK_PIECES","metadata":{"execution":{"iopub.status.busy":"2023-01-28T13:04:12.679154Z","iopub.execute_input":"2023-01-28T13:04:12.679550Z","iopub.status.idle":"2023-01-28T13:04:12.684459Z","shell.execute_reply.started":"2023-01-28T13:04:12.679515Z","shell.execute_reply":"2023-01-28T13:04:12.683278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SIZE","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:46:46.990891Z","iopub.execute_input":"2023-01-28T12:46:46.992014Z","iopub.status.idle":"2023-01-28T12:46:46.999986Z","shell.execute_reply.started":"2023-01-28T12:46:46.991965Z","shell.execute_reply":"2023-01-28T12:46:46.999019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CHUNK","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:51:26.090812Z","iopub.execute_input":"2023-01-28T12:51:26.091253Z","iopub.status.idle":"2023-01-28T12:51:26.100836Z","shell.execute_reply.started":"2023-01-28T12:51:26.091214Z","shell.execute_reply":"2023-01-28T12:51:26.099784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"1*CHUNK","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:52:44.612392Z","iopub.execute_input":"2023-01-28T12:52:44.612895Z","iopub.status.idle":"2023-01-28T12:52:44.619090Z","shell.execute_reply.started":"2023-01-28T12:52:44.612849Z","shell.execute_reply":"2023-01-28T12:52:44.618153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"READ_CT","metadata":{"execution":{"iopub.status.busy":"2023-01-28T12:57:10.082849Z","iopub.execute_input":"2023-01-28T12:57:10.083260Z","iopub.status.idle":"2023-01-28T12:57:10.089998Z","shell.execute_reply.started":"2023-01-28T12:57:10.083227Z","shell.execute_reply":"2023-01-28T12:57:10.088967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1) \"Carts Orders\" Co-visitation Matrix - Type Weighted","metadata":{"papermill":{"duration":0.004089,"end_time":"2022-11-10T16:03:24.100502","exception":false,"start_time":"2022-11-10T16:03:24.096413","status":"completed"},"tags":[]}},{"cell_type":"code","source":"#DISK_PIECES = 4\ntype_weight = {0:1, 1:9, 2:3}\n# COMPUTE IN PARTS FOR MEMORY MANGEMENT\nfor PART in range(DISK_PIECES):\n    print('---')\n    print('### DISK PART',PART+1)\n    \n    # MERGE IS FASTEST PROCESSING CHUNKS WITHIN CHUNKS\n    # => OUTER CHUNKS\n    for j in range(6):\n        a = j*CHUNK\n        b = min( (j+1)*CHUNK, len(files) )\n        print(f'Processing files {a} thru {b-1} in groups of {READ_CT}...')\n        \n        # => INNER CHUNKS\n        for k in range(a,b,READ_CT):\n            # READ FILE\n            df = [read_file(files[k])]\n            for i in range(1,READ_CT): \n                if k+i<b: df.append( read_file(files[k+i]) )\n            df = cudf.concat(df,ignore_index=True,axis=0)\n            ##df =df.loc[df['type'].isin([1,2])]  ##\n            \n            df = df.sort_values(['session','ts'],ascending=[True,False])\n            # USE TAIL OF SESSION\n            df = df.reset_index(drop=True)\n            df['n'] = df.groupby('session').cumcount()\n            df = df.loc[df.n<30].drop('n',axis=1)\n            # CREATE PAIRS\n            df = df.merge(df,on='session')\n            df = df.loc[ ((df.ts_x - df.ts_y).abs()< 24 * 60 * 60) & (df.aid_x != df.aid_y) ]\n            # MEMORY MANAGEMENT COMPUTE IN PARTS\n            df = df.loc[(df.aid_x >= PART*SIZE)&(df.aid_x < (PART+1)*SIZE)]\n            # ASSIGN WEIGHTS\n            df = df[['session', 'aid_x', 'aid_y','type_y']].drop_duplicates(['session', 'aid_x', 'aid_y'])\n            df['wgt'] = df.type_y.map(type_weight)\n            df = df[['aid_x','aid_y','wgt']]\n            df.wgt = df.wgt.astype('float32')\n            df = df.groupby(['aid_x','aid_y']).wgt.sum()\n            # COMBINE INNER CHUNKS\n            if k==a: tmp2 = df\n            else: tmp2 = tmp2.add(df, fill_value=0)\n            print(k,', ',end='')\n        print()\n        # COMBINE OUTER CHUNKS\n        if a==0: tmp = tmp2\n        else: tmp = tmp.add(tmp2, fill_value=0)\n        del tmp2, df\n        gc.collect()\n    # CONVERT MATRIX TO DICTIONARY\n    tmp = tmp.reset_index()\n    tmp = tmp.sort_values(['aid_x','wgt'],ascending=[True,False])\n    # SAVE TOP 40\n    tmp = tmp.reset_index(drop=True)\n    tmp['n'] = tmp.groupby('aid_x').aid_y.cumcount()\n    tmp = tmp.loc[tmp.n<20].drop('n',axis=1)\n    # SAVE PART TO DISK (convert to pandas first uses less memory)\n    tmp.to_pandas().to_parquet(f'top_15_carts_orders_v{VER}_{PART}.pqt')","metadata":{"papermill":{"duration":566.561189,"end_time":"2022-11-10T16:12:50.666123","exception":false,"start_time":"2022-11-10T16:03:24.104934","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-01-28T13:04:20.230946Z","iopub.execute_input":"2023-01-28T13:04:20.231558Z","iopub.status.idle":"2023-01-28T13:06:46.515327Z","shell.execute_reply.started":"2023-01-28T13:04:20.231511Z","shell.execute_reply":"2023-01-28T13:06:46.514346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2) \"Buy2Buy\" Co-visitation Matrix","metadata":{"papermill":{"duration":0.03219,"end_time":"2022-11-10T16:12:50.730634","exception":false,"start_time":"2022-11-10T16:12:50.698444","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time\n# USE SMALLEST DISK_PIECES POSSIBLE WITHOUT MEMORY ERROR\nDISK_PIECES =4\nSIZE = 1.86e6/DISK_PIECES\n\n# COMPUTE IN PARTS FOR MEMORY MANGEMENT\nfor PART in range(DISK_PIECES):\n    print()\n    print('### DISK PART',PART+1)\n    \n    # MERGE IS FASTEST PROCESSING CHUNKS WITHIN CHUNKS\n    # => OUTER CHUNKS\n    for j in range(6):\n        a = j*CHUNK\n        b = min( (j+1)*CHUNK, len(files) )\n        print(f'Processing files {a} thru {b-1} in groups of {READ_CT}...')\n        \n        # => INNER CHUNKS\n        for k in range(a,b,READ_CT):\n            # READ FILE\n            df = [read_file(files[k])]\n            for i in range(1,READ_CT): \n                if k+i<b: df.append( read_file(files[k+i]) )\n            df = cudf.concat(df,ignore_index=True,axis=0)\n            df.loc[df['type'].isin([1,2])] # ONLY WANT CARTS AND ORDERS    df =df.loc[df['type']==2] # \n            df = df.sort_values(['session','ts'],ascending=[True,False])\n            # USE TAIL OF SESSION\n            df = df.reset_index(drop=True)\n            df['n'] = df.groupby('session').cumcount()\n            df = df.loc[df.n<20].drop('n',axis=1)\n            # CREATE PAIRS\n            df = df.merge(df,on='session')\n            df = df.loc[ ((df.ts_x - df.ts_y).abs()< 14 * 24 * 60 * 60) & (df.aid_x != df.aid_y) ] # 14 DAYS\n            # MEMORY MANAGEMENT COMPUTE IN PARTS\n            df = df.loc[(df.aid_x >= PART*SIZE)&(df.aid_x < (PART+1)*SIZE)]\n            # ASSIGN WEIGHTS\n            df = df[['session', 'aid_x', 'aid_y','type_y']].drop_duplicates(['session', 'aid_x', 'aid_y'])\n            df['wgt'] = 1\n            df = df[['aid_x','aid_y','wgt']]\n            df.wgt = df.wgt.astype('float32')\n            df = df.groupby(['aid_x','aid_y']).wgt.sum()\n            # COMBINE INNER CHUNKS\n            if k==a: tmp2 = df\n            else: tmp2 = tmp2.add(df, fill_value=0)\n            print(k,', ',end='')\n        print()\n        # COMBINE OUTER CHUNKS\n        if a==0: tmp = tmp2\n        else: tmp = tmp.add(tmp2, fill_value=0)\n        del tmp2, df\n        gc.collect()\n    # CONVERT MATRIX TO DICTIONARY\n    tmp = tmp.reset_index()\n    tmp = tmp.sort_values(['aid_x','wgt'],ascending=[True,False])\n    # SAVE TOP 40\n    tmp = tmp.reset_index(drop=True)\n    tmp['n'] = tmp.groupby('aid_x').aid_y.cumcount()\n    tmp = tmp.loc[tmp.n<15].drop('n',axis=1)\n    # SAVE PART TO DISK (convert to pandas first uses less memory)\n    tmp.to_pandas().to_parquet(f'top_15_buy2buy_v{VER}_{PART}.pqt')\n        ","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"papermill":{"duration":113.735315,"end_time":"2022-11-10T16:14:44.498182","exception":false,"start_time":"2022-11-10T16:12:50.762867","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-01-28T13:06:51.884834Z","iopub.execute_input":"2023-01-28T13:06:51.885537Z","iopub.status.idle":"2023-01-28T13:08:02.797550Z","shell.execute_reply.started":"2023-01-28T13:06:51.885496Z","shell.execute_reply":"2023-01-28T13:08:02.796475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3) \"Clicks\" Co-visitation Matrix - Time Weighted","metadata":{"papermill":{"duration":0.04526,"end_time":"2022-11-10T16:14:44.58589","exception":false,"start_time":"2022-11-10T16:14:44.54063","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time\n# USE SMALLEST DISK_PIECES POSSIBLE WITHOUT MEMORY ERROR\nDISK_PIECES = 4\nSIZE = 1.86e6/DISK_PIECES\n\n# COMPUTE IN PARTS FOR MEMORY MANGEMENT\nfor PART in range(DISK_PIECES):\n    print()\n    print('### DISK PART',PART+1)\n    \n    # MERGE IS FASTEST PROCESSING CHUNKS WITHIN CHUNKS\n    # => OUTER CHUNKS\n    for j in range(6):\n        a = j*CHUNK\n        b = min( (j+1)*CHUNK, len(files) )\n        print(f'Processing files {a} thru {b-1} in groups of {READ_CT}...')\n        \n        # => INNER CHUNKS\n        for k in range(a,b,READ_CT):\n            # READ FILE\n            df = [read_file(files[k])]\n            for i in range(1,READ_CT): \n                if k+i<b: df.append( read_file(files[k+i]) )\n            df = cudf.concat(df,ignore_index=True,axis=0)\n            df = df.sort_values(['session','ts'],ascending=[True,False])\n            # USE TAIL OF SESSION\n            df = df.reset_index(drop=True)\n            df['n'] = df.groupby('session').cumcount()\n            df = df.loc[df.n<20].drop('n',axis=1)\n            # CREATE PAIRS\n            df = df.merge(df,on='session')\n            df = df.loc[ ((df.ts_x - df.ts_y).abs()< 24 * 60 * 60) & (df.aid_x != df.aid_y) ]\n            # MEMORY MANAGEMENT COMPUTE IN PARTS\n            df = df.loc[(df.aid_x >= PART*SIZE)&(df.aid_x < (PART+1)*SIZE)]\n            # ASSIGN WEIGHTS\n            df = df[['session', 'aid_x', 'aid_y','ts_x']].drop_duplicates(['session', 'aid_x', 'aid_y'])\n            df['wgt'] = 1 + 3*(df.ts_x - 1659304800)/(1662328791-1659304800)\n            df = df[['aid_x','aid_y','wgt']]\n            df.wgt = df.wgt.astype('float32')\n            df = df.groupby(['aid_x','aid_y']).wgt.sum()\n            # COMBINE INNER CHUNKS\n            if k==a: tmp2 = df\n            else: tmp2 = tmp2.add(df, fill_value=0)\n            print(k,', ',end='')\n        print()\n        # COMBINE OUTER CHUNKS\n        if a==0: tmp = tmp2\n        else: tmp = tmp.add(tmp2, fill_value=0)\n        del tmp2, df\n        gc.collect()\n    # CONVERT MATRIX TO DICTIONARY\n    tmp = tmp.reset_index()\n    tmp = tmp.sort_values(['aid_x','wgt'],ascending=[True,False])\n    # SAVE TOP 40\n    tmp = tmp.reset_index(drop=True)\n    tmp['n'] = tmp.groupby('aid_x').aid_y.cumcount()\n    tmp = tmp.loc[tmp.n<20].drop('n',axis=1)\n    # SAVE PART TO DISK (convert to pandas first uses less memory)\n    tmp.to_pandas().to_parquet(f'top_20_clicks_v{VER}_{PART}.pqt')","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"papermill":{"duration":null,"end_time":null,"exception":false,"start_time":"2022-11-10T16:14:44.629032","status":"running"},"tags":[],"execution":{"iopub.status.busy":"2023-01-28T13:08:14.865465Z","iopub.execute_input":"2023-01-28T13:08:14.865860Z","iopub.status.idle":"2023-01-28T13:09:00.173387Z","shell.execute_reply.started":"2023-01-28T13:08:14.865826Z","shell.execute_reply":"2023-01-28T13:09:00.172370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FREE MEMORY\ndel data_cache, tmp\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-26T05:13:54.066329Z","iopub.execute_input":"2023-01-26T05:13:54.067253Z","iopub.status.idle":"2023-01-26T05:13:54.224288Z","shell.execute_reply.started":"2023-01-26T05:13:54.067203Z","shell.execute_reply":"2023-01-26T05:13:54.222258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 2 - ReRank (choose 20) using handcrafted rules\nFor description of the handcrafted rules, read this notebook's intro.","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]}},{"cell_type":"code","source":"%%time\ndef load_test():    \n    dfs = []\n    for e, chunk_file in enumerate(glob.glob('../input/otto-chunk-data-inparquet-format/test_parquet/*')):\n        chunk = pd.read_parquet(chunk_file)\n        chunk.ts = (chunk.ts/1000).astype('int32')\n        chunk['type'] = chunk['type'].map(type_labels).astype('int8')\n        dfs.append(chunk)\n    return pd.concat(dfs).reset_index(drop=True) #.astype({\"ts\": \"datetime64[ms]\"})\n\ntest_df = load_test()\nprint('Test data has shape',test_df.shape)\ntest_df.head()","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-01-26T05:14:15.780642Z","iopub.execute_input":"2023-01-26T05:14:15.781628Z","iopub.status.idle":"2023-01-26T05:14:17.052780Z","shell.execute_reply.started":"2023-01-26T05:14:15.781591Z","shell.execute_reply":"2023-01-26T05:14:17.051772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndef pqt_to_dict(df):\n    return df.groupby('aid_x').aid_y.apply(list).to_dict()\n# LOAD THREE CO-VISITATION MATRICES\ntop_20_clicks = pqt_to_dict( pd.read_parquet(f'top_20_clicks_v{VER}_0.pqt') )\nfor k in range(1,DISK_PIECES): \n    top_20_clicks.update( pqt_to_dict( pd.read_parquet(f'top_20_clicks_v{VER}_{k}.pqt') ) )\n\ntop_20_buys = pqt_to_dict( pd.read_parquet(f'top_15_carts_orders_v{VER}_0.pqt') )\nfor k in range(1,DISK_PIECES): \n    top_20_buys.update( pqt_to_dict( pd.read_parquet(f'top_15_carts_orders_v{VER}_{k}.pqt') ) )\n\ntop_20_buy2buy = pqt_to_dict( pd.read_parquet(f'top_15_buy2buy_v{VER}_0.pqt') )\n\n# TOP CLICKS AND ORDERS IN TEST\ntop_clicks = test_df.loc[test_df['type']=='clicks','aid'].value_counts().index.values[:20]\ntop_carts = test_df.loc[test_df['type']=='carts','aid'].value_counts().index.values[:20]\ntop_orders = test_df.loc[test_df['type']=='orders','aid'].value_counts().index.values[:20]\n\n#top_40_clicks = test_df.loc[test_df['type']=='clicks','aid'].value_counts().index.values[:40]\nprint('Here are size of our 3 co-visitation matrices:')\nprint( len( top_20_clicks ), len( top_20_buy2buy ), len( top_20_buys ) )","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-01-26T05:14:41.230362Z","iopub.execute_input":"2023-01-26T05:14:41.230720Z","iopub.status.idle":"2023-01-26T05:16:20.798636Z","shell.execute_reply.started":"2023-01-26T05:14:41.230689Z","shell.execute_reply":"2023-01-26T05:16:20.797730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(top_clicks)","metadata":{"execution":{"iopub.status.busy":"2023-01-26T05:16:42.540618Z","iopub.execute_input":"2023-01-26T05:16:42.541214Z","iopub.status.idle":"2023-01-26T05:16:42.547469Z","shell.execute_reply.started":"2023-01-26T05:16:42.541177Z","shell.execute_reply":"2023-01-26T05:16:42.546434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(top_20_clicks)","metadata":{"execution":{"iopub.status.busy":"2023-01-26T05:17:07.727537Z","iopub.execute_input":"2023-01-26T05:17:07.727928Z","iopub.status.idle":"2023-01-26T05:17:07.734605Z","shell.execute_reply.started":"2023-01-26T05:17:07.727895Z","shell.execute_reply":"2023-01-26T05:17:07.733359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(top_20_clicks[0])","metadata":{"execution":{"iopub.status.busy":"2023-01-26T05:17:22.376208Z","iopub.execute_input":"2023-01-26T05:17:22.376653Z","iopub.status.idle":"2023-01-26T05:17:22.387854Z","shell.execute_reply.started":"2023-01-26T05:17:22.376614Z","shell.execute_reply":"2023-01-26T05:17:22.386841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(top_20_clicks[10000])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(top_20_clicks[532042])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(top_20_clicks)","metadata":{"execution":{"iopub.status.busy":"2023-01-26T05:17:43.356373Z","iopub.execute_input":"2023-01-26T05:17:43.356824Z","iopub.status.idle":"2023-01-26T05:17:43.364969Z","shell.execute_reply.started":"2023-01-26T05:17:43.356783Z","shell.execute_reply":"2023-01-26T05:17:43.363859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#type_weight_multipliers = {'clicks': 1, 'carts': 6, 'orders': 3}\ntype_weight_multipliers = {0: 1, 1:9, 2: 3}\n\n# def suggest_clicks(df):\n#     # USER HISTORY AIDS AND TYPES\n#     aids=df.aid.tolist()\n#     types = df.type.tolist()\n#     unique_aids = list(dict.fromkeys(aids[::-1] ))\n#     # RERANK CANDIDATES USING WEIGHTS\n#     if len(unique_aids)>=20:\n#         weights=np.logspace(0.1,1,len(aids),base=2, endpoint=True)-1\n#         aids_temp = Counter() \n#         # RERANK BASED ON REPEAT ITEMS AND TYPE OF ITEMS\n#         for aid,w,t in zip(aids,weights,types): \n#             aids_temp[aid] += w * type_weight_multipliers[t]\n#         sorted_aids = [k for k,v in aids_temp.most_common(20)]\n#         return sorted_aids\n#     # USE \"CLICKS\" CO-VISITATION MATRIX\n#     aids2 = list(itertools.chain(*[top_20_clicks[aid] for aid in unique_aids if aid in top_20_clicks]))\n#     # RERANK CANDIDATES\n#     top_aids2 = [aid2 for aid2, cnt in Counter(aids2).most_common(40) if aid2 not in unique_aids]    \n#     result = unique_aids + top_aids2[:40 - len(unique_aids)]\n#     # USE TOP20 TEST CLICKS\n#     return result + list(top_clicks)[:40-len(result)]\n\ndef suggest_40_clicks(df):\n    # USER HISTORY AIDS AND TYPES\n    aids=df.aid.tolist()\n    types = df.type.tolist()\n    unique_aids = list(dict.fromkeys(aids[::-1] ))\n    # RERANK CANDIDATES USING WEIGHTS\n    if len(unique_aids)>=20:\n        weights=np.logspace(0.1,1,len(aids),base=2, endpoint=True)-1\n        aids_temp = Counter() \n        # RERANK BASED ON REPEAT ITEMS AND TYPE OF ITEMS\n        for aid,w,t in zip(aids,weights,types): \n            aids_temp[aid] += w * type_weight_multipliers[t]\n        sorted_aids = [k for k,v in aids_temp.most_common(20)]\n        return sorted_aids\n    # USE \"CLICKS\" CO-VISITATION MATRIX\n    aids2 = list(itertools.chain(*[top_20_clicks[aid] for aid in unique_aids if aid in top_20_clicks]))\n    # RERANK CANDIDATES\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids2).most_common(20) if aid2 not in unique_aids]    \n    result = unique_aids + top_aids2[:20 - len(unique_aids)]\n    # USE TOP20 TEST CLICKS\n    result=result + list(top_clicks)[:20-len(result)]\n    \n    return result\n\ndef suggest_carts(df):\n    # User history aids and types\n    aids = df.aid.tolist()\n    types = df.type.tolist()\n\n    # UNIQUE AIDS AND UNIQUE BUYS\n    unique_aids = list(dict.fromkeys(aids[::-1]))\n    df = df.loc[(df['type'] == 0) | (df['type'] == 1)]\n    unique_buys = list(dict.fromkeys(df.aid.tolist()[::-1]))\n\n    # Rerank candidates using weights\n    if len(unique_aids) >= 20:\n        weights = np.logspace(0.5, 1, len(aids), base=2, endpoint=True) - 1\n        aids_temp = Counter()\n\n        # Rerank based on repeat items and types of items\n        for aid, w, t in zip(aids, weights, types):\n            aids_temp[aid] += w * type_weight_multipliers[t]\n\n        # Rerank candidates using\"top_20_carts\" co-visitation matrix\n        aids2 = list(itertools.chain(*[top_20_buys[aid] for aid in unique_buys if aid in top_20_buys]))\n        for aid in aids2: aids_temp[aid] += 0.1\n        sorted_aids = [k for k, v in aids_temp.most_common(20)]\n        return sorted_aids\n\n    # Use \"cart order\" and \"clicks\" co-visitation matrices\n    aids1 = list(itertools.chain(*[top_20_clicks[aid] for aid in unique_aids if aid in top_20_clicks]))\n    aids2 = list(itertools.chain(*[top_20_buys[aid] * 2 for aid in unique_aids if aid in top_20_buys]))\n\n    # RERANK CANDIDATES\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids1 + aids2).most_common(20) if aid2 not in unique_aids]\n    result = unique_aids + top_aids2[:20 - len(unique_aids)]\n\n    return result + list(top_carts)[:20 - len(result)]\n\ndef suggest_buys(df):\n    # USER HISTORY AIDS AND TYPES\n    aids=df.aid.tolist()\n    types = df.type.tolist()\n    # UNIQUE AIDS AND UNIQUE BUYS\n    unique_aids = list(dict.fromkeys(aids[::-1] ))\n    df.loc[(df['type']==1)|(df['type']==2)]  #df =df.loc[(df['type']==2)]  # \n    unique_buys = list(dict.fromkeys( df.aid.tolist()[::-1] ))\n    # RERANK CANDIDATES USING WEIGHTS\n    if len(unique_aids)>=20:\n        weights=np.logspace(0.5,1,len(aids),base=2, endpoint=True)-1\n        aids_temp = Counter() \n        # RERANK BASED ON REPEAT ITEMS AND TYPE OF ITEMS\n        for aid,w,t in zip(aids,weights,types): \n            aids_temp[aid] += w * type_weight_multipliers[t]\n        # RERANK CANDIDATES USING \"BUY2BUY\" CO-VISITATION MATRIX\n        aids3 = list(itertools.chain(*[top_20_buy2buy[aid] for aid in unique_buys if aid in top_20_buy2buy]))\n        for aid in aids3: aids_temp[aid] += 0.1\n        sorted_aids = [k for k,v in aids_temp.most_common(20)]\n        return sorted_aids\n    # USE \"CART ORDER\" CO-VISITATION MATRIX\n    aids2 = list(itertools.chain(*[top_20_buys[aid] for aid in unique_aids if aid in top_20_buys]))\n    # USE \"BUY2BUY\" CO-VISITATION MATRIX\n    aids3 = list(itertools.chain(*[top_20_buy2buy[aid] for aid in unique_buys if aid in top_20_buy2buy]))\n    # RERANK CANDIDATES\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids2+aids3).most_common(20) if aid2 not in unique_aids] \n    result = unique_aids + top_aids2[:20 - len(unique_aids)]\n    # USE TOP20 TEST ORDERS\n    return result + list(top_orders)[:20-len(result)]","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-01-26T05:23:49.834702Z","iopub.execute_input":"2023-01-26T05:23:49.835100Z","iopub.status.idle":"2023-01-26T05:23:49.862026Z","shell.execute_reply.started":"2023-01-26T05:23:49.835068Z","shell.execute_reply":"2023-01-26T05:23:49.860798Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Submission CSV\nInferring test data with Pandas groupby is slow. We need to accelerate the following code.","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[]}},{"cell_type":"code","source":"%%time\npred_df_clicks = test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_40_clicks(x)\n)\n\npred_df_carts = test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_carts(x)\n)\n\npred_df_buys = test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_buys(x)\n)","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-01-25T09:24:47.362191Z","iopub.execute_input":"2023-01-25T09:24:47.363061Z","iopub.status.idle":"2023-01-25T10:14:02.871087Z","shell.execute_reply.started":"2023-01-25T09:24:47.363025Z","shell.execute_reply":"2023-01-25T10:14:02.870041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(pred_df_clicks, columns=[\"labels\"]).reset_index().to_parquet(f'pred_df_clicks.pqt')\npd.DataFrame(pred_df_carts, columns=[\"labels\"]).reset_index().to_parquet(f'pred_df_carts.pqt')\npd.DataFrame(pred_df_buys, columns=[\"labels\"]).reset_index().to_parquet(f'pred_df_buys.pqt')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('pred_df_clicks:', pd.DataFrame(pred_df_clicks).shape)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('pred_df_carts:', pd.DataFrame(pred_df_carts).shape)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('pred_df_buys:', pd.DataFrame(pred_df_buys).shape)","metadata":{"execution":{"iopub.status.busy":"2023-01-26T05:59:46.122051Z","iopub.execute_input":"2023-01-26T05:59:46.122657Z","iopub.status.idle":"2023-01-26T05:59:46.128646Z","shell.execute_reply.started":"2023-01-26T05:59:46.122621Z","shell.execute_reply":"2023-01-26T05:59:46.127568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clicks_pred_df = pd.DataFrame(pred_df_clicks.add_suffix(\"_clicks\"), columns=[\"labels\"]).reset_index()\norders_pred_df = pd.DataFrame(pred_df_buys.add_suffix(\"_orders\"), columns=[\"labels\"]).reset_index()\ncarts_pred_df = pd.DataFrame(pred_df_carts.add_suffix(\"_carts\"), columns=[\"labels\"]).reset_index()","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-01-16T09:26:06.272261Z","iopub.execute_input":"2023-01-16T09:26:06.273282Z","iopub.status.idle":"2023-01-16T09:26:09.191316Z","shell.execute_reply.started":"2023-01-16T09:26:06.273245Z","shell.execute_reply":"2023-01-16T09:26:09.190324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clicks_pred_df.to_parquet(f'clicks_pred_df.pqt')\norders_pred_df.to_parquet(f'orders_pred_df.pqt')\ncarts_pred_df.to_parquet(f'carts_pred_df.pqt')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.concat([clicks_pred_df, orders_pred_df, carts_pred_df])\npred_df.columns = [\"session_type\", \"labels\"]\npred_df[\"labels\"] = pred_df.labels.apply(lambda x: \" \".join(map(str,x)))\npred_df.to_csv(\"submission.csv\", index=False)\npred_df.head()","metadata":{"papermill":{"duration":null,"end_time":null,"exception":null,"start_time":null,"status":"pending"},"tags":[],"execution":{"iopub.status.busy":"2023-01-16T09:26:38.077570Z","iopub.execute_input":"2023-01-16T09:26:38.077982Z","iopub.status.idle":"2023-01-16T09:27:17.319942Z","shell.execute_reply.started":"2023-01-16T09:26:38.077947Z","shell.execute_reply":"2023-01-16T09:27:17.318900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}