{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np\nimport os, sys, pickle, glob, gc, math\nimport cudf\nprint('We will use RAPIDS version',cudf.__version__)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-28T16:41:09.336159Z","iopub.execute_input":"2023-01-28T16:41:09.336857Z","iopub.status.idle":"2023-01-28T16:41:12.276976Z","shell.execute_reply.started":"2023-01-28T16:41:09.336761Z","shell.execute_reply":"2023-01-28T16:41:12.276045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"V1: https://www.kaggle.com/code/hoangnguyen719/otto-candidates-v1-tail50-top40-40-50-w136-c40/notebook\n<br>\nV2's differences from V1:\n- Use Chris's Covisitation matrices","metadata":{}},{"cell_type":"code","source":"VER = 2\nTEST = False # if True, use only `TEST_FRAC` fraction of total data\nTEST_FRAC = 1e-2\nINFERENCE = False\nRS = 719 # Random state\nMAX_SIZE = 1.86e6\nPARTITIONS = 40\nCHUNKS = 4\nVAL_PTH = '../input/otto-train-and-test-data-for-local-validation/'\nINFER_PTH = '../input/otto-chunk-data-inparquet-format/'\nMATRX_DS = '../input/ottoval-tail60-top303030-weight136/'\n\n# VARIABLE\nCANDIDATES = 20 if (TEST or INFERENCE) else 40\nTAIL = 40\nCART_ORDER_TOP = 40\nBUY2BUY_TOP = 40\nCLICK_TIME_TOP = 50\n\ntype_weight = {0:1, 1:3, 2:6}\ntype_label = {'clicks':0, 'carts':1, 'orders':2}","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:41:12.281474Z","iopub.execute_input":"2023-01-28T16:41:12.284224Z","iopub.status.idle":"2023-01-28T16:41:12.293413Z","shell.execute_reply.started":"2023-01-28T16:41:12.284178Z","shell.execute_reply":"2023-01-28T16:41:12.292416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Utility functions\ndef load_ds(path):\n    return pd.read_parquet(path)\n\ndef to_cudf(df):\n    return cudf.DataFrame(df)\n\ndef load2cudf(path):\n    return to_cudf(load_ds(path))\n\ndef save_to_parquet(df, pth):\n    df.to_pandas().to_parquet(pth)\n\n# Partitioning dataset into small chunks of the same size\n# For each session, its events are all in one same partition\ndef partition(df, partitions=PARTITIONS):\n    if 'chunk' in df.columns:\n        df = df.drop('chunk', axis=1)\n    # Calculate chunk\n    tmp = df.groupby(['session']).count()\n    tmp = tmp.reset_index()[['session', 'aid']]\n    tmp = tmp.sort_values(['aid'], ascending=[False]).reset_index(drop=True)\n    tmp['chunk'] = np.array(range(len(tmp))) % partitions\n    tmp['chunk'] = tmp.chunk.astype('int16')\n    tmp = tmp[['session', 'chunk']]\n    df = df.merge(tmp, on='session', how='left')\n    df = df.sort_values(['chunk', 'session'], ascending=[True, True])\n    df = df.reset_index(drop=True).reset_index()\n    df = df.astype({'chunk':'uint8'})\n    del tmp\n    gc.collect()\n    \n    partition_idx = df.groupby('chunk').agg({'index':['min','max'], 'aid':'count'}).reset_index()\n    partition_idx.columns = ['chunk', 'min', 'max', 'rows']\n    partition_idx = partition_idx.sort_values(['chunk'], ascending=[True]).reset_index(drop=True)\n    \n    partition_idx = partition_idx.drop('chunk', axis=1)\n#     df = df.drop(['index'], axis=1)\n    df = df.drop(['index', 'chunk'], axis=1)\n    return df, partition_idx","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:41:12.296799Z","iopub.execute_input":"2023-01-28T16:41:12.297424Z","iopub.status.idle":"2023-01-28T16:41:12.315291Z","shell.execute_reply.started":"2023-01-28T16:41:12.297388Z","shell.execute_reply":"2023-01-28T16:41:12.314257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"def truncate_sess(df):     \n    np.random.seed(RS)\n    sess = df.session.unique().to_array()\n    sess = np.random.choice(sess, size=int(len(sess)*TEST_FRAC), replace=False)\n    df = df.loc[df.session.isin(sess)].reset_index(drop=True)      \n    return df\n\ndef load_entire_file(file_paths):\n    print(f'Loading {len(file_paths)} files')\n    for idx, f in enumerate(file_paths):\n        tmp = load2cudf(f)\n        if idx==0: df = [tmp]\n        else: df.append(tmp)\n        tmp = (idx+1)\n        if (tmp%10)==0: print(f'{tmp}...', end='\\n' if (tmp%100)==0 else '')\n    df = cudf.concat(df, ignore_index=True)\n    return df\n\ndef process_raw_file(df):\n    df['ts'] = (df['ts'] / 1000)\n    df['type'] = df['type'].map(type_label)\n    df = df.astype({\n        'session': 'uint32'\n        , 'aid': 'uint32'\n        , 'ts': 'int32'\n        , 'type': 'int8'\n    })\n    return df\n\ndef load_process_truncate(file_paths, testing=TEST, infer=INFERENCE):\n    df = load_entire_file(file_paths)\n    print('Done loading!')\n    if testing:\n        df = truncate_sess(df)\n        print('Done truncating!')\n    if infer:\n        df = process_raw_file(df)\n    print('Done processing')\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:41:12.317579Z","iopub.execute_input":"2023-01-28T16:41:12.318026Z","iopub.status.idle":"2023-01-28T16:41:12.331673Z","shell.execute_reply.started":"2023-01-28T16:41:12.317968Z","shell.execute_reply":"2023-01-28T16:41:12.330931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Create cache\ndata_cache = {}\nif INFERENCE:\n    files_tr = glob.glob(INFER_PTH + 'train_parquet/*')\n    files_tr.sort()\n    data_cache['train_0'] = {\n        'path': [f for idx, f in enumerate(files_tr) if (idx%2)==0]\n        , 'ptn_n': PARTITIONS\n    }\n    data_cache['train_1'] = {\n        'path': [f for idx, f in enumerate(files_tr) if (idx%2)==1]\n        , 'ptn_n': PARTITIONS\n    }\n    data_cache['test'] = {\n        'path': glob.glob(INFER_PTH + 'test_parquet/*')\n        , 'ptn_n': PARTITIONS//5\n    }\nelse:\n    data_cache['train'] = {\n        'path': [VAL_PTH + 'train.parquet']\n        , 'ptn_n': PARTITIONS\n    }\n    data_cache['test'] = {\n        'path': [VAL_PTH + 'test.parquet']\n        , 'ptn_n': PARTITIONS//5\n    }\n\n# Load and truncate data if needed\nfor ds in data_cache:\n    print('='*30 + f'\\n{ds}')\n    data_cache[ds]['dataset'] = load_process_truncate(data_cache[ds]['path'])\n    data_cache[ds]['dataset'], data_cache[ds]['ptn'] = partition(\n        data_cache[ds]['dataset'], data_cache[ds]['ptn_n']\n    )\n    print(f'Len({ds})={len(data_cache[ds][\"dataset\"])}')","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:41:12.332999Z","iopub.execute_input":"2023-01-28T16:41:12.334038Z","iopub.status.idle":"2023-01-28T16:41:36.6085Z","shell.execute_reply.started":"2023-01-28T16:41:12.333971Z","shell.execute_reply":"2023-01-28T16:41:36.607447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for ds in data_cache:\n    print(data_cache[ds]['dataset'].dtypes)\n    break","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:41:36.610158Z","iopub.execute_input":"2023-01-28T16:41:36.610859Z","iopub.status.idle":"2023-01-28T16:41:36.618579Z","shell.execute_reply.started":"2023-01-28T16:41:36.610818Z","shell.execute_reply":"2023-01-28T16:41:36.617291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Co-visitation Matrices\nCovisitation matrices created externally","metadata":{}},{"cell_type":"code","source":"# # VARIABLES\n# name_list = list(data_cache.keys())\n# ds_list = [data_cache[ds]['dataset'] for ds in name_list]\n# ptn_list = [data_cache[ds]['ptn'] for ds in name_list]\n\n# # UTILITY FUNCTIONS\n# def tail_ds(df, session_tail=None):\n#     if session_tail:\n#         df = df.sort_values(['session', 'ts'], ascending=[True,False])\n#         df = df.reset_index(drop=True)\n#         df['n'] = df.groupby('session').cumcount() \n#         df = df.loc[df.n<session_tail].drop('n',axis=1)\n#     return df\n\n# def select_top_weight(df, top=15):\n#     df = df.reset_index()\n#     df = df.sort_values(['aid_x','wgt'],ascending=[True,False]).reset_index(drop=True)\n#     df['n'] = df.groupby('aid_x').aid_y.cumcount()\n#     df = df.loc[df.n < top].drop('n',axis=1).reset_index(drop=True)\n#     return df\n\n# def process_piece(\n#     dss, names, ptns\n#     , filter_fn, weight_fn, select_top_fn=select_top_weight\n#     , top_candidates=15, chunks=3, part=0, size=MAX_SIZE, tail=TAIL\n# ):\n#     if ptns is None:\n#         ptns = [cudf.DataFrame(pd.DataFrame({\n#             'min': [0], 'max':[len(df)-1], 'rows':[len(df)]\n#         })) for df in dss]\n#         chunks = 1\n#     for ds, ptn, name in zip(dss, ptns, names):\n#         ptn_len = len(ptn)\n#         ptn_chunks = np.array_split(range(ptn_len), chunks)\n#         print(f'Processing {name}, {ptn_len} partitions in {chunks} chunks...')\n#         for idx, chunk in enumerate(ptn_chunks):\n#             for p in chunk:\n#                 df = ds.loc[ptn.loc[p, 'min']:ptn.loc[p, 'max']]\n#                 df = filter_fn(df, tail)\n#                 df = weight_fn(df, part*size, (part+1)*size)\n#                 # COMBINE INNER CHUNK\n#                 if p==chunk[0]: tmp2 = df\n#                 else: tmp2 = tmp2.add(df, fill_value=0)\n#                 if p==chunk[-1]: print(p+1, ' , ', end='')\n#             # COMBINE OUTER CHUNK\n#             if (idx==0) & (name==names[0]): tmp = tmp2\n#             else: tmp = tmp.add(tmp2, fill_value=0)\n#             del df, tmp2\n#             gc.collect()\n#         print()\n#     tmp = select_top_fn(tmp, top_candidates)\n#     return tmp","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:41:36.624896Z","iopub.execute_input":"2023-01-28T16:41:36.627121Z","iopub.status.idle":"2023-01-28T16:41:36.635857Z","shell.execute_reply.started":"2023-01-28T16:41:36.627082Z","shell.execute_reply":"2023-01-28T16:41:36.634838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1) Carts & Orders Matrix","metadata":{}},{"cell_type":"code","source":"# %%time\n# DISK_PIECES = 4\n# SIZE = MAX_SIZE/DISK_PIECES # 1.86e6 is the maximum aid\n\n# def cart_order_filter(df, tail=None):\n#     df = tail_ds(df, tail)\n#     return df\n\n# def cart_order_weight(df, aid_min, aid_max):\n#     # CREATE PAIRS\n#     df = df.merge(df,on='session')\n#     df = df.loc[ ((df.ts_x - df.ts_y).abs()< 24 * 60 * 60) & (df.aid_x != df.aid_y) ]\n#     # MEMORY MANAGEMENT COMPUTE IN PARTS\n#     df = df.loc[(df.aid_x >= aid_min) & (df.aid_x < aid_max)]\n#     # ASSIGN WEIGHTS\n#     df = df[['session', 'aid_x', 'aid_y','type_y']].drop_duplicates(['session', 'aid_x', 'aid_y', 'type_y'])\n#     df['wgt'] = df.type_y.map(type_weight)\n#     df = df[['aid_x','aid_y','wgt']]\n#     df.wgt = df.wgt.astype('float32')\n#     df = df.groupby(['aid_x','aid_y']).wgt.sum()\n#     return df\n\n# for PART in range(DISK_PIECES):\n#     print('\\n### DISK PART',PART+1)\n#     tmp = process_piece(\n#         ds_list, name_list, ptn_list\n#         , cart_order_filter, cart_order_weight, select_top_weight\n#         , top_candidates=CART_ORDER_TOP, chunks=CHUNKS, part=PART, size=SIZE, tail=30\n#     )\n#     save_to_parquet(tmp, f'top_{CART_ORDER_TOP}_carts_orders_v{VER}_{PART}.pqt')\n#     del tmp\n#     gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:41:36.637206Z","iopub.execute_input":"2023-01-28T16:41:36.63843Z","iopub.status.idle":"2023-01-28T16:41:36.653648Z","shell.execute_reply.started":"2023-01-28T16:41:36.638393Z","shell.execute_reply":"2023-01-28T16:41:36.652606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2) \"Buy2Buy\" Co-visitation Matrix","metadata":{}},{"cell_type":"code","source":"# %%time\n# DISK_PIECES = 1\n# SIZE = MAX_SIZE/DISK_PIECES # 1.86e6 is the maximum aid\n\n# def buy2buy_filter(df, tail=None):\n#     df = df.loc[df['type'].isin([1,2])]\n#     df = tail_ds(df, tail)\n#     return df\n\n# def buy2buy_weight(df, aid_min, aid_max):\n#     # CREATE PAIRS\n#     df = df.merge(df,on='session')\n#     df = df.loc[ ((df.ts_x - df.ts_y).abs()< 14 * 24 * 60 * 60) & (df.aid_x != df.aid_y) ] # 14 days\n#     # MEMORY MANAGEMENT COMPUTE IN PARTS\n#     df = df.loc[(df.aid_x >= aid_min) & (df.aid_x < aid_max)]\n#     # ASSIGN WEIGHTS\n#     df = df[['session', 'aid_x', 'aid_y','type_y']]\n#     df = df.drop_duplicates(['session', 'aid_x', 'aid_y', 'type_y'])\n#     df['wgt'] = 1\n#     df = df[['aid_x','aid_y','wgt']]\n#     df.wgt = df.wgt.astype('float32')\n#     df = df.groupby(['aid_x','aid_y']).wgt.sum()\n#     return df \n\n# for PART in range(DISK_PIECES):\n#     print('\\n### DISK PART',PART+1)\n#     tmp = process_piece(\n#         ds_list, name_list, ptn_list\n#         , buy2buy_filter, buy2buy_weight, select_top_weight\n#         , top_candidates=BUY2BUY_TOP, chunks=CHUNKS, part=PART, size=SIZE, tail=30\n#     )\n#     save_to_parquet(tmp, f'top_{BUY2BUY_TOP}_buy2buy_v{VER}_{PART}.pqt')\n#     del tmp\n#     gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:41:36.655997Z","iopub.execute_input":"2023-01-28T16:41:36.656401Z","iopub.status.idle":"2023-01-28T16:41:36.668203Z","shell.execute_reply.started":"2023-01-28T16:41:36.656367Z","shell.execute_reply":"2023-01-28T16:41:36.667146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3) \"Clicks\" Co-visitation Matrix - Time Weighted","metadata":{}},{"cell_type":"code","source":"# %%time\n# DISK_PIECES = 4\n# SIZE = MAX_SIZE/DISK_PIECES # 1.86e6 is the maximum aid\n\n# def clicktime_filter(df, tail=None):\n#     df = tail_ds(df, tail)\n#     return df\n\n# def clicktime_weight(df, aid_min, aid_max):\n#     # CREATE PAIRS\n#     df = df.merge(df,on='session')\n#     df = df.loc[ ((df.ts_x - df.ts_y).abs()< 24 * 60 * 60) & (df.aid_x != df.aid_y) ]\n#     # MEMORY MANAGEMENT COMPUTE IN PARTS\n#     df = df.loc[(df.aid_x >= aid_min) & (df.aid_x < aid_max)]\n#     # ASSIGN WEIGHTS\n#     df = df[['session', 'aid_x', 'aid_y','ts_x']]\n#     df = df.drop_duplicates(['session', 'aid_x', 'aid_y'])\n#     df['wgt'] = 1 + 3*(df.ts_x - 1659304800)/(1662328791-1659304800)\n#     df = df[['aid_x','aid_y','wgt']]\n#     df.wgt = df.wgt.astype('float32')\n#     df = df.groupby(['aid_x','aid_y']).wgt.sum()\n#     return df \n    \n\n# for PART in range(DISK_PIECES):\n#     print()\n#     print('### DISK PART',PART+1)\n#     tmp = process_piece(\n#         ds_list, name_list, ptn_list\n#         , clicktime_filter, clicktime_weight, select_top_weight\n#         , top_candidates=CLICK_TIME_TOP, chunks=CHUNKS, part=PART, size=SIZE, tail=30\n#     )\n#     save_to_parquet(tmp, f'top_{CLICK_TIME_TOP}_clicks_v{VER}_{PART}.pqt')\n#     del tmp\n#     gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:41:36.670292Z","iopub.execute_input":"2023-01-28T16:41:36.670797Z","iopub.status.idle":"2023-01-28T16:41:36.683339Z","shell.execute_reply.started":"2023-01-28T16:41:36.670761Z","shell.execute_reply":"2023-01-28T16:41:36.682227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CANDIDATE SELECTION","metadata":{}},{"cell_type":"markdown","source":"## Load Data","metadata":{}},{"cell_type":"code","source":"# LOAD AND FILTER FUNCTIONS\n\ndef add_asc_pos_col(df, colname):\n    df = df.sort_values(['session', 'ts']).reset_index(drop=True)\n    df[colname] = df.groupby('session').cumcount()\n    return df\n\ndef add_row_count_col(df, colname):\n    tmp = df.groupby(['session']).count().reset_index()[['session', 'aid']]\n    tmp.columns = ['session', colname]\n    df = df.merge(tmp, on='session', how='left')\n    return df\n\ndef position_event(df):\n    # Calculate rows in each \n    df = add_row_count_col(df, 'rows')\n    \n    # Calculate position\n    df = add_asc_pos_col(df, 'n')\n    \n    # Calculate rows position - cart and order\n    tmp = df.loc[df.type.isin([1,2])][['session', 'ts', 'aid']]\n    tmp = add_row_count_col(tmp, 'rows_12')\n    tmp = add_asc_pos_col(tmp, 'n_12')\n    df = df.merge(tmp, on=['session', 'ts', 'aid'], how='left')\n    df['rows_12'] = df['rows_12'].fillna(0)\n    df['n_12'] = df['n_12'].fillna(-1) # -1 SIGNIFYING NULLS\n    \n    del tmp\n    df = df.astype({\n        'n':'int16', 'rows':'int16', 'n_12':'int16', 'rows_12':'int16'\n    })\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:41:36.684902Z","iopub.execute_input":"2023-01-28T16:41:36.685951Z","iopub.status.idle":"2023-01-28T16:41:36.704273Z","shell.execute_reply.started":"2023-01-28T16:41:36.685916Z","shell.execute_reply":"2023-01-28T16:41:36.703193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntype_map = {\n    'clicks': 0\n    , 'carts': 1\n    , 'orders': 2\n}\n\npred_ds = position_event(data_cache['test']['dataset'])\npred_ds, pred_ptn = partition(pred_ds, partitions=data_cache['test']['ptn_n'])\nprint(f'Predicting DS size: {len(pred_ds)}')\n\nif not INFERENCE:\n    pred_lbls = load2cudf(VAL_PTH + 'test_labels.parquet')\n    if TEST:\n        pred_sessions = pred_ds.session.unique()\n        print(f'Session count: {len(pred_sessions)}')\n        pred_lbls = pred_lbls.loc[pred_lbls.session.isin(pred_sessions)]\n        del pred_sessions\n        gc.collect()\n    pred_lbls['type'] = pred_lbls.type.map(type_map)\n    print(f'Label size: {len(pred_lbls)}')","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:41:36.706391Z","iopub.execute_input":"2023-01-28T16:41:36.709209Z","iopub.status.idle":"2023-01-28T16:41:40.546555Z","shell.execute_reply.started":"2023-01-28T16:41:36.709172Z","shell.execute_reply":"2023-01-28T16:41:40.545329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get Covisitation Matrices","metadata":{}},{"cell_type":"code","source":"DISK_PIECES = 4\ntmp_path = '/kaggle/input/otto-tr--matrixv2-tail40-top404050-w136/'\n# clk_cov_pth = [f'top_{CLICK_TIME_TOP}_clicks_v{VER}_{k}.pqt' for k in range(DISK_PIECES)]\n# buy_cov_pth = [f'top_{CART_ORDER_TOP}_carts_orders_v{VER}_{k}.pqt' for k in range(DISK_PIECES)]\n# b2b_cov_pth = [f'top_{BUY2BUY_TOP}_buy2buy_v{VER}_0.pqt']\nclk_cov_pth = [tmp_path + f'top_{CLICK_TIME_TOP}_clicks_v{VER}_{k}.pqt' for k in range(DISK_PIECES)]\nbuy_cov_pth = [tmp_path + f'top_{CART_ORDER_TOP}_carts_orders_v{VER}_{k}.pqt' for k in range(DISK_PIECES)]\nb2b_cov_pth = [tmp_path + f'top_{BUY2BUY_TOP}_buy2buy_v{VER}_0.pqt']\nif not INFERENCE:\n    clk_cov_pth = glob.glob(MATRX_DS + '/top_*_clicks_*')\n    buy_cov_pth = glob.glob(MATRX_DS + '/top_*_carts_orders_*')\n    b2b_cov_pth = glob.glob(MATRX_DS + '/top_*_buy2buy_*')","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:43:17.39212Z","iopub.execute_input":"2023-01-28T16:43:17.392705Z","iopub.status.idle":"2023-01-28T16:43:17.413175Z","shell.execute_reply.started":"2023-01-28T16:43:17.392634Z","shell.execute_reply":"2023-01-28T16:43:17.412271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Click\ntop_clk_cov = [load2cudf(pth) for pth in clk_cov_pth]\ntop_clk_cov = cudf.concat(top_clk_cov, ignore_index=True).reset_index(drop=True)\ntop_clk_cov.columns = ['aid', 'top_clk_cov', 'wgt']\n\n# Cart Order\ntop_buy_cov = [load2cudf(pth) for pth in buy_cov_pth]\ntop_buy_cov = cudf.concat(top_buy_cov, ignore_index=True).reset_index(drop=True)\ntop_buy_cov.columns = ['aid', 'top_buy_cov', 'wgt']\n\n# Buy2buy\ntop_b2b_cov = [load2cudf(pth) for pth in b2b_cov_pth]\ntop_b2b_cov = cudf.concat(top_b2b_cov, ignore_index=True).reset_index(drop=True)\ntop_b2b_cov.columns = ['aid', 'top_b2b_cov', 'wgt']\n\n# TOP CLICKS AND ORDERS IN TEST\ntop_clk_buy_wk = pred_ds.loc[pred_ds['type']==0].aid.value_counts().index.values[:CANDIDATES]\ntop_clk_buy_wk = cudf.DataFrame({'aid': top_clk_buy_wk})\ntop_clk_buy_wk['top_clk_wk'] = 1\n\ntmp = pred_ds.loc[pred_ds['type']==2].aid.value_counts().index.values[:CANDIDATES]\ntmp = cudf.DataFrame({'aid':tmp})\ntmp['top_buy_wk'] = 1\ntop_clk_buy_wk = top_clk_buy_wk.merge(tmp, on=['aid'], how='outer').fillna(0)\ntop_clk_buy_wk = top_clk_buy_wk.astype({\n    'top_clk_wk': 'uint8'\n    , 'top_buy_wk': 'uint8'\n})\n\nprint('Size of our 3 co-visitation matrices:')\nprint(f'---Top clicks: {len(top_clk_cov)}')\nprint(f'---Top buys: {len(top_buy_cov)}')\nprint(f'---Top buy2buy: {len(top_b2b_cov)}')\nprint(f'Size of top click/buy in week: {len(top_clk_buy_wk)}')","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:43:22.409411Z","iopub.execute_input":"2023-01-28T16:43:22.409783Z","iopub.status.idle":"2023-01-28T16:43:32.469259Z","shell.execute_reply.started":"2023-01-28T16:43:22.409749Z","shell.execute_reply":"2023-01-28T16:43:32.46819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"code","source":"day_quarters = ['12am_5am', '6am_11am', '12pm_5pm', '6pm_11pm']\ndows = ['mon', 'tue', 'wed', 'thu', 'fri', 'sat', 'sun']\nFEATURES_DF = pd.DataFrame(columns=['segment', 'group', 'name', 'path'])\nLABELS = {}\n\ndef set_multiplier(pos, rows, st, en, base=2):\n    row_offset = (pos / (rows - 1)).fillna(0)\n    null_offset = (pos > -1).astype('uint16')\n    return null_offset * (base**(st + (en-st) * row_offset) - 1)\n\ndef set_weight(df):\n    # ASSIGNING CHRONOLOGICALLY SCALED WEIGHT FOR EACH AID\n    df['type_wgt'] = df['type'].map(type_weight)\n    \n    # Click weight\n    tmp = set_multiplier(df['n'], df['rows'], 0.1, 1)\n    df['clk_wgt'] = df['type_wgt'] * tmp\n    \n    # Order/cart weight\n#     tmp = set_multiplier(df['n_12'], df['rows_12'], 0.5, 1)\n    tmp = set_multiplier(df['n'], df['rows'], 0.5, 1)\n    df['order_wgt'] = df['type_wgt'] * tmp\n    \n    del tmp\n    gc.collect()    \n    df = df.astype({\n        'type_wgt': 'int8'\n        , 'clk_wgt': 'float32'\n        , 'order_wgt': 'float32'\n    })\n    return df\n\ndef update_ft_df(segment, group, name, path):\n    idx = (FEATURES_DF.segment==segment) & (FEATURES_DF.group==group) & (FEATURES_DF.name==name)\n    if len(FEATURES_DF.loc[idx]) == 0:\n        FEATURES_DF.loc[len(FEATURES_DF)] = [segment, group, name, path]\n    else:\n        FEATURES_DF.loc[idx, 'path'] = path\n    \ndef save_ft(df, groupbys, segment, group, names, path):\n    if not INFERENCE:\n        for n in names:\n            update_ft_df(segment, group, n, path)\n        df.loc[:, groupbys + names].to_pandas().to_parquet(path)\n\ndef get_most_frequent(df, groupby_cols, suffix):\n    def _most_freq_col(df, col):\n        tmp = df.groupby(groupby_cols + [col]).size().reset_index()\n        tmp = tmp.sort_values(groupby_cols + [0], ascending=False).reset_index(drop=True)\n        tmp['n'] = tmp.groupby(groupby_cols).cumcount()\n        tmp = tmp.loc[tmp.n == 0, groupby_cols + [col]]\n        return tmp\n    freq = _most_freq_col(df, 'day_quarter')\n    freq = freq.merge(_most_freq_col(df, 'dow'))\n    freq = freq.astype({\n        'day_quarter': 'uint8'\n        , 'dow': 'uint8'\n    })\n    freq = freq.rename(columns={\n        'day_quarter': suffix + 'freq_quarter_day'\n        , 'dow': suffix + 'freq_dow'\n    })\n    return freq\n    \ndef get_item_ft_df(df):\n    # CREATE COLUMNS\n    # Event counts and weights\n    item_df = df.loc[:,['aid', 'clicks', 'carts', 'orders', 'clk_wgt', 'order_wgt']]\n    item_df = item_df.groupby('aid').sum().reset_index()\n    \n    # Session count\n    tmp = df[['aid', 'session']].groupby('aid').session.nunique().reset_index()\n    item_df = item_df.merge(tmp, on='aid', how='left')\n    \n    # Frequency\n    tmp = get_most_frequent(df, ['aid'], 'i_')\n    item_df = item_df.merge(tmp, on=['aid'], how='left')\n    \n    # RENAME COLUMNS to add `item` suffix\n    rename = {'session': 'i_sessions'}\n    for i in ['clicks', 'carts', 'orders', 'clk_wgt', 'order_wgt']:\n        rename[i] = 'i_' + str(i)\n    item_df = item_df.rename(columns=rename)\n    \n    # CONVERT DTYPES\n    item_df = item_df.astype({\n        'i_clicks': 'uint16'\n        , 'i_carts': 'uint16'\n        , 'i_orders': 'uint16'\n    })\n    \n    # SAVE DATA TO PARQUET\n    for group, names, path in [\n        ['event_cnt', ['i_clicks', 'i_carts', 'i_orders', 'i_sessions'], f'i_event_count_v{VER}.pqt']\n        , ['event_wgt', ['i_clk_wgt', 'i_order_wgt'], f'i_event_wgt_v{VER}.pqt']\n        , ['event_freq', ['i_freq_quarter_day', 'i_freq_dow'], f'i_freq_v{VER}.pqt']\n    ]:\n        save_ft(item_df, ['aid'], 'item', group, names, path)\n    return item_df\n\ndef get_user_ft_df(df):\n    # Event count\n    user_df = df.loc[:,['session', 'clicks', 'carts', 'orders', 'clk_wgt', 'order_wgt']]\n    user_df = user_df.groupby('session').sum().reset_index()\n    \n    # AID count and active days\n    tmp = df.loc[:, ['session', 'aid', 'doy']]\n    tmp = tmp.groupby('session')[['aid', 'doy']].nunique().reset_index()\n    user_df = user_df.merge(tmp, on='session', how='left')\n    \n    # Frequency\n    tmp = get_most_frequent(df, ['session'], 'u_')\n    user_df = user_df.merge(tmp, on='session', how='left')\n    \n    # Active periods\n    tmp = df.groupby('session').agg({'doy':['min', 'max']}).reset_index()\n    tmp.columns = ['session', 'doy_min', 'doy_max']\n    tmp['u_activ_period'] = (tmp.doy_max - tmp.doy_min + 1).astype('uint8')\n    user_df = user_df.merge(tmp[['session', 'u_activ_period']], on='session', how='left')\n    \n    # CONVERT DTYPE\n    dtypes = {\n        'aid': 'uint16'\n        , 'doy': 'uint8'\n    }\n    user_df = user_df.astype(dtypes)\n    \n    # RENAME\n    renames = {\n        'clicks': 'u_clicks'\n        , 'carts': 'u_carts'\n        , 'orders': 'u_orders'\n        , 'clk_wgt': 'u_clk_wgt'\n        , 'order_wgt': 'u_order_wgt'\n        , 'aid': 'u_aids'\n        , 'doy': 'u_active_days'\n    }\n    user_df = user_df.rename(columns=renames)\n    \n    # SAVE DATA\n    for group, names, path in [\n        ['event_cnt', ['u_aids', 'u_clicks', 'u_carts', 'u_orders'], f'u_event_count_v{VER}.pqt']\n        , ['event_wgt', ['u_clk_wgt', 'u_order_wgt'], f'u_event_wgt_v{VER}.pqt']\n        , ['event_freq', ['u_freq_quarter_day', 'u_freq_dow'], f'u_freq_v{VER}.pqt']\n        , ['activ_period', ['u_active_days', 'u_activ_period'], f'u_activ_period_v{VER}.pqt']\n    ]:\n        save_ft(user_df, ['session'], 'user', group, names, path)\n    return user_df\n\ndef get_user_item_ft(df):\n    first_ts = df.ts.min()\n    last_ts = df.ts.max()\n    config = {'agg': {}, 'rename': {}, 'dtype': {}}\n    \n    # CREATE NEW COLUMNS\n    # Frequency\n    tmp = get_most_frequent(df, ['session', 'aid'], 'ui_')\n    \n    # Event weight and counts\n    config['agg'] = {\n        'rows': 'max'\n        , 'n': 'max'\n        , 'ts': 'max'\n    }\n    \n    for i in day_quarters + dows + ['clk_wgt', 'order_wgt', 'clicks', 'carts', 'orders']:\n        config['agg'][i]='sum'\n        config['rename'][i] = 'ui_' + i\n    \n    df = df.groupby(['session', 'aid']).agg(config['agg'])\n    df = df.reset_index()\n    df = df.merge(tmp, on=['session', 'aid'], how='left')    \n    del tmp\n    gc.collect()\n    df['ui_last_n'] = df['n'] / df['rows']\n    df['ui_last_ts'] = (df['ts'] - first_ts) / (last_ts - first_ts)\n    \n    # RENAME COLUMNS to add `ui` (user-item) suffix\n    df = df.rename(columns=config['rename'])\n    \n    # CONVERT DTYPES\n    config['dtype'] = {\n        'ui_last_n': 'float32'\n        , 'ui_last_ts': 'float32'        \n    }\n    for i in day_quarters + dows + ['clicks', 'carts', 'orders']:\n        config['dtype']['ui_' + i] = 'uint16'\n    df = df.astype(config['dtype'])\n    df = df.drop(['n', 'ts'], axis=1)\n    \n    # Save df to parquet\n    for group, names, path in [\n        ['event_cnt', ['ui_clicks', 'ui_carts', 'ui_orders'], f'ui_event_count_v{VER}.pqt']\n        , ['event_wgt', ['ui_clk_wgt', 'ui_order_wgt'], f'ui_event_wgt_v{VER}.pqt']\n        , ['freq_quarter_day', ['ui_12am_5am', 'ui_6am_11am', 'ui_12pm_5pm', 'ui_6pm_11pm'], f'ui_freq_quarter_day_v{VER}.pqt']\n        , ['freq_dow', ['ui_mon', 'ui_tue', 'ui_wed', 'ui_thu', 'ui_fri', 'ui_sat', 'ui_sun'], f'ui_freq_dow_v{VER}.pqt']\n        , ['event_freq', ['ui_freq_quarter_day', 'ui_freq_dow'], f'ui_freq_v{VER}.pqt']\n        , ['ui_last_time', ['ui_last_n', 'ui_last_ts'], f'ui_last_time_v{VER}.pqt']\n    ]:\n        save_ft(df, ['session', 'aid'], 'user_item', group, names, path)\n    \n    return df\n\ndef preprocess_test(df):\n    df = set_weight(df)\n    \n    df['clicks'] = (df.type==0).astype('uint8')\n    df['carts'] = (df.type==1).astype('uint8')\n    df['orders'] = (df.type==2).astype('uint8')\n    df['ts_dt'] = cudf.to_datetime(df['ts'], unit='s')\n    df['day_quarter'] = df['ts_dt'].dt.hour // 6\n    for i, v in enumerate(day_quarters):\n        df[v] = (df['day_quarter'] == i).astype('uint16')\n    df['dow'] = df['ts_dt'].dt.dayofweek\n    for i, v in enumerate(dows):\n        df[v] = (df['dow'] == i).astype('uint16')\n    df['doy'] = df['ts_dt'].dt.dayofyear\n    \n    # Get user-item interaction features\n    ui_df = get_user_item_ft(df)\n    \n    if not INFERENCE:\n        # Get item features\n        tmp = get_item_ft_df(df)\n        update_ft_df('item', 'all', 'all', f'i_all_v{VER}.pqt')\n        tmp.to_pandas().to_parquet(f'i_all_v{VER}.pqt')\n\n        # Get user features\n        tmp = get_user_ft_df(df)\n        update_ft_df('user', 'all', 'all', f'u_all_v{VER}.pqt')\n        tmp.to_pandas().to_parquet(f'u_all_v{VER}.pqt')\n        \n        # Save user-item features\n        update_ft_df('user_item', 'all', 'all', f'ui_all_v{VER}.pqt')\n        ui_df.to_pandas().to_parquet(f'ui_all_v{VER}.pqt')\n        \n        # Save feature metadata\n        FEATURES_DF.to_parquet(f'feature_metadata_v{VER}.pqt')\n    \n    return ui_df\n\ndef join_aid(df):\n    df = df.groupby('session')['aid'].apply(lambda x: ' '.join(map(str,x)))\n    df = df.reset_index()\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:44:18.741397Z","iopub.execute_input":"2023-01-28T16:44:18.74177Z","iopub.status.idle":"2023-01-28T16:44:18.784857Z","shell.execute_reply.started":"2023-01-28T16:44:18.74173Z","shell.execute_reply":"2023-01-28T16:44:18.783786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Preprocess\npred_ds_prcess = preprocess_test(pred_ds)","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:44:29.143834Z","iopub.execute_input":"2023-01-28T16:44:29.144416Z","iopub.status.idle":"2023-01-28T16:44:40.587695Z","shell.execute_reply.started":"2023-01-28T16:44:29.14437Z","shell.execute_reply":"2023-01-28T16:44:40.585851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Clicks","metadata":{}},{"cell_type":"code","source":"if not INFERENCE:\n    def get_target_df(lbl_type, name, df=pred_lbls):\n        if isinstance(lbl_type, list) == False: lbl_type = [lbl_type]\n        lbl_subset = df.loc[df.type.isin(lbl_type)].reset_index(drop=True)\n        lbl_subset = lbl_subset.explode('ground_truth')[['session', 'ground_truth']]\n        lbl_subset = lbl_subset.drop_duplicates(ignore_index=True)\n        lbl_subset.columns = ['session', 'aid']\n        lbl_subset[name] = 1\n        lbl_subset[name] = lbl_subset[name].astype('uint8')\n\n        return lbl_subset","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:44:40.589877Z","iopub.execute_input":"2023-01-28T16:44:40.590546Z","iopub.status.idle":"2023-01-28T16:44:40.597765Z","shell.execute_reply.started":"2023-01-28T16:44:40.590504Z","shell.execute_reply":"2023-01-28T16:44:40.59651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_aid(df, col_name):\n    tmp = df[['session', 'aid']].groupby('session').count().reset_index()\n    tmp.columns = ['session', col_name]\n    df = df.merge(tmp, on='session', how='left')\n    return df\n\ndef get_clk_cov(df, candidates=CANDIDATES):\n    def get_wgt(df):\n        df = df.merge(top_clk_cov, on='aid', how='left')\n        df = df[['session', 'top_clk_cov', 'wgt']].groupby(['session', 'top_clk_cov']).sum()\n        df = df.reset_index()\n        df.columns = ['session', 'aid', 'clk_cov_wgt']\n        return df\n    \n    df = count_aid(df, 'aids')\n    \n    # Get top covisited clicks\n    tmp_more = df.loc[df.aids >= candidates, ['session', 'aid']]\n    tmp_more1 = get_wgt(tmp_more)\n    tmp_more = tmp_more.merge(tmp_more1, on=['session', 'aid'], how='left')\n    \n    tmp_fewer = df.loc[df.aids < candidates, ['session', 'aid']]\n    tmp_fewer = get_wgt(tmp_fewer)\n    \n    tmp = cudf.concat([tmp_more, tmp_fewer])\n    del tmp_more, tmp_fewer\n    gc.collect()\n    \n    df = df.merge(tmp, on=['session', 'aid'], how='outer')\n    df = df.drop('aids', axis=1).fillna(0)\n    return df\n\ndef get_clk_wk(df, candidates=CANDIDATES):\n    top_clk_wk = top_clk_buy_wk.loc[top_clk_buy_wk.top_clk_wk == 1, ['aid', 'top_clk_wk']]\n    df = count_aid(df, 'aids')\n    \n    df = df.merge(top_clk_wk, on='aid', how='left')\n    \n    tmp_fewer = df.loc[df.aids < candidates][['session']].drop_duplicates()\n    tmp_fewer = tmp_fewer.to_pandas()\n    top_clk_wk = top_clk_wk.to_pandas()\n    tmp_fewer = cudf.DataFrame(tmp_fewer.merge(top_clk_wk, how='cross'))\n    df = df.merge(tmp_fewer, on=['session', 'aid'], how='outer')\n    df['top_clk_wk'] = df.top_clk_wk_x.fillna(df.top_clk_wk_y)\n    del tmp_fewer, top_clk_wk\n    gc.collect()\n    \n    df = df.drop(['aids', 'top_clk_wk_x', 'top_clk_wk_y'], axis=1)\n    df = df.fillna(0)\n    return df\n\ndef suggest_click(df, candidates=CANDIDATES, add_target=True):\n    if 'order_wgt' in df.columns:\n        df = df.drop('order_wgt', axis=1)\n    df = get_clk_cov(df, candidates)\n    df = get_clk_wk(df, candidates)\n    \n    # Sort and get top candidates\n    df = df.sort_values(\n        ['session', 'ui_clk_wgt', 'clk_cov_wgt', 'top_clk_wk']\n        , ascending=[True, False, False, False]\n        , ignore_index=True\n    ).reset_index(drop=True)\n    df['n'] = df.groupby('session').cumcount()\n    df = df.loc[df.n < candidates]\n    df = df.drop(['n', 'ui_clk_wgt'], axis=1)\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:44:40.599493Z","iopub.execute_input":"2023-01-28T16:44:40.599907Z","iopub.status.idle":"2023-01-28T16:44:40.617056Z","shell.execute_reply.started":"2023-01-28T16:44:40.599872Z","shell.execute_reply":"2023-01-28T16:44:40.616037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def timer(sta):\n    return (dt.now() - sta).seconds\n\ndef create_suggestions(df, partitions, suggest_fn, candidates=CANDIDATES, verbose=False):\n    suggestions = []\n    for p in range(len(partitions)):\n        if verbose: print(f'Processing partition {p+1}...', end='')\n        part = df.loc[partitions.loc[p,'min']:partitions.loc[p,'max']]\n#         part = part.drop('chunk', axis=1)\n        part = suggest_fn(part, candidates=candidates)\n        suggestions.append(part)\n        if verbose: print('done!')\n        del part\n        gc.collect()\n    return cudf.concat(suggestions, ignore_index=True)\n\ndef check_positive_coverage(\n    df, partn, target_code\n    , suggest_fn, target_col\n    , cand_nums=[CANDIDATES]\n):\n    candidates = []\n    coverage = []\n    if not isinstance(target_code, list):\n        target_code = [target_code]\n    for cand_num in cand_nums:\n        print('\\n' + '='*20 + f'\\n{cand_num} candidates')\n        sta = dt.now()\n        suggestions = create_suggestions(df, partn, suggest_fn, cand_num)\n        print(f'Suggestions created ({round(timer(sta), 2)} sec)')\n        ground_truth =  len(pred_lbls.loc[pred_lbls.type.isin(target_code)].explode(\"ground_truth\"))\n        suggestions = suggestions[target_col].value_counts()[1]\n        ratio = round(suggestions/ground_truth,4)\n        if cand_num == cand_nums[0]: gain = 'none'\n        else: gain = round(ratio - coverage[-1],4)\n        candidates.append(cand_num)\n        coverage.append(ratio)\n        ratio_str = f'(Coverage={ratio}, gain={gain})'\n        print(f'Suggestion/Total positives: {suggestions}/{ground_truth} {ratio_str}')\n    plt.figure(figsize=(10,5))\n    plt.plot(candidates, coverage)\n    plt.title('Positive Coverage by Number of Candidates')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:44:40.620139Z","iopub.execute_input":"2023-01-28T16:44:40.620497Z","iopub.status.idle":"2023-01-28T16:44:40.635195Z","shell.execute_reply.started":"2023-01-28T16:44:40.620463Z","shell.execute_reply":"2023-01-28T16:44:40.634358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Partitioning before candidate suggestion\npred_ds_prcess_clk, clk_ptn = partition(\n    pred_ds_prcess.loc[:, ['session', 'aid', 'ui_clk_wgt']], 6\n)\n\ncandidates_clk = create_suggestions(pred_ds_prcess_clk, clk_ptn, suggest_click, CANDIDATES, True)\nif INFERENCE:\n    candidates_clk = candidates_clk[['session', 'aid']]\n    candidates_clk = join_aid(candidates_clk.to_pandas())\n    candidates_clk.session = candidates_clk.session.astype(str) + '_clicks'\n    candidates_clk.columns = ['session_type', 'labels']\nelse:\n    clk_labels = get_target_df(0, 'click_target')\n    candidates_clk = candidates_clk.merge(clk_labels, on=['session', 'aid'], how='left')\n    candidates_clk = candidates_clk.fillna(0)\n    del clk_labels\n    gc.collect()\ndel pred_ds_prcess_clk, clk_ptn\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:44:40.636572Z","iopub.execute_input":"2023-01-28T16:44:40.636922Z","iopub.status.idle":"2023-01-28T16:44:56.887313Z","shell.execute_reply.started":"2023-01-28T16:44:40.636886Z","shell.execute_reply":"2023-01-28T16:44:56.882501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Carts & Orders","metadata":{}},{"cell_type":"code","source":"def count_aid_12(df, col_name):\n    tmp = df.loc[df.ui_order_wgt > 0][['session', 'aid']]\n    tmp = tmp.groupby('session').count().reset_index()\n    tmp.columns = ['session', col_name]\n    df = df.merge(tmp, on='session', how='left').fillna(0).reset_index(drop=True)\n    return df\n\ndef get_cart_ord_cov(df, candidates=CANDIDATES):\n    def get_wgt(df):\n        # Buy cov\n        tmp = df.merge(top_buy_cov, on='aid', how='left')[['session', 'top_buy_cov', 'wgt']]\n        tmp.columns = ['session', 'aid', 'buy_cov_wgt']\n        tmp = tmp.groupby(['session', 'aid']).sum().reset_index()\n        # Buy2buy cov\n        df = df.merge(top_b2b_cov, on='aid', how='left')[['session', 'top_b2b_cov', 'wgt']]\n        df.columns = ['session', 'aid', 'b2b_cov_wgt']\n        df = df.groupby(['session', 'aid']).sum().reset_index()\n        # Merge covs\n        df = df.merge(tmp, on=['session', 'aid'], how='outer').fillna(0)\n        df['cov_wgt_sum'] = df.buy_cov_wgt + df.b2b_cov_wgt\n        return df\n    \n    # Get top covisited clicks\n    tmp_more = df.loc[df.aids_12 >= candidates, ['session', 'aid']]\n    tmp_more1 = get_wgt(tmp_more)\n    tmp_more = tmp_more.merge(tmp_more1, on=['session', 'aid'], how='left')\n    \n    tmp_fewer = df.loc[df.aids_12 < candidates, ['session', 'aid']]\n    tmp_fewer = get_wgt(tmp_fewer)\n    \n    tmp = cudf.concat([tmp_more, tmp_fewer])\n    del tmp_more, tmp_more1, tmp_fewer\n    gc.collect()\n    \n    df = df.merge(tmp, on=['session', 'aid'], how='outer')\n    df = df.drop('aids_12', axis=1).fillna(0)\n    return df\n\ndef get_order_wk(df, candidates=CANDIDATES):\n    def outer_merge(df1, df2):\n        df1 = df1.to_pandas()\n        df2 = df2.to_pandas()\n        return cudf.DataFrame(df1.merge(df2, how='cross'))\n    top_buy_wk = top_clk_buy_wk.loc[top_clk_buy_wk.top_buy_wk == 1, ['aid', 'top_buy_wk']]\n    if 'aid' not in df.columns:\n        df = outer_merge(df, top_buy_wk)\n    else:\n        df = df.merge(top_buy_wk, on='aid', how='left')\n        df = count_aid_12(df, 'aids_12')\n        tmp_fewer = df.loc[df.aids_12 < candidates][['session']].drop_duplicates()\n        tmp_fewer = outer_merge(tmp_fewer, top_buy_wk)\n        df = df.merge(tmp_fewer, on=['session', 'aid'], how='outer')\n        df['top_buy_wk'] = df.top_buy_wk_x.fillna(df.top_buy_wk_y)\n        df = df.drop(['aids_12', 'top_buy_wk_x', 'top_buy_wk_y'], axis=1)\n        df = df.fillna(0)\n        del tmp_fewer\n    del top_buy_wk\n    gc.collect()\n    return df\n\ndef suggest_cart_order(df, candidates=CANDIDATES, add_target=True):\n    if 'clk_wgt' in df.columns:\n        df = df.drop('clk_wgt', axis=1)\n    # Split data to non-order sessions and having-order sessions\n    df = count_aid_12(df, 'aids_12')\n    tmp_zero = df.loc[df.aids_12 == 0, ['session']].drop_duplicates(ignore_index=True)\n    tmp_non_z = df.loc[df.aids_12 > 0]\n    \n    # Get cov weight\n    tmp_non_z = get_cart_ord_cov(tmp_non_z, candidates)\n    # Get top order week\n    tmp_zero = get_order_wk(tmp_zero, candidates)\n    tmp_non_z = get_order_wk(tmp_non_z, candidates)\n    \n    # Concat data\n    for c in tmp_non_z.columns:\n        if c not in tmp_zero.columns:\n            tmp_zero[c] = 0\n    tmp_zero = tmp_zero.loc[:, tmp_non_z.columns]\n    df = cudf.concat([tmp_zero, tmp_non_z], ignore_index=True)\n    del tmp_zero, tmp_non_z\n    gc.collect()\n    \n    # Get top candidates\n    df = df.sort_values(\n        ['session', 'ui_order_wgt', 'cov_wgt_sum'\n         , 'b2b_cov_wgt', 'buy_cov_wgt', 'top_buy_wk']\n        , ascending=False\n        , ignore_index=True\n    ).reset_index(drop=True)\n    df['n'] = df.groupby('session').cumcount()\n    df = df.loc[df.n < candidates]\n    \n    # Post-processing\n    df = df.drop(['ui_order_wgt', 'n'], axis=1)\n    df = df.astype({\n        'aid': 'int32'\n        , 'buy_cov_wgt': 'float32'\n    })\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:44:56.889075Z","iopub.execute_input":"2023-01-28T16:44:56.88971Z","iopub.status.idle":"2023-01-28T16:44:56.921307Z","shell.execute_reply.started":"2023-01-28T16:44:56.88967Z","shell.execute_reply":"2023-01-28T16:44:56.919896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Partition before candidates selection\npred_ds_prcess_ord, ord_ptn = partition(\n    pred_ds_prcess.loc[:, ['session', 'aid', 'ui_order_wgt']], 3\n)\nprint(f'Done partitioning - {len(ord_ptn)} partitions!')\n\ncandidates_ord = create_suggestions(\n    pred_ds_prcess_ord, ord_ptn, suggest_cart_order, CANDIDATES, True\n)\nif INFERENCE:\n    candidates_ord = candidates_ord[['session', 'aid']]\n    candidates_ord = join_aid(candidates_ord.to_pandas())\n    candidates_ord.columns = ['session_type', 'labels']\n    candidates_cart = candidates_ord.copy()\n    candidates_ord.session_type = candidates_ord.session_type.astype(str) + '_orders'\n    candidates_cart.session_type = candidates_cart.session_type.astype(str) + '_carts'\nelse:\n    # Add labels\n    cart_labels = get_target_df(1, 'cart_target')\n    ord_labels = get_target_df(2, 'order_target')\n    candidates_ord = candidates_ord.merge(cart_labels, on=['session', 'aid'], how='left')\n    candidates_ord = candidates_ord.merge(ord_labels, on=['session', 'aid'], how='left')\n    candidates_ord = candidates_ord.fillna(0)\n    del ord_labels, cart_labels\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:44:56.922799Z","iopub.execute_input":"2023-01-28T16:44:56.923513Z","iopub.status.idle":"2023-01-28T16:45:20.232887Z","shell.execute_reply.started":"2023-01-28T16:44:56.923471Z","shell.execute_reply":"2023-01-28T16:45:20.231852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save result","metadata":{"execution":{"iopub.status.busy":"2023-01-12T07:07:10.046664Z","iopub.execute_input":"2023-01-12T07:07:10.047091Z","iopub.status.idle":"2023-01-12T07:07:10.059431Z","shell.execute_reply.started":"2023-01-12T07:07:10.047056Z","shell.execute_reply":"2023-01-12T07:07:10.058363Z"}}},{"cell_type":"code","source":"if INFERENCE:\n    print('Submitting results...')\n    submission = pd.concat([candidates_clk, candidates_ord, candidates_cart])\n    submission.to_csv('submission.csv', index=False)\n    print(submission.head())\nelse:\n    print('Saving results...')\n    candidates_clk.to_pandas().to_parquet(f'click_candidates_top{CANDIDATES}_v{VER}.pqt')\n    candidates_ord.to_pandas().to_parquet(f'order_carts_candidates_top{CANDIDATES}_v{VER}.pqt')\n    print(candidates_clk.head(), end='\\n\\n')\n    print(candidates_ord.head())","metadata":{"execution":{"iopub.status.busy":"2023-01-28T16:45:20.234358Z","iopub.execute_input":"2023-01-28T16:45:20.235036Z","iopub.status.idle":"2023-01-28T16:45:39.457405Z","shell.execute_reply.started":"2023-01-28T16:45:20.234994Z","shell.execute_reply":"2023-01-28T16:45:39.456197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}