{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Version details\n- Best version from Notebook version 1: LB 0.570\n- V1: LB 0.574 | otto-metadata V9\n- V5: first attemp at using 4 covisitations\n- V10: ~ V5 use parquet public -> better\n- V12: new pipeline for generating candidates\n- V17: ~V12, change the order of aids2,aids3\n- V18: ~V17, top_output=150\n- V19: ~V17, top_output=50\n- V20: add top common aid in test","metadata":{}},{"cell_type":"code","source":"!pip install pyarrow fastparquet","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEBUG = False\n\n# --- OTTO [Build metadata] notebook params\nTOP_K = 20\nTOP_OUTPUT = 100\nLEAK_DATA = True\nDELTA_TS = 24 # hours\nDISK_PIECES = 4 \nSIZE = 1.86e6/DISK_PIECES # total session\n# ---\n\nmetadata_path = \"/kaggle/input/otto-metadata\"\ntype_labels = {'clicks':0, 'carts':1, 'orders':2}","metadata":{"execution":{"iopub.status.busy":"2022-12-15T02:56:34.049057Z","iopub.execute_input":"2022-12-15T02:56:34.049549Z","iopub.status.idle":"2022-12-15T02:56:34.082394Z","shell.execute_reply.started":"2022-12-15T02:56:34.049453Z","shell.execute_reply":"2022-12-15T02:56:34.081004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd, numpy as np\nimport copy\nfrom tqdm.notebook import tqdm\nimport os, sys, pickle, glob, gc\nfrom collections import Counter\nimport itertools\n\nfrom multiprocessing import Pool\nimport psutil\nN_CPU = psutil.cpu_count()\nprint(\"Number of cpu:\", N_CPU)","metadata":{"execution":{"iopub.status.busy":"2022-12-15T02:56:34.084397Z","iopub.execute_input":"2022-12-15T02:56:34.084767Z","iopub.status.idle":"2022-12-15T02:56:34.228463Z","shell.execute_reply.started":"2022-12-15T02:56:34.084735Z","shell.execute_reply":"2022-12-15T02:56:34.227200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def df_parallelize_run(func, t_split):\n    \n    num_cores = np.min([N_CPU, len(t_split)])\n    pool = Pool(num_cores)\n    df = pool.map(func, t_split)\n    pool.close()\n    pool.join()\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-12-15T02:56:34.230547Z","iopub.execute_input":"2022-12-15T02:56:34.231001Z","iopub.status.idle":"2022-12-15T02:56:34.238400Z","shell.execute_reply.started":"2022-12-15T02:56:34.230966Z","shell.execute_reply":"2022-12-15T02:56:34.236795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pqt_to_dict(df):\n#     if USE_FREQUENCY_SCORE:\n#         return df.groupby('aid_x').apply(lambda df: Counter(dict(zip(df.aid_y, df.wgt))))\n    return df.groupby('aid_x').aid_y.apply(list).to_dict()","metadata":{"execution":{"iopub.status.busy":"2022-12-15T02:56:34.240279Z","iopub.execute_input":"2022-12-15T02:56:34.241062Z","iopub.status.idle":"2022-12-15T02:56:34.253570Z","shell.execute_reply.started":"2022-12-15T02:56:34.241015Z","shell.execute_reply":"2022-12-15T02:56:34.251523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#type_weight_multipliers = {'clicks': 1, 'carts': 6, 'orders': 3}\ntype_weight_multipliers = {0: 0.5, 1: 9, 2: 0.5}\n\ndef suggest_clicks(df):\n    session = df[0]\n    aids = df[1]\n    types = df[2]\n    unique_aids = list(dict.fromkeys(aids[::-1]))\n    \n    # history candidates\n    weights=np.logspace(0.1,1,len(aids),base=2, endpoint=True)-1\n    aids_temp = Counter() \n    # RERANK BASED ON REPEAT ITEMS AND TYPE OF ITEMS\n    for aid,w,t in zip(aids,weights,types): \n        aids_temp[aid] += w * type_weight_multipliers[t]\n\n    sorted_aids = []\n    weights = []\n    type_cands = []\n    \n    if len(unique_aids) >= TOP_OUTPUT:\n        for aid, cnt in aids_temp.most_common(TOP_OUTPUT):\n            sorted_aids.append(aid)\n            weights.append(cnt)\n            type_cands.append(1)\n        return session, sorted_aids, weights, type_cands\n    else:\n        for aid, cnt in aids_temp.most_common(len(unique_aids)):\n            sorted_aids.append(aid)\n            weights.append(cnt)\n            type_cands.append(1)\n            \n    # potential candidates\n    aids2 = list(itertools.chain(*[top_click[aid] for aid in unique_aids if aid in top_click]))\n    aids3 = list(itertools.chain(*[top_fulltype[aid] for aid in unique_aids if aid in top_fulltype]))\n    \n    top_aids2 = Counter(aids2+aids3)\n    top_aids2 = [(aid2,cnt) for aid2, cnt in top_aids2.most_common(TOP_OUTPUT) if aid2 not in unique_aids]\n    for aid, cnt in top_aids2[:TOP_OUTPUT - len(unique_aids)]:\n        sorted_aids.append(aid)\n        weights.append(cnt)\n        type_cands.append(0)    \n            \n    if len(sorted_aids) < TOP_OUTPUT:\n        top_common_aid = [aid for aid in top_clicks if aid not in sorted_aids]\n        for aid in top_common_aid[:TOP_OUTPUT - len(sorted_aids)]:\n            sorted_aids.append(aid)\n            weights.append(0)\n            type_cands.append(2)\n    return session, sorted_aids, weights, type_cands\n\ndef suggest_carts(df):\n    # USE USER HISTORY AIDS AND TYPES\n    session = df[0]\n    aids = df[1]\n    types = df[2]\n    unique_aids = list(dict.fromkeys(aids[::-1]))\n\n    # history candidates\n    weights=np.logspace(0.5,1,len(aids),base=2, endpoint=True)-1\n    aids_temp = Counter() \n    # RERANK BASED ON REPEAT ITEMS AND TYPE OF ITEMS\n    for aid,w,t in zip(aids,weights,types): \n        aids_temp[aid] += w * type_weight_multipliers[t]\n\n    sorted_aids = []\n    weights = []\n    type_cands = []\n    \n    if len(unique_aids) >= TOP_OUTPUT:\n        for aid, cnt in aids_temp.most_common(TOP_OUTPUT):\n            sorted_aids.append(aid)\n            weights.append(cnt)\n            type_cands.append(1)\n        return session, sorted_aids, weights, type_cands\n    else:\n        for aid, cnt in aids_temp.most_common(len(unique_aids)):\n            sorted_aids.append(aid)\n            weights.append(cnt)\n            type_cands.append(1)\n            \n    # potential candidates\n    aids2 = list(itertools.chain(*[top_cart[aid] for aid in unique_aids if aid in top_cart]))\n    aids3 = list(itertools.chain(*[top_fulltype[aid] for aid in unique_aids if aid in top_fulltype]))\n    \n    top_aids2 = Counter(aids3+aids2)\n    top_aids2 = [(aid2,cnt) for aid2, cnt in top_aids2.most_common(TOP_OUTPUT) if aid2 not in unique_aids]\n    for aid, cnt in top_aids2[:TOP_OUTPUT - len(unique_aids)]:\n        sorted_aids.append(aid)\n        weights.append(cnt)\n        type_cands.append(0)\n        \n    if len(sorted_aids) < TOP_OUTPUT:\n        top_common_aid = [aid for aid in top_carts if aid not in sorted_aids]\n        for aid in top_common_aid[:TOP_OUTPUT - len(sorted_aids)]:\n            sorted_aids.append(aid)\n            weights.append(0)\n            type_cands.append(2)\n            \n    return session, sorted_aids, weights, type_cands\n\n\ndef suggest_orders(df):\n    # USE USER HISTORY AIDS AND TYPES\n    session = df[0]\n    aids = df[1]\n    types = df[2]\n    unique_aids = list(dict.fromkeys(aids[::-1] ))\n    \n    # history candidates\n    weights=np.logspace(0.5,1,len(aids),base=2, endpoint=True)-1\n    aids_temp = Counter() \n    # RERANK BASED ON REPEAT ITEMS AND TYPE OF ITEMS\n    for aid,w,t in zip(aids,weights,types): \n        aids_temp[aid] += w * type_weight_multipliers[t]\n\n    sorted_aids = []\n    weights = []\n    type_cands = []\n    \n    if len(unique_aids) >= TOP_OUTPUT:\n        for aid, cnt in aids_temp.most_common(TOP_OUTPUT):\n            sorted_aids.append(aid)\n            weights.append(cnt)\n            type_cands.append(1)\n        return session, sorted_aids, weights, type_cands\n    else:\n        for aid, cnt in aids_temp.most_common(len(unique_aids)):\n            sorted_aids.append(aid)\n            weights.append(cnt)\n            type_cands.append(1)\n            \n    # potential candidates\n    aids2 = list(itertools.chain(*[top_purchase[aid] for aid in unique_aids if aid in top_purchase]))\n    aids3 = list(itertools.chain(*[top_fulltype[aid] for aid in unique_aids if aid in top_fulltype]))\n    \n    top_aids2 = Counter(aids3+aids2)\n    top_aids2 = [(aid2,cnt) for aid2, cnt in top_aids2.most_common(TOP_OUTPUT) if aid2 not in unique_aids]\n    for aid, cnt in top_aids2[:TOP_OUTPUT - len(unique_aids)]:\n        sorted_aids.append(aid)\n        weights.append(cnt)\n        type_cands.append(0)    \n            \n    if len(sorted_aids) < TOP_OUTPUT:\n        top_common_aid = [aid for aid in top_orders if aid not in sorted_aids]\n        for aid in top_common_aid[:TOP_OUTPUT - len(sorted_aids)]:\n            sorted_aids.append(aid)\n            weights.append(0)\n            type_cands.append(2)\n    return session, sorted_aids, weights, type_cands","metadata":{"execution":{"iopub.status.busy":"2022-12-15T02:56:34.255988Z","iopub.execute_input":"2022-12-15T02:56:34.256364Z","iopub.status.idle":"2022-12-15T02:56:34.270688Z","shell.execute_reply.started":"2022-12-15T02:56:34.256330Z","shell.execute_reply":"2022-12-15T02:56:34.268561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict\nHere a submission file is created, the same than [here](https://www.kaggle.com/code/cdeotte/candidate-rerank-model-lb-0-575) but faster.","metadata":{}},{"cell_type":"code","source":"def load_test(path):    \n    dfs = []\n    for e, chunk_file in sorted(enumerate(glob.glob(path))):\n        chunk = pd.read_parquet(chunk_file)\n        chunk.ts = (chunk.ts/1000).astype('int32')\n        chunk['type'] = chunk['type'].map(type_labels).astype('int8')\n        dfs.append(chunk)\n    return pd.concat(dfs).reset_index(drop=True) #.astype({\"ts\": \"datetime64[ms]\"})\n\ntest_df = load_test('/kaggle/input/otto-chunk-data-inparquet-format/test_parquet/*')\nprint('Test data has shape',test_df.shape)\ntest_df.head()\n\ntop_clicks = list(test_df.loc[test_df['type']== 0,'aid'].value_counts().index.values[:TOP_OUTPUT]) \ntop_carts = list(test_df.loc[test_df['type']== 1,'aid'].value_counts().index.values[:TOP_OUTPUT])\ntop_orders = list(test_df.loc[test_df['type']== 2,'aid'].value_counts().index.values[:TOP_OUTPUT])\ndel test_df\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nPIECES = 5\nVER = 1\ntest_bysession_list = []\nfor PART in range(PIECES):\n    with open(f'../input/otto-valid-test-list/test_group_tolist_{PART}_{VER}.pkl', 'rb') as f:\n        test_bysession_list.extend(pickle.load(f))\nprint(len(test_bysession_list))","metadata":{"execution":{"iopub.status.busy":"2022-12-15T02:56:34.330412Z","iopub.execute_input":"2022-12-15T02:56:34.331224Z","iopub.status.idle":"2022-12-15T02:56:46.301188Z","shell.execute_reply.started":"2022-12-15T02:56:34.331176Z","shell.execute_reply":"2022-12-15T02:56:46.299695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nTYPE_COO = \"fulltype\"\nDELTA_TS = 24\nOUTPUT_NAME = f\"top_{TOP_K}_{TYPE_COO}_{DELTA_TS}hours\"\ntry:\n    assert USE_PICKLE_COVI == True\n    with open(f'/kaggle/input/otto-metadata/{OUTPUT_NAME}_full.pickle', 'rb') as file:\n        top_fulltype = pickle.load(file)\n    print(f\"Load {TYPE_COO} covisitaion | dict format\")\nexcept:\n    print(f\"Load {TYPE_COO} covisitaion | dataframe format\")\n    top_fulltype = pqt_to_dict(pd.read_parquet(f'/kaggle/input/otto-metadata/{OUTPUT_NAME}_full.pqt'))","metadata":{"execution":{"iopub.status.busy":"2022-12-15T02:56:46.302926Z","iopub.execute_input":"2022-12-15T02:56:46.303297Z","iopub.status.idle":"2022-12-15T02:57:33.517865Z","shell.execute_reply.started":"2022-12-15T02:56:46.303255Z","shell.execute_reply":"2022-12-15T02:57:33.516906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Click","metadata":{}},{"cell_type":"code","source":"%%time\nTYPE_COO = \"click\"\nDELTA_TS = 24\nOUTPUT_NAME = f\"top_{TOP_K}_{TYPE_COO}_{DELTA_TS}hours\"\ntry:\n    assert USE_PICKLE_COVI == True\n    with open(f'/kaggle/input/otto-metadata/{OUTPUT_NAME}_full.pickle', 'rb') as file:\n        top_click = pickle.load(file)\n    print(f\"Load {TYPE_COO} covisitaion | dict format\")\nexcept:\n    print(f\"Load {TYPE_COO} covisitaion | dataframe format\")\n    top_click = pqt_to_dict(pd.read_parquet(f'/kaggle/input/otto-metadata/{OUTPUT_NAME}_full.pqt'))","metadata":{"execution":{"iopub.status.busy":"2022-12-15T02:57:33.521933Z","iopub.execute_input":"2022-12-15T02:57:33.523082Z","iopub.status.idle":"2022-12-15T02:58:01.306350Z","shell.execute_reply.started":"2022-12-15T02:57:33.523029Z","shell.execute_reply":"2022-12-15T02:58:01.304983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# # Predict on all sessions in parallel\ntemp = df_parallelize_run(suggest_clicks, test_bysession_list)\nsession, aids, weights, type_cands = [], [], [], []\nfor _session, _aids, _weights, _type in temp:\n    session.extend([_session]*len(_aids))\n    aids.extend(_aids)\n    weights.extend(_weights)\n    type_cands.extend(_type)\n    \ncandidates_df = pd.DataFrame({'session':session,'aid':aids,'score':weights,'type_candidate':type_cands})\ncandidates_df.to_parquet('/kaggle/working/clicks_candidates.pqt')","metadata":{"execution":{"iopub.status.busy":"2022-12-15T02:58:01.308424Z","iopub.execute_input":"2022-12-15T02:58:01.308979Z","iopub.status.idle":"2022-12-15T02:59:09.726807Z","shell.execute_reply.started":"2022-12-15T02:58:01.308933Z","shell.execute_reply":"2022-12-15T02:59:09.725554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del top_click, temp, candidates_df\ndel session, aids, weights, type_cands\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cart","metadata":{}},{"cell_type":"code","source":"%%time\nTYPE_COO = \"cart\"\nDELTA_TS = 24*7\nOUTPUT_NAME = f\"top_{TOP_K}_{TYPE_COO}_{DELTA_TS}hours\"\ntry:\n    assert USE_PICKLE_COVI == True\n    with open(f'/kaggle/input/otto-metadata/{OUTPUT_NAME}_full.pickle', 'rb') as file:\n        top_cart = pickle.load(file)\n    print(f\"Load {TYPE_COO} covisitaion | dict format\")\nexcept:\n    print(f\"Load {TYPE_COO} covisitaion | dataframe format\")\n    top_cart = pqt_to_dict(pd.read_parquet(f'/kaggle/input/otto-metadata/{OUTPUT_NAME}_full.pqt'))","metadata":{"execution":{"iopub.status.busy":"2022-12-15T02:59:09.728405Z","iopub.execute_input":"2022-12-15T02:59:09.729016Z","iopub.status.idle":"2022-12-15T02:59:54.915028Z","shell.execute_reply.started":"2022-12-15T02:59:09.728977Z","shell.execute_reply":"2022-12-15T02:59:54.912975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# # Predict on all sessions in parallel\ntemp = df_parallelize_run(suggest_carts, test_bysession_list)\nsession, aids, weights, type_cands = [], [], [], []\nfor _session, _aids, _weights, _type in temp:\n    session.extend([_session]*len(_aids))\n    aids.extend(_aids)\n    weights.extend(_weights)\n    type_cands.extend(_type)\n    \ncandidates_df = pd.DataFrame({'session':session,'aid':aids,'score':weights,'type_candidate':type_cands})\ncandidates_df.to_parquet('/kaggle/working/carts_candidates.pqt')","metadata":{"execution":{"iopub.status.busy":"2022-12-15T02:59:54.916868Z","iopub.execute_input":"2022-12-15T02:59:54.917335Z","iopub.status.idle":"2022-12-15T03:01:26.230646Z","shell.execute_reply.started":"2022-12-15T02:59:54.917292Z","shell.execute_reply":"2022-12-15T03:01:26.228908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del top_cart, temp, candidates_df\ndel session, aids, weights, type_cands\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Orders","metadata":{}},{"cell_type":"code","source":"%%time\nTYPE_COO = \"purchase\"\nDELTA_TS = 24*7\nOUTPUT_NAME = f\"top_{TOP_K}_{TYPE_COO}_{DELTA_TS}hours\"\ntry:\n    assert USE_PICKLE_COVI == True\n    with open(f'/kaggle/input/otto-metadata/{OUTPUT_NAME}_full.pickle', 'rb') as file:\n        top_purchase = pickle.load(file)\n    print(f\"Load {TYPE_COO} covisitaion | dict format\")\nexcept:\n    print(f\"Load {TYPE_COO} covisitaion | dataframe format\")\n    top_purchase = pqt_to_dict(pd.read_parquet(f'/kaggle/input/otto-metadata/{OUTPUT_NAME}_full.pqt'))","metadata":{"execution":{"iopub.status.busy":"2022-12-15T03:01:26.233193Z","iopub.execute_input":"2022-12-15T03:01:26.234455Z","iopub.status.idle":"2022-12-15T03:02:17.478308Z","shell.execute_reply.started":"2022-12-15T03:01:26.234403Z","shell.execute_reply":"2022-12-15T03:02:17.476698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# # Predict on all sessions in parallel\ntemp = df_parallelize_run(suggest_orders, test_bysession_list)\nsession, aids, weights, type_cands = [], [], [], []\nfor _session, _aids, _weights, _type in temp:\n    session.extend([_session]*len(_aids))\n    aids.extend(_aids)\n    weights.extend(_weights)\n    type_cands.extend(_type)\n    \ncandidates_df = pd.DataFrame({'session':session,'aid':aids,'score':weights,'type_candidate':type_cands})\ncandidates_df.to_parquet('/kaggle/working/orders_candidates.pqt')","metadata":{"execution":{"iopub.status.busy":"2022-12-15T03:02:17.480255Z","iopub.execute_input":"2022-12-15T03:02:17.481065Z","iopub.status.idle":"2022-12-15T03:03:58.305297Z","shell.execute_reply.started":"2022-12-15T03:02:17.481017Z","shell.execute_reply":"2022-12-15T03:03:58.303839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del top_purchase, temp, candidates_df\ndel session, aids, weights, type_cands\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]}]}