{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":38760,"databundleVersionId":4493939,"sourceType":"competition"},{"sourceId":4436180,"sourceType":"datasetVersion","datasetId":2597726},{"sourceId":7203605,"sourceType":"datasetVersion","datasetId":4167267}],"dockerImageVersionId":30627,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"colab":{"provenance":[]}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.simplefilter(action='ignore')\n\nimport pandas as pd, numpy as np\nfrom tqdm.notebook import tqdm\nimport os, sys, pickle, glob, gc\nfrom collections import Counter\nimport cudf, itertools\n\nfrom pathlib import Path","metadata":{"id":"5o8pYDK_2YPC","outputId":"136f4007-156a-4621-8c39-39456aeee201","execution":{"iopub.status.busy":"2023-12-14T20:34:51.842784Z","iopub.execute_input":"2023-12-14T20:34:51.843057Z","iopub.status.idle":"2023-12-14T20:34:57.259165Z","shell.execute_reply.started":"2023-12-14T20:34:51.843032Z","shell.execute_reply":"2023-12-14T20:34:57.25806Z"},"trusted":true},"execution_count":1,"outputs":[]},{"cell_type":"code","source":"# functions for reading files\ndef read_file(f):\n    return cudf.DataFrame( data_cache[f] )\ndef read_file_to_cache(f):\n    df = pd.read_parquet(f)\n    df.ts = (df.ts/1000).astype('int32')\n    df['type'] = df['type'].map(type_labels).astype('int8')\n    return df\n\n# cache data\ndata_cache = {}\ntype_labels = {'clicks':0, 'carts':1, 'orders':2}\nfiles = glob.glob('../input/otto-chunk-data-inparquet-format/*_parquet/*')\nfor f in files: data_cache[f] = read_file_to_cache(f)","metadata":{"id":"_qdEcKcs2YPE","outputId":"724af9fc-b0dd-4045-ec52-0458caa64760","execution":{"iopub.status.busy":"2023-12-14T20:38:12.688325Z","iopub.execute_input":"2023-12-14T20:38:12.688736Z","iopub.status.idle":"2023-12-14T20:39:21.509215Z","shell.execute_reply.started":"2023-12-14T20:38:12.688707Z","shell.execute_reply":"2023-12-14T20:39:21.508321Z"},"trusted":true},"execution_count":2,"outputs":[]},{"cell_type":"code","source":"# set ratio of weights by type (0=click, 1=cart, 2=order)\ntype_wt = {0:.5, 1:9, 2:.5}\nwt_mult = type_wt\n\n# chunk parameters\nREADS_PER = 5\nCHUNK = int( np.ceil( len(files)/6 ))\nprint(f'Processing {len(files)} files, {READS_PER} at a time, in chunks of {CHUNK}.')\n\n# use fewest disk pieces possible without causing memory error\nDISK_PIECES = 8\nSIZE = 1.86e6/DISK_PIECES","metadata":{"execution":{"iopub.status.busy":"2023-12-14T20:40:20.502851Z","iopub.execute_input":"2023-12-14T20:40:20.503197Z","iopub.status.idle":"2023-12-14T20:40:20.509849Z","shell.execute_reply.started":"2023-12-14T20:40:20.503171Z","shell.execute_reply":"2023-12-14T20:40:20.508801Z"},"trusted":true},"execution_count":3,"outputs":[{"name":"stdout","text":"Processing 146 files, 5 at a time, in chunks of 25.\n","output_type":"stream"}]},{"cell_type":"code","source":"def prep_for_cv(df, cv_type, drop=30, days=1):\n    \n    # sort by session and time stamp\n    df = df.sort_values(['session','ts'],ascending=[True,False])\n\n    # drop first # events of each session\n    df = df.reset_index(drop=True)\n    df['n'] = df.groupby('session').cumcount()\n    df = df.loc[df.n<drop].drop('n',axis=1)\n\n    # create pairs of events that occurred within 12 days\n    df = df.merge(df,on='session')\n    df = df.loc[ ((df.ts_x - df.ts_y).abs()< days * 24 * 60 * 60) & (df.aid_x != df.aid_y) ]\n    df = df.loc[(df.aid_x >= PART*SIZE)&(df.aid_x < (PART+1)*SIZE)]\n\n    # assign weights\n\n    \n    # set weights for co-visititation matrix type\n    match cv_type:\n        # cv1: type weighted co cv\n        case 1: \n            df = df[['session', 'aid_x', 'aid_y','type_y']].drop_duplicates(['session', 'aid_x', 'aid_y', 'type_y'])\n            df['wgt'] = df.type_y.map(type_wt)\n            df = df[['aid_x','aid_y','wgt']]\n        # cv2: time weighted click cv\n        case 2: \n            df = df[['session', 'aid_x', 'aid_y','ts_x']].drop_duplicates(['session', 'aid_x', 'aid_y'])\n            df['wgt'] = 1 + 3*(df.ts_x - 1659304800)/(1662328791-1659304800)\n            df = df[['aid_x','aid_y','wgt', 'ts_x']]\n        # cv3: buy to buy cv\n        case 3: \n            df['wgt'] = 1\n            df = df[['aid_x','aid_y','wgt']]\n        case _: \n            df['wgt'] = 1\n            df = df[['aid_x','aid_y','wgt']]\n    \n    df.wgt = df.wgt.astype('float32')\n    df = df.groupby(['aid_x','aid_y']).wgt.sum()\n    \n    return(df)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T20:56:12.023709Z","iopub.execute_input":"2023-12-14T20:56:12.024744Z","iopub.status.idle":"2023-12-14T20:56:12.035566Z","shell.execute_reply.started":"2023-12-14T20:56:12.024704Z","shell.execute_reply":"2023-12-14T20:56:12.034499Z"},"trusted":true},"execution_count":14,"outputs":[]},{"cell_type":"code","source":"%%time\n### Type-weighted carts/orders co-visitation matrix\n\n# keep top n cart aids\ncarts_n = 20\n\n# process data in parts\nfor PART in range(DISK_PIECES):\n    print('Part', PART+1, 'of', DISK_PIECES)\n\n    # outer chunks\n    for j in range(6):\n        a = j*CHUNK; b = min( (j+1)*CHUNK, len(files) )\n\n        # inner chunks\n        for k in range(a,b,READS_PER):\n            # read file from cache\n            df = [read_file(files[k])]\n            for i in range(1,READS_PER):\n                if k+i<b: df.append( read_file(files[k+i]) )\n            df = cudf.concat(df,ignore_index=True,axis=0)\n            df = prep_for_cv(df, cv_type=1, days=1)\n\n            # merge inner chunks\n            if k==a: tmp2 = df\n            else: tmp2 = tmp2.add(df, fill_value=0)\n            \n            del df\n            gc.collect()\n\n        # merge outer chunks\n        if a==0: tmp = tmp2\n        else: tmp = tmp.add(tmp2, fill_value=0)\n        del tmp2\n        gc.collect()\n\n    # convert for storage\n    tmp = tmp.reset_index()\n    tmp = tmp.sort_values(['aid_x','wgt'],ascending=[True,False])\n\n    # keep top results (by cumulative aid_y ccount with carts_n as cutline)\n    tmp = tmp.reset_index(drop=True)\n    tmp['n'] = tmp.groupby('aid_x').aid_y.cumcount()\n    tmp = tmp.loc[tmp.n<carts_n].drop('n',axis=1)\n\n    # convert to pandas and save as parquet\n    tmp.to_pandas().to_parquet(f'top_n_carts_orders_{PART}.pqt')","metadata":{"id":"PJioGgOw2YPF","outputId":"c2a50a05-da98-41cd-bf0f-cc9c82cf0ce1","execution":{"iopub.status.busy":"2023-12-14T20:40:25.657648Z","iopub.execute_input":"2023-12-14T20:40:25.658116Z","iopub.status.idle":"2023-12-14T20:45:46.089157Z","shell.execute_reply.started":"2023-12-14T20:40:25.658074Z","shell.execute_reply":"2023-12-14T20:45:46.088214Z"},"trusted":true},"execution_count":4,"outputs":[{"name":"stdout","text":"Part: 1 of 8\nPart: 2 of 8\nPart: 3 of 8\nPart: 4 of 8\nPart: 5 of 8\nPart: 6 of 8\nPart: 7 of 8\nPart: 8 of 8\nCPU times: user 3min 17s, sys: 2min 1s, total: 5min 19s\nWall time: 5min 20s\n","output_type":"stream"}]},{"cell_type":"code","source":"%%time\n### Cart/order co-visitation matrix given a user's previous cart/order\n\n# keep top n order aids\norders_n = 20\n\n# process data in parts\nfor PART in range(DISK_PIECES):\n    print('Part', PART+1, 'of', DISK_PIECES)\n\n    # outer chunks\n    for j in range(6):\n        a = j*CHUNK; b = min( (j+1)*CHUNK, len(files) )\n\n        # inner chunks\n        for k in range(a,b,READS_PER):\n\n            # read file from cache\n            df = [read_file(files[k])]\n            for i in range(1,READS_PER):\n                if k+i<b: df.append( read_file(files[k+i]) )\n            df = cudf.concat(df,ignore_index=True,axis=0)\n            df = df.loc[df['type'].isin([1,2])] # 1,2 is carts,orders\n            df = prep_for_cv(df, cv_type=3, days=14)\n\n            # merge inner\n            if k==a: tmp2 = df\n            else: tmp2 = tmp2.add(df, fill_value=0)\n            \n            del df\n            gc.collect()\n\n        # merge outer\n        if a==0: tmp = tmp2\n        else: tmp = tmp.add(tmp2, fill_value=0)\n            \n        del tmp2\n        gc.collect()\n\n    # convert for storage\n    tmp = tmp.reset_index()\n    tmp = tmp.sort_values(['aid_x','wgt'],ascending=[True,False])\n\n    # keep top results (by cumulative aid_y count with orders_n as cutline)\n    tmp = tmp.reset_index(drop=True)\n    tmp['n'] = tmp.groupby('aid_x').aid_y.cumcount()\n    tmp = tmp.loc[tmp.n<orders_n].drop('n',axis=1)\n\n    # convert to pandas and save as parquet\n    tmp.to_pandas().to_parquet(f'top_n_buy_buy_{PART}.pqt')","metadata":{"id":"W5shGKL-2YPF","outputId":"fdaaaf2d-ed75-4264-bffc-2c8b72085dd7","execution":{"iopub.status.busy":"2023-12-14T20:51:08.321544Z","iopub.execute_input":"2023-12-14T20:51:08.322454Z","iopub.status.idle":"2023-12-14T20:53:22.535975Z","shell.execute_reply.started":"2023-12-14T20:51:08.322418Z","shell.execute_reply":"2023-12-14T20:53:22.534959Z"},"trusted":true},"execution_count":12,"outputs":[{"name":"stdout","text":"Part 1 of 8\nPart 2 of 8\nPart 3 of 8\nPart 4 of 8\nPart 5 of 8\nPart 6 of 8\nPart 7 of 8\nPart 8 of 8\nCPU times: user 1min 40s, sys: 33.4 s, total: 2min 14s\nWall time: 2min 14s\n","output_type":"stream"}]},{"cell_type":"code","source":"%%time\n### Time-weighted co-visitation matrix for clicks\n\n# keep top n click aids\nclicks_n = 10\n\n# process in parts\nfor PART in range(DISK_PIECES):\n    print('Part', PART+1, 'of', DISK_PIECES)\n\n    # outer chunks\n    for j in range(6):\n        a = j*CHUNK\n        b = min( (j+1)*CHUNK, len(files) )\n\n        # inner chunks\n        for k in range(a,b,READS_PER):\n            \n            # read file from cache\n            df = [read_file(files[k])]\n            for i in range(1,READS_PER):\n                if k+i<b: df.append( read_file(files[k+i]) )\n            df = cudf.concat(df,ignore_index=True,axis=0)\n            df = df.sort_values(['session','ts'],ascending=[True,False])\n            \n            # drop first 30 events of each session\n            df = df.reset_index(drop=True)\n            df['n'] = df.groupby('session').cumcount()\n            df = df.loc[df.n<30].drop('n',axis=1)\n            \n            # create pairs of events that occurred on same day\n            df = df.merge(df,on='session')\n            df = df.loc[ ((df.ts_x - df.ts_y).abs()< 24 * 60 * 60) & (df.aid_x != df.aid_y) ]\n            df = df.loc[(df.aid_x >= PART*SIZE)&(df.aid_x < (PART+1)*SIZE)]\n\n            # assign weights by relative time proximity (min ts=1659304800, max ts=1662328791)\n            df = df[['session', 'aid_x', 'aid_y','ts_x']].drop_duplicates(['session', 'aid_x', 'aid_y'])\n            df['wgt'] = 1 + 3*(df.ts_x - 1659304800)/(1662328791-1659304800)\n            df = df[['aid_x','aid_y','wgt']]\n            df.wgt = df.wgt.astype('float32')\n            df = df.groupby(['aid_x','aid_y']).wgt.sum()\n\n            # merge inner chunks\n            if k==a: tmp2 = df\n            else: tmp2 = tmp2.add(df, fill_value=0)\n            \n            del df\n            gc.collect()\n\n        # merge outer chunks\n        if a==0: tmp = tmp2\n        else: tmp = tmp.add(tmp2, fill_value=0)\n            \n        del tmp2\n        gc.collect()\n\n    # convert for storage\n    tmp = tmp.reset_index()\n    tmp = tmp.sort_values(['aid_x','wgt'],ascending=[True,False])\n\n    # keep top results (by cumulative aid_y count with clicks_n as cutline)\n    tmp = tmp.reset_index(drop=True)\n    tmp['n'] = tmp.groupby('aid_x').aid_y.cumcount()\n    tmp = tmp.loc[tmp.n<clicks_n].drop('n',axis=1)\n\n    # convert to pandas and save as parquet\n    tmp.to_pandas().to_parquet(f'top_n_clicks_{PART}.pqt')","metadata":{"id":"LqUn4mty2YPG","outputId":"f2021511-89eb-488c-b610-4f28be4f5ae7","execution":{"iopub.status.busy":"2023-12-14T20:56:24.962026Z","iopub.execute_input":"2023-12-14T20:56:24.962533Z","iopub.status.idle":"2023-12-14T20:56:25.366837Z","shell.execute_reply.started":"2023-12-14T20:56:24.962494Z","shell.execute_reply":"2023-12-14T20:56:25.365903Z"},"trusted":true},"execution_count":15,"outputs":[{"name":"stdout","text":"Part 1 of 8\n","output_type":"stream"},{"traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mKeyError\u001b[0m                                  Traceback (most recent call last)","File \u001b[0;32m/opt/conda/lib/python3.10/site-packages/cudf/utils/utils.py:230\u001b[0m, in \u001b[0;36mGetAttrGetItemMixin.__getattr__\u001b[0;34m(self, key)\u001b[0m\n\u001b[1;32m    229\u001b[0m \u001b[38;5;28;01mtry\u001b[39;00m:\n\u001b[0;32m--> 230\u001b[0m     \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[38;5;28;43mself\u001b[39;49m\u001b[43m[\u001b[49m\u001b[43mkey\u001b[49m\u001b[43m]\u001b[49m\n\u001b[1;32m    231\u001b[0m \u001b[38;5;28;01mexcept\u001b[39;00m \u001b[38;5;167;01mKeyError\u001b[39;00m:\n","File \u001b[0;32m/opt/conda/lib/python3.10/site-packages/nvtx/nvtx.py:115\u001b[0m, in \u001b[0;36mannotate.__call__.<locals>.inner\u001b[0;34m(*args, **kwargs)\u001b[0m\n\u001b[1;32m    114\u001b[0m libnvtx_push_range(\u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mattributes, \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mdomain\u001b[38;5;241m.\u001b[39mhandle)\n\u001b[0;32m--> 115\u001b[0m result \u001b[38;5;241m=\u001b[39m \u001b[43mfunc\u001b[49m\u001b[43m(\u001b[49m\u001b[38;5;241;43m*\u001b[39;49m\u001b[43margs\u001b[49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[38;5;241;43m*\u001b[39;49m\u001b[38;5;241;43m*\u001b[39;49m\u001b[43mkwargs\u001b[49m\u001b[43m)\u001b[49m\n\u001b[1;32m    116\u001b[0m libnvtx_pop_range(\u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mdomain\u001b[38;5;241m.\u001b[39mhandle)\n","File \u001b[0;32m/opt/conda/lib/python3.10/site-packages/cudf/core/dataframe.py:1159\u001b[0m, in \u001b[0;36mDataFrame.__getitem__\u001b[0;34m(self, arg)\u001b[0m\n\u001b[1;32m   1158\u001b[0m \u001b[38;5;28;01mif\u001b[39;00m _is_scalar_or_zero_d_array(arg) \u001b[38;5;129;01mor\u001b[39;00m \u001b[38;5;28misinstance\u001b[39m(arg, \u001b[38;5;28mtuple\u001b[39m):\n\u001b[0;32m-> 1159\u001b[0m     \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[38;5;28;43mself\u001b[39;49m\u001b[38;5;241;43m.\u001b[39;49m\u001b[43m_get_columns_by_label\u001b[49m\u001b[43m(\u001b[49m\u001b[43marg\u001b[49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[43mdowncast\u001b[49m\u001b[38;5;241;43m=\u001b[39;49m\u001b[38;5;28;43;01mTrue\u001b[39;49;00m\u001b[43m)\u001b[49m\n\u001b[1;32m   1161\u001b[0m \u001b[38;5;28;01melif\u001b[39;00m \u001b[38;5;28misinstance\u001b[39m(arg, \u001b[38;5;28mslice\u001b[39m):\n","File \u001b[0;32m/opt/conda/lib/python3.10/site-packages/nvtx/nvtx.py:115\u001b[0m, in \u001b[0;36mannotate.__call__.<locals>.inner\u001b[0;34m(*args, **kwargs)\u001b[0m\n\u001b[1;32m    114\u001b[0m libnvtx_push_range(\u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mattributes, \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mdomain\u001b[38;5;241m.\u001b[39mhandle)\n\u001b[0;32m--> 115\u001b[0m result \u001b[38;5;241m=\u001b[39m \u001b[43mfunc\u001b[49m\u001b[43m(\u001b[49m\u001b[38;5;241;43m*\u001b[39;49m\u001b[43margs\u001b[49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[38;5;241;43m*\u001b[39;49m\u001b[38;5;241;43m*\u001b[39;49m\u001b[43mkwargs\u001b[49m\u001b[43m)\u001b[49m\n\u001b[1;32m    116\u001b[0m libnvtx_pop_range(\u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mdomain\u001b[38;5;241m.\u001b[39mhandle)\n","File \u001b[0;32m/opt/conda/lib/python3.10/site-packages/cudf/core/dataframe.py:1792\u001b[0m, in \u001b[0;36mDataFrame._get_columns_by_label\u001b[0;34m(self, labels, downcast)\u001b[0m\n\u001b[1;32m   1787\u001b[0m \u001b[38;5;250m\u001b[39m\u001b[38;5;124;03m\"\"\"\u001b[39;00m\n\u001b[1;32m   1788\u001b[0m \u001b[38;5;124;03mReturn columns of dataframe by `labels`\u001b[39;00m\n\u001b[1;32m   1789\u001b[0m \n\u001b[1;32m   1790\u001b[0m \u001b[38;5;124;03mIf downcast is True, try and downcast from a DataFrame to a Series\u001b[39;00m\n\u001b[1;32m   1791\u001b[0m \u001b[38;5;124;03m\"\"\"\u001b[39;00m\n\u001b[0;32m-> 1792\u001b[0m new_data \u001b[38;5;241m=\u001b[39m \u001b[38;5;28;43msuper\u001b[39;49m\u001b[43m(\u001b[49m\u001b[43m)\u001b[49m\u001b[38;5;241;43m.\u001b[39;49m\u001b[43m_get_columns_by_label\u001b[49m\u001b[43m(\u001b[49m\u001b[43mlabels\u001b[49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[43mdowncast\u001b[49m\u001b[43m)\u001b[49m\n\u001b[1;32m   1793\u001b[0m \u001b[38;5;28;01mif\u001b[39;00m downcast:\n","File \u001b[0;32m/opt/conda/lib/python3.10/site-packages/nvtx/nvtx.py:115\u001b[0m, in \u001b[0;36mannotate.__call__.<locals>.inner\u001b[0;34m(*args, **kwargs)\u001b[0m\n\u001b[1;32m    114\u001b[0m libnvtx_push_range(\u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mattributes, \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mdomain\u001b[38;5;241m.\u001b[39mhandle)\n\u001b[0;32m--> 115\u001b[0m result \u001b[38;5;241m=\u001b[39m \u001b[43mfunc\u001b[49m\u001b[43m(\u001b[49m\u001b[38;5;241;43m*\u001b[39;49m\u001b[43margs\u001b[49m\u001b[43m,\u001b[49m\u001b[43m \u001b[49m\u001b[38;5;241;43m*\u001b[39;49m\u001b[38;5;241;43m*\u001b[39;49m\u001b[43mkwargs\u001b[49m\u001b[43m)\u001b[49m\n\u001b[1;32m    116\u001b[0m libnvtx_pop_range(\u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mdomain\u001b[38;5;241m.\u001b[39mhandle)\n","File \u001b[0;32m/opt/conda/lib/python3.10/site-packages/cudf/core/frame.py:425\u001b[0m, in \u001b[0;36mFrame._get_columns_by_label\u001b[0;34m(self, labels, downcast)\u001b[0m\n\u001b[1;32m    421\u001b[0m \u001b[38;5;250m\u001b[39m\u001b[38;5;124;03m\"\"\"\u001b[39;00m\n\u001b[1;32m    422\u001b[0m \u001b[38;5;124;03mReturns columns of the Frame specified by `labels`\u001b[39;00m\n\u001b[1;32m    423\u001b[0m \n\u001b[1;32m    424\u001b[0m \u001b[38;5;124;03m\"\"\"\u001b[39;00m\n\u001b[0;32m--> 425\u001b[0m \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[38;5;28;43mself\u001b[39;49m\u001b[38;5;241;43m.\u001b[39;49m\u001b[43m_data\u001b[49m\u001b[38;5;241;43m.\u001b[39;49m\u001b[43mselect_by_label\u001b[49m\u001b[43m(\u001b[49m\u001b[43mlabels\u001b[49m\u001b[43m)\u001b[49m\n","File \u001b[0;32m/opt/conda/lib/python3.10/site-packages/cudf/core/column_accessor.py:357\u001b[0m, in \u001b[0;36mColumnAccessor.select_by_label\u001b[0;34m(self, key)\u001b[0m\n\u001b[1;32m    356\u001b[0m         \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39m_select_by_label_with_wildcard(key)\n\u001b[0;32m--> 357\u001b[0m \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[38;5;28;43mself\u001b[39;49m\u001b[38;5;241;43m.\u001b[39;49m\u001b[43m_select_by_label_grouped\u001b[49m\u001b[43m(\u001b[49m\u001b[43mkey\u001b[49m\u001b[43m)\u001b[49m\n","File \u001b[0;32m/opt/conda/lib/python3.10/site-packages/cudf/core/column_accessor.py:512\u001b[0m, in \u001b[0;36mColumnAccessor._select_by_label_grouped\u001b[0;34m(self, key)\u001b[0m\n\u001b[1;32m    511\u001b[0m \u001b[38;5;28;01mdef\u001b[39;00m \u001b[38;5;21m_select_by_label_grouped\u001b[39m(\u001b[38;5;28mself\u001b[39m, key: Any) \u001b[38;5;241m-\u001b[39m\u001b[38;5;241m>\u001b[39m ColumnAccessor:\n\u001b[0;32m--> 512\u001b[0m     result \u001b[38;5;241m=\u001b[39m \u001b[38;5;28;43mself\u001b[39;49m\u001b[38;5;241;43m.\u001b[39;49m\u001b[43m_grouped_data\u001b[49m\u001b[43m[\u001b[49m\u001b[43mkey\u001b[49m\u001b[43m]\u001b[49m\n\u001b[1;32m    513\u001b[0m     \u001b[38;5;28;01mif\u001b[39;00m \u001b[38;5;28misinstance\u001b[39m(result, cudf\u001b[38;5;241m.\u001b[39mcore\u001b[38;5;241m.\u001b[39mcolumn\u001b[38;5;241m.\u001b[39mColumnBase):\n","\u001b[0;31mKeyError\u001b[0m: 'ts_x'","\nDuring handling of the above exception, another exception occurred:\n","\u001b[0;31mAttributeError\u001b[0m                            Traceback (most recent call last)","File \u001b[0;32m<timed exec>:23\u001b[0m\n","Cell \u001b[0;32mIn[14], line 27\u001b[0m, in \u001b[0;36mprep_for_cv\u001b[0;34m(df, cv_type, drop, days)\u001b[0m\n\u001b[1;32m     25\u001b[0m \u001b[38;5;66;03m# cv2: time weighted click cv\u001b[39;00m\n\u001b[1;32m     26\u001b[0m \u001b[38;5;28;01mcase\u001b[39;00m \u001b[38;5;241m2\u001b[39m: \n\u001b[0;32m---> 27\u001b[0m     df[\u001b[38;5;124m'\u001b[39m\u001b[38;5;124mwgt\u001b[39m\u001b[38;5;124m'\u001b[39m] \u001b[38;5;241m=\u001b[39m \u001b[38;5;241m1\u001b[39m \u001b[38;5;241m+\u001b[39m \u001b[38;5;241m3\u001b[39m\u001b[38;5;241m*\u001b[39m(\u001b[43mdf\u001b[49m\u001b[38;5;241;43m.\u001b[39;49m\u001b[43mts_x\u001b[49m \u001b[38;5;241m-\u001b[39m \u001b[38;5;241m1659304800\u001b[39m)\u001b[38;5;241m/\u001b[39m(\u001b[38;5;241m1662328791\u001b[39m\u001b[38;5;241m-\u001b[39m\u001b[38;5;241m1659304800\u001b[39m)\n\u001b[1;32m     28\u001b[0m     df \u001b[38;5;241m=\u001b[39m df[[\u001b[38;5;124m'\u001b[39m\u001b[38;5;124maid_x\u001b[39m\u001b[38;5;124m'\u001b[39m,\u001b[38;5;124m'\u001b[39m\u001b[38;5;124maid_y\u001b[39m\u001b[38;5;124m'\u001b[39m,\u001b[38;5;124m'\u001b[39m\u001b[38;5;124mwgt\u001b[39m\u001b[38;5;124m'\u001b[39m, \u001b[38;5;124m'\u001b[39m\u001b[38;5;124mts_x\u001b[39m\u001b[38;5;124m'\u001b[39m]]\n\u001b[1;32m     29\u001b[0m \u001b[38;5;66;03m# cv3: buy to buy cv\u001b[39;00m\n","File \u001b[0;32m/opt/conda/lib/python3.10/site-packages/cudf/utils/utils.py:232\u001b[0m, in \u001b[0;36mGetAttrGetItemMixin.__getattr__\u001b[0;34m(self, key)\u001b[0m\n\u001b[1;32m    230\u001b[0m     \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[38;5;28mself\u001b[39m[key]\n\u001b[1;32m    231\u001b[0m \u001b[38;5;28;01mexcept\u001b[39;00m \u001b[38;5;167;01mKeyError\u001b[39;00m:\n\u001b[0;32m--> 232\u001b[0m     \u001b[38;5;28;01mraise\u001b[39;00m \u001b[38;5;167;01mAttributeError\u001b[39;00m(\n\u001b[1;32m    233\u001b[0m         \u001b[38;5;124mf\u001b[39m\u001b[38;5;124m\"\u001b[39m\u001b[38;5;132;01m{\u001b[39;00m\u001b[38;5;28mtype\u001b[39m(\u001b[38;5;28mself\u001b[39m)\u001b[38;5;241m.\u001b[39m\u001b[38;5;18m__name__\u001b[39m\u001b[38;5;132;01m}\u001b[39;00m\u001b[38;5;124m object has no attribute \u001b[39m\u001b[38;5;132;01m{\u001b[39;00mkey\u001b[38;5;132;01m}\u001b[39;00m\u001b[38;5;124m\"\u001b[39m\n\u001b[1;32m    234\u001b[0m     )\n","\u001b[0;31mAttributeError\u001b[0m: DataFrame object has no attribute ts_x"],"ename":"AttributeError","evalue":"DataFrame object has no attribute ts_x","output_type":"error"}]},{"cell_type":"code","source":"# release memory\ndel data_cache, tmp\n_ = gc.collect()","metadata":{"id":"1rNuEv4B2YPH","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-12-14T01:18:54.204155Z","iopub.execute_input":"2023-12-14T01:18:54.204569Z","iopub.status.idle":"2023-12-14T01:18:54.338701Z","shell.execute_reply.started":"2023-12-14T01:18:54.204511Z","shell.execute_reply":"2023-12-14T01:18:54.337427Z"},"trusted":true},"execution_count":9,"outputs":[]},{"cell_type":"code","source":"# load test data\ndef load_test():\n    dfs = []\n    for e, chunk_file in enumerate(glob.glob('../input/otto-chunk-data-inparquet-format/test_parquet/*')):\n        chunk = pd.read_parquet(chunk_file)\n        chunk.ts = (chunk.ts/1000).astype('int32')\n        chunk['type'] = chunk['type'].map(type_labels).astype('int8')\n        dfs.append(chunk)\n    return pd.concat(dfs).reset_index(drop=True) #.astype({\"ts\": \"datetime64[ms]\"})\n\ntest_df = load_test()\nprint('Test data dimensions:',test_df.shape)\ntest_df.head()","metadata":{"id":"gpY-Ac5y2YPH","outputId":"b0811e83-7887-44ef-dfc8-34a2add75710","execution":{"iopub.status.busy":"2023-12-14T01:18:54.343774Z","iopub.execute_input":"2023-12-14T01:18:54.344089Z","iopub.status.idle":"2023-12-14T01:18:56.018228Z","shell.execute_reply.started":"2023-12-14T01:18:54.344064Z","shell.execute_reply":"2023-12-14T01:18:56.017138Z"},"trusted":true},"execution_count":10,"outputs":[{"name":"stdout","text":"Test data dimensions: (6928123, 4)\n","output_type":"stream"},{"execution_count":10,"output_type":"execute_result","data":{"text/plain":"    session     aid          ts  type\n0  13099779  245308  1661795832     0\n1  13099779  245308  1661795862     1\n2  13099779  972319  1661795888     0\n3  13099779  972319  1661795898     1\n4  13099779  245308  1661795907     0","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>session</th>\n      <th>aid</th>\n      <th>ts</th>\n      <th>type</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>13099779</td>\n      <td>245308</td>\n      <td>1661795832</td>\n      <td>0</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>13099779</td>\n      <td>245308</td>\n      <td>1661795862</td>\n      <td>1</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>13099779</td>\n      <td>972319</td>\n      <td>1661795888</td>\n      <td>0</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>13099779</td>\n      <td>972319</td>\n      <td>1661795898</td>\n      <td>1</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>13099779</td>\n      <td>245308</td>\n      <td>1661795907</td>\n      <td>0</td>\n    </tr>\n  </tbody>\n</table>\n</div>"},"metadata":{}}]},{"cell_type":"code","source":"def load_comp_test():\n    dfs = []\n    for e, chunk_file in enumerate(glob.glob('../input/test-data-as-parquet/*')):\n        chunk = pd.read_parquet(chunk_file)\n        chunk.ts = (chunk.ts/1000).astype('int32')\n        chunk['type'] = chunk['type'].map(type_labels).astype('int8')\n        dfs.append(chunk)\n    return pd.concat(dfs).reset_index(drop=True) #.astype({\"ts\": \"datetime64[ms]\"})\n\ncomp_test_df = load_comp_test()\nprint('Complete test data dimensions:',comp_test_df.shape)\ncomp_test_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T01:18:56.019761Z","iopub.execute_input":"2023-12-14T01:18:56.02044Z","iopub.status.idle":"2023-12-14T01:18:59.046318Z","shell.execute_reply.started":"2023-12-14T01:18:56.020398Z","shell.execute_reply":"2023-12-14T01:18:59.045255Z"},"trusted":true},"execution_count":11,"outputs":[{"name":"stdout","text":"Complete test data dimensions: (13851293, 4)\n","output_type":"stream"},{"execution_count":11,"output_type":"execute_result","data":{"text/plain":"    session      aid          ts  type\n0  12899779    59625  1661724000     0\n1  12899779   875854  1661724026     0\n2  12899780  1142000  1661724000     0\n3  12899780   582732  1661724058     0\n4  12899780   973453  1661724109     0","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>session</th>\n      <th>aid</th>\n      <th>ts</th>\n      <th>type</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>12899779</td>\n      <td>59625</td>\n      <td>1661724000</td>\n      <td>0</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>12899779</td>\n      <td>875854</td>\n      <td>1661724026</td>\n      <td>0</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>12899780</td>\n      <td>1142000</td>\n      <td>1661724000</td>\n      <td>0</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>12899780</td>\n      <td>582732</td>\n      <td>1661724058</td>\n      <td>0</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>12899780</td>\n      <td>973453</td>\n      <td>1661724109</td>\n      <td>0</td>\n    </tr>\n  </tbody>\n</table>\n</div>"},"metadata":{}}]},{"cell_type":"code","source":"%%time\n### load most frequent click, cart, order aids for custom rules to adjust recommendation rankings\n\n# convert df from parquet file to dictionary\ndef pqt_to_dict(df):\n    return df.groupby('aid_x').aid_y.apply(list).to_dict()\n\n# load 0th co-visitation matrices\ntop_n_clicks = pqt_to_dict( pd.read_parquet(f'top_n_clicks_0.pqt') )\ntop_n_buys = pqt_to_dict( pd.read_parquet(f'top_n_carts_orders_0.pqt') )\ntop_n_buy_buy = pqt_to_dict( pd.read_parquet(f'top_n_buy_buy_0.pqt') )\n\n# update 0th co-visitation matrices with each successive iteration\nfor k in range(1,DISK_PIECES):\n    top_n_clicks.update( pqt_to_dict( pd.read_parquet(f'top_n_clicks_{k}.pqt') ) )\n    top_n_buys.update( pqt_to_dict( pd.read_parquet(f'top_n_carts_orders_{k}.pqt') ) )\n    top_n_buy_buy.update( pqt_to_dict( pd.read_parquet(f'top_n_buy_buy_{k}.pqt') ) )","metadata":{"id":"1MGOKloG2YPI","outputId":"6758654b-5c15-44aa-ebee-e02582dc497e","execution":{"iopub.status.busy":"2023-12-14T01:18:59.04762Z","iopub.execute_input":"2023-12-14T01:18:59.04795Z","iopub.status.idle":"2023-12-14T01:22:12.700939Z","shell.execute_reply.started":"2023-12-14T01:18:59.04792Z","shell.execute_reply":"2023-12-14T01:22:12.699904Z"},"trusted":true},"execution_count":12,"outputs":[{"name":"stdout","text":"CPU times: user 3min 14s, sys: 5.27 s, total: 3min 19s\nWall time: 3min 13s\n","output_type":"stream"}]},{"cell_type":"code","source":"# most frequent clicks and orders in test data\ntop_clicks = test_df.loc[test_df['type']== 0,'aid'].value_counts().index.values[:20]\ntop_carts = test_df.loc[test_df['type']== 1,'aid'].value_counts().index.values[:20]\ntop_orders = test_df.loc[test_df['type']== 2,'aid'].value_counts().index.values[:20]\nlen(set(list(top_clicks)+list(top_carts)+list(top_orders)))","metadata":{"id":"F3jrY4iY2YPI","execution":{"iopub.status.busy":"2023-12-14T01:22:12.70245Z","iopub.execute_input":"2023-12-14T01:22:12.703102Z","iopub.status.idle":"2023-12-14T01:22:13.178267Z","shell.execute_reply.started":"2023-12-14T01:22:12.70306Z","shell.execute_reply":"2023-12-14T01:22:13.177202Z"},"trusted":true},"execution_count":13,"outputs":[{"execution_count":13,"output_type":"execute_result","data":{"text/plain":"36"},"metadata":{}}]},{"cell_type":"code","source":"def suggest_clicks(df):\n    # user history by items and events\n    aids=df.aid.tolist()\n    types = df.type.tolist()\n    unique_aids = list(dict.fromkeys(aids[::-1] ))\n    \n    # adjust recommendations by weights\n    if len(unique_aids)>=20:\n        weights=np.logspace(0.1,1,len(aids),base=2, endpoint=True)-1\n        aids_temp = Counter()\n        \n        # adjust recommendations by user history\n        for aid,w,t in zip(aids,weights,types):\n            aids_temp[aid] += w * wt_mult[t]\n        sorted_aids = [k for k,v in aids_temp.most_common(20)]\n        return sorted_aids\n\n    # pad recommendations with most common clicks in training data if not in user history\n    aids2 = list(itertools.chain(*[top_n_clicks[aid] for aid in unique_aids if aid in top_clicks]))\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids2).most_common(20) if aid2 not in unique_aids]\n    result = unique_aids + top_aids2[:20 - len(unique_aids)]\n    \n    # pad recommendations with most common clicks in current week\n    return result + list(top_clicks)[:20-len(result)]","metadata":{"id":"BGBxxDHe2YPI","execution":{"iopub.status.busy":"2023-12-14T01:22:13.179646Z","iopub.execute_input":"2023-12-14T01:22:13.179988Z","iopub.status.idle":"2023-12-14T01:22:13.190185Z","shell.execute_reply.started":"2023-12-14T01:22:13.179959Z","shell.execute_reply":"2023-12-14T01:22:13.189035Z"},"trusted":true},"execution_count":14,"outputs":[]},{"cell_type":"code","source":"def suggest_carts(df):\n    # user history by items and events\n    aids = df.aid.tolist()\n    types = df.type.tolist()\n    unique_aids = list(dict.fromkeys(aids[::-1] ))\n    df = df.loc[(df['type'] == 0)|(df['type'] == 1)]\n    unique_buys = list(dict.fromkeys(df.aid.tolist()[::-1]))\n\n    # adjust recommendations by weights\n    if len(unique_aids) >= 20:\n        weights=np.logspace(0.5,1,len(aids),base=2, endpoint=True)-1\n        aids_temp = Counter()\n\n        # adjust recommendations by user history\n        for aid,w,t in zip(aids,weights,types):\n            aids_temp[aid] += w * wt_mult[t]\n\n        # pad recommendations with most frequent orders in training data given purchases in user history\n        aids2 = list(itertools.chain(*[top_n_buys[aid] for aid in unique_buys if aid in top_orders]))\n        for aid in aids2: aids_temp[aid] += 0.1\n        sorted_aids = [k for k,v in aids_temp.most_common(20)]\n        return sorted_aids\n\n    # pad recommendations with most frequent clicks and orders in training data given user history\n    aids1 = list(itertools.chain(*[top_n_clicks[aid] for aid in unique_aids if aid in top_clicks]))\n    aids2 = list(itertools.chain(*[top_n_buys[aid] for aid in unique_aids if aid in top_orders]))\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids1+aids2).most_common(20) if aid2 not in unique_aids]\n    result = unique_aids + top_aids2[:20 - len(unique_aids)]\n\n    # pad recommendations with most common carts in current week\n    return result + list(top_carts)[:20-len(result)]","metadata":{"id":"_HFDWhUE2YPJ","execution":{"iopub.status.busy":"2023-12-14T01:22:13.191705Z","iopub.execute_input":"2023-12-14T01:22:13.192147Z","iopub.status.idle":"2023-12-14T01:22:13.208244Z","shell.execute_reply.started":"2023-12-14T01:22:13.192063Z","shell.execute_reply":"2023-12-14T01:22:13.207282Z"},"trusted":true},"execution_count":15,"outputs":[]},{"cell_type":"code","source":"def suggest_buys(df):\n    # user history by items and events\n    aids=df.aid.tolist()\n    types = df.type.tolist()\n    unique_aids = list(dict.fromkeys(aids[::-1] ))\n    #u_aids = list(set(aids[::-1]))\n    df = df.loc[(df['type']==1)|(df['type']==2)]\n    #df2 = df.loc[df['type'] in [1,2]]\n    unique_buys = list(dict.fromkeys( df.aid.tolist()[::-1] ))\n    #u_buys = list(set(df.aid.tolist()[::-1]))\n\n    # adjust recommendations by weights\n    if len(unique_aids)>=20:\n        weights=np.logspace(0.5,1,len(aids),base=2, endpoint=True)-1\n        aids_temp = Counter()\n\n        # adjust recommendations by user history\n        for aid,w,t in zip(aids,weights,types):\n            aids_temp[aid] += w * wt_mult[t]\n        \n        aids3 = list(itertools.chain(*[top_n_buy_buy[aid] for aid in unique_buys if aid in top_n_buy_buy]))\n        for aid in aids3: aids_temp[aid] += 0.1\n        sorted_aids = [k for k,v in aids_temp.most_common(20)]\n        return sorted_aids\n    \n    # pad recommendations with most common orders in training data given user history\n    aids2 = list(itertools.chain(*[top_n_buys[aid] for aid in unique_aids if aid in top_n_buys]))\n    aids3 = list(itertools.chain(*[top_n_buy_buy[aid] for aid in unique_buys if aid in top_n_buy_buy]))\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids2+aids3).most_common(20) if aid2 not in unique_aids]\n    result = unique_aids + top_aids2[:20 - len(unique_aids)]\n\n    # pad recommendations with most common orders in current week\n    return result + list(top_orders)[:20-len(result)]","metadata":{"id":"il8dPBsu2YPJ","execution":{"iopub.status.busy":"2023-12-14T01:22:13.20965Z","iopub.execute_input":"2023-12-14T01:22:13.210036Z","iopub.status.idle":"2023-12-14T01:22:13.224328Z","shell.execute_reply.started":"2023-12-14T01:22:13.209999Z","shell.execute_reply":"2023-12-14T01:22:13.223126Z"},"trusted":true},"execution_count":16,"outputs":[]},{"cell_type":"code","source":"%%time\n### truncate test data sessions to test predictions (~5 min @ len / 30)\nimport random\n\ntrunc_test_df = pd.DataFrame()\nsolution_df = pd.DataFrame()\n\n# Get list of sesssions in test_df\ntest_sessions = list(test_df['session'].unique())\nsession_subset = random.sample(range(len(test_sessions)), round(len(test_sessions)/10))\n\n# randomly truncate sessions placing earlier data in trunc_test_df and remainder in solution_df\nfor session in session_subset:\n    \n    session_df = test_df[test_df['session']==test_sessions[session]]\n    if len(session_df['aid']) < 22: continue\n    \n    split = random.randint(2, len(session_df['aid'])-20)\n    trunc_test_df = pd.concat([trunc_test_df, session_df[0:split]])\n    solution_df = pd.concat([solution_df, session_df[split:len(session_df['aid'])]])\n    \nsolution_df.reset_index(inplace=True)\nsolution_df.sort_values(by=[\"session\", \"ts\"], axis=0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T01:22:13.225653Z","iopub.execute_input":"2023-12-14T01:22:13.225984Z","iopub.status.idle":"2023-12-14T01:38:35.255531Z","shell.execute_reply.started":"2023-12-14T01:22:13.225958Z","shell.execute_reply":"2023-12-14T01:38:35.25461Z"},"trusted":true},"execution_count":17,"outputs":[{"name":"stdout","text":"CPU times: user 38min 5s, sys: 25.3 s, total: 38min 30s\nWall time: 16min 22s\n","output_type":"stream"}]},{"cell_type":"code","source":"%%time\n# creating recommendations for truncated test data (<30 sec. @ len / 10)\npred_clicks = trunc_test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_clicks(x))\n\npred_carts = trunc_test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_carts(x))\n\npred_buys = trunc_test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_buys(x))","metadata":{"execution":{"iopub.status.busy":"2023-12-14T07:22:14.722831Z","iopub.execute_input":"2023-12-14T07:22:14.723547Z","iopub.status.idle":"2023-12-14T07:22:15.687327Z","shell.execute_reply.started":"2023-12-14T07:22:14.723505Z","shell.execute_reply":"2023-12-14T07:22:15.686211Z"},"trusted":true},"execution_count":59,"outputs":[{"name":"stdout","text":"CPU times: user 954 ms, sys: 5.97 ms, total: 959 ms\nWall time: 958 ms\n","output_type":"stream"}]},{"cell_type":"code","source":"# compare predictions for truncated test df to solution df for truncated test data (<30 sec. @ len / 10)\nclick_matches=0\ncart_matches=0\norder_matches=0\nnum_carts=0\nnum_orders=0\n\nfor i in range(len(pred_clicks.index)):\n    session = pred_clicks.index[i]\n    session_df = solution_df[solution_df['session']==session]\n    # count correct click predictions\n    if 0 in list(session_df['type']):\n        clicks_df = session_df[session_df['type']==0]\n        if clicks_df.aid[:1].item() in pred_clicks[session]:\n            click_matches += 1\n    # count correct cart predictions\n    if 1 in list(session_df['type']):\n        carts_df = session_df[session_df['type']==1]\n        for item in carts_df.aid:\n            if item in pred_carts[session]:\n                cart_matches += 1\n                num_carts += min(20, len(carts_df.aid))\n    # count correct order predictions\n    if 2 in list(session_df['type']):\n        orders_df = session_df[session_df['type']==2]\n        for item in orders_df.aid:\n            if item in pred_buys[pred_clicks.index[i]]:\n                order_matches += 1\n                num_orders += min(20, len(orders_df.aid))\n            \nprint(click_matches/len(pred_clicks))\nprint(cart_matches/num_carts)\nprint(order_matches/num_orders)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T01:38:44.68772Z","iopub.execute_input":"2023-12-14T01:38:44.68803Z","iopub.status.idle":"2023-12-14T01:38:51.963316Z","shell.execute_reply.started":"2023-12-14T01:38:44.688004Z","shell.execute_reply":"2023-12-14T01:38:51.962193Z"},"trusted":true},"execution_count":20,"outputs":[{"name":"stdout","text":"0.28735135135135137\n0.10899158935229342\n0.18148450244698205\n","output_type":"stream"}]},{"cell_type":"code","source":"%%time\n### open complete test data, merge with test df, drop duplicated rows, reset index and sort\n\nsolution_df = pd.concat([test_df, comp_test_df])\nsolution_df = solution_df[solution_df.duplicated(keep=False)]\nsolution_df.reset_index(inplace=True)\nsolution_df.sort_values(by=[\"session\", \"ts\"], axis=0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T01:38:51.964769Z","iopub.execute_input":"2023-12-14T01:38:51.965114Z","iopub.status.idle":"2023-12-14T01:39:03.240157Z","shell.execute_reply.started":"2023-12-14T01:38:51.965083Z","shell.execute_reply":"2023-12-14T01:39:03.239075Z"},"trusted":true},"execution_count":21,"outputs":[{"name":"stdout","text":"CPU times: user 9.77 s, sys: 1.5 s, total: 11.3 s\nWall time: 11.3 s\n","output_type":"stream"}]},{"cell_type":"code","source":"%%time\n# creating recommendations for test data is slow (~45 min.)\npred_clicks = test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_clicks(x))\n\npred_carts = test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_carts(x))\n\npred_buys = test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_buys(x))","metadata":{"id":"t5iW7N1O2YPJ","outputId":"48547adf-4e3e-493e-d28f-7cc2c7f7a411","execution":{"iopub.status.busy":"2023-12-14T07:23:39.464626Z","iopub.execute_input":"2023-12-14T07:23:39.465039Z","iopub.status.idle":"2023-12-14T08:17:11.561902Z","shell.execute_reply.started":"2023-12-14T07:23:39.465004Z","shell.execute_reply":"2023-12-14T08:17:11.560617Z"},"trusted":true},"execution_count":65,"outputs":[{"name":"stdout","text":"CPU times: user 53min 26s, sys: 7.75 s, total: 53min 33s\nWall time: 53min 32s\n","output_type":"stream"}]},{"cell_type":"code","source":"# compare predictions on test df to solution df (<30 sec. @ len / 10)\nclick_matches=0\ncart_matches=0\norder_matches=0\nnum_carts=0\nnum_orders=0\n\nfor i in range(len(pred_clicks.index)):\n    session = pred_clicks.index[i]\n    session_df = solution_df[solution_df['session']==session]\n    # count correct click predictions\n    if 0 in list(session_df['type']):\n        clicks_df = session_df[session_df['type']==0]\n        if clicks_df.aid[:1].item() in pred_clicks[session]:\n            click_matches += 1\n    # count correct cart predictions\n    if 1 in list(session_df['type']):\n        carts_df = session_df[session_df['type']==1]\n        for item in carts_df.aid:\n            if item in pred_carts[session]:\n                cart_matches += 1\n                num_carts += min(20, len(carts_df.aid))\n    # count correct order predictions\n    if 2 in list(session_df['type']):\n        orders_df = session_df[session_df['type']==2]\n        for item in orders_df.aid:\n            if item in pred_buys[pred_clicks.index[i]]:\n                order_matches += 1\n                num_orders += min(20, len(orders_df.aid))\n    if i % 100000 == 0: print(f\"{i} of {len(pred_clicks.index)} complete\")\n            \nprint(click_matches/len(pred_clicks))\nprint(cart_matches/num_carts)\nprint(order_matches/num_orders)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"click_matches","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clicks_pred = pd.DataFrame(pred_clicks.add_suffix(\"_clicks\"), columns=[\"labels\"]).reset_index()\norders_pred = pd.DataFrame(pred_buys.add_suffix(\"_orders\"), columns=[\"labels\"]).reset_index()\ncarts_pred = pd.DataFrame(pred_carts.add_suffix(\"_carts\"), columns=[\"labels\"]).reset_index()","metadata":{"id":"gOZ2OpZo2YPK","execution":{"iopub.status.busy":"2023-12-14T07:10:37.088703Z","iopub.status.idle":"2023-12-14T07:10:37.089049Z","shell.execute_reply.started":"2023-12-14T07:10:37.08888Z","shell.execute_reply":"2023-12-14T07:10:37.088897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.concat([clicks_pred, orders_pred, carts_pred])\npred_df.columns = [\"session_type\", \"labels\"]\npred_df[\"labels\"] = pred_df.labels.apply(lambda x: \" \".join(map(str,x)))\npred_df.to_csv(\"submission.csv\", index=False)\npred_df.head()","metadata":{"id":"7nCjOUfI2YPK","outputId":"0461e834-788b-4721-9ea5-e9cb27e0f372","execution":{"iopub.status.busy":"2023-12-14T07:10:37.090559Z","iopub.status.idle":"2023-12-14T07:10:37.090938Z","shell.execute_reply.started":"2023-12-14T07:10:37.090753Z","shell.execute_reply":"2023-12-14T07:10:37.09077Z"},"trusted":true},"execution_count":null,"outputs":[]}]}