{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":38760,"databundleVersionId":4493939,"sourceType":"competition"},{"sourceId":4436180,"sourceType":"datasetVersion","datasetId":2597726},{"sourceId":7203605,"sourceType":"datasetVersion","datasetId":4167267}],"dockerImageVersionId":30626,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"colab":{"provenance":[]}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nwarnings.simplefilter(action='ignore')\n\nimport pandas as pd, numpy as np\nfrom tqdm.notebook import tqdm\nimport os, sys, pickle, glob, gc\nfrom collections import Counter\nimport cudf, itertools\n\nfrom pathlib import Path","metadata":{"id":"5o8pYDK_2YPC","outputId":"136f4007-156a-4621-8c39-39456aeee201","execution":{"iopub.status.busy":"2023-12-15T22:26:29.457249Z","iopub.execute_input":"2023-12-15T22:26:29.457632Z","iopub.status.idle":"2023-12-15T22:26:29.463307Z","shell.execute_reply.started":"2023-12-15T22:26:29.457599Z","shell.execute_reply":"2023-12-15T22:26:29.462319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# functions for reading files\ndef read_file(f):\n    return cudf.DataFrame( data_cache[f] )\ndef read_file_to_cache(f):\n    df = pd.read_parquet(f)\n    df.ts = (df.ts/1000).astype('int32')\n    df['type'] = df['type'].map(type_labels).astype('int8')\n    return df\n\n# cache data\ndata_cache = {}\ntype_labels = {'clicks':0, 'carts':1, 'orders':2}\nfiles = glob.glob('../input/otto-chunk-data-inparquet-format/*_parquet/*')\nfor f in files: data_cache[f] = read_file_to_cache(f)","metadata":{"id":"_qdEcKcs2YPE","outputId":"724af9fc-b0dd-4045-ec52-0458caa64760","execution":{"iopub.status.busy":"2023-12-15T22:26:29.465653Z","iopub.execute_input":"2023-12-15T22:26:29.466009Z","iopub.status.idle":"2023-12-15T22:27:09.846532Z","shell.execute_reply.started":"2023-12-15T22:26:29.465974Z","shell.execute_reply":"2023-12-15T22:27:09.845595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set ratio of weights by type (0=click, 1=cart, 2=order)\ntype_wt = {0:.5, 1:9, 2:.5}\nwt_mult = type_wt\n\n# chunk parameters\nREADS_PER = 5\nCHUNK = int( np.ceil( len(files)/6 ))\nprint(f'Processing {len(files)} files, {READS_PER} at a time, in chunks of {CHUNK}.')\n\n# use fewest disk pieces possible without causing memory error\nDISK_PIECES = 8\nSIZE = 1.86e6/DISK_PIECES","metadata":{"execution":{"iopub.status.busy":"2023-12-15T22:27:09.847857Z","iopub.execute_input":"2023-12-15T22:27:09.848257Z","iopub.status.idle":"2023-12-15T22:27:09.855409Z","shell.execute_reply.started":"2023-12-15T22:27:09.848222Z","shell.execute_reply":"2023-12-15T22:27:09.854582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prep_for_cv(df, cv_type, drop=30, days=1):\n    \n    # sort by session and time stamp\n    df = df.sort_values(['session','ts'],ascending=[True,False])\n\n    # drop first # events of each session\n    df = df.reset_index(drop=True)\n    df['n'] = df.groupby('session').cumcount()\n    df = df.loc[df.n<drop].drop('n',axis=1)\n\n    # create pairs of events that occurred within 12 days\n    df = df.merge(df,on='session')\n    df = df.loc[ ((df.ts_x - df.ts_y).abs()< days * 24 * 60 * 60) & (df.aid_x != df.aid_y) ]\n    df = df.loc[(df.aid_x >= PART*SIZE)&(df.aid_x < (PART+1)*SIZE)]\n\n    # assign weights by co-visititation matrix type\n    match cv_type:\n        # cv1: type weighted co cv\n        case 1: \n            df = df[['session', 'aid_x', 'aid_y','type_y']].drop_duplicates(['session', 'aid_x', 'aid_y', 'type_y'])\n            df['wgt'] = df.type_y.map(type_wt)\n            df = df[['aid_x','aid_y','wgt']]\n        # cv2: time weighted click cv\n        case 2: \n            df = df[['session', 'aid_x', 'aid_y','ts_x']].drop_duplicates(['session', 'aid_x', 'aid_y'])\n            df['wgt'] = 1 + 3*(df.ts_x - 1659304800)/(1662328791-1659304800)\n            df = df[['aid_x','aid_y','wgt', 'ts_x']]\n        # cv3: buy to buy cv\n        case 3: \n            df = df[['session', 'aid_x', 'aid_y']].drop_duplicates(['session', 'aid_x', 'aid_y'])\n            df['wgt'] = 1\n            df = df[['aid_x','aid_y','wgt']]\n        case _: \n            df = df[['session', 'aid_x', 'aid_y']].drop_duplicates(['session', 'aid_x', 'aid_y'])\n            df['wgt'] = 1\n            df = df[['aid_x','aid_y','wgt']]\n    \n    df.wgt = df.wgt.astype('float32')\n    df = df.groupby(['aid_x','aid_y']).wgt.sum()\n    \n    return(df)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T22:27:09.856932Z","iopub.execute_input":"2023-12-15T22:27:09.857283Z","iopub.status.idle":"2023-12-15T22:27:09.872017Z","shell.execute_reply.started":"2023-12-15T22:27:09.857249Z","shell.execute_reply":"2023-12-15T22:27:09.871114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n### Type-weighted carts/orders co-visitation matrix (5 min.)\n\n# keep top n cart aids\ncarts_n = 20\n\n# process data in parts\nfor PART in range(DISK_PIECES):\n    print('Part', PART+1, 'of', DISK_PIECES)\n\n    # outer chunks\n    for j in range(6):\n        a = j*CHUNK; b = min( (j+1)*CHUNK, len(files) )\n\n        # inner chunks\n        for k in range(a,b,READS_PER):\n            # read file from cache\n            df = [read_file(files[k])]\n            for i in range(1,READS_PER):\n                if k+i<b: df.append( read_file(files[k+i]) )\n            df = cudf.concat(df,ignore_index=True,axis=0)\n            df = prep_for_cv(df, cv_type=1, days=1)\n\n            # merge inner chunks\n            if k==a: tmp2 = df\n            else: tmp2 = tmp2.add(df, fill_value=0)\n            \n            del df\n            gc.collect()\n\n        # merge outer chunks\n        if a==0: tmp = tmp2\n        else: tmp = tmp.add(tmp2, fill_value=0)\n        del tmp2\n        gc.collect()\n\n    # convert for storage\n    tmp = tmp.reset_index()\n    tmp = tmp.sort_values(['aid_x','wgt'],ascending=[True,False])\n\n    # keep top results (by cumulative aid_y ccount with carts_n as cutline)\n    tmp = tmp.reset_index(drop=True)\n    tmp['n'] = tmp.groupby('aid_x').aid_y.cumcount()\n    tmp = tmp.loc[tmp.n<carts_n].drop('n',axis=1)\n\n    # convert to pandas and save as parquet\n    tmp.to_pandas().to_parquet(f'top_n_carts_orders_{PART}.pqt')","metadata":{"id":"PJioGgOw2YPF","outputId":"c2a50a05-da98-41cd-bf0f-cc9c82cf0ce1","execution":{"iopub.status.busy":"2023-12-15T22:27:09.874618Z","iopub.execute_input":"2023-12-15T22:27:09.874908Z","iopub.status.idle":"2023-12-15T22:38:17.287411Z","shell.execute_reply.started":"2023-12-15T22:27:09.874884Z","shell.execute_reply":"2023-12-15T22:38:17.286323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n### Cart/order co-visitation matrix given a user's previous cart/order (2 min.)\n\n# keep top n order aids\norders_n = 20\n\n# process data in parts\nfor PART in range(DISK_PIECES):\n    print('Part', PART+1, 'of', DISK_PIECES)\n\n    # outer chunks\n    for j in range(6):\n        a = j*CHUNK; b = min( (j+1)*CHUNK, len(files) )\n\n        # inner chunks\n        for k in range(a,b,READS_PER):\n\n            # read file from cache\n            df = [read_file(files[k])]\n            for i in range(1,READS_PER):\n                if k+i<b: df.append( read_file(files[k+i]) )\n            df = cudf.concat(df,ignore_index=True,axis=0)\n            df = df.loc[df['type'].isin([1,2])] # 1,2 is carts,orders\n            df = prep_for_cv(df, cv_type=3, days=14)\n\n            # merge inner\n            if k==a: tmp2 = df\n            else: tmp2 = tmp2.add(df, fill_value=0)\n            \n            del df\n            gc.collect()\n\n        # merge outer\n        if a==0: tmp = tmp2\n        else: tmp = tmp.add(tmp2, fill_value=0)\n            \n        del tmp2\n        gc.collect()\n\n    # convert for storage\n    tmp = tmp.reset_index()\n    tmp = tmp.sort_values(['aid_x','wgt'],ascending=[True,False])\n\n    # keep top results (by cumulative aid_y count with orders_n as cutline)\n    tmp = tmp.reset_index(drop=True)\n    tmp['n'] = tmp.groupby('aid_x').aid_y.cumcount()\n    tmp = tmp.loc[tmp.n<orders_n].drop('n',axis=1)\n\n    # convert to pandas and save as parquet\n    tmp.to_pandas().to_parquet(f'top_n_buy_buy_{PART}.pqt')","metadata":{"id":"W5shGKL-2YPF","outputId":"fdaaaf2d-ed75-4264-bffc-2c8b72085dd7","execution":{"iopub.status.busy":"2023-12-15T22:38:17.289087Z","iopub.execute_input":"2023-12-15T22:38:17.289518Z","iopub.status.idle":"2023-12-15T22:46:16.278943Z","shell.execute_reply.started":"2023-12-15T22:38:17.289476Z","shell.execute_reply":"2023-12-15T22:46:16.278027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n### Time-weighted co-visitation matrix for clicks (5 min.)\n\n# keep top n click aids\nclicks_n = 15\n\n# process in parts\nfor PART in range(DISK_PIECES):\n    print('Part', PART+1, 'of', DISK_PIECES)\n\n    # outer chunks\n    for j in range(6):\n        a = j*CHUNK\n        b = min( (j+1)*CHUNK, len(files) )\n\n        # inner chunks\n        for k in range(a,b,READS_PER):\n            \n            # read file from cache\n            df = [read_file(files[k])]\n            for i in range(1,READS_PER):\n                if k+i<b: df.append( read_file(files[k+i]) )\n            df = cudf.concat(df,ignore_index=True,axis=0)\n            df = prep_for_cv(df, cv_type=2, days=1)\n\n            # merge inner chunks\n            if k==a: tmp2 = df\n            else: tmp2 = tmp2.add(df, fill_value=0)\n            \n            del df\n            gc.collect()\n\n        # merge outer chunks\n        if a==0: tmp = tmp2\n        else: tmp = tmp.add(tmp2, fill_value=0)\n            \n        del tmp2\n        gc.collect()\n\n    # convert for storage\n    tmp = tmp.reset_index()\n    tmp = tmp.sort_values(['aid_x','wgt'],ascending=[True,False])\n\n    # keep top results (by cumulative aid_y count with clicks_n as cutline)\n    tmp = tmp.reset_index(drop=True)\n    tmp['n'] = tmp.groupby('aid_x').aid_y.cumcount()\n    tmp = tmp.loc[tmp.n<clicks_n].drop('n',axis=1)\n\n    # convert to pandas and save as parquet\n    tmp.to_pandas().to_parquet(f'top_n_clicks_{PART}.pqt')","metadata":{"id":"LqUn4mty2YPG","outputId":"f2021511-89eb-488c-b610-4f28be4f5ae7","execution":{"iopub.status.busy":"2023-12-15T22:46:16.280307Z","iopub.execute_input":"2023-12-15T22:46:16.280641Z","iopub.status.idle":"2023-12-15T22:57:20.259928Z","shell.execute_reply.started":"2023-12-15T22:46:16.280603Z","shell.execute_reply":"2023-12-15T22:57:20.259017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# release memory\ndel data_cache, tmp\n_ = gc.collect()","metadata":{"id":"1rNuEv4B2YPH","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-12-15T22:57:20.261368Z","iopub.execute_input":"2023-12-15T22:57:20.261685Z","iopub.status.idle":"2023-12-15T22:57:21.709902Z","shell.execute_reply.started":"2023-12-15T22:57:20.261658Z","shell.execute_reply":"2023-12-15T22:57:21.708694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load test data\ndef load_test():\n    dfs = []\n    for e, chunk_file in enumerate(glob.glob('../input/otto-chunk-data-inparquet-format/test_parquet/*')):\n        chunk = pd.read_parquet(chunk_file)\n        chunk.ts = (chunk.ts/1000).astype('int32')\n        chunk['type'] = chunk['type'].map(type_labels).astype('int8')\n        dfs.append(chunk)\n    return pd.concat(dfs).reset_index(drop=True) #.astype({\"ts\": \"datetime64[ms]\"})\n\ntest_df = load_test()\nprint('Test data dimensions:',test_df.shape)\ntest_df.head()","metadata":{"id":"gpY-Ac5y2YPH","outputId":"b0811e83-7887-44ef-dfc8-34a2add75710","execution":{"iopub.status.busy":"2023-12-15T22:57:21.711863Z","iopub.execute_input":"2023-12-15T22:57:21.712631Z","iopub.status.idle":"2023-12-15T22:57:23.186656Z","shell.execute_reply.started":"2023-12-15T22:57:21.712601Z","shell.execute_reply":"2023-12-15T22:57:23.185745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_comp_test():\n    dfs = []\n    for e, chunk_file in enumerate(glob.glob('/kaggle/input/comp-test-data/comp_test_parquet/*')):\n        chunk = pd.read_parquet(chunk_file)\n        chunk.ts = (chunk.ts/1000).astype('int32')\n        chunk['type'] = chunk['type'].map(type_labels).astype('int8')\n        dfs.append(chunk)\n    return pd.concat(dfs).reset_index(drop=True) #.astype({\"ts\": \"datetime64[ms]\"})\n\ncomp_test_df = load_comp_test()\nprint('Complete test data dimensions:',comp_test_df.shape)\ncomp_test_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-15T22:57:23.187965Z","iopub.execute_input":"2023-12-15T22:57:23.188440Z","iopub.status.idle":"2023-12-15T22:57:25.903520Z","shell.execute_reply.started":"2023-12-15T22:57:23.188396Z","shell.execute_reply":"2023-12-15T22:57:25.902595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n### load most frequent click, cart, order aids for custom rules to adjust recommendation rankings (3 min)\n\n# convert df from parquet file to dictionary\ndef pqt_to_dict(df):\n    return df.groupby('aid_x').aid_y.apply(list).to_dict()\n\n# load 0th co-visitation matrices\ntop_n_clicks = pqt_to_dict( pd.read_parquet(f'top_n_clicks_0.pqt') )\ntop_n_buys = pqt_to_dict( pd.read_parquet(f'top_n_carts_orders_0.pqt') )\ntop_n_buy_buy = pqt_to_dict( pd.read_parquet(f'top_n_buy_buy_0.pqt') )\n\n# update 0th co-visitation matrices with each successive iteration\nfor k in range(1,DISK_PIECES):\n    top_n_clicks.update( pqt_to_dict( pd.read_parquet(f'top_n_clicks_{k}.pqt') ) )\n    top_n_buys.update( pqt_to_dict( pd.read_parquet(f'top_n_carts_orders_{k}.pqt') ) )\n    top_n_buy_buy.update( pqt_to_dict( pd.read_parquet(f'top_n_buy_buy_{k}.pqt') ) )","metadata":{"id":"1MGOKloG2YPI","outputId":"6758654b-5c15-44aa-ebee-e02582dc497e","execution":{"iopub.status.busy":"2023-12-15T22:57:25.904901Z","iopub.execute_input":"2023-12-15T22:57:25.905320Z","iopub.status.idle":"2023-12-15T23:00:11.613466Z","shell.execute_reply.started":"2023-12-15T22:57:25.905283Z","shell.execute_reply":"2023-12-15T23:00:11.612457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# most frequent clicks carts and orders in test data\ntop_clicks = test_df.loc[test_df['type']== 0,'aid'].value_counts().index.values[:20]\ntop_carts = test_df.loc[test_df['type']== 1,'aid'].value_counts().index.values[:20]\ntop_orders = test_df.loc[test_df['type']== 2,'aid'].value_counts().index.values[:20]\n# check amount of overlap between most frequent clicks carts and orders\nlen(set(list(top_clicks)+list(top_carts)+list(top_orders)))","metadata":{"id":"F3jrY4iY2YPI","execution":{"iopub.status.busy":"2023-12-15T23:00:11.614929Z","iopub.execute_input":"2023-12-15T23:00:11.615296Z","iopub.status.idle":"2023-12-15T23:00:12.059959Z","shell.execute_reply.started":"2023-12-15T23:00:11.615261Z","shell.execute_reply":"2023-12-15T23:00:12.059014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def suggest_clicks(df):\n    # user history by items and events\n    aids=df.aid.tolist()\n    types = df.type.tolist()\n    unique_aids = list(dict.fromkeys(aids[::-1] ))\n    \n    # adjust recommendations by weights\n    if len(unique_aids)>=20:\n        weights=np.logspace(0.1,1,len(aids),base=2, endpoint=True)-1\n        aids_temp = Counter()\n        \n        # adjust recommendations by user history\n        for aid,w,t in zip(aids,weights,types):\n            aids_temp[aid] += w * wt_mult[t]\n        sorted_aids = [k for k,v in aids_temp.most_common(20)]\n        return sorted_aids\n\n    # pad recommendations with most common clicks in training data if not in user history\n    aids2 = list(itertools.chain(*[top_n_clicks[aid] for aid in unique_aids if aid in top_clicks]))\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids2).most_common(20) if aid2 not in unique_aids]\n    result = unique_aids + top_aids2[:20 - len(unique_aids)]\n    \n    # pad recommendations with most common clicks in current week\n    return result + list(top_clicks)[:20-len(result)]","metadata":{"id":"BGBxxDHe2YPI","execution":{"iopub.status.busy":"2023-12-15T23:00:12.061085Z","iopub.execute_input":"2023-12-15T23:00:12.061412Z","iopub.status.idle":"2023-12-15T23:00:12.070224Z","shell.execute_reply.started":"2023-12-15T23:00:12.061386Z","shell.execute_reply":"2023-12-15T23:00:12.069343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def suggest_carts(df):\n    # user history by items and events\n    aids = df.aid.tolist()\n    types = df.type.tolist()\n    unique_aids = list(dict.fromkeys(aids[::-1] ))\n    df = df.loc[(df['type'] == 0)|(df['type'] == 1)]\n    unique_buys = list(dict.fromkeys(df.aid.tolist()[::-1]))\n\n    # adjust recommendations by weights\n    if len(unique_aids) >= 20:\n        weights=np.logspace(0.5,1,len(aids),base=2, endpoint=True)-1\n        aids_temp = Counter()\n\n        # adjust recommendations by user history\n        for aid,w,t in zip(aids,weights,types):\n            aids_temp[aid] += w * wt_mult[t]\n\n        # pad recommendations with most frequent orders in training data given purchases in user history\n        aids2 = list(itertools.chain(*[top_n_buys[aid] for aid in unique_buys if aid in top_orders]))\n        for aid in aids2: aids_temp[aid] += 0.1\n        sorted_aids = [k for k,v in aids_temp.most_common(20)]\n        return sorted_aids\n\n    # pad recommendations with most frequent clicks and orders in training data given user history\n    aids1 = list(itertools.chain(*[top_n_clicks[aid] for aid in unique_aids if aid in top_clicks]))\n    aids2 = list(itertools.chain(*[top_n_buys[aid] for aid in unique_aids if aid in top_orders]))\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids1+aids2).most_common(20) if aid2 not in unique_aids]\n    result = unique_aids + top_aids2[:20 - len(unique_aids)]\n\n    # pad recommendations with most common carts in current week\n    return result + list(top_carts)[:20-len(result)]","metadata":{"id":"_HFDWhUE2YPJ","execution":{"iopub.status.busy":"2023-12-15T23:00:12.075009Z","iopub.execute_input":"2023-12-15T23:00:12.075908Z","iopub.status.idle":"2023-12-15T23:00:12.089794Z","shell.execute_reply.started":"2023-12-15T23:00:12.075873Z","shell.execute_reply":"2023-12-15T23:00:12.088805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def suggest_buys(df):\n    # user history by items and events\n    aids=df.aid.tolist()\n    types = df.type.tolist()\n    unique_aids = list(dict.fromkeys(aids[::-1] ))\n    #u_aids = list(set(aids[::-1]))\n    df = df.loc[(df['type']==1)|(df['type']==2)]\n    #df2 = df.loc[df['type'] in [1,2]]\n    unique_buys = list(dict.fromkeys( df.aid.tolist()[::-1] ))\n    #u_buys = list(set(df.aid.tolist()[::-1]))\n\n    # adjust recommendations by weights\n    if len(unique_aids)>=20:\n        weights=np.logspace(0.5,1,len(aids),base=2, endpoint=True)-1\n        aids_temp = Counter()\n\n        # adjust recommendations by user history\n        for aid,w,t in zip(aids,weights,types):\n            aids_temp[aid] += w * wt_mult[t]\n        \n        aids3 = list(itertools.chain(*[top_n_buy_buy[aid] for aid in unique_buys if aid in top_n_buy_buy]))\n        for aid in aids3: aids_temp[aid] += 0.1\n        sorted_aids = [k for k,v in aids_temp.most_common(20)]\n        return sorted_aids\n    \n    # pad recommendations with most common orders in training data given user history\n    aids2 = list(itertools.chain(*[top_n_buys[aid] for aid in unique_aids if aid in top_n_buys]))\n    aids3 = list(itertools.chain(*[top_n_buy_buy[aid] for aid in unique_buys if aid in top_n_buy_buy]))\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids2+aids3).most_common(20) if aid2 not in unique_aids]\n    result = unique_aids + top_aids2[:20 - len(unique_aids)]\n\n    # pad recommendations with most common orders in current week\n    return result + list(top_orders)[:20-len(result)]","metadata":{"id":"il8dPBsu2YPJ","execution":{"iopub.status.busy":"2023-12-15T23:00:12.091504Z","iopub.execute_input":"2023-12-15T23:00:12.091885Z","iopub.status.idle":"2023-12-15T23:00:12.106303Z","shell.execute_reply.started":"2023-12-15T23:00:12.091851Z","shell.execute_reply":"2023-12-15T23:00:12.105187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n### truncate test data sessions to test predictions (<2 hrs)\nimport random\n\ntrunc_test_df = pd.DataFrame()\nsolution_df = pd.DataFrame()\n\n# Get list of sesssions in test_df\ntest_sessions = list(test_df['session'].unique())\n\n# randomly truncate sessions placing earlier data in trunc_test_df and remainder in solution_df\nfor session in test_sessions[:500000]:\n    \n    session_df = test_df[test_df['session']==session]\n    if len(session_df['aid']) < 24: continue\n    \n    trunc_test_df = pd.concat([trunc_test_df, session_df[0:len(session_df['aid'])-20]])\n    solution_df = pd.concat([solution_df, session_df[len(session_df['aid'])-20:len(session_df['aid'])]])\n    \nsolution_df.reset_index(inplace=True)\nsolution_df.sort_values(by=[\"session\", \"ts\"], axis=0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T23:00:12.107955Z","iopub.execute_input":"2023-12-15T23:00:12.108344Z","iopub.status.idle":"2023-12-15T23:45:05.859274Z","shell.execute_reply.started":"2023-12-15T23:00:12.108317Z","shell.execute_reply":"2023-12-15T23:45:05.858247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# creating recommendations for truncated test data (~30 sec.)\npred_clicks = trunc_test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_clicks(x))\n\npred_carts = trunc_test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_carts(x))\n\npred_buys = trunc_test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_buys(x))","metadata":{"execution":{"iopub.status.busy":"2023-12-15T23:45:05.860798Z","iopub.execute_input":"2023-12-15T23:45:05.861176Z","iopub.status.idle":"2023-12-15T23:45:30.438012Z","shell.execute_reply.started":"2023-12-15T23:45:05.861140Z","shell.execute_reply":"2023-12-15T23:45:30.436967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# compare predictions for truncated test df to solution df for truncated test data (<30 sec.)\nclick_matches=0\ncart_matches=0\norder_matches=0\nnum_carts=0\nnum_orders=0\n\nfor i in range(len(pred_clicks.index)):\n    session = pred_clicks.index[i]\n    session_df = solution_df[solution_df['session']==session]\n    # count correct click predictions\n    if 0 in list(session_df['type']):\n        clicks_df = session_df[session_df['type']==0]\n        if clicks_df.aid[:1].item() in pred_clicks[session]:\n            click_matches += 1\n    # count correct cart predictions\n    if 1 in list(session_df['type']):\n        carts_df = session_df[session_df['type']==1]\n        for item in carts_df.aid:\n            if item in pred_carts[session]:\n                cart_matches += 1\n        num_carts += min(20, len(carts_df.aid.unique()))\n    # count correct order predictions\n    if 2 in list(session_df['type']):\n        orders_df = session_df[session_df['type']==2]\n        for item in orders_df.aid:\n            if item in pred_buys[pred_clicks.index[i]]:\n                order_matches += 1\n        num_orders += min(20, len(orders_df.aid.unique()))\n            \nprint(\"Next click:\", click_matches/len(pred_clicks))\nprint(\"Carts:\", cart_matches/num_carts)\nprint(\"Orders:\", order_matches/num_orders)\nprint(\"Recall@20 =\",.1 * click_matches/len(pred_clicks) + .3 * cart_matches/num_carts + .6 * order_matches/num_orders)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T23:45:30.439521Z","iopub.execute_input":"2023-12-15T23:45:30.439910Z","iopub.status.idle":"2023-12-15T23:45:49.129577Z","shell.execute_reply.started":"2023-12-15T23:45:30.439875Z","shell.execute_reply":"2023-12-15T23:45:49.128616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n### open complete test data, merge with test df, drop duplicated rows, reset index and sort\n\nsolution_df = pd.concat([test_df, comp_test_df])\nsolution_df = solution_df[solution_df.duplicated(keep=False)]\nsolution_df.reset_index(inplace=True)\nsolution_df.sort_values(by=[\"session\", \"ts\"], axis=0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T23:45:49.131031Z","iopub.execute_input":"2023-12-15T23:45:49.131360Z","iopub.status.idle":"2023-12-15T23:46:00.425263Z","shell.execute_reply.started":"2023-12-15T23:45:49.131302Z","shell.execute_reply":"2023-12-15T23:46:00.424199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# creating recommendations for complete test data is slow (~45 min.)\npred_clicks = test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_clicks(x))\n\npred_carts = test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_carts(x))\n\npred_buys = test_df.sort_values([\"session\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_buys(x))","metadata":{"id":"t5iW7N1O2YPJ","outputId":"48547adf-4e3e-493e-d28f-7cc2c7f7a411","execution":{"iopub.status.busy":"2023-12-16T01:20:53.078968Z","iopub.execute_input":"2023-12-16T01:20:53.079355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# compare predictions on complete test df to solution df (~5 hours for comp_test_df)\nclick_matches=0\ncart_matches=0\norder_matches=0\nnum_carts=0\nnum_orders=0\n\nfor i in range(len(pred_clicks.index)):\n    session = pred_clicks.index[i]\n    session_df = solution_df[solution_df['session']==session]\n    # count correct click predictions\n    if 0 in list(session_df['type']):\n        clicks_df = session_df[session_df['type']==0]\n        if clicks_df.aid[:1].item() in pred_clicks[session]:\n            click_matches += 1\n    # count correct cart predictions\n    if 1 in list(session_df['type']):\n        carts_df = session_df[session_df['type']==1]\n        for item in carts_df.aid.unique():\n            if item in pred_carts[session]:\n                cart_matches += 1\n        num_carts += min(20, len(carts_df.aid))\n    # count correct order predictions\n    if 2 in list(session_df['type']):\n        orders_df = session_df[session_df['type']==2]\n        for item in orders_df.aid.unique():\n            if item in pred_buys[pred_clicks.index[i]]:\n                order_matches += 1\n        num_orders += min(20, len(orders_df.aid))\n    if i % 100000 == 0: print(f\"{i} of {len(pred_clicks.index)} complete\")\n            \nprint(\"Next click:\", click_matches/len(pred_clicks))\nprint(\"Carts:\", cart_matches/num_carts)\nprint(\"Orders:\", order_matches/num_orders)\nprint(\"Recall@20=\",.1 * click_matches/len(pred_clicks) + .3 * cart_matches/num_carts + .6 * order_matches/num_orders)","metadata":{"execution":{"iopub.status.busy":"2023-12-16T01:20:16.598054Z","iopub.execute_input":"2023-12-16T01:20:16.598350Z","iopub.status.idle":"2023-12-16T01:20:17.212594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clicks_pred = pd.DataFrame(pred_clicks.add_suffix(\"_clicks\"), columns=[\"labels\"]).reset_index()\norders_pred = pd.DataFrame(pred_buys.add_suffix(\"_orders\"), columns=[\"labels\"]).reset_index()\ncarts_pred = pd.DataFrame(pred_carts.add_suffix(\"_carts\"), columns=[\"labels\"]).reset_index()","metadata":{"id":"gOZ2OpZo2YPK","execution":{"iopub.status.busy":"2023-12-15T23:51:04.205936Z","iopub.execute_input":"2023-12-15T23:51:04.206232Z","iopub.status.idle":"2023-12-15T23:51:04.259374Z","shell.execute_reply.started":"2023-12-15T23:51:04.206206Z","shell.execute_reply":"2023-12-15T23:51:04.258567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.concat([clicks_pred, orders_pred, carts_pred])\npred_df.columns = [\"session_type\", \"labels\"]\npred_df[\"labels\"] = pred_df.labels.apply(lambda x: \" \".join(map(str,x)))\npred_df.to_csv(\"submission.csv\", index=False)\npred_df.head()","metadata":{"id":"7nCjOUfI2YPK","outputId":"0461e834-788b-4721-9ea5-e9cb27e0f372","execution":{"iopub.status.busy":"2023-12-15T23:51:04.260614Z","iopub.execute_input":"2023-12-15T23:51:04.260921Z","iopub.status.idle":"2023-12-15T23:51:05.844086Z","shell.execute_reply.started":"2023-12-15T23:51:04.260895Z","shell.execute_reply":"2023-12-15T23:51:05.843144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip -r file.zip /kaggle/working\nfrom IPython.display import FileLink\nFileLink(r'file.zip')","metadata":{"execution":{"iopub.status.busy":"2023-12-15T23:51:05.845938Z","iopub.execute_input":"2023-12-15T23:51:05.846360Z","iopub.status.idle":"2023-12-15T23:52:01.613530Z","shell.execute_reply.started":"2023-12-15T23:51:05.846321Z","shell.execute_reply":"2023-12-15T23:52:01.612372Z"},"trusted":true},"execution_count":null,"outputs":[]}]}