{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Thank's to [Chris Deotte](https://www.kaggle.com/cdeotte) for https://www.kaggle.com/code/cdeotte/candidate-rerank-model-lb-0-575\n\nIn this notebook, I put in a dataset, co-visitation matrices for valid and test, calculated with GPU. [Here](https://www.kaggle.com/code/adaubas/otto-fast-handcrafted-model/log) those matrices are used to do fast predictions with parallel CPU.\n\nI wasn't able to both in the same notebook because of an unresolved error while using GPU and CPU in parallel in the same notebook : [here](https://www.kaggle.com/code/adaubas/compute-validation-score-cv-565-086787-faster)","metadata":{}},{"cell_type":"code","source":"VER = 1\n\nimport pandas as pd, numpy as np\nimport glob, gc\nimport cudf\nprint('We will use RAPIDS version', cudf.__version__)","metadata":{"papermill":{"duration":2.845087,"end_time":"2022-11-10T16:03:27.484916","exception":false,"start_time":"2022-11-10T16:03:24.639829","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-12-02T05:48:00.733739Z","iopub.execute_input":"2022-12-02T05:48:00.734107Z","iopub.status.idle":"2022-12-02T05:48:03.643852Z","shell.execute_reply.started":"2022-12-02T05:48:00.73402Z","shell.execute_reply":"2022-12-02T05:48:03.642648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Co validation matrices for validation","metadata":{}},{"cell_type":"code","source":"%%time\n\n# CACHE FUNCTIONS\ndef read_file(f):\n    return cudf.DataFrame( data_cache[f] )\n\ndef read_file_to_cache(f):\n    df = pd.read_parquet(f)\n    df.ts = (df.ts / 1000).astype('int32')\n    df['type'] = df['type'].map(type_labels).astype('int8')\n    return df\n\n# CACHE THE DATA ON CPU BEFORE PROCESSING ON GPU\ndata_cache = {}\ntype_labels = {'clicks':0, 'carts':1, 'orders':2}\nfiles = glob.glob('../input/otto-validation/*_parquet/*')\nfor f in files: data_cache[f] = read_file_to_cache(f)\n\n# CHUNK PARAMETERS\nREAD_CT = 5\nCHUNK = int( np.ceil( len(files) / 6 ))\nprint(f'We will process {len(files)} files, in groups of {READ_CT} and chunks of {CHUNK}.')","metadata":{"papermill":{"duration":0.044165,"end_time":"2022-11-10T16:03:27.542811","exception":false,"start_time":"2022-11-10T16:03:27.498646","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-12-02T05:48:03.646059Z","iopub.execute_input":"2022-12-02T05:48:03.646809Z","iopub.status.idle":"2022-12-02T05:48:48.501803Z","shell.execute_reply.started":"2022-12-02T05:48:03.646767Z","shell.execute_reply":"2022-12-02T05:48:48.500597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_covisitation(fn_heart, DISK_PIECES, save_top, output_name):\n    \n    SIZE = 1.86e6 / DISK_PIECES\n    \n    # COMPUTE IN PARTS FOR MEMORY MANGEMENT\n    for PART in range(DISK_PIECES):\n        print()\n        print('### DISK PART',PART+1)\n    \n        # MERGE IS FASTEST PROCESSING CHUNKS WITHIN CHUNKS\n        # => OUTER CHUNKS\n        for j in range(6):\n            a = j * CHUNK\n            b = min( (j + 1) * CHUNK, len(files) )\n            print(f'Processing files {a} thru {b - 1} in groups of {READ_CT}...')\n        \n            # => INNER CHUNKS\n            for k in range(a,b,READ_CT):\n                # READ FILE\n                df = [read_file(files[k])]\n                for i in range(1, READ_CT): \n                    if k+i<b: df.append( read_file(files[k+i]) )\n                df = cudf.concat(df, ignore_index=True,axis=0)\n                \n                df = fn_heart(df, from_ = PART * SIZE, to_ = (PART+1) *SIZE)\n\n                # COMBINE INNER CHUNKS\n                if k == a: tmp2 = df\n                else: tmp2 = tmp2.add(df, fill_value = 0)\n                print(k,', ',end = '')\n            print()\n            # COMBINE OUTER CHUNKS\n            if a == 0: tmp = tmp2\n            else: tmp = tmp.add(tmp2, fill_value = 0)\n            del tmp2, df\n            gc.collect()\n            \n        # CONVERT MATRIX TO DICTIONARY\n        tmp = tmp.reset_index()\n        tmp = tmp.sort_values(['aid_x', 'wgt'], ascending = [True,False])\n        \n        # SAVE TOP \n        tmp = tmp.reset_index(drop=True)\n        tmp['n'] = tmp.groupby('aid_x').aid_y.cumcount()\n        tmp = tmp.loc[tmp.n < save_top].drop('n',axis=1)\n        \n        # SAVE PART TO DISK (convert to pandas first uses less memory)\n        tmp.to_pandas().to_parquet(f'top_{save_top}_{output_name}_v{VER}_{PART}.pqt')","metadata":{"execution":{"iopub.status.busy":"2022-11-29T20:43:43.639789Z","iopub.execute_input":"2022-11-29T20:43:43.640181Z","iopub.status.idle":"2022-11-29T20:43:43.661991Z","shell.execute_reply.started":"2022-11-29T20:43:43.640137Z","shell.execute_reply":"2022-11-29T20:43:43.66004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def heart_carts_order(df, from_, to_, ndays = 1, type_weight = {0:1, 1:6, 2:3}):\n    \n    df = df.sort_values(['session','ts'], ascending = [True, False])\n    \n    # USE TAIL OF SESSION\n    df = df.reset_index(drop = True)\n    df['n'] = df.groupby('session').cumcount()\n    df = df.loc[df.n < 30].drop('n', axis = 1)\n    \n    # CREATE PAIRS\n    df = df.merge(df,on='session')\n    df = df.loc[ ((df.ts_x - df.ts_y).abs()< ndays * 24 * 60 * 60) & (df.aid_x != df.aid_y) ]\n    \n    # MEMORY MANAGEMENT COMPUTE IN PARTS\n    df = df.loc[(df.aid_x >= from_) & (df.aid_x < to_)]\n    \n    # ASSIGN WEIGHTS\n    df = df[['session', 'aid_x', 'aid_y', 'type_y']].drop_duplicates(['session', 'aid_x', 'aid_y'])\n    df['wgt'] = df.type_y.map(type_weight)\n    df = df[['aid_x', 'aid_y', 'wgt']]\n    df.wgt = df.wgt.astype('float32')\n\n    return df.groupby(['aid_x', 'aid_y']).wgt.sum()","metadata":{"execution":{"iopub.status.busy":"2022-11-29T20:43:43.665828Z","iopub.execute_input":"2022-11-29T20:43:43.666332Z","iopub.status.idle":"2022-11-29T20:43:43.685044Z","shell.execute_reply.started":"2022-11-29T20:43:43.666297Z","shell.execute_reply":"2022-11-29T20:43:43.683946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_covisitation(fn_heart=heart_carts_order, DISK_PIECES=4, save_top = 15, output_name = \"valid_carts_orders\")","metadata":{"execution":{"iopub.status.busy":"2022-11-29T20:43:43.689876Z","iopub.execute_input":"2022-11-29T20:43:43.690244Z","iopub.status.idle":"2022-11-29T20:45:56.228382Z","shell.execute_reply.started":"2022-11-29T20:43:43.690201Z","shell.execute_reply":"2022-11-29T20:45:56.22739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def heart_buy2buy(df, from_, to_, ndays = 14):\n    \n    # ONLY WANT CARTS AND ORDERS\n    df = df.loc[df['type'].isin([1,2])] \n    df = df.sort_values(['session','ts'],ascending=[True,False])\n    \n    # USE TAIL OF SESSION\n    df = df.reset_index(drop=True)\n    df['n'] = df.groupby('session').cumcount()\n    df = df.loc[df.n<30].drop('n',axis=1)\n    \n    # CREATE PAIRS\n    df = df.merge(df,on='session')\n    df = df.loc[ ((df.ts_x - df.ts_y).abs()< ndays * 24 * 60 * 60) & (df.aid_x != df.aid_y) ] \n    \n    # MEMORY MANAGEMENT COMPUTE IN PARTS\n    df = df.loc[(df.aid_x >= from_) & (df.aid_x < to_)]\n    \n    # ASSIGN WEIGHTS\n    df = df[['session', 'aid_x', 'aid_y','type_y']].drop_duplicates(['session', 'aid_x', 'aid_y'])\n    df['wgt'] = 1\n    df = df[['aid_x','aid_y','wgt']]\n    df.wgt = df.wgt.astype('float32')\n    \n    return df.groupby(['aid_x','aid_y']).wgt.sum()","metadata":{"execution":{"iopub.status.busy":"2022-11-29T20:46:04.91959Z","iopub.execute_input":"2022-11-29T20:46:04.919963Z","iopub.status.idle":"2022-11-29T20:46:04.929082Z","shell.execute_reply.started":"2022-11-29T20:46:04.919931Z","shell.execute_reply":"2022-11-29T20:46:04.928044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_covisitation(fn_heart=heart_buy2buy, DISK_PIECES = 1, save_top = 15, output_name = \"valid_buy2buy\")","metadata":{"execution":{"iopub.status.busy":"2022-11-29T20:46:07.835438Z","iopub.execute_input":"2022-11-29T20:46:07.835808Z","iopub.status.idle":"2022-11-29T20:46:26.879578Z","shell.execute_reply.started":"2022-11-29T20:46:07.835757Z","shell.execute_reply":"2022-11-29T20:46:26.878577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def heart_clicks(df, from_, to_, ndays = 1):\n\n    df = df.sort_values(['session','ts'],ascending=[True,False])\n    \n    # USE TAIL OF SESSION\n    df = df.reset_index(drop=True)\n    df['n'] = df.groupby('session').cumcount()\n    df = df.loc[df.n<30].drop('n',axis=1)\n    \n    # CREATE PAIRS\n    df = df.merge(df,on='session')\n    df = df.loc[ ((df.ts_x - df.ts_y).abs()< ndays * 24 * 60 * 60) & (df.aid_x != df.aid_y) ]\n    \n    # MEMORY MANAGEMENT COMPUTE IN PARTS\n    df = df.loc[(df.aid_x >= from_)&(df.aid_x < to_)]\n    \n    # ASSIGN WEIGHTS\n    df = df[['session', 'aid_x', 'aid_y','ts_x']].drop_duplicates(['session', 'aid_x', 'aid_y'])\n    df['wgt'] = 1 + 3 * (df.ts_x - 1659304800)/(1662328791-1659304800)\n    df = df[['aid_x','aid_y','wgt']]\n    df.wgt = df.wgt.astype('float32')\n                                     \n    return df.groupby(['aid_x','aid_y']).wgt.sum()","metadata":{"execution":{"iopub.status.busy":"2022-11-29T20:46:27.664423Z","iopub.execute_input":"2022-11-29T20:46:27.66481Z","iopub.status.idle":"2022-11-29T20:46:27.674849Z","shell.execute_reply.started":"2022-11-29T20:46:27.664755Z","shell.execute_reply":"2022-11-29T20:46:27.673756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_covisitation(fn_heart=heart_clicks, DISK_PIECES = 4, save_top = 20, output_name = \"valid_clicks\")","metadata":{"execution":{"iopub.status.busy":"2022-11-29T20:46:28.477749Z","iopub.execute_input":"2022-11-29T20:46:28.478281Z","iopub.status.idle":"2022-11-29T20:48:37.561434Z","shell.execute_reply.started":"2022-11-29T20:46:28.478231Z","shell.execute_reply":"2022-11-29T20:48:37.560416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del data_cache\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-29T20:49:16.185592Z","iopub.execute_input":"2022-11-29T20:49:16.186304Z","iopub.status.idle":"2022-11-29T20:49:16.334803Z","shell.execute_reply.started":"2022-11-29T20:49:16.186266Z","shell.execute_reply":"2022-11-29T20:49:16.333775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Co validation matrices for test","metadata":{}},{"cell_type":"code","source":"%%time\ndata_cache = {}\ntype_labels = {'clicks':0, 'carts':1, 'orders':2}\nfiles = glob.glob('../input/otto-chunk-data-inparquet-format/*_parquet/*')\nfor f in files: data_cache[f] = read_file_to_cache(f)\n\n# CHUNK PARAMETERS\nREAD_CT = 5\nCHUNK = int( np.ceil( len(files) / 6 ))\nprint(f'We will process {len(files)} files, in groups of {READ_CT} and chunks of {CHUNK}.')","metadata":{"execution":{"iopub.status.busy":"2022-11-29T20:49:52.292994Z","iopub.execute_input":"2022-11-29T20:49:52.293999Z","iopub.status.idle":"2022-11-29T20:50:58.60698Z","shell.execute_reply.started":"2022-11-29T20:49:52.293951Z","shell.execute_reply":"2022-11-29T20:50:58.605844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_covisitation(fn_heart = heart_carts_order, DISK_PIECES = 4, save_top = 15, output_name = \"test_carts_orders\")","metadata":{"execution":{"iopub.status.busy":"2022-11-29T20:50:58.608837Z","iopub.execute_input":"2022-11-29T20:50:58.610029Z","iopub.status.idle":"2022-11-29T20:54:17.79768Z","shell.execute_reply.started":"2022-11-29T20:50:58.609989Z","shell.execute_reply":"2022-11-29T20:54:17.796631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_covisitation(fn_heart = heart_buy2buy, DISK_PIECES = 1, save_top = 15, output_name = \"test_buy2buy\")","metadata":{"execution":{"iopub.status.busy":"2022-11-29T20:54:35.838207Z","iopub.execute_input":"2022-11-29T20:54:35.838555Z","iopub.status.idle":"2022-11-29T20:55:06.164156Z","shell.execute_reply.started":"2022-11-29T20:54:35.838523Z","shell.execute_reply":"2022-11-29T20:55:06.16314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_covisitation(fn_heart = heart_clicks, DISK_PIECES = 4, save_top = 20, output_name = \"test_clicks\")","metadata":{},"execution_count":null,"outputs":[]}]}