{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Generate Candidates","metadata":{}},{"cell_type":"code","source":"import pandas as pd, numpy as np\nfrom tqdm.notebook import tqdm\nimport os, sys, pickle, glob, gc\nfrom collections import Counter,defaultdict\nimport  itertools\nimport pyarrow.parquet as pq\nimport json\n","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:01:10.643921Z","iopub.execute_input":"2022-12-22T08:01:10.644798Z","iopub.status.idle":"2022-12-22T08:01:10.771894Z","shell.execute_reply.started":"2022-12-22T08:01:10.644704Z","shell.execute_reply":"2022-12-22T08:01:10.771073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# CACHE FUNCTIONS\ndef read_file(f):\n    return cudf.DataFrame( data_cache[f] )\ndef read_file_to_cache(f):\n    df = pd.read_parquet(f)\n    df.ts = (df.ts/1000).astype('int32')\n    df['type'] = df['type'].map(type_labels).astype('int8')\n    return df\n\n# CACHE THE DATA ON CPU BEFORE PROCESSING ON GPU\ndata_cache = {}\ntype_labels = {'clicks':0, 'carts':1, 'orders':2}\nfiles = glob.glob('../input/otto-chunk-data-inparquet-format/*_parquet/*')\n#for f in files: data_cache[f] = read_file_to_cache(f)\n    \n# CHUNK PARAMETERS\nREAD_CT = 5\nVER = 5\nCHUNK = int( np.ceil( len(files)/6 ))\nprint(f'We will process {len(files)} files, in groups of {READ_CT} and chunks of {CHUNK}.')","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-12-22T08:01:10.773473Z","iopub.execute_input":"2022-12-22T08:01:10.773973Z","iopub.status.idle":"2022-12-22T08:01:10.800759Z","shell.execute_reply.started":"2022-12-22T08:01:10.773942Z","shell.execute_reply":"2022-12-22T08:01:10.799716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files[0]","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:01:10.801986Z","iopub.execute_input":"2022-12-22T08:01:10.802307Z","iopub.status.idle":"2022-12-22T08:01:10.811068Z","shell.execute_reply.started":"2022-12-22T08:01:10.802277Z","shell.execute_reply":"2022-12-22T08:01:10.809996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Generate train & test dataset","metadata":{}},{"cell_type":"code","source":"def read_file_to_cache(f):\n    df = pd.read_parquet(f)\n    df.ts = (df.ts/1000).astype('int32')\n    return df # cudf.DataFrame(df)\n\ndf =  read_file_to_cache('/kaggle/input/train-test-split-feature-generate/train/000000000_000100000.parquet')\ndf","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:01:10.813923Z","iopub.execute_input":"2022-12-22T08:01:10.814535Z","iopub.status.idle":"2022-12-22T08:01:12.283636Z","shell.execute_reply.started":"2022-12-22T08:01:10.814493Z","shell.execute_reply":"2022-12-22T08:01:12.282493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### add more candidate as negative label","metadata":{}},{"cell_type":"code","source":"%%time\nDISK_PIECES=4\ndef pqt_to_dict(df):\n    return df.groupby('aid_x').aid_y.apply(list).to_dict()\n# LOAD THREE CO-VISITATION MATRICES\ntop_20_clicks = pqt_to_dict( pd.read_parquet(f'/kaggle/input/co-visitation-matrix-generation/top_20_clicks_v{VER}_0.pqt') )\nfor k in range(1,DISK_PIECES): \n    top_20_clicks.update( pqt_to_dict( pd.read_parquet(f'/kaggle/input/co-visitation-matrix-generation/top_20_clicks_v{VER}_{k}.pqt') ) )\ntop_20_buys = pqt_to_dict( pd.read_parquet(f'/kaggle/input/co-visitation-matrix-generation/top_15_carts_orders_v{VER}_0.pqt') )\nfor k in range(1,DISK_PIECES): \n    top_20_buys.update( pqt_to_dict( pd.read_parquet(f'/kaggle/input/co-visitation-matrix-generation/top_15_carts_orders_v{VER}_{k}.pqt') ) )\ntop_20_buy2buy = pqt_to_dict( pd.read_parquet(f'/kaggle/input/co-visitation-matrix-generation/top_15_buy2buy_v{VER}_0.pqt') )\n\n# TOP CLICKS AND ORDERS IN Train\ntop_clicks = df.loc[df['type']=='clicks','aid'].value_counts().index.values[:20]\ntop_orders = df.loc[df['type']=='orders','aid'].value_counts().index.values[:20]\n\nprint('Here are size of our 3 co-visitation matrices:')\nprint( len( top_20_clicks ), len( top_20_buy2buy ), len( top_20_buys ) )","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:01:12.285288Z","iopub.execute_input":"2022-12-22T08:01:12.285619Z","iopub.status.idle":"2022-12-22T08:03:11.658221Z","shell.execute_reply.started":"2022-12-22T08:01:12.285589Z","shell.execute_reply":"2022-12-22T08:03:11.657035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type_weight_multipliers = {'clicks': 1, 'carts': 6, 'orders': 3}\n#type_weight_multipliers = {0: 1, 1: 6, 2: 3}\n\ndef suggest_clicks(df):\n    # USER HISTORY AIDS AND TYPES\n    aids=df.aid.tolist()\n    types = df.type.tolist()\n    unique_aids = list(dict.fromkeys(aids[::-1] ))\n    # RERANK CANDIDATES USING WEIGHTS\n    if len(unique_aids)>=20:\n        weights=np.logspace(0.1,1,len(aids),base=2, endpoint=True)-1\n        aids_temp = Counter() \n        # RERANK BASED ON REPEAT ITEMS AND TYPE OF ITEMS\n        for aid,w,t in zip(aids,weights,types): \n            aids_temp[aid] += w * type_weight_multipliers[t]\n        sorted_aids = [k for k,v in aids_temp.most_common(20)]\n        return sorted_aids\n    # USE \"CLICKS\" CO-VISITATION MATRIX\n    aids2 = list(itertools.chain(*[top_20_clicks[aid] for aid in unique_aids if aid in top_20_clicks]))\n    # RERANK CANDIDATES\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids2).most_common(20) if aid2 not in unique_aids]    \n    result = unique_aids + top_aids2[:20 - len(unique_aids)]\n    # USE TOP20 TEST CLICKS\n    return result + list(top_clicks)[:20-len(result)]\n\ndef suggest_buys(df):\n    # USER HISTORY AIDS AND TYPES\n    aids=df.aid.tolist()\n    types = df.type.tolist()\n    # UNIQUE AIDS AND UNIQUE BUYS\n    unique_aids = list(dict.fromkeys(aids[::-1] ))\n    df = df.loc[(df['type']==1)|(df['type']==2)]\n    unique_buys = list(dict.fromkeys( df.aid.tolist()[::-1] ))\n    # RERANK CANDIDATES USING WEIGHTS\n    if len(unique_aids)>=20:\n        weights=np.logspace(0.5,1,len(aids),base=2, endpoint=True)-1\n        aids_temp = Counter() \n        # RERANK BASED ON REPEAT ITEMS AND TYPE OF ITEMS\n        for aid,w,t in zip(aids,weights,types): \n            aids_temp[aid] += w * type_weight_multipliers[t]\n        # RERANK CANDIDATES USING \"BUY2BUY\" CO-VISITATION MATRIX\n        aids3 = list(itertools.chain(*[top_20_buy2buy[aid] for aid in unique_buys if aid in top_20_buy2buy]))\n        for aid in aids3: aids_temp[aid] += 0.1\n        sorted_aids = [k for k,v in aids_temp.most_common(20)]\n        return sorted_aids\n    # USE \"CART ORDER\" CO-VISITATION MATRIX\n    aids2 = list(itertools.chain(*[top_20_buys[aid] for aid in unique_aids if aid in top_20_buys]))\n    # USE \"BUY2BUY\" CO-VISITATION MATRIX\n    aids3 = list(itertools.chain(*[top_20_buy2buy[aid] for aid in unique_buys if aid in top_20_buy2buy]))\n    # RERANK CANDIDATES\n    top_aids2 = [aid2 for aid2, cnt in Counter(aids2+aids3).most_common(20) if aid2 not in unique_aids] \n    result = unique_aids + top_aids2[:20 - len(unique_aids)]\n    # USE TOP20 TEST ORDERS\n    return result + list(top_orders)[:20-len(result)]","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:03:11.660177Z","iopub.execute_input":"2022-12-22T08:03:11.660623Z","iopub.status.idle":"2022-12-22T08:03:11.751379Z","shell.execute_reply.started":"2022-12-22T08:03:11.660581Z","shell.execute_reply":"2022-12-22T08:03:11.750120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnegative_df_clicks = df.sort_values([\"session\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_clicks(x)\n)\n\nnegative_df_buys = df.sort_values([\"session\"]).groupby([\"session\"]).apply(\n    lambda x: suggest_buys(x)\n)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:03:11.753796Z","iopub.execute_input":"2022-12-22T08:03:11.754498Z","iopub.status.idle":"2022-12-22T08:05:22.668647Z","shell.execute_reply.started":"2022-12-22T08:03:11.754447Z","shell.execute_reply":"2022-12-22T08:05:22.667315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"negative_df_clicks","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:05:22.672676Z","iopub.execute_input":"2022-12-22T08:05:22.673031Z","iopub.status.idle":"2022-12-22T08:05:22.684082Z","shell.execute_reply.started":"2022-12-22T08:05:22.672999Z","shell.execute_reply":"2022-12-22T08:05:22.682834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clicks_pred_df = pd.DataFrame(negative_df_clicks, columns=[\"aid\"]).reset_index().explode('aid').reset_index(drop=True)\norders_pred_df = pd.DataFrame(negative_df_buys, columns=[\"aid\"]).reset_index().explode('aid').reset_index(drop=True)\n#carts_pred_df = pd.DataFrame(negative_df_buys, columns=[\"aids\"]).reset_index().explode('aids').reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:05:22.686025Z","iopub.execute_input":"2022-12-22T08:05:22.686499Z","iopub.status.idle":"2022-12-22T08:05:23.700429Z","shell.execute_reply.started":"2022-12-22T08:05:22.686446Z","shell.execute_reply":"2022-12-22T08:05:23.699442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clicks_pred_df ","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:05:23.703585Z","iopub.execute_input":"2022-12-22T08:05:23.703937Z","iopub.status.idle":"2022-12-22T08:05:23.717325Z","shell.execute_reply.started":"2022-12-22T08:05:23.703906Z","shell.execute_reply":"2022-12-22T08:05:23.716080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"negative_df = pd.concat([clicks_pred_df, orders_pred_df])\nnegative_df['Label_carts'] = 0\nnegative_df['Label_clicks'] = 0\nnegative_df['Label_orders'] = 0\nnegative_df","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:05:23.718885Z","iopub.execute_input":"2022-12-22T08:05:23.719265Z","iopub.status.idle":"2022-12-22T08:05:23.910350Z","shell.execute_reply.started":"2022-12-22T08:05:23.719233Z","shell.execute_reply":"2022-12-22T08:05:23.909190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## convert clicks into label","metadata":{}},{"cell_type":"code","source":"df = pd.get_dummies(df, columns = ['type'] ,prefix='Label')\ndf = df.groupby(['session', 'aid']).agg({'Label_carts':'sum', 'Label_clicks':'sum', 'Label_orders':'sum'}).reset_index().sort_values('session').reset_index(drop=True)\ndf.loc[df['Label_carts']>0, 'Label_carts'] = 1\ndf.loc[df['Label_clicks']>0, 'Label_clicks'] = 1\ndf.loc[df['Label_orders']>0, 'Label_orders'] = 1\n","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:05:23.912284Z","iopub.execute_input":"2022-12-22T08:05:23.912731Z","iopub.status.idle":"2022-12-22T08:05:27.742533Z","shell.execute_reply.started":"2022-12-22T08:05:23.912688Z","shell.execute_reply":"2022-12-22T08:05:27.741346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.concat([df, negative_df])","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:05:27.743841Z","iopub.execute_input":"2022-12-22T08:05:27.744198Z","iopub.status.idle":"2022-12-22T08:05:28.098082Z","shell.execute_reply.started":"2022-12-22T08:05:27.744166Z","shell.execute_reply":"2022-12-22T08:05:28.096753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:05:28.099353Z","iopub.execute_input":"2022-12-22T08:05:28.099687Z","iopub.status.idle":"2022-12-22T08:05:28.118787Z","shell.execute_reply.started":"2022-12-22T08:05:28.099657Z","shell.execute_reply":"2022-12-22T08:05:28.117298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## merge features","metadata":{}},{"cell_type":"code","source":"most_clicks_per_users = pd.read_parquet(\"/kaggle/input/train-test-split-feature-generate/most_clicks_per_users.parquet\")\nmost_carts_per_users = pd.read_parquet(\"/kaggle/input/train-test-split-feature-generate/most_carts_per_users.parquet\")\nmost_orders_per_users = pd.read_parquet(\"/kaggle/input/train-test-split-feature-generate/most_orders_per_users.parquet\")\nmost_clicks_for_all = pd.read_parquet(\"/kaggle/input/train-test-split-feature-generate/most_clicks_for_all.parquet\")\nmost_carts_for_all = pd.read_parquet(\"/kaggle/input/train-test-split-feature-generate/most_carts_for_all.parquet\")\nmost_orders_for_all = pd.read_parquet(\"/kaggle/input/train-test-split-feature-generate/most_orders_for_all.parquet\")\ndf_carts_orders_ratio = pd.read_parquet(\"/kaggle/input/train-test-split-feature-generate/df_carts_orders_ratio.parquet\")","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:05:28.120144Z","iopub.execute_input":"2022-12-22T08:05:28.120489Z","iopub.status.idle":"2022-12-22T08:05:50.327713Z","shell.execute_reply.started":"2022-12-22T08:05:28.120459Z","shell.execute_reply":"2022-12-22T08:05:50.326774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_carts_orders_ratio","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:14:30.343719Z","iopub.execute_input":"2022-12-22T08:14:30.344163Z","iopub.status.idle":"2022-12-22T08:14:30.364643Z","shell.execute_reply.started":"2022-12-22T08:14:30.344118Z","shell.execute_reply":"2022-12-22T08:14:30.363723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.merge(most_clicks_per_users, on=['session', 'aid'], how= 'left')","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:05:50.343383Z","iopub.execute_input":"2022-12-22T08:05:50.343694Z","iopub.status.idle":"2022-12-22T08:08:30.576563Z","shell.execute_reply.started":"2022-12-22T08:05:50.343666Z","shell.execute_reply":"2022-12-22T08:08:30.574691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.rename(columns={\"item_count_per_session\": \"item_clicks_per_users\"})\ndf = df.merge(most_carts_per_users, on=['session', 'aid'], how= 'left')\ndf = df.rename(columns={\"item_count_per_session\": \"item_carts_per_users\"})\ndf = df.merge(most_orders_per_users, on=['session', 'aid'], how= 'left')\ndf = df.rename(columns={\"item_count_per_session\": \"item_orders_per_users\"})","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:08:30.579364Z","iopub.execute_input":"2022-12-22T08:08:30.579857Z","iopub.status.idle":"2022-12-22T08:09:01.592363Z","shell.execute_reply.started":"2022-12-22T08:08:30.579807Z","shell.execute_reply":"2022-12-22T08:09:01.591292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.merge(most_clicks_for_all, on=['aid'], how= 'left')\ndf = df.rename(columns={\"item_count_per_session\": \"most_clicks_for_all\"})\n\ndf = df.merge(most_carts_for_all, on=['aid'], how= 'left')\ndf = df.rename(columns={\"item_count_per_session\": \"most_carts_for_all\"})\n\ndf = df.merge(most_orders_for_all, on=['aid'], how= 'left')\ndf = df.rename(columns={\"item_count_per_session\": \"most_orders_for_all\"})","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:13:45.354371Z","iopub.execute_input":"2022-12-22T08:13:45.355721Z","iopub.status.idle":"2022-12-22T08:14:06.251474Z","shell.execute_reply.started":"2022-12-22T08:13:45.355663Z","shell.execute_reply":"2022-12-22T08:14:06.250445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.merge(df_carts_orders_ratio, on=['session'], how= 'left')\n","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:14:48.340559Z","iopub.execute_input":"2022-12-22T08:14:48.340993Z","iopub.status.idle":"2022-12-22T08:14:57.347067Z","shell.execute_reply.started":"2022-12-22T08:14:48.340958Z","shell.execute_reply":"2022-12-22T08:14:57.345867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.fillna(0)\ndf","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:15:19.546402Z","iopub.execute_input":"2022-12-22T08:15:19.546789Z","iopub.status.idle":"2022-12-22T08:15:23.193069Z","shell.execute_reply.started":"2022-12-22T08:15:19.546760Z","shell.execute_reply":"2022-12-22T08:15:23.192048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df[['session', 'aid',\n       'item_clicks_per_users', 'item_carts_per_users',\n       'item_orders_per_users', 'most_clicks_for_all', 'most_carts_for_all',\n       'most_orders_for_all', 'total_click', 'total_carts', 'total_order',\n       'cart_2_order_ratio', 'click_2_order_ratio', 'click_2_cart_ratio',\n        'Label_clicks','Label_carts', 'Label_orders']]\ndf['cart_2_order_ratio'] = np.where(df['cart_2_order_ratio']>1, 1, df['cart_2_order_ratio']>1)","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:19:07.457219Z","iopub.execute_input":"2022-12-22T08:19:07.457668Z","iopub.status.idle":"2022-12-22T08:19:08.967333Z","shell.execute_reply.started":"2022-12-22T08:19:07.457630Z","shell.execute_reply":"2022-12-22T08:19:08.966087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## train xgb","metadata":{}},{"cell_type":"code","source":"import xgboost as xgb\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.multioutput import MultiOutputClassifier\n\ndf.aid = df.aid.astype('int32')\ndf = df.fillna(0)\nskf = GroupKFold(n_splits=5)\nfor fold,(train_idx, valid_idx) in enumerate(skf.split(df, groups=df['session'] )):\n\n    X_train = df.iloc[train_idx, :-3]\n    y_train = df.iloc[train_idx, -3:]\n    X_valid = df.iloc[valid_idx, :-3]\n    y_valid = df.iloc[valid_idx, -3:]\n\n    \n    # create XGBoost instance with default hyper-parameters\n    xgb_estimator = xgb.XGBClassifier(objective='binary:logistic', verbosity=1, tree_method='hist' )\n\n    model = xgb_estimator.fit(\n        X_train, y_train)\n    model.save_model(f'XGB_fold{fold}_click.xgb')","metadata":{"execution":{"iopub.status.busy":"2022-12-22T08:19:44.619921Z","iopub.execute_input":"2022-12-22T08:19:44.620835Z","iopub.status.idle":"2022-12-22T08:28:40.061243Z","shell.execute_reply.started":"2022-12-22T08:19:44.620792Z","shell.execute_reply":"2022-12-22T08:28:40.059861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{}}]}