{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np\nimport os, sys, pickle, glob, gc, itertools, math, json\nimport cudf\n# from datetime import datetime as dt\n# import matplotlib.pyplot as plt\n# from sklearn.model_selection import StratifiedKFold as skfold\n# from collections import Counter\n# import xgboost as xgb\n\nprint('We will use RAPIDS version',cudf.__version__)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-31T19:49:37.625981Z","iopub.execute_input":"2023-01-31T19:49:37.626338Z","iopub.status.idle":"2023-01-31T19:49:40.511386Z","shell.execute_reply.started":"2023-01-31T19:49:37.626261Z","shell.execute_reply":"2023-01-31T19:49:40.510341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INFER_PTH = {\n    'clicks': {\n        'xgb': {\n            'path': '/kaggle/input/otto--infer-clk-can40-v2-tail40-top404050/XGBRanker_Click_Infer_Can40_V2_Tail40_Top404050.pqt'\n            , 'weight': 1\n        }\n    }\n    , 'carts': {\n        'xgb': {\n            'path': '/kaggle/input/otto--infer-cart-can40-v2-tail40-top404050/XGBRanker_Cart_Infer_Can40_V2_Tail40_Top404050.pqt'\n            , 'weight': 1\n        }\n    }\n    , 'orders': {\n        'xgb': {\n            'path': '/kaggle/input/otto--infer-ord-can40-v2-tail40-top404050/XGBRanker_Ord_Infer_Can40_V2_Tail40_Top404050.pqt'\n            , 'weight': 1\n        }\n    }\n}","metadata":{"execution":{"iopub.status.busy":"2023-01-31T19:49:40.516283Z","iopub.execute_input":"2023-01-31T19:49:40.518557Z","iopub.status.idle":"2023-01-31T19:49:40.526803Z","shell.execute_reply.started":"2023-01-31T19:49:40.518519Z","shell.execute_reply":"2023-01-31T19:49:40.525319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utility Functions","metadata":{}},{"cell_type":"code","source":"def timer(sta):\n    return round((dt.now() - sta).seconds, 3)\n\ndef load_pqt(path):\n    return pd.read_parquet(path)\n\ndef load2cudf(path):\n    return cudf.from_pandas(load_pqt(path))\n\ndef pd2cudf(df):\n    return cudf.from_pandas(df)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T19:49:40.531879Z","iopub.execute_input":"2023-01-31T19:49:40.534772Z","iopub.status.idle":"2023-01-31T19:49:40.542932Z","shell.execute_reply.started":"2023-01-31T19:49:40.534736Z","shell.execute_reply":"2023-01-31T19:49:40.541844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsubmission = []\nfor e in INFER_PTH:\n    print('='*20 + '\\n' + e)\n    print(f'Inferencing {e}...', end='')\n    model_n = len(INFER_PTH[e])\n    for idx, model in enumerate(INFER_PTH[e]):\n        print(f'{model}...', end='')\n        path = INFER_PTH[e][model]['path']\n        weight = INFER_PTH[e][model]['weight']\n        if idx==0:\n            e_cand = load2cudf(path)\n            e_cand = e_cand.rename(columns={'predict_final': model})\n            e_cand[model] = (e_cand[model] * weight).astype('float32')\n        else:\n            tmp = load2cudf(path)\n            tmp = tmp.rename(columns={'predict_final': model})\n            tmp[model] = (tmp[model] * weight).astype('float32')\n            e_cand = e_cand.merge(tmp, on=['session', 'aid'], how='outer')\n    print('done!')\n    \n    print('Selecting top 20...', end='')\n    e_cand = e_cand.fillna(0)\n    cols = list(INFER_PTH[e].keys())\n    e_cand['final_score'] = e_cand[cols].sum(axis=1).astype('float32')\n    e_cand = e_cand.sort_values(['session', 'final_score'], ascending=False)\n    e_cand = e_cand.reset_index(drop=True)\n    e_cand['n'] = e_cand.groupby(['session']).cumcount()\n    e_cand = e_cand.loc[e_cand['n'] < 20]\n    print('done!')\n    \n    print('Joining aid...', end='')\n    e_cand['session'] = e_cand['session'].astype(str) + '_' + e\n    e_cand = e_cand.to_pandas()\n    e_cand = e_cand.groupby('session')['aid'].apply(lambda x: ' '.join(map(str,x)))\n    \n    e_cand = e_cand.reset_index()\n    print('done!')\n    submission.append(e_cand)\n    del e_cand\n    _ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-31T19:49:40.549079Z","iopub.execute_input":"2023-01-31T19:49:40.551390Z","iopub.status.idle":"2023-01-31T19:52:01.434240Z","shell.execute_reply.started":"2023-01-31T19:49:40.551354Z","shell.execute_reply":"2023-01-31T19:52:01.433164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsubmission = pd.concat(submission)\nsubmission = submission.rename(columns={'session':'session_type', 'aid':'labels'})\nsubmission.to_csv('submission.csv', index=False)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-31T19:52:01.435637Z","iopub.execute_input":"2023-01-31T19:52:01.436524Z","iopub.status.idle":"2023-01-31T19:52:18.381109Z","shell.execute_reply.started":"2023-01-31T19:52:01.436484Z","shell.execute_reply":"2023-01-31T19:52:18.379967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ses = 1671803\nlen_submit = len(submission)\nprint('Validating')\nprint(f'len={len_submit}')\nprint(f'sessions = {round(len_submit/3)}/{test_ses} ({round(len_submit/len_submit,2)})')","metadata":{"execution":{"iopub.status.busy":"2023-01-31T19:52:18.382676Z","iopub.execute_input":"2023-01-31T19:52:18.383242Z","iopub.status.idle":"2023-01-31T19:52:18.399555Z","shell.execute_reply.started":"2023-01-31T19:52:18.383203Z","shell.execute_reply":"2023-01-31T19:52:18.398528Z"},"trusted":true},"execution_count":null,"outputs":[]}]}