{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Setup","metadata":{}},{"cell_type":"code","source":"GLOBAL_SEED = 42\n\nimport os\nos.environ['PYTHONHASHSEED'] = str(GLOBAL_SEED)\nimport sys\nfrom multiprocessing import cpu_count\nimport gc\nfrom tqdm import tqdm\nimport datetime\nimport pickle\nimport random as rnd\nfrom glob import glob\n\nimport pandas as pd\nimport numpy as np\nfrom numpy import random as np_rnd\nfrom collections import Counter, defaultdict\n\nfrom scipy.stats import rankdata","metadata":{"execution":{"iopub.status.busy":"2023-01-31T14:34:27.987938Z","iopub.execute_input":"2023-01-31T14:34:27.988532Z","iopub.status.idle":"2023-01-31T14:34:28.947203Z","shell.execute_reply.started":"2023-01-31T14:34:27.988421Z","shell.execute_reply":"2023-01-31T14:34:28.945857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=42):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    # python random\n    rnd.seed(seed)\n    # numpy random\n    np_rnd.seed(seed)\n    # tf random\n    try:\n        tf_rnd.set_seed(seed)\n    except:\n        pass\n    # RAPIDS random\n    try:\n        cupy.random.seed(seed)\n    except:\n        pass\n    # pytorch random\n    try:\n        torch.manual_seed(seed)\n        torch.cuda.manual_seed(seed)\n        torch.backends.cudnn.deterministic = True\n    except:\n        pass\n\ndef create_get_ts(ts):\n    return int((ts.replace(tzinfo=CFG.tz) - CFG.ts_zero).total_seconds())\n\ndef pickleIO(obj, src, op=\"w\"):\n    if op==\"w\":\n        with open(src, op + \"b\") as f:\n            pickle.dump(obj, f)\n    elif op==\"r\":\n        with open(src, op + \"b\") as f:\n            tmp = pickle.load(f)\n        return tmp\n    else:\n        print(\"unknown operation\")\n        return obj","metadata":{"execution":{"iopub.status.busy":"2023-01-31T14:34:28.952001Z","iopub.execute_input":"2023-01-31T14:34:28.953350Z","iopub.status.idle":"2023-01-31T14:34:29.138267Z","shell.execute_reply.started":"2023-01-31T14:34:28.953284Z","shell.execute_reply":"2023-01-31T14:34:29.136137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    debug = False\n    embed_dim = 32\n    tz = datetime.timezone.utc\n    ts_zero = datetime.datetime(1970, 1, 1, tzinfo=tz)\n    contentType_mapper = pd.Series([\"clicks\", \"carts\", \"orders\"], index=[0, 1, 2])\n    target_weight = (0.1, 0.3, 0.6)","metadata":{"execution":{"iopub.status.busy":"2023-01-31T14:34:29.140583Z","iopub.execute_input":"2023-01-31T14:34:29.141210Z","iopub.status.idle":"2023-01-31T14:34:29.159539Z","shell.execute_reply.started":"2023-01-31T14:34:29.141156Z","shell.execute_reply":"2023-01-31T14:34:29.157595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CFG.train_dates = (create_get_ts(datetime.datetime(2022, 8, 15, 0, 0)), create_get_ts(datetime.datetime(2022, 8, 22, 0, 0)))\nCFG.valid_dates = (create_get_ts(datetime.datetime(2022, 8, 22, 0, 0)), create_get_ts(datetime.datetime(2022, 8, 29, 0, 0)))","metadata":{"execution":{"iopub.status.busy":"2023-01-31T14:34:29.164920Z","iopub.execute_input":"2023-01-31T14:34:29.166001Z","iopub.status.idle":"2023-01-31T14:34:29.174815Z","shell.execute_reply.started":"2023-01-31T14:34:29.165938Z","shell.execute_reply":"2023-01-31T14:34:29.173519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Retrieval From Multiple Algorithms & Ranking With Handcrafted Rule","metadata":{}},{"cell_type":"code","source":"weight_dic = {\n    \"rerank_raw_output\": {\"weight\": 0.2 * 0.5, \"pred_type\": \"same\"},\n    \"rerankbyaid_raw_output\": {\"weight\": 0.4 * 0.5, \"pred_type\": \"diff\"},\n    \"rerankbytime_raw_output\": {\"weight\": 0.4 * 0.5, \"pred_type\": \"diff\"},\n    \"covisit_raw_output\": {\"weight\": 0.2 * 0.3, \"pred_type\": \"same\"},\n    \"covisit_withtime_raw_output\": {\"weight\": 0.2 * 0.3, \"pred_type\": \"same\"},\n    \"o2o_raw_output\": {\"weight\": 0.6 * 0.3, \"pred_type\": \"diff\"},\n    \"w2v_raw_output\": {\"weight\": 0.5 * 0.2, \"pred_type\": \"same\"},\n    \"mf_raw_output\": {\"weight\": 0.5 * 0.2, \"pred_type\": \"same\"},\n#     \"ubf_raw_output\": {\"weight\": 0.1 * 0.2, \"pred_type\": \"same\"},\n#     \"trend_raw_output\": {\"weight\": 0.1 * 0.2, \"pred_type\": \"same\"},\n    \"candidates_rerank\": {\"weight\": 0.25, \"pred_type\": \"diff\"},\n    \"tuning_rerank\": {\"weight\": 0.25, \"pred_type\": \"diff\"},\n    \"otto_pipeline2\": {\"weight\": 0.25, \"pred_type\": \"diff\"},\n    \"final_ensemble\": {\"weight\": 0.25, \"pred_type\": \"diff\"},\n#     \"otto-easy-understanding-for-beginner-\": {\"weight\": 0.03, \"pred_type\": \"diff\"},\n}\n\nsum([weight_dic[i][\"weight\"] for i in weight_dic.keys()])","metadata":{"execution":{"iopub.status.busy":"2023-01-31T14:34:29.177811Z","iopub.execute_input":"2023-01-31T14:34:29.178422Z","iopub.status.idle":"2023-01-31T14:34:29.196992Z","shell.execute_reply.started":"2023-01-31T14:34:29.178372Z","shell.execute_reply":"2023-01-31T14:34:29.195419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mymodel_paths = [i for i in sorted(glob(\"/kaggle/input/otto-retrieval-stage-raw-output/*\"))if i.split(\".\")[-1] == \"parquet\"]\nmymodel_list = [i.split(\"/\")[-1].split(\".\")[0] for i in mymodel_paths]\nmymodel_paths = [i for i in mymodel_paths if i.split(\"/\")[-1].split(\".\")[0] in weight_dic]\n\npublic_nb_paths = [i for i in sorted(glob(\"/kaggle/input/otto-preprocessing-public-notebooks/*\")) if i.split(\".\")[-1] == \"parquet\"]\npublic_nb_list = [i.split(\"/\")[-1].split(\".\")[0] for i in public_nb_paths]\npublic_nb_paths = [i for i in public_nb_paths if i.split(\"/\")[-1].split(\".\")[0] in weight_dic]\n\nmodel_paths = mymodel_paths + public_nb_paths","metadata":{"execution":{"iopub.status.busy":"2023-01-29T04:54:31.161387Z","iopub.execute_input":"2023-01-29T04:54:31.161812Z","iopub.status.idle":"2023-01-29T04:54:31.172839Z","shell.execute_reply.started":"2023-01-29T04:54:31.161772Z","shell.execute_reply":"2023-01-29T04:54:31.171814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{}},{"cell_type":"code","source":"def get_retrieval_items(aid_type):\n    t = aid_type\n    retrieval_aids = defaultdict(Counter)\n    for i in tqdm(model_paths[:2] if CFG.debug else model_paths):\n        df_output = pd.read_parquet(i).iloc[:1000] if CFG.debug else pd.read_parquet(i)\n        df_output = df_output[df_output[\"type\"] == t].reset_index(drop=True)\n        \n        model_aids = df_output[\"rec\"].apply(lambda x: [int(j) for j in x.split()])\n        model_score = df_output[\"score\"].apply(lambda x: [float(j) for j in x.split()])\n        model_weight = weight_dic[i.split(\"/\")[-1].split(\".\")[0]][\"weight\"]\n        model_pred_type = weight_dic[i.split(\"/\")[-1].split(\".\")[0]][\"pred_type\"]\n        mymodel_weight = 0.7\n        model_source_weight = mymodel_weight if i.split(\"/\")[-1].split(\".\")[0] in mymodel_list else (1 - mymodel_weight)\n        \n#         # Method 1 : Each model's candidates are equal weight\n#         for sess, aid, score in tqdm(zip(model_aids.index, model_aids, model_score), total=len(model_aids)):\n#             retrieval_aids[sess] += Counter(dict(zip(aid, np.ones(len(aid)) * model_weight)))\n\n        # Method 2 : Each model's candidates are normalized weight\n        for sess, aid, score in zip(df_output[\"session\"].values, model_aids, model_score):\n            score = rankdata(score) / len(score)\n            candidates = Counter(dict(zip(aid, score * model_weight * model_source_weight)))\n            retrieval_aids[sess] += candidates\n        \n#         # Method 3 : Each model's candidates are normalized weight with multiplier\n#         for sess, aid, score in tqdm(zip(model_aids.index, model_aids, model_score), total=len(model_aids)):\n#             # Transform Lienar shape score to Exponential shape score\n#             score = (rankdata(score) / len(score)) ** 3\n#             retrieval_aids[t][sess] += Counter(dict(zip(aid, score * model_weight)))\n\n    return retrieval_aids","metadata":{"execution":{"iopub.status.busy":"2023-01-27T04:46:04.432508Z","iopub.execute_input":"2023-01-27T04:46:04.432957Z","iopub.status.idle":"2023-01-27T04:46:04.442575Z","shell.execute_reply.started":"2023-01-27T04:46:04.432931Z","shell.execute_reply":"2023-01-27T04:46:04.441432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nn_aids = 20\n\nfor t in [0, 1, 2]:\n    print(f\"=== type {t} ===\")\n    output = {\n        \"session\": [],\n        \"type\": [],\n        \"rec\": [],\n        \"score\": [],\n    }\n    \n    # Get Items\n    retrieval_aids = get_retrieval_items(t)\n    \n    # Inference\n    for SESS in tqdm(retrieval_aids.keys()):\n        \n        rec, score = zip(*retrieval_aids[SESS].most_common(n_aids))\n        \n        output[\"session\"].append(SESS)\n        output[\"type\"].append(t)\n        output[\"rec\"].append(\" \".join(pd.Series(rec, dtype=\"str\").values))\n        output[\"score\"].append(\" \".join(pd.Series(np.round(score, 5), dtype=\"str\").values))\n        \n    pd.DataFrame({\n        'session':pd.Series(output[\"session\"], dtype='int32'),\n        'type':pd.Series(output[\"type\"], dtype='int8'),\n        'rec':pd.Series(output[\"rec\"], dtype='str'),\n        'score':pd.Series(output[\"score\"], dtype='str'),\n    }).to_parquet(f\"./ranking_raw_output_type{t}.parquet\")\n    \n    del output, retrieval_aids; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-27T04:46:04.608422Z","iopub.execute_input":"2023-01-27T04:46:04.608773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.concat([pd.read_parquet(f\"./ranking_raw_output_type{t}.parquet\") for t in [0, 1, 2]], axis=0, ignore_index=False).sort_values([\"session\", \"type\"]).set_index([\"session\", \"type\"])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.reset_index().to_parquet(\"./ranking_raw_output.parquet\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"output[\"session_type\"] = [str(i[0]) + \"_\" + str(CFG.contentType_mapper[i[1]]) for i in output.index]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(\"/kaggle/input/otto-recommender-system/sample_submission.csv\")\nsubmission = submission.set_index(\"session_type\")\nsubmission.loc[output[\"session_type\"].values, \"labels\"] = output[\"rec\"].values\nsubmission = submission.reset_index()\nsubmission.to_csv(\"./submission.csv\", index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del output, submission; gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get Local Recall Score","metadata":{}},{"cell_type":"code","source":"def get_submission_with(labels):\n    labels_as_strings = [' '.join([str(l) for l in lls]) for lls in labels]\n    predictions = pd.DataFrame(data={'session_type': test_session_AIDs.index, \n                                    'labels': labels_as_strings})\n    prediction_dfs = []\n\n    for st in session_types:\n        modified_predictions = predictions.copy() \n        modified_predictions.session_type = modified_predictions.session_type.astype('str') + f'_{st}'\n        prediction_dfs.append(modified_predictions)\n\n    submission = pd.concat(prediction_dfs).reset_index(drop=True) \n    return submission\n\nmapper = {\"clicks\": 0, \"carts\": 1, \"orders\": 2}\n# get_submission_with 에서 나온 submission (session_type 과 labels 칼럼으로 구성) 된 것을 아래 get_local_score에 활용\ndef get_local_score(submission):\n    submission['session'] = submission.session_type.apply(lambda x: int(x.split('_')[0]))\n    submission['type'] = submission.session_type.apply(lambda x: x.split('_')[1])\n    submission['type'] = submission['type'].map(mapper).astype(\"int32\").values\n    try:\n        submission.labels = submission.labels.apply(lambda x: [int(i) for i in x.split(' ')[:20]])\n    except:\n        submission.labels.apply(lambda x: [int(i) for i in x[:20]])\n    \n    test_labels = pd.read_parquet('/kaggle/input/otto-full-optimized-memory-footprint/test.parquet')\n    test_labels = test_labels.reset_index(drop =True).groupby(['session','type'])['aid'].apply(list).reset_index()\n    test_labels.columns = ['session', 'type', 'ground_truth']\n    test_labels = test_labels.merge(submission, how='left', on=['session', 'type'])\n    test_labels['hits'] = test_labels.apply(lambda df: len(set(df.ground_truth).intersection(set(df.labels))), axis=1)\n    test_labels['gt_count'] = test_labels.ground_truth.apply(lambda x : len(set(x))).clip(0, 20)\n\n    recall_per_type = test_labels.groupby(['type'])['hits'].sum() / test_labels.groupby(['type'])['gt_count'].sum() \n\n    score = (recall_per_type * pd.Series({0: 0.10, 1: 0.30, 2: 0.60})).sum()\n    \n    return score, test_labels","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"local_recall_score = get_local_score(pd.read_csv(\"./submission.csv\"))\nprint(\"Recall :\", local_recall_score[0])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}