{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Otto - GPU Inference\n\n* This notebook implements **GPU inference** of [Chris Deotte's notebook] using CUDF.\n* evaluation time becomes from **27min to 4sec**, which means **x400 faster** than the original notebook \n* I minor-changed suggestion logic for faster inference.\n\n---\n\nThis notebook is based on [Chris Deotte's notebook].\nIf you think this notebook usefull, **please also upvote the original notebook**.\n\n[Chris Deotte's notebook]: https://www.kaggle.com/code/cdeotte/compute-validation-score-cv-565","metadata":{}},{"cell_type":"markdown","source":"# Credits\n\nWe thank many Kagglers who have shared ideas. We use co-visitation matrix idea from Vladimir [here][1]. We use groupby sort logic from Sinan in comment section [here][4]. We use duplicate prediction removal logic from Radek [here][5]. We use multiple visit logic from Pietro [here][2]. We use type weighting logic from Ingvaras [here][3]. We use leaky test data from my previous notebook [here][4]. And some ideas may have originated from Tawara [here][6] and KJ [here][7]. We use Colum2131's parquets [here][8]. Above image is from Ravi's discussion about candidate rerank models [here][9]\n\n[1]: https://www.kaggle.com/code/vslaykovsky/co-visitation-matrix\n[2]: https://www.kaggle.com/code/pietromaldini1/multiple-clicks-vs-latest-items\n[3]: https://www.kaggle.com/code/ingvarasgalinskas/item-type-vs-multiple-clicks-vs-latest-items\n[4]: https://www.kaggle.com/code/cdeotte/test-data-leak-lb-boost\n[5]: https://www.kaggle.com/code/radek1/co-visitation-matrix-simplified-imprvd-logic\n[6]: https://www.kaggle.com/code/ttahara/otto-mors-aid-frequency-baseline\n[7]: https://www.kaggle.com/code/whitelily/co-occurrence-baseline\n[8]: https://www.kaggle.com/datasets/columbia2131/otto-chunk-data-inparquet-format\n[9]: https://www.kaggle.com/competitions/otto-recommender-system/discussion/364721\n[10]: https://www.kaggle.com/cdeotte/candidate-rerank-model-lb-0-574\n[11]: https://www.kaggle.com/competitions/otto-recommender-system/discussion/364991","metadata":{"execution":{"iopub.status.busy":"2022-12-18T18:45:14.266436Z","iopub.execute_input":"2022-12-18T18:45:14.267092Z","iopub.status.idle":"2022-12-18T18:45:14.312651Z","shell.execute_reply.started":"2022-12-18T18:45:14.267056Z","shell.execute_reply":"2022-12-18T18:45:14.310089Z"}}},{"cell_type":"markdown","source":"# Step 1 - Candidate Generation with RAPIDS\n\nThis logic was moved to [other notebook].\n\n[other notebook]: https://www.kaggle.com/code/tatamikenn/otto-gpu-inference-make-covis-mat","metadata":{}},{"cell_type":"markdown","source":"# Step 2 - ReRank (choose 20) using handcrafted rules","metadata":{"papermill":{"duration":0.017902,"end_time":"2022-11-16T20:18:07.176982","exception":false,"start_time":"2022-11-16T20:18:07.159080","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%writefile \"predict.py\"\n\nimport sys\n\nimport os, sys, pickle, glob, gc\nfrom collections.abc import Iterable\nfrom pathlib import Path\n\nimport pandas as pd, numpy as np\nfrom tqdm.notebook import tqdm\nfrom collections import Counter\nimport cudf, cupy, itertools\n\nsys.path.append(\"/kaggle/input/otto-gpu-inference-make-covis-mat\")\nfrom config import *\n\nprint(\"We will use RAPIDS version\", cudf.__version__)\n\n\ndef load_parquet_by_glob(glob, path=Path(\".\")):\n    return pd.concat(\n        pd.read_parquet(parquet_file)\n        for parquet_file in path.glob(glob)\n    )\n\n\ndef load_test(delta=1e-16, evaluate=True):\n    if evaluate:\n        test_df = load_parquet_by_glob(\"../input/otto-validation/test_parquet/*\")\n    else:\n        # TODO: adapt to multiple parquet file\n        test_df = load_parquet_by_glob(\"../input/otto-parquet/test.parquet\")        \n        code2type = {0: \"clicks\", 1: \"carts\", 2: \"orders\"}\n        test_df[\"type\"] = test_df[\"type\"].map(code2type)\n\n    # optimize memory usage\n    test_df[\"session\"] = test_df[\"session\"].astype(\"uint32\")\n    test_df[\"aid\"] = test_df[\"aid\"].astype(\"uint32\")\n    test_df[\"ts\"] = (test_df[\"ts\"] / 1000).astype(\"uint32\")\n\n    # type to w_type\n    test_df[\"type\"] = test_df[\"type\"].map(type_labels).astype(np.int8)\n    type2weight = {0: 1, 1: 6, 2: 3}\n    test_df[\"type\"] = test_df[\"type\"].map(type2weight).astype(\"float32\")\n\n    # CPU -> GPU\n    test_df = cudf.from_pandas(test_df)\n\n    # normalize ts\n    test_df = test_df.merge(\n        test_df.groupby(\"session\")[\"ts\"].agg([\"min\", \"max\"]).reset_index(), on=\"session\"\n    )\n    test_df[\"ts\"] = (\n        (test_df[\"ts\"] - test_df[\"min\"]) / (test_df[\"max\"] - test_df[\"min\"] + delta)\n    ).astype(\"float32\")\n\n    # wgt\n    test_df[\"wgt\"] = (test_df[\"type\"] * test_df[\"ts\"]).astype(\"float32\")\n    return test_df.reset_index(drop=True)\n\n\ndef normalize_weight(df):\n    df[\"wgt\"] = df[\"wgt\"] / df[\"wgt\"] / df[\"aid_x\"].map(df.groupby(\"aid_x\")[\"wgt\"].sum())\n    return df\n\n\ndef assert_weight_normalized(df):\n    assert np.isclose(1.0, df.groupby(\"aid_x\")[\"wgt\"].transform(\"sum\").to_numpy()).all()\n    return True\n\n\ndef suggest(\n    df, covis_mats, n_suggest=20, agg_method=\"sum\", sort_key=\"ts\", wgt_gain=10.0\n):\n    df = df.copy()\n\n    unique_aids = df[\"aid\"].unique()\n    df = df.sort_values([\"session\", sort_key])\n    df = df.drop_duplicates(subset=[\"session\", \"aid\"], keep=\"last\")\n    df = df.merge(\n        cudf.concat(\n            [\n                *covis_mats,\n                cudf.DataFrame(\n                    {\"aid_x\": unique_aids, \"aid_y\": unique_aids, \"wgt\": wgt_gain}\n                ),\n            ]\n        ),\n        left_on=\"aid\",\n        right_on=\"aid_x\",\n        how=\"left\",\n    )\n    df = df.groupby([\"session\", \"aid_y\"])[\"wgt_y\"].agg(agg_method).reset_index()\n    df = df.sort_values([\"session\", \"wgt_y\"], ascending=False).reset_index(drop=True)\n    df[\"priority\"] = df.groupby(\"session\").cumcount()\n    df = df[df[\"priority\"] < 20]\n    df = df.groupby(\"session\")[\"aid_y\"].agg(list).rename(\"labels\")\n    return df.reset_index()\n\n\ndef evaluate(pred_df, gt_labels, n_suggest=20):\n    # Pred labels\n    pred_df = pred_df.copy()\n    pred_df[\"priority\"] = pred_df.groupby([\"type\", \"session\"]).cumcount()\n    pred_df = pred_df[pred_df[\"priority\"] < n_suggest]\n    pred_df = pred_df.set_index([\"session\", \"type\", \"labels\"])\n\n    # GT labels\n    gt_labels = gt_labels.rename({\"ground_truth\": \"labels\"}, axis=1)\n    gt_labels = gt_labels.explode(\"labels\")\n    gt_labels = gt_labels.set_index([\"session\", \"type\", \"labels\"])\n\n    # Correct labels\n    correct_index = gt_labels.index.intersection(pred_df.index)\n    correct_df = pred_df.loc[correct_index].reset_index()\n\n    # Correct counts\n    correct_counts = correct_df.groupby(\"type\")[[\"labels\"]].agg(\"count\")\n\n    # GT counts\n    gt_counts = (\n        gt_labels.reset_index().groupby([\"session\", \"type\"]).agg(\"count\").reset_index()\n    )\n    filter_label_count = gt_counts[\"labels\"] > n_suggest\n    gt_counts.loc[filter_label_count, \"labels\"] = n_suggest\n    gt_counts = gt_counts.groupby(\"type\")[[\"labels\"]].agg(\"sum\")\n\n    eval_score = (correct_counts / gt_counts).T\n    eval_score[\"total\"] = (\n        eval_score[\"clicks\"] * 0.1\n        + eval_score[\"carts\"] * 0.3\n        + eval_score[\"orders\"] * 0.6\n    )\n    eval_score.rename(\n        {\n            \"clicks\": f\"recall@{n_suggest}(click)\",\n            \"carts\": f\"recall@{n_suggest}(cart)\",\n            \"orders\": f\"recall@{n_suggest}(order)\",\n            \"total\": f\"recall@{n_suggest}\",\n        },\n        axis=1,\n    )\n    return eval_score\n\n\ndef load_covis_mat(path, glob):\n    covis_mat = load_parquet_by_glob(glob, path)\n    covis_mat[\"aid_x\"] = covis_mat[\"aid_x\"].astype(\"uint32\")\n    covis_mat[\"aid_y\"] = covis_mat[\"aid_y\"].astype(\"uint32\")\n    covis_mat[\"wgt\"] = covis_mat[\"wgt\"].astype(\"float32\")\n    covis_mat = cudf.from_pandas(covis_mat)\n    return covis_mat\n\n\ntest_df = load_test(evaluate=EVALUATE)\nprint(\"Test data has shape\", test_df.shape)\n\nfrom pathlib import Path\nINPUT = Path(\"/kaggle/input/otto-gpu-inference-make-covis-mat\")\n\ncovis_clicks = load_covis_mat(INPUT, f\"top_{TOP_N_CLICKS}_clicks_v{VER}{suffix}_*\")\ncovis_carts_orders = load_covis_mat(INPUT, f\"top_{TOP_N_CARTS_ORDERS}_carts_orders_v{VER}{suffix}_*\")\n#covis_buy2buy = load_covis_mat(INPUT, f\"top_{TOP_N_BUY2BUY}_buy2buy_v{VER}{suffix}_*\")\n\ncovis_carts_orders = normalize_weight(covis_carts_orders)\ncovis_clicks = normalize_weight(covis_clicks)\n#covis_buy2buy = normalize_weight(covis_buy2buy)\n#covis_buy2buy[\"wgt\"] = covis_buy2buy[\"wgt\"] * 0.75\n\ncovis_carts_orders = cudf.concat([covis_carts_orders, covis_clicks])\ncovis_carts_orders = covis_carts_orders.groupby([\"aid_x\", \"aid_y\"])[\"wgt\"].sum().reset_index()\n\nparams = dict(agg_method=\"sum\", sort_key=\"wgt\", wgt_gain=10.0)\nsuggest_click = suggest(test_df, [covis_clicks], **params)\ndel covis_clicks\nsuggest_buy = suggest(test_df, [covis_carts_orders], **params)\n\n\npred_clicks = suggest_click.copy()\npred_clicks[\"type\"] = \"clicks\"\npred_carts = suggest_buy.copy()\npred_carts[\"type\"] = \"carts\"\npred_orders = suggest_buy.copy()\npred_orders[\"type\"] = \"orders\"\n\npred_df = cudf.concat([pred_clicks, pred_carts, pred_orders])\npred_df = pred_df.explode(\"labels\").reset_index(drop=True)\ndel suggest_buy, suggest_click, pred_clicks, pred_carts, pred_orders\n\nif EVALUATE:\n    gt_labels = cudf.read_parquet(\"../input/otto-validation/test_labels.parquet\")\n    print(evaluate(pred_df, gt_labels, 20))\nelse:\n    assert len(set(test_df[\"session\"].unique().to_arrow().to_pylist()).difference(\n        set(pred_df[\"session\"].unique().to_arrow().to_pylist()))) == 0\n    \n    pred_df[\"labels\"] = pred_df[\"labels\"].astype(str)\n    pred_df = pred_df.groupby([\"type\", \"session\"])[\"labels\"].agg(list).reset_index()\n    pred_df[\"session_type\"] = pred_df[\"session\"].astype(str) + \"_\" + pred_df[\"type\"].astype(str)\n    pred_df = pred_df.drop([\"type\", \"session\"], axis=1)\n    pred_df = pred_df.to_pandas()\n    pred_df[\"labels\"] = pred_df[\"labels\"].transform(lambda x: \" \".join(x))\n    pred_df = pred_df[[\"session_type\", \"labels\"]]\n    pred_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-19T01:21:44.756956Z","iopub.execute_input":"2022-12-19T01:21:44.757325Z","iopub.status.idle":"2022-12-19T01:21:44.767721Z","shell.execute_reply.started":"2022-12-19T01:21:44.757290Z","shell.execute_reply":"2022-12-19T01:21:44.766661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!time python predict.py","metadata":{"execution":{"iopub.status.busy":"2022-12-19T01:21:45.204353Z","iopub.execute_input":"2022-12-19T01:21:45.204826Z","iopub.status.idle":"2022-12-19T01:23:04.043191Z","shell.execute_reply.started":"2022-12-19T01:21:45.204787Z","shell.execute_reply":"2022-12-19T01:23:04.042016Z"},"trusted":true},"execution_count":null,"outputs":[]}]}