{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**\\[edited\\]**\n\nThe unusually high score 0.947 (before recalculation)  was probably due to a bug in the metrics\n\nThe recalculated score 0.467 appears to be a normal score\n\nThanks to kaggler for finding the bug and to the organizer s for the  quick response!\n\n\n**-- below this was written before the recalculation --**\n\n\nI wrote a very simple implementation of the first sample submit, with code to output the last time 20 `aid` for each sessino of the provided test data\n\nThis code got 0.947 in the Leaderboard...\n\nI think, is there an overlap of `ts` between the provided test data and the Leaderboard computed data?\n\nIf anyone knows the reason for this high score, please let me know. :pray:","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom tqdm import tqdm\n\ndef load_jsonl(load_path, max_load_chunk):\n    chunks = pd.read_json(load_path, lines=True, chunksize=100_000)\n    \n    dfs = []\n    for e, chunk in tqdm(enumerate(chunks)):\n        if e > max_load_chunk:\n            break\n        event_dict = {\"session\": [], \"aid\": [], \"ts\": [], \"type\": []}\n        for session, events in zip(chunk[\"session\"].tolist(), chunk[\"events\"].tolist()):\n            for event in events:\n                event_dict[\"session\"].append(session)\n                event_dict[\"aid\"].append(event[\"aid\"])\n                event_dict[\"ts\"].append(event[\"ts\"])\n                event_dict[\"type\"].append(event[\"type\"])\n        dfs.append(pd.DataFrame(event_dict))\n\n    return pd.concat(dfs).reset_index(drop=True).astype({\"ts\": \"datetime64[ms]\"})","metadata":{"execution":{"iopub.status.busy":"2022-11-02T13:27:05.838274Z","iopub.execute_input":"2022-11-02T13:27:05.839422Z","iopub.status.idle":"2022-11-02T13:27:05.850423Z","shell.execute_reply.started":"2022-11-02T13:27:05.839377Z","shell.execute_reply":"2022-11-02T13:27:05.848522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = load_jsonl(\"../input/otto-recommender-system/test.jsonl\", max_load_chunk=10**5)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T13:27:05.852918Z","iopub.execute_input":"2022-11-02T13:27:05.853363Z","iopub.status.idle":"2022-11-02T13:27:50.435335Z","shell.execute_reply.started":"2022-11-02T13:27:05.853325Z","shell.execute_reply":"2022-11-02T13:27:50.433905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = test_df.sort_values([\"session\", \"type\", \"ts\"]).groupby([\"session\"]).apply(\n    lambda x: x.tail(20).aid.tolist()\n)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-02T13:27:50.437106Z","iopub.execute_input":"2022-11-02T13:27:50.437511Z","iopub.status.idle":"2022-11-02T13:32:19.147624Z","shell.execute_reply.started":"2022-11-02T13:27:50.437475Z","shell.execute_reply":"2022-11-02T13:32:19.146385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clicks_pred_df = pd.DataFrame(pred_df.add_suffix(\"_clicks\"), columns=[\"labels\"]).reset_index()\norders_pred_df = pd.DataFrame(pred_df.add_suffix(\"_orders\"), columns=[\"labels\"]).reset_index()\ncarts_pred_df = pd.DataFrame(pred_df.add_suffix(\"_carts\"), columns=[\"labels\"]).reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-11-02T13:32:19.149614Z","iopub.execute_input":"2022-11-02T13:32:19.150066Z","iopub.status.idle":"2022-11-02T13:32:22.710314Z","shell.execute_reply.started":"2022-11-02T13:32:19.150033Z","shell.execute_reply":"2022-11-02T13:32:22.709017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.concat(\n    [clicks_pred_df, orders_pred_df, carts_pred_df]\n)\npred_df.columns = [\"session_type\", \"labels\"]\npred_df[\"labels\"] = pred_df.labels.apply(lambda x: \" \".join(map(str,x)))\npred_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T13:32:22.711865Z","iopub.execute_input":"2022-11-02T13:32:22.712497Z","iopub.status.idle":"2022-11-02T13:32:38.921554Z","shell.execute_reply.started":"2022-11-02T13:32:22.712446Z","shell.execute_reply":"2022-11-02T13:32:38.920377Z"},"trusted":true},"execution_count":null,"outputs":[]}]}