{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook is to explore and experiment on @radek1's notebook [simplified without need for chunking](https://www.kaggle.com/code/radek1/last-20-aids) and the [original code](https://www.kaggle.com/code/ttahara/last-aid-20) is by @ttahara","metadata":{}},{"cell_type":"markdown","source":"## rd: recsys - otto - last 20 aids - The purpose of this notebook - Grabbing the last 20 aid for each session of the test set, use them as prediction and submit or run local validation to see how powerful can the last 20 aids be.","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"## import dataset and functions","metadata":{}},{"cell_type":"code","source":"import os\n\ntry: import fastkaggle\nexcept ModuleNotFoundError:\n    os.system(\"pip install -Uq fastkaggle\")\n\nfrom fastkaggle import *\n\n# use fastdebug.utils \nif iskaggle: os.system(\"pip install nbdev snoop\")\n\nif iskaggle:\n    path = \"../input/fastdebugutils0\"\n    import sys\n    sys.path\n    sys.path.insert(1, path)\n    import utils as fu\n    from utils import *\nelse: \n    from fastdebug.utils import *\n    import fastdebug.utils as fu","metadata":{"execution":{"iopub.status.busy":"2022-11-18T13:30:29.494242Z","iopub.execute_input":"2022-11-18T13:30:29.494684Z","iopub.status.idle":"2022-11-18T13:30:51.590731Z","shell.execute_reply.started":"2022-11-18T13:30:29.49465Z","shell.execute_reply":"2022-11-18T13:30:51.588479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ntrain = pd.read_parquet('../input/otto-full-optimized-memory-footprint/train.parquet')\ntest = pd.read_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')\n\n!pip install pickle5\nimport pickle5 as pickle\n\nwith open('../input/otto-full-optimized-memory-footprint/id2type.pkl', \"rb\") as fh:\n    id2type = pickle.load(fh)\nwith open('../input/otto-full-optimized-memory-footprint/type2id.pkl', \"rb\") as fh:\n    type2id = pickle.load(fh)\n    \nsample_sub = pd.read_csv('../input/otto-recommender-system/sample_submission.csv')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-18T13:20:03.724545Z","iopub.execute_input":"2022-11-18T13:20:03.725074Z","iopub.status.idle":"2022-11-18T13:20:42.772998Z","shell.execute_reply.started":"2022-11-18T13:20:03.72496Z","shell.execute_reply":"2022-11-18T13:20:42.771612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-18T13:20:57.615352Z","iopub.execute_input":"2022-11-18T13:20:57.615774Z","iopub.status.idle":"2022-11-18T13:20:57.635736Z","shell.execute_reply.started":"2022-11-18T13:20:57.615737Z","shell.execute_reply":"2022-11-18T13:20:57.634198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-18T13:21:02.223228Z","iopub.execute_input":"2022-11-18T13:21:02.223659Z","iopub.status.idle":"2022-11-18T13:21:02.235787Z","shell.execute_reply.started":"2022-11-18T13:21:02.223615Z","shell.execute_reply":"2022-11-18T13:21:02.234449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### rd: recsys - otto - last 20 aid - sort df by two columns - test.sort_values(['session', 'ts'])","metadata":{}},{"cell_type":"code","source":"%%time\n\ntest = test.sort_values(['session', 'ts'])\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-18T13:21:59.647169Z","iopub.execute_input":"2022-11-18T13:21:59.647685Z","iopub.status.idle":"2022-11-18T13:22:03.527072Z","shell.execute_reply.started":"2022-11-18T13:21:59.647643Z","shell.execute_reply":"2022-11-18T13:22:03.525976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### rd: recsys - otto - last 20 aid - take last 20 aids from each session - test.groupby('session')['aid'].apply(lambda x: list(x)[-20:])","metadata":{}},{"cell_type":"code","source":"session_aids = test.groupby('session')['aid'].apply(lambda x: list(x)[-20:])","metadata":{"execution":{"iopub.status.busy":"2022-11-18T13:26:04.517776Z","iopub.execute_input":"2022-11-18T13:26:04.51821Z","iopub.status.idle":"2022-11-18T13:26:37.206532Z","shell.execute_reply.started":"2022-11-18T13:26:04.518176Z","shell.execute_reply":"2022-11-18T13:26:37.205383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-11-18T13:27:08.943689Z","iopub.execute_input":"2022-11-18T13:27:08.944114Z","iopub.status.idle":"2022-11-18T13:27:08.949609Z","shell.execute_reply.started":"2022-11-18T13:27:08.944081Z","shell.execute_reply":"2022-11-18T13:27:08.948242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"session_aids.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-18T13:31:04.637319Z","iopub.execute_input":"2022-11-18T13:31:04.638323Z","iopub.status.idle":"2022-11-18T13:31:04.648716Z","shell.execute_reply.started":"2022-11-18T13:31:04.638285Z","shell.execute_reply":"2022-11-18T13:31:04.647641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"session_aids.index","metadata":{"execution":{"iopub.status.busy":"2022-11-18T13:35:19.23793Z","iopub.execute_input":"2022-11-18T13:35:19.238457Z","iopub.status.idle":"2022-11-18T13:35:19.24633Z","shell.execute_reply.started":"2022-11-18T13:35:19.238421Z","shell.execute_reply":"2022-11-18T13:35:19.245143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### rd: recsys - otto - last 20 aid - loop through a Series with session as index with a list as the only column's value - for session, aids in session_aids.iteritems():","metadata":{}},{"cell_type":"markdown","source":"### rd: recsys - otto - last 20 aid - turn a list into a string with values connected with empty space - labels.append(' '.join([str(a) for a in aids]))","metadata":{}},{"cell_type":"code","source":"%%time\n\nsession_type = []\nlabels = []\nsession_types = ['clicks', 'carts', 'orders']\n\nfor session, aids in session_aids.iteritems():\n    for st in session_types:\n        session_type.append(f'{session}_{st}')\n        labels.append(' '.join([str(a) for a in aids]))","metadata":{"execution":{"iopub.status.busy":"2022-11-18T13:38:59.637744Z","iopub.execute_input":"2022-11-18T13:38:59.638148Z","iopub.status.idle":"2022-11-18T13:39:09.762999Z","shell.execute_reply.started":"2022-11-18T13:38:59.638116Z","shell.execute_reply":"2022-11-18T13:39:09.762196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### rd: recsys - otto - last 20 aid - make a df from a dict with two lists as values - pd.DataFrame({'session_type': session_type, 'labels': labels})","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame({'session_type': session_type, 'labels': labels})","metadata":{"execution":{"iopub.status.busy":"2022-11-18T13:40:05.700925Z","iopub.execute_input":"2022-11-18T13:40:05.701327Z","iopub.status.idle":"2022-11-18T13:40:06.806699Z","shell.execute_reply.started":"2022-11-18T13:40:05.701295Z","shell.execute_reply":"2022-11-18T13:40:06.805484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-18T13:40:06.809192Z","iopub.execute_input":"2022-11-18T13:40:06.809691Z","iopub.status.idle":"2022-11-18T13:40:06.820881Z","shell.execute_reply.started":"2022-11-18T13:40:06.809645Z","shell.execute_reply":"2022-11-18T13:40:06.819607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}