{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-19T11:21:35.148552Z","iopub.execute_input":"2022-12-19T11:21:35.149070Z","iopub.status.idle":"2022-12-19T11:21:35.180187Z","shell.execute_reply.started":"2022-12-19T11:21:35.148945Z","shell.execute_reply":"2022-12-19T11:21:35.178870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datetime import datetime\nfrom tqdm import tqdm\n\nfrom collections import defaultdict\nimport math\nimport numpy as np\nimport random\nimport copy\nfrom collections import Counter","metadata":{"execution":{"iopub.status.busy":"2022-12-19T11:21:35.182431Z","iopub.execute_input":"2022-12-19T11:21:35.183174Z","iopub.status.idle":"2022-12-19T11:21:35.188853Z","shell.execute_reply.started":"2022-12-19T11:21:35.183127Z","shell.execute_reply":"2022-12-19T11:21:35.187916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pickle5","metadata":{"execution":{"iopub.status.busy":"2022-12-19T11:31:43.798517Z","iopub.execute_input":"2022-12-19T11:31:43.798935Z","iopub.status.idle":"2022-12-19T11:31:58.397147Z","shell.execute_reply.started":"2022-12-19T11:31:43.798890Z","shell.execute_reply":"2022-12-19T11:31:58.395887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle5 as pickle","metadata":{"execution":{"iopub.status.busy":"2022-12-19T11:32:11.195165Z","iopub.execute_input":"2022-12-19T11:32:11.195581Z","iopub.status.idle":"2022-12-19T11:32:11.208575Z","shell.execute_reply.started":"2022-12-19T11:32:11.195537Z","shell.execute_reply":"2022-12-19T11:32:11.207579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_parquet('../input/otto-full-optimized-memory-footprint/train.parquet')\ntest_df = pd.read_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-12-19T11:32:13.706908Z","iopub.execute_input":"2022-12-19T11:32:13.707359Z","iopub.status.idle":"2022-12-19T11:32:35.298587Z","shell.execute_reply.started":"2022-12-19T11:32:13.707322Z","shell.execute_reply":"2022-12-19T11:32:35.297603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-16T09:22:38.162635Z","iopub.execute_input":"2022-12-16T09:22:38.163047Z","iopub.status.idle":"2022-12-16T09:22:38.172148Z","shell.execute_reply.started":"2022-12-16T09:22:38.163015Z","shell.execute_reply":"2022-12-16T09:22:38.170591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('../input/otto-full-optimized-memory-footprint/id2type.pkl', \"rb\") as fh:\n    id2type = pickle.load(fh)\nwith open('../input/otto-full-optimized-memory-footprint/type2id.pkl', \"rb\") as fh:\n    type2id = pickle.load(fh)\n    \nsample_sub_df = pd.read_csv('../input/otto-recommender-system/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-19T11:33:04.905953Z","iopub.execute_input":"2022-12-19T11:33:04.906409Z","iopub.status.idle":"2022-12-19T11:33:11.704077Z","shell.execute_reply.started":"2022-12-19T11:33:04.906367Z","shell.execute_reply":"2022-12-19T11:33:11.703160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(id2type)","metadata":{"execution":{"iopub.status.busy":"2022-12-16T09:23:05.054826Z","iopub.execute_input":"2022-12-16T09:23:05.055291Z","iopub.status.idle":"2022-12-16T09:23:05.062910Z","shell.execute_reply.started":"2022-12-16T09:23:05.055240Z","shell.execute_reply":"2022-12-16T09:23:05.061736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(id2type)","metadata":{"execution":{"iopub.status.busy":"2022-12-16T09:23:08.196604Z","iopub.execute_input":"2022-12-16T09:23:08.196996Z","iopub.status.idle":"2022-12-16T09:23:08.207525Z","shell.execute_reply.started":"2022-12-16T09:23:08.196965Z","shell.execute_reply":"2022-12-16T09:23:08.205975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(type2id)","metadata":{"execution":{"iopub.status.busy":"2022-12-16T09:23:20.826908Z","iopub.execute_input":"2022-12-16T09:23:20.827353Z","iopub.status.idle":"2022-12-16T09:23:20.832942Z","shell.execute_reply.started":"2022-12-16T09:23:20.827314Z","shell.execute_reply":"2022-12-16T09:23:20.832094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub_df","metadata":{"execution":{"iopub.status.busy":"2022-12-16T09:24:03.159754Z","iopub.execute_input":"2022-12-16T09:24:03.160174Z","iopub.status.idle":"2022-12-16T09:24:03.178816Z","shell.execute_reply.started":"2022-12-16T09:24:03.160141Z","shell.execute_reply":"2022-12-16T09:24:03.177247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config = {\n    'train_session_num':3000000,\n}\n","metadata":{"execution":{"iopub.status.busy":"2022-12-19T11:33:30.433770Z","iopub.execute_input":"2022-12-19T11:33:30.434208Z","iopub.status.idle":"2022-12-19T11:33:30.439375Z","shell.execute_reply.started":"2022-12-19T11:33:30.434174Z","shell.execute_reply":"2022-12-19T11:33:30.438050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_session = random.sample(list(train_df['session'].unique()),config['train_session_num'])\ntrain_df = train_df.query('session in @train_session').reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-19T11:36:20.576297Z","iopub.execute_input":"2022-12-19T11:36:20.578748Z","iopub.status.idle":"2022-12-19T11:36:39.582942Z","shell.execute_reply.started":"2022-12-19T11:36:20.578690Z","shell.execute_reply":"2022-12-19T11:36:39.581431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['aid'] = train_df['aid'].astype('int32').astype('str')\ntest_df['aid'] = test_df['aid'].astype('int32').astype('str')\n","metadata":{"execution":{"iopub.status.busy":"2022-12-19T11:36:54.017044Z","iopub.execute_input":"2022-12-19T11:36:54.017814Z","iopub.status.idle":"2022-12-19T11:37:31.855710Z","shell.execute_reply.started":"2022-12-19T11:36:54.017769Z","shell.execute_reply":"2022-12-19T11:37:31.854586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-19T11:37:41.194239Z","iopub.execute_input":"2022-12-19T11:37:41.194654Z","iopub.status.idle":"2022-12-19T11:37:41.203170Z","shell.execute_reply.started":"2022-12-19T11:37:41.194621Z","shell.execute_reply":"2022-12-19T11:37:41.201973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-12-19T11:37:45.060214Z","iopub.execute_input":"2022-12-19T11:37:45.060604Z","iopub.status.idle":"2022-12-19T11:37:45.079040Z","shell.execute_reply.started":"2022-12-19T11:37:45.060573Z","shell.execute_reply":"2022-12-19T11:37:45.077823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['time_stamp'] = pd.to_datetime(train_df['ts'],unit='s').dt.strftime('%Y-%m-%d')\ntest_df['time_stamp'] = pd.to_datetime(test_df['ts'],unit='s').dt.strftime('%Y-%m-%d')","metadata":{"execution":{"iopub.status.busy":"2022-12-19T11:38:47.305508Z","iopub.execute_input":"2022-12-19T11:38:47.306052Z","iopub.status.idle":"2022-12-19T11:43:35.570512Z","shell.execute_reply.started":"2022-12-19T11:38:47.306006Z","shell.execute_reply":"2022-12-19T11:43:35.569379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['time_stamp']","metadata":{"execution":{"iopub.status.busy":"2022-12-19T11:46:33.852666Z","iopub.execute_input":"2022-12-19T11:46:33.854251Z","iopub.status.idle":"2022-12-19T11:46:33.869593Z","shell.execute_reply.started":"2022-12-19T11:46:33.854170Z","shell.execute_reply":"2022-12-19T11:46:33.868267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def gen_pairs(df):\n    df=df.sort_values(by=['session','ts'])\n    df['aid_next'] = df['aid'].shift(-1)\n    df['session_day'] = df['session'].astype('str')+'_'+df['time_stamp']\n    df['session_day_count'] = df['session_day'].map(df['session_day'].value_counts())\n    df['ranking'] = df.groupby(['session_day'])['ts'].rank(method='first', ascending=True)\n    df = df.query('session_day_count!=ranking').reset_index(drop=True)\n    \n    sim_aids = df.groupby('aid').apply(lambda df: Counter(df.aid_next).most_common(50)).to_dict()\n    sim_aids = {aid: Counter(dict(top)) for aid, top in sim_aids.items()}\n    return sim_aids\n    ","metadata":{"execution":{"iopub.status.busy":"2022-12-19T11:46:44.215738Z","iopub.execute_input":"2022-12-19T11:46:44.216199Z","iopub.status.idle":"2022-12-19T11:46:44.229756Z","shell.execute_reply.started":"2022-12-19T11:46:44.216157Z","shell.execute_reply":"2022-12-19T11:46:44.228317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sim_aids = gen_pairs(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-12-19T11:46:51.086873Z","iopub.execute_input":"2022-12-19T11:46:51.087372Z","iopub.status.idle":"2022-12-19T11:53:32.050449Z","shell.execute_reply.started":"2022-12-19T11:46:51.087327Z","shell.execute_reply":"2022-12-19T11:53:32.049219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sim_aids","metadata":{"execution":{"iopub.status.busy":"2022-12-19T12:02:01.637277Z","iopub.execute_input":"2022-12-19T12:02:01.638107Z","iopub.status.idle":"2022-12-19T12:02:01.996218Z","shell.execute_reply.started":"2022-12-19T12:02:01.638059Z","shell.execute_reply":"2022-12-19T12:02:01.995068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def recommend(aids,popular_items):\n    \n    if len(aids) >= 20:\n        return aids[-20:]\n\n    aids = set(aids)\n    new_aids = Counter()\n    for aid in aids:\n        new_aids.update(sim_aids.get(aid, Counter()))\n    \n    top_aids2 = [aid2 for aid2, cnt in new_aids.most_common(40) if aid2 not in aids] \n    final_rec_list = list(aids) + top_aids2[:20 - len(aids)]\n    \n    if len(final_rec_list)<20:\n        return final_rec_list + popular_items[:20-len(final_rec_list)]\n    else:\n        return final_rec_list","metadata":{"execution":{"iopub.status.busy":"2022-12-19T12:08:09.288017Z","iopub.execute_input":"2022-12-19T12:08:09.288580Z","iopub.status.idle":"2022-12-19T12:08:09.299217Z","shell.execute_reply.started":"2022-12-19T12:08:09.288535Z","shell.execute_reply":"2022-12-19T12:08:09.297659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_df.sort_values([\"session\", \"type\", \"ts\"])\ntest_session_dict = test_df.groupby('session')['aid'].agg(list).to_dict()\nsession_id_list = []\nitem_id_list = []\n\npopular_items = list(train_df['aid'].value_counts().index)\n\nfor session_id,session_item_list in tqdm(test_session_dict.items()):\n    item_list = recommend(session_item_list,popular_items)\n    \n    session_id_list.append(session_id)\n    item_id_list.append(list(item_list))\n\nres_df = pd.DataFrame()\nres_df['session_type'] = session_id_list\nres_df['labels'] = [' '.join([str(l) for l in lls]) for lls in item_id_list]","metadata":{"execution":{"iopub.status.busy":"2022-12-19T12:08:34.241203Z","iopub.execute_input":"2022-12-19T12:08:34.241634Z","iopub.status.idle":"2022-12-19T12:12:33.874183Z","shell.execute_reply.started":"2022-12-19T12:08:34.241599Z","shell.execute_reply":"2022-12-19T12:12:33.872781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res_list = []\nfor type_ in [0,1,2]:\n    temp_df = copy.deepcopy(res_df)\n    temp_df['session_type'] = temp_df['session_type'].apply(lambda x:'{}_{}'.format(x,id2type[type_]))\n    res_list.append(temp_df)\nres_df = pd.concat(res_list,axis=0)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-19T12:15:27.874102Z","iopub.execute_input":"2022-12-19T12:15:27.875533Z","iopub.status.idle":"2022-12-19T12:15:31.649468Z","shell.execute_reply.started":"2022-12-19T12:15:27.875467Z","shell.execute_reply":"2022-12-19T12:15:31.648299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res_df.to_csv('submission1.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-19T12:15:38.070885Z","iopub.execute_input":"2022-12-19T12:15:38.072250Z","iopub.status.idle":"2022-12-19T12:15:59.137423Z","shell.execute_reply.started":"2022-12-19T12:15:38.072189Z","shell.execute_reply":"2022-12-19T12:15:59.136153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}