{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-24T10:13:44.418366Z","iopub.execute_input":"2022-12-24T10:13:44.418977Z","iopub.status.idle":"2022-12-24T10:13:44.466057Z","shell.execute_reply.started":"2022-12-24T10:13:44.418852Z","shell.execute_reply":"2022-12-24T10:13:44.464933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datetime import datetime\nfrom tqdm import tqdm\n\nfrom collections import defaultdict\nimport math\nimport numpy as np\nimport random\nimport copy\nfrom collections import Counter","metadata":{"execution":{"iopub.status.busy":"2022-12-24T10:13:47.198051Z","iopub.execute_input":"2022-12-24T10:13:47.198488Z","iopub.status.idle":"2022-12-24T10:13:47.204880Z","shell.execute_reply.started":"2022-12-24T10:13:47.198454Z","shell.execute_reply":"2022-12-24T10:13:47.203286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"對物件結構提供了一個二進制序列化與反序列化功能的模組。\n1. 資料永久儲存\n2. 資料傳輸\n3. 最小化資料","metadata":{}},{"cell_type":"code","source":"!pip install pickle5","metadata":{"execution":{"iopub.status.busy":"2022-12-24T10:13:49.292365Z","iopub.execute_input":"2022-12-24T10:13:49.292779Z","iopub.status.idle":"2022-12-24T10:14:04.847689Z","shell.execute_reply.started":"2022-12-24T10:13:49.292748Z","shell.execute_reply":"2022-12-24T10:14:04.846220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle5 as pickle","metadata":{"execution":{"iopub.status.busy":"2022-12-24T10:14:07.873273Z","iopub.execute_input":"2022-12-24T10:14:07.873764Z","iopub.status.idle":"2022-12-24T10:14:07.888801Z","shell.execute_reply.started":"2022-12-24T10:14:07.873723Z","shell.execute_reply":"2022-12-24T10:14:07.887382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_parquet('/kaggle/input/clear-train/clear_train.parquet')\ntest_df = pd.read_parquet('/kaggle/input/otto-full-optimized-memory-footprint/test.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-12-24T10:14:09.638819Z","iopub.execute_input":"2022-12-24T10:14:09.639313Z","iopub.status.idle":"2022-12-24T10:14:41.223139Z","shell.execute_reply.started":"2022-12-24T10:14:09.639273Z","shell.execute_reply":"2022-12-24T10:14:41.221345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-24T09:51:49.619055Z","iopub.execute_input":"2022-12-24T09:51:49.619460Z","iopub.status.idle":"2022-12-24T09:51:49.630794Z","shell.execute_reply.started":"2022-12-24T09:51:49.619429Z","shell.execute_reply":"2022-12-24T09:51:49.629143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"rb : read binary","metadata":{}},{"cell_type":"code","source":"with open('../input/otto-full-optimized-memory-footprint/id2type.pkl', \"rb\") as fh:\n    id2type = pickle.load(fh)\nwith open('../input/otto-full-optimized-memory-footprint/type2id.pkl', \"rb\") as fh:\n    type2id = pickle.load(fh)\n    \nsample_sub_df = pd.read_csv('../input/otto-recommender-system/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-24T10:14:51.931395Z","iopub.execute_input":"2022-12-24T10:14:51.931831Z","iopub.status.idle":"2022-12-24T10:14:59.651566Z","shell.execute_reply.started":"2022-12-24T10:14:51.931798Z","shell.execute_reply":"2022-12-24T10:14:59.650214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(id2type)","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:58:12.863048Z","iopub.execute_input":"2022-12-24T08:58:12.863419Z","iopub.status.idle":"2022-12-24T08:58:12.869456Z","shell.execute_reply.started":"2022-12-24T08:58:12.863395Z","shell.execute_reply":"2022-12-24T08:58:12.868445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(id2type)","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:58:18.387899Z","iopub.execute_input":"2022-12-24T08:58:18.388550Z","iopub.status.idle":"2022-12-24T08:58:18.394824Z","shell.execute_reply.started":"2022-12-24T08:58:18.388516Z","shell.execute_reply":"2022-12-24T08:58:18.393561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(type2id)","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:58:20.570051Z","iopub.execute_input":"2022-12-24T08:58:20.570426Z","iopub.status.idle":"2022-12-24T08:58:20.576657Z","shell.execute_reply.started":"2022-12-24T08:58:20.570401Z","shell.execute_reply":"2022-12-24T08:58:20.575308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub_df","metadata":{"execution":{"iopub.status.busy":"2022-12-16T09:24:03.159754Z","iopub.execute_input":"2022-12-16T09:24:03.160174Z","iopub.status.idle":"2022-12-16T09:24:03.178816Z","shell.execute_reply.started":"2022-12-16T09:24:03.160141Z","shell.execute_reply":"2022-12-16T09:24:03.177247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"定義取樣的 unique session 數量","metadata":{}},{"cell_type":"code","source":"config = {\n    'train_session_num':3500000,\n}\n","metadata":{"execution":{"iopub.status.busy":"2022-12-24T10:15:12.377181Z","iopub.execute_input":"2022-12-24T10:15:12.377606Z","iopub.status.idle":"2022-12-24T10:15:12.384004Z","shell.execute_reply.started":"2022-12-24T10:15:12.377569Z","shell.execute_reply":"2022-12-24T10:15:12.382269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Syntax : random.sample(sequence, k)\n\nParameters:\nsequence: Can be a list, tuple, string, or set.\nk: An Integer value, it specify the length of a sample.\n\nReturns: k length new list of elements chosen from the sequence.","metadata":{}},{"cell_type":"code","source":"train_session = random.sample(list(train_df['session'].unique()),config['train_session_num'])\ntrain_df = train_df.query('session in @train_session').reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-24T10:15:15.458126Z","iopub.execute_input":"2022-12-24T10:15:15.458598Z","iopub.status.idle":"2022-12-24T10:15:33.659120Z","shell.execute_reply.started":"2022-12-24T10:15:15.458564Z","shell.execute_reply":"2022-12-24T10:15:33.657613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"astype() : convert to type","metadata":{}},{"cell_type":"code","source":"train_df['aid'] = train_df['aid'].astype('int32').astype('str')\ntest_df['aid'] = test_df['aid'].astype('int32').astype('str')\n","metadata":{"execution":{"iopub.status.busy":"2022-12-24T10:15:36.872730Z","iopub.execute_input":"2022-12-24T10:15:36.873405Z","iopub.status.idle":"2022-12-24T10:16:18.870521Z","shell.execute_reply.started":"2022-12-24T10:15:36.873356Z","shell.execute_reply":"2022-12-24T10:16:18.869044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-12-24T07:43:46.410075Z","iopub.execute_input":"2022-12-24T07:43:46.410461Z","iopub.status.idle":"2022-12-24T07:43:46.420402Z","shell.execute_reply.started":"2022-12-24T07:43:46.410431Z","shell.execute_reply":"2022-12-24T07:43:46.418948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-12-24T07:43:50.341413Z","iopub.execute_input":"2022-12-24T07:43:50.341775Z","iopub.status.idle":"2022-12-24T07:43:50.365508Z","shell.execute_reply.started":"2022-12-24T07:43:50.341748Z","shell.execute_reply":"2022-12-24T07:43:50.364155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"strftime 格式化時間\n\nSeries.dt.strftime(\\*args, \\*\\*kwargs)\nConvert to Index using specified date_format.\n\nReturn an Index of formatted strings specified by date_format, which supports the same string format as the python standard library. ","metadata":{}},{"cell_type":"code","source":"train_df['time_stamp'] = pd.to_datetime(train_df['ts'],unit='s').dt.strftime('%Y-%m-%d')\ntest_df['time_stamp'] = pd.to_datetime(test_df['ts'],unit='s').dt.strftime('%Y-%m-%d')","metadata":{"execution":{"iopub.status.busy":"2022-12-24T10:17:43.734972Z","iopub.execute_input":"2022-12-24T10:17:43.736604Z","iopub.status.idle":"2022-12-24T10:23:38.116379Z","shell.execute_reply.started":"2022-12-24T10:17:43.736527Z","shell.execute_reply":"2022-12-24T10:23:38.112503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['time_stamp']","metadata":{"execution":{"iopub.status.busy":"2022-12-24T07:51:07.396280Z","iopub.execute_input":"2022-12-24T07:51:07.396790Z","iopub.status.idle":"2022-12-24T07:51:07.409212Z","shell.execute_reply.started":"2022-12-24T07:51:07.396745Z","shell.execute_reply":"2022-12-24T07:51:07.407821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Series.value_counts(normalize=False, sort=True, ascending=False, bins=None, dropna=True)\n\nReturn a Series containing counts of unique values.","metadata":{}},{"cell_type":"code","source":"def gen_pairs(df):\n    df=df.sort_values(by=['session','ts'])\n    #平移取下一樣商品\n    df['aid_next'] = df['aid'].shift(-1)\n    #把最後一筆刪掉 start\n    #分天\n    df['session_day'] = df['session'].astype('str')+'_'+df['time_stamp']\n    #操作天數紀錄\n    df['session_day_count'] = df['session_day'].map(df['session_day'].value_counts())\n    #每天做ranking first:相同時依照在array的order排序 ascending:時間早的在前面\n    df['ranking'] = df.groupby(['session_day'])['ts'].rank(method='first', ascending=True)\n    #把最後一筆刪掉 end\n    df = df.query('session_day_count!=ranking').reset_index(drop=True)\n    \n    #依照商品做運算 計算最常接著後50個商品\n    sim_aids = df.groupby('aid').apply(lambda df: Counter(df.aid_next).most_common(50)).to_dict()\n    #算出出現次數\n    sim_aids = {aid: Counter(dict(top)) for aid, top in sim_aids.items()}\n    return sim_aids\n    ","metadata":{"execution":{"iopub.status.busy":"2022-12-24T10:25:29.300234Z","iopub.execute_input":"2022-12-24T10:25:29.300905Z","iopub.status.idle":"2022-12-24T10:25:29.316090Z","shell.execute_reply.started":"2022-12-24T10:25:29.300854Z","shell.execute_reply":"2022-12-24T10:25:29.314281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sim_aids = gen_pairs(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-12-24T10:25:33.094871Z","iopub.execute_input":"2022-12-24T10:25:33.095371Z","iopub.status.idle":"2022-12-24T10:34:10.203418Z","shell.execute_reply.started":"2022-12-24T10:25:33.095332Z","shell.execute_reply":"2022-12-24T10:34:10.201665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sim_aids","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:36:07.572749Z","iopub.execute_input":"2022-12-24T08:36:07.573098Z","iopub.status.idle":"2022-12-24T08:36:07.916460Z","shell.execute_reply.started":"2022-12-24T08:36:07.573067Z","shell.execute_reply":"2022-12-24T08:36:07.915501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def recommend(aids,popular_items):\n    \n    #足夠20個直接拿\n    if len(aids) >= 20:\n        return aids[-20:]\n\n    #不夠的在從裡面的生成\n    aids = set(aids)\n    new_aids = Counter()\n    for aid in aids:\n        new_aids.update(sim_aids.get(aid, Counter()))\n    \n    #最常出現然後不在一開始的列表中\n    top_aids2 = [aid2 for aid2, cnt in new_aids.most_common(40) if aid2 not in aids] \n    final_rec_list = list(aids) + top_aids2[:20 - len(aids)]\n    \n    #如果還是不夠再去從最熱門商品直接拿，再不夠就算了\n    if len(final_rec_list)<20:\n        return final_rec_list + popular_items[:20-len(final_rec_list)]\n    else:\n        return final_rec_list","metadata":{"execution":{"iopub.status.busy":"2022-12-24T10:34:54.253328Z","iopub.execute_input":"2022-12-24T10:34:54.253758Z","iopub.status.idle":"2022-12-24T10:34:54.264467Z","shell.execute_reply.started":"2022-12-24T10:34:54.253725Z","shell.execute_reply":"2022-12-24T10:34:54.262702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_df.sort_values([\"session\", \"type\", \"ts\"])\n#session 對應的 aid 轉成 list 在轉成 dict\n# session : [aid1, aid2,...]\ntest_session_dict = test_df.groupby('session')['aid'].agg(list).to_dict()\nsession_id_list = []\nitem_id_list = []\n\npopular_items = list(train_df['aid'].value_counts().index)\n\n#tqdm 進度條\nfor session_id,session_item_list in tqdm(test_session_dict.items()):\n    item_list = recommend(session_item_list,popular_items)\n    \n    session_id_list.append(session_id)\n    item_id_list.append(list(item_list))\n\nres_df = pd.DataFrame()\nres_df['session_type'] = session_id_list\nres_df['labels'] = [' '.join([str(l) for l in lls]) for lls in item_id_list]","metadata":{"execution":{"iopub.status.busy":"2022-12-24T10:34:57.291342Z","iopub.execute_input":"2022-12-24T10:34:57.292093Z","iopub.status.idle":"2022-12-24T10:39:28.786934Z","shell.execute_reply.started":"2022-12-24T10:34:57.292056Z","shell.execute_reply":"2022-12-24T10:39:28.785541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res_list = []\nfor type_ in [0,1,2]:\n    temp_df = copy.deepcopy(res_df)\n    temp_df['session_type'] = temp_df['session_type'].apply(lambda x:'{}_{}'.format(x,id2type[type_]))\n    res_list.append(temp_df)\nres_df = pd.concat(res_list,axis=0)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-24T10:40:49.845316Z","iopub.execute_input":"2022-12-24T10:40:49.846530Z","iopub.status.idle":"2022-12-24T10:40:53.219494Z","shell.execute_reply.started":"2022-12-24T10:40:49.846476Z","shell.execute_reply":"2022-12-24T10:40:53.217666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"res_df.to_csv('submission2.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-12-24T10:40:56.114897Z","iopub.execute_input":"2022-12-24T10:40:56.115442Z","iopub.status.idle":"2022-12-24T10:41:16.895977Z","shell.execute_reply.started":"2022-12-24T10:40:56.115399Z","shell.execute_reply":"2022-12-24T10:41:16.894518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import FileLink\nFileLink(r'submission2.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-24T10:42:30.492310Z","iopub.execute_input":"2022-12-24T10:42:30.492820Z","iopub.status.idle":"2022-12-24T10:42:30.505252Z","shell.execute_reply.started":"2022-12-24T10:42:30.492782Z","shell.execute_reply":"2022-12-24T10:42:30.503752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}