{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":38760,"databundleVersionId":4493939,"sourceType":"competition"}],"dockerImageVersionId":30301,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# word2vec Tutorial\nFor a detailed explanation of the word2vec, please refer to the following paper.\n\n[Efficient Estation of Word Representations in Vector Space](https://arxiv.org/abs/1301.3781)","metadata":{}},{"cell_type":"markdown","source":"# Read Train&Test Data","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\n\nfrom pathlib import Path\n\ndata_path = Path('/kaggle/input/otto-recommender-system/')","metadata":{"execution":{"iopub.status.busy":"2022-12-05T18:29:17.922196Z","iopub.execute_input":"2022-12-05T18:29:17.922638Z","iopub.status.idle":"2022-12-05T18:29:18.045433Z","shell.execute_reply.started":"2022-12-05T18:29:17.922599Z","shell.execute_reply":"2022-12-05T18:29:18.044315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_jsonl(target: str) -> pd.DataFrame():\n    sessions = pd.DataFrame()\n    chunks = pd.read_json(data_path / f'{target}.jsonl', lines=True, chunksize=100_000)\n\n    for e, chunk in enumerate(chunks):\n        event_dict = {\n            'session': [],\n            'aid': [],\n            'ts': [],\n            'type': [],\n        }\n        if e < 2:\n            for session, events in zip(chunk['session'].tolist(), chunk['events'].tolist()):\n                for event in events:\n                    event_dict['session'].append(session)\n                    event_dict['aid'].append(event['aid'])\n                    event_dict['ts'].append(event['ts'])\n                    event_dict['type'].append(event['type'])\n            chunk_session = pd.DataFrame(event_dict)\n            sessions = pd.concat([sessions, chunk_session])\n        else:\n            break\n    return sessions.reset_index(drop=True)\n\ntrain_sessions = read_jsonl('train')\ntest_sessions = read_jsonl('test')","metadata":{"execution":{"iopub.status.busy":"2022-12-05T18:29:20.241201Z","iopub.execute_input":"2022-12-05T18:29:20.241625Z","iopub.status.idle":"2022-12-05T18:30:24.348078Z","shell.execute_reply.started":"2022-12-05T18:29:20.241587Z","shell.execute_reply":"2022-12-05T18:30:24.346572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# word2vec","metadata":{}},{"cell_type":"code","source":"import hashlib\nimport os\nfrom gensim.models import word2vec\nfrom gensim.models import KeyedVectors\nos.environ[\"PYTHONHASHSEED\"] = str(42)\ndef hashfxn(x):\n    return int(hashlib.md5(str(x).encode()).hexdigest(), 16)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-05T18:40:46.544225Z","iopub.execute_input":"2022-12-05T18:40:46.545601Z","iopub.status.idle":"2022-12-05T18:40:47.006737Z","shell.execute_reply.started":"2022-12-05T18:40:46.545546Z","shell.execute_reply":"2022-12-05T18:40:47.005362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_corpus = []\nfor session, group_df in tqdm(train_sessions.groupby(['session'])):\n    raw_corpus.append(list(group_df['aid'].astype(str) + '_' + group_df['type']))\nfor session, group_df in tqdm(test_sessions.groupby(['session'])):\n    raw_corpus.append(list(group_df['aid'].astype(str) + '_' + group_df['type']))","metadata":{"execution":{"iopub.status.busy":"2022-12-05T18:40:55.773929Z","iopub.execute_input":"2022-12-05T18:40:55.774332Z","iopub.status.idle":"2022-12-05T18:44:59.951583Z","shell.execute_reply.started":"2022-12-05T18:40:55.774302Z","shell.execute_reply":"2022-12-05T18:44:59.925816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aid2vec_model = word2vec.Word2Vec(sentences=raw_corpus, vector_size=100, window=5, min_count=1, sg=0, workers=-1, seed=42, hashfxn=hashfxn)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T18:50:22.961886Z","iopub.execute_input":"2022-12-05T18:50:22.962311Z","iopub.status.idle":"2022-12-05T18:51:05.526171Z","shell.execute_reply.started":"2022-12-05T18:50:22.962277Z","shell.execute_reply":"2022-12-05T18:51:05.245178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aid2vec_model.wv.save_word2vec_format('otto_aid2vec.bin', binary=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T18:51:15.136879Z","iopub.execute_input":"2022-12-05T18:51:15.137296Z","iopub.status.idle":"2022-12-05T18:51:23.157373Z","shell.execute_reply.started":"2022-12-05T18:51:15.13726Z","shell.execute_reply":"2022-12-05T18:51:23.156302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for session, group_df in tqdm(test_sessions.groupby(['session'])):\n    aid_list = []\n    results = aid2vec_model.wv.most_similar(positive=list(group_df['aid'].astype(str) + '_' + group_df['type']), topn=100)\n    for result in results:\n        aid = result[0].split('_')[0]\n        if aid not in aid_list:\n            aid_list.append(aid)\n        if len(aid_list) == 20:\n            aid_list = ' '.join(aid_list)\n            break\n    break","metadata":{"execution":{"iopub.status.busy":"2022-12-05T18:52:09.094799Z","iopub.execute_input":"2022-12-05T18:52:09.09543Z","iopub.status.idle":"2022-12-05T18:52:12.565899Z","shell.execute_reply.started":"2022-12-05T18:52:09.095379Z","shell.execute_reply":"2022-12-05T18:52:12.564293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Input:', list(group_df['aid'].astype(str) + '_' + group_df['type']))\nprint('Output: ', aid_list)","metadata":{"execution":{"iopub.status.busy":"2022-12-05T18:52:13.367989Z","iopub.execute_input":"2022-12-05T18:52:13.369432Z","iopub.status.idle":"2022-12-05T18:52:13.377995Z","shell.execute_reply.started":"2022-12-05T18:52:13.369388Z","shell.execute_reply":"2022-12-05T18:52:13.376703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results","metadata":{"execution":{"iopub.status.busy":"2022-12-05T18:57:57.720542Z","iopub.execute_input":"2022-12-05T18:57:57.721146Z","iopub.status.idle":"2022-12-05T18:57:57.73787Z","shell.execute_reply.started":"2022-12-05T18:57:57.721088Z","shell.execute_reply":"2022-12-05T18:57:57.736285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_sessions","metadata":{"execution":{"iopub.status.busy":"2022-12-05T18:55:06.142903Z","iopub.execute_input":"2022-12-05T18:55:06.143404Z","iopub.status.idle":"2022-12-05T18:55:06.161395Z","shell.execute_reply.started":"2022-12-05T18:55:06.143367Z","shell.execute_reply":"2022-12-05T18:55:06.160112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('/kaggle/input/otto-recommender-system/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-05T18:53:38.401274Z","iopub.execute_input":"2022-12-05T18:53:38.40181Z","iopub.status.idle":"2022-12-05T18:53:45.409916Z","shell.execute_reply.started":"2022-12-05T18:53:38.401763Z","shell.execute_reply":"2022-12-05T18:53:45.408286Z"},"trusted":true},"execution_count":null,"outputs":[]}]}