{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# word2vec Tutorial\nFor a detailed explanation of the word2vec, please refer to the following paper.\n\n[Efficient Estation of Word Representations in Vector Space](https://arxiv.org/abs/1301.3781)","metadata":{}},{"cell_type":"markdown","source":"# Read Train&Test Data","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\n\nfrom pathlib import Path\n\ndata_path = Path('/kaggle/input/otto-recommender-system/')","metadata":{"execution":{"iopub.status.busy":"2022-11-03T00:23:08.520420Z","iopub.execute_input":"2022-11-03T00:23:08.521045Z","iopub.status.idle":"2022-11-03T00:23:08.691123Z","shell.execute_reply.started":"2022-11-03T00:23:08.520889Z","shell.execute_reply":"2022-11-03T00:23:08.690068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_jsonl(target: str) -> pd.DataFrame():\n    sessions = pd.DataFrame()\n    chunks = pd.read_json(data_path / f'{target}.jsonl', lines=True, chunksize=100_000)\n\n    for e, chunk in enumerate(chunks):\n        event_dict = {\n            'session': [],\n            'aid': [],\n            'ts': [],\n            'type': [],\n        }\n        if e < 2:\n            for session, events in zip(chunk['session'].tolist(), chunk['events'].tolist()):\n                for event in events:\n                    event_dict['session'].append(session)\n                    event_dict['aid'].append(event['aid'])\n                    event_dict['ts'].append(event['ts'])\n                    event_dict['type'].append(event['type'])\n            chunk_session = pd.DataFrame(event_dict)\n            sessions = pd.concat([sessions, chunk_session])\n        else:\n            break\n    return sessions.reset_index(drop=True)\n\ntrain_sessions = read_jsonl('train')\ntest_sessions = read_jsonl('test')","metadata":{"execution":{"iopub.status.busy":"2022-11-03T00:23:08.693492Z","iopub.execute_input":"2022-11-03T00:23:08.694234Z","iopub.status.idle":"2022-11-03T00:24:11.227918Z","shell.execute_reply.started":"2022-11-03T00:23:08.694182Z","shell.execute_reply":"2022-11-03T00:24:11.226646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# word2vec","metadata":{}},{"cell_type":"code","source":"import hashlib\nimport os\nfrom gensim.models import word2vec\nfrom gensim.models import KeyedVectors\nos.environ[\"PYTHONHASHSEED\"] = str(42)\ndef hashfxn(x):\n    return int(hashlib.md5(str(x).encode()).hexdigest(), 16)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-03T00:24:11.229523Z","iopub.execute_input":"2022-11-03T00:24:11.230021Z","iopub.status.idle":"2022-11-03T00:24:11.896498Z","shell.execute_reply.started":"2022-11-03T00:24:11.229955Z","shell.execute_reply":"2022-11-03T00:24:11.895305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_corpus = []\nfor session, group_df in tqdm(train_sessions.groupby(['session'])):\n    raw_corpus.append(list(group_df['aid'].astype(str) + '_' + group_df['type']))\nfor session, group_df in tqdm(test_sessions.groupby(['session'])):\n    raw_corpus.append(list(group_df['aid'].astype(str) + '_' + group_df['type']))","metadata":{"execution":{"iopub.status.busy":"2022-11-03T00:24:11.899862Z","iopub.execute_input":"2022-11-03T00:24:11.900391Z","iopub.status.idle":"2022-11-03T00:28:12.985483Z","shell.execute_reply.started":"2022-11-03T00:24:11.900343Z","shell.execute_reply":"2022-11-03T00:28:12.984431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aid2vec_model = word2vec.Word2Vec(sentences=raw_corpus, vector_size=100, window=5, min_count=1, sg=0, workers=-1, seed=42, hashfxn=hashfxn)","metadata":{"execution":{"iopub.status.busy":"2022-11-03T00:28:12.986895Z","iopub.execute_input":"2022-11-03T00:28:12.987259Z","iopub.status.idle":"2022-11-03T00:28:54.338465Z","shell.execute_reply.started":"2022-11-03T00:28:12.987228Z","shell.execute_reply":"2022-11-03T00:28:54.265447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aid2vec_model.wv.save_word2vec_format('otto_aid2vec.bin', binary=True)","metadata":{"execution":{"iopub.status.busy":"2022-11-03T00:28:54.371391Z","iopub.execute_input":"2022-11-03T00:28:54.371803Z","iopub.status.idle":"2022-11-03T00:29:02.902394Z","shell.execute_reply.started":"2022-11-03T00:28:54.371768Z","shell.execute_reply":"2022-11-03T00:29:02.901172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for session, group_df in tqdm(test_sessions.groupby(['session'])):\n    aid_list = []\n    results = aid2vec_model.wv.most_similar(positive=list(group_df['aid'].astype(str) + '_' + group_df['type']), topn=100)\n    for result in results:\n        aid = result[0].split('_')[0]\n        if aid not in aid_list:\n            aid_list.append(aid)\n        if len(aid_list) == 20:\n            aid_list = ' '.join(aid_list)\n            break\n    break","metadata":{"execution":{"iopub.status.busy":"2022-11-03T00:29:02.906636Z","iopub.execute_input":"2022-11-03T00:29:02.907050Z","iopub.status.idle":"2022-11-03T00:29:06.841848Z","shell.execute_reply.started":"2022-11-03T00:29:02.907012Z","shell.execute_reply":"2022-11-03T00:29:06.840078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Input:', list(group_df['aid'].astype(str) + '_' + group_df['type']))\nprint('Output: ', aid_list)","metadata":{"execution":{"iopub.status.busy":"2022-11-03T00:29:06.843890Z","iopub.execute_input":"2022-11-03T00:29:06.844372Z","iopub.status.idle":"2022-11-03T00:29:06.865468Z","shell.execute_reply.started":"2022-11-03T00:29:06.844325Z","shell.execute_reply":"2022-11-03T00:29:06.862161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}