{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Topic Model Tutorial\nWe want to obtain user and item features from users' past purchase information.\nThere are several ways to do this.\n1. Memory-based\n2. Model-based\n3. Hybrid\n4. Deep-Learning\n\nI present the second method(2. Model-based), which uses a topic model.\nWithout going into the details of the topic model, we consider users as sentences and purchase items as words to cluster users.\nFor a detailed explanation of the topic model, please refer to the following paper.\n\n[Latent Dirichlet Allocation](https://web.archive.org/web/20120501152722/http://jmlr.csail.mit.edu/papers/v3/blei03a.html)\nBlei, David M.; Ng, Andrew Y.; Jordan, Michael I. Journal of Machine Learning Research. 3 (4–5): pp. 993–1022.","metadata":{}},{"cell_type":"markdown","source":"# Read Train&Test Data\nRead Data by [OTTO - Read a chunk of jsonl to manageable DF](https://www.kaggle.com/code/columbia2131/otto-read-a-chunk-of-jsonl-to-manageable-df?scriptVersionId=109793998)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom pathlib import Path\n\ndata_path = Path('/kaggle/input/otto-recommender-system/')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T11:18:41.162112Z","iopub.execute_input":"2022-11-02T11:18:41.162709Z","iopub.status.idle":"2022-11-02T11:18:41.190269Z","shell.execute_reply.started":"2022-11-02T11:18:41.162581Z","shell.execute_reply":"2022-11-02T11:18:41.189358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_json(target: str) -> pd.DataFrame():\n    sessions = pd.DataFrame()\n    chunks = pd.read_json(data_path / f'{target}.jsonl', lines=True, chunksize=100_000)\n\n    for e, chunk in enumerate(chunks):\n        event_dict = {\n            'session': [],\n            'aid': [],\n            'ts': [],\n            'type': [],\n        }\n        if e < 2:\n            for session, events in zip(chunk['session'].tolist(), chunk['events'].tolist()):\n                for event in events:\n                    event_dict['session'].append(session)\n                    event_dict['aid'].append(event['aid'])\n                    event_dict['ts'].append(event['ts'])\n                    event_dict['type'].append(event['type'])\n            chunk_session = pd.DataFrame(event_dict)\n            sessions = pd.concat([sessions, chunk_session])\n        else:\n            break\n    return sessions.reset_index(drop=True)\ntrain_sessions = read_json('train')\n# test_sessions = read_json('test')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T11:18:41.215352Z","iopub.execute_input":"2022-11-02T11:18:41.216094Z","iopub.status.idle":"2022-11-02T11:19:43.611181Z","shell.execute_reply.started":"2022-11-02T11:18:41.216029Z","shell.execute_reply":"2022-11-02T11:19:43.609954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Topic Model","metadata":{}},{"cell_type":"code","source":"from gensim.models import LdaModel # https://radimrehurek.com/gensim/models/ldamodel.html\nfrom gensim.corpora.dictionary import Dictionary\nimport pyLDAvis.gensim\n\nfrom tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-11-02T11:19:43.635210Z","iopub.execute_input":"2022-11-02T11:19:43.635716Z","iopub.status.idle":"2022-11-02T11:19:44.808683Z","shell.execute_reply.started":"2022-11-02T11:19:43.635684Z","shell.execute_reply":"2022-11-02T11:19:44.807680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_corpus = []\nfor session, group_df in tqdm(train_sessions.groupby(['session'])):\n    raw_corpus.append(list(group_df['aid'].astype(str) + '_' + group_df['type']))","metadata":{"execution":{"iopub.status.busy":"2022-11-02T11:19:44.810901Z","iopub.execute_input":"2022-11-02T11:19:44.811250Z","iopub.status.idle":"2022-11-02T11:21:49.981796Z","shell.execute_reply.started":"2022-11-02T11:19:44.811219Z","shell.execute_reply":"2022-11-02T11:21:49.975166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a dictionary that maps words to word ids\ndictionary = Dictionary(raw_corpus)\n# Convert to BoW format that can be read by LdaModel\ncorpus = [dictionary.doc2bow(aid_type) for aid_type in raw_corpus]","metadata":{"execution":{"iopub.status.busy":"2022-11-02T11:21:49.995762Z","iopub.execute_input":"2022-11-02T11:21:49.996333Z","iopub.status.idle":"2022-11-02T11:22:23.851226Z","shell.execute_reply.started":"2022-11-02T11:21:49.996280Z","shell.execute_reply":"2022-11-02T11:22:23.850070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_topics = 10\nlda = LdaModel(corpus=corpus, num_topics=num_topics, id2word=dictionary, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T11:22:23.853026Z","iopub.execute_input":"2022-11-02T11:22:23.853473Z","iopub.status.idle":"2022-11-02T11:26:14.052273Z","shell.execute_reply.started":"2022-11-02T11:22:23.853428Z","shell.execute_reply":"2022-11-02T11:26:14.051174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lda.save('lda_model.pickle')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T11:26:14.053854Z","iopub.execute_input":"2022-11-02T11:26:14.054215Z","iopub.status.idle":"2022-11-02T11:26:15.249364Z","shell.execute_reply.started":"2022-11-02T11:26:14.054182Z","shell.execute_reply":"2022-11-02T11:26:15.248209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualization\nvis = pyLDAvis.gensim.prepare(lda, corpus, dictionary, n_jobs = -1, sort_topics = False)\npyLDAvis.save_html(vis, 'OTTO_topic.html')\npyLDAvis.display(vis)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T11:26:15.250792Z","iopub.execute_input":"2022-11-02T11:26:15.251150Z","iopub.status.idle":"2022-11-02T11:27:28.540448Z","shell.execute_reply.started":"2022-11-02T11:26:15.251120Z","shell.execute_reply":"2022-11-02T11:27:28.539162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}