{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This is a fork of [this](https://www.kaggle.com/code/jdoesv/topics-identification) notebook by [jdoesv](https://www.kaggle.com/jdoesv).\n\nI have added some code to save the model and the BERTopic package so I can use it in an offline inference kernel. ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-05T11:43:53.919432Z","iopub.execute_input":"2022-06-05T11:43:53.920273Z","iopub.status.idle":"2022-06-05T11:45:48.500077Z","shell.execute_reply.started":"2022-06-05T11:43:53.920174Z","shell.execute_reply":"2022-06-05T11:45:48.499268Z"}}},{"cell_type":"code","source":"import glob, pandas as pd, numpy as np, re\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\n\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-07-05T23:40:37.959763Z","iopub.execute_input":"2022-07-05T23:40:37.960704Z","iopub.status.idle":"2022-07-05T23:40:39.742368Z","shell.execute_reply.started":"2022-07-05T23:40:37.960601Z","shell.execute_reply":"2022-07-05T23:40:39.740828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install bertopic -qq --target=/kaggle/working/site-packages\nimport sys\nsys.path.append('/kaggle/working/site-packages')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T23:40:39.744371Z","iopub.execute_input":"2022-07-05T23:40:39.744764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from bertopic import BERTopic","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sws = stopwords.words(\"english\") + [\"n't\",  \"'s\", \"'ve\"]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fls = glob.glob(\"../input/feedback-prize-2021/train/*.txt\")\ndocs = []\nfor fl in tqdm(fls):\n    with open(fl) as f:\n        txt = f.read()\n        word_tokens = word_tokenize(txt)\n        txt = \" \".join([w for w in word_tokens if not w.lower() in sws])\n    docs.append(txt)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"topic_model = BERTopic(n_gram_range=(1, 3), top_n_words=5, verbose=True)\ntopics, probs = topic_model.fit_transform(docs)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tm_meta = topic_model.get_topic_info()\ntm_meta.to_csv(\"./topic_model_metadata.csv\", index=False)\ndisplay(tm_meta)\n\npred_topics = pd.DataFrame()\ndids = list(map(lambda fl: fl.split(\"/\")[-1].split(\".\")[0], fls))\npred_topics[\"id\"] = dids\npred_topics[\"topic\"] = topics\npred_topics['prob'] = probs\npred_topics.to_csv(\"./topic_model_feedback.csv\", index=False)\npred_topics","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"topic_model.save(\"./feedback_2021_topic_model\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ntopics_meta = pd.read_csv('./topic_model_metadata.csv')\ntopics_df = pd.read_csv('./topic_model_feedback.csv')\n\ndef get_topic(txt):\n    txt = txt.split('_')[1:]\n    txt = \" \".join(txt)\n    return txt\n\ntopics_meta['topic'] = topics_meta['Name'].apply(get_topic)\n\ntopics_df = topics_df.merge(topics_meta[['Topic', 'topic']], left_on='topic', right_on='Topic', how='left')\ntopics_df = topics_df[['id', 'topic_y']]\ntopics_df = topics_df.rename(columns={'id': 'essay_id', 'topic_y': 'topic'})\nprint(topics_df.head())\ntopics_df.to_csv('./topics.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}