{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import spacy\nimport pandas as pd\n\ntrain = pd.read_csv('/kaggle/input/train-nlpproject/Train.csv')\ntest = pd.read_csv('/kaggle/input/train-nlpproject/Test.csv')\n\n# Loading the language library\nnlp = spacy.load('en_core_web_sm')\n\n# Building a Pipeline Object\ndoc = nlp(u'''I always wrote this series off as being a complete stink-fest because Jim Belushi was involved in it...''')\n\n# Using Tokens\nfor token in doc:\n    print(f\"{token.text:{12}}{token.pos_:{12}}{token.dep_:{12}}{token.lemma_}\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-30T15:37:52.356642Z","iopub.execute_input":"2023-05-30T15:37:52.357081Z","iopub.status.idle":"2023-05-30T15:37:53.953551Z","shell.execute_reply.started":"2023-05-30T15:37:52.357046Z","shell.execute_reply":"2023-05-30T15:37:53.951835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:37:56.059073Z","iopub.execute_input":"2023-05-30T15:37:56.059529Z","iopub.status.idle":"2023-05-30T15:37:56.071115Z","shell.execute_reply.started":"2023-05-30T15:37:56.059473Z","shell.execute_reply":"2023-05-30T15:37:56.069917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:37:58.286995Z","iopub.execute_input":"2023-05-30T15:37:58.287393Z","iopub.status.idle":"2023-05-30T15:37:58.296789Z","shell.execute_reply.started":"2023-05-30T15:37:58.287368Z","shell.execute_reply":"2023-05-30T15:37:58.295180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:00.576058Z","iopub.execute_input":"2023-05-30T15:38:00.576866Z","iopub.status.idle":"2023-05-30T15:38:00.602223Z","shell.execute_reply.started":"2023-05-30T15:38:00.576827Z","shell.execute_reply":"2023-05-30T15:38:00.600597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"visaulize\n","metadata":{}},{"cell_type":"code","source":"positive=(train['label']==1).sum()\nnegative=(train['label']==0).sum()","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:02.445075Z","iopub.execute_input":"2023-05-30T15:38:02.445548Z","iopub.status.idle":"2023-05-30T15:38:02.454361Z","shell.execute_reply.started":"2023-05-30T15:38:02.445512Z","shell.execute_reply":"2023-05-30T15:38:02.452231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt          \nimport random  \nfig = plt.figure(figsize=(5, 5))\n\n# labels for the two classes\nlabels = 1,0\n\n# Sizes for each slide\nsizes = [positive.size, negative.size]\n\n# Declare pie chart, where the slices will be ordered and plotted counter-clockwise:\nplt.pie(sizes, labels=labels, autopct='%1.1f%%',\n        shadow=True, startangle=90)\n\n# Equal aspect ratio ensures that pie is drawn as a circle.\nplt.axis('equal')  \n\n# Display the chart\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:04.441448Z","iopub.execute_input":"2023-05-30T15:38:04.441933Z","iopub.status.idle":"2023-05-30T15:38:04.560829Z","shell.execute_reply.started":"2023-05-30T15:38:04.441901Z","shell.execute_reply":"2023-05-30T15:38:04.559378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:07.900358Z","iopub.execute_input":"2023-05-30T15:38:07.900857Z","iopub.status.idle":"2023-05-30T15:38:07.913753Z","shell.execute_reply.started":"2023-05-30T15:38:07.900821Z","shell.execute_reply":"2023-05-30T15:38:07.911307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['text'].duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:09.438744Z","iopub.execute_input":"2023-05-30T15:38:09.440533Z","iopub.status.idle":"2023-05-30T15:38:09.477786Z","shell.execute_reply.started":"2023-05-30T15:38:09.440442Z","shell.execute_reply":"2023-05-30T15:38:09.476668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.drop_duplicates(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:10.928591Z","iopub.execute_input":"2023-05-30T15:38:10.929410Z","iopub.status.idle":"2023-05-30T15:38:11.070072Z","shell.execute_reply.started":"2023-05-30T15:38:10.929349Z","shell.execute_reply":"2023-05-30T15:38:11.069215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['text'].duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:12.600962Z","iopub.execute_input":"2023-05-30T15:38:12.601775Z","iopub.status.idle":"2023-05-30T15:38:12.620807Z","shell.execute_reply.started":"2023-05-30T15:38:12.601740Z","shell.execute_reply":"2023-05-30T15:38:12.618936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['text'][1]","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:14.520850Z","iopub.execute_input":"2023-05-30T15:38:14.521287Z","iopub.status.idle":"2023-05-30T15:38:14.529585Z","shell.execute_reply.started":"2023-05-30T15:38:14.521257Z","shell.execute_reply":"2023-05-30T15:38:14.528470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['text'][300]","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:16.821949Z","iopub.execute_input":"2023-05-30T15:38:16.822311Z","iopub.status.idle":"2023-05-30T15:38:16.829145Z","shell.execute_reply.started":"2023-05-30T15:38:16.822287Z","shell.execute_reply":"2023-05-30T15:38:16.827655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nlp.pipeline","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:18.903475Z","iopub.execute_input":"2023-05-30T15:38:18.903879Z","iopub.status.idle":"2023-05-30T15:38:18.915417Z","shell.execute_reply.started":"2023-05-30T15:38:18.903850Z","shell.execute_reply":"2023-05-30T15:38:18.913331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nlp.pipe_names","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:21.311649Z","iopub.execute_input":"2023-05-30T15:38:21.312042Z","iopub.status.idle":"2023-05-30T15:38:21.323228Z","shell.execute_reply.started":"2023-05-30T15:38:21.312014Z","shell.execute_reply":"2023-05-30T15:38:21.322248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text = \"\"\"\nWhen I put this movie in my DVD player, and sat down with a coke and some chips, I had some expectat...\"\"\"\ndoc = nlp(text)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:22.913075Z","iopub.execute_input":"2023-05-30T15:38:22.913524Z","iopub.status.idle":"2023-05-30T15:38:22.933156Z","shell.execute_reply.started":"2023-05-30T15:38:22.913474Z","shell.execute_reply":"2023-05-30T15:38:22.931961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"quote = doc[30:50]\nprint(quote)\nprint(type(quote))","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:24.911849Z","iopub.execute_input":"2023-05-30T15:38:24.913171Z","iopub.status.idle":"2023-05-30T15:38:24.920908Z","shell.execute_reply.started":"2023-05-30T15:38:24.913073Z","shell.execute_reply":"2023-05-30T15:38:24.918891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, sentence in enumerate(doc.sents, 1):\n    print(f\"{i} - {sentence}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:26.574906Z","iopub.execute_input":"2023-05-30T15:38:26.575361Z","iopub.status.idle":"2023-05-30T15:38:26.583422Z","shell.execute_reply.started":"2023-05-30T15:38:26.575326Z","shell.execute_reply":"2023-05-30T15:38:26.581455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Named Entity","metadata":{}},{"cell_type":"code","source":"for entity in doc.ents:\n    print(f\"{entity.text:-<{20}}{entity.label_:-<{20}}{str(spacy.explain(entity.label_))}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:32.014301Z","iopub.execute_input":"2023-05-30T15:38:32.015473Z","iopub.status.idle":"2023-05-30T15:38:32.022769Z","shell.execute_reply.started":"2023-05-30T15:38:32.015423Z","shell.execute_reply":"2023-05-30T15:38:32.021460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Noun Chunks","metadata":{}},{"cell_type":"code","source":"for chunk in doc.noun_chunks:\n    print(chunk.text)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:33.585395Z","iopub.execute_input":"2023-05-30T15:38:33.585870Z","iopub.status.idle":"2023-05-30T15:38:33.593065Z","shell.execute_reply.started":"2023-05-30T15:38:33.585834Z","shell.execute_reply":"2023-05-30T15:38:33.591476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Built-in Visualizers","metadata":{}},{"cell_type":"code","source":"from spacy import displacy\n\ndisplacy.render(doc, style='dep', jupyter=True, options={'distance':90})","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:37.008637Z","iopub.execute_input":"2023-05-30T15:38:37.009026Z","iopub.status.idle":"2023-05-30T15:38:37.022864Z","shell.execute_reply.started":"2023-05-30T15:38:37.008997Z","shell.execute_reply":"2023-05-30T15:38:37.021573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Stemming","metadata":{}},{"cell_type":"code","source":"import nltk\nnltk.download('all')","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:40.022337Z","iopub.execute_input":"2023-05-30T15:38:40.022682Z","iopub.status.idle":"2023-05-30T15:38:45.377658Z","shell.execute_reply.started":"2023-05-30T15:38:40.022659Z","shell.execute_reply":"2023-05-30T15:38:45.376369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk.stem.porter import PorterStemmer\n\ntext = \"Some text to be split and stemmed\"\nwords = text.split()\np_stemmer = PorterStemmer()\n\nfor word in words:\n    print(f\"{word} --------> {p_stemmer.stem(word)}\")\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['text'][0]","metadata":{"execution":{"iopub.status.busy":"2023-05-30T14:56:22.015876Z","iopub.execute_input":"2023-05-30T14:56:22.016352Z","iopub.status.idle":"2023-05-30T14:56:22.028627Z","shell.execute_reply.started":"2023-05-30T14:56:22.016318Z","shell.execute_reply":"2023-05-30T14:56:22.027332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk.stem.snowball import SnowballStemmer\nstemmer = SnowballStemmer(\"english\")\ntrainl = []\nfor txt in train['text']:\n    tokens = [stemmer.stem(word) for word in txt]\n    trainl.append(tokens)\n\ntrain['text'] = trainl\n","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:39:12.395213Z","iopub.execute_input":"2023-05-30T15:39:12.395652Z","iopub.status.idle":"2023-05-30T15:39:35.708441Z","shell.execute_reply.started":"2023-05-30T15:39:12.395619Z","shell.execute_reply":"2023-05-30T15:39:35.707377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Lemmatization\n","metadata":{}},{"cell_type":"code","source":"text = nlp(u\"My name is John Mourby and this is my story about Paperhouse: In May 2003 I saw Alfred Hitchcock's p...\")\n\nfor token in text:\n    print(f\"{token.text:{12}}{token.pos_:{10}}\\t{token.lemma:{20}}\\t{token.lemma_}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-30T14:59:46.650279Z","iopub.execute_input":"2023-05-30T14:59:46.650722Z","iopub.status.idle":"2023-05-30T14:59:46.670913Z","shell.execute_reply.started":"2023-05-30T14:59:46.650689Z","shell.execute_reply":"2023-05-30T14:59:46.669284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Stop Words\n\n","metadata":{}},{"cell_type":"code","source":"nlp = spacy.load('en_core_web_sm')\n\nprint(nlp.Defaults.stop_words)\nprint(len(nlp.Defaults.stop_words))","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:39:35.712895Z","iopub.execute_input":"2023-05-30T15:39:35.713205Z","iopub.status.idle":"2023-05-30T15:39:36.553364Z","shell.execute_reply.started":"2023-05-30T15:39:35.713180Z","shell.execute_reply":"2023-05-30T15:39:36.551601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"words = ['do', 'perhaps', 'six', 'you', 'for', 'all']\n\nfor word in words:\n    print(f\"{word}: is stop word: {nlp.vocab[word].is_stop}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:40:23.384374Z","iopub.execute_input":"2023-05-30T15:40:23.384857Z","iopub.status.idle":"2023-05-30T15:40:23.392048Z","shell.execute_reply.started":"2023-05-30T15:40:23.384825Z","shell.execute_reply":"2023-05-30T15:40:23.390840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can add our own stop word\nnlp.Defaults.stop_words.add('btw')\nnlp.Defaults.stop_words.add('u')\n\nsentence = 'Where was u ? I was looking for you btw session...'\nfor word in sentence.split():\n    print(f\"{word:{20}}: is stop word: {nlp.vocab[word].is_stop}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:40:25.033127Z","iopub.execute_input":"2023-05-30T15:40:25.033578Z","iopub.status.idle":"2023-05-30T15:40:25.041634Z","shell.execute_reply.started":"2023-05-30T15:40:25.033543Z","shell.execute_reply":"2023-05-30T15:40:25.039722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can also remove stop word\nnlp.vocab['for'].is_stop = False\nsentence = 'Where was u ? I was looking for you btw session...'\nfor word in sentence.split():\n    print(f\"{word:{20}}: is stop word: {nlp.vocab[word].is_stop}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:40:26.896787Z","iopub.execute_input":"2023-05-30T15:40:26.897213Z","iopub.status.idle":"2023-05-30T15:40:26.903763Z","shell.execute_reply.started":"2023-05-30T15:40:26.897179Z","shell.execute_reply":"2023-05-30T15:40:26.902581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Phrase Matching and Vocabulary\n","metadata":{}},{"cell_type":"code","source":"from spacy.matcher import Matcher\n\nmatcher = Matcher(nlp.vocab)\npattern_1 = [{'LOWER': 'solarpower'}] # ----> SolarPower\npattern_2 = [{'LOWER': 'solar'}, {'IS_PUNCT': True}, {'LOWER': 'power'}] # ---> Solar-Power\npattern_3 = [{'LOWER': 'solar'}, {'LOWER': 'power'}] # ---> Solar Power\n\nmatcher.add('SolarPower', [pattern_1, pattern_2, pattern_3])\n\ntext = u'''\nSolar Power is the conversion of energy from sunlight into electricity, \neither directly using photovoltaics (PV), indirectly using concentrated SolarPower, \nor a combination. Concentrated Solar-Power systems use lenses or mirrors and solar \ntracking systems to focus a large area of sunlight into a small beam.\n'''\ndoc = nlp(text)\nfound_matches = matcher(doc)\nprint(found_matches)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:40:28.375445Z","iopub.execute_input":"2023-05-30T15:40:28.375859Z","iopub.status.idle":"2023-05-30T15:40:28.424120Z","shell.execute_reply.started":"2023-05-30T15:40:28.375827Z","shell.execute_reply":"2023-05-30T15:40:28.422347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Word Vectors and Semantic Similarity\n","metadata":{}},{"cell_type":"code","source":"!python3 -m spacy download en_core_web_md","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-05-30T15:40:30.238209Z","iopub.execute_input":"2023-05-30T15:40:30.238609Z","iopub.status.idle":"2023-05-30T15:40:49.153679Z","shell.execute_reply.started":"2023-05-30T15:40:30.238583Z","shell.execute_reply":"2023-05-30T15:40:49.152704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import en_core_web_md\n\n# Load a larger model with vectors\nnlp = en_core_web_md.load()\n\n# Compare two documents\ndoc_1 = nlp(\"I like fast food\")\ndoc_2 = nlp(\"I like pizza\")\n\nprint(doc_1.similarity(doc_2))\nprint(doc_2.similarity(doc_1))","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:04:20.345237Z","iopub.execute_input":"2023-05-30T15:04:20.345542Z","iopub.status.idle":"2023-05-30T15:04:21.975187Z","shell.execute_reply.started":"2023-05-30T15:04:20.345517Z","shell.execute_reply":"2023-05-30T15:04:21.973631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compare two tokens\ndoc = nlp(\"fatma like pizza and pasta\")\n\ntoken_1 = doc[2]\ntoken_2 = doc[4]\nprint(token_1.similarity(token_2))","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:04:51.261775Z","iopub.execute_input":"2023-05-30T15:04:51.262193Z","iopub.status.idle":"2023-05-30T15:04:51.281110Z","shell.execute_reply.started":"2023-05-30T15:04:51.262161Z","shell.execute_reply":"2023-05-30T15:04:51.279235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compare a span with a document\nspan = nlp(\"I just found the IMDb and searched this film and I was moved almost to tears by the comments of all ...\")[2:5]\ndoc = nlp(\"Dragon Fighter is the first Sci-Fi Channel (although I guess it's now called Syfy?) original movie I..\")\n\nprint(span.similarity(doc))","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:04:31.911625Z","iopub.execute_input":"2023-05-30T15:04:31.913107Z","iopub.status.idle":"2023-05-30T15:04:31.939283Z","shell.execute_reply.started":"2023-05-30T15:04:31.913050Z","shell.execute_reply":"2023-05-30T15:04:31.937999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## View token tags\n","metadata":{}},{"cell_type":"code","source":"text = u'''\nFans of creature feature films have to endure a lot of awful movies lately. Blood Surf shamelessly j...\n'''\ndoc = nlp(text)\nfor token in doc:\n    print(f\"{token.text:{10}} {token.pos_:{8}} {token.tag_:{6}} {spacy.explain(token.tag_)}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:40:49.155596Z","iopub.execute_input":"2023-05-30T15:40:49.155903Z","iopub.status.idle":"2023-05-30T15:40:49.185313Z","shell.execute_reply.started":"2023-05-30T15:40:49.155873Z","shell.execute_reply":"2023-05-30T15:40:49.183415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Working with POS Tags\n","metadata":{}},{"cell_type":"code","source":"doc = nlp(\"If you haven't seen the gong show TV series then you won't like this movie much at all, not that kno.\")\nr = doc[1]\nprint(f\"{r.text:{10}} {r.pos_:{8}} {r.tag_:{6}} {spacy.explain(r.tag_)}\\n\")\n\ndoc = nlp(\"I sat through this film and i have to say it only just managed to keep my attention. The film would ...\")\nr = doc[2]\nprint(f\"{r.text:{10}} {r.pos_:{8}} {r.tag_:{6}} {spacy.explain(r.tag_)}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:40:49.187934Z","iopub.execute_input":"2023-05-30T15:40:49.188329Z","iopub.status.idle":"2023-05-30T15:40:49.225041Z","shell.execute_reply.started":"2023-05-30T15:40:49.188297Z","shell.execute_reply":"2023-05-30T15:40:49.222884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Counting POS Tags\n\nThe `Doc.count_by()` method accepts a specific token attribute as its argument, and returns a frequency count of the given attribute as a dictionary object.","metadata":{}},{"cell_type":"code","source":"doc = nlp(\"The Argentinian music poet, Atahualpa Yupanqui, once said that some folk music repeats similarly at ...\")\n\npos_count = doc.count_by(spacy.attrs.POS)\nprint(pos_count)\nnew = {}\nfor key, value in pos_count.items():\n    new[doc.vocab[key].text] = value\n    \nprint(new)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:40:49.228761Z","iopub.execute_input":"2023-05-30T15:40:49.229219Z","iopub.status.idle":"2023-05-30T15:40:49.251352Z","shell.execute_reply.started":"2023-05-30T15:40:49.229188Z","shell.execute_reply":"2023-05-30T15:40:49.249581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Gensim\n\nGensim is an open-source library for unsupervised topic modeling and natural language processing, using modern statistical machine learning. \n","metadata":{}},{"cell_type":"code","source":"!wget https://www.gutenberg.org/files/1342/1342-0.txt","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:40:49.253912Z","iopub.execute_input":"2023-05-30T15:40:49.254530Z","iopub.status.idle":"2023-05-30T15:40:50.928211Z","shell.execute_reply.started":"2023-05-30T15:40:49.254449Z","shell.execute_reply":"2023-05-30T15:40:50.925740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport spacy\n\nnlp = spacy.load('en_core_web_sm')\n\ndef read_file(file_name):\n    with open(file_name, 'r') as file:\n        return file.read()\n    \ntext = read_file('1342-0.txt')\nprocessed_text = nlp(text)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:42:10.794541Z","iopub.execute_input":"2023-05-30T15:42:10.796430Z","iopub.status.idle":"2023-05-30T15:42:33.008298Z","shell.execute_reply.started":"2023-05-30T15:42:10.796355Z","shell.execute_reply":"2023-05-30T15:42:33.007145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Corpus\n","metadata":{}},{"cell_type":"code","source":"# An example of corpus, consists of 7045 sentences\nsentences = [s for s in processed_text.sents]\n\nprint(len(sentences))","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:42:37.123365Z","iopub.execute_input":"2023-05-30T15:42:37.124625Z","iopub.status.idle":"2023-05-30T15:42:37.145990Z","shell.execute_reply.started":"2023-05-30T15:42:37.124564Z","shell.execute_reply":"2023-05-30T15:42:37.143663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(sentences[30:34])","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:42:39.512165Z","iopub.execute_input":"2023-05-30T15:42:39.512893Z","iopub.status.idle":"2023-05-30T15:42:39.523102Z","shell.execute_reply.started":"2023-05-30T15:42:39.512860Z","shell.execute_reply":"2023-05-30T15:42:39.521745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(processed_text.text.split())","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:42:41.774237Z","iopub.execute_input":"2023-05-30T15:42:41.774697Z","iopub.status.idle":"2023-05-30T15:42:41.895192Z","shell.execute_reply.started":"2023-05-30T15:42:41.774663Z","shell.execute_reply":"2023-05-30T15:42:41.893522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gensim\nfrom gensim.models import Word2Vec\n\nprint(f\"Gensim Version: {gensim.__version__}\")\n\n# We need data for training the model\nprocessed_sentences = [sent.lemma_.split() for sent in processed_text.sents]\nprocessed_sentences[0]","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:42:44.281414Z","iopub.execute_input":"2023-05-30T15:42:44.281949Z","iopub.status.idle":"2023-05-30T15:42:44.397894Z","shell.execute_reply.started":"2023-05-30T15:42:44.281912Z","shell.execute_reply":"2023-05-30T15:42:44.396778Z"},"trusted":true},"execution_count":null,"outputs":[]}]}