{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Strings to Hashes","metadata":{}},{"cell_type":"code","source":"import spacy\n\nprint(spacy.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:17:23.837321Z","iopub.execute_input":"2023-01-29T23:17:23.838298Z","iopub.status.idle":"2023-01-29T23:17:23.844169Z","shell.execute_reply.started":"2023-01-29T23:17:23.838247Z","shell.execute_reply":"2023-01-29T23:17:23.842850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from spacy.lang.en import English\n\nnlp = English()\ndoc = nlp(\"I have a cat\")\n\ncat_hash = nlp.vocab.strings['cat']\nprint(cat_hash)\n\ncat_string = nlp.vocab.strings[cat_hash]\nprint(cat_string)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:16:27.189165Z","iopub.execute_input":"2023-01-29T23:16:27.190353Z","iopub.status.idle":"2023-01-29T23:16:38.045449Z","shell.execute_reply.started":"2023-01-29T23:16:27.190213Z","shell.execute_reply":"2023-01-29T23:16:38.044325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"person_hash = nlp.vocab.strings['PERSON']\nprint(person_hash)\n\nperson_string = nlp.vocab.strings[person_hash]\nprint(person_string)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:16:38.047340Z","iopub.execute_input":"2023-01-29T23:16:38.047903Z","iopub.status.idle":"2023-01-29T23:16:38.056499Z","shell.execute_reply.started":"2023-01-29T23:16:38.047870Z","shell.execute_reply":"2023-01-29T23:16:38.055392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Creating a Doc","metadata":{}},{"cell_type":"code","source":"from spacy.lang.en import English\nfrom spacy.tokens import Doc\n\nnlp = English()\n\nwords = ['spaCy', 'is', 'cool', '!']\nspaces = [True, True, False, False]\n\ndoc = Doc(nlp.vocab, words=words, spaces=spaces)\nprint(doc.text)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:16:38.058196Z","iopub.execute_input":"2023-01-29T23:16:38.058572Z","iopub.status.idle":"2023-01-29T23:16:38.213320Z","shell.execute_reply.started":"2023-01-29T23:16:38.058540Z","shell.execute_reply":"2023-01-29T23:16:38.212132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"words = ['Go', ',', 'get', 'started', '!']\nspaces = [False, True, True, False, False]\n\ndoc = Doc(nlp.vocab, words=words, spaces=spaces)\nprint(doc.text)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:16:38.215898Z","iopub.execute_input":"2023-01-29T23:16:38.216240Z","iopub.status.idle":"2023-01-29T23:16:38.222976Z","shell.execute_reply.started":"2023-01-29T23:16:38.216209Z","shell.execute_reply":"2023-01-29T23:16:38.221890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Docs, Spans, and entities from Scratch","metadata":{}},{"cell_type":"code","source":"from spacy.lang.en import English\nfrom spacy.tokens import Doc, Span\n\nnlp = English()\n\nwords = ['I', 'like', 'David', 'Bowie']\nspaces = [True, True, True, False]\n\ndoc = Doc(nlp.vocab, words=words, spaces=spaces)\nprint(doc.text)\n\nspan = Span(doc, start=2, end=4, label='PERSON')\nprint(span.text, span.label_)\n\ndoc.ents = [span]\n\nprint([(ent.text, ent.label_) for ent in doc.ents])","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:16:38.224673Z","iopub.execute_input":"2023-01-29T23:16:38.225057Z","iopub.status.idle":"2023-01-29T23:16:38.377445Z","shell.execute_reply.started":"2023-01-29T23:16:38.224973Z","shell.execute_reply":"2023-01-29T23:16:38.376127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import spacy\n\nnlp = spacy.load(\"en_core_web_sm\")\ndoc = nlp(\"Berlin looks like a nice city\")\n\n# Collect all proper nouns that are followed by a verb\nfor token in doc:\n    if token.pos_ == \"PROPN\":\n        if doc[token.i + 1].pos_ == \"VERB\":\n            print(\"Found proper noun before a verb: \", token.text)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:16:38.378912Z","iopub.execute_input":"2023-01-29T23:16:38.379226Z","iopub.status.idle":"2023-01-29T23:16:39.183870Z","shell.execute_reply.started":"2023-01-29T23:16:38.379198Z","shell.execute_reply":"2023-01-29T23:16:39.182392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Inspecting Word Vectors","metadata":{}},{"cell_type":"code","source":"!python3 -m spacy download en_core_web_md","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-01-29T23:16:39.185315Z","iopub.execute_input":"2023-01-29T23:16:39.186239Z","iopub.status.idle":"2023-01-29T23:17:13.565841Z","shell.execute_reply.started":"2023-01-29T23:16:39.186200Z","shell.execute_reply":"2023-01-29T23:17:13.564172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import en_core_web_md\n\nnlp = en_core_web_md.load()\n\ndoc = nlp(\"Two bananas in pyjamas\")\n\nbananas_vector = doc[1].vector\nprint(bananas_vector)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:17:13.568093Z","iopub.execute_input":"2023-01-29T23:17:13.568494Z","iopub.status.idle":"2023-01-29T23:17:15.429060Z","shell.execute_reply.started":"2023-01-29T23:17:13.568454Z","shell.execute_reply":"2023-01-29T23:17:15.427788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Comparing Similarities","metadata":{}},{"cell_type":"code","source":"doc_1 = nlp(\"It's a warm summer day\")\ndoc_2 = nlp(\"It's sunny outside\")\n\nsimilarity = doc_1.similarity(doc_2)\nprint(similarity)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:17:15.430638Z","iopub.execute_input":"2023-01-29T23:17:15.431738Z","iopub.status.idle":"2023-01-29T23:17:15.455812Z","shell.execute_reply.started":"2023-01-29T23:17:15.431691Z","shell.execute_reply":"2023-01-29T23:17:15.454806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"doc = nlp(\"TV and Books\")\ntoken_1, token_2 = doc[0], doc[2]\n\nsimilarity = token_1.similarity(token_2)\nprint(similarity)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:17:15.459036Z","iopub.execute_input":"2023-01-29T23:17:15.459990Z","iopub.status.idle":"2023-01-29T23:17:15.478022Z","shell.execute_reply.started":"2023-01-29T23:17:15.459953Z","shell.execute_reply":"2023-01-29T23:17:15.476258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"doc = nlp(\"This was a great restaurant. Afterwards, we went to a really nice bar.\")\n\nspan_1 = doc[3:5]\nspan_2 = doc[12:15]\n\nprint(span_1)\nprint(span_2)\n\nsimilarity = span_1.similarity(span_2)\nprint(similarity)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:17:15.479590Z","iopub.execute_input":"2023-01-29T23:17:15.480082Z","iopub.status.idle":"2023-01-29T23:17:15.499455Z","shell.execute_reply.started":"2023-01-29T23:17:15.480047Z","shell.execute_reply":"2023-01-29T23:17:15.498319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. Debugging Patterns","metadata":{}},{"cell_type":"code","source":"import spacy\nfrom spacy.matcher import Matcher\n\nnlp = spacy.load(\"en_core_web_sm\")\ndoc = nlp(\n    \"Twitch Prime, the perks program for Amazon Prime members offering free \"\n    \"loot, games and other benefits, is ditching one of its best features: \"\n    \"ad-free viewing. According to an email sent out to Amazon Prime members \"\n    \"today, ad-free viewing will no longer be included as a part of Twitch \"\n    \"Prime for new members, beginning on September 14. However, members with \"\n    \"existing annual subscriptions will be able to continue to enjoy ad-free \"\n    \"viewing until their subscription comes up for renewal. Those with \"\n    \"monthly subscriptions will have access to ad-free viewing until October 15.\"\n)\n\npattern_1 = [{\"LOWER\": \"amazon\"}, {\"IS_TITLE\": True, \"POS\": \"PROPN\"}]\npattern_2 = [{\"LOWER\": \"ad\"}, {\"IS_PUNCT\":True}, {\"LOWER\":\"free\"}, {\"POS\": \"NOUN\"}]\n\nmatcher = Matcher(nlp.vocab)\nmatcher.add(\"PATTERN_1\", [pattern_1])\nmatcher.add(\"PATTERN_2\", [pattern_2])\n\nfor match_id, start, end in matcher(doc):\n    print(doc.vocab.strings[match_id], doc[start:end].text)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:17:44.855973Z","iopub.execute_input":"2023-01-29T23:17:44.856402Z","iopub.status.idle":"2023-01-29T23:17:45.576519Z","shell.execute_reply.started":"2023-01-29T23:17:44.856368Z","shell.execute_reply":"2023-01-29T23:17:45.575401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}