{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Getting Started\n\nSpacy is available in 55+ languages.\n\n[Language Available in spaCy](https://spacy.io/usage/models/#languages)\n\nThe general syntax to import a language is: `from spacy.lang.__ import Language`","metadata":{}},{"cell_type":"code","source":"# Import the English language class\nfrom spacy.lang.en import English\n\n# Create the nlp object\nnlp = English()\n\n# Process a text\ndoc = nlp(\"This is a sentence.\")\n\n# Print the document text\nprint(doc.text)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:10:59.589345Z","iopub.execute_input":"2023-01-29T23:10:59.590273Z","iopub.status.idle":"2023-01-29T23:11:12.605711Z","shell.execute_reply.started":"2023-01-29T23:10:59.590154Z","shell.execute_reply":"2023-01-29T23:11:12.604377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import the German language class\nfrom spacy.lang.de import German\n\n# Create the nlp object\nnlp = German()\n\n# Process a text (this is German for: \"Kind regards!\")\ndoc = nlp(\"Liebe Grüße!\")\n\n# Print the document text\nprint(doc.text)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:11:12.608919Z","iopub.execute_input":"2023-01-29T23:11:12.610164Z","iopub.status.idle":"2023-01-29T23:11:12.770206Z","shell.execute_reply.started":"2023-01-29T23:11:12.610124Z","shell.execute_reply":"2023-01-29T23:11:12.768933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import the spannish language class\nfrom spacy.lang.es import Spanish\n\n# Create the nlp object\nnlp = Spanish()\n\n# Process a text (this is a Spanish for: )\ndoc = nlp(\"¿Cómo estás?\")\nprint(doc)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:11:12.771573Z","iopub.execute_input":"2023-01-29T23:11:12.771933Z","iopub.status.idle":"2023-01-29T23:11:12.878833Z","shell.execute_reply.started":"2023-01-29T23:11:12.771900Z","shell.execute_reply":"2023-01-29T23:11:12.877549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Documents, spans and tokens\n\nWhen you call `nlp` on a string, spaCy first tokenizes the text and creates a document object.","metadata":{}},{"cell_type":"code","source":"# Import the English language class and create the nlp object\nfrom spacy.lang.en import English\n\nnlp = English()\n\n# Process the text \ndoc = nlp('I like tree kangaroos and narwhals.')\nprint(doc)\n\n# Select the first token\nfirst_token = doc[0]\n# Print the first token\nprint(\"First word: \", first_token.text)\n\n# A slice of the Doc for \"tree kangaroos\"\ntree_kangaroos = doc[2:4]\nprint(\"Tree Kangarous slice: \", tree_kangaroos.text)\n\n# A slice of the Doc for \"tree kangaroos and narwhales\" (without the '.')\ntree_kangaroos_and_narwhals = doc[2:6]\nprint(\"Tree Kangaroos and Narwhals: \", tree_kangaroos_and_narwhals)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:11:12.880877Z","iopub.execute_input":"2023-01-29T23:11:12.881253Z","iopub.status.idle":"2023-01-29T23:11:13.036973Z","shell.execute_reply.started":"2023-01-29T23:11:12.881219Z","shell.execute_reply":"2023-01-29T23:11:13.035600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Lexical Attributes\n\nWe can use spaCy's `Doc` and `Token` objects, and lexical attributes to find a pattern in a text.","metadata":{}},{"cell_type":"code","source":"# In this example we will be looking for two subsequent tokens: a number and a percent sign\n\nfrom spacy.lang.en import English\n\nnlp = English()\n\n# Process the text\ndoc = nlp(\n    \"In 1990, more than 60% of people in East Asia were in extreme poverty. \"\n    \"Now less than 4% are.\"\n)\n\n# Iterate over the tokens in the doc\nfor token in doc:\n    # Check if the token resembles a number using `like_num` token attribute\n    if token.like_num: \n        # Get the next token in the document\n        # The index of the next token in the `doc` is `token.i + 1`\n        next_token = doc[token.i + 1]\n        # Check if the next token's text equals \"%\"\n        if next_token.text == \"%\":\n            print(f\"Percentage found: {token.text}%\")","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:11:13.040224Z","iopub.execute_input":"2023-01-29T23:11:13.040632Z","iopub.status.idle":"2023-01-29T23:11:13.202864Z","shell.execute_reply.started":"2023-01-29T23:11:13.040596Z","shell.execute_reply":"2023-01-29T23:11:13.201465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Loading models\n    \nYou can install spaCy models using this command: `python -m spacy download en_core_web_sm`","metadata":{}},{"cell_type":"code","source":"import spacy\n\n# Load the \"en_core_web_sm\"\nnlp = spacy.load('en_core_web_sm')\n\ntext = \"It's official: Apple is the first U.S. puplic company to reach a $1 trillion market value\"\n\n# Process the text\ndoc = nlp(text)\n\n# Print the document text\nprint(doc.text)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:11:13.204297Z","iopub.execute_input":"2023-01-29T23:11:13.204777Z","iopub.status.idle":"2023-01-29T23:11:14.036024Z","shell.execute_reply.started":"2023-01-29T23:11:13.204728Z","shell.execute_reply":"2023-01-29T23:11:14.035001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Predicting linguistic annotations\n\nWe will get to try one of spaCy's pre-trained model packages and see its predictions in action.","metadata":{}},{"cell_type":"code","source":"for token in doc:\n    token_text = token.text\n    token_pos = token.pos_\n    token_dep = token.dep_\n    \n    print(f\"{token_text:<12}{token_pos:<10}{token_dep:<10}\")","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:11:14.037621Z","iopub.execute_input":"2023-01-29T23:11:14.038019Z","iopub.status.idle":"2023-01-29T23:11:14.045688Z","shell.execute_reply.started":"2023-01-29T23:11:14.037983Z","shell.execute_reply":"2023-01-29T23:11:14.044336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Additional Token Attributes\nWe'll see these again in upcoming lectures. For now we just want to illustrate some of the other information that spaCy assigns to tokens:\n\n|Tag|Description|doc2[0].tag|\n|:------|:------:|:------|\n|`.text`|The original word text<!-- .element: style=\"text-align:left;\" -->|`Tesla`|\n|`.lemma_`|The base form of the word|`tesla`|\n|`.pos_`|The simple part-of-speech tag|`PROPN`/`proper noun`|\n|`.tag_`|The detailed part-of-speech tag|`NNP`/`noun, proper singular`|\n|`.shape_`|The word shape – capitalization, punctuation, digits|`Xxxxx`|\n|`.is_alpha`|Is the token an alpha character?|`True`|\n|`.is_stop`|Is the token part of a stop list, i.e. the most common words of the language?|`False`|","metadata":{}},{"cell_type":"code","source":"for ent in doc.ents:\n    print(ent.text, ent.label_)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:11:14.047635Z","iopub.execute_input":"2023-01-29T23:11:14.048125Z","iopub.status.idle":"2023-01-29T23:11:14.058095Z","shell.execute_reply.started":"2023-01-29T23:11:14.048076Z","shell.execute_reply":"2023-01-29T23:11:14.056801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. Predicting named entities in context\n\nModels are statistical and not always right. Whether their predictions are correct depends on the training data and the text you're processing.","metadata":{}},{"cell_type":"code","source":"import spacy\n\nnlp = spacy.load('en_core_web_sm')\ntext = \"Upcoming iPhone X release date leaked as Apple reveals pre-orders\"\n\ndoc = nlp(text)\nfor ent in doc.ents:\n    print(ent.text, ent.label_)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:11:14.059668Z","iopub.execute_input":"2023-01-29T23:11:14.060842Z","iopub.status.idle":"2023-01-29T23:11:14.816547Z","shell.execute_reply.started":"2023-01-29T23:11:14.060794Z","shell.execute_reply":"2023-01-29T23:11:14.815355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 7. Using the Matcher","metadata":{}},{"cell_type":"code","source":"import spacy\nfrom spacy.matcher import Matcher\n\nnlp = spacy.load(\"en_core_web_sm\")\ndoc = nlp(\"Upcoming iPhone X release date leaked as Apple reveals pre-orders\")\n\nmatcher = Matcher(nlp.vocab)\n\npattern = [{'TEXT': 'iPhone'}, {'TEXT': 'X'}]\nmatcher.add('IPHONE_X_PATTERN', [pattern])\n\nmatches = matcher(doc)\nprint(\"Matches:\", [doc[start:end].text for match_id, start, end in matches])","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:13:05.091230Z","iopub.execute_input":"2023-01-29T23:13:05.092411Z","iopub.status.idle":"2023-01-29T23:13:05.863832Z","shell.execute_reply.started":"2023-01-29T23:13:05.092355Z","shell.execute_reply":"2023-01-29T23:13:05.862227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 8. Writing match patterns\n\n- Write one pattern that only matches mentions of full iOS versions: \"iOS 7\", \"iOS 11\" and \"iOS 10\".","metadata":{}},{"cell_type":"code","source":"import spacy\nfrom spacy.matcher import Matcher\n\nnlp = spacy.load(\"en_core_web_sm\")\nmatcher = Matcher(nlp.vocab)\n\ndoc = nlp(\n    \"After making the iOS update you won't notice a radical system-wide \"\n    \"redesign: nothing like the aesthetic upheaval we got with iOS 7. Most of \"\n    \"iOS 11's furniture remains the same as in iOS 10. But you will discover \"\n    \"some tweaks once you delve a little deeper.\"\n)\n\npattern = [{\"TEXT\": \"iOS\"}, {\"IS_DIGIT\": True}]\n\nmatcher.add(\"IOS_VERSION_PATTERN\", [pattern])\nmatches = matcher(doc)\nprint(\"Total matches found: \", len(matches))\n\nfor match_id, start, end in matches:\n    print(\"Match Found: \", doc[start:end].text)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:13:19.153170Z","iopub.execute_input":"2023-01-29T23:13:19.153653Z","iopub.status.idle":"2023-01-29T23:13:19.866070Z","shell.execute_reply.started":"2023-01-29T23:13:19.153615Z","shell.execute_reply":"2023-01-29T23:13:19.864770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Write one pattern that only matches forms of \"download\" (tokens with the lemma \"download\"), followed by a token with the part-of-speech tag `PROPN` (proper noun)","metadata":{}},{"cell_type":"code","source":"import spacy\nfrom spacy.matcher import Matcher\n\nnlp = spacy.load(\"en_core_web_sm\")\nmatcher = Matcher(nlp.vocab)\n\ndoc = nlp(\n    \"i downloaded Fortnite on my laptop and can't open the game at all. Help? \"\n    \"so when I was downloading Minecraft, I got the Windows version where it \"\n    \"is the '.zip' folder and I used the default program to unpack it... do \"\n    \"I also need to download Winzip?\"\n)\n\npattern = [{'LEMMA': 'download'}, {'POS': 'PROPN'}]\nmatcher.add('DOWNLOAD_THINGS_PATTERN', [pattern])\nmatches = matcher(doc)\n\nprint(\"Total matches found: \", len(matches))\n\nfor match_id, start, end in matches:\n    print(\"Match Found: \", doc[start:end].text)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:13:41.823911Z","iopub.execute_input":"2023-01-29T23:13:41.824344Z","iopub.status.idle":"2023-01-29T23:13:42.830413Z","shell.execute_reply.started":"2023-01-29T23:13:41.824308Z","shell.execute_reply":"2023-01-29T23:13:42.829029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Write one pattern that matches adjectives (`\"ADJ\"`) followed by one or two `\"NOUN\"`s (one noun and one optional name).","metadata":{}},{"cell_type":"code","source":"import spacy\nfrom spacy.matcher import Matcher\n\nnlp = spacy.load('en_core_web_sm')\nmatcher = Matcher(nlp.vocab)\n\ndoc = nlp(\n    \"Features of the app include a beautiful design, smart search, automatic \"\n    \"labels and optional voice responses.\"\n)\n\npattern = [{\"POS\": \"ADJ\"}, {\"POS\": \"NOUN\"}, {\"POS\": \"NOUN\", \"OP\":\"?\"}]\nmatcher.add(\"ADJ_NOUN_PATTERN\", [pattern])\nmatches = matcher(doc)\n\nprint(\"Total matches found: \", len(matches))\n\nfor match_id, start, end in matches:\n    print(\"Match Found: \", doc[start:end].text)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:13:59.202888Z","iopub.execute_input":"2023-01-29T23:13:59.203303Z","iopub.status.idle":"2023-01-29T23:13:59.942103Z","shell.execute_reply.started":"2023-01-29T23:13:59.203271Z","shell.execute_reply":"2023-01-29T23:13:59.940696Z"},"trusted":true},"execution_count":null,"outputs":[]}]}