{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. What happens when you call nlp?\n\n- Tokenize the text and apply each pipeline component in order. The tokenizer turns a string of text into a `Doc` object. spaCy then applies every component in the pipeline on document, in order.","metadata":{}},{"cell_type":"markdown","source":"# 2. Inspecting the Pipeline","metadata":{}},{"cell_type":"code","source":"import spacy\n\nprint(spacy.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:21:47.210075Z","iopub.execute_input":"2023-01-29T23:21:47.210534Z","iopub.status.idle":"2023-01-29T23:21:47.216574Z","shell.execute_reply.started":"2023-01-29T23:21:47.210485Z","shell.execute_reply":"2023-01-29T23:21:47.215524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nlp = spacy.load('en_core_web_sm')\n\nprint(nlp.pipe_names)\nprint(nlp.pipeline)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:21:47.218521Z","iopub.execute_input":"2023-01-29T23:21:47.219305Z","iopub.status.idle":"2023-01-29T23:21:47.896656Z","shell.execute_reply.started":"2023-01-29T23:21:47.219271Z","shell.execute_reply":"2023-01-29T23:21:47.895354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Simple Components\n","metadata":{}},{"cell_type":"code","source":"import spacy\nfrom spacy.language import Language\n\n@Language.component(\"Length\")\ndef length_component(doc):\n    doc_length = len(doc)\n    print(f\"This document is {doc_length} tokens long.\")\n    return doc\n\nnlp = spacy.load(\"en_core_web_sm\")\n\nnlp.add_pipe(\"Length\", first=True)\nprint(nlp.pipe_names)\n\ndoc = nlp(\"This is a sentence.\")","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:26:10.064607Z","iopub.execute_input":"2023-01-29T23:26:10.065761Z","iopub.status.idle":"2023-01-29T23:26:10.789585Z","shell.execute_reply.started":"2023-01-29T23:26:10.065718Z","shell.execute_reply":"2023-01-29T23:26:10.788358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Complex Components","metadata":{}},{"cell_type":"code","source":"import spacy\nfrom spacy.matcher import PhraseMatcher\nfrom spacy.tokens import Span\nfrom spacy.language import Language\n\n\nnlp = spacy.load(\"en_core_web_sm\")\nanimals = [\"Golden Retriever\", \"cat\", \"trutule\", \"Rattus norvegicus\"]\nanimal_patterns = list(nlp.pipe(animals))\nprint(\"animal_patterns: \", animal_patterns)\nmatcher = PhraseMatcher(nlp.vocab)\nmatcher.add(\"ANIMAL\", None, *animal_patterns)\n\n@Language.component(\"Animal\")\ndef animal_component(doc):\n    matches = matcher(doc)\n    spans = [Span(doc, start, end, label=\"ANIMAL\") for match_id, start, end in matches]\n    \n    doc.ents = spans\n    return doc\n\nnlp.add_pipe(\"Animal\", after=\"ner\")\nprint(nlp.pipe_names)\n\ndoc = nlp(\"I have a cat and a Golden Retriever\")\nprint([(ent.text, ent.label_) for ent in doc.ents])","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:27:21.280945Z","iopub.execute_input":"2023-01-29T23:27:21.281355Z","iopub.status.idle":"2023-01-29T23:27:21.975694Z","shell.execute_reply.started":"2023-01-29T23:27:21.281313Z","shell.execute_reply":"2023-01-29T23:27:21.974346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Setting Extension Attributes","metadata":{}},{"cell_type":"code","source":"from spacy.lang.en import English\nfrom spacy.tokens import Token\n\nnlp = English()\n\nToken.set_extension(\"is_country\", default=False)\n\ndoc = nlp(\"I live in Spain.\")\ndoc[3]._.is_country = True\n\nprint([(token.text, token._.is_country) for token in doc])","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:27:31.311155Z","iopub.execute_input":"2023-01-29T23:27:31.311582Z","iopub.status.idle":"2023-01-29T23:27:31.466159Z","shell.execute_reply.started":"2023-01-29T23:27:31.311543Z","shell.execute_reply":"2023-01-29T23:27:31.464968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from spacy.lang.en import English\nfrom spacy.tokens import Token\n\nnlp = English()\n\ndef get_reversed(token):\n    return token.text[::-1]\n\nToken.set_extension(\"reversed\", getter=get_reversed)\n\ndoc = nlp(\"All generalizations are false, including this one.\")\n\nfor token in doc:\n    print(\"reversed:\", token._.reversed)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:27:40.212086Z","iopub.execute_input":"2023-01-29T23:27:40.212860Z","iopub.status.idle":"2023-01-29T23:27:40.363134Z","shell.execute_reply.started":"2023-01-29T23:27:40.212812Z","shell.execute_reply":"2023-01-29T23:27:40.362062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from spacy.lang.en import English\nfrom spacy.tokens import Doc\n\nnlp = English()\n\ndef get_has_number(doc):\n    return any(token.like_num for token in doc)\n\nDoc.set_extension(\"has_number\", getter=get_has_number)\n\ndoc = nlp(\"The museum closed for five years in 2012.\")\nprint(\"has_number:\", doc._.has_number)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:27:58.513129Z","iopub.execute_input":"2023-01-29T23:27:58.513554Z","iopub.status.idle":"2023-01-29T23:27:58.668219Z","shell.execute_reply.started":"2023-01-29T23:27:58.513518Z","shell.execute_reply":"2023-01-29T23:27:58.667361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from spacy.lang.en import English\nfrom spacy.tokens import Span\n\nnlp = English()\n\ndef to_html(span, tag):\n    return f\"<{tag}>{span.text}</{tag}>\"\n\nSpan.set_extension(\"to_html\", method=to_html)\n\ndoc = nlp(\"Hello world, this is a sentence.\")\nspan = doc[0:2]\nprint(\"to_html\", span._.to_html('strong'))","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:28:06.098806Z","iopub.execute_input":"2023-01-29T23:28:06.099783Z","iopub.status.idle":"2023-01-29T23:28:06.253393Z","shell.execute_reply.started":"2023-01-29T23:28:06.099744Z","shell.execute_reply":"2023-01-29T23:28:06.252172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. Entities and Extensions","metadata":{}},{"cell_type":"code","source":"import spacy\nfrom spacy.tokens import Span\n\nnlp = spacy.load(\"en_core_web_sm\")\n\ndef get_wikipedia_url(span):\n    if span.label_ in (\"PERSON\", \"ORG\", \"GPE\", \"LOCATION\"):\n        entity_text = span.text.replace(\" \", \"_\")\n        return \"https://en.wikipedia.org/w/index.php?search=\"+entity_text\n    \n    \nSpan.set_extension(\"wikipedia_url\", getter=get_wikipedia_url)\n\ndoc = nlp(\n    \"In over fifty years from his very first recordings right through to his \"\n    \"last album, David Bowie was at the vanguard of contemporary culture.\"\n)\n\nfor ent in doc.ents:\n    print(ent.text, ent._.wikipedia_url)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T23:28:13.067024Z","iopub.execute_input":"2023-01-29T23:28:13.068211Z","iopub.status.idle":"2023-01-29T23:28:14.009611Z","shell.execute_reply.started":"2023-01-29T23:28:13.068166Z","shell.execute_reply":"2023-01-29T23:28:14.008493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}