{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59575,"databundleVersionId":8060720,"sourceType":"competition"},{"sourceId":8479599,"sourceType":"datasetVersion","datasetId":4517815},{"sourceId":174185912,"sourceType":"kernelVersion"}],"dockerImageVersionId":30732,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import re\nimport os\nimport ast\nimport random\nfrom collections import Counter, defaultdict\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\n\nimport whoosh_utils\nimport whoosh\n\nfrom tqdm import tqdm\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-20T03:09:50.286942Z","iopub.execute_input":"2024-06-20T03:09:50.287308Z","iopub.status.idle":"2024-06-20T03:10:25.46147Z","shell.execute_reply.started":"2024-06-20T03:09:50.287279Z","shell.execute_reply":"2024-06-20T03:10:25.460344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### CONFIGS\nCOLUMN = \"title\"","metadata":{"execution":{"iopub.status.busy":"2024-06-20T03:10:25.463448Z","iopub.execute_input":"2024-06-20T03:10:25.463922Z","iopub.status.idle":"2024-06-20T03:10:25.46861Z","shell.execute_reply.started":"2024-06-20T03:10:25.46389Z","shell.execute_reply":"2024-06-20T03:10:25.467475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUMBER_REGEX = re.compile(r\"^(\\d+|\\d{1,3}(,\\d{3})*)(\\.\\d+)?$\")\n\n\nclass NumberFilter(whoosh.analysis.Filter):\n    def __call__(self, tokens):\n        for t in tokens:\n            if not NUMBER_REGEX.match(t.text):\n                yield t\n\n\nBRS_STOPWORDS = [\n    \"an\",\n    \"are\",\n    \"by\",\n    \"for\",\n    \"if\",\n    \"into\",\n    \"is\",\n    \"no\",\n    \"not\",\n    \"of\",\n    \"on\",\n    \"such\",\n    \"that\",\n    \"the\",\n    \"their\",\n    \"then\",\n    \"there\",\n    \"these\",\n    \"they\",\n    \"this\",\n    \"to\",\n    \"was\",\n    \"will\",\n]\n# Prevent both stopwords and numbers from ever being indexed.\ncustom_analyzer = whoosh.analysis.StandardAnalyzer(stoplist=BRS_STOPWORDS) | NumberFilter()","metadata":{"execution":{"iopub.status.busy":"2024-06-19T14:14:37.142086Z","iopub.execute_input":"2024-06-19T14:14:37.142762Z","iopub.status.idle":"2024-06-19T14:14:37.15333Z","shell.execute_reply.started":"2024-06-19T14:14:37.142722Z","shell.execute_reply":"2024-06-19T14:14:37.151901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import ast\nimport itertools\nimport os\nimport pickle\nimport random\nimport re\nfrom collections import Counter, defaultdict\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport whoosh\nimport whoosh_utils\nfrom tqdm import tqdm\n\nINPUT_DIR = Path(\"../input/\")\nNUMBER_REGEX = re.compile(r\"^(\\d+|\\d{1,3}(,\\d{3})*)(\\.\\d+)?$\")\n\n\nclass NumberFilter(whoosh.analysis.Filter):\n    def __call__(self, tokens):\n        for t in tokens:\n            if not NUMBER_REGEX.match(t.text):\n                yield t\n\n\nBRS_STOPWORDS = [\n    \"an\",\n    \"are\",\n    \"by\",\n    \"for\",\n    \"if\",\n    \"into\",\n    \"is\",\n    \"no\",\n    \"not\",\n    \"of\",\n    \"on\",\n    \"such\",\n    \"that\",\n    \"the\",\n    \"their\",\n    \"then\",\n    \"there\",\n    \"these\",\n    \"they\",\n    \"this\",\n    \"to\",\n    \"was\",\n    \"will\",\n]\n# Prevent both stopwords and numbers from ever being indexed.\ncustom_analyzer = whoosh.analysis.StandardAnalyzer(stoplist=BRS_STOPWORDS) | NumberFilter()\n\n\ndef extract_ngrams(text, n):\n    words = [t.text for t in custom_analyzer(text)]\n    if n == 1:\n        return set(words)\n    ngrams = zip(*[words[i:] for i in range(n)], strict=False)\n    return list(set(ngrams))\n\n\nglobal_word_counter = Counter()\npatent_files = sorted(os.listdir(INPUT_DIR / \"uspto-explainable-ai/patent_data\"))\n\nfor patent_file in tqdm(patent_files):\n    df = pl.read_parquet(\n        INPUT_DIR / \"uspto-explainable-ai/patent_data\" / patent_file,\n        columns=[\"publication_number\", COLUMN],\n    )\n    word_list = []\n    for text in df[COLUMN]:\n        text = text.lower()\n        word_list.extend(extract_ngrams(text, 1))\n    _counter = Counter(word_list)\n    global_word_counter.update(_counter)\n\n\npickle.dump(global_word_counter, open(f\"global_{COLUMN}_word_counter.pkl\", \"wb\"))\n","metadata":{},"execution_count":null,"outputs":[]}]}