{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59575,"databundleVersionId":8060720,"sourceType":"competition"},{"sourceId":7731449,"sourceType":"datasetVersion","datasetId":4517815},{"sourceId":174185912,"sourceType":"kernelVersion"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Create your own index. \n\nI ran out of disk on Kaggle notebooks trying to create an index from 4k patents. 2k should definitely work and maybe a little more. \n\nIf you don't want to create your own I have created an index which can be found at this link: https://www.kaggle.com/datasets/devinanzelmo/uspto-explainable-ai-validation-index/. The index I created is based on 4k target patents and contains ~200k patents. The included patents are the neighbors of the target patents making it suitable for validation. ","metadata":{}},{"cell_type":"code","source":"import polars as pl\nimport whoosh_utils\nimport random\nimport os","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SET these variables with your desired values. \n\nnum_patents = 2 # WARNING number of patents in index will be 50 * num_patents\ndata_dir = \"/kaggle/input/uspto-explainable-ai/\"\noutput_dir = \"/kaggle/working/\"\nseed = 1","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create directory to store the validation index in\nif not os.path.exists(os.path.join(output_dir, \"validation_index\")):\n    os.makedirs(os.path.join(output_dir, \"validation_index\"))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# select num_patents publications numbers from 1975 or later\np_meta = pl.read_parquet(os.path.join(data_dir, \"patent_metadata.parquet\"), columns=[\"publication_number\", \"publication_date\"])\np_meta = p_meta.filter(pl.col(\"publication_date\") >= pl.date(1975, 1,1))\nrandom.seed(seed)\nval_pub_nums = p_meta.sample(num_patents)[:,0].to_list()\ndel p_meta","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:13:49.155919Z","iopub.execute_input":"2024-04-26T20:13:49.156141Z","iopub.status.idle":"2024-04-26T20:13:49.789663Z","shell.execute_reply.started":"2024-04-26T20:13:49.156122Z","shell.execute_reply":"2024-04-26T20:13:49.788920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load the relevent neighbors data based on selected patents\nneighbors = pl.read_csv(os.path.join(data_dir, \"nearest_neighbors.csv\"))\nneighbors = neighbors.filter(pl.col(\"publication_number\").is_in(val_pub_nums))","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:13:49.790846Z","iopub.execute_input":"2024-04-26T20:13:49.791175Z","iopub.status.idle":"2024-04-26T20:14:43.066698Z","shell.execute_reply.started":"2024-04-26T20:13:49.791145Z","shell.execute_reply":"2024-04-26T20:14:43.064609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save the publication numbers that have all their neighbors\npub_nums = neighbors[\"publication_number\"]\npl.DataFrame(pub_nums).write_csv(os.path.join(output_dir, \"validation_publication_numbers.csv\"))","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:14:43.070350Z","iopub.execute_input":"2024-04-26T20:14:43.070869Z","iopub.status.idle":"2024-04-26T20:14:43.078392Z","shell.execute_reply.started":"2024-04-26T20:14:43.070785Z","shell.execute_reply":"2024-04-26T20:14:43.077343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get the publication numbers we need to put into the index\nall_pub_nums = neighbors[:,1:].melt()[:,1].unique()\ndel neighbors, pub_nums","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:14:43.080113Z","iopub.execute_input":"2024-04-26T20:14:43.080839Z","iopub.status.idle":"2024-04-26T20:14:44.028066Z","shell.execute_reply.started":"2024-04-26T20:14:43.080768Z","shell.execute_reply":"2024-04-26T20:14:44.026867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare meta data to allow easy loading of relevent files\np_meta = pl.read_parquet(os.path.join(data_dir, \"patent_metadata.parquet\"), columns=[\"publication_number\", \"publication_date\", \"cpc_codes\"])\np_meta = p_meta.filter(pl.col(\"publication_number\").is_in(all_pub_nums))\n\np_meta = p_meta.with_columns(pl.col(\"publication_date\").dt.year().alias(\"year\"))\np_meta = p_meta.with_columns(pl.col(\"publication_date\").dt.month().alias(\"month\"))","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:14:44.029310Z","iopub.execute_input":"2024-04-26T20:14:44.029579Z","iopub.status.idle":"2024-04-26T20:14:48.258709Z","shell.execute_reply.started":"2024-04-26T20:14:44.029559Z","shell.execute_reply":"2024-04-26T20:14:48.257831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# generate the documents that will go in the index\ndocuments = list()\nfor (year, month), meta_df in p_meta.group_by([\"year\", \"month\"], maintain_order=True):\n    meta_df = meta_df.with_columns(pl.col(\"cpc_codes\").list.join(\" \"))\n    patents = pl.read_parquet(os.path.join(data_dir, f\"patent_data/{year}_{month}.parquet\"))\n    patents = patents.filter(pl.col(\"publication_number\").is_in(meta_df[\"publication_number\"]))\n    for i in range(meta_df.shape[0]):\n        d = dict()\n        p = patents.filter(pl.col(\"publication_number\") == meta_df[i, \"publication_number\"])\n        if p.shape[0] > 0:\n            d[\"publication_number\"] = p[0, \"publication_number\"]\n            d[\"title\"] = p[0, \"title\"]\n            d[\"abstract\"] = p[0, \"abstract\"]\n            d[\"claims\"] = p[0, \"claims\"]\n            d[\"description\"] = p[0, \"description\"]\n            d[\"cpc\"] = meta_df[i, \"cpc_codes\"]\n            documents.append(d)\n    del patents","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:14:48.259783Z","iopub.execute_input":"2024-04-26T20:14:48.260066Z","iopub.status.idle":"2024-04-26T20:40:58.793844Z","shell.execute_reply.started":"2024-04-26T20:14:48.260045Z","shell.execute_reply":"2024-04-26T20:40:58.792174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make the index\nwhoosh_utils.create_index(os.path.join(output_dir, \"validation_index\"), documents)","metadata":{"execution":{"iopub.status.busy":"2024-04-26T20:42:23.175002Z","iopub.execute_input":"2024-04-26T20:42:23.175369Z","iopub.status.idle":"2024-04-26T20:42:45.058680Z","shell.execute_reply.started":"2024-04-26T20:42:23.175345Z","shell.execute_reply":"2024-04-26T20:42:45.057649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}