{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59575,"databundleVersionId":8060720,"sourceType":"competition"},{"sourceId":8479599,"sourceType":"datasetVersion","datasetId":4517815},{"sourceId":174185912,"sourceType":"kernelVersion"}],"dockerImageVersionId":30698,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from collections import Counter\nfrom tqdm import tqdm\nimport pandas as pd \nimport whoosh_utils\nimport numpy as np \nimport os","metadata":{"execution":{"iopub.status.busy":"2024-07-03T05:34:13.875000Z","iopub.execute_input":"2024-07-03T05:34:13.875451Z","iopub.status.idle":"2024-07-03T05:34:49.980313Z","shell.execute_reply.started":"2024-07-03T05:34:13.875414Z","shell.execute_reply":"2024-07-03T05:34:49.979087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(\"/kaggle/input/uspto-explainable-ai/test.csv\")\ntest_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-07-03T05:34:49.982475Z","iopub.execute_input":"2024-07-03T05:34:49.983015Z","iopub.status.idle":"2024-07-03T05:34:50.036809Z","shell.execute_reply.started":"2024-07-03T05:34:49.982981Z","shell.execute_reply":"2024-07-03T05:34:50.035651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# store the unique [publication_number] from the test dataset\nallunique_pubnums = []\nfor num,chunk in tqdm(enumerate(pd.read_csv(\"/kaggle/input/uspto-explainable-ai/test.csv\",chunksize=1000))):\n    \n    unique_values = pd.unique(chunk.values.ravel())\n    unique_values = list(unique_values)\n    \n    allunique_pubnums.extend(unique_values)\n    allunique_pubnums = list(set(allunique_pubnums))\n    \nlen(allunique_pubnums)","metadata":{"execution":{"iopub.status.busy":"2024-07-03T05:34:50.038474Z","iopub.execute_input":"2024-07-03T05:34:50.038933Z","iopub.status.idle":"2024-07-03T05:34:50.068964Z","shell.execute_reply.started":"2024-07-03T05:34:50.038889Z","shell.execute_reply":"2024-07-03T05:34:50.067723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_df = pd.read_parquet(\"/kaggle/input/uspto-explainable-ai/patent_metadata.parquet\")\nmeta_df.head(5)\n\n# meta_df[meta_df['publication_number'] == 'US-695233-A']","metadata":{"execution":{"iopub.status.busy":"2024-07-03T05:34:50.072117Z","iopub.execute_input":"2024-07-03T05:34:50.072555Z","iopub.status.idle":"2024-07-03T05:35:22.353145Z","shell.execute_reply.started":"2024-07-03T05:34:50.072522Z","shell.execute_reply":"2024-07-03T05:35:22.351891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# filter metadata to keep only patents in test\nmeta_df_v2 = meta_df[meta_df['publication_number'].isin(allunique_pubnums)].reset_index(drop=True)\nmeta_df_v2","metadata":{"execution":{"iopub.status.busy":"2024-07-03T05:35:22.354523Z","iopub.execute_input":"2024-07-03T05:35:22.354898Z","iopub.status.idle":"2024-07-03T05:35:23.741666Z","shell.execute_reply.started":"2024-07-03T05:35:22.354868Z","shell.execute_reply":"2024-07-03T05:35:23.740398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.0 Feature Engineering","metadata":{}},{"cell_type":"code","source":"meta_df_v2['total_cpc_codes'] = meta_df_v2['cpc_codes'].apply(lambda x:len(x))\nmeta_df_v2.head(3)\n\n#max_value = meta_df_v2['total_cpc_codes'].max()\n#max_value","metadata":{"execution":{"iopub.status.busy":"2024-07-03T05:35:23.743316Z","iopub.execute_input":"2024-07-03T05:35:23.743789Z","iopub.status.idle":"2024-07-03T05:35:23.766045Z","shell.execute_reply.started":"2024-07-03T05:35:23.743750Z","shell.execute_reply":"2024-07-03T05:35:23.764708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to create a query string for Whoosh\ndef build_cpc_query(cpc_codes):\n    # Ensure cpc_codes is a list of strings\n    if isinstance(cpc_codes, np.ndarray):\n        cpc_codes = cpc_codes.tolist()\n    elif isinstance(cpc_codes, str):\n        cpc_codes = cpc_codes.split(\", \")\n    return \" OR \".join([f'cpc:{code.strip()}' for code in cpc_codes])\n\n# create queries\nqueries = []\nfor index, row in tqdm(meta_df_v2.iterrows(), total=len(meta_df)):\n    query = build_cpc_query(row['cpc_codes'])\n    queries.append(query)\n\nqueries","metadata":{"execution":{"iopub.status.busy":"2024-07-03T05:35:23.767717Z","iopub.execute_input":"2024-07-03T05:35:23.768509Z","iopub.status.idle":"2024-07-03T05:35:23.841781Z","shell.execute_reply.started":"2024-07-03T05:35:23.768466Z","shell.execute_reply":"2024-07-03T05:35:23.840416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.0 Whoosh\n\nWhoosh is a fast, featureful full-text indexing and searching library implemented in pure Python. It allows us to add search functionality to our applications, helping users find information quickly and efficiently. Think of it as a way to create a mini Google for our own data.\n\n**Key Concepts of Whoosh**\n* **Indexing:** This is the process of organizing data so that it can be searched efficiently. Whoosh takes the text data and creates an index, similar to an index in a book, which helps in fast retrieval.\n* **Searching:** Once data is indexed, we can perform searches on it. We can search for specific terms, phrases, or even perform complex queries.\n\nHowever, since here we are using a very small dataset, therefore I didn't set any indexing \n\n**How Whoosh Works**\n1. **Create an Index:** You define what data you want to index (e.g., titles, content of articles, tags).\n2. **Add Documents:** You add pieces of data (called documents) to the index.\n3. **Search the Index:** You perform searches on the indexed data to find relevant information.\n\nThese 3 steps work together to enable you to search\n* Loading the index gives you access to your data. Think of it like trying to access the library\n* Creating a searcher allows you to search through that data. It's like the librarian who helps you find the books you're looking for\n* Creating a query parser ensures that the searches you perform are understood and executed correctly. When users want to search for something, they usually type in a query, like a search term or a phrase. The query parser takes that query and turns it into a format that the searcher can understand and use to find relevant documents\n\n\n\n","metadata":{}},{"cell_type":"code","source":"# transform string to Whoosh query (less than 50 tokens) \nqueryValidator = whoosh_utils.QueryValidator()\n\nfinal_queries = []\nfor query in queries:\n    final_query = query\n    try:\n        # if validation is successful, final_query remains as the current query\n        queryValidator.validate_query(query)\n        final_query = query\n    except:\n        # if validation fails, set final_query to \"ti:device\".\n        final_query = \"ti:device\"\n    \n    # token count check - if the num of tokens in the query exceeds 50, set final_query to \"ti:mobile\"\n    if whoosh_utils.count_query_tokens(query) > 50:\n        final_query = \"ti:mobile\"\n    final_queries.append(final_query)\n\nfinal_queries\n# print the num of tokens and the text of each query in the final_queries list, but only for the first 15 queries\n#for query in final_queries[0:15]:\n    #print(\"\\n\", whoosh_utils.count_query_tokens(query), query)","metadata":{"execution":{"iopub.status.busy":"2024-07-03T05:35:23.843548Z","iopub.execute_input":"2024-07-03T05:35:23.843996Z","iopub.status.idle":"2024-07-03T05:35:23.916952Z","shell.execute_reply.started":"2024-07-03T05:35:23.843957Z","shell.execute_reply":"2024-07-03T05:35:23.915620Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.0 Submission","metadata":{}},{"cell_type":"code","source":"submission_df = pd.DataFrame(meta_df_v2['publication_number'])\nsubmission_df['query'] = final_queries","metadata":{"execution":{"iopub.status.busy":"2024-07-03T05:35:23.918481Z","iopub.execute_input":"2024-07-03T05:35:23.918908Z","iopub.status.idle":"2024-07-03T05:35:23.927259Z","shell.execute_reply.started":"2024-07-03T05:35:23.918875Z","shell.execute_reply":"2024-07-03T05:35:23.925500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# store the unique [allunique_pubnums_submission] from the sample submission dataset\nallunique_pubnums_submission = []\nfor num,chunk in tqdm(enumerate(pd.read_csv(\"/kaggle/input/uspto-explainable-ai/sample_submission.csv\",chunksize=1000))):\n    \n    unique_values = pd.unique(chunk.values.ravel())\n    unique_values = list(unique_values)\n    \n    allunique_pubnums_submission.extend(unique_values)\n    allunique_pubnums_submission = list(set(allunique_pubnums_submission))\n    \nlen(allunique_pubnums_submission)","metadata":{"execution":{"iopub.status.busy":"2024-07-03T05:35:23.931065Z","iopub.execute_input":"2024-07-03T05:35:23.931468Z","iopub.status.idle":"2024-07-03T05:35:23.959482Z","shell.execute_reply.started":"2024-07-03T05:35:23.931436Z","shell.execute_reply":"2024-07-03T05:35:23.958100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# filter meta_df_v2 (submission_df) to keep only patents in sample submission dataset\ndff_sample = pd.read_csv('/kaggle/input/uspto-explainable-ai/sample_submission.csv')\n\nsubmission_df = submission_df[submission_df['publication_number'].isin(allunique_pubnums_submission)].reset_index(drop=True)\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2024-07-03T05:35:23.960838Z","iopub.execute_input":"2024-07-03T05:35:23.961188Z","iopub.status.idle":"2024-07-03T05:35:23.980978Z","shell.execute_reply.started":"2024-07-03T05:35:23.961157Z","shell.execute_reply":"2024-07-03T05:35:23.979806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# store the submission dataset in csv \nsubmission_df.to_csv('submission.csv', index=False) \nsubmission_df = pd.read_csv('/kaggle/working/submission.csv')\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2024-07-03T05:35:23.982335Z","iopub.execute_input":"2024-07-03T05:35:23.982736Z","iopub.status.idle":"2024-07-03T05:35:24.007991Z","shell.execute_reply.started":"2024-07-03T05:35:23.982706Z","shell.execute_reply":"2024-07-03T05:35:24.006754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sample submission data type\nprint(\"sample submission dataset:\")\ndff_sample.dtypes","metadata":{"execution":{"iopub.status.busy":"2024-07-03T05:37:23.923823Z","iopub.execute_input":"2024-07-03T05:37:23.924280Z","iopub.status.idle":"2024-07-03T05:37:23.934446Z","shell.execute_reply.started":"2024-07-03T05:37:23.924240Z","shell.execute_reply":"2024-07-03T05:37:23.933178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# my submission data type\ndff = pd.read_csv('/kaggle/working/submission.csv')\nprint(\"my submission dataset:\")\ndff.dtypes","metadata":{"execution":{"iopub.status.busy":"2024-07-03T05:37:26.506891Z","iopub.execute_input":"2024-07-03T05:37:26.507312Z","iopub.status.idle":"2024-07-03T05:37:26.519987Z","shell.execute_reply.started":"2024-07-03T05:37:26.507279Z","shell.execute_reply":"2024-07-03T05:37:26.518583Z"},"trusted":true},"execution_count":null,"outputs":[]}]}