{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59575,"databundleVersionId":8060720,"sourceType":"competition"},{"sourceId":8479599,"sourceType":"datasetVersion","datasetId":4517815},{"sourceId":174185912,"sourceType":"kernelVersion"}],"dockerImageVersionId":30698,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom glob import glob\n#import whoosh_utils #are we overwriting patched version?\nimport tqdm, re, gc\nimport pyarrow.parquet as pq\nfrom collections import Counter\nimport warnings\nwarnings.filterwarnings('ignore')\n\ngc.enable()\np = '/kaggle/input/uspto-explainable-ai/'\n\n#meta = pd.read_parquet(p+'patent_metadata.parquet')\ntest = pd.read_csv(p+'test.csv') #test is also train\n#sub = pd.read_csv(p+'sample_submission.csv')\n#nn = pd.read_csv(p+'nearest_neighbors.csv')\n\n#Holding from Submission for ** MEMORY **\n#train_idx = whoosh_utils.load_index(p + 'train_index')\n#searcher = whoosh_utils.get_searcher(train_idx)\n#qp = whoosh_utils.get_query_parser()\n\nsw = ['an', 'are', 'by', 'for', 'if', 'into', 'is', 'no', 'not', 'of', 'on', 'such', 'that', 'the', 'their', 'then', 'there', 'these', 'they', 'this', 'to', 'was', 'will']\nsw += ['and', 'or', 'xor', 'not']","metadata":{"execution":{"iopub.status.busy":"2024-05-24T08:52:20.631521Z","iopub.execute_input":"2024-05-24T08:52:20.632672Z","iopub.status.idle":"2024-05-24T08:52:21.864019Z","shell.execute_reply.started":"2024-05-24T08:52:20.632618Z","shell.execute_reply":"2024-05-24T08:52:21.862813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Add Simple Token Info to Test","metadata":{}},{"cell_type":"code","source":"%%time\n\n#initiate with value - similar to NA replacement\ntest['query'] = 'ti:amplitude OR ti:mississippi'\ntest['tokens'] = ''\ntest = test[['publication_number', 'query', 'tokens']]\nids = list(test['publication_number'].values)\ntest = test.set_index('publication_number')\n\nfs = sorted(glob(p + 'patent_data/**'))\nfor f in tqdm.tqdm(fs):\n    df = pd.read_parquet(f,  columns=['publication_number', 'title'])\n    df = df[df['publication_number'].isin(ids)]\n    if len(df)>0:\n        df = df.reset_index(drop=True)\n        for i in range(len(df)):\n            test.loc[df['publication_number'][i], 'tokens'] = df['title'][i] #+ ' ' + df['abstract'][i] + ' ' + df['claims'][i] + ' ' + df['description'][i]","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-05-24T08:52:21.866486Z","iopub.execute_input":"2024-05-24T08:52:21.866910Z","iopub.status.idle":"2024-05-24T08:53:33.764730Z","shell.execute_reply.started":"2024-05-24T08:52:21.866879Z","shell.execute_reply":"2024-05-24T08:53:33.763737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def getQuery(t, limit=8): #50 token max limit\n    global sw\n    t = t.lower()\n    t = re.sub('[^a-z ]', ' ', t)\n    t = t.split(' ')\n    t = [w for w in t if w not in sw]\n    t = [w for w in t if len(w)>1]\n    t = t[:limit*2] #First Pass - Limit\n    t = list(set(t)) #Unique only\n    t = t[:limit] #Second Pass - Limit\n    \n    #ti: title\n    #ab: abstract\n    #clm: claims\n    #detd: description\n    #cpc: cpc codes\n    #OR, AND, NOT, XOR, *, ?, $\n    #[WORD1] ADJ[0-9] [WORD2], [WORD1] NEAR[0-9] [WORD2] \n\n    t = ['ti:' + w for w in t] #add for title search only\n\n    if len(t)>2:\n        q = '('*(len(t)-1) + t[0] + ' OR '\n        q += ') OR '.join(t[1:]) + ')'\n    elif len(t)==2:\n        q = ' OR '.join(t)\n    elif len(t)==1:\n        q = t[0]\n    else:\n        q = 'ti:amplitude OR ti:mississippi' #Standard Submission Query\n    #Add more clean up and query options later\n    return q\n\ntest['query'] = test['tokens'].map(lambda x: getQuery(str(x)))\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-24T08:55:46.543514Z","iopub.execute_input":"2024-05-24T08:55:46.543926Z","iopub.status.idle":"2024-05-24T08:55:46.570341Z","shell.execute_reply.started":"2024-05-24T08:55:46.543894Z","shell.execute_reply":"2024-05-24T08:55:46.569231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Add Query Evaluation","metadata":{}},{"cell_type":"code","source":"#def wqt(query):\n    #global qp\n    #try:\n        #print(qp.parse(query))\n        #return whoosh_utils.execute_query(query, qp, searcher)[:50]\n    #except:\n        #print(\"You can't use ADJ or NEAR on the CPC field.\")\n        #return ['Error 404 Search Result!']\n\n#test['results'] = test['query'].map(lambda x: wqt(str(x)))\n\n#Add Evaluation Here\n\n#test.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-24T08:53:34.913753Z","iopub.status.idle":"2024-05-24T08:53:34.914296Z","shell.execute_reply.started":"2024-05-24T08:53:34.914005Z","shell.execute_reply":"2024-05-24T08:53:34.914029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = test.reset_index()\ntest[['publication_number','query']].to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-24T08:56:21.550059Z","iopub.execute_input":"2024-05-24T08:56:21.551196Z","iopub.status.idle":"2024-05-24T08:56:21.562740Z","shell.execute_reply.started":"2024-05-24T08:56:21.551155Z","shell.execute_reply":"2024-05-24T08:56:21.561301Z"},"trusted":true},"execution_count":null,"outputs":[]}]}