{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59575,"databundleVersionId":8060720,"sourceType":"competition"},{"sourceId":9049626,"sourceType":"datasetVersion","datasetId":5456218}],"dockerImageVersionId":30746,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport gc\nimport re\nimport glob\nfrom pathlib import Path\nimport random\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-28T03:39:04.662423Z","iopub.execute_input":"2024-07-28T03:39:04.662841Z","iopub.status.idle":"2024-07-28T03:39:05.171979Z","shell.execute_reply.started":"2024-07-28T03:39:04.662809Z","shell.execute_reply":"2024-07-28T03:39:05.170458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root_input_path = Path(\"/kaggle/input/uspto-explainable-ai/\")\npdata_root_path = os.path.join(root_input_path, \"patent_data\")\npmeta_path = os.path.join(root_input_path, \"patent_metadata.parquet\")\ntest_path = os.path.join(root_input_path, \"test.csv\")\nsubmission_path = os.path.join(root_input_path, \"sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-07-28T03:39:05.622359Z","iopub.execute_input":"2024-07-28T03:39:05.622908Z","iopub.status.idle":"2024-07-28T03:39:05.629904Z","shell.execute_reply.started":"2024-07-28T03:39:05.622874Z","shell.execute_reply":"2024-07-28T03:39:05.628481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pdata_files = glob.glob(os.path.join(pdata_root_path,\"*.parquet\"))\nyear_extraction_pattern = r\"(?:.+\\/)(\\d{4})\"\n\ndef extract_year(file_path):\n    year_extraction_pattern = r\"(?:.+\\/)(\\d{4})\"\n    return re.match(year_extraction_pattern,file_path)\n\ndef get_year(file_path):\n    match = extract_year(file_path)\n    if match:\n        return match.group(1)\ndef is_filed_after(file_path,year=1950):\n    file_year = get_year(file_path)\n    if file_year is not None:\n        if int(file_year) >=year:\n            return True\n    else:\n        return False\n    return False\n\nfiled_years = [int(get_year(file)) for file in pdata_files if get_year(file) is not None]\nprint(max(filed_years),min(filed_years))\nprint(len(pdata_files))\npdata_files = [file for file in pdata_files if is_filed_after(file,year=1950)]\nprint(len(pdata_files))\nmatch = re.match(year_extraction_pattern,pdata_files[0])\nif match:\n    print(match.group(1))\nprint(get_year(pdata_files[0]))\nrandom_files=random.sample(pdata_files,k=20)\nprint(random_files[:4])","metadata":{"execution":{"iopub.status.busy":"2024-07-28T03:39:06.313528Z","iopub.execute_input":"2024-07-28T03:39:06.313936Z","iopub.status.idle":"2024-07-28T03:39:06.751412Z","shell.execute_reply.started":"2024-07-28T03:39:06.313903Z","shell.execute_reply":"2024-07-28T03:39:06.750101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(2023-1950)*12","metadata":{"execution":{"iopub.status.busy":"2024-07-28T03:39:08.158374Z","iopub.execute_input":"2024-07-28T03:39:08.158797Z","iopub.status.idle":"2024-07-28T03:39:08.168705Z","shell.execute_reply.started":"2024-07-28T03:39:08.158763Z","shell.execute_reply":"2024-07-28T03:39:08.167277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pdata_files[0]","metadata":{"execution":{"iopub.status.busy":"2024-07-28T03:39:12.379899Z","iopub.execute_input":"2024-07-28T03:39:12.380412Z","iopub.status.idle":"2024-07-28T03:39:12.388896Z","shell.execute_reply.started":"2024-07-28T03:39:12.380372Z","shell.execute_reply":"2024-07-28T03:39:12.387578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pdata = [pd.read_parquet(file,columns=['publication_number','claims'],engine='pyarrow') for file in random_files]\nprint(len(pdata),type(pdata),len(pdata[0]))\npdata_df = pd.concat(pdata,ignore_index=True)\npdata_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-28T03:39:13.978789Z","iopub.execute_input":"2024-07-28T03:39:13.979220Z","iopub.status.idle":"2024-07-28T03:39:27.588442Z","shell.execute_reply.started":"2024-07-28T03:39:13.979185Z","shell.execute_reply":"2024-07-28T03:39:27.587123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nlen(pdata_df)","metadata":{"execution":{"iopub.status.busy":"2024-07-28T03:39:27.590645Z","iopub.execute_input":"2024-07-28T03:39:27.591046Z","iopub.status.idle":"2024-07-28T03:39:27.672367Z","shell.execute_reply.started":"2024-07-28T03:39:27.590995Z","shell.execute_reply":"2024-07-28T03:39:27.671064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pdata_df.iloc[0]['claims']","metadata":{"execution":{"iopub.status.busy":"2024-07-28T03:39:29.329342Z","iopub.execute_input":"2024-07-28T03:39:29.329740Z","iopub.status.idle":"2024-07-28T03:39:29.337815Z","shell.execute_reply.started":"2024-07-28T03:39:29.329709Z","shell.execute_reply":"2024-07-28T03:39:29.336639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def should_discard_split_0(split_0):\n    pattern = r'([^\\d{1,2}])'\n    match = re.match(pattern,split_0)\n    if match:\n        return True\n    return False\n\ndef claim_text_df(row,idx):\n    claims = row['claims']\n    split_pattern = r'(?:\\n\\s{5,9}\\d{1,2}\\s{0,1}.)'\n#                    r'\\n\\s{5,9}\\d{1,2}\\s{0,2}\\.'\n    cleaned_splits = [re.sub(r'[\\s]+$',\"\",re.sub(r'^\\s',\"\",re.sub(r'\\n',\"\", text))) for text in re.split(split_pattern,claims)]\n    cleaned_splits= cleaned_splits[1:] if should_discard_split_0(cleaned_splits[0]) else cleaned_splits\n    new_rows = [{\"publication_number\": row['publication_number'],'sequence_id':idx,'tokens': split} for idx,split in enumerate(cleaned_splits)]\n    intermediate_df = pd.DataFrame(new_rows)\n    if idx%30000 == 0:\n        print(len(intermediate_df),row['publication_number'])\n    return intermediate_df\n\nexample_df = claim_text_df(pdata_df.iloc[0],idx=0)\nexample_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-28T03:39:31.222952Z","iopub.execute_input":"2024-07-28T03:39:31.223832Z","iopub.status.idle":"2024-07-28T03:39:31.245861Z","shell.execute_reply.started":"2024-07-28T03:39:31.223792Z","shell.execute_reply":"2024-07-28T03:39:31.244314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-07-28T03:39:35.191873Z","iopub.execute_input":"2024-07-28T03:39:35.192306Z","iopub.status.idle":"2024-07-28T03:39:35.265856Z","shell.execute_reply.started":"2024-07-28T03:39:35.192271Z","shell.execute_reply":"2024-07-28T03:39:35.264303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"claim_text_dfs = [claim_text_df(row,idx) for idx,row in pdata_df.iterrows()]\nannotation_df = pd.concat(claim_text_dfs,ignore_index=True)\nlen(annotation_df)","metadata":{"execution":{"iopub.status.busy":"2024-07-28T03:39:36.652470Z","iopub.execute_input":"2024-07-28T03:39:36.652875Z","iopub.status.idle":"2024-07-28T03:50:11.984310Z","shell.execute_reply.started":"2024-07-28T03:39:36.652844Z","shell.execute_reply":"2024-07-28T03:50:11.982627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#pdata_df[pdata_df['publication_number']=='US-D619669-S'].iloc[0]['claims']","metadata":{"execution":{"iopub.status.busy":"2024-07-26T13:55:51.347062Z","iopub.execute_input":"2024-07-26T13:55:51.347520Z","iopub.status.idle":"2024-07-26T13:55:51.353094Z","shell.execute_reply.started":"2024-07-26T13:55:51.347479Z","shell.execute_reply":"2024-07-26T13:55:51.351722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"annotation_df = annotation_df[annotation_df['tokens'] != \"\"]\n","metadata":{"execution":{"iopub.status.busy":"2024-07-26T13:55:51.354712Z","iopub.execute_input":"2024-07-26T13:55:51.355618Z","iopub.status.idle":"2024-07-26T13:55:51.929997Z","shell.execute_reply.started":"2024-07-26T13:55:51.355579Z","shell.execute_reply":"2024-07-26T13:55:51.928591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"annotation_df.tail(30)","metadata":{"execution":{"iopub.status.busy":"2024-07-26T13:55:51.931713Z","iopub.execute_input":"2024-07-26T13:55:51.932141Z","iopub.status.idle":"2024-07-26T13:55:51.948267Z","shell.execute_reply.started":"2024-07-26T13:55:51.932100Z","shell.execute_reply":"2024-07-26T13:55:51.947063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"annotation_df.to_csv(\"/kaggle/working/annotation_df.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2024-07-26T13:55:51.952039Z","iopub.execute_input":"2024-07-26T13:55:51.952484Z","iopub.status.idle":"2024-07-26T13:56:26.980943Z","shell.execute_reply.started":"2024-07-26T13:55:51.952444Z","shell.execute_reply":"2024-07-26T13:56:26.979505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\nannotation_df.groupby(by='publication_number')\npublication_ids = annotation_df['publication_number'].unique()\nsample_publication_ids = random.sample(sorted(publication_ids),k=10)\nsample_publications = annotation_df[annotation_df['publication_number'].isin(sample_publication_ids)]\nsample_publications.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-07-28T04:54:44.150112Z","iopub.execute_input":"2024-07-28T04:54:44.150857Z","iopub.status.idle":"2024-07-28T04:54:45.054854Z","shell.execute_reply.started":"2024-07-28T04:54:44.150820Z","shell.execute_reply":"2024-07-28T04:54:45.053681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(sample_publications))","metadata":{"execution":{"iopub.status.busy":"2024-07-28T04:55:13.477317Z","iopub.execute_input":"2024-07-28T04:55:13.478079Z","iopub.status.idle":"2024-07-28T04:55:13.485948Z","shell.execute_reply.started":"2024-07-28T04:55:13.478006Z","shell.execute_reply":"2024-07-28T04:55:13.484280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n#os.remove(\"/kaggle/working/sample_publications.csv\")\nos.remove(\"/kaggle/working/annotation_df.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-07-28T04:57:00.324518Z","iopub.execute_input":"2024-07-28T04:57:00.324927Z","iopub.status.idle":"2024-07-28T04:57:00.705575Z","shell.execute_reply.started":"2024-07-28T04:57:00.324894Z","shell.execute_reply":"2024-07-28T04:57:00.703884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_publications.to_csv(\"/kaggle/working/sample_publications.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2024-07-28T04:57:05.131975Z","iopub.execute_input":"2024-07-28T04:57:05.132387Z","iopub.status.idle":"2024-07-28T04:57:05.145037Z","shell.execute_reply.started":"2024-07-28T04:57:05.132357Z","shell.execute_reply":"2024-07-28T04:57:05.143573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport math\n\ndef human_readable_size(size_bytes):\n    if size_bytes == 0:\n        return \"0B\"\n    size_name = (\"B\", \"KB\", \"MB\", \"GB\", \"TB\", \"PB\", \"EB\", \"ZB\", \"YB\")\n    i = int(math.floor(math.log(size_bytes, 1024)))\n    p = math.pow(1024, i)\n    s = round(size_bytes / p, 2)\n    return \"%s %s\" % (s, size_name[i])\n\nfile_path = '/kaggle/working/sample_publications.csv'\nfile_size = os.path.getsize(file_path)\n\nprint(f\"The size of the file is {human_readable_size(file_size)}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-07-28T04:57:16.794234Z","iopub.execute_input":"2024-07-28T04:57:16.794689Z","iopub.status.idle":"2024-07-28T04:57:16.806103Z","shell.execute_reply.started":"2024-07-28T04:57:16.794654Z","shell.execute_reply":"2024-07-28T04:57:16.804430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -qqq argilla --pre","metadata":{"execution":{"iopub.status.busy":"2024-07-28T04:57:51.642979Z","iopub.execute_input":"2024-07-28T04:57:51.646185Z","iopub.status.idle":"2024-07-28T04:58:13.444540Z","shell.execute_reply.started":"2024-07-28T04:57:51.646138Z","shell.execute_reply":"2024-07-28T04:58:13.442815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import transformers\n\nimport re\nimport argilla as rg\nfrom datasets import load_dataset, Dataset, DatasetDict\nfrom transformers import TrainingArguments\ntransformers.__version__, rg.__version__","metadata":{"execution":{"iopub.status.busy":"2024-07-28T04:58:13.447601Z","iopub.execute_input":"2024-07-28T04:58:13.448063Z","iopub.status.idle":"2024-07-28T04:58:26.995811Z","shell.execute_reply.started":"2024-07-28T04:58:13.447999Z","shell.execute_reply":"2024-07-28T04:58:26.994539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row = sample_publications.iloc[0]\nprint(row['tokens'])","metadata":{"execution":{"iopub.status.busy":"2024-07-28T04:58:55.427773Z","iopub.execute_input":"2024-07-28T04:58:55.428585Z","iopub.status.idle":"2024-07-28T04:58:55.435916Z","shell.execute_reply.started":"2024-07-28T04:58:55.428544Z","shell.execute_reply":"2024-07-28T04:58:55.434580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"records = [\n    rg.Record(\n        fields=\n            {\"tokens\": \"\".join(row[\"tokens\"])\n            ,'document_id':str(row['publication_number'])\n            ,'sentence_id':str(row['sequence_id'])\n            })\n  for _,row in sample_publications.iterrows()\n  ]\nprint(records[0])\n#print(re.sub(r\"\\s\",\"\",records[0].fields['tokens']))","metadata":{"execution":{"iopub.status.busy":"2024-07-28T04:59:01.640267Z","iopub.execute_input":"2024-07-28T04:59:01.640723Z","iopub.status.idle":"2024-07-28T04:59:01.682560Z","shell.execute_reply.started":"2024-07-28T04:59:01.640689Z","shell.execute_reply":"2024-07-28T04:59:01.681050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}