{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59575,"databundleVersionId":8060720,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import polars as pl\nimport pyarrow.dataset as ds\nimport pyarrow as pa\nimport os\nfrom tqdm import tqdm\nimport gc\nimport csv","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-09T08:00:41.693652Z","iopub.execute_input":"2024-07-09T08:00:41.694049Z","iopub.status.idle":"2024-07-09T08:00:43.791912Z","shell.execute_reply.started":"2024-07-09T08:00:41.694018Z","shell.execute_reply":"2024-07-09T08:00:43.790621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root_dir = '/kaggle/input/uspto-explainable-ai'","metadata":{"execution":{"iopub.status.busy":"2024-07-09T08:00:43.794583Z","iopub.execute_input":"2024-07-09T08:00:43.795228Z","iopub.status.idle":"2024-07-09T08:00:43.801163Z","shell.execute_reply.started":"2024-07-09T08:00:43.795185Z","shell.execute_reply":"2024-07-09T08:00:43.799611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_dir = os.path.join(root_dir, 'patent_data')\n\npartitioning = ds.FilenamePartitioning(pa.schema([('year', pa.int16()), ('month', pa.int8())]))\n\nvalid_partitions = []\nfor f in os.listdir(dataset_dir):\n    if f == 'nan_nan.parquet':\n        continue\n    partitioning.parse(f)\n    valid_partitions.append(f)","metadata":{"execution":{"iopub.status.busy":"2024-07-09T08:00:43.803263Z","iopub.execute_input":"2024-07-09T08:00:43.804149Z","iopub.status.idle":"2024-07-09T08:00:44.299070Z","shell.execute_reply.started":"2024-07-09T08:00:43.804104Z","shell.execute_reply":"2024-07-09T08:00:44.297504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_lf = pl.scan_parquet(os.path.join(root_dir, 'patent_metadata.parquet'))\nmetadata_lf = metadata_lf.with_columns(\n    pl.format(\"{}_{}.parquet\", pl.col('publication_date').dt.year(), pl.col('publication_date').dt.month()).alias(\"filename\")\n)\nmetadata_df = metadata_lf.select(['publication_number', 'filename']).collect()\n\nmetadata_df","metadata":{"execution":{"iopub.status.busy":"2024-07-09T08:00:44.301568Z","iopub.execute_input":"2024-07-09T08:00:44.302062Z","iopub.status.idle":"2024-07-09T08:00:46.666692Z","shell.execute_reply.started":"2024-07-09T08:00:44.302014Z","shell.execute_reply":"2024-07-09T08:00:46.664606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_lf.select('publication_date').collect().null_count()","metadata":{"execution":{"iopub.status.busy":"2024-07-09T08:00:46.671525Z","iopub.execute_input":"2024-07-09T08:00:46.672499Z","iopub.status.idle":"2024-07-09T08:00:46.796240Z","shell.execute_reply.started":"2024-07-09T08:00:46.672451Z","shell.execute_reply":"2024-07-09T08:00:46.794016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count = 0\n\nmissing_meta_raw = pl.DataFrame()\nmissing_raw_meta = pl.DataFrame()\n\nfor filename, data in metadata_df.group_by(['filename']):\n    if count%100 == 0:\n        print(missing_meta_raw)\n        print(missing_raw_meta)\n    if filename[0] is None:\n        missing_meta_raw = pl.concat([missing_meta_raw, data]) # publication date/filename is None, no way to fetch raw data\n        continue\n    raw_data = pl.read_parquet(os.path.join(dataset_dir, filename[0]))\n    meta_raw = data.join(raw_data, on='publication_number', how='anti')\n    if not meta_raw.is_empty():\n        print(f'Missing raw data {meta_raw.shape} {filename}')\n        missing_meta_raw = pl.concat([missing_meta_raw, meta_raw])\n    \n    raw_meta = raw_data.join(data, on='publication_number', how='anti')\n    if not raw_meta.is_empty():\n        print(f'Missing metadata {raw_meta.shape} {filename}')\n        missing_raw_meta = pl.concat([missing_raw_meta, raw_meta])\n        \n    del raw_data, meta_raw, raw_meta\n    gc.collect()\n    count += 1","metadata":{"execution":{"iopub.status.busy":"2024-07-09T08:00:46.798261Z","iopub.execute_input":"2024-07-09T08:00:46.799015Z","iopub.status.idle":"2024-07-09T09:07:01.224792Z","shell.execute_reply.started":"2024-07-09T08:00:46.798968Z","shell.execute_reply":"2024-07-09T09:07:01.223229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_meta_raw","metadata":{"execution":{"iopub.status.busy":"2024-07-09T09:07:01.226736Z","iopub.execute_input":"2024-07-09T09:07:01.227279Z","iopub.status.idle":"2024-07-09T09:07:01.240627Z","shell.execute_reply.started":"2024-07-09T09:07:01.227245Z","shell.execute_reply":"2024-07-09T09:07:01.238577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_raw_meta","metadata":{"execution":{"iopub.status.busy":"2024-07-09T09:07:01.242256Z","iopub.execute_input":"2024-07-09T09:07:01.243647Z","iopub.status.idle":"2024-07-09T09:07:01.269296Z","shell.execute_reply.started":"2024-07-09T09:07:01.243608Z","shell.execute_reply":"2024-07-09T09:07:01.268015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_dict = dict(metadata_df.iter_rows())","metadata":{"execution":{"iopub.status.busy":"2024-07-09T09:07:01.270648Z","iopub.execute_input":"2024-07-09T09:07:01.271077Z","iopub.status.idle":"2024-07-09T09:07:16.148238Z","shell.execute_reply.started":"2024-07-09T09:07:01.271039Z","shell.execute_reply":"2024-07-09T09:07:16.146218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(os.path.join(root_dir, 'nearest_neighbors.csv')) as csvfile:  \n    data = csv.reader(csvfile)\n    \n    count = 0\n    for row in tqdm(data):\n        if count == 0: # skip header\n            count += 1\n            continue\n            \n        for i in row:\n            if i not in metadata_dict:\n                print(i)","metadata":{"execution":{"iopub.status.busy":"2024-07-09T09:07:16.150599Z","iopub.execute_input":"2024-07-09T09:07:16.151178Z","iopub.status.idle":"2024-07-09T09:24:22.547232Z","shell.execute_reply.started":"2024-07-09T09:07:16.151135Z","shell.execute_reply":"2024-07-09T09:24:22.545867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_meta_raw.write_parquet('missing_meta_raw.parquet')","metadata":{"execution":{"iopub.status.busy":"2024-07-09T09:24:22.548967Z","iopub.execute_input":"2024-07-09T09:24:22.549354Z","iopub.status.idle":"2024-07-09T09:24:22.602535Z","shell.execute_reply.started":"2024-07-09T09:24:22.549322Z","shell.execute_reply":"2024-07-09T09:24:22.601307Z"},"trusted":true},"execution_count":null,"outputs":[]}]}