{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59575,"databundleVersionId":8060720,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"We are provided with a train index of 200_000 samples. However, the nearest neighbors are not filtered accordingly. Instead, we find > 13_000_000 neighbours. Loading this full file will result in out-of-memory in kaggle.\n\nTo avoid this issue I prepared this little notebook as something you can import to get the relevant subset (which is not exactly 200k rows). This should save you some minutes in trying to get the relevant subset.","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\n\nos.mkdir(\"data\")\nos.mkdir(\"data/indexed_neighbors\")\n\ntr_idx = pd.read_json(\"/kaggle/input/uspto-explainable-ai/train_index_patent_ids.json\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-27T14:28:37.347159Z","iopub.execute_input":"2024-04-27T14:28:37.348408Z","iopub.status.idle":"2024-04-27T14:28:37.815260Z","shell.execute_reply.started":"2024-04-27T14:28:37.348365Z","shell.execute_reply":"2024-04-27T14:28:37.814506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# subsets of 2M rows each will fit into memory\nfor i in range(7): # 7 since 7*2_000_00 > 13_xxx_xxx\n    n = 2_000_000*i\n    # no need to skip rows on first load\n    if n == 0: \n        nn = pd.read_csv(\"/kaggle/input/uspto-explainable-ai/nearest_neighbors.csv\", nrows=2_000_000)\n    else: \n        # skip the already loaded rows; still load first row, as it is the header\n        skip = np.arange(1,n).tolist() # keep header\n        nn = pd.read_csv(\"/kaggle/input/uspto-explainable-ai/nearest_neighbors.csv\", skiprows=skip, nrows=2_000_000)\n    print(f\"loaded rows from {i*2_000_000} to {(i+1)*2_000_000}\")\n    \n    # filter loaded subset to only include patents from the train_idx\n    tr_idx_nn = nn[nn.publication_number.isin(tr_idx[0])]\n    print(f\"added {tr_idx_nn.shape[0]} rows that belong to the train index\")\n\n    # save 2M subset\n    tr_idx_nn.to_csv(f\"data/indexed_neighbors/tr_{i*2_000_000}_nn.csv\", index=False)\n    print(\"saved\")","metadata":{"execution":{"iopub.status.busy":"2024-04-27T14:28:37.817006Z","iopub.execute_input":"2024-04-27T14:28:37.817906Z","iopub.status.idle":"2024-04-27T14:39:51.729870Z","shell.execute_reply.started":"2024-04-27T14:28:37.817870Z","shell.execute_reply":"2024-04-27T14:39:51.728601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files = os.listdir(\"data/indexed_neighbors\")\n\ndf = pd.read_csv(\"data/indexed_neighbors/\"+files[0])\nfor f in files[1:]:\n    d = pd.read_csv(\"data/indexed_neighbors/\"+f)\n    df = pd.concat([df,d])\n    \ndf.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-27T14:39:51.733863Z","iopub.execute_input":"2024-04-27T14:39:51.734204Z","iopub.status.idle":"2024-04-27T14:39:58.039156Z","shell.execute_reply.started":"2024-04-27T14:39:51.734169Z","shell.execute_reply":"2024-04-27T14:39:58.038120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv(\"data/tr_idx_nn.csv\")\nprint(df.drop_duplicates(\"publication_number\").shape)\ndf.drop_duplicates(\"publication_number\").to_csv(\"data/tr_idx_nn_deduped.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-27T14:39:58.041373Z","iopub.execute_input":"2024-04-27T14:39:58.041670Z","iopub.status.idle":"2024-04-27T14:40:11.631798Z","shell.execute_reply.started":"2024-04-27T14:39:58.041646Z","shell.execute_reply":"2024-04-27T14:40:11.630895Z"},"trusted":true},"execution_count":null,"outputs":[]}]}