{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59575,"databundleVersionId":8060720,"sourceType":"competition"},{"sourceId":184355701,"sourceType":"kernelVersion"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Years for filtration\nstart_year = 2010\nend_year = 2013","metadata":{"execution":{"iopub.status.busy":"2024-07-10T11:34:38.061098Z","iopub.execute_input":"2024-07-10T11:34:38.061546Z","iopub.status.idle":"2024-07-10T11:34:38.06883Z","shell.execute_reply.started":"2024-07-10T11:34:38.061509Z","shell.execute_reply":"2024-07-10T11:34:38.067213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport h5py\nimport numpy as np\nimport ctypes\n\n# To clean RAM\nlibc = ctypes.CDLL(\"libc.so.6\")\n\n# Data paths\npatent_data = '/kaggle/input/uspto-explainable-ai/patent_data'\noutput_dir = '/kaggle/working/'\n\n# Function for converting a string into a byte array (memory saving)\ndef str_to_bytes(s):\n    return np.frombuffer(s.encode(), dtype=np.uint8)\n\n# Reading data from Parquet and saving to HDF5\nfor file_name in os.listdir(patent_data):\n    # Patent_data contains nan.parquet, which is not needed\n    if file_name.startswith('nan'):\n        continue\n        \n    else:\n        # Extracting year_month from filename\n        year_month = file_name.split('.')[0]\n\n        # Check that year_month falls within the given year range\n        year = int(year_month.split('_')[0])\n        \n        if start_year <= year <= end_year:\n            \n            # Path to the current Parquet file\n            file_path = os.path.join(patent_data, file_name)\n            \n            # Reading data from Parquet file\n            df = pd.read_parquet(file_path, columns=['publication_number', 'description'])\n            \n            # Saving each row to a separate HDF5 dataset\n            hdf5_file = os.path.join(output_dir, f\"{year_month}.h5\")\n            \n            with h5py.File(hdf5_file, 'w') as hf:\n                for index, row in df.iterrows():\n                    publication_number = row['publication_number']\n                    description = row['description']\n                    \n                    # Convert string to byte array to reduce size\n                    description_bytes = str_to_bytes(description)\n                    \n                    dataset_name = f\"{publication_number}\"\n                    hf.create_dataset(dataset_name, data=description_bytes, compression=\"gzip\", compression_opts=9)\n\n            print(f\"Processed {file_name} and saved to {hdf5_file}\")\n            libc.malloc_trim(0)\n","metadata":{"execution":{"iopub.status.busy":"2024-07-10T11:36:05.615202Z","iopub.execute_input":"2024-07-10T11:36:05.615635Z","iopub.status.idle":"2024-07-10T11:36:31.651305Z","shell.execute_reply.started":"2024-07-10T11:36:05.6156Z","shell.execute_reply":"2024-07-10T11:36:31.649917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}