{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"},{"sourceId":8118152,"sourceType":"datasetVersion","datasetId":4796619},{"sourceId":8119369,"sourceType":"datasetVersion","datasetId":4797480}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook is meant to compress the BELKA Dataset and give a nice interface to interact with the compressed DB.\n\nThe compression strategy is as follows: \n1. Treating all the building block columns as categorical variables. Instead of keeping the whole SMILES string, we keep a\nindex (np.int16) and then we keep a mapping between the index and the full SMILES string\n2. The original DB keeps a separate row for each of the proteins, even though the building blocks and molecules are the same. We save it all in one row instead.\n3. Drop the ID column\n\nThe code goes over the dataset row by row and compresses each separately. this means that theoretically we never need to have the whole DB in memory. But practically I could not run this on the Kaggle CPU (with 30GB RAM), because the compression process uses some extra variables for more efficient search when compressing which takes too much memory, this might be solvable but will likely slow down the compression\n\n\nYou can find the dataset here https://www.kaggle.com/datasets/elongm/belka-compressed-dataset\nand a notebook of how to load and use it here https://www.kaggle.com/code/elongm/belka-compressed-dataset-usage","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport os\nimport pandas as pd\nimport pyarrow.parquet as pq\nimport tqdm\nimport pickle\nimport sys","metadata":{"execution":{"iopub.status.busy":"2024-04-14T21:16:08.463482Z","iopub.execute_input":"2024-04-14T21:16:08.463968Z","iopub.status.idle":"2024-04-14T21:16:08.470279Z","shell.execute_reply.started":"2024-04-14T21:16:08.463926Z","shell.execute_reply":"2024-04-14T21:16:08.469008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class bi_directional_dict(object):\n    def __init__(self):\n        self._next_index = 0\n        self.key_to_index = {}\n        self.index_to_key = []\n    \n    def get_or_create_index(self, element):\n        if not element in self.key_to_index:\n            indx = self._next_index  \n            self.key_to_index[element] = indx\n            self.index_to_key.append(element)\n            self._next_index += 1\n            return indx\n        return self.key_to_index[element]\n    \n    def get_key_of_index(self, index):\n        return self.index_to_key[index]","metadata":{"execution":{"iopub.status.busy":"2024-04-14T21:29:25.783183Z","iopub.execute_input":"2024-04-14T21:29:25.783684Z","iopub.status.idle":"2024-04-14T21:29:25.791838Z","shell.execute_reply.started":"2024-04-14T21:29:25.783643Z","shell.execute_reply":"2024-04-14T21:29:25.790493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"org_db_columns = ['id', # 0\n 'buildingblock1_smiles', # 1  \n 'buildingblock2_smiles',# 2\n 'buildingblock3_smiles', # 3\n 'molecule_smiles', # 4\n 'protein_name', # 5\n 'binds' # 6\n]\n\norg_db_col_2_inx = {k:v for v,k in enumerate(org_db_columns)}\n\n\ncompressed_db_columns = [\"buildingblock1_smiles\", # index for building_block_keys \n        \"buildingblock2_smiles\", \n        \"buildingblock3_smiles\", \n        \"molecule_smiles\", \n        \"binds_sEH\", \n        \"binds_HSA\", \n        \"binds_BRD4\"]\n    \ncompressed_db_column_to_index = {k:v for v,k in enumerate(compressed_db_columns)}\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-14T21:29:26.148973Z","iopub.execute_input":"2024-04-14T21:29:26.149434Z","iopub.status.idle":"2024-04-14T21:29:26.156888Z","shell.execute_reply.started":"2024-04-14T21:29:26.149389Z","shell.execute_reply":"2024-04-14T21:29:26.155633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CreateCompressedDB(object):\n    def __init__(self):\n        \n        # building_block_keys key - the SMILE building block/protein string, value - a unique int index.\n        self.building_blocks_hash = bi_directional_dict()\n\n        self.building_block_combos_to_db_index = {} # key is the contactating the 4 building_block_key and value is the db index of that combo\n        self.db_next_index = 0\n        \n        self.db_columns = compressed_db_columns\n        self.db_column_to_index = compressed_db_column_to_index\n        \n        self.empy_db_line = [None] * len(self.db_columns)\n\n        self.compressed_db = []\n        \n    def save_final_db_to_disk(self, path=\"/kaggle/working/\"):\n        \n        df_a = np.array(self.compressed_db, dtype=object)\n        df = pd.DataFrame({'buildingblock1_smiles': np.asarray(df_a[:,0], dtype = np.int16),\n                           'buildingblock2_smiles': np.asarray(df_a[:,1], dtype = np.int16),\n                           'buildingblock3_smiles': np.asarray(df_a[:,2], dtype = np.int16),\n                           'molecule_smiles': np.asarray(df_a[:,3], dtype = str),\n                           'binds_sEH': np.asarray(df_a[:,4], dtype = np.byte),\n                           'binds_HSA': np.asarray(df_a[:,5], dtype = np.byte),\n                           'binds_BRD4' :np.asarray(df_a[:,6], dtype = np.byte)})\n\n        df.to_parquet(os.path.join(path, \"df.parquet\"), index=False)\n        df.to_csv(os.path.join(path, \"df.csv\"), index=False)\n              \n        with open(os.path.join(path, \"building_block_hashes.pkl\"), \"wb\") as f:\n            f.write(pickle.dumps(self.building_blocks_hash))\n        \n        \n    def create_empy_db_line(self):\n        self.compressed_db.append(self.empy_db_line.copy())\n        \n    def update_db_line(self, row_id, values_dict_to_update):\n        for k in values_dict_to_update.keys():\n            column_index = self.db_column_to_index[k]\n            self.compressed_db[row_id][column_index] = values_dict_to_update[k]\n            \n    def get_or_create_building_block_index(self, building_block):\n        return self.building_blocks_hash.get_or_create_index(building_block)\n        \n    def get_or_create_db_index(self, block_1_index, block_2_index, block_3_index, mol):\n        indices_concat = f\"{block_1_index}_{block_2_index}_{block_3_index}\"\n\n        if not indices_concat in self.building_block_combos_to_db_index: \n            indx = self.db_next_index\n            self.building_block_combos_to_db_index[indices_concat] = indx\n\n            #create db\n            self.create_empy_db_line()\n            self.update_db_line(indx, {   \"buildingblock1_smiles\": block_1_index,\n                                          \"buildingblock2_smiles\": block_2_index,\n                                          \"buildingblock3_smiles\": block_3_index,\n                                          \"molecule_smiles\":mol ,\n                                          \"binds_sEH\": -1, \n                                          \"binds_HSA\": -1, \n                                          \"binds_BRD4\": -1})\n\n            self.db_next_index += 1\n            return indx\n        \n        indx = self.building_block_combos_to_db_index[indices_concat]\n        \n        if self.compressed_db[indx][self.db_column_to_index[\"molecule_smiles\"]] != mol:\n            raise Exception(\"Same building blocks but not same mol!\")\n        \n        return indx \n    \n    def get_db_index(self, buildingblock1_smiles, buildingblock2_smiles, buildingblock3_smiles, molecule_smiles):\n            \n        block_1_index = self.get_or_create_building_block_index(buildingblock1_smiles)\n        block_2_index = self.get_or_create_building_block_index(buildingblock2_smiles)\n        block_3_index = self.get_or_create_building_block_index(buildingblock3_smiles)\n\n        return self.get_or_create_db_index(block_1_index, block_2_index, block_3_index, molecule_smiles)\n\n\n    def add_row_from_org_db(self, buildingblock1_smiles, buildingblock2_smiles, buildingblock3_smiles, molecule_smiles, protein_name, idx, binds):\n        db_index = self.get_db_index(buildingblock1_smiles, buildingblock2_smiles, buildingblock3_smiles, molecule_smiles)\n\n        update_dict = {#f\"id_{protein_name}\": idx,\n                      f\"binds_{protein_name}\": bool(binds) }\n\n        self.update_db_line(db_index, update_dict)\n        \n    def add_numpy_df(self, numpy_df):\n        for i in range(numpy_df.shape[0]):\n            row = numpy_df[i]\n            self.add_row_from_org_db(row[1], \n                               row[2], \n                               row[3], \n                               row[4], \n                               row[5], \n                               row[0], \n                               row[6])\n            if i % 10000000 == 0:\n                print(i)\n    \n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-14T21:29:26.644428Z","iopub.execute_input":"2024-04-14T21:29:26.644897Z","iopub.status.idle":"2024-04-14T21:29:26.666595Z","shell.execute_reply.started":"2024-04-14T21:29:26.644859Z","shell.execute_reply":"2024-04-14T21:29:26.665326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Accessing the Compressed Dataset Helper\n\nCompressedDB provides a nice way to access the compressed Dataset, and decompress the data on access.\n\nusing slicing return will return the decompressed numpy array.\n\nif you want to get a DataFrame in return you can use compresseddb.slice_as_df()\n\nNote that these are only copies of the comressedDB! if you want to edit the underlying dataset you can access it using the compressed_db variable.","metadata":{}},{"cell_type":"code","source":"class CompressedDB(object):\n    def __init__(self, compressed_db, building_blocks_hashes):\n        \"\"\"\n        compressed_db : can be the actual object or path to csv file\n        building_blocks_hashes : can be the actual object or path to pickle file\n        \"\"\"\n        self.compressed_db = compressed_db\n        self.building_blocks_hash = building_blocks_hashes\n        \n        if type(self.compressed_db) == str:\n            self.compressed_db = pd.read_csv(self.compressed_db, dtype={'buildingblock1_smiles': np.int16,\n                   'buildingblock2_smiles': np.int16,\n                   'buildingblock3_smiles': np.int16,\n                   'molecule_smiles': str,\n                   'binds_sEH': np.byte,\n                   'binds_HSA':  np.byte,\n                   'binds_BRD4': np.byte})\n        \n        if type(self.building_blocks_hash) == str:\n            with open(self.building_blocks_hash, \"rb\") as f:\n                self.building_blocks_hash = pickle.loads(f.read())\n        \n    def __getitem__(self, index):\n        if isinstance(index, slice):\n            sliced_db = self.compressed_db.__getitem__(index).to_numpy().copy()\n            bb1_indices = sliced_db[:,0]\n            bb2_indices = sliced_db[:,1]\n            bb3_indices = sliced_db[:,2]\n\n            exapnded_bb1 = np.array([self.building_blocks_hash.get_key_of_index(i) for i in bb1_indices])\n            exapnded_bb2 = np.array([self.building_blocks_hash.get_key_of_index(i) for i in bb2_indices])\n            exapnded_bb3 = np.array([self.building_blocks_hash.get_key_of_index(i) for i in bb3_indices])\n\n            sliced_db[:,0] = exapnded_bb1\n            sliced_db[:,1] = exapnded_bb2\n            sliced_db[:,2] = exapnded_bb3\n            return sliced_db\n        \n        compressed_db_row = self.compressed_db.iloc[index].to_numpy().copy()\n        compressed_db_row[0] = self.building_blocks_hash.get_key_of_index(compressed_db_row[0])\n        compressed_db_row[1] = self.building_blocks_hash.get_key_of_index(compressed_db_row[1])\n        compressed_db_row[2] = self.building_blocks_hash.get_key_of_index(compressed_db_row[2])\n        return compressed_db_row\n    \n    \n\n    def slice_as_df(self, start, stop):\n        slice_db = self[start: stop]\n        df = pd.DataFrame(slice_db, columns=compressed_db_columns)\n        return df.astype({'binds_sEH': np.float16, # boolean dtype allow NA \n                        'binds_HSA': np.float16, \n                        'binds_BRD4': np.float16,\n                        'buildingblock1_smiles': 'str',\n                        'buildingblock2_smiles': 'str',\n                        'buildingblock3_smiles': 'str', \n                        'molecule_smiles':\"str\"})\n        \n        ","metadata":{"execution":{"iopub.status.busy":"2024-04-14T21:43:16.037218Z","iopub.execute_input":"2024-04-14T21:43:16.037669Z","iopub.status.idle":"2024-04-14T21:43:16.052239Z","shell.execute_reply.started":"2024-04-14T21:43:16.037637Z","shell.execute_reply":"2024-04-14T21:43:16.050646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create DB IF TPU","metadata":{}},{"cell_type":"code","source":"%%time\nparquet_file = pd.read_parquet(r\"/kaggle/input/leash-BELKA/train.parquet\")\nnumpy_df = parquet_file.to_numpy()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ncd = CreateCompressedDB()\ncd.add_numpy_df(numpy_df)\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create DB if no tpu\nOne the kaggle CPU with 30GB this currenly still use too much memory :/","metadata":{}},{"cell_type":"code","source":"# %%time\n# cd = CreateCompressedDB()\n# parquet_file = pq.ParquetFile(r\"/kaggle/input/leash-BELKA/train.parquet\")\n# for g in range(0, parquet_file.num_row_groups):\n#     table = parquet_file.read_row_group(g)\n#     chunk_df = table.to_pandas(split_blocks=True, self_destruct=True)\n#     numpy_df = chunk_df.to_numpy()\n#     cd.add_numpy_df(numpy_df)\n#     break","metadata":{"execution":{"iopub.status.busy":"2024-04-14T21:29:35.553049Z","iopub.execute_input":"2024-04-14T21:29:35.553818Z","iopub.status.idle":"2024-04-14T21:29:42.212926Z","shell.execute_reply.started":"2024-04-14T21:29:35.553772Z","shell.execute_reply":"2024-04-14T21:29:42.211458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save DB to Disk","metadata":{}},{"cell_type":"code","source":"cd.save_final_db_to_disk()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T21:29:56.492583Z","iopub.execute_input":"2024-04-14T21:29:56.493458Z","iopub.status.idle":"2024-04-14T21:29:58.600313Z","shell.execute_reply.started":"2024-04-14T21:29:56.493417Z","shell.execute_reply":"2024-04-14T21:29:58.599004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# load and use Dataset\n\nTo load the DB you need to provide df.csv (holds the compressed Dataset) and building_block_hashes.pkl (hold the map to full building blocks). \n\n","metadata":{}},{"cell_type":"code","source":"# # make sure you have this class defined since we need it to unpickle\n\n# class bi_directional_dict(object):\n#     def __init__(self):\n#         self._next_index = 0\n#         self.key_to_index = {}\n#         self.index_to_key = []\n    \n#     def get_or_create_index(self, element):\n#         if not element in self.key_to_index:\n#             indx = self._next_index  \n#             self.key_to_index[element] = indx\n#             self.index_to_key.append(element)\n#             self._next_index += 1\n#             return indx\n#         return self.key_to_index[element]\n    \n#     def get_key_of_index(self, index):\n#         return self.index_to_key[index]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cdb = CompressedDB(r\"/kaggle/input/belka-compressed-db/df.csv\", r\"/kaggle/input/belka-compressed-db/building_block_hashes.pkl\")\n# cdb.slice_as_df(10, 100)\n# cdb[0]\n# cdb[0:10]","metadata":{"execution":{"iopub.status.busy":"2024-04-14T21:43:19.302218Z","iopub.execute_input":"2024-04-14T21:43:19.303077Z","iopub.status.idle":"2024-04-14T21:43:19.921394Z","shell.execute_reply.started":"2024-04-14T21:43:19.303040Z","shell.execute_reply":"2024-04-14T21:43:19.920302Z"},"trusted":true},"execution_count":null,"outputs":[]}]}