{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"},{"sourceId":8125974,"sourceType":"datasetVersion","datasetId":4802323}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook will load and use my version of the compressed BELKA dataset\n\nYou can check the compression notebook here https://www.kaggle.com/code/elongm/belka-create-compressed-db\nand find the dataset here https://www.kaggle.com/datasets/elongm/belka-compressed-dataset","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport os\nimport pandas as pd\nimport pickle\nimport sys\n\n\ncompressed_db_columns = [\"buildingblock1_smiles\",\n        \"buildingblock2_smiles\", \n        \"buildingblock3_smiles\", \n        \"molecule_smiles\", \n        \"binds_sEH\", \n        \"binds_HSA\", \n        \"binds_BRD4\"]\n    \ncompressed_db_column_to_index = {k:v for v,k in enumerate(compressed_db_columns)}\n\n\nclass bi_directional_dict(object):\n    def __init__(self):\n        self._next_index = 0\n        self.key_to_index = {}\n        self.index_to_key = []\n    \n    def get_or_create_index(self, element):\n        if not element in self.key_to_index:\n            indx = self._next_index  \n            self.key_to_index[element] = indx\n            self.index_to_key.append(element)\n            self._next_index += 1\n            return indx\n        return self.key_to_index[element]\n    \n    def get_key_of_index(self, index):\n        return self.index_to_key[index]\n    \n    \nclass CompressedDB(object):\n    def __init__(self, compressed_db, building_blocks_hashes):\n        \"\"\"\n        compressed_db : can be the actual object or path to csv file\n        building_blocks_hashes : can be the actual object or path to pickle file\n        \"\"\"\n        self.compressed_db = compressed_db\n        self.building_blocks_hash = building_blocks_hashes\n        \n        if type(self.compressed_db) == str:\n            self.compressed_db = pd.read_csv(self.compressed_db, dtype={'buildingblock1_smiles': np.int16,\n                   'buildingblock2_smiles': np.int16,\n                   'buildingblock3_smiles': np.int16,\n                   'molecule_smiles': str,\n                   'binds_sEH': np.byte,\n                   'binds_HSA':  np.byte,\n                   'binds_BRD4': np.byte})\n        \n        if type(self.building_blocks_hash) == str:\n            with open(self.building_blocks_hash, \"rb\") as f:\n                self.building_blocks_hash = pickle.loads(f.read())\n        \n    def __getitem__(self, index):\n        if isinstance(index, slice):\n            sliced_db = self.compressed_db.__getitem__(index).to_numpy().copy()\n            bb1_indices = sliced_db[:,0]\n            bb2_indices = sliced_db[:,1]\n            bb3_indices = sliced_db[:,2]\n\n            exapnded_bb1 = np.array([self.building_blocks_hash.get_key_of_index(i) for i in bb1_indices])\n            exapnded_bb2 = np.array([self.building_blocks_hash.get_key_of_index(i) for i in bb2_indices])\n            exapnded_bb3 = np.array([self.building_blocks_hash.get_key_of_index(i) for i in bb3_indices])\n\n            sliced_db[:,0] = exapnded_bb1\n            sliced_db[:,1] = exapnded_bb2\n            sliced_db[:,2] = exapnded_bb3\n            return sliced_db\n        \n        compressed_db_row = self.compressed_db.iloc[index].to_numpy().copy()\n        compressed_db_row[0] = self.building_blocks_hash.get_key_of_index(compressed_db_row[0])\n        compressed_db_row[1] = self.building_blocks_hash.get_key_of_index(compressed_db_row[1])\n        compressed_db_row[2] = self.building_blocks_hash.get_key_of_index(compressed_db_row[2])\n        return compressed_db_row\n    \n    \n\n    def slice_as_df(self, start, stop):\n        slice_db = self[start: stop]\n        df = pd.DataFrame(slice_db, columns=compressed_db_columns)\n        return df.astype({'binds_sEH': np.byte, \n                        'binds_HSA': np.byte, \n                        'binds_BRD4': np.byte,\n                        'buildingblock1_smiles': 'str',\n                        'buildingblock2_smiles': 'str',\n                        'buildingblock3_smiles': 'str', \n                        'molecule_smiles':\"str\"})\n        \n        ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the compressed dataset\ncdb = CompressedDB(r\"/kaggle/input/belka-compressed-dataset/df.csv\", r\"/kaggle/input/belka-compressed-dataset/building_block_hashes.pkl\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# decompress and return the first row, returned as a np array\ncdb[0]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# decompress and return the slice, returned as a np array\ncdb[10:15]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# decompress and return slice as Dataframe\ncdb.slice_as_df(0, 10)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# accessing buildingblock1_smiles column of the first row\nbuildingblock1_smiles_col_index = compressed_db_column_to_index[\"buildingblock1_smiles\"]\ncdb[0][buildingblock1_smiles_col_index]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# accessing binds_sEH column of row at index ten\ncdb[10][compressed_db_column_to_index[\"binds_sEH\"]]","metadata":{},"execution_count":null,"outputs":[]}]}