{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30746,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"The purpose of this notebook is to compare the shape and electrostatic potential similarities of the new building blocks in the test set and the building blocks in the train set using the espsim package: https://github.com/hesther/espsim.git\n\nThe results could be used to augment the training dataset to increase the generalization of the model for the new building blocks with the triazine core.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-13T09:03:06.425634Z","iopub.execute_input":"2024-07-13T09:03:06.426922Z","iopub.status.idle":"2024-07-13T09:03:07.577167Z","shell.execute_reply.started":"2024-07-13T09:03:06.426815Z","shell.execute_reply":"2024-07-13T09:03:07.575751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install duckdb\n!pip install mapply\n!pip install rdkit\n!pip install tqdm\n!pip install py3Dmol\n!pip install espsim","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:03:07.579669Z","iopub.execute_input":"2024-07-13T09:03:07.580301Z","iopub.status.idle":"2024-07-13T09:04:57.869073Z","shell.execute_reply.started":"2024-07-13T09:03:07.580253Z","shell.execute_reply":"2024-07-13T09:04:57.867733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import necessary libraries\nimport numpy as np\nimport pandas as pd\nimport duckdb\nimport os\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom rdkit.Chem import AllChem, MolFromSmiles, Draw, rdFMCS, rdDistGeom, rdMolAlign\nimport mapply\nfrom matplotlib_venn import venn2\nimport tqdm\nfrom rdkit import Chem, RDLogger\nfrom rdkit.Chem.Draw import IPythonConsole\nimport py3Dmol\nimport espsim\nfrom espsim import EmbedAlignConstrainedScore, EmbedAlignScore\nfrom IPython.display import display\nimport gc\n\nfrom rdkit.Chem import rdMolAlign\nfrom rdkit.Chem import rdMolDescriptors","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:04:57.870900Z","iopub.execute_input":"2024-07-13T09:04:57.871308Z","iopub.status.idle":"2024-07-13T09:04:59.409545Z","shell.execute_reply.started":"2024-07-13T09:04:57.871270Z","shell.execute_reply":"2024-07-13T09:04:59.408352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#initialize mapply for parallel processing\nmapply.init(\n    n_workers=-1,\n    progressbar=False,\n)\n\nRDLogger.DisableLog(\"rdApp.*\")  # Disable all RDKit logging, including warnings","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:04:59.412546Z","iopub.execute_input":"2024-07-13T09:04:59.412966Z","iopub.status.idle":"2024-07-13T09:04:59.418810Z","shell.execute_reply.started":"2024-07-13T09:04:59.412932Z","shell.execute_reply":"2024-07-13T09:04:59.417665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check for triazine\ndef check_for_triazine(x: str):\n    triazine = MolFromSmiles('C1=NC=NC=N1')\n    check = MolFromSmiles(x).HasSubstructMatch(triazine)\n    return check\n\ndef check_building_blocks(row):\n    # Check each building block against the list\n    exists = any(block in new_bb_list for block in [row['buildingblock1_smiles'], row['buildingblock2_smiles'], row['buildingblock3_smiles']])\n    # Return True if any building block exists, else False\n    return exists\n","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:04:59.420174Z","iopub.execute_input":"2024-07-13T09:04:59.420538Z","iopub.status.idle":"2024-07-13T09:04:59.439815Z","shell.execute_reply.started":"2024-07-13T09:04:59.420508Z","shell.execute_reply":"2024-07-13T09:04:59.438674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Create a list of new and shared building blocks for each position (bb1, bb2, bb3)","metadata":{}},{"cell_type":"code","source":"test_path = \"/kaggle/input/leash-BELKA/test.parquet\"\ntrain_path = \"/kaggle/input/leash-BELKA/train.parquet\"\n\ncon = duckdb.connect()\nall_building_blocks_df = con.query(f\"\"\"(SELECT * FROM (\n    SELECT distinct buildingblock1_smiles as smile, 'buildingblock1' as group, 'test' as split, protein_name, 2 as binds  FROM parquet_scan('{test_path}')\n    UNION\n    SELECT distinct buildingblock1_smiles as smile, 'buildingblock1' as group, 'train' as split, protein_name, binds FROM parquet_scan('{train_path}')\n    UNION\n    SELECT distinct buildingblock2_smiles as smile, 'buildingblock2' as group, 'test' as split, protein_name, 2 as binds  FROM parquet_scan('{test_path}')\n    UNION\n    SELECT distinct buildingblock2_smiles as smile, 'buildingblock2' as group, 'train' as split, protein_name, binds FROM parquet_scan('{train_path}')\n    UNION\n    SELECT distinct buildingblock3_smiles as smile, 'buildingblock3' as group, 'test' as split, protein_name, 2 as binds  FROM parquet_scan('{test_path}')\n    UNION\n    SELECT distinct buildingblock3_smiles as smile, 'buildingblock3' as group, 'train' as split, protein_name, binds FROM parquet_scan('{train_path}')\n    ) as t)\"\"\").df()\ncon.close()","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:04:59.441406Z","iopub.execute_input":"2024-07-13T09:04:59.441978Z","iopub.status.idle":"2024-07-13T09:05:47.708990Z","shell.execute_reply.started":"2024-07-13T09:04:59.441934Z","shell.execute_reply":"2024-07-13T09:05:47.707788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#extract the bbs from train and test set\ntrain_smiles_bb_HSA = set(all_building_blocks_df[ (all_building_blocks_df['split'] == 'train')  & (all_building_blocks_df['protein_name'] == 'HSA')]['smile'])\ntest_smiles_bb_HSA = set(all_building_blocks_df[  (all_building_blocks_df['split'] == 'test')  & (all_building_blocks_df['protein_name'] == 'HSA')]['smile'])","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:05:47.710735Z","iopub.execute_input":"2024-07-13T09:05:47.711115Z","iopub.status.idle":"2024-07-13T09:05:47.749216Z","shell.execute_reply.started":"2024-07-13T09:05:47.711082Z","shell.execute_reply":"2024-07-13T09:05:47.748005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax  = plt.subplots(1,1, figsize=(15,5))\nax.set_title(\"HSA\")\nvenn2([train_smiles_bb_HSA, test_smiles_bb_HSA], set_labels = ('Train', 'Test'), ax=ax)\nfig.suptitle('New building blocks in the test set')","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:05:47.750750Z","iopub.execute_input":"2024-07-13T09:05:47.751170Z","iopub.status.idle":"2024-07-13T09:05:47.956096Z","shell.execute_reply.started":"2024-07-13T09:05:47.751132Z","shell.execute_reply":"2024-07-13T09:05:47.954653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_bbs_in_test_list_HSA = sorted(list(test_smiles_bb_HSA - train_smiles_bb_HSA ))","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:05:47.958397Z","iopub.execute_input":"2024-07-13T09:05:47.959343Z","iopub.status.idle":"2024-07-13T09:05:47.969435Z","shell.execute_reply.started":"2024-07-13T09:05:47.959275Z","shell.execute_reply":"2024-07-13T09:05:47.967973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_path = \"/kaggle/input/leash-BELKA/test.parquet\"\nprotein_names = ['HSA']\nnew_bb_lists = [new_bbs_in_test_list_HSA]\nfor protein_name,new_bb_list in zip(protein_names,new_bb_lists):\n    con = duckdb.connect()\n\n    df_test = con.query(f\"\"\"(SELECT *\n                        FROM parquet_scan('{test_path}')\n                        WHERE protein_name = '{protein_name}'\n                        )\"\"\").df()\n\n    con.close()\n    # Function to check if any building block exists in new_bbs_in_test_list_sEH\n\n\n    # Apply the function to each row in df_test\n    df_test['new_bbs'] = df_test.mapply(check_building_blocks, axis=1)\n    df_test['triazine'] = df_test['molecule_smiles'].mapply(check_for_triazine)\n    # Display the updated DataFrame\n   \n    new_bbs = df_test[['id','new_bbs','triazine']]\n","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:05:47.977344Z","iopub.execute_input":"2024-07-13T09:05:47.978607Z","iopub.status.idle":"2024-07-13T09:08:10.560361Z","shell.execute_reply.started":"2024-07-13T09:05:47.978517Z","shell.execute_reply":"2024-07-13T09:08:10.558576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#save data to working directory\nnew_bbs.to_csv(f\"/kaggle/working/{protein_name}_new_bbs.csv\", index=False)  \nprint(new_bbs[\"new_bbs\"].value_counts())\nprint(new_bbs[\"triazine\"].value_counts())","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:10.562463Z","iopub.execute_input":"2024-07-13T09:08:10.562980Z","iopub.status.idle":"2024-07-13T09:08:11.744265Z","shell.execute_reply.started":"2024-07-13T09:08:10.562932Z","shell.execute_reply":"2024-07-13T09:08:11.743136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del all_building_blocks_df, df_test\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:11.745699Z","iopub.execute_input":"2024-07-13T09:08:11.746039Z","iopub.status.idle":"2024-07-13T09:08:12.070085Z","shell.execute_reply.started":"2024-07-13T09:08:11.746010Z","shell.execute_reply":"2024-07-13T09:08:12.068852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n\ncon = duckdb.connect()\n\n# as the same molecules are used to test three targets, query protein_name = \"HSA\" to expedite the queery process\nquery_bb1 = f\"SELECT distinct buildingblock1_smiles FROM parquet_scan('{train_path}') WHERE protein_name = 'HSA'\"\nquery_bb2 = f\"SELECT distinct buildingblock2_smiles FROM parquet_scan('{train_path}') WHERE protein_name = 'HSA'\"\nquery_bb3 = f\"SELECT distinct buildingblock3_smiles FROM parquet_scan('{train_path}') WHERE protein_name = 'HSA'\"\n\nbb1_train = con.execute(query_bb1).fetchdf()[\"buildingblock1_smiles\"].to_list()\nbb2_train = con.execute(query_bb2).fetchdf()[\"buildingblock2_smiles\"].to_list()\nbb3_train = con.execute(query_bb3).fetchdf()[\"buildingblock3_smiles\"].to_list()\n\nbbs_test = con.query(\n    f\"\"\"(SELECT id, buildingblock1_smiles, buildingblock2_smiles, buildingblock3_smiles\n                    FROM parquet_scan('{test_path}')\n                    WHERE protein_name = 'HSA'\n                    )\"\"\"\n).df()\n\ncon.close()\n","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:12.071644Z","iopub.execute_input":"2024-07-13T09:08:12.072095Z","iopub.status.idle":"2024-07-13T09:08:29.464060Z","shell.execute_reply.started":"2024-07-13T09:08:12.072052Z","shell.execute_reply":"2024-07-13T09:08:29.462535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del query_bb1, query_bb2, query_bb3","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:29.465445Z","iopub.execute_input":"2024-07-13T09:08:29.465823Z","iopub.status.idle":"2024-07-13T09:08:29.471208Z","shell.execute_reply.started":"2024-07-13T09:08:29.465792Z","shell.execute_reply":"2024-07-13T09:08:29.469962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#merge \nbbs_test = bbs_test.merge(new_bbs, on=\"id\")\nbbs_test.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:29.472900Z","iopub.execute_input":"2024-07-13T09:08:29.473352Z","iopub.status.idle":"2024-07-13T09:08:29.550059Z","shell.execute_reply.started":"2024-07-13T09:08:29.473310Z","shell.execute_reply":"2024-07-13T09:08:29.548924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#subset the new building blocks in the test with the triazine core\nnew_bb1_test_triazine = set(bbs_test[(bbs_test[\"new_bbs\"] == 1) & (bbs_test[\"triazine\"] == 1)][\"buildingblock1_smiles\"].tolist())\nnew_bb2_test_triazine = set(bbs_test[(bbs_test[\"new_bbs\"] == 1) & (bbs_test[\"triazine\"] == 1)][\"buildingblock2_smiles\"].tolist())\nnew_bb3_test_triazine = set(bbs_test[(bbs_test[\"new_bbs\"] == 1) & (bbs_test[\"triazine\"] == 1)][\"buildingblock3_smiles\"].tolist())\n\n\n#subset the new building blocks in the test without the triazine core\nnew_bb1_test_no_triazine = set(bbs_test[(bbs_test[\"new_bbs\"] == 1) & (bbs_test[\"triazine\"] == 0)][\"buildingblock1_smiles\"].tolist())\nnew_bb2_test_no_triazine = set(bbs_test[(bbs_test[\"new_bbs\"] == 1) & (bbs_test[\"triazine\"] == 0)][\"buildingblock2_smiles\"].tolist())\nnew_bb3_test_no_triazine = set(bbs_test[(bbs_test[\"new_bbs\"] == 1) & (bbs_test[\"triazine\"] == 0)][\"buildingblock3_smiles\"].tolist())\n\n\n\n#subset the shared building blocks in the test with the triazine core\nshared_bb1_test_triazine = set(bbs_test[(bbs_test[\"new_bbs\"] == 0) & (bbs_test[\"triazine\"] == 1)][\"buildingblock1_smiles\"].tolist())\nshared_bb2_test_triazine = set(bbs_test[(bbs_test[\"new_bbs\"] == 0) & (bbs_test[\"triazine\"] == 1)][\"buildingblock2_smiles\"].tolist())\nshared_bb3_test_triazine = set(bbs_test[(bbs_test[\"new_bbs\"] == 0) & (bbs_test[\"triazine\"] == 1)][\"buildingblock3_smiles\"].tolist())","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:29.552036Z","iopub.execute_input":"2024-07-13T09:08:29.552515Z","iopub.status.idle":"2024-07-13T09:08:29.893954Z","shell.execute_reply.started":"2024-07-13T09:08:29.552442Z","shell.execute_reply":"2024-07-13T09:08:29.892793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot number of building blocks with triazine core and without triazine core in the test set\nfig, ax = plt.subplots(1, 3, figsize=(8, 3))\n\n# Data for plotting\ncategories = [\"bb1\", \"bb2\", \"bb3\"]\nvalues1 = [len(shared_bb1_test_triazine), len(shared_bb2_test_triazine), len(shared_bb3_test_triazine)]\nvalues2 = [len(new_bb1_test_triazine), len(new_bb2_test_triazine), len(new_bb3_test_triazine)]\nvalues3 = [len(new_bb1_test_no_triazine), len(new_bb2_test_no_triazine), len(new_bb3_test_no_triazine)]\n\n# Shared building blocks with triazine core\nbars1 = ax[0].bar(categories, values1)\nax[0].set_title(\"Shared building blocks \\nwith triazine core\")\nax[0].set_ylabel(\"Number of building blocks\")\n# Display values on top of bars\nfor bar in bars1:\n    height = bar.get_height()\n    ax[0].text(bar.get_x() + bar.get_width() / 2., 1.05*height, '%d' % int(height), ha='center', va='bottom')\nax[0].set_ylim(0, max(values1)*1.2)\n\n# New building blocks with triazine core\nbars2 = ax[1].bar(categories, values2)\nax[1].set_title(\"New building blocks \\nwith triazine core\")\n# Display values on top of bars\nfor bar in bars2:\n    height = bar.get_height()\n    ax[1].text(bar.get_x() + bar.get_width() / 2., 1.05*height, '%d' % int(height), ha='center', va='bottom')\nax[1].set_ylim(0, max(values2)*1.2)\n\n# New building blocks without triazine core\nbars3 = ax[2].bar(categories, values3)\nax[2].set_title(\"New building blocks \\nwithout triazine core\")\n# Display values on top of bars\nfor bar in bars3:\n    height = bar.get_height()\n    ax[2].text(bar.get_x() + bar.get_width() / 2., 1.05*height, '%d' % int(height), ha='center', va='bottom')\nax[2].set_ylim(0, max(values3)*1.2)\nplt.tight_layout()\n","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:29.895335Z","iopub.execute_input":"2024-07-13T09:08:29.895714Z","iopub.status.idle":"2024-07-13T09:08:30.647228Z","shell.execute_reply.started":"2024-07-13T09:08:29.895684Z","shell.execute_reply":"2024-07-13T09:08:30.646031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Building blocks in tranining dataset\")\nprint(\"bb1\")\ndisplay(Draw.MolsToGridImage([Chem.MolFromSmiles(x) for x in bb1_train], molsPerRow=5, maxMols=5))\nprint(\"bb2\")\ndisplay(Draw.MolsToGridImage([Chem.MolFromSmiles(x) for x in bb2_train], molsPerRow=5, maxMols=5))\nprint(\"bb3\")\ndisplay(Draw.MolsToGridImage([Chem.MolFromSmiles(x) for x in bb3_train], molsPerRow=5, maxMols=5))","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:30.648855Z","iopub.execute_input":"2024-07-13T09:08:30.649213Z","iopub.status.idle":"2024-07-13T09:08:30.975227Z","shell.execute_reply.started":"2024-07-13T09:08:30.649182Z","shell.execute_reply":"2024-07-13T09:08:30.974022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"New building blocks with triazine core\")\nprint(\"bb1\")\ndisplay(Draw.MolsToGridImage([Chem.MolFromSmiles(x) for x in new_bb1_test_triazine], molsPerRow=5, maxMols=5))\nprint(\"bb2\")\ndisplay(Draw.MolsToGridImage([Chem.MolFromSmiles(x) for x in new_bb2_test_triazine], molsPerRow=5, maxMols=5))\nprint(\"bb3\")\ndisplay(Draw.MolsToGridImage([Chem.MolFromSmiles(x) for x in new_bb3_test_triazine], molsPerRow=5, maxMols=5))","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:30.976975Z","iopub.execute_input":"2024-07-13T09:08:30.977434Z","iopub.status.idle":"2024-07-13T09:08:31.084316Z","shell.execute_reply.started":"2024-07-13T09:08:30.977395Z","shell.execute_reply":"2024-07-13T09:08:31.083134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To compare the similarity between the building blocks, it is necessary to remove the protecting groups in the building blocks. Moreover, to ensure the building blocks are compared in the correct orientation, they are attached to the triazine core.","metadata":{}},{"cell_type":"code","source":"#define protection groups, drawing using openbabel website: https://www.cheminfo.org/Chemistry/Cheminformatics/FormatConverter/index.html\nprotection_group_triazine = (\"n1cncnc1N[*:1]\",\n                'c1ccc2-c3c(C(c2c1)COC(=O)N[*:1])cccc3',\n                 'N[*:1]',\n                )\n#replace protection group with the triazine core\nprotection_group_triazine = [(Chem.MolFromSmiles(x)) for x in protection_group_triazine]\nDraw.MolsToGridImage(protection_group_triazine)","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:31.085619Z","iopub.execute_input":"2024-07-13T09:08:31.085974Z","iopub.status.idle":"2024-07-13T09:08:31.107025Z","shell.execute_reply.started":"2024-07-13T09:08:31.085944Z","shell.execute_reply":"2024-07-13T09:08:31.105798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#reaction to replace the protection group\ndef buildIsostereReaction(start,replacement):\n    start = Chem.AddHs(start)\n    qps = Chem.AdjustQueryParameters()\n    qps.adjustDegree = False\n    qps.adjustHeavyDegree = False\n    qps.adjustRingCount = False\n    qps.aromatizeIfPossible = False\n    qps.makeAtomsGeneric = False\n    qps.makeBondsGeneric = False\n    qps.makeDummiesQueries = True\n    start = Chem.AdjustQueryProperties(start,qps)\n    replacement = Chem.AdjustQueryProperties(replacement,qps)\n    product = AllChem.ChemicalReaction()\n    product.AddReactantTemplate(start)\n    product.AddProductTemplate(replacement)\n    product.Initialize()\n    return product\n\n\ndef doReplacement(mol,rxn):\n    mol = Chem.AddHs(mol)\n    ps = rxn.RunReactants((mol,))\n    res = [x[0] for x in ps]\n    return res\ndef doReplacements(mol,rxns):\n    res = []\n    mol = Chem.AddHs(mol)\n    for i,rxn in enumerate(rxns):\n        seenSoFar=set()\n        ims = doReplacement(mol,rxn)\n        p0 = rxn.GetProductTemplate(0)\n        # save where the match is\n        for im in ims:\n            smi = Chem.MolToSmiles(im,True)\n            if smi not in seenSoFar:\n                im.coreMatches = im.GetSubstructMatch(p0)\n                res.append(im)\n                seenSoFar.add(smi)\n    return res","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:31.109195Z","iopub.execute_input":"2024-07-13T09:08:31.109965Z","iopub.status.idle":"2024-07-13T09:08:31.123888Z","shell.execute_reply.started":"2024-07-13T09:08:31.109924Z","shell.execute_reply":"2024-07-13T09:08:31.122666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#remove protection group for the new building blocks with triazine core\nnew_bbs = [list(new_bb1_test_triazine), \n           list(new_bb2_test_triazine), \n           list(new_bb3_test_triazine)]\n\n#initialize the list of the new building blocks without the protection group\nnew_bb1_test_triazine_protection_group = []\nnew_bb2_test_triazine_protection_group = []\nnew_bb3_test_triazine_protection_group = []\n\nnew_bbs_remove_protection_group = [new_bb1_test_triazine_protection_group,\n                                   new_bb2_test_triazine_protection_group,\n                                   new_bb3_test_triazine_protection_group]\n\nfor new_bb_remove_protection_group, new_bb in zip(new_bbs_remove_protection_group, new_bbs):\n    for a in range(1, len(protection_group_triazine)):\n        for i in tqdm.tqdm(range(len(new_bb)), desc='Molecules'):\n            isostereReactions = [buildIsostereReaction(protection_group_triazine[a], protection_group_triazine[0])]\n            mol = Chem.MolFromSmiles(new_bb[i])\n            if mol is None:\n                continue  # Skip if the molecule could not be created\n            subs = doReplacements(mol, isostereReactions)\n            if len(subs) > 0:\n                sub = subs[0] #extract only the first reaction\n                # Attempt to sanitize the molecule\n                try:\n                    Chem.SanitizeMol(sub)\n                    new_bb_remove_protection_group.append(Chem.MolToSmiles(Chem.RemoveHs(sub)))\n                except Exception as e:\n                    print(f\"Error processing molecule\")\n                    end  # Skip this molecule if an error occurs\n\n            \n\n              ","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:31.126261Z","iopub.execute_input":"2024-07-13T09:08:31.126773Z","iopub.status.idle":"2024-07-13T09:08:31.438026Z","shell.execute_reply.started":"2024-07-13T09:08:31.126734Z","shell.execute_reply":"2024-07-13T09:08:31.436817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"New building blocks with triazine core\")\nprint(\"bb1\")\ndisplay(Draw.MolsToGridImage([Chem.MolFromSmiles(x) for x in new_bb1_test_triazine_protection_group], molsPerRow=5, maxMols=5))\nprint(\"bb2\")\ndisplay(Draw.MolsToGridImage([Chem.MolFromSmiles(x) for x in new_bb2_test_triazine_protection_group ], molsPerRow=5, maxMols=5))\nprint(\"bb3\")\ndisplay(Draw.MolsToGridImage([Chem.MolFromSmiles(x) for x in new_bb3_test_triazine_protection_group ], molsPerRow=5, maxMols=5))","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:31.439558Z","iopub.execute_input":"2024-07-13T09:08:31.440005Z","iopub.status.idle":"2024-07-13T09:08:31.552855Z","shell.execute_reply.started":"2024-07-13T09:08:31.439967Z","shell.execute_reply":"2024-07-13T09:08:31.551474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#remove protection group for the shared building blocks with triazine core\ntrain_bbs = [bb1_train, \n           bb2_train, \n           bb3_train]\n\n#initialize the list of the new building blocks without the protection group\nbb1_train_protection_group = []\nbb2_train_protection_group = []\nbb3_train_protection_group = []\nbbs_train_remove_protection_group = [bb1_train_protection_group,\n                                   bb2_train_protection_group,\n                                   bb3_train_protection_group]\nfor bb_train_remove_protection_group, train_bb in zip(bbs_train_remove_protection_group, train_bbs):\n    for a in range(1, len(protection_group_triazine)):\n        for i in tqdm.tqdm(range(len(train_bb)), desc='Molecules'):\n            isostereReactions = [buildIsostereReaction(protection_group_triazine[a], protection_group_triazine[0])]\n            mol = Chem.MolFromSmiles(train_bb[i])\n            if mol is None:\n                continue  # Skip if the molecule could not be created\n            subs = doReplacements(mol, isostereReactions)\n            if len(subs) > 0:\n                sub = subs[0] #extract only the first reaction\n                # Attempt to sanitize the molecule\n                try:\n                    Chem.SanitizeMol(sub)\n                    bb_train_remove_protection_group.append(Chem.MolToSmiles(Chem.RemoveHs(sub)))\n                except Exception as e:\n                    print(\"Error processing molecule\")\n                    end  # Skip this molecule if an error occurs\n            \n\n              ","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:31.554408Z","iopub.execute_input":"2024-07-13T09:08:31.554791Z","iopub.status.idle":"2024-07-13T09:08:34.312962Z","shell.execute_reply.started":"2024-07-13T09:08:31.554761Z","shell.execute_reply":"2024-07-13T09:08:34.311859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Building blocks in training dataset\")\nprint(\"bb1\")\ndisplay(Draw.MolsToGridImage([Chem.MolFromSmiles(x) for x in bb1_train_protection_group], molsPerRow=5, maxMols=5))\nprint(\"bb2\")\ndisplay(Draw.MolsToGridImage([Chem.MolFromSmiles(x) for x in bb2_train_protection_group ], molsPerRow=5, maxMols=5))\nprint(\"bb3\")\ndisplay(Draw.MolsToGridImage([Chem.MolFromSmiles(x) for x in bb3_train_protection_group ], molsPerRow=5, maxMols=5))","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:34.314661Z","iopub.execute_input":"2024-07-13T09:08:34.315320Z","iopub.status.idle":"2024-07-13T09:08:34.706237Z","shell.execute_reply.started":"2024-07-13T09:08:34.315278Z","shell.execute_reply":"2024-07-13T09:08:34.705087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#remove salt\nfor new_bb_remove_protection_group in new_bbs_remove_protection_group:\n    for i in range(len(new_bb_remove_protection_group)):\n        new_bb_remove_protection_group[i] = max(new_bb_remove_protection_group[i].split('.'), key=len)\nfor bb_train_remove_protection_group in bbs_train_remove_protection_group:\n    for i in range(len(bb_train_remove_protection_group)):\n        bb_train_remove_protection_group[i] = max(bb_train_remove_protection_group[i].split('.'), key=len)","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:34.707731Z","iopub.execute_input":"2024-07-13T09:08:34.708145Z","iopub.status.idle":"2024-07-13T09:08:34.718245Z","shell.execute_reply.started":"2024-07-13T09:08:34.708108Z","shell.execute_reply":"2024-07-13T09:08:34.716549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fragment_similarity(smiles: str):\n    try:\n        mol_1 = smiles.split(',')[0]\n        mol_1 = Chem.MolFromSmiles(mol_1)\n        mol_1 = Chem.AddHs(mol_1)\n        AllChem.EmbedMolecule(mol_1)\n\n        mol_2 = smiles.split(',')[1]\n        mol_2 = Chem.MolFromSmiles(mol_2)\n        mol_2 = Chem.AddHs(mol_2)\n        AllChem.EmbedMolecule(mol_2)\n        mols = [mol_1, mol_2]\n        # try:\n        #assign the triazine as a core\n        mcsmol = Chem.MolFromSmiles(\"n1cncnc1N\")\n        AllChem.AddHs(mcsmol)\n        AllChem.EmbedMolecule(mcsmol, AllChem.ETKDGv2())\n        core = AllChem.DeleteSubstructs(AllChem.ReplaceSidechains(mol_1,mcsmol),Chem.MolFromSmiles('*'))\n        shapesim, espsim = EmbedAlignConstrainedScore(mol_1, [mol_2],core\n                        , prbNumConfs=mol_1.GetNumConformers()\n                        , refNumConfs=mol_2.GetNumConformers())\n        return shapesim[0], espsim[0]\n    except Exception as e:  \n        print(f\"An error occurred: {e}\")  # Optionally log the error\n        return 0, 0","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:34.719552Z","iopub.execute_input":"2024-07-13T09:08:34.720859Z","iopub.status.idle":"2024-07-13T09:08:34.732987Z","shell.execute_reply.started":"2024-07-13T09:08:34.720818Z","shell.execute_reply":"2024-07-13T09:08:34.731612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_rows = len(bbs_train_remove_protection_group[0])\nnum_columns = len(new_bbs_remove_protection_group[0])\ndf_smiles_bb1 = pd.DataFrame(np.full((num_rows, num_columns), \"\"))\nfor i in tqdm.tqdm(range(num_rows), total=num_rows):\n    for j in range(num_columns):\n        df_smiles_bb1.iloc[i, j] = (\n            str(bbs_train_remove_protection_group[0][i])\n            + \",\"\n            + str(new_bbs_remove_protection_group[0][j])\n        )\ndf_2D_bb1 = pd.DataFrame(np.full((num_rows, num_columns), np.nan))\ndf_3D_bb1 = pd.DataFrame(np.full((num_rows, num_columns), np.nan))\nfor i in tqdm.tqdm(range(num_columns),total=num_columns):\n    a = df_smiles_bb1[i].mapply(fragment_similarity)\n    a_df = pd.DataFrame(list(a))\n    df_2D_bb1.iloc[:,i] = a_df.iloc[:,0]\n    df_3D_bb1.iloc[:,i] = a_df.iloc[:,1]\n\n#saving data\ndf_2D_bb1.to_csv(\"/kaggle/working/df_2D_bb1.csv\",index=False)\ndf_3D_bb1.to_csv(\"/kaggle/working/df_3D_bb1.csv\",index=False)\ndf_smiles_bb1.to_csv(\"/kaggle/working/df_smiles_bb1.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:08:34.740623Z","iopub.execute_input":"2024-07-13T09:08:34.741419Z","iopub.status.idle":"2024-07-13T09:18:00.601114Z","shell.execute_reply.started":"2024-07-13T09:08:34.741370Z","shell.execute_reply":"2024-07-13T09:18:00.599692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Compare similarity of the new building blocks 2 with the shared building blocks 2\n\nnum_rows = len(bbs_train_remove_protection_group[1])\nnum_columns = len(new_bbs_remove_protection_group[1])\ndf_smiles_bb2 = pd.DataFrame(np.full((num_rows, num_columns), \"\"))\nfor i in tqdm.tqdm(range(num_rows), total=num_rows):\n    for j in range(num_columns):\n        df_smiles_bb2.iloc[i, j] = (\n            str(bbs_train_remove_protection_group[1][i])\n            + \",\"\n            + str(new_bbs_remove_protection_group[1][j])\n        )\ndf_2D_bb2 = pd.DataFrame(np.full((num_rows, num_columns), np.nan))\ndf_3D_bb2 = pd.DataFrame(np.full((num_rows, num_columns), np.nan))\nfor i in tqdm.tqdm(range(num_columns),total=num_columns):\n    a = df_smiles_bb2[i].mapply(fragment_similarity)\n    a_df = pd.DataFrame(list(a))\n    df_2D_bb2.iloc[:,i] = a_df.iloc[:,0]\n    df_3D_bb2.iloc[:,i] = a_df.iloc[:,1]\n\n#saving data\ndf_2D_bb2.to_csv(\"/kaggle/working/df_2D_bb2.csv\",index=False)\ndf_3D_bb2.to_csv(\"/kaggle/working/df_3D_bb2.csv\",index=False)\ndf_smiles_bb2.to_csv(\"/kaggle/working/df_smiles_bb2.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2024-07-13T09:18:00.603093Z","iopub.execute_input":"2024-07-13T09:18:00.603452Z","iopub.status.idle":"2024-07-13T10:02:15.429021Z","shell.execute_reply.started":"2024-07-13T09:18:00.603419Z","shell.execute_reply":"2024-07-13T10:02:15.427544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Compare similarity of the new building blocks 3 with the shared building blocks 3\n\nnum_rows = len(bbs_train_remove_protection_group[2])\nnum_columns = len(new_bbs_remove_protection_group[2])\n\ndf_smiles_bb3 = pd.DataFrame(np.full((num_rows, num_columns), \"\"))\nfor i in tqdm.tqdm(range(num_rows), total=num_rows):\n    for j in range(num_columns):\n        df_smiles_bb3.iloc[i, j] = (\n            str(bbs_train_remove_protection_group[2][i])\n            + \",\"\n            + str(new_bbs_remove_protection_group[2][j])\n        )\ndf_2D_bb3 = pd.DataFrame(np.full((num_rows, num_columns), np.nan))\ndf_3D_bb3 = pd.DataFrame(np.full((num_rows, num_columns), np.nan))\nfor i in tqdm.tqdm(range(num_columns),total=num_columns):\n    a = df_smiles_bb3[i].mapply(fragment_similarity)\n    a_df = pd.DataFrame(list(a))\n    df_2D_bb3.iloc[:,i] = a_df.iloc[:,0]\n    df_3D_bb3.iloc[:,i] = a_df.iloc[:,1]\n\n#saving data\ndf_2D_bb3.to_csv(\"/kaggle/working/df_2D_bb3.csv\",index=False)\ndf_3D_bb3.to_csv(\"/kaggle/working/df_3D_bb3.csv\",index=False)\ndf_smiles_bb3.to_csv(\"/kaggle/working/df_smiles_bb3.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2024-07-13T10:02:15.431255Z","iopub.execute_input":"2024-07-13T10:02:15.431667Z","iopub.status.idle":"2024-07-13T11:00:19.300885Z","shell.execute_reply.started":"2024-07-13T10:02:15.431629Z","shell.execute_reply":"2024-07-13T11:00:19.299426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def draw(ms,p=None, confIds=None):\n    if p is None:\n        p = py3Dmol.view(width=400, height=400)\n    if confIds is None:\n        confIds = [-1]*len(ms)\n    p.removeAllModels()\n    for i,m in enumerate(ms):\n        mb=Chem.MolToMolBlock(m, confId=confIds[i])\n        p.addModel(mb, 'sdf')\n        p.setStyle({'model':i},{'stick':{'radius':0.15}})\n    p.setBackgroundColor('white')#'0xeeeeee')\n    p.zoomTo()\n    return p.show()\n    \ndef align_two_mol(mol1, mol2, core):\n    mol1_conf = mol1.GetConformer()\n    mol2_conf = mol2.GetConformer()\n    match1 = mol1.GetSubstructMatch(core)\n    match2 = mol2.GetSubstructMatch(core) \n    AllChem.AlignMol(mol2, mol1, atomMap = list(zip(match2, match1)))\n\n    new_mol2_conf = mol2.GetConformer()\n    new_mol2_coord = new_mol2_conf.GetPositions()\n    mol1_coord = mol1_conf.GetPositions()\n\n    return mol1_coord, new_mol2_coord ","metadata":{"execution":{"iopub.status.busy":"2024-07-13T11:00:19.303101Z","iopub.execute_input":"2024-07-13T11:00:19.303607Z","iopub.status.idle":"2024-07-13T11:00:19.315889Z","shell.execute_reply.started":"2024-07-13T11:00:19.303544Z","shell.execute_reply":"2024-07-13T11:00:19.314667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a heatmap for df_2D\n\nvmin, vmax = 0, 1\nfig, ax = plt.subplots(1,2,figsize=(10, 5), sharey=True,sharex=True)\nsns.heatmap(df_2D_bb1, cmap='flare', ax = ax[0],cbar=True,vmin=vmin, vmax=vmax)  # 'coolwarm' is one example of a colormap. You can choose others as per your preference.\nsns.heatmap(df_3D_bb1, cmap='flare', ax = ax[1],vmin=vmin, vmax=vmax)\nfig.suptitle('Building block 1', fontsize=16)\nax[0].set_title(\"Shape similarity\")\nax[1].set_title(\"Electrostatic potential similarity\")\nax[0].set_xlabel(\"New building blocks\")\nax[0].set_ylabel(\"Shared building blocks\")\nax[1].set_ylabel(\"Shared building blocks\")\nax[1].set_xlabel(\"New building blocks\")\n# Display the plot\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-07-13T11:24:27.141160Z","iopub.execute_input":"2024-07-13T11:24:27.141959Z","iopub.status.idle":"2024-07-13T11:24:28.762115Z","shell.execute_reply.started":"2024-07-13T11:24:27.141915Z","shell.execute_reply":"2024-07-13T11:24:28.760469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a heatmap for df_2D\n\nvmin, vmax = 0, 1\nfig, ax = plt.subplots(1,2,figsize=(10, 5), sharey=True,sharex=True)\nsns.heatmap(df_2D_bb2, cmap='flare', ax = ax[0],cbar=True,vmin=vmin, vmax=vmax)  # 'coolwarm' is one example of a colormap. You can choose others as per your preference.\nsns.heatmap(df_3D_bb2, cmap='flare', ax = ax[1],vmin=vmin, vmax=vmax)\nfig.suptitle('Building block 2', fontsize=16)\nax[0].set_title(\"Shape similarity\")\nax[1].set_title(\"Electrostatic potential similarity\")\nax[0].set_xlabel(\"New building blocks\")\nax[1].set_xlabel(\"New building blocks\")\nax[0].set_ylabel(\"Shared building blocks\")\nax[1].set_ylabel(\"Shared building blocks\")\n# Display the plot\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-07-13T11:24:49.674732Z","iopub.execute_input":"2024-07-13T11:24:49.675136Z","iopub.status.idle":"2024-07-13T11:24:51.183992Z","shell.execute_reply.started":"2024-07-13T11:24:49.675102Z","shell.execute_reply":"2024-07-13T11:24:51.182761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a heatmap for df_2D\n\nvmin, vmax = 0, 1\nfig, ax = plt.subplots(1,2,figsize=(10, 5), sharey=True,sharex=True)\nsns.heatmap(df_2D_bb3, cmap='flare', ax = ax[0],cbar=True,vmin=vmin, vmax=vmax)  # 'coolwarm' is one example of a colormap. You can choose others as per your preference.\nsns.heatmap(df_3D_bb3, cmap='flare', ax = ax[1],vmin=vmin, vmax=vmax)\nfig.suptitle('Building block 3', fontsize=16)\n\nax[0].set_title(\"Shape similarity\")\nax[1].set_title(\"Electrostatic potential similarity\")\nax[0].set_xlabel(\"New building blocks\")\nax[1].set_xlabel(\"New building blocks\")\nax[0].set_ylabel(\"Shared building blocks\")\nax[1].set_ylabel(\"Shared building blocks\")\n# Display the plot\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-07-13T11:25:08.102497Z","iopub.execute_input":"2024-07-13T11:25:08.103355Z","iopub.status.idle":"2024-07-13T11:25:09.547856Z","shell.execute_reply.started":"2024-07-13T11:25:08.103316Z","shell.execute_reply":"2024-07-13T11:25:09.546613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#find best match for each new BBs\ndef find_best_match(df, df_smiles):\n    best_match = []\n    best_match_score = []\n    for column in df.columns:\n        max_values_per_column = df[column].max()\n        idmax = df[column].idxmax()\n        x = df_smiles.iloc[int(idmax),int(column)]\n        best_match.append(x)\n        best_match_score.append(max_values_per_column)\n    return best_match, best_match_score","metadata":{"execution":{"iopub.status.busy":"2024-07-13T11:05:39.571877Z","iopub.execute_input":"2024-07-13T11:05:39.572269Z","iopub.status.idle":"2024-07-13T11:05:39.579944Z","shell.execute_reply.started":"2024-07-13T11:05:39.572237Z","shell.execute_reply":"2024-07-13T11:05:39.578582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract the top 5 similarities between the new building blocks and the shared BB1\nbest_match, best_match_score  = find_best_match(df_3D_bb1, df_smiles_bb1)\nbb1_similarity = pd.DataFrame( {'smiles': best_match, 'best_match_score': best_match_score})\nbb1_similarity.sort_values(by=\"best_match_score\",ascending=False,inplace=True, ignore_index=True)\nfrom IPython.display import display\nfor i in range(0,5):\n    print(f\"Electrostatic potential similarity score: {bb1_similarity.loc[i,'best_match_score']}\")\n    display(Draw.MolsToGridImage([MolFromSmiles(x) for x in bb1_similarity.loc[i,\"smiles\"].split(',')], molsPerRow=4, maxMols=20))\n    \nbest_match, best_match_score  = find_best_match(df_2D_bb1, df_smiles_bb1)\nbb1_similarity = pd.DataFrame( {'smiles': best_match, 'best_match_score': best_match_score})\nbb1_similarity.sort_values(by=\"best_match_score\",ascending=False,inplace=True, ignore_index=True)\nfrom IPython.display import display\nfor i in range(0,5):\n    print(f\"Shape similarity score: {bb1_similarity.loc[i,'best_match_score']}\")\n    display(Draw.MolsToGridImage([MolFromSmiles(x) for x in bb1_similarity.loc[i,\"smiles\"].split(',')], molsPerRow=4, maxMols=20))\n    ","metadata":{"execution":{"iopub.status.busy":"2024-07-13T11:06:33.760983Z","iopub.execute_input":"2024-07-13T11:06:33.761434Z","iopub.status.idle":"2024-07-13T11:06:33.977330Z","shell.execute_reply.started":"2024-07-13T11:06:33.761396Z","shell.execute_reply":"2024-07-13T11:06:33.975512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract the top 5 similarities between the new building blocks and the shared BB2\n\nbest_match, best_match_score  = find_best_match(df_3D_bb2, df_smiles_bb2)\nbb2_similarity = pd.DataFrame( {'smiles': best_match, 'best_match_score': best_match_score})\nbb2_similarity.sort_values(by=\"best_match_score\",ascending=False,inplace=True, ignore_index=True)\nfrom IPython.display import display\nfor i in range(0,5):\n    print(f\"Electrostatic potential similarities score: {bb2_similarity.loc[i,'best_match_score']}\")\n    display(Draw.MolsToGridImage([MolFromSmiles(x) for x in bb2_similarity.loc[i,\"smiles\"].split(',')], molsPerRow=4, maxMols=20))\n    \nbest_match, best_match_score  = find_best_match(df_2D_bb2, df_smiles_bb2)\nbb2_similarity = pd.DataFrame( {'smiles': best_match, 'best_match_score': best_match_score})\nbb2_similarity.sort_values(by=\"best_match_score\",ascending=False,inplace=True, ignore_index=True)\nfrom IPython.display import display\nfor i in range(0,5):\n    print(f\"Shape similarity score: {bb2_similarity.loc[i,'best_match_score']}\")\n    display(Draw.MolsToGridImage([MolFromSmiles(x) for x in bb2_similarity.loc[i,\"smiles\"].split(',')], molsPerRow=4, maxMols=20))\n    ","metadata":{"execution":{"iopub.status.busy":"2024-07-13T11:08:58.194131Z","iopub.execute_input":"2024-07-13T11:08:58.194548Z","iopub.status.idle":"2024-07-13T11:08:58.401843Z","shell.execute_reply.started":"2024-07-13T11:08:58.194518Z","shell.execute_reply":"2024-07-13T11:08:58.400713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract the top 5 similarities between the new building blocks and the shared BB3\n\nbest_match, best_match_score  = find_best_match(df_3D_bb3, df_smiles_bb3)\nbb3_similarity = pd.DataFrame( {'smiles': best_match, 'best_match_score': best_match_score})\nbb3_similarity.sort_values(by=\"best_match_score\",ascending=False,inplace=True, ignore_index=True)\nfrom IPython.display import display\nfor i in range(0,5):\n    print(f\"Electrostatic potential similarities score: {bb3_similarity.loc[i,'best_match_score']}\")\n    display(Draw.MolsToGridImage([MolFromSmiles(x) for x in bb3_similarity.loc[i,\"smiles\"].split(',')], molsPerRow=4, maxMols=20))\n    \nbest_match, best_match_score  = find_best_match(df_2D_bb3, df_smiles_bb3)\nbb3_similarity = pd.DataFrame( {'smiles': best_match, 'best_match_score': best_match_score})\nbb3_similarity.sort_values(by=\"best_match_score\",ascending=False,inplace=True, ignore_index=True)\nfrom IPython.display import display\nfor i in range(0,5):\n    print(f\"Shape similarity score: {bb3_similarity.loc[i,'best_match_score']}\")\n    display(Draw.MolsToGridImage([MolFromSmiles(x) for x in bb3_similarity.loc[i,\"smiles\"].split(',')], molsPerRow=4, maxMols=20))\n    ","metadata":{"execution":{"iopub.status.busy":"2024-07-13T11:08:36.175848Z","iopub.execute_input":"2024-07-13T11:08:36.176272Z","iopub.status.idle":"2024-07-13T11:08:36.384156Z","shell.execute_reply.started":"2024-07-13T11:08:36.176229Z","shell.execute_reply":"2024-07-13T11:08:36.382891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#superimpose the bbs\nm1 = bb1_similarity.loc[0,\"smiles\"].split(',')[0]\nm2 = bb1_similarity.loc[0,\"smiles\"].split(',')[1]\ncore = Chem.MolFromSmiles(\"n1cncnc1N\")\nAllChem.EmbedMolecule(core,AllChem.ETKDG())\nm1 = MolFromSmiles(m1)\nAllChem.EmbedMolecule(m1,AllChem.ETKDG()) \nm2 = MolFromSmiles(m2)\nAllChem.EmbedMolecule(m2,AllChem.ETKDG()) \nalign_two_mol(m1,m2,core)\ndraw([m1,m2])","metadata":{"execution":{"iopub.status.busy":"2024-07-13T11:11:27.692445Z","iopub.execute_input":"2024-07-13T11:11:27.692886Z","iopub.status.idle":"2024-07-13T11:11:27.718373Z","shell.execute_reply.started":"2024-07-13T11:11:27.692852Z","shell.execute_reply":"2024-07-13T11:11:27.717155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}