{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-20T22:48:14.150489Z","iopub.execute_input":"2024-05-20T22:48:14.150831Z","iopub.status.idle":"2024-05-20T22:48:15.307478Z","shell.execute_reply.started":"2024-05-20T22:48:14.150802Z","shell.execute_reply":"2024-05-20T22:48:15.306385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install rdkit","metadata":{"execution":{"iopub.status.busy":"2024-05-20T22:48:15.309452Z","iopub.execute_input":"2024-05-20T22:48:15.309956Z","iopub.status.idle":"2024-05-20T22:48:15.313896Z","shell.execute_reply.started":"2024-05-20T22:48:15.309923Z","shell.execute_reply":"2024-05-20T22:48:15.313015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install rdkit\n!pip install duckdb\n!pip install deepchem\n# !pip install dgllife==0.3.2\n# !pip istall rdkit==rdkit-2023.3.1","metadata":{"execution":{"iopub.status.busy":"2024-05-20T22:56:35.654283Z","iopub.execute_input":"2024-05-20T22:56:35.655097Z","iopub.status.idle":"2024-05-20T22:57:20.899171Z","shell.execute_reply.started":"2024-05-20T22:56:35.655063Z","shell.execute_reply":"2024-05-20T22:57:20.898239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import duckdb\nimport pandas as pd\n\ntrain_path = '/kaggle/input/leash-BELKA/train.parquet'\n\ntest_path = '/kaggle/input/leash-BELKA/test.parquet'\n\ncon = duckdb.connect()\n\ndf = con.query(f\"\"\"(SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0\n                        ORDER BY random()\n                        LIMIT 50000)\n                        UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1\n                        ORDER BY random()\n                        LIMIT 50000)\"\"\").df()\n\ncon.close()","metadata":{"execution":{"iopub.status.busy":"2024-05-15T18:52:22.127421Z","iopub.execute_input":"2024-05-15T18:52:22.127768Z","iopub.status.idle":"2024-05-15T18:53:09.335638Z","shell.execute_reply.started":"2024-05-15T18:52:22.127734Z","shell.execute_reply":"2024-05-15T18:53:09.334829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()\nprint(len(df))","metadata":{"execution":{"iopub.status.busy":"2024-05-15T18:53:09.336768Z","iopub.execute_input":"2024-05-15T18:53:09.337050Z","iopub.status.idle":"2024-05-15T18:53:09.342119Z","shell.execute_reply.started":"2024-05-15T18:53:09.337025Z","shell.execute_reply":"2024-05-15T18:53:09.341273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import rdkit.Chem as Chem\nfrom rdkit.Chem import Draw","metadata":{"execution":{"iopub.status.busy":"2024-05-20T23:32:40.261673Z","iopub.execute_input":"2024-05-20T23:32:40.262650Z","iopub.status.idle":"2024-05-20T23:32:40.377713Z","shell.execute_reply.started":"2024-05-20T23:32:40.262615Z","shell.execute_reply":"2024-05-20T23:32:40.376678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(df.head())","metadata":{"execution":{"iopub.status.busy":"2024-05-15T18:53:09.483351Z","iopub.execute_input":"2024-05-15T18:53:09.483780Z","iopub.status.idle":"2024-05-15T18:53:09.508394Z","shell.execute_reply.started":"2024-05-15T18:53:09.483747Z","shell.execute_reply":"2024-05-15T18:53:09.507521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"demo_smiles = df['molecule_smiles'].values[0]\ndemo_smiles_bb1 = df['buildingblock1_smiles'].values[0]\ndemo_smiles_bb2 = df['buildingblock2_smiles'].values[0]\ndemo_smiles_bb3 = df['buildingblock3_smiles'].values[0]\nmol = Chem.MolFromSmiles(demo_smiles)\nprint(\"SMILES:\", demo_smiles)\nDraw.MolToImage(mol)\n\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-15T18:53:09.509651Z","iopub.execute_input":"2024-05-15T18:53:09.510295Z","iopub.status.idle":"2024-05-15T18:53:09.547054Z","shell.execute_reply.started":"2024-05-15T18:53:09.510258Z","shell.execute_reply":"2024-05-15T18:53:09.546257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mol_bb1 = Chem.MolFromSmiles(demo_smiles_bb1)\nprint(\"SMILES:\", demo_smiles_bb1)\nDraw.MolToImage(mol_bb1)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T18:53:09.547971Z","iopub.execute_input":"2024-05-15T18:53:09.548199Z","iopub.status.idle":"2024-05-15T18:53:09.568793Z","shell.execute_reply.started":"2024-05-15T18:53:09.548178Z","shell.execute_reply":"2024-05-15T18:53:09.567869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mol_bb2 = Chem.MolFromSmiles(demo_smiles_bb2)\nprint(\"SMILES:\", demo_smiles_bb2)\nDraw.MolToImage(mol_bb2)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T18:53:09.571363Z","iopub.execute_input":"2024-05-15T18:53:09.571707Z","iopub.status.idle":"2024-05-15T18:53:09.591116Z","shell.execute_reply.started":"2024-05-15T18:53:09.571683Z","shell.execute_reply":"2024-05-15T18:53:09.590306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mol_bb3 = Chem.MolFromSmiles(demo_smiles_bb3)\nprint(\"SMILES:\", demo_smiles_bb3)\nDraw.MolToImage(mol_bb3)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T18:53:09.592105Z","iopub.execute_input":"2024-05-15T18:53:09.592349Z","iopub.status.idle":"2024-05-15T18:53:09.612287Z","shell.execute_reply.started":"2024-05-15T18:53:09.592328Z","shell.execute_reply":"2024-05-15T18:53:09.611487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#@title Basic graph structure\n\nprint(\"ATOMS (vertices)\")\nfor idx, atom in enumerate(mol.GetAtoms()):\n  print(\" - [%d] %s\" %(idx, atom.GetSymbol()))\nprint()\nprint()\nprint(\"BONDS (edges)\")\nfor bond in mol.GetBonds():\n  print(\" - [%d:%d] \" % (bond.GetBeginAtomIdx(), bond.GetEndAtomIdx()))","metadata":{"execution":{"iopub.status.busy":"2024-05-15T18:53:09.613394Z","iopub.execute_input":"2024-05-15T18:53:09.613638Z","iopub.status.idle":"2024-05-15T18:53:09.620287Z","shell.execute_reply.started":"2024-05-15T18:53:09.613616Z","shell.execute_reply":"2024-05-15T18:53:09.619428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import deepchem\n# from deepchem.feat import MolGraphConvFeaturizer as MGCF\n","metadata":{"execution":{"iopub.status.busy":"2024-05-15T18:53:09.621276Z","iopub.execute_input":"2024-05-15T18:53:09.621977Z","iopub.status.idle":"2024-05-15T18:53:30.157867Z","shell.execute_reply.started":"2024-05-15T18:53:09.621944Z","shell.execute_reply":"2024-05-15T18:53:30.156902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from deepchem.utils.typing import RDKitMol\nfrom deepchem.feat.base_classes import MolecularFeaturizer\nfrom deepchem.feat import MolGraphConvFeaturizer as MGCF\nimport torch\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-15T18:57:41.690476Z","iopub.execute_input":"2024-05-15T18:57:41.691252Z","iopub.status.idle":"2024-05-15T18:57:41.696156Z","shell.execute_reply.started":"2024-05-15T18:57:41.691218Z","shell.execute_reply":"2024-05-15T18:57:41.694988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport torch\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nos.environ['TORCH'] = torch.__version__\nprint(torch.__version__)\n\n!pip install -q torch-scatter -f https://data.pyg.org/whl/torch-${TORCH}.html\n!pip install -q torch-sparse -f https://data.pyg.org/whl/torch-${TORCH}.html\n!pip install -q git+https://github.com/pyg-team/pytorch_geometric.git","metadata":{"execution":{"iopub.status.busy":"2024-05-15T19:02:18.328605Z","iopub.execute_input":"2024-05-15T19:02:18.329537Z","iopub.status.idle":"2024-05-15T19:03:12.393759Z","shell.execute_reply.started":"2024-05-15T19:02:18.329497Z","shell.execute_reply":"2024-05-15T19:03:12.392371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# try:\n#     import torch_geometric\n# # except ModuleNotFoundError:\n# except:\n#     # Installing torch geometric packages with specific CUDA+PyTorch version.\n#     # See https://pytorch-geometric.readthedocs.io/en/latest/notes/installation.html for details\n#     TORCH = torch.__version__.split('+')[0]\n#     CUDA = 'cu' + torch.version.cuda.replace('.','')\n\n#     !pip install torch-scatter     -f https://pytorch-geometric.com/whl/torch-{TORCH}+{CUDA}.html\n#     !pip install torch-sparse      -f https://pytorch-geometric.com/whl/torch-{TORCH}+{CUDA}.html\n#     !pip install torch-cluster     -f https://pytorch-geometric.com/whl/torch-{TORCH}+{CUDA}.html\n#     !pip install torch-spline-conv -f https://pytorch-geometric.com/whl/torch-{TORCH}+{CUDA}.html\n#     !pip install torch-geometric\n#     import torch_geometric\n# import torch_geometric.nn as geom_nn\n# import torch_geometric.data as geom_data","metadata":{"execution":{"iopub.status.busy":"2024-05-15T19:04:59.102521Z","iopub.execute_input":"2024-05-15T19:04:59.102908Z","iopub.status.idle":"2024-05-15T19:06:02.769283Z","shell.execute_reply.started":"2024-05-15T19:04:59.102880Z","shell.execute_reply":"2024-05-15T19:06:02.767793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfeat_graph = MGCF(use_edges=True).featurize([demo_smiles])[0] # node_features=[21, 30], edge_index=[2, 46], edge_features=[46, 11], pos=[0]\n\nx = torch.tensor(feat_graph.node_features, dtype=torch.float32) # node input features from deepchem\nedge_index = torch.tensor(feat_graph.edge_index, dtype=torch.long)\nedge_attr = torch.tensor(feat_graph.edge_features, dtype=torch.long)\n# pos = feat_graph.pos\n# y = logD.values[0] # logD is our label\nprint(x)\nprint(edge_index)\nprint(edge_attr)","metadata":{"execution":{"iopub.status.busy":"2024-05-15T19:07:58.873082Z","iopub.execute_input":"2024-05-15T19:07:58.873445Z","iopub.status.idle":"2024-05-15T19:07:58.915722Z","shell.execute_reply.started":"2024-05-15T19:07:58.873417Z","shell.execute_reply":"2024-05-15T19:07:58.914868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-05-15T19:39:34.275178Z","iopub.execute_input":"2024-05-15T19:39:34.275601Z","iopub.status.idle":"2024-05-15T19:39:34.343719Z","shell.execute_reply.started":"2024-05-15T19:39:34.275571Z","shell.execute_reply":"2024-05-15T19:39:34.342413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import duckdb\nimport pandas as pd\n\ntrain_path = '/kaggle/input/leash-BELKA/train.parquet'\n\ntest_path = '/kaggle/input/leash-BELKA/test.parquet'\n\ncon = duckdb.connect()\n\ndf_bind = con.query(f\"\"\"(SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1\n                        ORDER BY random()\n                        LIMIT 50000)\"\"\").df()\n\ncon.close()","metadata":{"execution":{"iopub.status.busy":"2024-05-20T23:02:14.850940Z","iopub.execute_input":"2024-05-20T23:02:14.851364Z","iopub.status.idle":"2024-05-20T23:02:33.293708Z","shell.execute_reply.started":"2024-05-20T23:02:14.851330Z","shell.execute_reply":"2024-05-20T23:02:33.292824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_bind_sEH = df_bind[df_bind['protein_name'] == 'sEH']\ndf_bind_HSA = df_bind[df_bind['protein_name'] == 'HSA']\ndf_bind_BRD4 = df_bind[df_bind['protein_name'] == 'BRD4']","metadata":{"execution":{"iopub.status.busy":"2024-05-20T23:03:52.950004Z","iopub.execute_input":"2024-05-20T23:03:52.950375Z","iopub.status.idle":"2024-05-20T23:03:52.991710Z","shell.execute_reply.started":"2024-05-20T23:03:52.950346Z","shell.execute_reply":"2024-05-20T23:03:52.990906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_bind_BRD4","metadata":{"execution":{"iopub.status.busy":"2024-05-20T23:03:57.761993Z","iopub.execute_input":"2024-05-20T23:03:57.762902Z","iopub.status.idle":"2024-05-20T23:03:57.778299Z","shell.execute_reply.started":"2024-05-20T23:03:57.762863Z","shell.execute_reply":"2024-05-20T23:03:57.777226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_bind_BRD4['buildingblock1_smiles'].value_counts()[:30]","metadata":{"execution":{"iopub.status.busy":"2024-05-20T23:10:06.549276Z","iopub.execute_input":"2024-05-20T23:10:06.549649Z","iopub.status.idle":"2024-05-20T23:10:06.561497Z","shell.execute_reply.started":"2024-05-20T23:10:06.549622Z","shell.execute_reply":"2024-05-20T23:10:06.560484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_bind_sEH['buildingblock1_smiles'].value_counts()[:30]","metadata":{"execution":{"iopub.status.busy":"2024-05-20T23:10:31.388807Z","iopub.execute_input":"2024-05-20T23:10:31.389144Z","iopub.status.idle":"2024-05-20T23:10:31.404521Z","shell.execute_reply.started":"2024-05-20T23:10:31.389120Z","shell.execute_reply":"2024-05-20T23:10:31.403550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N = 10\nprint('buildingblock1 bind BRD4', df_bind_BRD4['buildingblock1_smiles'].value_counts()[:N])\nprint('buildingblock2 bind BRD4', df_bind_BRD4['buildingblock2_smiles'].value_counts()[:N])\nprint('buildingblock3 bind BRD4', df_bind_BRD4['buildingblock3_smiles'].value_counts()[:N])","metadata":{"execution":{"iopub.status.busy":"2024-05-20T23:27:12.283790Z","iopub.execute_input":"2024-05-20T23:27:12.284552Z","iopub.status.idle":"2024-05-20T23:27:12.302242Z","shell.execute_reply.started":"2024-05-20T23:27:12.284522Z","shell.execute_reply":"2024-05-20T23:27:12.301339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit.Chem.Draw import IPythonConsole\nfrom IPython.display import SVG\n\nfrom rdkit.Chem import rdDepictor\nfrom rdkit.Chem.Draw import rdMolDraw2D\ndef moltosvg(mol,molSize=(450,150),kekulize=True):\n    mc = Chem.Mol(mol.ToBinary())\n    if kekulize:\n        try:\n            Chem.Kekulize(mc)\n        except:\n            mc = Chem.Mol(mol.ToBinary())\n    if not mc.GetNumConformers():\n        rdDepictor.Compute2DCoords(mc)\n    drawer = rdMolDraw2D.MolDraw2DSVG(molSize[0],molSize[1])\n    drawer.DrawMolecule(mc)\n    drawer.FinishDrawing()\n    svg = drawer.GetDrawingText()\n    return svg\n\ndef render_svg(svg):\n    # It seems that the svg renderer used doesn't quite hit the spec.\n    # Here are some fixes to make it work in the notebook, although I think\n    # the underlying issue needs to be resolved at the generation step\n    return SVG(svg.replace('svg:',''))","metadata":{"execution":{"iopub.status.busy":"2024-05-20T23:51:12.154395Z","iopub.execute_input":"2024-05-20T23:51:12.154947Z","iopub.status.idle":"2024-05-20T23:51:12.163442Z","shell.execute_reply.started":"2024-05-20T23:51:12.154919Z","shell.execute_reply":"2024-05-20T23:51:12.162346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# top 10\nmol_bind_BRD4 = df_bind_BRD4['buildingblock1_smiles'].value_counts()[:N].index.tolist()\nfor idx in range(N):\n    demo_mol = mol_bind_BRD4[idx]\n    mol = Chem.MolFromSmiles(demo_mol)\n    print('bind to BRD4')\n    print('binding count:', df_bind_BRD4['buildingblock1_smiles'].value_counts().iloc[idx])\n    run = render_svg(moltosvg(mol))\n    display(run)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-20T23:54:35.497414Z","iopub.execute_input":"2024-05-20T23:54:35.498234Z","iopub.status.idle":"2024-05-20T23:54:35.625971Z","shell.execute_reply.started":"2024-05-20T23:54:35.498200Z","shell.execute_reply":"2024-05-20T23:54:35.625091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mol_bind_sEH = df_bind_sEH['buildingblock1_smiles'].value_counts()[:N].index.tolist()\nfor idx in range(N):\n    demo_mol = mol_bind_sEH[idx]\n    mol = Chem.MolFromSmiles(demo_mol)\n    print('bind to sEH')\n    print('binding count:', df_bind_sEH['buildingblock1_smiles'].value_counts().iloc[idx])\n    run = render_svg(moltosvg(mol))\n    display(run)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-20T23:57:12.439015Z","iopub.execute_input":"2024-05-20T23:57:12.439861Z","iopub.status.idle":"2024-05-20T23:57:12.583135Z","shell.execute_reply.started":"2024-05-20T23:57:12.439826Z","shell.execute_reply":"2024-05-20T23:57:12.582281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mol_bind_HSA = df_bind_HSA['buildingblock1_smiles'].value_counts()[:N].index.tolist()\nfor idx in range(N):\n    demo_mol = mol_bind_HSA[idx]\n    mol = Chem.MolFromSmiles(demo_mol)\n    print('bind to HSA')\n    print('binding count:', df_bind_HSA['buildingblock1_smiles'].value_counts().iloc[idx])\n    run = render_svg(moltosvg(mol))\n    display(run)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-20T23:58:39.358328Z","iopub.execute_input":"2024-05-20T23:58:39.359100Z","iopub.status.idle":"2024-05-20T23:58:39.498406Z","shell.execute_reply.started":"2024-05-20T23:58:39.359065Z","shell.execute_reply":"2024-05-20T23:58:39.497588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mol_bind_HSA = df_bind_HSA['buildingblock2_smiles'].value_counts()[:N].index.tolist()\nfor idx in range(N):\n    demo_mol = mol_bind_HSA[idx]\n    mol = Chem.MolFromSmiles(demo_mol)\n    print('bind to HSA')\n    print('binding count:', df_bind_HSA['buildingblock2_smiles'].value_counts().iloc[idx])\n    run = render_svg(moltosvg(mol))\n    display(run)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-20T23:59:03.480055Z","iopub.execute_input":"2024-05-20T23:59:03.480398Z","iopub.status.idle":"2024-05-20T23:59:03.576857Z","shell.execute_reply.started":"2024-05-20T23:59:03.480373Z","shell.execute_reply":"2024-05-20T23:59:03.575883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}