{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Visualization and clustering of drugs**\n\nThe drugs used in this competition were visualized and clustered by following steps:\n\n1. calculate the fingerprint of drugs using  Physicochemical property fingerprints  <br>\n(“Topological torsion: a new molecular descriptor for SAR applications. Comparison with other descriptors” J. Chem. Inf. Comput. Sci. 1987, 27, 82-85.)\n\n2. calculate the Dice similarity for each drugs\n\n3. visualize and clustering using UMAP <br>\n* Previous notebook introducing the UMAP plot ⇒ [Visualization of gene expression using UMAP](https://www.kaggle.com/code/yusaku5739/visualization-of-gene-expression-using-umap)\n\n<br>\n\n### Physicochemical property fingerprints \n* Physicochemical property fingerprints are a method for calculating a fingerprint reflecting information regarding the entire chemical structure. This fingerprint is calculated using general chemical specificity, such as cation, anion, and hydrogen-donor. Other fingerprints may calculate the distance of bioisostere, such as carbonic acid and tetrazole, as far, whereas physicochemical property fingerprints can calculate it as near. <br> \n* In my opinion, this fingerprint is suitable because drugs are used in cell in this competition. \n\n## **Please Upvote if you Find this Useful :)**","metadata":{}},{"cell_type":"markdown","source":"### Install and import Package","metadata":{}},{"cell_type":"code","source":"!pip install rdkit-pypi","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-09-26T17:42:51.781437Z","iopub.execute_input":"2023-09-26T17:42:51.782457Z","iopub.status.idle":"2023-09-26T17:43:05.748366Z","shell.execute_reply.started":"2023-09-26T17:42:51.782414Z","shell.execute_reply":"2023-09-26T17:43:05.746969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit import rdBase, Chem, DataStructs\nfrom rdkit.Avalon import pyAvalonTools\nfrom rdkit.Chem import AllChem, Draw\nfrom rdkit.Chem.Fingerprints import FingerprintMols\nfrom rdkit.Chem.AtomPairs import Pairs, Torsions\nfrom tqdm.notebook import tqdm\nfrom sklearn.cluster import KMeans\nimport umap\nimport numpy as np\nimport pandas as pd\nimport time\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport collections\nsns.set_theme()","metadata":{"execution":{"iopub.status.busy":"2023-09-26T17:43:05.751405Z","iopub.execute_input":"2023-09-26T17:43:05.751851Z","iopub.status.idle":"2023-09-26T17:43:18.130965Z","shell.execute_reply.started":"2023-09-26T17:43:05.751807Z","shell.execute_reply":"2023-09-26T17:43:18.129572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prepare data","metadata":{}},{"cell_type":"code","source":"df = pd.read_parquet(\"/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet\")\ndf_chem = df[[\"sm_name\", \"SMILES\"]]\ndf_chem = df_chem.drop_duplicates().reset_index(drop=True).set_index(\"sm_name\")\ndf_chem","metadata":{"execution":{"iopub.status.busy":"2023-09-26T17:43:18.132615Z","iopub.execute_input":"2023-09-26T17:43:18.133333Z","iopub.status.idle":"2023-09-26T17:43:19.649761Z","shell.execute_reply.started":"2023-09-26T17:43:18.133295Z","shell.execute_reply":"2023-09-26T17:43:19.648921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Calculate fingerprint","metadata":{}},{"cell_type":"code","source":"from rdkit.Chem.AtomPairs import Sheridan\n\ndf_chem[\"fingerprint\"] = None\ndf_chem[\"mol\"] = None\ndf_chem[\"mol\"] = df_chem[\"SMILES\"].apply(lambda x: Chem.MolFromSmiles(x))\n\nfor name in df_chem.index:\n    fg = Sheridan.GetBPFingerprint(df_chem.at[name, \"mol\"])\n    df_chem.at[name, \"fingerprint\"] = fg","metadata":{"execution":{"iopub.status.busy":"2023-09-26T17:43:19.652091Z","iopub.execute_input":"2023-09-26T17:43:19.652740Z","iopub.status.idle":"2023-09-26T17:43:19.878771Z","shell.execute_reply.started":"2023-09-26T17:43:19.652704Z","shell.execute_reply":"2023-09-26T17:43:19.877879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Calculate Dice similarity","metadata":{}},{"cell_type":"code","source":"df_chem[\"distance\"] = 0\ndf_chem[\"distance\"] = df_chem[\"fingerprint\"].apply(\n    lambda x: np.array(DataStructs.cDataStructs.BulkDiceSimilarity(x, df_chem[\"fingerprint\"].values.tolist(), returnDistance=True))\n)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T17:43:19.880242Z","iopub.execute_input":"2023-09-26T17:43:19.880874Z","iopub.status.idle":"2023-09-26T17:43:19.983872Z","shell.execute_reply.started":"2023-09-26T17:43:19.880840Z","shell.execute_reply":"2023-09-26T17:43:19.982882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### UMAP plotting","metadata":{}},{"cell_type":"code","source":"# do UMAP clustering\nmapper = umap.UMAP(random_state=42,\n                   n_neighbors=3,\n                   min_dist=0.6,\n                   metric=\"correlation\")\nembedding = mapper.fit_transform(df_chem[\"distance\"].values.tolist())\n\nembedding_x = embedding[:, 0]\nembedding_y = embedding[:, 1]\n\nplt.scatter(embedding_x, embedding_y, s=30)\n\nplt.grid()\n#plt.legend(loc='upper left', bbox_to_anchor=(1, 1))\nplt.title(\"Visualization of Coumpunds\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-26T17:44:01.633306Z","iopub.execute_input":"2023-09-26T17:44:01.633726Z","iopub.status.idle":"2023-09-26T17:44:03.301111Z","shell.execute_reply.started":"2023-09-26T17:44:01.633693Z","shell.execute_reply":"2023-09-26T17:44:03.299820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kmeans = KMeans(n_clusters=5, random_state=1)\nkmeans.fit(embedding)\ndf_chem[\"label\"] = kmeans.labels_\ncolors=[\"r\", \"b\", \"g\", \"y\", \"m\", \"c\", \"k\"]\n\nfor i in range(5):\n    lab_idx = np.where(kmeans.labels_==i)\n    if i >= len(colors):\n        marker = \"^\"\n    else:\n        marker = \"o\"\n    c_i = i%len(colors)\n    plt.scatter(embedding_x[lab_idx], embedding_y[lab_idx], marker=marker, label=i, s=30, color=colors[c_i])\n\nplt.title(\"visualization of coumpaunds after clustering\")","metadata":{"execution":{"iopub.status.busy":"2023-09-26T17:44:35.308800Z","iopub.execute_input":"2023-09-26T17:44:35.309224Z","iopub.status.idle":"2023-09-26T17:44:35.834218Z","shell.execute_reply.started":"2023-09-26T17:44:35.309192Z","shell.execute_reply":"2023-09-26T17:44:35.832972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualize the drugs ","metadata":{}},{"cell_type":"code","source":"from IPython.display import display\nfor label in range(5):\n    df_chem_lab = df_chem[df_chem[\"label\"]==label]\n    mols = df_chem_lab[\"mol\"].values.tolist()\n    legends = df_chem_lab.index.values.tolist()\n    \n    img = Draw.MolsToGridImage(mols[:5],\n                               molsPerRow=5, \n                               subImgSize=(200,200),\n                               legends=legends[:5]\n                               )\n    print(f\"label : {label}\")\n    display(img)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T17:44:38.637298Z","iopub.execute_input":"2023-09-26T17:44:38.638331Z","iopub.status.idle":"2023-09-26T17:44:38.815471Z","shell.execute_reply.started":"2023-09-26T17:44:38.638279Z","shell.execute_reply":"2023-09-26T17:44:38.814348Z"},"trusted":true},"execution_count":null,"outputs":[]}]}