{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Single-Cell Perturbations chemical space analysis\nThe notebook is dedicated to analyse chemical agents used to treat PBMС, describe it and extract as much features as possible.","metadata":{}},{"cell_type":"markdown","source":"## Imports and data load","metadata":{}},{"cell_type":"code","source":"!pip install rdkit","metadata":{"execution":{"iopub.status.busy":"2023-09-26T11:58:03.477610Z","iopub.execute_input":"2023-09-26T11:58:03.478311Z","iopub.status.idle":"2023-09-26T11:58:15.901102Z","shell.execute_reply.started":"2023-09-26T11:58:03.478207Z","shell.execute_reply":"2023-09-26T11:58:15.899600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom enum import Enum\n\nfrom sklearn.manifold import TSNE\nfrom sklearn.cluster import OPTICS\nfrom umap import UMAP\n\nimport requests\nfrom urllib.parse import quote_plus\n\nimport rdkit\nfrom rdkit import Chem\nfrom rdkit.Chem import Descriptors, Descriptors3D, Draw\nfrom rdkit import DataStructs\nfrom rdkit.Chem import AllChem\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport os\nfrom pathlib import Path\ndata_dir = Path(\"/kaggle/input/open-problems-single-cell-perturbations\")\nfor p in data_dir.iterdir():\n        print(p)\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-26T11:58:15.903491Z","iopub.execute_input":"2023-09-26T11:58:15.903892Z","iopub.status.idle":"2023-09-26T11:58:15.916389Z","shell.execute_reply.started":"2023-09-26T11:58:15.903866Z","shell.execute_reply":"2023-09-26T11:58:15.915329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train_df = pd.read_parquet(data_dir/\"de_train.parquet\")\nde_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-26T11:58:15.917817Z","iopub.execute_input":"2023-09-26T11:58:15.918130Z","iopub.status.idle":"2023-09-26T11:58:17.409452Z","shell.execute_reply.started":"2023-09-26T11:58:15.918105Z","shell.execute_reply":"2023-09-26T11:58:17.408395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chemicals = de_train_df.loc[:, [\"cell_type\", \"sm_name\", \"sm_lincs_id\", \"SMILES\", \"control\"]]","metadata":{"execution":{"iopub.status.busy":"2023-09-26T11:58:17.411612Z","iopub.execute_input":"2023-09-26T11:58:17.411900Z","iopub.status.idle":"2023-09-26T11:58:17.418483Z","shell.execute_reply.started":"2023-09-26T11:58:17.411876Z","shell.execute_reply":"2023-09-26T11:58:17.417821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chemicals.to_csv(\"/kaggle/working/chemicals.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-09-26T13:45:23.400755Z","iopub.execute_input":"2023-09-26T13:45:23.401719Z","iopub.status.idle":"2023-09-26T13:45:23.413254Z","shell.execute_reply.started":"2023-09-26T13:45:23.401681Z","shell.execute_reply":"2023-09-26T13:45:23.412312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Clean chemical data with Pubchem ID assignment","metadata":{}},{"cell_type":"code","source":"# Check for 1:1 SMILES vs naming\nfor smi, sdf in chemicals.groupby(\"SMILES\"):\n    if len(sdf[\"sm_name\"].unique()) != 1:\n        print(smi, sdf[\"sm_name\"].unique())\nfor name, sdf in chemicals.groupby(\"sm_name\"):\n    if len(sdf[\"SMILES\"].unique()) != 1:\n        print(smi, sdf[\"SMILES\"].unique())","metadata":{"execution":{"iopub.status.busy":"2023-09-26T11:58:17.419593Z","iopub.execute_input":"2023-09-26T11:58:17.420052Z","iopub.status.idle":"2023-09-26T11:58:17.465913Z","shell.execute_reply.started":"2023-09-26T11:58:17.420028Z","shell.execute_reply":"2023-09-26T11:58:17.464781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_pubchem_cmpd_id_from_smiles(smiles: str) -> str:\n    url = f\"https://pubchem.ncbi.nlm.nih.gov/rest/pug/compound/smiles/cids/JSON?smiles={quote_plus(smiles)}\"\n    response = requests.get(url)\n    if response.ok:\n        cid_list = response.json()[\"IdentifierList\"][\"CID\"]\n        if len(cid_list) != 1:\n            print(f\"Multiple compounds: {cid_list}\")\n            cid = -1\n        else:\n            cid, *_ = cid_list\n    else:\n        print(f\"Invalid SMILES: {smiles}\")\n        cid = -1\n    \n    return cid","metadata":{"execution":{"iopub.status.busy":"2023-09-26T11:58:17.467351Z","iopub.execute_input":"2023-09-26T11:58:17.467680Z","iopub.status.idle":"2023-09-26T11:58:17.476730Z","shell.execute_reply.started":"2023-09-26T11:58:17.467651Z","shell.execute_reply":"2023-09-26T11:58:17.475395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign Pubchemm CID\ncid = {\n    smiles: get_pubchem_cmpd_id_from_smiles(smiles)\n    for smiles in chemicals[\"SMILES\"].unique()\n}","metadata":{"execution":{"iopub.status.busy":"2023-09-26T11:58:17.477951Z","iopub.execute_input":"2023-09-26T11:58:17.478279Z","iopub.status.idle":"2023-09-26T11:59:56.747043Z","shell.execute_reply.started":"2023-09-26T11:58:17.478236Z","shell.execute_reply":"2023-09-26T11:59:56.745718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chemicals[\"cid\"] = chemicals[\"SMILES\"].map(cid)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T11:59:56.748643Z","iopub.execute_input":"2023-09-26T11:59:56.748999Z","iopub.status.idle":"2023-09-26T11:59:56.756857Z","shell.execute_reply.started":"2023-09-26T11:59:56.748969Z","shell.execute_reply":"2023-09-26T11:59:56.755863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we are sure that all molecules are in PC as individual compounds","metadata":{}},{"cell_type":"code","source":"chemicals","metadata":{"execution":{"iopub.status.busy":"2023-09-26T11:59:56.757990Z","iopub.execute_input":"2023-09-26T11:59:56.758832Z","iopub.status.idle":"2023-09-26T11:59:56.782079Z","shell.execute_reply.started":"2023-09-26T11:59:56.758803Z","shell.execute_reply":"2023-09-26T11:59:56.781101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"# Calculate Descriptors","metadata":{}},{"cell_type":"code","source":"descriptors = []\nfor smiles in cid:\n    entry = {\"SMILES\": smiles}\n    mol = Chem.MolFromSmiles(smiles)\n    entry[\"RDMol\"] = mol\n    entry.update(\n        Descriptors.CalcMolDescriptors(mol)\n    )\n    descriptors.append(entry)\ndescriptors = pd.DataFrame.from_records(descriptors)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T13:15:22.001371Z","iopub.execute_input":"2023-09-26T13:15:22.001907Z","iopub.status.idle":"2023-09-26T13:15:25.306749Z","shell.execute_reply.started":"2023-09-26T13:15:22.001867Z","shell.execute_reply":"2023-09-26T13:15:25.305537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"descriptors.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:00:00.007354Z","iopub.execute_input":"2023-09-26T12:00:00.007689Z","iopub.status.idle":"2023-09-26T12:00:00.029697Z","shell.execute_reply.started":"2023-09-26T12:00:00.007660Z","shell.execute_reply":"2023-09-26T12:00:00.028577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Clustering via descriptors","metadata":{}},{"cell_type":"code","source":"chem_embedding = UMAP()\nchem_embedding.fit(descriptors.iloc[:, 2:])","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:00:00.031106Z","iopub.execute_input":"2023-09-26T12:00:00.031471Z","iopub.status.idle":"2023-09-26T12:00:02.940177Z","shell.execute_reply.started":"2023-09-26T12:00:00.031436Z","shell.execute_reply":"2023-09-26T12:00:02.939269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chem_cluster = OPTICS(\n    min_samples=4,\n    max_eps=np.inf\n)\nchem_cluster.fit(descriptors.iloc[:, 2:])","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:00:02.941449Z","iopub.execute_input":"2023-09-26T12:00:02.941740Z","iopub.status.idle":"2023-09-26T12:00:03.255096Z","shell.execute_reply.started":"2023-09-26T12:00:02.941716Z","shell.execute_reply":"2023-09-26T12:00:03.254209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8,7))\nax.scatter(\n    *chem_embedding.embedding_.T,\n    c=chem_cluster.labels_,\n)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:00:03.259181Z","iopub.execute_input":"2023-09-26T12:00:03.261228Z","iopub.status.idle":"2023-09-26T12:00:03.548532Z","shell.execute_reply.started":"2023-09-26T12:00:03.261194Z","shell.execute_reply":"2023-09-26T12:00:03.547553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Clustering by fingerprints","metadata":{}},{"cell_type":"code","source":"fpgen = AllChem.GetRDKitFPGenerator()\ndescriptors[\"fingerprint\"] = descriptors[\"RDMol\"].map(fpgen.GetFingerprint)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T13:15:29.680849Z","iopub.execute_input":"2023-09-26T13:15:29.681232Z","iopub.status.idle":"2023-09-26T13:15:30.028501Z","shell.execute_reply.started":"2023-09-26T13:15:29.681203Z","shell.execute_reply":"2023-09-26T13:15:30.027257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"descriptors[\"fingerprint\"]","metadata":{"execution":{"iopub.status.busy":"2023-09-26T13:15:38.069267Z","iopub.execute_input":"2023-09-26T13:15:38.069690Z","iopub.status.idle":"2023-09-26T13:15:38.088553Z","shell.execute_reply.started":"2023-09-26T13:15:38.069654Z","shell.execute_reply":"2023-09-26T13:15:38.087466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Similarity Metrics\nTanimoto = DataStructs.TanimotoSimilarity\nDice = DataStructs.DiceSimilarity\nCosine = DataStructs.CosineSimilarity","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:56:57.623484Z","iopub.execute_input":"2023-09-26T12:56:57.624552Z","iopub.status.idle":"2023-09-26T12:56:57.629079Z","shell.execute_reply.started":"2023-09-26T12:56:57.624463Z","shell.execute_reply":"2023-09-26T12:56:57.627999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculate_distance_matrix(method=Tanimoto):\n    similarity_matrix = np.zeros((146, 146))\n    for i, rowi in descriptors.iterrows():\n        fpi = rowi[\"fingerprint\"]\n        for j, rowj in descriptors.iterrows():\n            fpj = rowj[\"fingerprint\"]\n            sim = method(fpi, fpj)\n            similarity_matrix[i, j] = sim\n    return 1/similarity_matrix - 1","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:56:57.871560Z","iopub.execute_input":"2023-09-26T12:56:57.872126Z","iopub.status.idle":"2023-09-26T12:56:57.878013Z","shell.execute_reply.started":"2023-09-26T12:56:57.872097Z","shell.execute_reply":"2023-09-26T12:56:57.876802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculate_similarity_matrix(method=Tanimoto):\n    similarity_matrix = np.zeros((146, 146))\n    for i, rowi in descriptors.iterrows():\n        fpi = rowi[\"fingerprint\"]\n        for j, rowj in descriptors.iterrows():\n            fpj = rowj[\"fingerprint\"]\n            sim = method(fpi, fpj)\n            similarity_matrix[i, j] = sim\n    return similarity_matrix","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:57:01.451332Z","iopub.execute_input":"2023-09-26T12:57:01.452555Z","iopub.status.idle":"2023-09-26T12:57:01.459022Z","shell.execute_reply.started":"2023-09-26T12:57:01.452509Z","shell.execute_reply":"2023-09-26T12:57:01.457953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8,7))\nsns.heatmap(\n    (calculate_similarity_matrix(Tanimoto)),\n    ax=ax,\n)\nax.set_xlabel(\"Compound\")\nax.set_title(\"Tanimoto Similarity\")","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:57:01.623304Z","iopub.execute_input":"2023-09-26T12:57:01.624076Z","iopub.status.idle":"2023-09-26T12:57:04.397148Z","shell.execute_reply.started":"2023-09-26T12:57:01.624034Z","shell.execute_reply":"2023-09-26T12:57:04.396009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Median similarity\nmediansim = np.median(\n    calculate_similarity_matrix(Tanimoto),\n    axis=1,\n)\nfig, ax = plt.subplots(figsize=(8,7))\nax.scatter(\n    range(146),\n    mediansim,\n    c=np.log(mediansim),\n)\nax.set_xlabel(\"Compound\")\nax.set_ylabel(\"Median Similarity\")","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:12:52.923525Z","iopub.execute_input":"2023-09-26T12:12:52.923964Z","iopub.status.idle":"2023-09-26T12:12:54.993683Z","shell.execute_reply.started":"2023-09-26T12:12:52.923934Z","shell.execute_reply":"2023-09-26T12:12:54.992545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8,7))\nsns.heatmap(\n    np.log(1 + calculate_distance_matrix(Tanimoto)),\n    ax=ax,\n)\nax.set_xlabel(\"Compound\")\nax.set_title(\"log Tanimoto distance\")","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:00:03.970411Z","iopub.execute_input":"2023-09-26T12:00:03.970748Z","iopub.status.idle":"2023-09-26T12:00:08.211392Z","shell.execute_reply.started":"2023-09-26T12:00:03.970721Z","shell.execute_reply":"2023-09-26T12:00:08.210609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(np.log(1 + calculate_distance_matrix(Dice)))","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:00:08.212872Z","iopub.execute_input":"2023-09-26T12:00:08.213453Z","iopub.status.idle":"2023-09-26T12:00:10.625949Z","shell.execute_reply.started":"2023-09-26T12:00:08.213421Z","shell.execute_reply":"2023-09-26T12:00:10.624893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(np.log(1 + calculate_distance_matrix(Cosine)))","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:00:10.627632Z","iopub.execute_input":"2023-09-26T12:00:10.628425Z","iopub.status.idle":"2023-09-26T12:00:13.005256Z","shell.execute_reply.started":"2023-09-26T12:00:10.628379Z","shell.execute_reply":"2023-09-26T12:00:13.004410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Median distance\nmedians = np.median(\n    calculate_distance_matrix(Tanimoto),\n    axis=1,\n)\nfig, ax = plt.subplots(figsize=(8,7))\nax.scatter(\n    range(146),\n    medians,\n    c=np.log(medians),\n)\nax.set_xlabel(\"Compound\")\nax.set_ylabel(\"Median Distance\")","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:00:13.006587Z","iopub.execute_input":"2023-09-26T12:00:13.007346Z","iopub.status.idle":"2023-09-26T12:00:15.057060Z","shell.execute_reply.started":"2023-09-26T12:00:13.007317Z","shell.execute_reply":"2023-09-26T12:00:15.055940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Outliers\nnp.where(medians >= 5)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:00:15.058377Z","iopub.execute_input":"2023-09-26T12:00:15.058717Z","iopub.status.idle":"2023-09-26T12:00:15.065856Z","shell.execute_reply.started":"2023-09-26T12:00:15.058690Z","shell.execute_reply":"2023-09-26T12:00:15.064690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We have one strong outlier – very distinct compound\n# compound 52 – hydroxyurea\nDraw.MolToImage(descriptors.loc[52, \"RDMol\"])","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:00:15.067363Z","iopub.execute_input":"2023-09-26T12:00:15.068017Z","iopub.status.idle":"2023-09-26T12:00:15.093664Z","shell.execute_reply.started":"2023-09-26T12:00:15.067981Z","shell.execute_reply":"2023-09-26T12:00:15.092840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# All Outliers\nDraw.MolsToImage(descriptors.loc[np.where(medians >= 5), \"RDMol\"])","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:00:15.094895Z","iopub.execute_input":"2023-09-26T12:00:15.095205Z","iopub.status.idle":"2023-09-26T12:00:15.145311Z","shell.execute_reply.started":"2023-09-26T12:00:15.095180Z","shell.execute_reply":"2023-09-26T12:00:15.144200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Distance Clustering","metadata":{}},{"cell_type":"code","source":"distance_umap = UMAP()\ndistance_umap.fit(calculate_distance_matrix(Tanimoto))","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:00:15.146488Z","iopub.execute_input":"2023-09-26T12:00:15.146786Z","iopub.status.idle":"2023-09-26T12:00:19.640117Z","shell.execute_reply.started":"2023-09-26T12:00:15.146761Z","shell.execute_reply":"2023-09-26T12:00:19.639069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"distance_cluster = OPTICS(\n    min_samples=4,\n    max_eps=np.inf\n)\ndistance_cluster.fit(calculate_distance_matrix(Tanimoto))","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:00:19.641233Z","iopub.execute_input":"2023-09-26T12:00:19.641529Z","iopub.status.idle":"2023-09-26T12:00:21.548498Z","shell.execute_reply.started":"2023-09-26T12:00:19.641504Z","shell.execute_reply":"2023-09-26T12:00:21.547591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8,7))\nax.scatter(\n    *distance_umap.embedding_.T,\n    c=distance_cluster.labels_,\n)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:00:21.555241Z","iopub.execute_input":"2023-09-26T12:00:21.555756Z","iopub.status.idle":"2023-09-26T12:00:21.834972Z","shell.execute_reply.started":"2023-09-26T12:00:21.555724Z","shell.execute_reply":"2023-09-26T12:00:21.833953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Combining Chemical Features with main dataset","metadata":{}},{"cell_type":"code","source":"chem_descriptors = chemicals.join(descriptors.drop(\"fingerprint\", axis=1).set_index(\"SMILES\"), on=\"SMILES\", how=\"left\")\nchem_descriptors.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-26T12:46:36.943834Z","iopub.execute_input":"2023-09-26T12:46:36.944199Z","iopub.status.idle":"2023-09-26T12:46:36.971971Z","shell.execute_reply.started":"2023-09-26T12:46:36.944172Z","shell.execute_reply":"2023-09-26T12:46:36.970967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chem_features = pd.concat([descriptors.drop(\"fingerprint\", axis=1), descriptors[\"fingerprint\"].map(lambda fp: fp.ToList()).apply(pd.Series).add_prefix(\"fp_\")], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T13:18:09.303559Z","iopub.execute_input":"2023-09-26T13:18:09.303988Z","iopub.status.idle":"2023-09-26T13:18:09.534121Z","shell.execute_reply.started":"2023-09-26T13:18:09.303948Z","shell.execute_reply":"2023-09-26T13:18:09.533031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chem_features_descriptors = list(chem_features.columns[2:126])\nchem_features_fragments = list(chem_features.columns[126:211])\nchem_features_fingerprints = list(chem_features.columns[211:])\n\ngene_features = list(de_train_df.columns[5:])","metadata":{"execution":{"iopub.status.busy":"2023-09-26T13:40:40.377416Z","iopub.execute_input":"2023-09-26T13:40:40.378288Z","iopub.status.idle":"2023-09-26T13:40:40.387570Z","shell.execute_reply.started":"2023-09-26T13:40:40.378250Z","shell.execute_reply":"2023-09-26T13:40:40.386441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_umap = UMAP()\nall_umap.fit(chem_features.iloc[:, 2:])\nall_cluster = OPTICS(\n    min_samples=7,\n    max_eps=np.inf\n)\nall_cluster.fit(chem_features.iloc[:, 2:])","metadata":{"execution":{"iopub.status.busy":"2023-09-26T13:34:14.747447Z","iopub.execute_input":"2023-09-26T13:34:14.747898Z","iopub.status.idle":"2023-09-26T13:34:17.866346Z","shell.execute_reply.started":"2023-09-26T13:34:14.747862Z","shell.execute_reply":"2023-09-26T13:34:17.865366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8,7))\nax.scatter(\n    *all_umap.embedding_.T,\n    c=all_cluster.labels_,\n    s=5*(all_cluster.labels_+1.1)\n)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T13:34:37.421569Z","iopub.execute_input":"2023-09-26T13:34:37.422486Z","iopub.status.idle":"2023-09-26T13:34:37.704144Z","shell.execute_reply.started":"2023-09-26T13:34:37.422439Z","shell.execute_reply":"2023-09-26T13:34:37.703080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_with_chemicals = de_train_df.join(chem_features.set_index(\"SMILES\"), on=\"SMILES\", how=\"left\")","metadata":{"execution":{"iopub.status.busy":"2023-09-26T13:29:31.517989Z","iopub.execute_input":"2023-09-26T13:29:31.518401Z","iopub.status.idle":"2023-09-26T13:29:31.597331Z","shell.execute_reply.started":"2023-09-26T13:29:31.518367Z","shell.execute_reply":"2023-09-26T13:29:31.596487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_with_chemicals[\"cid\"] = data_with_chemicals[\"SMILES\"].map(cid)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T15:14:22.079593Z","iopub.execute_input":"2023-09-26T15:14:22.080058Z","iopub.status.idle":"2023-09-26T15:14:22.088484Z","shell.execute_reply.started":"2023-09-26T15:14:22.080024Z","shell.execute_reply":"2023-09-26T15:14:22.087307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_with_chemicals.drop(\"RDMol\", axis=1).to_parquet(\"/kaggle/working/data_with_chemicals.parquet\")","metadata":{"execution":{"iopub.status.busy":"2023-09-26T15:14:30.046737Z","iopub.execute_input":"2023-09-26T15:14:30.047113Z","iopub.status.idle":"2023-09-26T15:14:34.328900Z","shell.execute_reply.started":"2023-09-26T15:14:30.047082Z","shell.execute_reply":"2023-09-26T15:14:34.327699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_with_chemicals[\"cid\"]","metadata":{"execution":{"iopub.status.busy":"2023-09-26T15:14:36.230650Z","iopub.execute_input":"2023-09-26T15:14:36.231289Z","iopub.status.idle":"2023-09-26T15:14:36.240732Z","shell.execute_reply.started":"2023-09-26T15:14:36.231253Z","shell.execute_reply":"2023-09-26T15:14:36.239563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = data_with_chemicals[chem_features_descriptors + chem_features_fragments + chem_features_fingerprints]","metadata":{"execution":{"iopub.status.busy":"2023-09-26T13:41:16.600597Z","iopub.execute_input":"2023-09-26T13:41:16.601361Z","iopub.status.idle":"2023-09-26T13:41:16.614719Z","shell.execute_reply.started":"2023-09-26T13:41:16.601321Z","shell.execute_reply":"2023-09-26T13:41:16.613519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y = data_with_chemicals[gene_features]","metadata":{"execution":{"iopub.status.busy":"2023-09-26T13:41:32.543417Z","iopub.execute_input":"2023-09-26T13:41:32.543783Z","iopub.status.idle":"2023-09-26T13:41:32.591894Z","shell.execute_reply.started":"2023-09-26T13:41:32.543756Z","shell.execute_reply":"2023-09-26T13:41:32.590986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pubchem / Chembl assay data","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}