{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install chembl_webresource_client\n!pip install rdkit\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom pathlib import Path\n\nimport pandas as pd\nfrom chembl_webresource_client.new_client import new_client\nfrom rdkit import Chem\nfrom rdkit.Chem import AllChem\n\nimport seaborn as sns\nfrom sklearn.manifold import TSNE\nimport matplotlib.pyplot as plt\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        path = Path(dirname).joinpath(filename)\n        print(path)\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-18T04:29:26.952974Z","iopub.execute_input":"2023-09-18T04:29:26.953469Z","iopub.status.idle":"2023-09-18T04:29:53.924197Z","shell.execute_reply.started":"2023-09-18T04:29:26.953434Z","shell.execute_reply":"2023-09-18T04:29:53.922646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def smile_to_morgan_bitfp(smile, radius=2, nBits=512):\n    m = Chem.MolFromSmiles(smile)\n    return AllChem.GetMorganFingerprintAsBitVect(m, radius=radius, nBits=nBits)\n\ndef fetch_molecule_by_smiles(smiles):\n    molecule = new_client.molecule\n    res = molecule.filter(molecule_structures__canonical_smiles=smiles)\n    return res[0]['molecule_chembl_id'] if res else None\n\ndef fetch_mechanism_data(chembl_id):\n    mechanism = new_client.mechanism\n    mechanism_data = mechanism.filter(molecule_chembl_id=chembl_id)\n    \n    unique_mechanisms = {}\n    for mech in mechanism_data:\n        target_id = mech['target_chembl_id']\n        mechanism_of_action = mech['mechanism_of_action']\n        \n        # Store mechanism of action for each unique target\n        unique_mechanisms[target_id] = mechanism_of_action\n    \n    return unique_mechanisms\n\ndef fetch_target_name(target_id):\n    target = new_client.target\n    target_data = target.get(target_id)\n    return target_data['pref_name']\n\ndef fetch_moa_by_smiles(smiles_list):\n    df_list = []\n    for smiles in smiles_list:\n        chembl_id = fetch_molecule_by_smiles(smiles)\n        if chembl_id is None:\n            print(f\"No ChEMBL ID found for SMILES: {smiles}\")\n            continue\n        \n        unique_mechanisms = fetch_mechanism_data(chembl_id)\n        for target_id, mechanism_of_action in unique_mechanisms.items():\n            if target_id is None:\n                print(f\"No target ID found for ChEMBL ID: {chembl_id}\")\n                continue\n            \n            try:\n                target_name = fetch_target_name(target_id)\n            except Exception as e:\n                print(f\"An error occurred while fetching target name for target ID: {target_id}. Error: {e}\")\n                continue\n\n            df_list.append({\n                'SMILES': smiles,\n                'ChEMBL_ID': chembl_id,\n                'Target_ChEMBL_ID': target_id,\n                'Target_Name': target_name,\n                'Mechanism_of_Action': mechanism_of_action\n            })\n            \n    return pd.DataFrame(df_list)","metadata":{"execution":{"iopub.status.busy":"2023-09-18T04:29:53.928309Z","iopub.execute_input":"2023-09-18T04:29:53.928761Z","iopub.status.idle":"2023-09-18T04:29:53.945047Z","shell.execute_reply.started":"2023-09-18T04:29:53.928716Z","shell.execute_reply":"2023-09-18T04:29:53.943786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = Path(\"/kaggle/input/open-problems-single-cell-perturbations\")\nmeta = pd.read_csv(path.joinpath(\"adata_obs_meta.csv\"))\nmeta.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-18T04:29:53.946780Z","iopub.execute_input":"2023-09-18T04:29:53.947192Z","iopub.status.idle":"2023-09-18T04:29:54.865880Z","shell.execute_reply.started":"2023-09-18T04:29:53.947160Z","shell.execute_reply":"2023-09-18T04:29:54.864651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fetch MOA\nWIP, not retrieving many MOA or targets","metadata":{}},{"cell_type":"code","source":"moa_file = Path(\"/kaggle/working/moa.csv\")\nif moa_file.exists():\n    moa = pd.read_csv(moa_file)\nelse:\n    moa = fetch_moa_by_smiles(meta['SMILES'].unique())\n    print(f\"Recovered MOA for {moa['SMILES'].nunique()} compounds\")\n    moa.index = moa['SMILES'].map(meta.drop_duplicates('SMILES').set_index('SMILES')['sm_name'])\n    moa.index = moa.index.rename('sm_name')\n    moa.to_csv(moa_file)","metadata":{"execution":{"iopub.status.busy":"2023-09-18T04:30:27.553390Z","iopub.execute_input":"2023-09-18T04:30:27.553860Z","iopub.status.idle":"2023-09-18T04:30:28.605858Z","shell.execute_reply.started":"2023-09-18T04:30:27.553825Z","shell.execute_reply":"2023-09-18T04:30:28.604591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create Morgan Bit Fingerprints","metadata":{}},{"cell_type":"code","source":"fp = np.array([smile_to_morgan_bitfp(s) for s in meta['SMILES'].unique()])\nfp = pd.DataFrame(fp, index= meta['sm_name'].unique())\nfp.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-18T04:30:33.116292Z","iopub.execute_input":"2023-09-18T04:30:33.116674Z","iopub.status.idle":"2023-09-18T04:30:33.366684Z","shell.execute_reply.started":"2023-09-18T04:30:33.116643Z","shell.execute_reply":"2023-09-18T04:30:33.365488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Perform t-SNE dimensionality reduction\ntsne = TSNE(n_components=2, random_state=0)\nreduced_data = tsne.fit_transform(fp)\n\nreduced_data = pd.DataFrame(reduced_data, index=fp.index)\nreduced_data = reduced_data.join(moa, how='left')","metadata":{"execution":{"iopub.status.busy":"2023-09-18T04:30:35.216382Z","iopub.execute_input":"2023-09-18T04:30:35.217213Z","iopub.status.idle":"2023-09-18T04:30:36.265863Z","shell.execute_reply.started":"2023-09-18T04:30:35.217164Z","shell.execute_reply":"2023-09-18T04:30:36.264953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get most common MOA\nval_counts = reduced_data['Mechanism_of_Action'].value_counts()\nfreq_moa = val_counts[val_counts>2].index\n\n# categorize for viz\nreduced_data['MOA'] = [i if i in freq_moa else 'other' for i in reduced_data['Mechanism_of_Action']]\n# reorder to plot colors last\nreduced_data = pd.concat((reduced_data[reduced_data['MOA']=='other'], reduced_data[reduced_data['MOA']!='other']))\nreduced_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-18T04:30:36.267771Z","iopub.execute_input":"2023-09-18T04:30:36.268724Z","iopub.status.idle":"2023-09-18T04:30:36.294209Z","shell.execute_reply.started":"2023-09-18T04:30:36.268681Z","shell.execute_reply":"2023-09-18T04:30:36.293023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the reduced data\nsns.scatterplot(x=0, y=1, data=reduced_data, hue='MOA')\n\n# Optional: Annotate points, add labels, color by some metric, etc.\n\nplt.xlabel('t-SNE 1')\nplt.ylabel('t-SNE 2')\nplt.title('Structural Diversity of Compounds')\nplt.legend(loc='upper left', bbox_to_anchor=(1, 1))\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-09-18T04:30:36.531784Z","iopub.execute_input":"2023-09-18T04:30:36.532165Z","iopub.status.idle":"2023-09-18T04:30:37.174662Z","shell.execute_reply.started":"2023-09-18T04:30:36.532136Z","shell.execute_reply":"2023-09-18T04:30:37.173501Z"},"trusted":true},"execution_count":null,"outputs":[]}]}