{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kernel_info":{"name":"python310-sdkv2"},"microsoft":{"host":{"AzureML":{"notebookHasBeenCompleted":true}}},"nteract":{"version":"nteract-front-end@1.0.0"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"},{"sourceId":8042988,"sourceType":"datasetVersion","datasetId":4740586}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Outline\nTo explore the chemical space of both train and test molecules and gain insights, I utilized [Mordred](https://github.com/mordred-descriptor/mordred)  to compute over 1,000 molecular descriptors. Subsequently, I reduced these high-dimensional vectors using PCA and t-SNE.","metadata":{}},{"cell_type":"code","source":"!pip install duckdb mapply rdkit mordred","metadata":{"nteract":{"transient":{"deleting":false}},"gather":{"logged":1717381410240},"execution":{"iopub.status.busy":"2024-06-03T12:59:24.464703Z","iopub.execute_input":"2024-06-03T12:59:24.465245Z","iopub.status.idle":"2024-06-03T12:59:40.394378Z","shell.execute_reply.started":"2024-06-03T12:59:24.465202Z","shell.execute_reply":"2024-06-03T12:59:40.392789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\nimport duckdb\nimport pandas as pd\nimport numpy as np\nfrom mordred import Calculator, descriptors\n\nSEED = 427\ndatadir = Path('/kaggle/input/leash-BELKA')\nbelka_shrunken_dir = Path('/kaggle/input/belka-shrunken-train-set')","metadata":{"gather":{"logged":1717381419307},"execution":{"iopub.status.busy":"2024-06-03T14:04:36.311307Z","iopub.execute_input":"2024-06-03T14:04:36.311849Z","iopub.status.idle":"2024-06-03T14:04:36.318689Z","shell.execute_reply.started":"2024-06-03T14:04:36.311811Z","shell.execute_reply":"2024-06-03T14:04:36.317531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import mapply\nmapply.init(\n    n_workers=-1,\n    progressbar=True,\n)","metadata":{"nteract":{"transient":{"deleting":false}},"gather":{"logged":1717381419410},"execution":{"iopub.status.busy":"2024-06-03T12:59:43.803120Z","iopub.execute_input":"2024-06-03T12:59:43.803675Z","iopub.status.idle":"2024-06-03T12:59:43.920515Z","shell.execute_reply.started":"2024-06-03T12:59:43.803641Z","shell.execute_reply":"2024-06-03T12:59:43.919411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import AllChem\nDy = Chem.MolFromSmiles('[Dy]')\nmethyl = Chem.MolFromSmiles('C')\ntriazine = Chem.MolFromSmiles('C1=NC=NC=N1')\n\n\ndef replace_linker(smiles, newgroup=methyl):\n    mol = Chem.MolFromSmiles(smiles)\n    newmol = AllChem.ReplaceSubstructs(mol, Dy, newgroup)[0]\n    Chem.SanitizeMol(newmol)\n    return newmol\n\n\ndef from_smiles_to_3dmol(smiles, seed=SEED):\n    mol = replace_linker(smiles)\n    molh = Chem.AddHs(mol)\n    AllChem.EmbedMolecule(molh, randomSeed=seed)\n    try:\n        AllChem.MMFFOptimizeMolecule(molh, maxIters=500, nonBondedThresh=200.0)\n    except:\n        pass\n    return molh\n\n\ndef check_for_triazine(x):\n    # https://www.kaggle.com/code/chemdatafarmer/scaffold-exploration\n    check = Chem.MolFromSmiles(x).HasSubstructMatch(triazine)\n    return check","metadata":{"nteract":{"transient":{"deleting":false}},"gather":{"logged":1717381419911},"execution":{"iopub.status.busy":"2024-06-03T12:59:43.923067Z","iopub.execute_input":"2024-06-03T12:59:43.923638Z","iopub.status.idle":"2024-06-03T12:59:44.154608Z","shell.execute_reply.started":"2024-06-03T12:59:43.923603Z","shell.execute_reply":"2024-06-03T12:59:44.153445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preliminary Analysis\nI generated a small dataset comprising 1,000 rows and computed its Mordred values to identify relevant descriptors.","metadata":{}},{"cell_type":"code","source":"try:\n    train1k = pd.read_pickle('train1k.df.pcl.zip')\nexcept FileNotFoundError:\n    train_path = datadir/'train.parquet'\n    con = duckdb.connect()\n    N = 500\n    train1k = (con.query(f\"\"\"(SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1\n                        ORDER BY random()\n                        LIMIT {N})\n                        UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0\n                        ORDER BY random()\n                        LIMIT {N})\"\"\")\n               .df()\n               .drop(['buildingblock1_smiles', 'buildingblock2_smiles',\n                      'buildingblock3_smiles'], axis=1)\n               .set_index('id')\n               .sort_index()\n               )\n    con.close()\n    train1k.to_pickle('train1k.df.pcl.zip')","metadata":{"nteract":{"transient":{"deleting":false}},"gather":{"logged":1717381445526},"execution":{"iopub.status.busy":"2024-06-03T12:59:44.156011Z","iopub.execute_input":"2024-06-03T12:59:44.156431Z","iopub.status.idle":"2024-06-03T12:59:44.173491Z","shell.execute_reply.started":"2024-06-03T12:59:44.156396Z","shell.execute_reply":"2024-06-03T12:59:44.170719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    mordred_train1k = pd.read_pickle('mordred_train1k.df.pcl.zip')\nexcept FileNotFoundError:\n    calc = Calculator(descriptors, ignore_3D=False)\n    mols = train1k.molecule_smiles.mapply(from_smiles_to_3dmol)\n    mordred_train1k = calc.pandas(mols).select_dtypes('number')\n    mordred_train1k.to_pickle('mordred_train1k.df.pcl.zip')","metadata":{"nteract":{"transient":{"deleting":false}},"gather":{"logged":1717381806653},"execution":{"iopub.status.busy":"2024-06-03T12:59:44.175397Z","iopub.execute_input":"2024-06-03T12:59:44.175915Z","iopub.status.idle":"2024-06-03T12:59:44.334772Z","shell.execute_reply.started":"2024-06-03T12:59:44.175869Z","shell.execute_reply":"2024-06-03T12:59:44.333232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Out of the 1,488 Mordred descriptors, 159 had a single unique value across the dataset and were thus considered uninformative for further analyses. I proceeded with the remaining 1,329 descriptors in subsequent analyses.","metadata":{}},{"cell_type":"code","source":"calc = Calculator(descriptors, ignore_3D=False)\nn_unique = mordred_train1k.nunique()\nusecols = list(n_unique[n_unique > 1].index)\nmy_descs = []\nfor i, desc in enumerate(calc.descriptors):\n    if desc.__str__() in usecols:\n        my_descs.append(desc)\nprint(f'#columns:{mordred_train1k.shape[1]}')\nprint(f'#columns with 2 or more values: {len(usecols)}')","metadata":{"execution":{"iopub.status.busy":"2024-06-03T14:15:09.953549Z","iopub.execute_input":"2024-06-03T14:15:09.954160Z","iopub.status.idle":"2024-06-03T14:15:10.246890Z","shell.execute_reply.started":"2024-06-03T14:15:09.954111Z","shell.execute_reply":"2024-06-03T14:15:10.245549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfig, ax = plt.subplots()\nn_unique.hist(ax=ax)\nax.set_xlabel('Number of unique values')\nax.set_title('Mordred descriptors')","metadata":{"jupyter":{"source_hidden":false,"outputs_hidden":false},"nteract":{"transient":{"deleting":false}},"gather":{"logged":1716809098928},"collapsed":false,"execution":{"iopub.status.busy":"2024-06-03T14:09:55.845682Z","iopub.execute_input":"2024-06-03T14:09:55.847208Z","iopub.status.idle":"2024-06-03T14:09:56.577643Z","shell.execute_reply.started":"2024-06-03T14:09:55.847092Z","shell.execute_reply":"2024-06-03T14:09:56.576360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Analyses\nI sampled the following groups of molecules for my analyses.\n- ~1,000 train molecules each positive for one of the three proteins\n- 3,000 train molecules negative for all the three proteins\n- 1,000 test molecules with the triazine core and shared or non-shared building blocks each\n- 1,000 test molecules with non-triazine cores and non-shared building blocks","metadata":{}},{"cell_type":"code","source":"def get_mordred(smiles_ser, descs=descriptors):\n    calc = Calculator(descs)\n    mols = smiles_ser.mapply(from_smiles_to_3dmol)\n    return calc.pandas(mols).select_dtypes('number')","metadata":{"jupyter":{"source_hidden":false,"outputs_hidden":false},"nteract":{"transient":{"deleting":false}},"gather":{"logged":1717374468223},"collapsed":false,"execution":{"iopub.status.busy":"2024-06-03T12:59:45.066896Z","iopub.execute_input":"2024-06-03T12:59:45.067411Z","iopub.status.idle":"2024-06-03T12:59:45.074528Z","shell.execute_reply.started":"2024-06-03T12:59:45.067365Z","shell.execute_reply":"2024-06-03T12:59:45.073201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nbelka_shrunken_dir = Path('/kaggle/input/belka-shrunken-train-set')\nBB1train = pickle.load(\n    open(belka_shrunken_dir/'train_dicts/BBs_dict_1.p', 'br'))\nBB2train = pickle.load(\n    open(belka_shrunken_dir/'train_dicts/BBs_dict_2.p', 'br'))\nBB3train = pickle.load(\n    open(belka_shrunken_dir/'train_dicts/BBs_dict_3.p', 'br'))\nBB1test = pickle.load(\n    open(belka_shrunken_dir/'test_dicts/BBs_dict_1_test.p', 'br'))\nBB2test = pickle.load(\n    open(belka_shrunken_dir/'test_dicts/BBs_dict_2_test.p', 'br'))\nBB3test = pickle.load(\n    open(belka_shrunken_dir/'test_dicts/BBs_dict_3_test.p', 'br'))","metadata":{"jupyter":{"source_hidden":false,"outputs_hidden":false},"nteract":{"transient":{"deleting":false}},"gather":{"logged":1717374470567},"collapsed":false,"execution":{"iopub.status.busy":"2024-06-03T12:59:45.076060Z","iopub.execute_input":"2024-06-03T12:59:45.076596Z","iopub.status.idle":"2024-06-03T12:59:45.144214Z","shell.execute_reply.started":"2024-06-03T12:59:45.076560Z","shell.execute_reply":"2024-06-03T12:59:45.143062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N = 1000\ntry:\n    train = pd.read_pickle('train.df.pcl.zip')\nexcept FileNotFoundError:\n    train_path = datadir/'train.parquet'\n    protein_names=['BRD4','HSA','sEH']\n    con = duckdb.connect()\n    selected_ids = (con.query(\"\"\"UNION ALL\n            \"\"\".join([f\"\"\"(SELECT *\n            FROM parquet_scan('{train_path}')\n            WHERE protein_name = '{protein_name}' AND binds = 1\n            ORDER BY random()\n            LIMIT {N})\n            \"\"\" for protein_name in protein_names])+\n            f\"\"\"\n            UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE protein_name = 'sEH' AND binds = 1\n                        ORDER BY random()\n                        LIMIT {N})\n                        UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0\n                        ORDER BY random()\n                        LIMIT {3*N})\"\"\")\n                    .df()\n                    ['id']\n                    )\n\n    con.close()\n    userows = sorted(\n        set(np.hstack([3*(selected_ids.values//3) + k for k in range(3)])))\n\n    train = (pd.read_parquet(train_path,\n                             columns=['id', 'molecule_smiles',\n                                      'protein_name', 'binds'],\n                             filters=[('id', 'in', userows)])\n             .astype({'binds': 'boolean'})\n             .assign(**{'id': lambda df: 3*(df.id//3)})\n             .pivot(index=['id', 'molecule_smiles'], columns='protein_name', values='binds')\n             .reset_index(level=1))\n    train.to_pickle('train.df.pcl.zip')","metadata":{"execution":{"iopub.status.busy":"2024-06-03T14:08:58.269182Z","iopub.execute_input":"2024-06-03T14:08:58.270384Z","iopub.status.idle":"2024-06-03T14:08:58.291823Z","shell.execute_reply.started":"2024-06-03T14:08:58.270337Z","shell.execute_reply":"2024-06-03T14:08:58.289962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    test = pd.read_pickle('test.df.pcl.zip')\nexcept FileNotFoundError:\n    trainBBs = set(BB1train.keys()).union(\n        BB2train.keys()).union(BB3train.keys())\n    test = (pd.read_parquet(datadir/'test.parquet')\n            .drop_duplicates(subset=['molecule_smiles'])\n            .assign(BB1shared=lambda df: df.buildingblock1_smiles.isin(trainBBs),\n                    BB2shared=lambda df: df.buildingblock2_smiles.isin(\n                        trainBBs),\n                    BB3shared=lambda df: df.buildingblock3_smiles.isin(\n                        trainBBs),\n                    triazine=lambda df: df.molecule_smiles.mapply(check_for_triazine))\n            .drop(['buildingblock1_smiles', 'buildingblock2_smiles',\n                   'buildingblock3_smiles', 'protein_name'],\n                  axis=1))\n    test.to_pickle('test.df.pcl.zip')","metadata":{"nteract":{"transient":{"deleting":false}},"gather":{"logged":1717374674584},"execution":{"iopub.status.busy":"2024-06-03T15:03:34.074658Z","iopub.execute_input":"2024-06-03T15:03:34.075250Z","iopub.status.idle":"2024-06-03T15:03:35.537718Z","shell.execute_reply.started":"2024-06-03T15:03:34.075205Z","shell.execute_reply":"2024-06-03T15:03:35.536186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Test molecules breakdown\nIn each test molecule, the three building blocks are either shared or not shared at all. Also, all the molecules with non-traizine cores have non-shared building blocks.","metadata":{}},{"cell_type":"code","source":"breakdown = test.groupby(\n    ['BB1shared', 'BB2shared', 'BB3shared', 'triazine']).size().sort_values()\nbreakdown","metadata":{"execution":{"iopub.status.busy":"2024-06-03T13:02:19.087215Z","iopub.execute_input":"2024-06-03T13:02:19.089491Z","iopub.status.idle":"2024-06-03T13:02:19.179852Z","shell.execute_reply.started":"2024-06-03T13:02:19.089437Z","shell.execute_reply":"2024-06-03T13:02:19.178628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nfig, ax = plt.subplots()\nax.pie(breakdown,\n       labels=['non-shared, triazine', 'shared, triazine',\n               'non-shared, non-triazine'],\n       startangle=90, colors=sns.color_palette('colorblind'))\nax.set_title('Test dataset')","metadata":{"execution":{"iopub.status.busy":"2024-06-03T13:02:19.181229Z","iopub.execute_input":"2024-06-03T13:02:19.181768Z","iopub.status.idle":"2024-06-03T13:02:20.615465Z","shell.execute_reply.started":"2024-06-03T13:02:19.181735Z","shell.execute_reply":"2024-06-03T13:02:20.614175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shared_triazine = test[test.BB1shared & test.triazine].sample(\n    n=N, random_state=SEED).sort_index()\nnonshared_triazine = test[~test.BB1shared & test.triazine].sample(\n    n=N, random_state=SEED).sort_index()\nnonshared_nontriazine = test[~test.BB1shared & ~test.triazine].sample(\n    n=N, random_state=SEED).sort_index()","metadata":{"jupyter":{"source_hidden":false,"outputs_hidden":false},"nteract":{"transient":{"deleting":false}},"gather":{"logged":1717375248870},"collapsed":false,"execution":{"iopub.status.busy":"2024-06-03T13:02:20.617672Z","iopub.execute_input":"2024-06-03T13:02:20.618355Z","iopub.status.idle":"2024-06-03T13:02:20.764637Z","shell.execute_reply.started":"2024-06-03T13:02:20.618303Z","shell.execute_reply":"2024-06-03T13:02:20.763308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"names = ['train', 'shared_triazine',\n         'nonshared_triazine', 'nonshared_nontriazine']\nfor name in names:\n    try:\n        exec(f'mordred_{name}=pd.read_pickle(\"mordred_{name}.df.pcl.zip\")')\n    except FileNotFoundError:\n        exec(\n            f'mordred_{name}=get_mordred({name}.molecule_smiles,descs=my_descs)')\n        exec(f'mordred_{name}.to_pickle(\"mordred_{name}.df.pcl.zip\")')","metadata":{"jupyter":{"source_hidden":false,"outputs_hidden":false},"nteract":{"transient":{"deleting":false}},"gather":{"logged":1717150405708},"collapsed":false,"execution":{"iopub.status.busy":"2024-06-03T13:02:20.767130Z","iopub.execute_input":"2024-06-03T13:02:20.768161Z","iopub.status.idle":"2024-06-03T13:02:21.876707Z","shell.execute_reply.started":"2024-06-03T13:02:20.768113Z","shell.execute_reply":"2024-06-03T13:02:21.875194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"commoncols = list(set(mordred_train.columns)\n                  .intersection(mordred_shared_triazine.columns)\n                  .intersection(mordred_nonshared_triazine.columns)\n                  .intersection(mordred_nonshared_nontriazine.columns))","metadata":{"execution":{"iopub.status.busy":"2024-06-03T13:02:21.878501Z","iopub.execute_input":"2024-06-03T13:02:21.879082Z","iopub.status.idle":"2024-06-03T13:02:21.886761Z","shell.execute_reply.started":"2024-06-03T13:02:21.879031Z","shell.execute_reply":"2024-06-03T13:02:21.885303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mordred = (pd.concat([mordred_train.assign(group='train'),\n                      mordred_shared_triazine.assign(group='shared, triazine'),\n                      mordred_nonshared_triazine.assign(\n    group='non-shared, tirazine'),\n    mordred_nonshared_nontriazine.assign(group='non-shared, non-triazine')])\n    [commoncols+['group']]\n    .sample(frac=1, random_state=SEED)\n)\ngroups = ['train', 'shared, triazine', 'non-shared, tirazine',\n          'non-shared, non-triazine']","metadata":{"execution":{"iopub.status.busy":"2024-06-03T13:02:21.888460Z","iopub.execute_input":"2024-06-03T13:02:21.888830Z","iopub.status.idle":"2024-06-03T13:02:22.080859Z","shell.execute_reply.started":"2024-06-03T13:02:21.888801Z","shell.execute_reply":"2024-06-03T13:02:22.079531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### PCA","metadata":{}},{"cell_type":"markdown","source":"#### Train molecules\nWhen examining the molecules positive for each protein, distinct patterns emerged in the principal component analysis (PCA).\n","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nscaler = StandardScaler()\nn_components = 2\npca = PCA(n_components=n_components)\nX = (pd.DataFrame(pca.fit_transform(scaler.fit_transform(mordred.drop('group', axis=1))),\n                  index=mordred.index,\n                  columns=[f'pc{k+1}' for k in range(n_components)])\n     .merge(mordred.group, left_index=True, right_index=True)\n     )","metadata":{"jupyter":{"source_hidden":false,"outputs_hidden":false},"nteract":{"transient":{"deleting":false}},"gather":{"logged":1717375108603},"collapsed":false,"execution":{"iopub.status.busy":"2024-06-03T13:02:22.082629Z","iopub.execute_input":"2024-06-03T13:02:22.083022Z","iopub.status.idle":"2024-06-03T13:02:23.473278Z","shell.execute_reply.started":"2024-06-03T13:02:22.082989Z","shell.execute_reply":"2024-06-03T13:02:23.472003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom matplotlib import rcParams\nrcParams.update({'legend.frameon': False,\n                'legend.edgecolor': 'none'})\nbincolors = [sns.color_palette('colorblind')[k] for k in [7,3]]\nfig = plt.figure(figsize=(9, 3))\nfor i, protein in enumerate(['BRD4', 'HSA', 'sEH']):\n    ax = fig.add_subplot(1, 3, i+1)\n    sns.scatterplot(ax=ax, data=X.loc[train.index], x='pc1',\n                    y='pc2', s=7, hue=train[protein].astype('int'), alpha=.5)\nfig.suptitle('Train molecules')\nplt.tight_layout()","metadata":{"jupyter":{"source_hidden":false,"outputs_hidden":false},"nteract":{"transient":{"deleting":false}},"gather":{"logged":1717244314561},"collapsed":false,"execution":{"iopub.status.busy":"2024-06-03T13:13:13.205189Z","iopub.execute_input":"2024-06-03T13:13:13.206489Z","iopub.status.idle":"2024-06-03T13:13:15.396363Z","shell.execute_reply.started":"2024-06-03T13:13:13.206439Z","shell.execute_reply":"2024-06-03T13:13:15.395224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Train vs. test molecues\nOverall, these protein groups did not exhibit any discernible distribution patterns in the figure","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots()\ncolors = [sns.color_palette('colorblind')[k] for k in [0, 1, 2, 4]]\nsns.scatterplot(ax=ax, data=X, x='pc1', y='pc2', hue='group',\n                hue_order=groups, s=7, palette=colors, alpha=.5)","metadata":{"execution":{"iopub.status.busy":"2024-06-03T13:02:25.807209Z","iopub.execute_input":"2024-06-03T13:02:25.807567Z","iopub.status.idle":"2024-06-03T13:02:26.915822Z","shell.execute_reply.started":"2024-06-03T13:02:25.807540Z","shell.execute_reply":"2024-06-03T13:02:26.914868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So, I created separate plots for each group of test molecules. Those containing the triazine core overlapped significantly with the train molecules, while those with non-triazine cores were predominantly located in the lower part.","metadata":{}},{"cell_type":"code","source":"fig = plt.figure(figsize=(9, 3))\nfor i, group in enumerate(groups[1:]):\n    ax = fig.add_subplot(1, 3, i+1)\n    sns.scatterplot(ax=ax, data=X[lambda df:df.group.isin(['train', group])],\n                    x='pc1', y='pc2', hue='group', hue_order=['train', group], s=7,\n                    palette=[colors[k] for k in [0, groups.index(group)]],\n                    alpha=.5)\n    legend = ax.legend(loc='upper left')\n    ax.set_xlim(X.pc1.min()-5, X.pc1.max()+5)\n    ax.set_ylim(X.pc2.min()-5, X.pc2.max()+20)\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2024-06-03T13:02:26.917220Z","iopub.execute_input":"2024-06-03T13:02:26.917757Z","iopub.status.idle":"2024-06-03T13:02:28.453778Z","shell.execute_reply.started":"2024-06-03T13:02:26.917725Z","shell.execute_reply":"2024-06-03T13:02:28.452003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### t-SNE","metadata":{}},{"cell_type":"code","source":"from sklearn.manifold import TSNE\ntsne = TSNE(n_components=2, random_state=SEED)\nXtsne = (pd.DataFrame(\n    tsne.fit_transform(scaler.fit_transform(mordred.drop('group', axis=1))),\n    index=mordred.index,\n    columns=['tsne1', 'tsne2'])\n    .merge(mordred.group, left_index=True, right_index=True))","metadata":{"jupyter":{"source_hidden":false,"outputs_hidden":false},"nteract":{"transient":{"deleting":false}},"gather":{"logged":1717244314573},"collapsed":false,"execution":{"iopub.status.busy":"2024-06-03T13:02:28.455667Z","iopub.execute_input":"2024-06-03T13:02:28.456089Z","iopub.status.idle":"2024-06-03T13:03:21.648576Z","shell.execute_reply.started":"2024-06-03T13:02:28.456053Z","shell.execute_reply":"2024-06-03T13:03:21.647423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Train molecules","metadata":{}},{"cell_type":"code","source":"rcParams.update({'legend.frameon': True})\nfig = plt.figure(figsize=(9, 3))\nfor i, protein in enumerate(['BRD4', 'HSA', 'sEH']):\n    ax = fig.add_subplot(1, 3, i+1)\n    sns.scatterplot(ax=ax, data=Xtsne.loc[train.index], x='tsne1',\n                    y='tsne2', s=7, hue=train[protein].astype('int'), alpha=.5)\nfig.suptitle('Binding molecules')\nplt.tight_layout()","metadata":{"jupyter":{"source_hidden":false,"outputs_hidden":false},"nteract":{"transient":{"deleting":false}},"gather":{"logged":1717244314584},"collapsed":false,"execution":{"iopub.status.busy":"2024-06-03T13:03:21.650531Z","iopub.execute_input":"2024-06-03T13:03:21.651082Z","iopub.status.idle":"2024-06-03T13:03:23.908609Z","shell.execute_reply.started":"2024-06-03T13:03:21.651034Z","shell.execute_reply":"2024-06-03T13:03:23.906758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Train vs. test molecues\nWait a minute! Test molecules with non-shared BB and non-triazine cores exhibited distinct characteristics compared to both the train molecules and the test molecules in the other groups.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots()\nsns.scatterplot(ax=ax, data=Xtsne, x='tsne1', y='tsne2', hue='group',\n                hue_order=groups, s=7, palette=colors, alpha=.5)\nlegend = ax.legend(loc='upper left')","metadata":{"jupyter":{"source_hidden":false,"outputs_hidden":false},"nteract":{"transient":{"deleting":false}},"gather":{"logged":1717244314595},"collapsed":false,"execution":{"iopub.status.busy":"2024-06-03T13:03:23.909889Z","iopub.execute_input":"2024-06-03T13:03:23.910683Z","iopub.status.idle":"2024-06-03T13:03:24.615990Z","shell.execute_reply.started":"2024-06-03T13:03:23.910646Z","shell.execute_reply":"2024-06-03T13:03:24.614659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rcParams.update({'legend.frameon': False})\nfig = plt.figure(figsize=(9, 3))\nfor i, group in enumerate(groups[1:]):\n    ax = fig.add_subplot(1, 3, i+1)\n    sns.scatterplot(ax=ax, data=Xtsne[lambda df:df.group.isin(['train', group])],\n                    x='tsne1', y='tsne2', hue='group', hue_order=['train', group], s=7,\n                    palette=[colors[k] for k in [0, groups.index(group)]],\n                    alpha=.5)\n    legend = ax.legend(loc='upper left')\n    ax.set_xlim(Xtsne.tsne1.min()-5, Xtsne.tsne1.max()+5)\n    ax.set_ylim(Xtsne.tsne2.min()-5, Xtsne.tsne2.max()+50)\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2024-06-03T13:10:01.158023Z","iopub.execute_input":"2024-06-03T13:10:01.158861Z","iopub.status.idle":"2024-06-03T13:10:02.848045Z","shell.execute_reply.started":"2024-06-03T13:10:01.158824Z","shell.execute_reply":"2024-06-03T13:10:02.846550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Conclusion\nIn the chemical space, molecules binding to each protein exhibited distinct patterns, which is promising. However, test molecules with non-triazine cores appeared isolated from other molecules, particularly in t-SNE. This observation may explain why they pose significant challenges.","metadata":{}}]}