{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Published on September 13, 2023. By Marília Prata, mpwolke","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-12T19:52:41.302086Z","iopub.execute_input":"2023-09-12T19:52:41.302739Z","iopub.status.idle":"2023-09-12T19:52:41.671717Z","shell.execute_reply.started":"2023-09-12T19:52:41.302708Z","shell.execute_reply":"2023-09-12T19:52:41.670854Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Competition Citation:\n\n@misc{open-problems-single-cell-perturbations,\n\n    author = {Andrew Benz, Ashley Chow, Avsecz, Daniel Burkhardt, HCL-Rantig, Ivana Jelic, \n    lancerunstats, Malte Luecken, Robrecht Cannoodt, Ryan Holbrook, Scott Gigante},\n    \n    title = {Open Problems – Single-Cell Perturbations},\n    \n    publisher = {Kaggle},\n    \n    year = {2023},\n    \n    url = {https://kaggle.com/competitions/open-problems-single-cell-perturbations}\n}","metadata":{}},{"cell_type":"markdown","source":"![](https://miro.medium.com/v2/resize:fit:306/1*Sdp5deTO43FQYuhFz2Y8zQ.png)https://sharifsuliman.medium.com/validating-smiles-with-rdkit-pysmiles-molvs-and-partialsmiles-5b65e800235f","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd\nimport os","metadata":{"execution":{"iopub.status.busy":"2023-09-14T00:42:01.834582Z","iopub.execute_input":"2023-09-14T00:42:01.835002Z","iopub.status.idle":"2023-09-14T00:42:02.298051Z","shell.execute_reply.started":"2023-09-14T00:42:01.834970Z","shell.execute_reply":"2023-09-14T00:42:02.296872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#No Module Perturbnet!\n\nhttps://github.com/welch-lab/PerturbNet/blob/main/examples/perturbnet_gi_example_notebook.ipynb","metadata":{}},{"cell_type":"code","source":"!pip install perturbnet","metadata":{"execution":{"iopub.status.busy":"2023-09-14T00:44:26.543155Z","iopub.execute_input":"2023-09-14T00:44:26.543646Z","iopub.status.idle":"2023-09-14T00:44:29.512058Z","shell.execute_reply.started":"2023-09-14T00:44:26.543613Z","shell.execute_reply":"2023-09-14T00:44:29.510115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from perturbnet.perturb.util import * \nfrom perturbnet.genetic_perturb.cinn.modules.flow import * \nfrom perturbnet.genetic_perturb.genotypevae.genotypeVAE import *\nfrom perturbnet.drug_perturb.cinn.modules.flow_generate import SCVIZ_CheckNet2Net","metadata":{"execution":{"iopub.status.busy":"2023-09-14T00:43:22.134176Z","iopub.execute_input":"2023-09-14T00:43:22.134693Z","iopub.status.idle":"2023-09-14T00:43:22.586543Z","shell.execute_reply.started":"2023-09-14T00:43:22.134646Z","shell.execute_reply":"2023-09-14T00:43:22.584614Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adata = pd.read_csv('../input/open-problems-single-cell-perturbations/adata_obs_meta.csv', encoding='utf8')\npd.set_option('display.max_columns', None)\nadata.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-14T00:42:09.969266Z","iopub.execute_input":"2023-09-14T00:42:09.969851Z","iopub.status.idle":"2023-09-14T00:42:11.620781Z","shell.execute_reply.started":"2023-09-14T00:42:09.969802Z","shell.execute_reply":"2023-09-14T00:42:11.619474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adata[\"cell_type\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-09-13T16:30:52.523812Z","iopub.execute_input":"2023-09-13T16:30:52.526059Z","iopub.status.idle":"2023-09-13T16:30:52.591768Z","shell.execute_reply.started":"2023-09-13T16:30:52.525960Z","shell.execute_reply":"2023-09-13T16:30:52.590047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Rob Mulla https://www.kaggle.com/code/robikscube/sign-language-recognition-eda-twitch-stream\n\nfig, ax = plt.subplots(figsize=(4, 4))\nadata[\"cell_type\"].value_counts().head(7).sort_values(ascending=True).plot(\n    kind=\"barh\", color='g', ax=ax, title=\"Cell Types\"\n)\nax.set_xlabel(\"Number of Training Examples\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-14T02:54:52.529227Z","iopub.execute_input":"2023-09-14T02:54:52.529791Z","iopub.status.idle":"2023-09-14T02:54:53.011747Z","shell.execute_reply.started":"2023-09-14T02:54:52.529725Z","shell.execute_reply":"2023-09-14T02:54:53.010369Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"obs = pd.read_csv('../input/open-problems-single-cell-perturbations/multiome_obs_meta.csv', encoding='utf8')\npd.set_option('display.max_columns', None)\nobs.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-13T15:41:20.597799Z","iopub.execute_input":"2023-09-13T15:41:20.598306Z","iopub.status.idle":"2023-09-13T15:41:20.666062Z","shell.execute_reply.started":"2023-09-13T15:41:20.598262Z","shell.execute_reply":"2023-09-13T15:41:20.664639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"idm = pd.read_csv('../input/open-problems-single-cell-perturbations/id_map.csv', encoding='utf8')\npd.set_option('display.max_columns', None)\nidm.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-13T15:41:28.382781Z","iopub.execute_input":"2023-09-13T15:41:28.383157Z","iopub.status.idle":"2023-09-13T15:41:28.401887Z","shell.execute_reply.started":"2023-09-13T15:41:28.383128Z","shell.execute_reply":"2023-09-13T15:41:28.401000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#SMILES Algorithm\n\nCitation: SMILES. 2. Algorithm for generation of unique SMILES notation\n\nDavid Weininger, Arthur Weininger, and Joseph L. Weininger\n\nJournal of Chemical Information and Computer Sciences 1989 29 (2), 97-101 - DOI: 10.1021/ci00062a008\n\n\"The SMILES chemical notation language was introduced in the first paper of this series.’ Processing chemical information with greater efficiency than conventional methods, it represents a new approach to computerized chemical nomenclature. SMILES is simple to write because rules and hierarchical procedures, which are inherently difficult for the chemist, are relegated to computer algorithms. For a given \nchemical structure, arbitrary SMILES notation can take many equally valid forms. One must emerge as “unique” to serve as the identifier of the structure for database and other computer applications.\"\n\n\"This is accomplished by a method called CANGEN that combines two separate algorithms, CANON and GENES. \nThe first stage, CANON, labels a molecular structure with canonical labels. The structure is treated as a graph with nodes (atoms) and edges (bonds).\"\n\n\"Each atom is given a numerical label on the basis of its topology. In the second stage, GENES \ngenerates the unique SMILES notation as a tree representation of the molecular graph. GENES selects the starting atom and makes branching decisions by referring to the canonical labels as needed.\"\n\n\"The combined procedure designates a unique SMILES notation for each chemical structure regardless of the many possible equivalent descriptions of the structure that might be input.\" \n\n\"The CANGEN process consists of a two-stage algorithm. The first stage involves canonicalization of structure, whereby the molecular structure is treated as a graph with nodes (atoms) and edges (bonds). All atoms are canonically ranked on the basis of a suitable set of invariant node properties and \nare labeled numerically.\"\n\n\"In the second stage, starting with the lowest ranked atom, a tree (molecular graph) is constructed that is the unique SMILES notation regardless of which of various valid original linear SMILES notations was originally specified. The generation of unique SMILES by this process provides the key to solving the basic problem of chemical nomenclature, namely, that a single chemical compound may have many different names. When a unique notation for a structure is obtained, chemical nomenclature is amenable to many applications for databases where SMILES can be associated with any number of synonyms, identifiers, and structure keys, such as common names, Collective Index names, IUPAC names, and CAS numbers.\"\n\n\"A unique SMILES notation serves extremely well as an identifier for a chemical database. This will be illustrated in one of the next publications in this series which describes a SMILES-oriented, extremely efficient database.\"\n\nhttp://organica1.org/seminario/smile_2_1988.pdf","metadata":{}},{"cell_type":"code","source":"var = pd.read_csv('../input/open-problems-single-cell-perturbations/multiome_var_meta.csv', encoding='utf8')\npd.set_option('display.max_columns', None)\nvar.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-13T15:41:34.874472Z","iopub.execute_input":"2023-09-13T15:41:34.874884Z","iopub.status.idle":"2023-09-13T15:41:35.453427Z","shell.execute_reply.started":"2023-09-13T15:41:34.874851Z","shell.execute_reply":"2023-09-13T15:41:35.452358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Parquet files - Reading them without Dask","metadata":{}},{"cell_type":"code","source":"multiome_train = pd.read_parquet('../input/open-problems-single-cell-perturbations/multiome_train.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-09-13T15:54:11.397300Z","iopub.execute_input":"2023-09-13T15:54:11.397727Z","iopub.status.idle":"2023-09-13T15:55:33.636424Z","shell.execute_reply.started":"2023-09-13T15:54:11.397697Z","shell.execute_reply":"2023-09-13T15:55:33.635225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"multiome_train. head()","metadata":{"execution":{"iopub.status.busy":"2023-09-13T15:55:43.914693Z","iopub.execute_input":"2023-09-13T15:55:43.916089Z","iopub.status.idle":"2023-09-13T15:55:43.932359Z","shell.execute_reply.started":"2023-09-13T15:55:43.916048Z","shell.execute_reply":"2023-09-13T15:55:43.931259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adata_train = pd.read_parquet('../input/open-problems-single-cell-perturbations/multiome_train.parquet')\nadata_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-13T15:58:37.832627Z","iopub.execute_input":"2023-09-13T15:58:37.833055Z","iopub.status.idle":"2023-09-13T15:59:51.200062Z","shell.execute_reply.started":"2023-09-13T15:58:37.833026Z","shell.execute_reply":"2023-09-13T15:59:51.198737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train = pd.read_parquet('../input/open-problems-single-cell-perturbations/de_train.parquet')\nde_train.tail()","metadata":{"execution":{"iopub.status.busy":"2023-09-13T16:48:40.153254Z","iopub.execute_input":"2023-09-13T16:48:40.153789Z","iopub.status.idle":"2023-09-13T16:48:56.412901Z","shell.execute_reply.started":"2023-09-13T16:48:40.153752Z","shell.execute_reply":"2023-09-13T16:48:56.411810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Riociguat sm_name\n\nde_train.iloc[612,3]","metadata":{"execution":{"iopub.status.busy":"2023-09-13T16:50:51.633510Z","iopub.execute_input":"2023-09-13T16:50:51.634222Z","iopub.status.idle":"2023-09-13T16:50:51.644495Z","shell.execute_reply.started":"2023-09-13T16:50:51.634173Z","shell.execute_reply":"2023-09-13T16:50:51.641968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train.iloc[600,3]","metadata":{"execution":{"iopub.status.busy":"2023-09-13T16:52:16.980745Z","iopub.execute_input":"2023-09-13T16:52:16.981182Z","iopub.status.idle":"2023-09-13T16:52:16.990196Z","shell.execute_reply.started":"2023-09-13T16:52:16.981152Z","shell.execute_reply":"2023-09-13T16:52:16.989180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#SMILES (Simplified Molecular Input Line Entry System) \n\nWhat is SMILES?\n\n\"SMILES (Simplified Molecular Input Line Entry System) is a chemical notation that allows a user to represent a chemical structure in a way that can be used by the computer. SMILES is an easily learned and flexible notation. The SMILES notation requires that you learn a handful of rules. You do not need to worry about ambiguous representations because the software will automatically reorder your entry into a unique SMILES string when necessary.\"\n\n\"SMILES was developed through funding from the U.S. Environmental Protection Agency, Mid-Continent Ecology Division-Duluth, (MED-Duluth) Duluth, MN to the Medicinal Chemistry Project at Pomona College, Claremont, CA and the Computer Sciences Corporation, Duluth, MN. Several publications discuss SMILES in more detail, including Anderson et al. 1987, Weininger 1988, Weininger et al. 1989, and Hunter et al., 1987.\"\n\n\"SMILES has five basic syntax rules which must be observed. If basic rules of chemistry are not followed in SMILES entry, the system will warn the user and ask that the structure be edited or reentered. For example, if the user places too many bonds on an atom, a SMILES warning will appear that the structure is impossible. The rules are described below and some examples are provided.\"\n\nhttps://archive.epa.gov/med/med_archive_03/web/html/smiles.html","metadata":{}},{"cell_type":"code","source":"!pip install rdkit","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-09-14T00:57:25.919733Z","iopub.execute_input":"2023-09-14T00:57:25.920760Z","iopub.status.idle":"2023-09-14T00:57:43.499914Z","shell.execute_reply.started":"2023-09-14T00:57:25.920703Z","shell.execute_reply":"2023-09-14T00:57:43.498084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Meer Atif https://www.kaggle.com/code/meeratif/smiles-open-problems\n\nfrom rdkit import Chem\nfrom rdkit.Chem import Draw","metadata":{"execution":{"iopub.status.busy":"2023-09-14T00:57:47.647517Z","iopub.execute_input":"2023-09-14T00:57:47.648057Z","iopub.status.idle":"2023-09-14T00:57:47.771551Z","shell.execute_reply.started":"2023-09-14T00:57:47.648011Z","shell.execute_reply":"2023-09-14T00:57:47.770208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I can Not say if below is an aromatic or a kekulized SMILE.","metadata":{}},{"cell_type":"code","source":"#https://mattermodeling.stackexchange.com/questions/6460/rdkit-and-pysmiles-results-differ-on-some-smiles-strings\n#Answered by RapelPy, Jul 31, 2021\n\nfrom rdkit import Chem\nfrom rdkit.Chem.Draw import IPythonConsole\nfrom rdkit.Chem import Draw\nimport pysmiles\n\ns1 = 'O=C(/C=C/c1cccnc1)NCCCCC1CCN(C(=O)c2ccccc2)CC1'  # aromatic\ns2 = 'CCCCOc1c(C(=O)c2c(F)cc(C)cc2F)cnc2[nH]ncc12'  # kekulized\n\n\ndef show_implicit_h(smiles):\n    m = Chem.MolFromSmiles(smiles)\n    for atom in m.GetAtoms():\n        atom.SetProp('atomLabel', str(atom.GetIdx()))\n    m = Chem.AddHs(m)\n    return Draw.MolToImage(m, size=(300, 300))\n\n\nshow_implicit_h(s1)","metadata":{"execution":{"iopub.status.busy":"2023-09-14T01:58:34.627569Z","iopub.execute_input":"2023-09-14T01:58:34.628400Z","iopub.status.idle":"2023-09-14T01:58:34.668260Z","shell.execute_reply.started":"2023-09-14T01:58:34.628361Z","shell.execute_reply":"2023-09-14T01:58:34.667025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Aromaticity and Kekulization \n\n\"Counting the hydrogen on both smiles shows that pysmiles finds hydrogen in the aromatic, but not in the kekulized SMILES. RDKit finds none in both.\"\n\nColumns: Index, Symbol, hcount pysmiles, hcount RDKit\n\n\"Aromaticity must be detected in a system that generates an unambiguous chemical nomenclature. As will be discussed in following papers, this is needed both for the generation of a unique nomenclature and for effective substructure recognition. There can be no definition of \"aromaticity\" that is both rigorous and all encompassing; the word implies something about \"reactivity\" to a synthetic chemist, \"ring current\" to a NMR spectroscopist, \"symmetry\" to a crystallographer, and presumably \"odor\" to the original user of the word. Our objective in defining aromaticity is to provide an automatic and rigorous definition for the purposes of generating an unambiguous chemical nomenclature. Although the SMILES algorithm produces results that most chemists find natural, nothing is implied by this definition about physical properties.\"\n\n\"The most important application of the DS for SMILES readers is kekulization. Kekulization is the process of assigning double bonds to a molecular graph using the DS as a guide. Kekulization occurs before the assignment of virtual hydrogens.\"\n\n\"The problem of assigning alternating double bonds is just one instance of a broader problem in graph theory known as matching. A matching is a subgraph in which each node has degree one. Three kinds of matching are of interest: maximal (no additional edges can be added); maximum (all possible edges have been added); and perfect (all nodes have been added).\"\n\nhttps://depth-first.com/articles/2020/02/10/a-comprehensive-treatment-of-aromaticity-in-the-smiles-language/","metadata":{}},{"cell_type":"code","source":"#https://mattermodeling.stackexchange.com/questions/6460/rdkit-and-pysmiles-results-differ-on-some-smiles-strings\n\nshow_implicit_h(s2)","metadata":{"execution":{"iopub.status.busy":"2023-09-14T01:59:32.509572Z","iopub.execute_input":"2023-09-14T01:59:32.510064Z","iopub.status.idle":"2023-09-14T01:59:32.542214Z","shell.execute_reply.started":"2023-09-14T01:59:32.510029Z","shell.execute_reply":"2023-09-14T01:59:32.540553Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Rule One: Atoms and Bonds\n\n\"SMILES supports all elements in the periodic table. An atom is represented using its respective atomic symbol. Upper case letters refer to non-aromatic atoms; lower case letters refer to aromatic atoms. If the atomic symbol has more than one letter the second letter must be lower case.\"\n\nBonds are denoted as shown below:\n\n-Single bond\n=Double bond\n#Triple bond\n*Aromatic bond\n.Disconnected structures\n\n\"Single bonds are the default and therefore need not be entered. For example, 'CC' would mean that there is a non-aromatic carbon attached to another non-aromatic carbon by a single bond, and the computer would identify the structure as the chemical ethane. It is also assumed that the bond between two lower case atom symbols is aromatic. A blank terminates the SMILES string.\"\n\nhttps://archive.epa.gov/med/med_archive_03/web/html/smiles.html","metadata":{}},{"cell_type":"code","source":"#By ButtonWood https://mattermodeling.stackexchange.com/questions/6460/rdkit-and-pysmiles-results-differ-on-some-smiles-strings\n\nfrom rdkit import Chem\nfrom rdkit.Chem.Draw import IPythonConsole\nfrom rdkit.Chem import Draw\nfrom rdkit.Chem import rdDepictor\nfrom rdkit.Chem.Draw import rdMolDraw2D\n\nfrom IPython.display import SVG\nIPythonConsole.ipython_useSVG=True  \n\n\ndef mol_with_atom_index(mol):\n    for atom in mol.GetAtoms():\n        atom.SetAtomMapNum(atom.GetIdx())\n    return mol\n\n\nmol = Chem.MolFromSmiles(\"O=C(/C=C/c1cccnc1)NCCCCC1CCN(C(=O)c2ccccc2)CC1\")\nmol = mol_with_atom_index(mol)\nmc = Chem.Mol(mol.ToBinary())\n\n\ndrawer = rdMolDraw2D.MolDraw2DSVG(450, 200) \ndrawer.DrawMolecule(mc)\ndrawer.FinishDrawing()\n\nsvg = drawer.GetDrawingText()\ndisplay(SVG(svg.replace('svg:','')))","metadata":{"execution":{"iopub.status.busy":"2023-09-14T02:03:53.592516Z","iopub.execute_input":"2023-09-14T02:03:53.593087Z","iopub.status.idle":"2023-09-14T02:03:53.637164Z","shell.execute_reply.started":"2023-09-14T02:03:53.593049Z","shell.execute_reply":"2023-09-14T02:03:53.635694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By ButtonWood https://mattermodeling.stackexchange.com/questions/6460/rdkit-and-pysmiles-results-differ-on-some-smiles-strings\n\nfrom pysmiles import read_smiles\n\nsmiles = \"O=C(/C=C/c1cccnc1)NCCCCC1CCN(C(=O)c2ccccc2)CC1\"\nmolecule = read_smiles(smiles)\n\npysmiles_list = zip(molecule.nodes(data=\"SMILES\"), molecule.nodes(data=\"hcount\"))#Don't change hcount\nfor element in pysmiles_list:\n    print(element)","metadata":{"execution":{"iopub.status.busy":"2023-09-14T02:07:18.598576Z","iopub.execute_input":"2023-09-14T02:07:18.599068Z","iopub.status.idle":"2023-09-14T02:07:18.611410Z","shell.execute_reply.started":"2023-09-14T02:07:18.599036Z","shell.execute_reply":"2023-09-14T02:07:18.610198Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\"A similar listing of atom index, atom type, and number of hydrogens bond to the non-H may be achieved iterating with RDKit over the non-H atoms\"","metadata":{}},{"cell_type":"code","source":"#By ButtonWood https://mattermodeling.stackexchange.com/questions/6460/rdkit-and-pysmiles-results-differ-on-some-smiles-strings\n\nfor atom in mol.GetAtoms():\n    print(\"{:2} {:2} {}\".format(atom.GetIdx(), atom.GetSymbol(),\n                                atom.GetTotalNumHs()))","metadata":{"execution":{"iopub.status.busy":"2023-09-14T02:08:09.688611Z","iopub.execute_input":"2023-09-14T02:08:09.689128Z","iopub.status.idle":"2023-09-14T02:08:09.697131Z","shell.execute_reply.started":"2023-09-14T02:08:09.689092Z","shell.execute_reply":"2023-09-14T02:08:09.695840Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train['SMILES'][5]","metadata":{"execution":{"iopub.status.busy":"2023-09-13T16:38:46.118023Z","iopub.execute_input":"2023-09-13T16:38:46.118508Z","iopub.status.idle":"2023-09-13T16:38:46.131664Z","shell.execute_reply.started":"2023-09-13T16:38:46.118472Z","shell.execute_reply":"2023-09-13T16:38:46.129978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Meer Atif https://www.kaggle.com/code/meeratif/smiles-open-problems\n\n# Your SMILES string\nsmiles = \"C[C@@H]1C[C@H]2[C@@H]3CCC4=CC(=O)C=C[C@]4(C)[C@@]3(Cl)[C@@H](O)C[C@]2(C)[C@@]1(OC(=O)c1ccco1)C(=O)CCl\"\n\n# Convert the SMILES string to an RDKit molecule object\nmol = Chem.MolFromSmiles(smiles)\n\n# Check if the conversion was successful\nif mol is not None:\n    # Generate a 2D depiction of the molecule\n    img = Draw.MolToImage(mol)    \n    # Save the image to a file\n#     img.save(\"molecule.png\")\nelse:\n    print(\"Invalid SMILES string\")","metadata":{"execution":{"iopub.status.busy":"2023-09-13T16:41:30.053823Z","iopub.execute_input":"2023-09-13T16:41:30.054601Z","iopub.status.idle":"2023-09-13T16:41:30.101819Z","shell.execute_reply.started":"2023-09-13T16:41:30.054364Z","shell.execute_reply":"2023-09-13T16:41:30.099967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Rule Two: Simple Chains\n\n\"By combining atomic symbols and bond symbols simple chain structures can be represented. The structures that are entered using SMILES are hydrogen-suppressed, that is to say that the molecules are represented without hydrogens. The SMILES software understands the number of possible connections that an atom can have. If enough bonds are not identified by the user through SMILES notation, the system will automatically assume that the other connections are satisfied by hydrogen bonds.\"\n\nSome examples:\n\nCC CH3CH3 Ethane\n\nC=C CH2CH2 Ethene\n\nCBr CH3Br Bromomethane\n\nC#N C=N Hydrocyanic acid\n\nNa.Cl  NaCl  Sodium chloride\n\n\"The user can explicitly identify the hydrogen bonds, but if one hydrogen bond is identified in the string, the SMILES interpreter will assume that the user has identified all hydrogens for that molecule.\"\n\nHC(H)=C(H)(H) Ethene\n\n\"Because SMILES allows entry of all elements in the periodic table, and also utilizes hydrogen suppression, the user should be aware of chemicals with two letters that could be misinterpreted by the computer. For example, 'Sc' could be interpreted as a sulfur atom connected to an aromatic carbon by a single bond, or it could be the symbol for scandium. The SMILES interpreter gives priority to the interpretation of a single bond connecting a sulfur atom and an aromatic carbon. To identify scandium the user should enter [Sc].\"\n\nhttps://archive.epa.gov/med/med_archive_03/web/html/smiles.html","metadata":{}},{"cell_type":"code","source":"#By Meer Atif https://www.kaggle.com/code/meeratif/smiles-open-problems\n\nimg","metadata":{"execution":{"iopub.status.busy":"2023-09-13T16:41:48.125467Z","iopub.execute_input":"2023-09-13T16:41:48.126050Z","iopub.status.idle":"2023-09-13T16:41:48.151072Z","shell.execute_reply.started":"2023-09-13T16:41:48.126004Z","shell.execute_reply":"2023-09-13T16:41:48.149107Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It's strange since I chose a different SMILE from Meer Atif, however the image above is similar to what he made on his Notebook.\n\nhttps://www.rdkit.org/docs/GettingStartedInPython.html\n\nhttps://github.com/rdkit/rdkit/discussions/5769","metadata":{}},{"cell_type":"markdown","source":"#PySmiles","metadata":{}},{"cell_type":"code","source":"!pip install pysmiles","metadata":{"execution":{"iopub.status.busy":"2023-09-14T01:11:30.564655Z","iopub.execute_input":"2023-09-14T01:11:30.565217Z","iopub.status.idle":"2023-09-14T01:11:47.347935Z","shell.execute_reply.started":"2023-09-14T01:11:30.565165Z","shell.execute_reply":"2023-09-14T01:11:47.346714Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Rule Three: Branches\n\n\"A branch from a chain is specified by placing the SMILES symbol(s) for the branch between parenthesis. The string in parentheses is placed directly after the symbol for the atom to which it is connected. If it is connected by a double or triple bond, the bond symbol immediately follows the left parenthesis. Some examples:\"\n\nCC(O)C 2-Propanol\n\nCC(=O)C 2-Propanone\n\nCC(CC)C 2-Methylbutane\n\nCC(C)CC(=O) 2-Methylbutanal\n\nc1c(N(=O)=O)cccc1 Nitrobenzene\n\nCC(C)(C)CC 2,2-Dimethylbutane\n\nhttps://archive.epa.gov/med/med_archive_03/web/html/smiles.html","metadata":{}},{"cell_type":"code","source":"#https://stackoverflow.com/questions/57062757/how-to-generate-a-graph-from-a-smiles-molecule-representation\n#By Davide Fiocco answered Jul 16, 2019\n\nfrom pysmiles import read_smiles\nimport networkx as nx\n    \nsmiles = 'O=C(/C=C/c1cccnc1)NCCCCC1CCN(C(=O)c2ccccc2)CC1'\nmol = read_smiles(smiles)\n    \n# atom vector (C only)\nprint(mol.nodes(data='SMILES'))#Don't change data\n# adjacency matrix\nprint(nx.to_numpy_matrix(mol))","metadata":{"execution":{"iopub.status.busy":"2023-09-14T01:12:25.939500Z","iopub.execute_input":"2023-09-14T01:12:25.939994Z","iopub.status.idle":"2023-09-14T01:12:25.959644Z","shell.execute_reply.started":"2023-09-14T01:12:25.939960Z","shell.execute_reply":"2023-09-14T01:12:25.958456Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Rule Four: Rings\n\n\"SMILES allows a user to identify ring structures by using numbers to identify the opening and closing ring atom. For example, in C1CCCCC1, the first carbon has a number '1' which connects by a single bond with the last carbon which also has a number '1'. The resulting structure is cyclohexane. Chemicals that have multiple rings may be identified by using different numbers for each ring. If a double, single, or aromatic bond is used for the ring closure, the bond symbol is placed before the ring closure number. Some examples:\"\n\nor C=1CCCCC1   Cyclohexene\n\nC*1*C*C*C*C*C1\n\nc1ccccc1    Benzene\n\nC1OC1CC Ethyloxirane\n\nc1cc2ccccc2cc1 Naphthalene\n\nhttps://archive.epa.gov/med/med_archive_03/web/html/smiles.html","metadata":{}},{"cell_type":"code","source":"#By Davide Fiocco https://stackoverflow.com/questions/57062757/how-to-generate-a-graph-from-a-smiles-molecule-representation\n\nimport matplotlib.pyplot as plt\nelements = nx.get_node_attributes(mol, name = \"SMILES\")\nnx.draw(mol, with_labels=True, labels = elements, pos=nx.spring_layout(mol))\nplt.gca().set_aspect('equal')","metadata":{"execution":{"iopub.status.busy":"2023-09-14T01:14:47.769795Z","iopub.execute_input":"2023-09-14T01:14:47.770267Z","iopub.status.idle":"2023-09-14T01:14:48.052496Z","shell.execute_reply.started":"2023-09-14T01:14:47.770236Z","shell.execute_reply":"2023-09-14T01:14:48.051280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Let's check if I can make a different chart. With another Smile.","metadata":{}},{"cell_type":"code","source":"#By Davide Fiocco https://stackoverflow.com/questions/57062757/how-to-generate-a-graph-from-a-smiles-molecule-representation\n\nsmiles1 = 'CCCCOc1c(C(=O)c2c(F)cc(C)cc2F)cnc2[nH]ncc12'\nmol = read_smiles(smiles1)\n    \n# atom vector (C only)\nprint(mol.nodes(data='SMILES'))#Don't change data\n# adjacency matrix\nprint(nx.to_numpy_matrix(mol))","metadata":{"execution":{"iopub.status.busy":"2023-09-14T01:27:13.724188Z","iopub.execute_input":"2023-09-14T01:27:13.724705Z","iopub.status.idle":"2023-09-14T01:27:13.742052Z","shell.execute_reply.started":"2023-09-14T01:27:13.724667Z","shell.execute_reply":"2023-09-14T01:27:13.740719Z"},"_kg_hide-output":true,"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Rule Five: Charged Atoms\n\n\"Charges on an atom can be used to override the knowledge regarding valence that is built into SMILES software. The format for identifying a charged atom consists of the atom followed by brackets which enclose the charge on the atom. The number of charges may be explicitly stated ({-1}) or not ({-}). For example:\"\n\n \n\nor CCC(=O)O{-1} Ionized form of propanoic acid\n\nCCC(=O)O{-}\n\nc1ccccn{+1}1CC(=O)O 1-Carboxylmethyl pyridinium\n\nhttps://archive.epa.gov/med/med_archive_03/web/html/smiles.html","metadata":{}},{"cell_type":"code","source":"#By Davide Fiocco https://stackoverflow.com/questions/57062757/how-to-generate-a-graph-from-a-smiles-molecule-representation\n\nelements = nx.get_node_attributes(mol, name = \"SMILES\")\nnx.draw(mol, with_labels=True, labels = elements, pos=nx.spring_layout(mol))\nplt.gca().set_aspect('equal')","metadata":{"execution":{"iopub.status.busy":"2023-09-14T01:28:45.069047Z","iopub.execute_input":"2023-09-14T01:28:45.069569Z","iopub.status.idle":"2023-09-14T01:28:45.303627Z","shell.execute_reply.started":"2023-09-14T01:28:45.069510Z","shell.execute_reply":"2023-09-14T01:28:45.302122Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Interpretable single-cell perturbation modeling using a compositional perturbation autoencoder (CPA)\n\n\"CPA is a framework to learn effects of perturbations at the single-cell level. CPA encodes and learns phenotypic drug response across different cell types, doses and drug combinations. CPA allows:\"\n\n\"Out-of-distribution predicitons of unseen drug combinations at various doses and among different cell types.\"\n\n\"Learn interpretable drug and cell type latent spaces.\"\n\n\"Estimate dose response curve for each perturbation and their combinations.\"\n\n\"Access the uncertainty of the estimations of the model.\"\n\n![](https://github.com/facebookresearch/CPA/raw/main/Figure1.png)https://github.com/facebookresearch/CPA","metadata":{}},{"cell_type":"markdown","source":"Now that Perturbation is starting to make a sense to me.","metadata":{}},{"cell_type":"markdown","source":"#Acknowledgements:\n\nU.S. Environmental Agency (Mid-Continent Ecology Division)\nhttps://archive.epa.gov/med/med_archive_03/web/html/smiles.html\n\nMeer Atif https://www.kaggle.com/code/meeratif/smiles-open-problems","metadata":{}}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}