{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"},{"sourceId":8368538,"sourceType":"datasetVersion","datasetId":4974788}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# SmilesEnumerator: Augmentation for SMILES\n\n> SMILES enumeration is the process of writing out all possible SMILES forms of a molecule.   \n> It's a useful technique for data augmentation before sequence based modeling of molecules. \n\nOne of the way to address this competition is feeding SMILES into transformer or 1dcnn based models.  \nSMILES can take many forms per molecule.  \nIn this notebook, I introduce you SmiilesEnumerator, generating various form of SMILES from one SMILES.   \nYou can use it for augmentation of SMILES.    \n  \nIf you want more information, please visit (https://github.com/EBjerrum/SMILES-enumeration)\n\n### Please Upvote if you Find this Useful :)","metadata":{}},{"cell_type":"code","source":"!pip install rdkit -q","metadata":{"execution":{"iopub.status.busy":"2024-05-09T17:22:19.033833Z","iopub.execute_input":"2024-05-09T17:22:19.034233Z","iopub.status.idle":"2024-05-09T17:22:39.705528Z","shell.execute_reply.started":"2024-05-09T17:22:19.034203Z","shell.execute_reply":"2024-05-09T17:22:39.703766Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import library","metadata":{}},{"cell_type":"code","source":"import sys\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom rdkit import Chem\nfrom rdkit.Chem import Draw\n\nsys.path.append(\"/kaggle/input/smiles-enumerator\")\n\nfrom SmilesEnumerator import SmilesEnumerator","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-09T17:29:40.415125Z","iopub.execute_input":"2024-05-09T17:29:40.415512Z","iopub.status.idle":"2024-05-09T17:29:40.422972Z","shell.execute_reply.started":"2024-05-09T17:29:40.415481Z","shell.execute_reply":"2024-05-09T17:29:40.421605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load test data","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv(\"/kaggle/input/leash-BELKA/test.csv\")\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-09T17:23:03.372624Z","iopub.execute_input":"2024-05-09T17:23:03.373854Z","iopub.status.idle":"2024-05-09T17:23:12.093372Z","shell.execute_reply.started":"2024-05-09T17:23:03.373782Z","shell.execute_reply":"2024-05-09T17:23:12.091792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Choose example SMILES","metadata":{}},{"cell_type":"code","source":"example = test.loc[0, \"molecule_smiles\"]\nexample","metadata":{"execution":{"iopub.status.busy":"2024-05-09T17:24:00.083394Z","iopub.execute_input":"2024-05-09T17:24:00.083831Z","iopub.status.idle":"2024-05-09T17:24:00.096899Z","shell.execute_reply.started":"2024-05-09T17:24:00.083797Z","shell.execute_reply":"2024-05-09T17:24:00.095459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get augmented SMILES","metadata":{}},{"cell_type":"code","source":"sme = SmilesEnumerator()\nfor i in range(10):\n    example_randomized = sme.randomize_smiles(example)\n    print(example_randomized)","metadata":{"execution":{"iopub.status.busy":"2024-05-09T17:30:34.927996Z","iopub.execute_input":"2024-05-09T17:30:34.928908Z","iopub.status.idle":"2024-05-09T17:30:34.943189Z","shell.execute_reply.started":"2024-05-09T17:30:34.928860Z","shell.execute_reply":"2024-05-09T17:30:34.941989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check raw and augmented SMILES","metadata":{}},{"cell_type":"code","source":"mols = [Chem.MolFromSmiles(example), Chem.MolFromSmiles(example_randomized)]\nDraw.MolsToGridImage(mols, subImgSize=(500, 500), legends=[f\"Raw: {example}\", f\"Augmented: {example_randomized}\"])","metadata":{"execution":{"iopub.status.busy":"2024-05-09T17:39:11.952694Z","iopub.execute_input":"2024-05-09T17:39:11.954000Z","iopub.status.idle":"2024-05-09T17:39:12.024617Z","shell.execute_reply.started":"2024-05-09T17:39:11.953948Z","shell.execute_reply":"2024-05-09T17:39:12.023085Z"},"trusted":true},"execution_count":null,"outputs":[]}]}