{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install rdkit-pypi openai langchain ","metadata":{"execution":{"iopub.status.busy":"2023-09-12T22:20:40.555386Z","iopub.execute_input":"2023-09-12T22:20:40.556686Z","iopub.status.idle":"2023-09-12T22:21:03.881963Z","shell.execute_reply.started":"2023-09-12T22:20:40.556646Z","shell.execute_reply":"2023-09-12T22:21:03.880200Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DISCLAIMER\n\n\nThis is an ongoing project, I will keep updating this notebook :)","metadata":{}},{"cell_type":"markdown","source":"# Libraries","metadata":{}},{"cell_type":"code","source":"### General ###\nimport os\nfrom typing import Optional\nfrom dotenv import load_dotenv\nfrom IPython.display import Markdown\n\n### Data Wrangling ###\nimport pandas as pd\n\n### Viz ###\nfrom colorama import Fore\nimport matplotlib.pyplot as plt\n\n### Chemoinformatics ###\nfrom rdkit import Chem\nfrom rdkit.Chem import Draw\nfrom rdkit.Chem import Descriptors\n\n### LLM ###\nimport openai\nfrom langchain.tools import tool\nfrom langchain.agents import AgentType\nfrom langchain.chat_models import ChatOpenAI\nfrom langchain.agents import initialize_agent\nfrom langchain.memory import ConversationBufferMemory\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-12T22:21:03.885459Z","iopub.execute_input":"2023-09-12T22:21:03.886028Z","iopub.status.idle":"2023-09-12T22:21:09.206022Z","shell.execute_reply.started":"2023-09-12T22:21:03.885972Z","shell.execute_reply":"2023-09-12T22:21:09.204713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"b_ = Fore.BLUE","metadata":{"execution":{"iopub.status.busy":"2023-09-12T22:21:09.207764Z","iopub.execute_input":"2023-09-12T22:21:09.208520Z","iopub.status.idle":"2023-09-12T22:21:09.216267Z","shell.execute_reply.started":"2023-09-12T22:21:09.208474Z","shell.execute_reply":"2023-09-12T22:21:09.214578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load_dotenv(\"/kaggle/input/langchaincoursedata/.env\")\n\nos.environ[\"OPENAI_API_KEY\"] = os.getenv(\"OPENAI_API_KEY\")\nopenai.api_key  = os.environ[\"OPENAI_API_KEY\"]","metadata":{"execution":{"iopub.status.busy":"2023-09-12T22:21:09.220283Z","iopub.execute_input":"2023-09-12T22:21:09.221598Z","iopub.status.idle":"2023-09-12T22:21:09.248741Z","shell.execute_reply.started":"2023-09-12T22:21:09.221544Z","shell.execute_reply":"2023-09-12T22:21:09.247211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the Data","metadata":{}},{"cell_type":"code","source":"de_train = pd.read_parquet(\"/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet\")\nde_train","metadata":{"execution":{"iopub.status.busy":"2023-09-12T22:21:09.251001Z","iopub.execute_input":"2023-09-12T22:21:09.251697Z","iopub.status.idle":"2023-09-12T22:21:12.788223Z","shell.execute_reply.started":"2023-09-12T22:21:09.251661Z","shell.execute_reply":"2023-09-12T22:21:12.787223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tools for the Agent","metadata":{}},{"cell_type":"code","source":"tools = []\nstorage_dict = {}","metadata":{"execution":{"iopub.status.busy":"2023-09-12T22:21:12.789529Z","iopub.execute_input":"2023-09-12T22:21:12.790083Z","iopub.status.idle":"2023-09-12T22:21:12.795430Z","shell.execute_reply.started":"2023-09-12T22:21:12.790036Z","shell.execute_reply":"2023-09-12T22:21:12.793475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@tools.append\n@tool\ndef draw_molecule(smiles_string):\n    \"\"\"Útil para dibujar moléculas basado en el formato SMILES\n    \"\"\"\n    mol = Chem.MolFromSmiles(smiles_string)\n    if mol:\n        img = Draw.MolToImage(mol)\n        plt.imshow(img)\n        plt.axis('off')  # Oculta los ejes\n        plt.show()\n        \n        return f\"La cadena smiles {smiles_string} ha sido dibujada\"\n    else:\n        return \"SMILES inválido\"","metadata":{"execution":{"iopub.status.busy":"2023-09-12T22:21:12.796929Z","iopub.execute_input":"2023-09-12T22:21:12.797356Z","iopub.status.idle":"2023-09-12T22:21:12.826393Z","shell.execute_reply.started":"2023-09-12T22:21:12.797322Z","shell.execute_reply":"2023-09-12T22:21:12.825430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@tools.append\n@tool\ndef calculate_descriptors(smiles_string):\n    \"\"\"\n    Útil para calcular descriptores moleculares a partir de SMILES\n    \"\"\"\n    mol = Chem.MolFromSmiles(smiles_string)\n    \n    if not mol:\n        return \"SMILES inválido\"\n    \n    # Calcula descriptores\n    mw = Descriptors.MolWt(mol)\n    logp = Descriptors.MolLogP(mol)\n    tpsa = Descriptors.TPSA(mol)\n    \n    descriptors = {\n        \"Molecular_Weight\": mw,\n        \"LogP\": logp,\n        \"TPSA\": tpsa\n    }\n\n    descriptor_name = \"descriptors_\" + smiles_string\n    storage_dict[descriptor_name] = descriptors\n    \n    return f\"Descriptores para {smiles_string} son {storage_dict[descriptor_name]}\"","metadata":{"execution":{"iopub.status.busy":"2023-09-12T22:21:12.827583Z","iopub.execute_input":"2023-09-12T22:21:12.830181Z","iopub.status.idle":"2023-09-12T22:21:12.849085Z","shell.execute_reply.started":"2023-09-12T22:21:12.830114Z","shell.execute_reply":"2023-09-12T22:21:12.847370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Agent","metadata":{}},{"cell_type":"code","source":"memory = ConversationBufferMemory(\n    memory_key = \"chat_history\", \n    return_messages = True\n)\nllm = ChatOpenAI(temperature = 0)\nagent = initialize_agent(\n    tools, \n    llm, \n    agent = AgentType.CHAT_CONVERSATIONAL_REACT_DESCRIPTION, \n    memory = memory\n)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T22:21:12.851111Z","iopub.execute_input":"2023-09-12T22:21:12.851497Z","iopub.status.idle":"2023-09-12T22:21:12.871142Z","shell.execute_reply.started":"2023-09-12T22:21:12.851465Z","shell.execute_reply":"2023-09-12T22:21:12.869427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Queries","metadata":{}},{"cell_type":"markdown","source":"### Querie 1","metadata":{}},{"cell_type":"code","source":"respuesta1 = agent.run(\n    f\"\"\"\n    1. Dibuja la secuencia SMILES siguiente: {de_train[\"SMILES\"][4]}\n    \"\"\"\n)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T22:21:12.875891Z","iopub.execute_input":"2023-09-12T22:21:12.876344Z","iopub.status.idle":"2023-09-12T22:21:17.641171Z","shell.execute_reply.started":"2023-09-12T22:21:12.876285Z","shell.execute_reply":"2023-09-12T22:21:17.639867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Querie 2","metadata":{}},{"cell_type":"code","source":"respuesta1 = agent.run(\n    f\"\"\"\n    1. Dibuja la secuencia SMILES siguiente: {de_train[\"SMILES\"][613]}\n    2. Después de dibujar esa molécula, calcula algunos descriptores moleculares\n    \"\"\"\n)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T22:21:17.643250Z","iopub.execute_input":"2023-09-12T22:21:17.643718Z","iopub.status.idle":"2023-09-12T22:21:24.144796Z","shell.execute_reply.started":"2023-09-12T22:21:17.643678Z","shell.execute_reply":"2023-09-12T22:21:24.143622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"respuesta2 = llm.predict(\n    f\"\"\"\n    1. Toma los descriptores moléculares que aparecen en {respuesta1} y dale formato de tabla\n    2. Extiende la tabla con una columna en la que expliques dichos descriptores moleculares brevemente, 15 palabras máximo\n    3. Traduce al inglés\n    \"\"\"\n)\nrespuesta3 = llm.predict(\n    f\"\"\"\n    Retorna solo la última tabla en inglés del siguiente texto: {respuesta2}\n    \"\"\"\n)\ndisplay(Markdown(respuesta3))","metadata":{"execution":{"iopub.status.busy":"2023-09-12T22:21:24.146832Z","iopub.execute_input":"2023-09-12T22:21:24.147965Z","iopub.status.idle":"2023-09-12T22:21:34.659149Z","shell.execute_reply.started":"2023-09-12T22:21:24.147921Z","shell.execute_reply":"2023-09-12T22:21:34.657333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Adding Chemical Descriptor","metadata":{}},{"cell_type":"code","source":"def calculate_mol_weight(smiles: str) -> Optional[float]:\n    mol = Chem.MolFromSmiles(smiles)\n    if mol is None:\n        return None  # or some other value to indicate that the SMILES was invalid\n    return Descriptors.MolWt(mol)\n\ndef calculate_mol_logp(smiles: str) -> Optional[float]:\n    mol = Chem.MolFromSmiles(smiles)\n    if mol is None:\n        return None  # or some other value to indicate that the SMILES was invalid\n    return Descriptors.MolLogP(mol)\n\ndef calculate_tpsa(smiles: str) -> Optional[float]:\n    mol = Chem.MolFromSmiles(smiles)\n    if mol is None:\n        return None  # or some other value to indicate that the SMILES was invalid\n    return Descriptors.TPSA(mol)","metadata":{"execution":{"iopub.status.busy":"2023-09-12T22:21:34.660779Z","iopub.execute_input":"2023-09-12T22:21:34.661856Z","iopub.status.idle":"2023-09-12T22:21:34.673172Z","shell.execute_reply.started":"2023-09-12T22:21:34.661806Z","shell.execute_reply":"2023-09-12T22:21:34.671208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train[\"MolWt\"] = de_train[\"SMILES\"].apply(calculate_mol_weight)\nde_train[\"MolLogP\"] = de_train[\"SMILES\"].apply(calculate_mol_logp)\nde_train[\"TPSA\"] = de_train[\"SMILES\"].apply(calculate_tpsa)\n\nde_train[[\"SMILES\", \"MolWt\", \"MolLogP\", \"TPSA\"]]","metadata":{"execution":{"iopub.status.busy":"2023-09-12T22:21:34.675284Z","iopub.execute_input":"2023-09-12T22:21:34.676966Z","iopub.status.idle":"2023-09-12T22:21:36.026835Z","shell.execute_reply.started":"2023-09-12T22:21:34.676916Z","shell.execute_reply":"2023-09-12T22:21:36.025663Z"},"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}