{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"},{"sourceId":8042988,"sourceType":"datasetVersion","datasetId":4740586},{"sourceId":11371,"sourceType":"modelInstanceVersion","modelInstanceId":5171},{"sourceId":26140,"sourceType":"modelInstanceVersion","modelInstanceId":22003}],"dockerImageVersionId":30699,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport time\nt0start = time.time() \n\nimport pickle\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport warnings\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-25T17:41:45.483361Z","iopub.execute_input":"2024-04-25T17:41:45.484297Z","iopub.status.idle":"2024-04-25T17:41:46.450797Z","shell.execute_reply.started":"2024-04-25T17:41:45.484259Z","shell.execute_reply":"2024-04-25T17:41:46.449893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Dataset\ndata_train = pd.read_csv('/kaggle/input/leash-BELKA/train.csv', nrows = 2)\ndata_train","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:41:46.452396Z","iopub.execute_input":"2024-04-25T17:41:46.452802Z","iopub.status.idle":"2024-04-25T17:41:46.491383Z","shell.execute_reply.started":"2024-04-25T17:41:46.452776Z","shell.execute_reply":"2024-04-25T17:41:46.490466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualizing first 5 data\ndata_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:41:46.492867Z","iopub.execute_input":"2024-04-25T17:41:46.493240Z","iopub.status.idle":"2024-04-25T17:41:46.505022Z","shell.execute_reply.started":"2024-04-25T17:41:46.493208Z","shell.execute_reply":"2024-04-25T17:41:46.504161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Info data\ndata_train.info()","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:41:46.507466Z","iopub.execute_input":"2024-04-25T17:41:46.508185Z","iopub.status.idle":"2024-04-25T17:41:46.533350Z","shell.execute_reply.started":"2024-04-25T17:41:46.508151Z","shell.execute_reply":"2024-04-25T17:41:46.532512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data type\ndata_train.dtypes","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:41:46.534432Z","iopub.execute_input":"2024-04-25T17:41:46.534716Z","iopub.status.idle":"2024-04-25T17:41:46.546723Z","shell.execute_reply.started":"2024-04-25T17:41:46.534693Z","shell.execute_reply":"2024-04-25T17:41:46.545881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Loads the BBs (Building Blocks) dictionaries from three files using the pickle library. These dictionaries are stored in the files 'BBs_dict_reverse_1.p', 'BBs_dict_reverse_2.p', and 'BBs_dict_reverse_3.p'. \n# The loaded dictionaries are stored in the variables BBs_dict_reverse_1, BBs_dict_reverse_2, and BBs_dict_reverse_3, respectively.\n\n# Defines a list named data_org that contains the values corresponding to the first 1000 items of the column buildingblock3_smiles from the data_train DataFrame. \n# These values are obtained by accessing the BBs_dict_reverse_3 dictionary using the keys specified in the buildingblock3_smiles column.\ndtypes_data = {'buildingblock1_smiles': np.int16,\n               'buildingblock2_smiles': np.int16,\n               'buildingblock3_smiles': np.int16,\n               'binds_BRD4':np.byte,\n               'binds_HSA':np.byte,\n               'binds_sEH':np.byte}","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:41:46.547848Z","iopub.execute_input":"2024-04-25T17:41:46.548372Z","iopub.status.idle":"2024-04-25T17:41:46.557980Z","shell.execute_reply.started":"2024-04-25T17:41:46.548340Z","shell.execute_reply":"2024-04-25T17:41:46.557201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# New dataset\ndata_train = pd.read_csv('/kaggle/input/belka-shrunken-train-set/train.csv', dtype = dtypes_data,  nrows = 10000 )\ndata_train","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:41:46.559167Z","iopub.execute_input":"2024-04-25T17:41:46.559541Z","iopub.status.idle":"2024-04-25T17:41:46.613779Z","shell.execute_reply.started":"2024-04-25T17:41:46.559510Z","shell.execute_reply":"2024-04-25T17:41:46.612972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualizing first 5 data\ndata_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:41:46.614778Z","iopub.execute_input":"2024-04-25T17:41:46.615000Z","iopub.status.idle":"2024-04-25T17:41:46.624907Z","shell.execute_reply.started":"2024-04-25T17:41:46.614980Z","shell.execute_reply":"2024-04-25T17:41:46.623955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20.5, 10))\nsns.countplot(x=\"buildingblock2_smiles\", data=data_train)\nplt.title(\"Certain molecules\")\nplt.ylabel(\"Total\")\nplt.xlabel(\"Molecules\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:41:46.625931Z","iopub.execute_input":"2024-04-25T17:41:46.626213Z","iopub.status.idle":"2024-04-25T17:41:47.074858Z","shell.execute_reply.started":"2024-04-25T17:41:46.626190Z","shell.execute_reply":"2024-04-25T17:41:47.073938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the subplots array\nfig, axes = plt.subplots(1, 3, figsize=(15, 5))\n\n# Plot 1: binds_BRD4\nsns.countplot(x=\"binds_BRD4\", data=data_train, ax=axes[0])\naxes[0].set_title('binds_BRD4')\n\n# Plot 2: binds_HSA\nsns.countplot(x=\"binds_HSA\", data=data_train, ax=axes[1])\naxes[1].set_title('binds_HSA')\n\n# Plot 3: binds_sEH\nsns.countplot(x=\"binds_sEH\", data=data_train, ax=axes[2])\naxes[2].set_title('binds_sEH')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:41:47.077999Z","iopub.execute_input":"2024-04-25T17:41:47.078297Z","iopub.status.idle":"2024-04-25T17:41:47.628722Z","shell.execute_reply.started":"2024-04-25T17:41:47.078272Z","shell.execute_reply":"2024-04-25T17:41:47.627767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot countplot\nplt.figure(figsize=(10, 6))\nsns.countplot(y=\"molecule_smiles\", data=data_train, order=data_train['molecule_smiles'].value_counts().index[:10])\nplt.title('Top 10 Molecule SMILES Count')\nplt.xlabel('Count')\nplt.ylabel('Molecule SMILES')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:41:47.630337Z","iopub.execute_input":"2024-04-25T17:41:47.630679Z","iopub.status.idle":"2024-04-25T17:41:48.037079Z","shell.execute_reply.started":"2024-04-25T17:41:47.630645Z","shell.execute_reply":"2024-04-25T17:41:48.036198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # plot pie\nplt.figure(figsize=(8, 8))\ndata_train['molecule_smiles'].value_counts().head(5).plot(kind='pie', autopct='%1.1f%%')\nplt.title('Top 5 Molecule SMILES')\nplt.ylabel('')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:41:48.038219Z","iopub.execute_input":"2024-04-25T17:41:48.038483Z","iopub.status.idle":"2024-04-25T17:41:48.343865Z","shell.execute_reply.started":"2024-04-25T17:41:48.038458Z","shell.execute_reply":"2024-04-25T17:41:48.342984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# This code snippet loads three BBs (Building Blocks) dictionaries from three different files and then creates a new list named data_org. \n# This list is composed of elements from the BBs dictionaries, where the keys are obtained from the first 1000 elements of the buildingblock3_smiles column of \n# the data_train DataFrame.\nBBs_dict_reverse_1 = pickle.load(open('/kaggle/input/belka-shrunken-train-set/train_dicts/BBs_dict_reverse_1.p', 'br'))\nBBs_dict_reverse_2 = pickle.load(open('/kaggle/input/belka-shrunken-train-set/train_dicts/BBs_dict_reverse_2.p', 'br'))\nBBs_dict_reverse_3 = pickle.load(open('/kaggle/input/belka-shrunken-train-set/train_dicts/BBs_dict_reverse_3.p', 'br'))\n\n# Viewing dataframe\ndata_org = [BBs_dict_reverse_3[x] for x in data_train.buildingblock3_smiles[:1000]]","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:41:48.345414Z","iopub.execute_input":"2024-04-25T17:41:48.346005Z","iopub.status.idle":"2024-04-25T17:41:48.364709Z","shell.execute_reply.started":"2024-04-25T17:41:48.345969Z","shell.execute_reply":"2024-04-25T17:41:48.363911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Model LLM - ChemBERTa**","metadata":{}},{"cell_type":"code","source":"# Import libers\nimport torch\nfrom transformers import AutoModelForMaskedLM, AutoTokenizer","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:41:48.365769Z","iopub.execute_input":"2024-04-25T17:41:48.366023Z","iopub.status.idle":"2024-04-25T17:41:53.728557Z","shell.execute_reply.started":"2024-04-25T17:41:48.366000Z","shell.execute_reply":"2024-04-25T17:41:53.727778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tokenizer\ntokenizer = AutoTokenizer.from_pretrained(\"DeepChem/ChemBERTa-77M-MTR\")","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:41:53.729782Z","iopub.execute_input":"2024-04-25T17:41:53.730692Z","iopub.status.idle":"2024-04-25T17:41:55.286900Z","shell.execute_reply.started":"2024-04-25T17:41:53.730653Z","shell.execute_reply":"2024-04-25T17:41:55.286076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model\nmodel_ChemBERTa = AutoModelForMaskedLM.from_pretrained(\"DeepChem/ChemBERTa-77M-MTR\")","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:41:55.288170Z","iopub.execute_input":"2024-04-25T17:41:55.288444Z","iopub.status.idle":"2024-04-25T17:41:57.147133Z","shell.execute_reply.started":"2024-04-25T17:41:55.288419Z","shell.execute_reply":"2024-04-25T17:41:57.146424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom tqdm import tqdm\n\n# Defines a value N as 1 and extracts a sample from the molecule_smiles of the first entry in the training dataset (data_train).\nN = 1\ntrain_data = data_train['molecule_smiles'].iat[0]#[:30]\n\n# Uses a Language Model (LLM) to compute embeddings (numerical representations) for the training input. \n# It utilizes the ChemBERTa model with the Hugging Face transformers library for this purpose. \n# It calculates two types of embeddings: one using only the special token [CLS] and another by averaging the embeddings of all tokens in the sequence.\nwith torch.no_grad():\n    model_llm_padding = True\n    model_llm_encoded_input = tokenizer(train_data, return_tensors=\"pt\",padding=model_llm_padding,truncation=True)\n    model_llm_output = model_ChemBERTa(**model_llm_encoded_input)\n    model_llm_embedding = model_llm_output[0][:,0,:]\n    model_llm_embeddings_cls = model_llm_embedding\n    model_llm_embedding = torch.mean(model_llm_output[0],1)\n    model_llm_embeddings_mean = model_llm_embedding\n    \n# Defines a function model_feat that takes a list of SMILES and computes embeddings for each SMILES using the ChemBERTa model. \n# This function iterates over the list of SMILES, computing embeddings for each individual SMILES. \n# Again, two types of embeddings are calculated \n# one using only the special token [CLS] and another by averaging the embeddings of all tokens in the sequence.\nmodel_ChemBERTa.eval()\n\n# Loads a training sample consisting of the first N items from the molecule_smiles column of the data_train DataFrame.\ndef model_feat(smiles_list, padding=True):\n    model_llm_embeddings_cls = torch.zeros(len(smiles_list), 600)\n    model_llm_embeddings_mean = torch.zeros(len(smiles_list), 600)\n\n    # Utilizes the model_feat function to compute embeddings for the loaded training sample.\n    with torch.no_grad():\n        for x, train in enumerate(tqdm(smiles_list)):\n            model_llm_encoded_input2 = tokenizer(train, return_tensors=\"pt\", padding=padding, truncation=True)\n            model_llm_output = model_ChemBERTa(**model_llm_encoded_input2)\n            model_llm_embedding = model_llm_output[0][:, 0, :]\n            model_llm_embeddings_cls[x] = model_llm_embedding\n            model_llm_embedding = torch.mean(model_llm_output[0], 1)\n            model_llm_embeddings_mean[x] = model_llm_embedding\n            \n    return model_llm_embeddings_cls.numpy(), model_llm_embeddings_mean.numpy()\n\n# Train\nN = int(1e4)\ntrain = data_train['molecule_smiles'].iloc[:N].to_list()\n\n# Fit Model\nx_train_model, y_train_model_mean = model_feat(train)","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:41:57.148708Z","iopub.execute_input":"2024-04-25T17:41:57.149337Z","iopub.status.idle":"2024-04-25T17:43:13.651275Z","shell.execute_reply.started":"2024-04-25T17:41:57.149301Z","shell.execute_reply":"2024-04-25T17:43:13.650313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Viewing cluster","metadata":{}},{"cell_type":"code","source":"# Imports the necessary libraries, including TSNE from scikit-learn for t-SNE (t-distributed stochastic neighbor embedding) \n# and umap for UMAP (Uniform Manifold Approximation and Projection).\nfrom sklearn.manifold import TSNE\nimport umap\n\n# Initializes a UMAP model.\numap_model = TSNE()\numap_model = umap.UMAP()\n\n# Computes UMAP projections using the input embeddings x_train_model.\numap_result = umap_model.fit_transform(x_train_model)\n\n# Iterates over the datasets x_train_model and y_train_model_mean (which seem to contain input and output embeddings, respectively).a. \n# For each dataset, computes UMAP projections.b. Plots a scatter plot of the UMAP projections.c. \n# Computes the correlation between the dimensions of the UMAP projections and displays the correlation DataFrame.d. \n# For each specified column ('binds_BRD4', 'binds_HSA', 'binds_sEH') in the original DataFrame data_train, plots a scatter plot of the UMAP projections colored according to the values of that column.\nfor x in[x_train_model, y_train_model_mean]:\n    y = umap_model.fit_transform(x)\n    sns.scatterplot(x=y[:,0], y=y[:,1])\n    plt.show()\n    df = pd.DataFrame(y)\n    df = df.reset_index()\n    display(df.corr())\n    \n    for x_col in ['binds_BRD4', 'binds_HSA', 'binds_sEH']:\n        sns.scatterplot(x = y[:,0], y=y[:,1], hue=data_train[x_col])\n        plt.title(\"UMAP - LLM ChemBERTa\", fontsize=20)\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:47:11.687797Z","iopub.execute_input":"2024-04-25T17:47:11.688685Z","iopub.status.idle":"2024-04-25T17:47:51.498379Z","shell.execute_reply.started":"2024-04-25T17:47:11.688645Z","shell.execute_reply":"2024-04-25T17:47:51.497347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Imports the TSNE class from the manifold module of the scikit-learn library.\nfrom sklearn.manifold import TSNE\n\n# Initializes a TSNE model.\ntsne_model = TSNE()\n\n# Computes t-SNE projections using the input embeddings x_train_model.\ntsne_result = tsne_model.fit_transform(x_train_model)\n\n# Iterates over the datasets x_train_model and y_train_model_mean (which seem to contain input and output embeddings, respectively).a. \n# For each dataset, computes t-SNE projections.b. Plots a scatter plot of the t-SNE projections.c. \n# Computes the correlation between the dimensions of the t-SNE projections and displays the correlation DataFrame.d. \n# For each specified column ('binds_BRD4', 'binds_HSA', 'binds_sEH') in the original DataFrame data_train.\n# plots a scatter plot of the t-SNE projections colored according to the values of that column.\nfor x in[x_train_model, y_train_model_mean]:\n    x = tsne_model.fit_transform(x)\n    sns.scatterplot(x=x[:,0], y=x[:,1])\n    plt.show()\n    df = pd.DataFrame(x)\n    df = df.reset_index()\n    display(df.corr())\n    \n    for x_col in ['binds_BRD4', 'binds_HSA', 'binds_sEH']:\n        sns.scatterplot(x = x[:,0], y=x[:,1], hue=data_train[x_col])\n        plt.title(\"TSNE - LLM ChemBERTa\", fontsize=20)\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-25T17:47:51.500355Z","iopub.execute_input":"2024-04-25T17:47:51.500733Z","iopub.status.idle":"2024-04-25T17:50:40.536561Z","shell.execute_reply.started":"2024-04-25T17:47:51.500699Z","shell.execute_reply":"2024-04-25T17:50:40.535605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Reference**\n\nChemBERTa-77M-MTR https://huggingface.co/DeepChem/ChemBERTa-77M-MTR\n\nhugging face https://huggingface.co/\n\n","metadata":{}},{"cell_type":"markdown","source":"## **Acknowledgements**\n\nALEXANDER CHERVOV  https://www.kaggle.com/code/alexandervc/chembert2a-smiles-embeddings-for-beginners/notebook","metadata":{}},{"cell_type":"markdown","source":"## **Conclusion**\n\nIn this project, the goal is to predict new medications using molecules. The LLM ChemBERTa model, specifically designed for Bio in the Health field, was employed for this purpose. The project consists of several stages:\n\nData preprocessing: This involves preparing the data for analysis and modeling.\nExploratory data analysis: A small-scale analysis of the data is conducted to gain insights into the molecules.\nModel development: An LLM model is developed for cluster creation. Two dimensionality reduction techniques, t-SNE and UMAP, are utilized to reduce the dimensions of the data and visualize the clusters. The plots of the clusters are used to identify the molecules.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}