{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ?\n\n\nExamples of models for DNA embeddings \n\nFor RNA change might be neccessary change U (uracil) in RNA for   thymine (T)  - in DNA","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\ncc = 0\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        cc += 1\n        if cc < 50:\n            print(os.path.join(dirname, filename))\n        else:\n            break\n    if cc < 50:\n        pass\n    else:\n        break\n            \n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-10T20:15:09.939543Z","iopub.execute_input":"2023-09-10T20:15:09.940066Z","iopub.status.idle":"2023-09-10T20:15:10.155850Z","shell.execute_reply.started":"2023-09-10T20:15:09.939990Z","shell.execute_reply":"2023-09-10T20:15:10.154472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GENA LM  install with cloning git directory ","metadata":{}},{"cell_type":"code","source":"! git clone https://github.com/AIRI-Institute/GENA_LM.git\n\nfrom GENA_LM.src.gena_lm.modeling_bert import BertForSequenceClassification\nfrom transformers import AutoTokenizer\n\ntokenizer = AutoTokenizer.from_pretrained('AIRI-Institute/gena-lm-bert-base')\nmodel = BertForSequenceClassification.from_pretrained('AIRI-Institute/gena-lm-bert-base')","metadata":{"execution":{"iopub.status.busy":"2023-09-10T20:17:16.513837Z","iopub.execute_input":"2023-09-10T20:17:16.514644Z","iopub.status.idle":"2023-09-10T20:17:50.874800Z","shell.execute_reply.started":"2023-09-10T20:17:16.514595Z","shell.execute_reply":"2023-09-10T20:17:50.873330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re \nimport torch \ndevice = torch.device('cuda:0' if torch.cuda.is_available() else 'cpu')\nseq = 'AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA'\n\nsequence_examples = [\" \".join(list(re.sub(r\"[UZOB]\", \"X\", seq)))]\n\nids = tokenizer.batch_encode_plus(sequence_examples, add_special_tokens=True, padding=\"longest\")\n\ninput_ids = torch.tensor(ids['input_ids']).to(device)\nattention_mask = torch.tensor(ids['attention_mask']).to(device)\n\n# generate embeddings\nwith torch.no_grad():\n    embedding_repr = model(input_ids=input_ids, output_hidden_states = True , \n                           attention_mask=attention_mask)\n\n# dir(embedding_repr)\ne = embedding_repr['logits']# .keys()\ne.shape\nembedding_repr","metadata":{"execution":{"iopub.status.busy":"2023-09-10T20:17:57.700428Z","iopub.execute_input":"2023-09-10T20:17:57.700854Z","iopub.status.idle":"2023-09-10T20:17:58.010991Z","shell.execute_reply.started":"2023-09-10T20:17:57.700813Z","shell.execute_reply":"2023-09-10T20:17:58.009657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DNA BERT - seems works - despite warning \n\nBut some problem with tokenizer: # tokenizer = AutoTokenizer.from_pretrained('Peltarion/dnabert-minilm-small')\n\nSo let us use previous from GENA_lm\n","metadata":{"execution":{"iopub.status.busy":"2023-04-28T09:11:53.289445Z","iopub.execute_input":"2023-04-28T09:11:53.289920Z","iopub.status.idle":"2023-04-28T09:11:53.684068Z","shell.execute_reply.started":"2023-04-28T09:11:53.289879Z","shell.execute_reply":"2023-04-28T09:11:53.682550Z"}}},{"cell_type":"code","source":"%%time\nimport re\nimport torch\nfrom transformers import BertForSequenceClassification, AutoTokenizer\n# tokenizer = AutoTokenizer.from_pretrained('Peltarion/dnabert-minilm-small')\ntokenizer = AutoTokenizer.from_pretrained('AIRI-Institute/gena-lm-bert-base')\n\n\nmodel = BertForSequenceClassification.from_pretrained('Peltarion/dnabert-minilm-small')\n\nseq = 'AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAATTTTTTTTTTTTTTTTAA'\nprint(len(seq))\n\nsequence_examples = [\" \".join(list(re.sub(r\"[UZOB]\", \"X\", seq)))]\n\nids = tokenizer.batch_encode_plus(sequence_examples, add_special_tokens=True, padding=\"longest\")\n\ninput_ids = torch.tensor(ids['input_ids']).to(device)\nattention_mask = torch.tensor(ids['attention_mask']).to(device)\n\n# generate embeddings\nwith torch.no_grad():\n    embedding_repr = model(input_ids=input_ids,\n                           attention_mask=attention_mask)\n\n# dir(embedding_repr)\ne = embedding_repr['logits']# .keys()\n\n\nimport numpy as np\nv = embedding_repr['hidden_states'][-1].numpy()[0,:,:].mean(0)\nv.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-10T20:18:52.216285Z","iopub.execute_input":"2023-09-10T20:18:52.216753Z","iopub.status.idle":"2023-09-10T20:18:56.368604Z","shell.execute_reply.started":"2023-09-10T20:18:52.216710Z","shell.execute_reply":"2023-09-10T20:18:56.367248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_repr['hidden_states'][0].shape","metadata":{"execution":{"iopub.status.busy":"2023-09-10T20:18:59.003309Z","iopub.execute_input":"2023-09-10T20:18:59.004767Z","iopub.status.idle":"2023-09-10T20:18:59.013453Z","shell.execute_reply.started":"2023-09-10T20:18:59.004701Z","shell.execute_reply":"2023-09-10T20:18:59.012085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_repr['hidden_states'][-1].shape","metadata":{"execution":{"iopub.status.busy":"2023-09-10T20:19:00.467473Z","iopub.execute_input":"2023-09-10T20:19:00.467932Z","iopub.status.idle":"2023-09-10T20:19:00.480459Z","shell.execute_reply.started":"2023-09-10T20:19:00.467891Z","shell.execute_reply":"2023-09-10T20:19:00.478612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-09-10T20:19:00.980167Z","iopub.execute_input":"2023-09-10T20:19:00.981512Z","iopub.status.idle":"2023-09-10T20:19:00.987744Z","shell.execute_reply.started":"2023-09-10T20:19:00.981446Z","shell.execute_reply":"2023-09-10T20:19:00.986325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"v = embedding_repr['hidden_states'][-1].numpy()[0,:,:].mean(0)\nv.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-10T20:19:01.519217Z","iopub.execute_input":"2023-09-10T20:19:01.520079Z","iopub.status.idle":"2023-09-10T20:19:01.530030Z","shell.execute_reply.started":"2023-09-10T20:19:01.519982Z","shell.execute_reply":"2023-09-10T20:19:01.528605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"v = embedding_repr['hidden_states'][-1].numpy()[0,:,:].mean(0)\nv.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-10T20:19:02.338245Z","iopub.execute_input":"2023-09-10T20:19:02.338662Z","iopub.status.idle":"2023-09-10T20:19:02.348290Z","shell.execute_reply.started":"2023-09-10T20:19:02.338626Z","shell.execute_reply":"2023-09-10T20:19:02.346920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_repr['hidden_states'][-1].numpy()[0,:,:].mean(0)","metadata":{"execution":{"iopub.status.busy":"2023-09-10T20:19:03.343525Z","iopub.execute_input":"2023-09-10T20:19:03.344495Z","iopub.status.idle":"2023-09-10T20:19:03.367644Z","shell.execute_reply.started":"2023-09-10T20:19:03.344441Z","shell.execute_reply":"2023-09-10T20:19:03.365770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"v = embedding_repr['hidden_states'][-1].numpy()[0,:,:].mean(0)\nv.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-10T20:19:03.556064Z","iopub.execute_input":"2023-09-10T20:19:03.557212Z","iopub.status.idle":"2023-09-10T20:19:03.567347Z","shell.execute_reply.started":"2023-09-10T20:19:03.557167Z","shell.execute_reply":"2023-09-10T20:19:03.566239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# BigBird - seems to be no embeddings ","metadata":{}},{"cell_type":"code","source":"from transformers import AutoTokenizer, BigBirdForMaskedLM\n\ntokenizer = AutoTokenizer.from_pretrained('AIRI-Institute/gena-lm-bigbird-base-t2t')\nmodel = BigBirdForMaskedLM.from_pretrained('AIRI-Institute/gena-lm-bigbird-base-t2t')\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-09-10T20:19:05.295524Z","iopub.execute_input":"2023-09-10T20:19:05.296629Z","iopub.status.idle":"2023-09-10T20:19:21.828450Z","shell.execute_reply.started":"2023-09-10T20:19:05.296583Z","shell.execute_reply":"2023-09-10T20:19:21.827091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\ndevice = torch.device('cuda:0' if torch.cuda.is_available() else 'cpu')","metadata":{"execution":{"iopub.status.busy":"2023-09-10T20:20:03.021615Z","iopub.execute_input":"2023-09-10T20:20:03.022095Z","iopub.status.idle":"2023-09-10T20:20:03.027644Z","shell.execute_reply.started":"2023-09-10T20:20:03.022047Z","shell.execute_reply":"2023-09-10T20:20:03.026487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(seq)","metadata":{"execution":{"iopub.status.busy":"2023-09-10T20:20:03.508063Z","iopub.execute_input":"2023-09-10T20:20:03.508484Z","iopub.status.idle":"2023-09-10T20:20:03.516369Z","shell.execute_reply.started":"2023-09-10T20:20:03.508446Z","shell.execute_reply":"2023-09-10T20:20:03.515340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seq = 'AAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAA'\n\nsequence_examples = [\" \".join(list(re.sub(r\"[UZOB]\", \"X\", seq)))]\n\nids = tokenizer.batch_encode_plus(sequence_examples, add_special_tokens=True, padding=\"longest\")\n\ninput_ids = torch.tensor(ids['input_ids']).to(device)\nattention_mask = torch.tensor(ids['attention_mask']).to(device)\n\n# generate embeddings\nwith torch.no_grad():\n    embedding_repr = model(input_ids=input_ids,\n                           attention_mask=attention_mask)\n\n# dir(embedding_repr)\ne = embedding_repr['logits']# .keys()\ne.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-10T20:20:04.484964Z","iopub.execute_input":"2023-09-10T20:20:04.485396Z","iopub.status.idle":"2023-09-10T20:20:04.900439Z","shell.execute_reply.started":"2023-09-10T20:20:04.485360Z","shell.execute_reply":"2023-09-10T20:20:04.899062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_repr.__dict__","metadata":{"execution":{"iopub.status.busy":"2023-09-10T20:20:05.000463Z","iopub.execute_input":"2023-09-10T20:20:05.001166Z","iopub.status.idle":"2023-09-10T20:20:05.009337Z","shell.execute_reply.started":"2023-09-10T20:20:05.001126Z","shell.execute_reply":"2023-09-10T20:20:05.008376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\ndef get_embeddings(seq):\n    sequence_examples = [\" \".join(list(re.sub(r\"[UZOB]\", \"X\", seq)))]\n\n    ids = tokenizer.batch_encode_plus(sequence_examples, add_special_tokens=True, padding=\"longest\")\n\n    input_ids = torch.tensor(ids['input_ids']).to(device)\n    attention_mask = torch.tensor(ids['attention_mask']).to(device)\n\n    # generate embeddings\n    with torch.no_grad():\n        embedding_repr = model(input_ids=input_ids,\n                               attention_mask=attention_mask)\n\n    # extract residue embeddings for the first ([0,:]) sequence in the batch and remove padded & special tokens ([0,:7]) \n    \n    if 1:\n        emb_0 = embedding_repr.last_hidden_state[0]\n    emb_0_per_protein = emb_0.mean(dim=0)\n    \n    return emb_0_per_protein\n\nget_embeddings('AAAAAAAAAAAAAAA')","metadata":{"execution":{"iopub.status.busy":"2023-09-10T20:20:05.671258Z","iopub.execute_input":"2023-09-10T20:20:05.672438Z","iopub.status.idle":"2023-09-10T20:20:06.200864Z","shell.execute_reply.started":"2023-09-10T20:20:05.672386Z","shell.execute_reply":"2023-09-10T20:20:06.199096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from src.gena_lm.modeling_bert import BertForSequenceClassification\nfrom transformers import AutoTokenizer\n\ntokenizer = AutoTokenizer.from_pretrained('AIRI-Institute/gena-lm-bert-base')\nmodel = BertForSequenceClassification.from_pretrained('AIRI-Institute/gena-lm-bert-base')","metadata":{"execution":{"iopub.status.busy":"2023-09-10T20:20:06.214105Z","iopub.execute_input":"2023-09-10T20:20:06.214712Z","iopub.status.idle":"2023-09-10T20:20:06.244299Z","shell.execute_reply.started":"2023-09-10T20:20:06.214674Z","shell.execute_reply":"2023-09-10T20:20:06.242769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}