{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30733,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Fine Tuning with BERT\nI used the HuggingFace🤗 transformers library for the models and tokenizers. Checkpoint used: bert-base-uncased. For finetuning the model, the I concatenated the SMILES and the protein names using the special separator [SEP] token. The [SEP] token is commonly used in BERT to separate different segments of texts in tasks like question answering where the input consists of two distinct parts (e.g., a question and a context).\n\nFor example:\n```python\nsmiles = ''Cc1conc1CNc1nc(Nc2cccnc2C)nc(N[C@H](CC(=O)N[Dy])c2ccc(Cl)cc2)n1'\nprotein_name = 'BRD4'\ninput_text = f{smiles} [SEP] {protein_name}\n#input text\n'Cc1conc1CNc1nc(Nc2cccnc2C)nc(N[C@H](CC(=O)N[Dy])c2ccc(Cl)cc2)n1[SEP]BRD4'\n# tokenize this input_text\n```","metadata":{}},{"cell_type":"code","source":"!pip install duckdb -q\n!pip install rdkit -q","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:38:35.177744Z","iopub.execute_input":"2024-06-05T13:38:35.178470Z","iopub.status.idle":"2024-06-05T13:39:03.306954Z","shell.execute_reply.started":"2024-06-05T13:38:35.178433Z","shell.execute_reply":"2024-06-05T13:39:03.305915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport duckdb\nimport matplotlib.pyplot as plt\nfrom datasets import*\nfrom rdkit import Chem\nfrom rdkit.Chem import AllChem\nimport seaborn as sns ","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:39:03.945753Z","iopub.execute_input":"2024-06-05T13:39:03.946037Z","iopub.status.idle":"2024-06-05T13:39:03.950921Z","shell.execute_reply.started":"2024-06-05T13:39:03.946013Z","shell.execute_reply":"2024-06-05T13:39:03.949910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The dataset is way too big, getting a small one to work with","metadata":{}},{"cell_type":"code","source":"data_train = '/kaggle/input/leash-BELKA/train.parquet'\ntest_path = '/kaggle/input/leash-BELKA/test.parquet'\n\ncon = duckdb.connect()\n\n# Query\ndata = con.query(f\"\"\"(SELECT * FROM parquet_scan('{data_train}') \nWHERE binds = 0\nORDER BY random()\nLIMIT 30000)\nUNION ALL\n(SELECT * FROM parquet_scan('{data_train}')\nWHERE binds = 1\nORDER BY random()\nLIMIT 30000)\"\"\").df()\n\n# Closing database\ncon.close()\n\n# Saving dataset\ndata.to_csv(\"/kaggle/working/small_dataset.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:39:03.953413Z","iopub.execute_input":"2024-06-05T13:39:03.953970Z","iopub.status.idle":"2024-06-05T13:39:53.324924Z","shell.execute_reply.started":"2024-06-05T13:39:03.953932Z","shell.execute_reply":"2024-06-05T13:39:53.323769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"small_df = pd.read_csv('/kaggle/working/small_dataset.csv')\nsmall_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:39:53.326081Z","iopub.execute_input":"2024-06-05T13:39:53.326431Z","iopub.status.idle":"2024-06-05T13:39:53.514142Z","shell.execute_reply.started":"2024-06-05T13:39:53.326398Z","shell.execute_reply":"2024-06-05T13:39:53.513221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=small_df,x='binds');","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:39:53.515528Z","iopub.execute_input":"2024-06-05T13:39:53.515907Z","iopub.status.idle":"2024-06-05T13:39:54.083800Z","shell.execute_reply.started":"2024-06-05T13:39:53.515874Z","shell.execute_reply":"2024-06-05T13:39:54.082906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The `[SEP]` token is crucial as it separates the two text inputs. This allows the model to understand that it is processing two related but distinct segments of data.","metadata":{}},{"cell_type":"code","source":"small_df['text'] = small_df['molecule_smiles']+ '[SEP]' + small_df['protein_name']\nsmall_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:39:54.084987Z","iopub.execute_input":"2024-06-05T13:39:54.085268Z","iopub.status.idle":"2024-06-05T13:39:54.120437Z","shell.execute_reply.started":"2024-06-05T13:39:54.085244Z","shell.execute_reply":"2024-06-05T13:39:54.119632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"small_df.columns","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:39:54.121903Z","iopub.execute_input":"2024-06-05T13:39:54.122248Z","iopub.status.idle":"2024-06-05T13:39:54.129076Z","shell.execute_reply.started":"2024-06-05T13:39:54.122217Z","shell.execute_reply":"2024-06-05T13:39:54.128207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"small_df = small_df.drop(['Unnamed: 0', 'id', 'buildingblock1_smiles', 'buildingblock2_smiles',\n       'buildingblock3_smiles', 'molecule_smiles', 'protein_name'],axis=1)\nsmall_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:39:54.130249Z","iopub.execute_input":"2024-06-05T13:39:54.130828Z","iopub.status.idle":"2024-06-05T13:39:54.143274Z","shell.execute_reply.started":"2024-06-05T13:39:54.130802Z","shell.execute_reply":"2024-06-05T13:39:54.142447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = Dataset.from_pandas(small_df)\ndataset","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:39:54.146806Z","iopub.execute_input":"2024-06-05T13:39:54.147115Z","iopub.status.idle":"2024-06-05T13:39:54.202187Z","shell.execute_reply.started":"2024-06-05T13:39:54.147092Z","shell.execute_reply":"2024-06-05T13:39:54.201311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = dataset.rename_column('binds','label')\ndataset = dataset.train_test_split(test_size=0.2)\ndataset","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:39:54.203466Z","iopub.execute_input":"2024-06-05T13:39:54.203813Z","iopub.status.idle":"2024-06-05T13:39:54.246740Z","shell.execute_reply.started":"2024-06-05T13:39:54.203780Z","shell.execute_reply":"2024-06-05T13:39:54.245921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = dataset['train']\ntest_df = dataset['test']","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:39:54.247688Z","iopub.execute_input":"2024-06-05T13:39:54.248052Z","iopub.status.idle":"2024-06-05T13:39:54.252073Z","shell.execute_reply.started":"2024-06-05T13:39:54.248017Z","shell.execute_reply":"2024-06-05T13:39:54.251211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id2label = {0:'No bind',1:'Bind'}\nlabel2id = {'No bind':0,'Bind':1}","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:39:54.253276Z","iopub.execute_input":"2024-06-05T13:39:54.254037Z","iopub.status.idle":"2024-06-05T13:39:54.260686Z","shell.execute_reply.started":"2024-06-05T13:39:54.254003Z","shell.execute_reply":"2024-06-05T13:39:54.259883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import BertTokenizer, BertTokenizerFast,BertForSequenceClassification,Trainer,TrainingArguments","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:39:54.261639Z","iopub.execute_input":"2024-06-05T13:39:54.261930Z","iopub.status.idle":"2024-06-05T13:39:54.300100Z","shell.execute_reply.started":"2024-06-05T13:39:54.261901Z","shell.execute_reply":"2024-06-05T13:39:54.299451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint = \"bert-base-uncased\"\ntokenizer = BertTokenizerFast.from_pretrained(checkpoint)\nmodel = BertForSequenceClassification.from_pretrained(checkpoint,num_labels=2,label2id=label2id,id2label=id2label)","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:39:54.300980Z","iopub.execute_input":"2024-06-05T13:39:54.301213Z","iopub.status.idle":"2024-06-05T13:40:09.647845Z","shell.execute_reply.started":"2024-06-05T13:39:54.301186Z","shell.execute_reply":"2024-06-05T13:40:09.647125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_function(examples):\n    return tokenizer(examples['text'],truncation=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:40:09.649091Z","iopub.execute_input":"2024-06-05T13:40:09.649566Z","iopub.status.idle":"2024-06-05T13:40:09.654375Z","shell.execute_reply.started":"2024-06-05T13:40:09.649530Z","shell.execute_reply":"2024-06-05T13:40:09.653285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenize_train = train_df.map(preprocess_function,batched=True)\ntokenize_test = test_df.map(preprocess_function,batched=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:40:09.655562Z","iopub.execute_input":"2024-06-05T13:40:09.655902Z","iopub.status.idle":"2024-06-05T13:40:18.229019Z","shell.execute_reply.started":"2024-06-05T13:40:09.655870Z","shell.execute_reply":"2024-06-05T13:40:18.228235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import DataCollatorWithPadding\ndata_collator = DataCollatorWithPadding(tokenizer=tokenizer)","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:40:18.230036Z","iopub.execute_input":"2024-06-05T13:40:18.230305Z","iopub.status.idle":"2024-06-05T13:40:18.234707Z","shell.execute_reply.started":"2024-06-05T13:40:18.230282Z","shell.execute_reply":"2024-06-05T13:40:18.233777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install evaluate -q","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:40:18.235838Z","iopub.execute_input":"2024-06-05T13:40:18.236185Z","iopub.status.idle":"2024-06-05T13:40:30.893407Z","shell.execute_reply.started":"2024-06-05T13:40:18.236154Z","shell.execute_reply":"2024-06-05T13:40:30.892295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import evaluate\naccuracy = evaluate.load(\"accuracy\")","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:40:30.894996Z","iopub.execute_input":"2024-06-05T13:40:30.895307Z","iopub.status.idle":"2024-06-05T13:40:32.268550Z","shell.execute_reply.started":"2024-06-05T13:40:30.895280Z","shell.execute_reply":"2024-06-05T13:40:32.267794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def compute_metrics(eval_pred):\n    predictions, labels = eval_pred\n    predictions = np.argmax(predictions, axis=1)\n    return accuracy.compute(predictions=predictions, references=labels)","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:40:32.269621Z","iopub.execute_input":"2024-06-05T13:40:32.269915Z","iopub.status.idle":"2024-06-05T13:40:32.274488Z","shell.execute_reply.started":"2024-06-05T13:40:32.269888Z","shell.execute_reply":"2024-06-05T13:40:32.273617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_args = TrainingArguments(\n    output_dir='Belka-BERT',\n    overwrite_output_dir=True,\n    num_train_epochs=4,\n    per_device_train_batch_size=8,\n    per_device_eval_batch_size=8,\n    evaluation_strategy='epoch',\n    save_strategy='epoch',\n    logging_dir='logs',\n    report_to='none',\n    learning_rate=2e-5,\n)","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:40:32.275645Z","iopub.execute_input":"2024-06-05T13:40:32.275943Z","iopub.status.idle":"2024-06-05T13:40:32.355410Z","shell.execute_reply.started":"2024-06-05T13:40:32.275918Z","shell.execute_reply":"2024-06-05T13:40:32.354525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer = Trainer(\n    model=model,\n    args=training_args,\n    train_dataset=tokenize_train,\n    eval_dataset=tokenize_test,\n    tokenizer=tokenizer,\n    data_collator=data_collator,\n    compute_metrics=compute_metrics,\n)","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:40:32.356767Z","iopub.execute_input":"2024-06-05T13:40:32.357216Z","iopub.status.idle":"2024-06-05T13:40:32.603717Z","shell.execute_reply.started":"2024-06-05T13:40:32.357183Z","shell.execute_reply":"2024-06-05T13:40:32.602831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer.train()","metadata":{"execution":{"iopub.status.busy":"2024-06-05T12:41:07.748377Z","iopub.execute_input":"2024-06-05T12:41:07.749247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Sample Inference","metadata":{}},{"cell_type":"code","source":"from transformers import pipeline","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:40:32.604811Z","iopub.execute_input":"2024-06-05T13:40:32.605108Z","iopub.status.idle":"2024-06-05T13:40:32.609409Z","shell.execute_reply.started":"2024-06-05T13:40:32.605083Z","shell.execute_reply":"2024-06-05T13:40:32.608487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifier = pipeline('text-classification',model=\"/kaggle/working/Belka-BERT/checkpoint-12000\",tokenizer=tokenizer,device=0) #device=0 for GPU","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:40:33.557259Z","iopub.execute_input":"2024-06-05T13:40:33.557542Z","iopub.status.idle":"2024-06-05T13:40:33.941069Z","shell.execute_reply.started":"2024-06-05T13:40:33.557518Z","shell.execute_reply":"2024-06-05T13:40:33.940243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifier(train_df['text'][100])","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:40:33.945003Z","iopub.execute_input":"2024-06-05T13:40:33.945513Z","iopub.status.idle":"2024-06-05T13:40:35.122968Z","shell.execute_reply.started":"2024-06-05T13:40:33.945485Z","shell.execute_reply":"2024-06-05T13:40:35.122017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[100]","metadata":{"execution":{"iopub.status.busy":"2024-06-05T13:40:58.561161Z","iopub.execute_input":"2024-06-05T13:40:58.561590Z","iopub.status.idle":"2024-06-05T13:40:58.568085Z","shell.execute_reply.started":"2024-06-05T13:40:58.561559Z","shell.execute_reply":"2024-06-05T13:40:58.567188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## It works","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##Any suggestions to improve training performance on Kaggle kernel seems slow even with P100, I was able to improve locally??","metadata":{},"execution_count":null,"outputs":[]}]}