{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"},{"sourceId":8196925,"sourceType":"datasetVersion","datasetId":4855347}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install rdkit","metadata":{"execution":{"iopub.status.busy":"2024-07-02T17:59:37.537874Z","iopub.execute_input":"2024-07-02T17:59:37.538837Z","iopub.status.idle":"2024-07-02T17:59:52.767879Z","shell.execute_reply.started":"2024-07-02T17:59:37.538792Z","shell.execute_reply":"2024-07-02T17:59:52.766776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!pip install biopython","metadata":{"execution":{"iopub.status.busy":"2024-07-02T17:59:52.770501Z","iopub.execute_input":"2024-07-02T17:59:52.771444Z","iopub.status.idle":"2024-07-02T17:59:52.775891Z","shell.execute_reply.started":"2024-07-02T17:59:52.771400Z","shell.execute_reply":"2024-07-02T17:59:52.774896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install duckdb","metadata":{"execution":{"iopub.status.busy":"2024-07-02T17:59:52.777072Z","iopub.execute_input":"2024-07-02T17:59:52.777362Z","iopub.status.idle":"2024-07-02T18:00:05.709930Z","shell.execute_reply.started":"2024-07-02T17:59:52.777337Z","shell.execute_reply":"2024-07-02T18:00:05.708660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport os\nimport gc\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport lightgbm as lgb\n\nfrom sklearn.model_selection import (train_test_split, cross_val_score, \n                                     GridSearchCV, KFold)\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.metrics.pairwise import cosine_similarity\nfrom sklearn.metrics import (accuracy_score, precision_score, \n                             recall_score, f1_score, classification_report)\nfrom xgboost import XGBClassifier\n\nimport rdkit\nfrom rdkit import Chem \nfrom rdkit.Chem import (rdMolDescriptors, Descriptors, \n                        rdmolfiles, rdFingerprintGenerator, AllChem)\n\n# from Bio import ExPASy\n# from Bio import SwissProt\n# from Bio.SeqUtils.ProtParam import ProteinAnalysis\n\n\nimport duckdb\n\noriginal_data_dir = \"/kaggle/input/leash-BELKA\"\nworking_data_dir = '/kaggle/working/'\ntrain_file = \"train.parquet\"\ntest_file = \"test.csv\"\npp_file = '/kaggle/working/PreProc_BELKA_training_data.csv'\npptest_file = \"PreProc_Test_BELKA.csv\"\n\n\nchemical_radious = 2\nchemical_nbits = 512\ndata_cap = 100_000        # Total data for each binding option (i.e *2)\nnfolds = 4               # Number of CV folds for Grid Search and XGboost modeling\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-02T18:00:05.712808Z","iopub.execute_input":"2024-07-02T18:00:05.713151Z","iopub.status.idle":"2024-07-02T18:00:08.320156Z","shell.execute_reply.started":"2024-07-02T18:00:05.713113Z","shell.execute_reply":"2024-07-02T18:00:08.319076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Thanks to Mehra Kazeminia, et al. and @andrewdblevins for the following code segment which elegently reads a balanced dataset from the original data:\n\nhttps://www.kaggle.com/code/mehrankazeminia/p-1-6-belka-eda-data-separation","metadata":{}},{"cell_type":"code","source":"train_path = os.path.join(original_data_dir, train_file)\ncon = duckdb.connect()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:00:08.321651Z","iopub.execute_input":"2024-07-02T18:00:08.322556Z","iopub.status.idle":"2024-07-02T18:00:08.334665Z","shell.execute_reply.started":"2024-07-02T18:00:08.322516Z","shell.execute_reply":"2024-07-02T18:00:08.333662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_data(rows=data_cap) -> pd.DataFrame :\n    print(f\"Reading {rows * 6} rows\")\n    df = con.query(f\"\"\"(SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0 AND protein_name = 'HSA'\n                        ORDER BY random()\n                        LIMIT {rows})\n                        UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1 AND protein_name = 'HSA'\n                        ORDER BY random()\n                        LIMIT {rows})\n\n                        UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0 AND protein_name = 'BRD4'\n                        ORDER BY random()\n                        LIMIT {rows})\n                        UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1 AND protein_name = 'BRD4'\n                        ORDER BY random()\n                        LIMIT {rows})\n\n                        UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0 AND protein_name = 'sEH'\n                        ORDER BY random()\n                        LIMIT {rows})\n                        UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1 AND protein_name = 'sEH'\n                        ORDER BY random()\n                        LIMIT {rows})\"\"\").df()\n\n    con.close()\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:00:08.336305Z","iopub.execute_input":"2024-07-02T18:00:08.336649Z","iopub.status.idle":"2024-07-02T18:00:08.347412Z","shell.execute_reply.started":"2024-07-02T18:00:08.336621Z","shell.execute_reply":"2024-07-02T18:00:08.346404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"train_df[['protein_name', 'binds', 'molecule_smiles']].groupby(by=['protein_name', 'binds']).count()","metadata":{"execution":{"iopub.status.busy":"2024-07-01T14:36:57.948914Z","iopub.execute_input":"2024-07-01T14:36:57.949229Z","iopub.status.idle":"2024-07-01T14:36:58.115982Z","shell.execute_reply.started":"2024-07-01T14:36:57.949203Z","shell.execute_reply":"2024-07-01T14:36:58.114989Z"}}},{"cell_type":"raw","source":"train_df.sample(frac=1, random_state=42).reset_index(drop=True)  # Shuffle\ntrain_df.to_csv(os.path.join('train_BELKA_balanced.csv'), index=False)","metadata":{"execution":{"iopub.status.busy":"2024-07-01T14:36:58.117393Z","iopub.execute_input":"2024-07-01T14:36:58.118168Z","iopub.status.idle":"2024-07-01T14:37:03.192810Z","shell.execute_reply.started":"2024-07-01T14:36:58.118139Z","shell.execute_reply":"2024-07-01T14:37:03.190271Z"}}},{"cell_type":"raw","source":"train_df.shape, train_df.columns","metadata":{"execution":{"iopub.status.busy":"2024-07-01T14:37:03.195860Z","iopub.execute_input":"2024-07-01T14:37:03.197453Z","iopub.status.idle":"2024-07-01T14:37:03.209122Z","shell.execute_reply.started":"2024-07-01T14:37:03.197382Z","shell.execute_reply":"2024-07-01T14:37:03.207840Z"}}},{"cell_type":"raw","source":"plt.gca().set_facecolor('lightgray')\ntrain_df['binds'].value_counts(normalize=True).plot(kind='barh', figsize=(12,1), color=['darkcyan','red'])\n\npd.DataFrame(data= {'Number': train_df['binds'].value_counts(), \n                    'Percent': train_df['binds'].value_counts(normalize=True)})","metadata":{"execution":{"iopub.status.busy":"2024-07-01T14:37:03.214307Z","iopub.execute_input":"2024-07-01T14:37:03.215648Z","iopub.status.idle":"2024-07-01T14:37:03.534737Z","shell.execute_reply.started":"2024-07-01T14:37:03.215586Z","shell.execute_reply":"2024-07-01T14:37:03.533453Z"}}},{"cell_type":"raw","source":"plt.gca().set_facecolor('lightgray')\ntrain_df['protein_name'].value_counts(normalize=True).plot(kind='barh', figsize=(12,1.5), color=['darkcyan','red','violet'])\n\npd.DataFrame(data= {'Number': train_df['protein_name'].value_counts(), \n                    'Percent': train_df['protein_name'].value_counts(normalize=True)})","metadata":{"execution":{"iopub.status.busy":"2024-07-01T14:37:03.536307Z","iopub.execute_input":"2024-07-01T14:37:03.536887Z","iopub.status.idle":"2024-07-01T14:37:03.990631Z","shell.execute_reply.started":"2024-07-01T14:37:03.536842Z","shell.execute_reply":"2024-07-01T14:37:03.989476Z"}}},{"cell_type":"raw","source":"plt.gca().set_facecolor('lightgray')\ntrain_df.groupby('protein_name')['binds'].value_counts(normalize=True).plot(kind='barh', figsize=(12,3), color=['darkcyan','red'])\n\npd.DataFrame(data= {'Number': train_df.groupby('protein_name')['binds'].value_counts(), \n                    'Percent': train_df.groupby('protein_name')['binds'].value_counts(normalize=True)})","metadata":{"execution":{"iopub.status.busy":"2024-07-01T14:37:03.992283Z","iopub.execute_input":"2024-07-01T14:37:03.992770Z","iopub.status.idle":"2024-07-01T14:37:04.508751Z","shell.execute_reply.started":"2024-07-01T14:37:03.992728Z","shell.execute_reply":"2024-07-01T14:37:04.507377Z"}}},{"cell_type":"code","source":"#train_df = pd.read_csv(\"/kaggle/working/train_BELKA_balanced.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:00:08.348711Z","iopub.execute_input":"2024-07-02T18:00:08.349039Z","iopub.status.idle":"2024-07-02T18:00:08.361068Z","shell.execute_reply.started":"2024-07-02T18:00:08.349011Z","shell.execute_reply":"2024-07-02T18:00:08.360067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### UPdate to our effort so far\nAfter a number of Iterartions, it appears the most important factor in binding affinity \nis the MorganFingerPrint bit vector. After adding this feature, the data got too large for memory and\nwe had to read the data in chunks and proces them one chunk at a time.\n\nOnce completed all three train data files (15GB), we realized the MFP array had increased each datafile to the point where we had no further room to process and split the test data or even be able to zip the three train data !!!\n\nThis brings to today, when we decided to reduce, by half, the data we read and see if we can get a better result by adding the MFP to the model.\n\nWE also decided to inclide only the MFP data along with Protein Enzyme Analysis data. The Protein Ezyme analysis is, of course, constant for every row in the protein-specific train dataset. While the MFP data is binary 0/1, the Protien Analsis data is in the range 10-100. This causes a challange in normalizatin of data where the columns with constant protein data cannot be normalilzed. \n\nWe have now decided to forgoe any protine data, since they are constant throughout a given train data file and as such could be treated the same as the 'binds' column.","metadata":{}},{"cell_type":"code","source":"# UniProKB Entry Name for our proteins. Assuming these proteins are for Homo Sepians\n# The codes obtained from https://www.uniprot.org/uniprotkb?query=sEH\nUniProKB_Name = {\n    'HSA' : 'HSA_STRGC',\n    'BRD4' : 'BRD4_HUMAN',\n    'sEH' : 'SEH1_HUMAN'\n}","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:00:08.362330Z","iopub.execute_input":"2024-07-02T18:00:08.362660Z","iopub.status.idle":"2024-07-02T18:00:08.372233Z","shell.execute_reply.started":"2024-07-02T18:00:08.362635Z","shell.execute_reply":"2024-07-02T18:00:08.371257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the Morgan Finger Print Generator\nmfpgen = rdFingerprintGenerator.GetMorganGenerator(radius=chemical_radious,\n                                                   fpSize=chemical_nbits)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:00:08.373552Z","iopub.execute_input":"2024-07-02T18:00:08.373943Z","iopub.status.idle":"2024-07-02T18:00:08.385131Z","shell.execute_reply.started":"2024-07-02T18:00:08.373909Z","shell.execute_reply":"2024-07-02T18:00:08.384238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate Morgan fingerprints for molecules\ndef generate_morgan_fingerprints(smiles_list):\n    fingerprints = []\n    for smiles in smiles_list:\n        mol = Chem.MolFromSmiles(smiles)\n        if mol:\n            fp = mfpgen.GetFingerprint(mol)\n            # Convert ExplicitBitVect to a list of integers\n            fp_list = list(fp.ToBitString())\n            fingerprints.append(fp_list)\n        else:\n            fingerprints.append(None)\n    return fingerprints","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:00:08.387551Z","iopub.execute_input":"2024-07-02T18:00:08.387849Z","iopub.status.idle":"2024-07-02T18:00:08.395085Z","shell.execute_reply.started":"2024-07-02T18:00:08.387824Z","shell.execute_reply":"2024-07-02T18:00:08.394225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"def fetch_protein_sequence(protein_name):\n    # Fetch protein entries from UniProt using ExPASy\n    handle = ExPASy.get_sprot_raw(UniProKB_Name[protein_name])\n    records = SwissProt.parse(handle)\n\n    # Get the first record from the iterator\n    first_record = next(records)\n\n    # Return the sequence attribute of the first record\n    return first_record.sequence","metadata":{"execution":{"iopub.status.busy":"2024-07-01T14:40:26.559033Z","iopub.execute_input":"2024-07-01T14:40:26.559427Z","iopub.status.idle":"2024-07-01T14:40:26.566991Z","shell.execute_reply.started":"2024-07-01T14:40:26.559398Z","shell.execute_reply":"2024-07-01T14:40:26.565109Z"}}},{"cell_type":"raw","source":"# Function to fetch and analyze protein features for each row\ndef fetch_and_analyze_protein_features(protein_name):\n    protein_sequence = fetch_protein_sequence(protein_name)\n    protein_features = analyze_protein_sequence(protein_sequence)\n    return [protein_features[aa] for aa in 'ACDEFGHIKLMNPQRSTVWY']","metadata":{"execution":{"iopub.status.busy":"2024-07-01T14:40:29.618318Z","iopub.execute_input":"2024-07-01T14:40:29.618729Z","iopub.status.idle":"2024-07-01T14:40:29.625065Z","shell.execute_reply.started":"2024-07-01T14:40:29.618697Z","shell.execute_reply":"2024-07-01T14:40:29.623590Z"}}},{"cell_type":"raw","source":"def analyze_protein_sequence(sequence):\n    analysis = ProteinAnalysis(sequence)\n    aa_composition = analysis.count_amino_acids()\n    return aa_composition","metadata":{"execution":{"iopub.status.busy":"2024-07-01T14:40:33.500779Z","iopub.execute_input":"2024-07-01T14:40:33.501168Z","iopub.status.idle":"2024-07-01T14:40:33.507211Z","shell.execute_reply.started":"2024-07-01T14:40:33.501140Z","shell.execute_reply":"2024-07-01T14:40:33.505796Z"}}},{"cell_type":"code","source":"#fetch_and_analyze_protein_features('sEH') # HSA_STRGC","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:00:08.397050Z","iopub.execute_input":"2024-07-02T18:00:08.397347Z","iopub.status.idle":"2024-07-02T18:00:08.405883Z","shell.execute_reply.started":"2024-07-02T18:00:08.397322Z","shell.execute_reply":"2024-07-02T18:00:08.404910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"# Prefetch the protein sequences for the three protins we have\nproteins = ['HSA', 'BRD4', 'sEH']\nprotein_sequences_dict = {p: fetch_and_analyze_protein_features(p) for p in proteins}\nprotein_sequences_dict","metadata":{"execution":{"iopub.status.busy":"2024-07-01T14:40:50.722118Z","iopub.execute_input":"2024-07-01T14:40:50.723414Z","iopub.status.idle":"2024-07-01T14:40:56.789438Z","shell.execute_reply.started":"2024-07-01T14:40:50.723370Z","shell.execute_reply":"2024-07-01T14:40:56.787206Z"}}},{"cell_type":"code","source":"columns_to_drop = ['buildingblock1_smiles', 'buildingblock2_smiles', \n                   'buildingblock3_smiles', 'molecule_smiles']","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:00:08.407152Z","iopub.execute_input":"2024-07-02T18:00:08.407624Z","iopub.status.idle":"2024-07-02T18:00:08.416849Z","shell.execute_reply.started":"2024-07-02T18:00:08.407589Z","shell.execute_reply":"2024-07-02T18:00:08.415870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Testing datafrem creation for MorganFingerPrints and Protine Ensyme data ----->>","metadata":{}},{"cell_type":"raw","source":"tdf = train_df.sample(n=50)","metadata":{"execution":{"iopub.status.busy":"2024-06-29T03:13:08.406098Z","iopub.execute_input":"2024-06-29T03:13:08.407305Z","iopub.status.idle":"2024-06-29T03:13:08.430989Z","shell.execute_reply.started":"2024-06-29T03:13:08.407260Z","shell.execute_reply":"2024-06-29T03:13:08.429287Z"}}},{"cell_type":"raw","source":"tdf.columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2024-06-29T03:29:02.603094Z","iopub.execute_input":"2024-06-29T03:29:02.603916Z","iopub.status.idle":"2024-06-29T03:29:02.612589Z","shell.execute_reply.started":"2024-06-29T03:29:02.603876Z","shell.execute_reply":"2024-06-29T03:29:02.611244Z"}}},{"cell_type":"raw","source":"mfp = np.array(generate_morgan_fingerprints(tdf['molecule_smiles']), dtype=np.uint8)","metadata":{"execution":{"iopub.status.busy":"2024-06-29T03:13:12.178006Z","iopub.execute_input":"2024-06-29T03:13:12.178451Z","iopub.status.idle":"2024-06-29T03:13:12.229025Z","shell.execute_reply.started":"2024-06-29T03:13:12.178416Z","shell.execute_reply":"2024-06-29T03:13:12.227932Z"}}},{"cell_type":"raw","source":"mfp.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-29T03:13:16.912034Z","iopub.execute_input":"2024-06-29T03:13:16.912426Z","iopub.status.idle":"2024-06-29T03:13:16.920001Z","shell.execute_reply.started":"2024-06-29T03:13:16.912398Z","shell.execute_reply":"2024-06-29T03:13:16.918620Z"}}},{"cell_type":"raw","source":"mfp","metadata":{"execution":{"iopub.status.busy":"2024-06-29T03:23:48.373166Z","iopub.execute_input":"2024-06-29T03:23:48.373583Z","iopub.status.idle":"2024-06-29T03:23:48.382770Z","shell.execute_reply.started":"2024-06-29T03:23:48.373550Z","shell.execute_reply":"2024-06-29T03:23:48.381404Z"}}},{"cell_type":"raw","source":"mdf = pd.DataFrame(mfp, dtype=np.uint8)\nmdf.columns = [f'fp_{i+1:04d}' for i in range(mfp.shape[1])]\nmdf.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-29T03:13:20.196124Z","iopub.execute_input":"2024-06-29T03:13:20.196507Z","iopub.status.idle":"2024-06-29T03:13:20.221725Z","shell.execute_reply.started":"2024-06-29T03:13:20.196478Z","shell.execute_reply":"2024-06-29T03:13:20.220446Z"}}},{"cell_type":"raw","source":"pfv = tdf['protein_name'].apply(lambda protein_name: protein_sequences_dict[protein_name])\npfv = np.array(list(pfv), dtype=np.int32)\n    \n# Convert protein feature vectors to separate columns with uint8 data type\npdf = pd.DataFrame(pfv, dtype=np.int32)\npdf.columns = [f'pf_{i+1:02d}' for i in range(pdf.shape[1])]\npdf.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-29T03:27:17.125121Z","iopub.execute_input":"2024-06-29T03:27:17.125572Z","iopub.status.idle":"2024-06-29T03:27:17.152445Z","shell.execute_reply.started":"2024-06-29T03:27:17.125530Z","shell.execute_reply":"2024-06-29T03:27:17.150878Z"}}},{"cell_type":"raw","source":"pp_data = pd.concat([tdf.reset_index(drop=True), mdf, pdf], axis=1)\npp_data.drop(columns_to_drop, inplace=True, axis=1)\npp_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-29T03:35:39.104996Z","iopub.execute_input":"2024-06-29T03:35:39.105433Z","iopub.status.idle":"2024-06-29T03:35:39.134138Z","shell.execute_reply.started":"2024-06-29T03:35:39.105402Z","shell.execute_reply":"2024-06-29T03:35:39.132596Z"}}},{"cell_type":"markdown","source":"#### ---------------------------------<<","metadata":{}},{"cell_type":"code","source":"if not os.path.exists(pp_file):\n            \n    print(\"Reading data ...\")    \n    train_df = get_data(rows=150_000)\n    \n    print(\"shuffling data ...\")\n    train_df.sample(frac=1, random_state=42).reset_index(drop=True)  # Shuffle\n    \n    # Generate molecular fingerprints\n    mfp = np.array(generate_morgan_fingerprints(train_df['molecule_smiles']), dtype=np.uint8)\n    \n    # Convert molecular fingerprints to separate columns with uint8 data type\n    mdf = pd.DataFrame(mfp, dtype=np.uint8)\n    mdf.columns = [f'fp_{i+1:04d}' for i in range(mfp.shape[1])]\n    \n#     pfv = train_df['protein_name'].apply(lambda protein_name: protein_sequences_dict[protein_name])\n#     pfv = np.array(list(pfv), dtype=np.int32)\n\n#     # Convert protein feature vectors to separate columns with int32 data type\n#     pdf = pd.DataFrame(pfv, dtype=np.int32)\n#     pdf.columns = [f'pf_{i+1:02d}' for i in range(pdf.shape[1])]\n    \n    pp_data = pd.concat([train_df.reset_index(drop=True), mdf], axis=1)\n    pp_data.drop(columns_to_drop, inplace=True, axis=1)\n    \n    # Save to CSV\n    pp_data.to_parquet(pp_file, index=False)\nelse:\n    pp_data = pd.read_parquet(pp_file)\n    \ndel train_df, mdf, mfp\ngc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:00:18.911450Z","iopub.execute_input":"2024-07-02T18:00:18.912135Z","iopub.status.idle":"2024-07-02T18:11:33.725848Z","shell.execute_reply.started":"2024-07-02T18:00:18.912101Z","shell.execute_reply":"2024-07-02T18:11:33.724452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pp_data.info()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:11:33.727896Z","iopub.execute_input":"2024-07-02T18:11:33.728244Z","iopub.status.idle":"2024-07-02T18:11:33.757192Z","shell.execute_reply.started":"2024-07-02T18:11:33.728216Z","shell.execute_reply":"2024-07-02T18:11:33.756110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subset_data = pp_data.sample(n=10_000, random_state=42)\nX_subset = subset_data.drop(columns=['id', 'binds', 'protein_name']).values\ny_subset = subset_data['binds'].values\n","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:12:23.398493Z","iopub.execute_input":"2024-07-02T18:12:23.399334Z","iopub.status.idle":"2024-07-02T18:12:23.516155Z","shell.execute_reply.started":"2024-07-02T18:12:23.399297Z","shell.execute_reply":"2024-07-02T18:12:23.515104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subset_data['binds'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:12:26.219812Z","iopub.execute_input":"2024-07-02T18:12:26.220739Z","iopub.status.idle":"2024-07-02T18:12:26.230675Z","shell.execute_reply.started":"2024-07-02T18:12:26.220704Z","shell.execute_reply":"2024-07-02T18:12:26.229603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the model\nmodel = XGBClassifier(use_label_encoder=False, eval_metric='logloss')","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:12:38.103777Z","iopub.execute_input":"2024-07-02T18:12:38.104885Z","iopub.status.idle":"2024-07-02T18:12:38.109494Z","shell.execute_reply.started":"2024-07-02T18:12:38.104852Z","shell.execute_reply":"2024-07-02T18:12:38.108186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the parameter grid for hyperparameter tuning\nhyperparam_grid = {\n    'n_estimators': [200, 400, 500],\n    'learning_rate': [.001, 0.01, 0.1],\n    'max_depth': [7, 11, 15],\n    'subsample': [0.8, 1.0],\n    'colsample_bytree': [0.8, 1.0]\n}","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:13:09.867271Z","iopub.execute_input":"2024-07-02T18:13:09.868197Z","iopub.status.idle":"2024-07-02T18:13:09.874018Z","shell.execute_reply.started":"2024-07-02T18:13:09.868163Z","shell.execute_reply":"2024-07-02T18:13:09.872071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Perform k-fold cross-validation\nkfold = KFold(n_splits=nfolds, shuffle=True, random_state=42)\n\n# Perform Grid Search with K-Fold Cross Validation\nmodel = XGBClassifier(use_label_encoder=False, eval_metric='logloss')\n\ngs_cv = GridSearchCV(estimator=model, \n                     param_grid=hyperparam_grid, \n                     scoring='average_precision', \n                     cv=kfold, \n                     return_train_score=True, \n                     n_jobs=-1, \n                     verbose=1)\n\n# Fit the model with grid search on the subset of data\ngs_cv.fit(X_subset, y_subset)\n\n# Extract best parameters\nbest_params = gs_cv.best_params_","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:13:18.745330Z","iopub.execute_input":"2024-07-02T18:13:18.746125Z","iopub.status.idle":"2024-07-02T18:47:42.293180Z","shell.execute_reply.started":"2024-07-02T18:13:18.746092Z","shell.execute_reply":"2024-07-02T18:47:42.291567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Optimal params for model :')\nprint(best_params)\nprint('-'*66)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:49:21.614762Z","iopub.execute_input":"2024-07-02T18:49:21.615645Z","iopub.status.idle":"2024-07-02T18:49:21.621884Z","shell.execute_reply.started":"2024-07-02T18:49:21.615604Z","shell.execute_reply":"2024-07-02T18:49:21.620665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_full = pp_data.drop(columns=['id', 'binds', 'protein_name']).values\ny_full = pp_data['binds'].values\n","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:50:26.730288Z","iopub.execute_input":"2024-07-02T18:50:26.730744Z","iopub.status.idle":"2024-07-02T18:50:27.435305Z","shell.execute_reply.started":"2024-07-02T18:50:26.730711Z","shell.execute_reply":"2024-07-02T18:50:27.434179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train final model on the entire dataset with best hyperparameters\nX_train_full, X_test_full, y_train_full, y_test_full = train_test_split(X_full, y_full, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:50:35.580485Z","iopub.execute_input":"2024-07-02T18:50:35.581249Z","iopub.status.idle":"2024-07-02T18:50:47.057685Z","shell.execute_reply.started":"2024-07-02T18:50:35.581215Z","shell.execute_reply":"2024-07-02T18:50:47.056440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Train final model on the entire dataset with best hyperparameters\nfinal_model = XGBClassifier(**best_params, use_label_encoder=False, eval_metric='logloss')\nfinal_model.fit(X_train_full, y_train_full)\n","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:50:52.801419Z","iopub.execute_input":"2024-07-02T18:50:52.801798Z","iopub.status.idle":"2024-07-02T18:58:19.026164Z","shell.execute_reply.started":"2024-07-02T18:50:52.801772Z","shell.execute_reply":"2024-07-02T18:58:19.024964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\n\ny_pred = final_model.predict(X_test_full)\ny_pred_proba = final_model.predict_proba(X_test_full)[:, 1]\n\naccuracy = accuracy_score(y_test_full, y_pred)\nroc_auc = roc_auc_score(y_test_full, y_pred_proba)\n\n\nprint(f'Accuracy: {accuracy}')\nprint(f'ROC AUC: {roc_auc}')","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:58:55.621892Z","iopub.execute_input":"2024-07-02T18:58:55.622594Z","iopub.status.idle":"2024-07-02T18:59:01.980064Z","shell.execute_reply.started":"2024-07-02T18:58:55.622560Z","shell.execute_reply":"2024-07-02T18:59:01.978932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv_scores = cross_val_score(final_model, X_train_full, y_train_full, cv=nfolds, scoring='average_precision')\nprint('Cross validation scores')\nprint('Average precision:', str(round(cv_scores.mean(), 3)))\nprint('-'*40)","metadata":{"execution":{"iopub.status.busy":"2024-07-01T17:19:24.145328Z","iopub.execute_input":"2024-07-01T17:19:24.146036Z","iopub.status.idle":"2024-07-01T17:32:28.430861Z","shell.execute_reply.started":"2024-07-01T17:19:24.146000Z","shell.execute_reply":"2024-07-01T17:32:28.429448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib\njoblib.dump(final_model, f'best_XGB_model.pkl')","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:59:21.705228Z","iopub.execute_input":"2024-07-02T18:59:21.706166Z","iopub.status.idle":"2024-07-02T18:59:21.797003Z","shell.execute_reply.started":"2024-07-02T18:59:21.706129Z","shell.execute_reply":"2024-07-02T18:59:21.796005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc, RocCurveDisplay\n\n# Compute the ROC curve and AUC score\nfpr, tpr, thresholds = roc_curve(y_test_full, y_pred_proba)\nroc_auc = auc(fpr, tpr)\n\n# Plot the ROC curve\nplt.figure()\nplt.plot(fpr, tpr, color='darkorange', lw=2, label=f'ROC curve (area = {roc_auc:.2f})')\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver Operating Characteristic (ROC) Curve')\nplt.legend(loc='lower right')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:59:31.218518Z","iopub.execute_input":"2024-07-02T18:59:31.218888Z","iopub.status.idle":"2024-07-02T18:59:31.563964Z","shell.execute_reply.started":"2024-07-02T18:59:31.218861Z","shell.execute_reply":"2024-07-02T18:59:31.562899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Split the pre processed data into three files for each prrotein","metadata":{}},{"cell_type":"code","source":"df = pp_data[pp_data['protein_name'] == 'HSA']\ndf.to_parquet(\"/kaggle/working/train_HSA.parquet\", index=False)\n\ndf = pp_data[pp_data['protein_name'] == 'BRD4']\ndf.to_parquet(\"/kaggle/working/train_BRD4.parquet\", index=False)\n\ndf = pp_data[pp_data['protein_name'] == 'sEH']\ndf.to_parquet(\"/kaggle/working/train_sEH.parquet\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T18:59:45.786091Z","iopub.execute_input":"2024-07-02T18:59:45.786505Z","iopub.status.idle":"2024-07-02T18:59:58.543764Z","shell.execute_reply.started":"2024-07-02T18:59:45.786473Z","shell.execute_reply":"2024-07-02T18:59:58.542809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T19:00:59.911783Z","iopub.execute_input":"2024-07-02T19:00:59.912196Z","iopub.status.idle":"2024-07-02T19:01:00.091854Z","shell.execute_reply.started":"2024-07-02T19:00:59.912163Z","shell.execute_reply":"2024-07-02T19:01:00.090557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_train_full, X_test_full, y_train_full, y_test_full\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T19:02:43.870144Z","iopub.execute_input":"2024-07-02T19:02:43.870617Z","iopub.status.idle":"2024-07-02T19:02:44.013569Z","shell.execute_reply.started":"2024-07-02T19:02:43.870580Z","shell.execute_reply":"2024-07-02T19:02:44.012424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_path = os.path.join(original_data_dir, test_file)\ntest_df = pd.read_csv(test_path)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T19:02:53.066919Z","iopub.execute_input":"2024-07-02T19:02:53.067309Z","iopub.status.idle":"2024-07-02T19:02:59.292848Z","shell.execute_reply.started":"2024-07-02T19:02:53.067279Z","shell.execute_reply":"2024-07-02T19:02:59.291913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns_to_drop = ['buildingblock1_smiles', 'buildingblock2_smiles', \n                   'buildingblock3_smiles']","metadata":{"execution":{"iopub.status.busy":"2024-07-02T19:03:07.912718Z","iopub.execute_input":"2024-07-02T19:03:07.913669Z","iopub.status.idle":"2024-07-02T19:03:07.918016Z","shell.execute_reply.started":"2024-07-02T19:03:07.913634Z","shell.execute_reply":"2024-07-02T19:03:07.917006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tset_df = test_df.drop(columns=columns_to_drop, inplace=True, axis=1)\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T19:03:13.056957Z","iopub.execute_input":"2024-07-02T19:03:13.057595Z","iopub.status.idle":"2024-07-02T19:03:13.147654Z","shell.execute_reply.started":"2024-07-02T19:03:13.057565Z","shell.execute_reply":"2024-07-02T19:03:13.146617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" # Generate molecular fingerprints\nPreProc_test_file = os.path.join(working_data_dir, pptest_file)\n\nif not os.path.exists(PreProc_test_file):\n    mfp = np.array(generate_morgan_fingerprints(test_df['molecule_smiles']), dtype=np.uint8)\n\n    # Convert molecular fingerprints to separate columns with uint8 data type\n    mdf = pd.DataFrame(mfp, dtype=np.uint8)\n    mdf.columns = [f'fp_{i+1:04d}' for i in range(mfp.shape[1])]\n    pptest_data = pd.concat([test_df.reset_index(drop=True), mdf], axis=1)\n    pptest_data.to_parquet(pptest_file, index=False)\nelse:\n    pptest_data = pd.read_parquet(PreProc_test_file)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T19:03:37.387677Z","iopub.execute_input":"2024-07-02T19:03:37.388439Z","iopub.status.idle":"2024-07-02T19:20:48.210934Z","shell.execute_reply.started":"2024-07-02T19:03:37.388402Z","shell.execute_reply":"2024-07-02T19:20:48.209825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pptest_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T19:22:15.903558Z","iopub.execute_input":"2024-07-02T19:22:15.904431Z","iopub.status.idle":"2024-07-02T19:22:15.923621Z","shell.execute_reply.started":"2024-07-02T19:22:15.904395Z","shell.execute_reply":"2024-07-02T19:22:15.922548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_df, mdf\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T19:22:25.579870Z","iopub.execute_input":"2024-07-02T19:22:25.580291Z","iopub.status.idle":"2024-07-02T19:22:25.712592Z","shell.execute_reply.started":"2024-07-02T19:22:25.580260Z","shell.execute_reply":"2024-07-02T19:22:25.711443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Xtest = pptest_data.drop(columns=['id', 'molecule_smiles', 'protein_name']).values\npredictions = final_model.predict_proba(Xtest)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2024-07-02T19:22:57.045463Z","iopub.execute_input":"2024-07-02T19:22:57.046464Z","iopub.status.idle":"2024-07-02T19:23:38.605585Z","shell.execute_reply.started":"2024-07-02T19:22:57.046428Z","shell.execute_reply":"2024-07-02T19:23:38.604212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions.max(), predictions.min()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T19:23:42.999880Z","iopub.execute_input":"2024-07-02T19:23:43.000301Z","iopub.status.idle":"2024-07-02T19:23:43.008520Z","shell.execute_reply.started":"2024-07-02T19:23:43.000270Z","shell.execute_reply":"2024-07-02T19:23:43.007228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib as mpl\n\nplt.figure(figsize=(10, 6))\n\n# Get the original colormap\n#orig_cmap = mpl.colormaps('viridis')\n\n# Plot the histogram with 2 bins\nplt.hist(predictions, bins=6, alpha=0.75, color='green')\n\n# Set the x-axis limits to clearly show the two bins\nplt.xlim(0, 1.0)\n\nplt.title('Histogram of Predicted Probabilities')\nplt.xlabel('Predicted Probability')\nplt.ylabel('Frequency')\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-07-02T19:23:51.438000Z","iopub.execute_input":"2024-07-02T19:23:51.439023Z","iopub.status.idle":"2024-07-02T19:23:51.735887Z","shell.execute_reply.started":"2024-07-02T19:23:51.438986Z","shell.execute_reply":"2024-07-02T19:23:51.734821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sumission_df = {}\n        \n# Create the sub DataFrame for the current chunk\nsub_df = {\n    'id': pptest_data['id'],\n    'binds': predictions\n}\nsubmission_df = pd.DataFrame(sub_df)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T19:24:03.804495Z","iopub.execute_input":"2024-07-02T19:24:03.804900Z","iopub.status.idle":"2024-07-02T19:24:03.815136Z","shell.execute_reply.started":"2024-07-02T19:24:03.804867Z","shell.execute_reply":"2024-07-02T19:24:03.814162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"length of submisison fiule : \", submission_df.shape)\nprint(submission_df.head())","metadata":{"execution":{"iopub.status.busy":"2024-07-02T19:24:06.963589Z","iopub.execute_input":"2024-07-02T19:24:06.964490Z","iopub.status.idle":"2024-07-02T19:24:06.971855Z","shell.execute_reply.started":"2024-07-02T19:24:06.964457Z","shell.execute_reply":"2024-07-02T19:24:06.970821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-07-02T19:24:11.182647Z","iopub.execute_input":"2024-07-02T19:24:11.183617Z","iopub.status.idle":"2024-07-02T19:24:14.074240Z","shell.execute_reply.started":"2024-07-02T19:24:11.183578Z","shell.execute_reply":"2024-07-02T19:24:14.073159Z"},"trusted":true},"execution_count":null,"outputs":[]}]}