{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-02T12:58:17.979259Z","iopub.execute_input":"2024-05-02T12:58:17.979934Z","iopub.status.idle":"2024-05-02T12:58:19.012175Z","shell.execute_reply.started":"2024-05-02T12:58:17.979904Z","shell.execute_reply":"2024-05-02T12:58:19.011221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/leash-BELKA/test.csv')\ntest_df","metadata":{"execution":{"iopub.status.busy":"2024-05-02T12:58:19.014053Z","iopub.execute_input":"2024-05-02T12:58:19.014427Z","iopub.status.idle":"2024-05-02T12:58:25.307141Z","shell.execute_reply.started":"2024-05-02T12:58:19.014401Z","shell.execute_reply":"2024-05-02T12:58:25.306334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#columns = ['id' , 'molecule_smiles' , 'protein_name' , 'binds']","metadata":{"execution":{"iopub.status.busy":"2024-05-02T12:58:25.308518Z","iopub.execute_input":"2024-05-02T12:58:25.308937Z","iopub.status.idle":"2024-05-02T12:58:25.313335Z","shell.execute_reply.started":"2024-05-02T12:58:25.308902Z","shell.execute_reply":"2024-05-02T12:58:25.312473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Specify the path to your large CSV file\nfilename = '/kaggle/input/leash-BELKA/train.csv'\n\n# Read the first 3,000,000 rows\ntrain_df = pd.read_csv(filename, nrows=3000000)\n\n# Now 'df' contains the first 3,000,000 rows of your data\nprint(train_df.head()) ","metadata":{"execution":{"iopub.status.busy":"2024-05-02T12:58:25.314546Z","iopub.execute_input":"2024-05-02T12:58:25.314834Z","iopub.status.idle":"2024-05-02T12:58:35.462028Z","shell.execute_reply.started":"2024-05-02T12:58:25.314797Z","shell.execute_reply":"2024-05-02T12:58:35.461086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2024-05-02T12:58:35.464543Z","iopub.execute_input":"2024-05-02T12:58:35.464886Z","iopub.status.idle":"2024-05-02T12:58:35.480891Z","shell.execute_reply.started":"2024-05-02T12:58:35.464855Z","shell.execute_reply":"2024-05-02T12:58:35.480046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_uniques(col):\n    s = 0\n    r = []\n    names = []\n    l = list(col.unique())\n    for i in l:\n        a = col.value_counts()[i]\n        r.append(a)\n        s += a\n        names.append(i)\n    print(f\"\\n******     Report    ******\\n\")\n    print(f\"Sum = {s}\")\n    for j in range(0 , len(names)):\n        print(f\"Count {names[j]} = {r[j]}\")\n        print(f\"Percent {names[j]} = {'%.2f' %(((r[j])/s)*100)}%\")","metadata":{"execution":{"iopub.status.busy":"2024-05-02T12:58:35.481901Z","iopub.execute_input":"2024-05-02T12:58:35.482302Z","iopub.status.idle":"2024-05-02T12:58:35.488569Z","shell.execute_reply.started":"2024-05-02T12:58:35.482279Z","shell.execute_reply":"2024-05-02T12:58:35.487595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"molecule_smiles = []\nfor i in train_df['molecule_smiles']:\n    molecule_smiles.append(i)\n            \nprint(f'{len(pd.Series(molecule_smiles).unique())} / {len(molecule_smiles)}')","metadata":{"execution":{"iopub.status.busy":"2024-05-02T12:58:35.489786Z","iopub.execute_input":"2024-05-02T12:58:35.490194Z","iopub.status.idle":"2024-05-02T12:58:36.994552Z","shell.execute_reply.started":"2024-05-02T12:58:35.490160Z","shell.execute_reply":"2024-05-02T12:58:36.993627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **STOUT V2.0 - Smiles TO iUpac Translator Version 2.0**\nThis repository contains STOUT-V2, SMILES to IUPAC name translator using transformers. STOUT-V2 can translate SMILES to IUPAC names and IUPAC names back to a valid SMILES string.","metadata":{}},{"cell_type":"code","source":"!pip install rdkit","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-05-02T12:58:36.995607Z","iopub.execute_input":"2024-05-02T12:58:36.995883Z","iopub.status.idle":"2024-05-02T12:58:52.832641Z","shell.execute_reply.started":"2024-05-02T12:58:36.995859Z","shell.execute_reply":"2024-05-02T12:58:52.831519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**In drug molecules, various chemical properties such as atoms, bonds, molecular weights, logPs, hydrogen bond acceptors and donors, topological polar surface area (TPSA), and rotatable bonds play significant roles in determining their behavior, effectiveness, and pharmacokinetics.**\n\n1. ***Atoms and Bonds***: The atoms and bonds in a drug molecule determine its chemical structure, which in turn influences its properties and biological activities. Different arrangements of atoms and bonds can lead to different spatial orientations, affecting the molecule's binding affinity to target proteins and its selectivity.\n2. ***Molecular Weights***: Molecular weight, also known as molecular mass, is the mass of a molecule calculated as the sum of the atomic weights of its constituent atoms. In drug discovery and development, molecular weight is an important parameter to consider, as it can influence the molecule's absorption, distribution, metabolism, excretion, and toxicity (ADMET) properties.\n3. ***LogP (Partition Coefficient)***: LogP, or the partition coefficient, is a measure of a drug molecule's lipophilicity (fat-solubility) and hydrophilicity (water-solubility). It is the ratio of concentrations of a compound in a two-phase system consisting of octanol (representing lipids) and water. LogP can influence drug permeability, absorption, and distribution in the body.\n4. ***Hydrogen Bond Acceptors and Donors (H-bond Acceptors and Donors)***: These are atoms or functional groups in a molecule that can form hydrogen bonds with other molecules. Hydrogen bond acceptors are atoms or groups with a lone pair of electrons that can accept a hydrogen bond, while hydrogen bond donors are atoms or groups with a hydrogen atom bonded to an electronegative atom. H-bond acceptors and donors play a crucial role in molecular recognition, binding affinity, and specificity of drug molecules to their targets.\n5. ***Topological Polar Surface Area (TPSA)***: TPSA is a measure of the molecule's polar surface area, which is the sum of the surfaces of all polar atoms (usually oxygen and nitrogen). It is an essential parameter in drug design, as it correlates with a molecule's ability to permeate cell membranes and participate in hydrogen bonding interactions. TPSA is often used to assess drug-likeness, bioavailability, and permeability.\n6. ***Rotatable Bonds***: Rotatable bonds are single bonds that can rotate freely, allowing different conformations of a molecule. The number of rotatable bonds in a drug molecule can influence its flexibility, shape, and binding properties. Molecules with fewer rotatable bonds are generally more rigid and easier to optimize for binding affinity, whereas those with more rotatable bonds can adopt various conformations, potentially affecting their binding selectivity and pharmacological properties.\n\nIn summary, understanding and optimizing these chemical properties in drug molecules are crucial in drug discovery and development processes, ultimately affecting their pharmacological behavior and therapeutic effectiveness.","metadata":{}},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import AllChem, MolFromSmiles\nfrom rdkit.Chem.Descriptors import ExactMolWt \nfrom rdkit.Chem.rdMolDescriptors import CalcNumHBA , CalcNumHBD , CalcTPSA , CalcNumRotatableBonds\nfrom rdkit.Chem.Crippen import MolLogP \n \n\n\ndef count_atoms_bonds_charges(smiles):\n    # Parse SMILES into RDKit Mol object\n    mol = MolFromSmiles(smiles)    \n    # Calculate the number of atoms and bonds\n    atom_count = mol.GetNumAtoms()\n    bond_count = mol.GetNumBonds()\n    # Calculate the molecular weight\n    molecular_weight = ExactMolWt(mol)\n    molecular_weight = round(molecular_weight, 2)\n\n    logP = MolLogP(mol)\n    logP = round(logP , 2)\n    hb_acceptor = CalcNumHBA(mol)\n    hb_donor = CalcNumHBD(mol)\n    \n    tpsa = CalcTPSA(mol)\n    \n    rotatable_bond = CalcNumRotatableBonds(mol)\n    \n    # Calculate the number of charges (positive and negative)\n    try:\n        charges = [atom.GetProp('_Charge') for atom in mol.GetAtoms()]\n        pos_charges = len([c for c in charges if c > 0])\n        neg_charges = len([c for c in charges if c < 0])\n        \n        \n    except KeyError:\n        pos_charges = 0\n        neg_charges = 0\n    \n    return {\n        'atoms': atom_count,\n        'bonds': bond_count,\n        'positive_charges': pos_charges,\n        'negative_charges': neg_charges,\n        'molecular_weight' : molecular_weight,\n        'LogP' : logP ,\n        'HBA' : hb_acceptor ,\n        'HBD' : hb_donor,\n        'TPSA' : tpsa ,\n        'Rotatable Bonds' : rotatable_bond\n    }","metadata":{"execution":{"iopub.status.busy":"2024-05-02T12:58:52.834635Z","iopub.execute_input":"2024-05-02T12:58:52.835094Z","iopub.status.idle":"2024-05-02T12:58:53.107153Z","shell.execute_reply.started":"2024-05-02T12:58:52.835052Z","shell.execute_reply":"2024-05-02T12:58:53.106363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# block1 = df['buildingblock1_smiles']\n# r = []\n# for i in block1[0:1000]:\n#     result = count_atoms_bonds_charges(i)\n#     r.append(result)\n# for i in block1[1000:5000]:\n#     result = count_atoms_bonds_charges(i)\n#     r.append(result)\n    \n# for i in block1[5000:10000]:\n#     result = count_atoms_bonds_charges(i)\n#     r.append(result)\n\n# for i in block1[10000:50000]:\n#     result = count_atoms_bonds_charges(i)\n#     r.append(result)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-05-02T12:58:53.108302Z","iopub.execute_input":"2024-05-02T12:58:53.108575Z","iopub.status.idle":"2024-05-02T12:58:53.113128Z","shell.execute_reply.started":"2024-05-02T12:58:53.108552Z","shell.execute_reply":"2024-05-02T12:58:53.112169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data = r\n# ids = df['id'][0:50000]\n# atoms = []\n# bonds = []\n# positive_charges = []\n# negative_charges = []\n# ndfids = []\n# protein_names = []\n# molecular_weights = []\n# logPs = []\n# hb_acceptors = []\n# hb_donors = []\n# tpsas = [] \n# rotatable_bonds = []\n\n\n\n            \n# for i in range(0,len(data)):\n#     ndfids.append(ids[i])\n#     atoms.append(data[i]['atoms'])\n#     bonds.append(data[i]['bonds'])\n#     positive_charges.append(data[i]['positive_charges'])\n#     negative_charges.append(data[i]['negative_charges'])\n#     protein_names.append(df['protein_name'][i])\n#     molecular_weights.append(data[i]['molecular_weight'])\n#     logPs.append(data[i]['LogP'])\n#     hb_acceptors.append(data[i]['HBA'])\n#     hb_donors.append(data[i]['HBD'])\n#     tpsas.append(data[i]['TPSA'])\n#     rotatable_bonds.append(data[i]['Rotatable Bonds'])\n    \n    \n# df_block1 = pd.DataFrame()\n# df_block1['ids'] = ndfids\n# df_block1['protein names'] = protein_names\n# df_block1['Atoms'] = atoms\n# df_block1['Bonds'] = bonds \n# df_block1['molecular weight'] = molecular_weights\n# df_block1['LogP'] = logPs\n# df_block1['HBA'] = hb_acceptors\n# df_block1['HBD'] = hb_donors\n# df_block1['TPSA'] = tpsas\n# df_block1['Rotatable Bonds'] = rotatable_bonds\n\n# #there was no charge possitive or negative in these 5000 samples in molecule_smiles column\n# # df_block1['negative_charges'] = negative_charges \n# # df_block1['positive_charges'] = positive_charges\n\n\n# df_block1","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-05-02T12:58:53.114237Z","iopub.execute_input":"2024-05-02T12:58:53.114502Z","iopub.status.idle":"2024-05-02T12:58:53.125777Z","shell.execute_reply.started":"2024-05-02T12:58:53.114471Z","shell.execute_reply":"2024-05-02T12:58:53.124743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Split 3,000,000 Train data for train and test preprocessing**","metadata":{}},{"cell_type":"code","source":"pd.DataFrame(train_df)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T13:19:39.649805Z","iopub.execute_input":"2024-05-02T13:19:39.650549Z","iopub.status.idle":"2024-05-02T13:19:39.667032Z","shell.execute_reply.started":"2024-05-02T13:19:39.650518Z","shell.execute_reply":"2024-05-02T13:19:39.665851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nr = []\nfor i in tqdm(molecule_smiles[0:500000]):\n    result = count_atoms_bonds_charges(i)\n    r.append(result)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T16:02:08.925216Z","iopub.execute_input":"2024-05-02T16:02:08.925982Z","iopub.status.idle":"2024-05-02T16:14:34.828470Z","shell.execute_reply.started":"2024-05-02T16:02:08.925950Z","shell.execute_reply":"2024-05-02T16:14:34.827513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(molecule_smiles[500000:1000000]):\n    result = count_atoms_bonds_charges(i)\n    r.append(result)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T16:14:34.830178Z","iopub.execute_input":"2024-05-02T16:14:34.830467Z","iopub.status.idle":"2024-05-02T16:27:02.530651Z","shell.execute_reply.started":"2024-05-02T16:14:34.830441Z","shell.execute_reply":"2024-05-02T16:27:02.529564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(molecule_smiles[1000000:1500000]):\n    result = count_atoms_bonds_charges(i)\n    r.append(result)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T16:27:02.532007Z","iopub.execute_input":"2024-05-02T16:27:02.532303Z","iopub.status.idle":"2024-05-02T16:39:12.929503Z","shell.execute_reply.started":"2024-05-02T16:27:02.532277Z","shell.execute_reply":"2024-05-02T16:39:12.928622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(molecule_smiles[1500000:2000000]):\n    result = count_atoms_bonds_charges(i)\n    r.append(result)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T16:39:12.931369Z","iopub.execute_input":"2024-05-02T16:39:12.931651Z","iopub.status.idle":"2024-05-02T16:51:22.395153Z","shell.execute_reply.started":"2024-05-02T16:39:12.931626Z","shell.execute_reply":"2024-05-02T16:51:22.394079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(molecule_smiles[2000000:2500000]):\n    result = count_atoms_bonds_charges(i)\n    r.append(result)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T16:52:26.430347Z","iopub.execute_input":"2024-05-02T16:52:26.430698Z","iopub.status.idle":"2024-05-02T17:04:48.152966Z","shell.execute_reply.started":"2024-05-02T16:52:26.430673Z","shell.execute_reply":"2024-05-02T17:04:48.152089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(r)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T17:05:34.710722Z","iopub.execute_input":"2024-05-02T17:05:34.711332Z","iopub.status.idle":"2024-05-02T17:05:34.717016Z","shell.execute_reply.started":"2024-05-02T17:05:34.711303Z","shell.execute_reply":"2024-05-02T17:05:34.716075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(molecule_smiles[2500000:3000000]):\n    result = count_atoms_bonds_charges(i)\n    r.append(result)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T17:05:37.649568Z","iopub.execute_input":"2024-05-02T17:05:37.650463Z","iopub.status.idle":"2024-05-02T17:18:11.588061Z","shell.execute_reply.started":"2024-05-02T17:05:37.650431Z","shell.execute_reply":"2024-05-02T17:18:11.587130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list(train_df['binds'][2500000:].unique())","metadata":{"execution":{"iopub.status.busy":"2024-05-02T17:18:11.596921Z","iopub.execute_input":"2024-05-02T17:18:11.597191Z","iopub.status.idle":"2024-05-02T17:18:11.625176Z","shell.execute_reply.started":"2024-05-02T17:18:11.597161Z","shell.execute_reply":"2024-05-02T17:18:11.624186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(r)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T17:18:11.589804Z","iopub.execute_input":"2024-05-02T17:18:11.590108Z","iopub.status.idle":"2024-05-02T17:18:11.595860Z","shell.execute_reply.started":"2024-05-02T17:18:11.590084Z","shell.execute_reply":"2024-05-02T17:18:11.595003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = r\nids = train_df['id']\natoms = []\nbonds = []\npositive_charges = []\nnegative_charges = []\nndfids = []\nprotein_names = []\nmolecular_weights = []\nlogPs = []\nhb_acceptors = []\nhb_donors = []\ntpsas = [] \nrotatable_bonds = []\nbinds = []\n\nfor i in tqdm(range(0,len(data))):\n    ndfids.append(ids[i])\n    atoms.append(data[i]['atoms'])\n    bonds.append(data[i]['bonds'])\n    positive_charges.append(data[i]['positive_charges'])\n    negative_charges.append(data[i]['negative_charges'])\n    protein_names.append(train_df['protein_name'][i])\n    molecular_weights.append(data[i]['molecular_weight'])\n    logPs.append(data[i]['LogP'])\n    hb_acceptors.append(data[i]['HBA'])\n    hb_donors.append(data[i]['HBD'])\n    tpsas.append(data[i]['TPSA'])\n    rotatable_bonds.append(data[i]['Rotatable Bonds'])\n    \n    \ndf_molecule_smiles = pd.DataFrame()\ndf_molecule_smiles['ids'] = ndfids\ndf_molecule_smiles['protein_names'] = protein_names\ndf_molecule_smiles['Atoms'] = atoms\ndf_molecule_smiles['Bonds'] = bonds \ndf_molecule_smiles['molecular weight'] = molecular_weights\ndf_molecule_smiles['LogP'] = logPs\ndf_molecule_smiles['HBA'] = hb_acceptors\ndf_molecule_smiles['HBD'] = hb_donors\ndf_molecule_smiles['TPSA'] = tpsas\ndf_molecule_smiles['Rotatable Bonds'] = rotatable_bonds\ndf_molecule_smiles['binds'] = train_df['binds']\n\ndf_moleculesmiles = df_molecule_smiles\ndf_molecule_smiles","metadata":{"execution":{"iopub.status.busy":"2024-05-02T17:20:17.406780Z","iopub.execute_input":"2024-05-02T17:20:17.407148Z","iopub.status.idle":"2024-05-02T17:21:38.736360Z","shell.execute_reply.started":"2024-05-02T17:20:17.407122Z","shell.execute_reply":"2024-05-02T17:21:38.735444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_molecule_smiles['binds'] = train_df['binds']","metadata":{"execution":{"iopub.status.busy":"2024-05-02T17:24:02.031775Z","iopub.execute_input":"2024-05-02T17:24:02.032575Z","iopub.status.idle":"2024-05-02T17:24:02.041509Z","shell.execute_reply.started":"2024-05-02T17:24:02.032546Z","shell.execute_reply":"2024-05-02T17:24:02.040578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_moleculesmiles.loc[:,'Atoms':'binds']","metadata":{"execution":{"iopub.status.busy":"2024-05-02T17:24:06.269378Z","iopub.execute_input":"2024-05-02T17:24:06.270109Z","iopub.status.idle":"2024-05-02T17:24:06.398730Z","shell.execute_reply.started":"2024-05-02T17:24:06.270077Z","shell.execute_reply":"2024-05-02T17:24:06.397890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature selection","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nimport matplotlib.pyplot as plt\n\nX = df_moleculesmiles.loc[0:1000000:,'Atoms':'Rotatable Bonds']\ny = df_moleculesmiles['binds'][0:1000001]\n\nrf_classifier = RandomForestClassifier()\nrf_classifier.fit(X, y)\n\nrf_importances = rf_classifier.feature_importances_\nrf_sorted_indices = np.argsort(rf_importances)[::-1]\n\n# Select features with non-zero importance\nthreshold = 0.02  # Set a threshold for feature importance\nrf_selected_features = X.columns[rf_importances > threshold]\nrf_X_filtered = X[rf_selected_features]\n\n# Plot feature importances\nplt.figure(figsize=(10, 6))\nplt.title(\"Random Forest Feature Importances\")\nplt.bar(range(X.shape[1]), rf_importances[rf_sorted_indices], align='center')\nplt.xticks(range(X.shape[1]), X.columns[rf_sorted_indices], rotation=90)\nplt.xlabel(\"Feature\")\nplt.ylabel(\"Feature Importance\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T17:47:59.894884Z","iopub.execute_input":"2024-05-02T17:47:59.895565Z","iopub.status.idle":"2024-05-02T17:50:15.703942Z","shell.execute_reply.started":"2024-05-02T17:47:59.895531Z","shell.execute_reply":"2024-05-02T17:50:15.702719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingClassifier\n\ngb_classifier = GradientBoostingClassifier()\n\ngb_classifier.fit(X, y)\n\ngb_importances = gb_classifier.feature_importances_\n\ngb_sorted_indices = np.argsort(gb_importances)[::-1]\n# Select features with non-zero importance\nthreshold = 0.02  # Set a threshold for feature importance\ngb_selected_features = X.columns[gb_importances > threshold]\n\n# Filter X to keep only the selected features\ngb_X_filtered = X[gb_selected_features]\nplt.figure(figsize=(10, 6))\nplt.title(\"Gradient Boosting Feature Importances\")\nplt.bar(range(X.shape[1]), gb_importances[gb_sorted_indices], align='center')\nplt.xticks(range(X.shape[1]), X.columns[gb_sorted_indices], rotation=90)\nplt.xlabel(\"Feature\")\nplt.ylabel(\"Feature Importance\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T12:58:53.163099Z","iopub.status.idle":"2024-05-02T12:58:53.163406Z","shell.execute_reply.started":"2024-05-02T12:58:53.163261Z","shell.execute_reply":"2024-05-02T12:58:53.163273Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# find most possitive features to work on them\n\nmx = gb_importances[gb_sorted_indices].max()\nmn = gb_importances[gb_sorted_indices].min()\nn = X.columns[gb_sorted_indices]\ns = gb_importances[gb_sorted_indices]\nprint('\\n***** Feature Importance of Gradient Boost Classifier *****\\n')\nfor i in range(0 , len(s)):      \n    if s[i] == mx:\n        print(f'{n[i]} = {s[i]} ,\\n{n[i+1]} = {s[i+1]} ,\\n{n[i+2]} = {s[i+2]}')\n        break\n    else:\n        continue","metadata":{"execution":{"iopub.status.busy":"2024-05-02T12:58:53.164919Z","iopub.status.idle":"2024-05-02T12:58:53.165228Z","shell.execute_reply.started":"2024-05-02T12:58:53.165073Z","shell.execute_reply":"2024-05-02T12:58:53.165086Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mx = rf_importances[rf_sorted_indices].max()\nmn = rf_importances[rf_sorted_indices].min()\nn = X.columns[rf_sorted_indices]\ns = rf_importances[rf_sorted_indices]\nprint('\\n***** Feature Importance of Random Forest *****\\n')\nfor i in range(0 , len(s)):      \n    if s[i] == mx:\n        print(f'{n[i]} = {s[i]} ,\\n{n[i+1]} = {s[i+1]} ,\\n{n[i+2]} = {s[i+2]}')\n        break\n    else:\n        continue","metadata":{"execution":{"iopub.status.busy":"2024-05-02T17:57:06.142083Z","iopub.execute_input":"2024-05-02T17:57:06.142718Z","iopub.status.idle":"2024-05-02T17:57:06.149411Z","shell.execute_reply.started":"2024-05-02T17:57:06.142685Z","shell.execute_reply":"2024-05-02T17:57:06.148516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"df_moleculesmiles","metadata":{"execution":{"iopub.status.busy":"2024-05-02T12:58:53.168672Z","iopub.status.idle":"2024-05-02T12:58:53.169014Z","shell.execute_reply.started":"2024-05-02T12:58:53.168854Z","shell.execute_reply":"2024-05-02T12:58:53.168871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nsns.heatmap(df_moleculesmiles.loc[0:1000000:,'Atoms':'Rotatable Bonds'].corr(),annot=True)\nplt.title('Molecule SMILES')\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-02T17:58:23.890546Z","iopub.execute_input":"2024-05-02T17:58:23.891096Z","iopub.status.idle":"2024-05-02T17:58:24.694722Z","shell.execute_reply.started":"2024-05-02T17:58:23.891057Z","shell.execute_reply":"2024-05-02T17:58:24.693759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\nsns.heatmap(df_moleculesmiles.loc[0:1000000:,'Atoms':'Rotatable Bonds'].corr() < 0.7,annot=True)\nplt.title('Molecule SMILES')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-02T17:57:59.173258Z","iopub.execute_input":"2024-05-02T17:57:59.173611Z","iopub.status.idle":"2024-05-02T17:58:00.101161Z","shell.execute_reply.started":"2024-05-02T17:57:59.173582Z","shell.execute_reply":"2024-05-02T17:58:00.100334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, MinMaxScaler\n\n# Standardization (Z-score normalization)\nscaler = StandardScaler()\ngb_X_scaled_standardized = scaler.fit_transform(gb_X_filtered)\n\n# Normalization (Min-Max scaling)\nmin_max_scaler = MinMaxScaler()\ngb_X_scaled_normalized = min_max_scaler.fit_transform(X)\n\ngb_X_filtered.columns ","metadata":{"execution":{"iopub.status.busy":"2024-05-02T12:58:53.173279Z","iopub.status.idle":"2024-05-02T12:58:53.173731Z","shell.execute_reply.started":"2024-05-02T12:58:53.173479Z","shell.execute_reply":"2024-05-02T12:58:53.173525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler, MinMaxScaler\n\n\nscaler = StandardScaler()\n\nrf_X_scaled_standardized = scaler.fit_transform(rf_X_filtered)\n\n# Normalization (Min-Max scaling)\nmin_max_scaler = MinMaxScaler()\nrf_X_scaled_normalized = min_max_scaler.fit_transform(X)\n\nrf_X_filtered.columns","metadata":{"execution":{"iopub.status.busy":"2024-05-02T17:59:40.309464Z","iopub.execute_input":"2024-05-02T17:59:40.309842Z","iopub.status.idle":"2024-05-02T17:59:40.541611Z","shell.execute_reply.started":"2024-05-02T17:59:40.309795Z","shell.execute_reply":"2024-05-02T17:59:40.540643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_copy= df_molecule_smiles\ndf_f=df_copy[['Bonds','Atoms','molecular weight', 'LogP', 'HBA', 'TPSA','Rotatable Bonds']]","metadata":{"execution":{"iopub.status.busy":"2024-05-02T17:59:54.290518Z","iopub.execute_input":"2024-05-02T17:59:54.291276Z","iopub.status.idle":"2024-05-02T17:59:54.381927Z","shell.execute_reply.started":"2024-05-02T17:59:54.291239Z","shell.execute_reply":"2024-05-02T17:59:54.380853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_f.columns","metadata":{"execution":{"iopub.status.busy":"2024-05-02T17:59:58.090764Z","iopub.execute_input":"2024-05-02T17:59:58.091128Z","iopub.status.idle":"2024-05-02T17:59:58.097374Z","shell.execute_reply.started":"2024-05-02T17:59:58.091098Z","shell.execute_reply":"2024-05-02T17:59:58.096504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_f.loc[:, 'ids'] = df_molecule_smiles['ids']\ndf_f.loc[:, 'binds'] = df_molecule_smiles['binds']","metadata":{"execution":{"iopub.status.busy":"2024-05-02T18:05:35.345644Z","iopub.execute_input":"2024-05-02T18:05:35.346021Z","iopub.status.idle":"2024-05-02T18:05:35.374268Z","shell.execute_reply.started":"2024-05-02T18:05:35.345991Z","shell.execute_reply":"2024-05-02T18:05:35.373349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_f.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T18:05:38.112769Z","iopub.execute_input":"2024-05-02T18:05:38.113569Z","iopub.status.idle":"2024-05-02T18:05:38.125651Z","shell.execute_reply.started":"2024-05-02T18:05:38.113539Z","shell.execute_reply":"2024-05-02T18:05:38.124742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import train_test_split\n\n# X=df_f.drop('ids',axis=1)\n# y=df_f['ids']\n\n# X_train,X_test,y_train,y_test=train_test_split(X,y, test_size=0.2, random_state=42)\n# # Scale the features\n# scaler = StandardScaler()\n# X_train_scaled = scaler.fit_transform(X_train)\n# X_test_scaled = scaler.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T12:58:53.183880Z","iopub.status.idle":"2024-05-02T12:58:53.184239Z","shell.execute_reply.started":"2024-05-02T12:58:53.184059Z","shell.execute_reply":"2024-05-02T12:58:53.184074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = '/kaggle/input/leash-BELKA/test.csv'\ntest_df = pd.read_csv(test_df)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T18:00:34.133643Z","iopub.execute_input":"2024-05-02T18:00:34.134257Z","iopub.status.idle":"2024-05-02T18:00:37.668091Z","shell.execute_reply.started":"2024-05-02T18:00:34.134226Z","shell.execute_reply":"2024-05-02T18:00:37.667289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"molecule_smiles = []\nfor i in test_df['molecule_smiles']:\n    molecule_smiles.append(i)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T18:00:59.395013Z","iopub.execute_input":"2024-05-02T18:00:59.395354Z","iopub.status.idle":"2024-05-02T18:00:59.834718Z","shell.execute_reply.started":"2024-05-02T18:00:59.395329Z","shell.execute_reply":"2024-05-02T18:00:59.833876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r = []\n\nfor i in tqdm(molecule_smiles[:400000]):\n    result = count_atoms_bonds_charges(i)\n    r.append(result)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T18:07:58.847983Z","iopub.execute_input":"2024-05-02T18:07:58.848349Z","iopub.status.idle":"2024-05-02T18:18:33.965642Z","shell.execute_reply.started":"2024-05-02T18:07:58.848321Z","shell.execute_reply":"2024-05-02T18:18:33.964722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(molecule_smiles[400000:800000]):\n    result = count_atoms_bonds_charges(i)\n    r.append(result)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T18:18:33.967666Z","iopub.execute_input":"2024-05-02T18:18:33.968466Z","iopub.status.idle":"2024-05-02T18:29:09.367254Z","shell.execute_reply.started":"2024-05-02T18:18:33.968430Z","shell.execute_reply":"2024-05-02T18:29:09.366317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(molecule_smiles[800000:1200000]):\n    result = count_atoms_bonds_charges(i)\n    r.append(result)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T18:29:09.368508Z","iopub.execute_input":"2024-05-02T18:29:09.368813Z","iopub.status.idle":"2024-05-02T18:39:36.498000Z","shell.execute_reply.started":"2024-05-02T18:29:09.368788Z","shell.execute_reply":"2024-05-02T18:39:36.497093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tqdm(molecule_smiles[1200000 :]):\n    result = count_atoms_bonds_charges(i)\n    r.append(result)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T18:39:36.499648Z","iopub.execute_input":"2024-05-02T18:39:36.499938Z","iopub.status.idle":"2024-05-02T18:52:02.095431Z","shell.execute_reply.started":"2024-05-02T18:39:36.499914Z","shell.execute_reply":"2024-05-02T18:52:02.094442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r[0]","metadata":{"execution":{"iopub.status.busy":"2024-05-02T18:53:42.923446Z","iopub.execute_input":"2024-05-02T18:53:42.923862Z","iopub.status.idle":"2024-05-02T18:53:42.930116Z","shell.execute_reply.started":"2024-05-02T18:53:42.923835Z","shell.execute_reply":"2024-05-02T18:53:42.929180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = r\nids = test_df['id']\natoms = []\nbonds = []\npositive_charges = []\nnegative_charges = []\nndfids = []\nprotein_names = []\nmolecular_weights = []\nlogPs = []\nhb_acceptors = []\nhb_donors = []\ntpsas = [] \nrotatable_bonds = []\n\n\nfor i in range(0,len(data)):\n    ndfids.append(ids[i])\n    atoms.append(data[i]['atoms'])\n    bonds.append(data[i]['bonds'])\n    positive_charges.append(data[i]['positive_charges'])\n    negative_charges.append(data[i]['negative_charges'])\n    protein_names.append(test_df['protein_name'][i])\n    molecular_weights.append(data[i]['molecular_weight'])\n    logPs.append(data[i]['LogP'])\n    hb_acceptors.append(data[i]['HBA'])\n    hb_donors.append(data[i]['HBD'])\n    tpsas.append(data[i]['TPSA'])\n    rotatable_bonds.append(data[i]['Rotatable Bonds'])\n    \n    \ndf_molecule_smiles = pd.DataFrame()\ndf_molecule_smiles['ids'] = ndfids\ndf_molecule_smiles['protein_names'] = protein_names\ndf_molecule_smiles['Atoms'] = atoms\ndf_molecule_smiles['Bonds'] = bonds \ndf_molecule_smiles['molecular weight'] = molecular_weights\ndf_molecule_smiles['LogP'] = logPs\ndf_molecule_smiles['HBA'] = hb_acceptors\ndf_molecule_smiles['HBD'] = hb_donors\ndf_molecule_smiles['TPSA'] = tpsas\ndf_molecule_smiles['Rotatable Bonds'] = rotatable_bonds\n\n#there was no charge possitive or negative in these 5000 samples in molecule_smiles column\ndf_moleculesmiles = df_molecule_smiles\ndf_molecule_smiles","metadata":{"execution":{"iopub.status.busy":"2024-05-02T18:56:47.140227Z","iopub.execute_input":"2024-05-02T18:56:47.140579Z","iopub.status.idle":"2024-05-02T18:57:29.814753Z","shell.execute_reply.started":"2024-05-02T18:56:47.140552Z","shell.execute_reply":"2024-05-02T18:57:29.813866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn import svm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n\n# Assuming your train data is stored in a DataFrame called 'df_f'\nX_train = df_f.loc[:, \"Bonds\":\"Rotatable Bonds\"]\nY_train = df_f[\"binds\"]  # Assuming the target variable is named 'binds'\n\n# Encode the target variable (if it's not already binary)\nlabel_encoder = LabelEncoder()\nY_train = label_encoder.fit_transform(Y_train)\n\n# Create an SVM classifier with a linear kernel\nclf = svm.SVC(kernel='linear')\n\n# Train the model using the training sets\nclf.fit(X_train, Y_train)\n\n# Assuming your test data is stored in a DataFrame called 'df_molecule_smiles'\nX_test = df_molecule_smiles.loc[:, \"Bonds\":\"Rotatable Bonds\"]\n\n# Predict the response for the test dataset\nY_pred = clf.predict(X_test)\n\n# Create a DataFrame with the test IDs and their corresponding predicted values\nsubmission_df = pd.DataFrame({\"id\": df_molecule_smiles[\"ids\"], \"binds\": Y_pred})\n\n# Save the submission file\nsubmission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T19:00:16.426848Z","iopub.execute_input":"2024-05-02T19:00:16.427526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn import svm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n\n# Assuming your data is stored in a DataFrame called 'df_f'\nX = df_f.loc[:, \"Atoms\":\"Rotatable Bonds\"]\nY = df_f[\"Bonds\"]\n\n# Encode the target variable (if it's not already binary)\nlabel_encoder = LabelEncoder()\nY = label_encoder.fit_transform(Y)\n\n# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, Y, test_size=0.2, random_state=42)\n\n# Create an SVM classifier with a linear kernel\nclf = svm.SVC(kernel='linear')\n\n# Train the model using the training sets\nclf.fit(X_train, y_train)\n\n# Predict the response for the entire test dataset\ny_pred = clf.predict(X_test)\n\n# Create a DataFrame with the test IDs and their corresponding predicted probabilities\nsubmission_df = pd.DataFrame({\"id\": df_f[\"ids\"][0:2000], \"binds\": y_pred})\n\n# Save the submission file\nsubmission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T12:58:53.194697Z","iopub.status.idle":"2024-05-02T12:58:53.195026Z","shell.execute_reply.started":"2024-05-02T12:58:53.194864Z","shell.execute_reply":"2024-05-02T12:58:53.194881Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df_f[\"ids\"][0:2000])\nlen(y_pred)","metadata":{"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.DataFrame({\"id\": df_f[\"ids\"][0:2000], \"binds\": y_pred})\n\n# Save the submission file\nsubmission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T12:58:53.198263Z","iopub.status.idle":"2024-05-02T12:58:53.198560Z","shell.execute_reply.started":"2024-05-02T12:58:53.198412Z","shell.execute_reply":"2024-05-02T12:58:53.198424Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import svm\nfrom tqdm import tqdm  # Import tqdm\n\n# Create an SVM classifier\nclf = svm.SVC()\n\n# Train the classifier on the training data\nwith tqdm(total=len(X_train), desc=\"Training\") as pbar:  # Initialize the progress bar\n    clf.fit(X_train, y_train)\n    pbar.update(1)  # Update the progress bar\n\n# Test the classifier on the testing data\nwith tqdm(total=len(X_test), desc=\"Testing\") as pbar:  # Initialize the progress bar\n    accuracy = clf.score(X_test, y_test)\n    pbar.update(1)  # Update the progress bar\n\nprint(\"Accuracy:\", accuracy)","metadata":{"execution":{"iopub.status.busy":"2024-05-02T12:58:53.200026Z","iopub.status.idle":"2024-05-02T12:58:53.200338Z","shell.execute_reply.started":"2024-05-02T12:58:53.200181Z","shell.execute_reply":"2024-05-02T12:58:53.200194Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]}]}