{"metadata":{"colab":{"gpuType":"T4","provenance":[]},"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"},{"sourceId":8625343,"sourceType":"datasetVersion","datasetId":5163809}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# @title Packages\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import classification_report, accuracy_score\n\nimport math\nimport numpy as np","metadata":{"id":"BgG9wtad7L03","execution":{"iopub.status.busy":"2024-06-06T17:58:21.011782Z","iopub.execute_input":"2024-06-06T17:58:21.012185Z","iopub.status.idle":"2024-06-06T17:58:23.860369Z","shell.execute_reply.started":"2024-06-06T17:58:21.01215Z","shell.execute_reply":"2024-06-06T17:58:23.859151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# @title Load Dataset\nprocessed_data = pd.read_csv(\"/kaggle/input/balanced-belka/processed_data.csv\")\nprocessed_data.head()","metadata":{"id":"zlRzGrtE7O0L","outputId":"56491d5e-03d1-49dd-daf6-4adc2c2c9b8c","execution":{"iopub.status.busy":"2024-06-06T17:59:40.938241Z","iopub.execute_input":"2024-06-06T17:59:40.939484Z","iopub.status.idle":"2024-06-06T17:59:47.60557Z","shell.execute_reply.started":"2024-06-06T17:59:40.939444Z","shell.execute_reply":"2024-06-06T17:59:47.604281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = [\n    \"MolWt\",\n    \"MolLogP\",\n    \"NumHDonors\",\n    \"NumHAcceptors\",\n    \"TPSA\",\n    \"NumRotatableBonds\",\n]","metadata":{"id":"TNR7i4oXOMyZ","execution":{"iopub.status.busy":"2024-06-06T17:59:57.22113Z","iopub.execute_input":"2024-06-06T17:59:57.221968Z","iopub.status.idle":"2024-06-06T17:59:57.226877Z","shell.execute_reply.started":"2024-06-06T17:59:57.221929Z","shell.execute_reply":"2024-06-06T17:59:57.225572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# @title Prepare the feature matrix X and target vector y\nX = processed_data[columns]\ny = processed_data[\"binds\"]","metadata":{"id":"X8aangjvARyg","execution":{"iopub.status.busy":"2024-06-06T18:02:15.851329Z","iopub.execute_input":"2024-06-06T18:02:15.851858Z","iopub.status.idle":"2024-06-06T18:02:15.877147Z","shell.execute_reply.started":"2024-06-06T18:02:15.851805Z","shell.execute_reply":"2024-06-06T18:02:15.875851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# @title Standardize the features\nfrom sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)","metadata":{"id":"S1Q9UxpzXtWY","execution":{"iopub.status.busy":"2024-06-06T18:02:48.130969Z","iopub.execute_input":"2024-06-06T18:02:48.132002Z","iopub.status.idle":"2024-06-06T18:02:48.293726Z","shell.execute_reply.started":"2024-06-06T18:02:48.131952Z","shell.execute_reply":"2024-06-06T18:02:48.291995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the data\nfrom sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.2, random_state=10\n)","metadata":{"id":"CD1hqLLuAU4P","execution":{"iopub.status.busy":"2024-06-06T18:02:53.882761Z","iopub.execute_input":"2024-06-06T18:02:53.883202Z","iopub.status.idle":"2024-06-06T18:02:54.034866Z","shell.execute_reply.started":"2024-06-06T18:02:53.883168Z","shell.execute_reply":"2024-06-06T18:02:54.033619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\n    f\"The number of un-binding molecules in the training set is {(y_train == 0).sum()}\\n\\n\"\n)\nprint(\n    f\"The number of binding molecules in the training set is {(y_train == 1).sum()}\\n\\n\"\n)\n\nprint(\n    f\"The number of un-binding molecules in the testing set is {(y_test == 0).sum()}\\n\\n\"\n)\nprint(f\"The number of binding molecules in the testing set is {(y_test == 1).sum()}\")","metadata":{"id":"vv2-HmVdCHuj","outputId":"0d1c90db-2b9d-47ab-f5d0-bca180030bef","execution":{"iopub.status.busy":"2024-06-06T18:02:55.890047Z","iopub.execute_input":"2024-06-06T18:02:55.893044Z","iopub.status.idle":"2024-06-06T18:02:55.903907Z","shell.execute_reply.started":"2024-06-06T18:02:55.893002Z","shell.execute_reply":"2024-06-06T18:02:55.902774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nclf = RandomForestClassifier(n_estimators=100, random_state=10)\nclf.fit(X_train, y_train)","metadata":{"id":"E6pGtvVaBRk9","outputId":"a6c5cd2d-860f-4682-981e-a99dbf806f97","execution":{"iopub.status.busy":"2024-06-06T18:02:58.118756Z","iopub.execute_input":"2024-06-06T18:02:58.11917Z","iopub.status.idle":"2024-06-06T18:06:35.137867Z","shell.execute_reply.started":"2024-06-06T18:02:58.119137Z","shell.execute_reply":"2024-06-06T18:06:35.136682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import average_precision_score\n\n# Make predictions on the test set\ny_pred_proba = clf.predict_proba(X_test)[:, 1]  # Probability of the positive class\n\n# Calculate the mean average precision\nmap_score = average_precision_score(y_test, y_pred_proba)\nprint(f\"Mean Average Precision (mAP): {map_score:.2f}\")","metadata":{"execution":{"iopub.status.busy":"2024-06-06T18:06:58.394632Z","iopub.execute_input":"2024-06-06T18:06:58.395136Z","iopub.status.idle":"2024-06-06T18:07:06.368701Z","shell.execute_reply.started":"2024-06-06T18:06:58.39509Z","shell.execute_reply":"2024-06-06T18:07:06.367519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report, accuracy_score\n\ny_pred = clf.predict(X_test)\n\nprint(\"Accuracy:\", accuracy_score(y_test, y_pred))\nprint(\"Classfication Report:\\n\", classification_report(y_test, y_pred))","metadata":{"id":"ByNRT9uVBiRA","outputId":"2430847d-f574-421d-8f6c-385bf4613054","execution":{"iopub.status.busy":"2024-06-06T18:07:08.513451Z","iopub.execute_input":"2024-06-06T18:07:08.513832Z","iopub.status.idle":"2024-06-06T18:07:16.826896Z","shell.execute_reply.started":"2024-06-06T18:07:08.5138Z","shell.execute_reply":"2024-06-06T18:07:16.825593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install rdkit","metadata":{"execution":{"iopub.status.busy":"2024-06-06T18:07:20.094753Z","iopub.execute_input":"2024-06-06T18:07:20.095298Z","iopub.status.idle":"2024-06-06T18:07:36.882943Z","shell.execute_reply.started":"2024-06-06T18:07:20.095256Z","shell.execute_reply":"2024-06-06T18:07:36.881628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import Descriptors, AllChem","metadata":{"execution":{"iopub.status.busy":"2024-06-06T18:07:36.885443Z","iopub.execute_input":"2024-06-06T18:07:36.885873Z","iopub.status.idle":"2024-06-06T18:07:37.185428Z","shell.execute_reply.started":"2024-06-06T18:07:36.885837Z","shell.execute_reply":"2024-06-06T18:07:37.18427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# @title Function to compute molecular descriptors\n\ndef compute_descriptors(smiles):\n    try:\n        mol = Chem.MolFromSmiles(smiles)\n        if mol is not None:\n            descriptors = [\n                Descriptors.MolWt(mol),\n                Descriptors.MolLogP(mol),\n                Descriptors.NumHDonors(mol),\n                Descriptors.NumHAcceptors(mol),\n                Descriptors.TPSA(mol),\n                Descriptors.NumRotatableBonds(mol),\n                # Descriptors.FpDensityMorgan1(mol),\n                # Descriptors.FpDensityMorgan2(mol),\n                # Descriptors.FpDensityMorgan3(mol),\n                # Descriptors.BalabanJ(mol),\n                # Descriptors.BertzCT(mol),\n                # Descriptors.HallKierAlpha(mol),\n                # Descriptors.Ipc(mol),\n                # Descriptors.Kappa1(mol),\n                # Descriptors.Kappa2(mol),\n                # Descriptors.Kappa3(mol),\n                # # Descriptors.MaxPartialCharge(mol),\n                # # Descriptors.MinPartialCharge(mol),\n                # # Descriptors.MaxAbsPartialCharge(mol),\n                # # Descriptors.MinAbsPartialCharge(mol),\n                # Descriptors.MolMR(mol),\n                # Descriptors.LabuteASA(mol),\n                # Descriptors.EState_VSA1(mol),\n                # Descriptors.EState_VSA2(mol),\n                # Descriptors.SMR_VSA1(mol),\n                # Descriptors.SMR_VSA2(mol),\n            ]\n        else:\n            descriptors = [None] * 6\n            # 22\n    except:\n        descriptors = [None] * 6\n        # 22\n    return descriptors","metadata":{"execution":{"iopub.status.busy":"2024-06-06T18:07:39.538397Z","iopub.execute_input":"2024-06-06T18:07:39.538843Z","iopub.status.idle":"2024-06-06T18:07:39.547364Z","shell.execute_reply.started":"2024-06-06T18:07:39.538811Z","shell.execute_reply":"2024-06-06T18:07:39.545804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# Process the test.parquet file chunk by chunk\ntest_file = '/kaggle/input/leash-BELKA/test.csv'\noutput_file = '/kaggle/input/balanced-belka/processed_data.csv'  # Specify the path and filename for the output file","metadata":{"execution":{"iopub.status.busy":"2024-06-06T18:07:44.860074Z","iopub.execute_input":"2024-06-06T18:07:44.861059Z","iopub.status.idle":"2024-06-06T18:07:44.866044Z","shell.execute_reply.started":"2024-06-06T18:07:44.86102Z","shell.execute_reply":"2024-06-06T18:07:44.864647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(test_file)","metadata":{"execution":{"iopub.status.busy":"2024-06-06T18:07:51.261757Z","iopub.execute_input":"2024-06-06T18:07:51.262155Z","iopub.status.idle":"2024-06-06T18:07:57.866064Z","shell.execute_reply.started":"2024-06-06T18:07:51.26212Z","shell.execute_reply":"2024-06-06T18:07:57.864975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T18:07:59.214023Z","iopub.execute_input":"2024-06-06T18:07:59.21452Z","iopub.status.idle":"2024-06-06T18:07:59.228649Z","shell.execute_reply.started":"2024-06-06T18:07:59.214483Z","shell.execute_reply":"2024-06-06T18:07:59.227565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sampled_data = test_data.sample(n=1000, random_state=1)  \n","metadata":{"execution":{"iopub.status.busy":"2024-06-06T18:09:40.262478Z","iopub.execute_input":"2024-06-06T18:09:40.262863Z","iopub.status.idle":"2024-06-06T18:09:40.347767Z","shell.execute_reply.started":"2024-06-06T18:09:40.26283Z","shell.execute_reply":"2024-06-06T18:09:40.346562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sampled_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T18:16:08.394304Z","iopub.execute_input":"2024-06-06T18:16:08.394821Z","iopub.status.idle":"2024-06-06T18:16:08.41713Z","shell.execute_reply.started":"2024-06-06T18:16:08.394777Z","shell.execute_reply":"2024-06-06T18:16:08.41589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = sampled_data['protein_name'].apply(lambda x: 1 if x == 'HSA' else 0)\ny_test.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T18:16:21.66056Z","iopub.execute_input":"2024-06-06T18:16:21.6609Z","iopub.status.idle":"2024-06-06T18:16:21.670505Z","shell.execute_reply.started":"2024-06-06T18:16:21.660872Z","shell.execute_reply":"2024-06-06T18:16:21.669191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sampled_data['molecule_descriptors'] = sampled_data['molecule_smiles'].apply(compute_descriptors)\n\nmolecule_descriptors_df = pd.DataFrame(sampled_data['molecule_descriptors'].tolist(), columns=columns)\n\n\n\n\n# @title Log and drop rows with invalid SMILES\ninvalid_smiles = molecule_descriptors_df[molecule_descriptors_df['MolWt'].isna()]\nif not invalid_smiles.empty:\n    print(\"Invalid SMILES strings:\")\n    print(invalid_smiles)\n    \n# @title Check if there is any infinite values\nhas_inf = np.isinf(molecule_descriptors_df.values).any()\nprint(\"Contains inf:\", has_inf)\n\nX_test = molecule_descriptors_df.dropna(subset=columns)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-06T18:23:49.299319Z","iopub.execute_input":"2024-06-06T18:23:49.299774Z","iopub.status.idle":"2024-06-06T18:23:50.737684Z","shell.execute_reply.started":"2024-06-06T18:23:49.299741Z","shell.execute_reply":"2024-06-06T18:23:50.735854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-06T18:23:51.001617Z","iopub.execute_input":"2024-06-06T18:23:51.001985Z","iopub.status.idle":"2024-06-06T18:23:51.015Z","shell.execute_reply.started":"2024-06-06T18:23:51.001957Z","shell.execute_reply":"2024-06-06T18:23:51.013878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions on the test set\ny_pred_proba = clf.predict_proba(X_test)[:, 1]  # Probability of the positive class\n\n# Calculate the mean average precision\nmap_score = average_precision_score(y_test, y_pred_proba)\nprint(f\"Mean Average Precision (mAP): {map_score:.2f}\")\n\n\n# # Predict the probabilities\n# probabilities = clf.predict_proba(X_test)[:, 1]\n\n# # Create a DataFrame with 'id' and 'probability' columns\n# output_df = pd.DataFrame({'id': df_test['id'], 'binds': probabilities})\n\n# # Save the output DataFrame to a CSV file\n# output_df.to_csv(output_file, index=False, mode='a', header=not os.path.exists(output_file))","metadata":{"execution":{"iopub.status.busy":"2024-06-06T18:25:32.162383Z","iopub.execute_input":"2024-06-06T18:25:32.162899Z","iopub.status.idle":"2024-06-06T18:25:32.279431Z","shell.execute_reply.started":"2024-06-06T18:25:32.16286Z","shell.execute_reply":"2024-06-06T18:25:32.277773Z"},"trusted":true},"execution_count":null,"outputs":[]}]}