{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30674,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![](https://www.kaggle.com/competitions/67356/images/header)","metadata":{}},{"cell_type":"markdown","source":"### My notebook is enspired by Andrew D. Blevins and his notebook [Leash Tutorial - ECFPs and Random Forest](https://www.kaggle.com/code/andrewdblevins/leash-tutorial-ecfps-and-random-forest)","metadata":{}},{"cell_type":"markdown","source":"### I will change all his code in my notebook later after I completely figure out with competition","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-04-06T13:16:43.524410Z","iopub.execute_input":"2024-04-06T13:16:43.525516Z","iopub.status.idle":"2024-04-06T13:16:44.356401Z","shell.execute_reply.started":"2024-04-06T13:16:43.525433Z","shell.execute_reply":"2024-04-06T13:16:44.354997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n!pip install rdkit\n!pip install duckdb\nprint('OK')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-04-06T13:16:44.359173Z","iopub.execute_input":"2024-04-06T13:16:44.359938Z","iopub.status.idle":"2024-04-06T13:17:10.310850Z","shell.execute_reply.started":"2024-04-06T13:16:44.359889Z","shell.execute_reply":"2024-04-06T13:17:10.309051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"![](https://www.chitkara.edu.in/blogs/wp-content/uploads/2023/03/Medicinal-Chemistry-vs-Pharmaceutics.jpg)","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm import tqdm\nimport duckdb\nimport optuna\nfrom optuna.samplers import TPESampler\nimport pickle\nfrom rdkit import Chem\nfrom rdkit.Chem import AllChem\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import StackingClassifier\nfrom xgboost import XGBClassifier\nfrom catboost import CatBoostClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.metrics import average_precision_score","metadata":{"execution":{"iopub.status.busy":"2024-04-06T13:17:10.313159Z","iopub.execute_input":"2024-04-06T13:17:10.313587Z","iopub.status.idle":"2024-04-06T13:17:14.049663Z","shell.execute_reply.started":"2024-04-06T13:17:10.313546Z","shell.execute_reply":"2024-04-06T13:17:14.048367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_path = '/kaggle/input/leash-BELKA/train.parquet'\ntest_path = '/kaggle/input/leash-BELKA/test.parquet'\n\ncon = duckdb.connect()\n\ndf = con.query(f\"\"\"(SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0\n                        ORDER BY random()\n                        LIMIT 30000)\n                        UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1\n                        ORDER BY random()\n                        LIMIT 30000)\"\"\").df()\n\ncon.close()","metadata":{"execution":{"iopub.status.busy":"2024-04-06T13:17:14.053229Z","iopub.execute_input":"2024-04-06T13:17:14.054407Z","iopub.status.idle":"2024-04-06T13:18:16.296843Z","shell.execute_reply.started":"2024-04-06T13:17:14.054357Z","shell.execute_reply":"2024-04-06T13:18:16.295739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-06T13:18:16.298220Z","iopub.execute_input":"2024-04-06T13:18:16.299875Z","iopub.status.idle":"2024-04-06T13:18:16.323221Z","shell.execute_reply.started":"2024-04-06T13:18:16.299839Z","shell.execute_reply":"2024-04-06T13:18:16.321857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"![](https://previews.123rf.com/images/olegdudko/olegdudko1712/olegdudko171200585/91056161-chemistry-science-formula-and-tablets-medicine-symbol.jpg)","metadata":{}},{"cell_type":"code","source":"%%time\n# Convert SMILES to RDKit molecules\ndf['molecule'] = df['molecule_smiles'].apply(Chem.MolFromSmiles)\n\n# Generate ECFPs\ndef generate_ecfp(molecule, radius=2, bits=1024):\n    if molecule is None:\n        return None\n    return list(AllChem.GetMorganFingerprintAsBitVect(molecule, radius, nBits=bits))\n\ndf['ecfp'] = df['molecule'].apply(generate_ecfp)","metadata":{"execution":{"iopub.status.busy":"2024-04-06T13:18:16.324736Z","iopub.execute_input":"2024-04-06T13:18:16.325075Z","iopub.status.idle":"2024-04-06T13:19:50.575815Z","shell.execute_reply.started":"2024-04-06T13:18:16.325048Z","shell.execute_reply":"2024-04-06T13:19:50.574747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# One-hot encode the protein_name\nonehot_encoder = OneHotEncoder(sparse_output=False)\nprotein_onehot = onehot_encoder.fit_transform(df['protein_name'].values.reshape(-1, 1))\n\n# Combine ECFPs and one-hot encoded protein_name\nX = [ecfp + protein for ecfp, protein in zip(df['ecfp'].tolist(), protein_onehot.tolist())]\ny = df['binds'].tolist()\n\n# Split the data into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=27)\n\n# Create and train the random forest model\nrf_model = CatBoostClassifier(iterations=500, random_state=27)\nrf_model.fit(X_train, y_train)\n\n# Make predictions on the test set\ny_pred_proba = rf_model.predict_proba(X_test)[:, 1]  # Probability of the positive class\n\n# Calculate the mean average precision\nmap_score = average_precision_score(y_test, y_pred_proba)\nprint(f\"Mean Average Precision (mAP): {map_score:.8f}\")","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-04-06T13:19:50.577005Z","iopub.execute_input":"2024-04-06T13:19:50.577306Z","iopub.status.idle":"2024-04-06T13:20:44.687126Z","shell.execute_reply.started":"2024-04-06T13:19:50.577282Z","shell.execute_reply":"2024-04-06T13:20:44.685976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"![](https://ars.els-cdn.com/content/image/1-s2.0-S2001037020304967-gr3.jpg)","metadata":{}},{"cell_type":"code","source":"%%time\n# Process the test.parquet file chunk by chunk\ntest_file = '/kaggle/input/leash-BELKA/test.csv'\noutput_file = 'submission.csv'  # Specify the path and filename for the output file\n\n# Read the test.parquet file into a pandas DataFrame\n# Wrap the loop with tqdm to display progress\nfor df_test in tqdm(pd.read_csv(test_file, chunksize=10_000)):\n    \n    # Generate ECFPs for the molecule_smiles\n    df_test['molecule'] = df_test['molecule_smiles'].apply(Chem.MolFromSmiles)\n    df_test['ecfp'] = df_test['molecule'].apply(generate_ecfp)\n\n    # One-hot encode the protein_name\n    protein_onehot = onehot_encoder.transform(df_test['protein_name'].values.reshape(-1, 1))\n\n    # Combine ECFPs and one-hot encoded protein_name\n    X_test = [ecfp + protein for ecfp, protein in zip(df_test['ecfp'].tolist(), protein_onehot.tolist())]\n\n    # Predict the probabilities\n    probabilities = rf_model.predict_proba(X_test)[:, 1]\n\n    # Create a DataFrame with 'id' and 'probability' columns\n    output_df = pd.DataFrame({'id': df_test['id'], 'binds': probabilities})\n\n    # Save the output DataFrame to a CSV file\n    output_df.to_csv(output_file, index=False, mode='a', header=not os.path.exists(output_file))","metadata":{"execution":{"iopub.status.busy":"2024-04-06T13:20:44.688707Z","iopub.execute_input":"2024-04-06T13:20:44.689458Z","iopub.status.idle":"2024-04-06T14:20:21.393859Z","shell.execute_reply.started":"2024-04-06T13:20:44.689405Z","shell.execute_reply":"2024-04-06T14:20:21.392978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"![](https://m.media-amazon.com/images/I/41Xf5xwc25L._SR600%2C315_PIWhiteStrip%2CBottomLeft%2C0%2C35_SCLZZZZZZZ_FMpng_BG255%2C255%2C255.jpg)","metadata":{}}]}