{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install rdkit\n!pip install duckdb","metadata":{"_uuid":"7cbe54e2-9655-443e-bb2f-2d0132033012","_cell_guid":"24b7ae79-0947-45a6-ae57-beb1ddef2ac3","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-06T16:26:16.636297Z","iopub.execute_input":"2024-04-06T16:26:16.637403Z","iopub.status.idle":"2024-04-06T16:26:52.601311Z","shell.execute_reply.started":"2024-04-06T16:26:16.637333Z","shell.execute_reply":"2024-04-06T16:26:52.599596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import duckdb\nimport pandas as pd\n\ntrain_path = '/kaggle/input/leash-predict-chemical-bindings/train.parquet'\ntest_path = '/kaggle/input/leash-predict-chemical-bindings/test.parquet'\n\ncon = duckdb.connect()\n\ndf = con.query(f\"\"\"(SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0\n                        ORDER BY random()\n                        LIMIT 30000)\n                        UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1\n                        ORDER BY random()\n                        LIMIT 30000)\"\"\").df()\n\ncon.close()","metadata":{"_uuid":"afcb1325-2e05-4c62-974c-c0d49f12760f","_cell_guid":"3561cf9e-3957-4370-9658-8818058b8345","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-06T16:26:52.605130Z","iopub.execute_input":"2024-04-06T16:26:52.605776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"_uuid":"9458b0cc-e19a-479a-af2f-6cc82bf248b6","_cell_guid":"71cd2efa-bb95-46a2-93a9-4dc12e3263f0","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Preprocessing\n\nLets grab the smiles for the fully assembled molecule `molecule_smiles` and generate ecfps for it. We could choose different radiuses or bits, but 2 and 1024 is pretty standard.","metadata":{"_uuid":"815cd2d3-317b-4962-8dc2-379d35ff2605","_cell_guid":"7af88eb4-517f-4fe1-94c6-1ec67f737565","trusted":true}},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import AllChem\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.preprocessing import OneHotEncoder\n\n# Convert SMILES to RDKit molecules\ndf['molecule'] = df['molecule_smiles'].apply(Chem.MolFromSmiles)\n\n# Generate ECFPs\ndef generate_ecfp(molecule, radius=2, bits=2048):\n    if molecule is None:\n        return None\n    return list(AllChem.GetMorganFingerprintAsBitVect(molecule, radius, nBits=bits))\n\ndf['ecfp'] = df['molecule'].apply(generate_ecfp)","metadata":{"_uuid":"167d1164-f066-44e2-8a2e-6e1a07f8f0d7","_cell_guid":"9b72a690-6f02-4055-ad71-9071b2c577a3","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-04-04T22:55:20.047210Z","iopub.execute_input":"2024-04-04T22:55:20.047614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train Model","metadata":{"_uuid":"7ddf0349-0f6f-4671-9ad6-fe63c4c1bcab","_cell_guid":"ad95e54b-7964-4bb1-8acd-5a4eb0d5d8c5","trusted":true}},{"cell_type":"code","source":"from xgboost import XGBClassifier\n\n# One-hot encode the protein_name\nonehot_encoder = OneHotEncoder(sparse_output=False)\nprotein_onehot = onehot_encoder.fit_transform(df['protein_name'].values.reshape(-1, 1))\n\n# Combine ECFPs and one-hot encoded protein_name\nX = [ecfp + protein for ecfp, protein in zip(df['ecfp'].tolist(), protein_onehot.tolist())]\ny = df['binds'].tolist()\n\n# Split the data into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Create and train the random forest model\nxgb_model = XGBClassifier(use_label_encoder=False, n_estimators=100, random_state=42)\nxgb_model.fit(X_train, y_train)\n\n# Make predictions on the test set\ny_pred_proba = xgb_model.predict_proba(X_test)[:, 1]  # Probability of the positive class\n\n# Calculate the mean average precision\nmap_score = average_precision_score(y_test, y_pred_proba)\nprint(f\"Mean Average Precision (mAP): {map_score:.2f}\")","metadata":{"_uuid":"085d9c3b-b8cc-4f89-b6db-8b14d0625cbf","_cell_guid":"e80deeaf-ef13-4f59-bd66-100a014782b1","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Look at that Average Precision score. We did amazing! \n\nActually no, we just overfit. This is likely recurring theme for this competition. It is easy to predict molecules that come from the same corner of chemical space, but generalizing to new molecules is extremely difficult.","metadata":{"_uuid":"952c2936-8867-4c5a-8539-3e18928a4631","_cell_guid":"268ae9d9-530d-4c81-8011-0fd21735457c","trusted":true}},{"cell_type":"markdown","source":"## Test Prediction\n\n The trained Random Forest model is then used to predict the binding probabilities. These predictions are saved to a CSV file, which serves as the submission file for the Kaggle competition.","metadata":{"_uuid":"005a0b72-efc6-4b93-860c-412dccd4e6e2","_cell_guid":"b069a642-573f-4b7f-8ebd-61fccfce852d","trusted":true}},{"cell_type":"code","source":"import os\n\n# Process the test.parquet file chunk by chunk\ntest_file = '/kaggle/input/leash-predict-chemical-bindings/test.csv'\noutput_file = 'submission.csv'  # Specify the path and filename for the output file\n\n# Read the test.parquet file into a pandas DataFrame\nfor df_test in pd.read_csv(test_file, chunksize=100000):\n\n    # Generate ECFPs for the molecule_smiles\n    df_test['molecule'] = df_test['molecule_smiles'].apply(Chem.MolFromSmiles)\n    df_test['ecfp'] = df_test['molecule'].apply(generate_ecfp)\n\n    # One-hot encode the protein_name\n    protein_onehot = onehot_encoder.transform(df_test['protein_name'].values.reshape(-1, 1))\n\n    # Combine ECFPs and one-hot encoded protein_name\n    X_test = [ecfp + protein for ecfp, protein in zip(df_test['ecfp'].tolist(), protein_onehot.tolist())]\n\n    # Predict the probabilities\n    probabilities = xgb_model.predict_proba(X_test)[:, 1]\n\n    # Create a DataFrame with 'id' and 'probability' columns\n    output_df = pd.DataFrame({'id': df_test['id'], 'binds': probabilities})\n\n    # Save the output DataFrame to a CSV file\n    output_df.to_csv(output_file, index=False, mode='a', header=not os.path.exists(output_file))","metadata":{"_uuid":"0af1ed6a-3608-444d-bf34-f238e297723d","_cell_guid":"00d315ef-b073-4c59-849e-217923503356","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]}]}