{"metadata":{"colab":{"machine_shape":"hm","provenance":[]},"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"widgets":{"application/vnd.jupyter.widget-state+json":{"1354e02a2203421b8690d495e3c6c8c4":{"model_module":"@jupyter-widgets/controls","model_name":"FloatProgressModel","model_module_version":"1.5.0","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"","description":"","description_tooltip":null,"layout":"IPY_MODEL_bf596737fb08446d9c8d06a29726b1a0","max":100,"min":0,"orientation":"horizontal","style":"IPY_MODEL_55ab1620acff437db2b6a98445f6ce70","value":50}},"bf596737fb08446d9c8d06a29726b1a0":{"model_module":"@jupyter-widgets/base","model_name":"LayoutModel","model_module_version":"1.2.0","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"auto"}},"55ab1620acff437db2b6a98445f6ce70":{"model_module":"@jupyter-widgets/controls","model_name":"ProgressStyleModel","model_module_version":"1.5.0","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":"black","description_width":""}}}},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Description\n\nThe following notebook contains all the necessary steps to:\n1) load the data\n\n2) convert the smiles to ECFP. For this part we us the **scikit-fingerprints** library instead of **rdkit**.\n\n3) Fine tune a XGBoost model for each protein\n\n4) Train the final model with a dataset of size 2M rows.\n\nHere I limited th model to 2M due to memoryy constrains. I assume that an increased size in the dataset would throw an even better score.","metadata":{}},{"cell_type":"code","source":"!pip install -qq duckdb scikit-fingerprints scikit-optimize imbalanced-learn lightgbm","metadata":{"id":"23cd70b7","executionInfo":{"status":"ok","timestamp":1718234059450,"user_tz":420,"elapsed":12051,"user":{"displayName":"Pablo Di Giusto","userId":"10282521815242520690"}},"outputId":"f07c8a00-235a-49cb-9a07-437753d4d47e"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check if GPU is enabled\nimport tensorflow as tf\nprint(\"Num GPUs Available: \", len(tf.config.experimental.list_physical_devices('GPU')))\n# Check GPU type\n!nvidia-smi","metadata":{"executionInfo":{"elapsed":4364,"status":"ok","timestamp":1718234063810,"user":{"displayName":"Pablo Di Giusto","userId":"10282521815242520690"},"user_tz":420},"id":"8s6xbWggNenE","outputId":"3166dd6b-e6ce-41f9-c506-5f1729dcfc39"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nimport time\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport multiprocessing as mp\nfrom joblib import Parallel, delayed\nimport gc\nimport duckdb\nimport numpy as np\nimport pandas as pd\nimport zipfile\nfrom scipy.sparse import csr_matrix, hstack\n\n# Obtain Fingerprints from SMILES\nfrom skfp.fingerprints import AtomPairFingerprint, ECFPFingerprint, MACCSFingerprint\n\n# Importing ML libraries\nfrom sklearn.model_selection import train_test_split, KFold\nfrom skopt import BayesSearchCV\nfrom skopt.space import Real, Integer, Categorical\nfrom sklearn.model_selection import ParameterGrid, GridSearchCV, RandomizedSearchCV, cross_val_score, StratifiedKFold\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.metrics import roc_curve, auc, confusion_matrix, accuracy_score, classification_report, average_precision_score, make_scorer, roc_auc_score\n\n# Data Sampling\nfrom imblearn.over_sampling import SMOTE\nfrom imblearn.under_sampling import RandomUnderSampler\nfrom imblearn.pipeline import Pipeline\n\n# LightGBM\nimport xgboost as xgb","metadata":{"id":"32627701","executionInfo":{"status":"ok","timestamp":1718218369057,"user_tz":420,"elapsed":553,"user":{"displayName":"Pablo Di Giusto","userId":"10282521815242520690"}}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from google.colab import files\nfiles.upload()  # Upload kaggle.json\n\n# Set up Kaggle API credentials\n!mkdir -p ~/.kaggle\n!cp kaggle.json ~/.kaggle/\n!chmod 600 ~/.kaggle/kaggle.json\n\n# Download the dataset\n!kaggle competitions download -c leash-BELKA\n\n# Extract specific files\nwith zipfile.ZipFile('leash-BELKA.zip', 'r') as zip_ref:\n    zip_ref.extract('test.parquet', path='/content/')\n    zip_ref.extract('train.parquet', path='/content/')\n","metadata":{"executionInfo":{"elapsed":288711,"status":"ok","timestamp":1718216890798,"user":{"displayName":"Pablo Di Giusto","userId":"10282521815242520690"},"user_tz":420},"id":"442861a7","outputId":"8003a9c0-89b2-4fbc-8924-9be94152726c"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fp_ap_transformer = AtomPairFingerprint(fp_size=2048, n_jobs=-1)\nfp_ecfp_transformer = ECFPFingerprint(fp_size=2048, radius=3, n_jobs=-1)\nfp_maccs_transformer = MACCSFingerprint(n_jobs=-1)\n\nELEMENTS_PER_WORKER = 100_000\n\ndef calculate_fp(x, fp_name: str):\n    if fp_name == 'ecfp':\n        fp_transformer = fp_ecfp_transformer\n    elif fp_name == 'ap':\n        fp_transformer = fp_ap_transformer\n    elif fp_name == 'maccs':\n        fp_transformer = fp_maccs_transformer\n    else:\n        raise Exception('Wrong fingerprint name!')\n\n    middle_parts = []\n    k_splits = x.shape[0] // ELEMENTS_PER_WORKER\n\n    for i in tqdm(range(k_splits)):\n        middle_parts.append(fp_transformer.transform(x[i * ELEMENTS_PER_WORKER: (i + 1) * ELEMENTS_PER_WORKER]))\n\n    if x.shape[0] % ELEMENTS_PER_WORKER > 0:\n        middle_parts.append(fp_transformer.transform(x[k_splits * ELEMENTS_PER_WORKER:]))\n\n    return np.concatenate(middle_parts)\n\n\ndef load_data(data_size, protein_name, train_path):\n    \"\"\"\n    Load data for a specific protein and data size from a parquet file.\n\n    Parameters:\n    data_size (int): The size of the data to load.\n    protein_name (str): The name of the protein to filter by.\n    train_path (str): The path to the parquet file.\n\n    Returns:\n    pd.DataFrame: The filtered dataframe.\n    \"\"\"\n    con = duckdb.connect()\n\n    query = f\"\"\"\n        (SELECT molecule_smiles, binds\n        FROM parquet_scan('{train_path}')\n        WHERE protein_name = '{protein_name}' AND binds = 0\n        ORDER BY random()\n        LIMIT {data_size})\n        UNION ALL\n        (SELECT molecule_smiles, binds\n        FROM parquet_scan('{train_path}')\n        WHERE protein_name = '{protein_name}' AND binds = 1\n        ORDER BY random()\n        LIMIT {data_size})\n    \"\"\"\n\n    df_train = con.query(query).df()\n    con.close()\n\n    return df_train","metadata":{"id":"89bb856a","executionInfo":{"status":"ok","timestamp":1718216890798,"user_tz":420,"elapsed":6,"user":{"displayName":"Pablo Di Giusto","userId":"10282521815242520690"}}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"proteins = ['BRD4', 'HSA', 'sEH']\nDATA_SIZE = 1_000_000\nDATA_SIZE_REDUCED = 500_000\ntrain_path = '/content/train.parquet'\nfp_name = 'ecfp'","metadata":{"id":"vq3VYn9D4DnC","executionInfo":{"status":"ok","timestamp":1718216890798,"user_tz":420,"elapsed":5,"user":{"displayName":"Pablo Di Giusto","userId":"10282521815242520690"}}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Verify GPU support\nprint(f'XGBoost built with GPU support: {xgb.__version__}')","metadata":{"executionInfo":{"elapsed":4,"status":"ok","timestamp":1718216890798,"user":{"displayName":"Pablo Di Giusto","userId":"10282521815242520690"},"user_tz":420},"id":"41b7sLavP3Xf","outputId":"01748544-e60e-4bc7-e6b5-b903121cf99e"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Hyperparametere fine tunning\n\n\n# Dictionary to store the best hyperparameters for each protein\nbest_params_dict = {}\n\nfor protein in proteins:\n    print('------------------------------------------------------------------------------------')\n    print(f'Hyperparameter fine tunning for: {protein} with a data size of {DATA_SIZE_REDUCED}')\n    print('------------------------------------------------------------------------------------')\n    # Load a reduced dataset\n    df_train_reduced = load_data(DATA_SIZE_REDUCED, protein, train_path)\n\n    # Calculate fingerprints for the reduced dataset\n    X_train_reduced = calculate_fp(df_train_reduced['molecule_smiles'], fp_name=fp_name)\n    y_train_reduced = df_train_reduced['binds']\n\n    # Define the hyperparameter grid\n    param_dist = {\n        'max_depth': Integer(3, 15),\n        'learning_rate': Real(0.01, 0.2, prior='log-uniform'),\n        'subsample': Real(0.6, 1.0),\n        'colsample_bytree': Real(0.6, 1.0),\n        'gamma': Real(1e-9, 1.0, prior='log-uniform'),\n        'reg_alpha': Real(1e-9, 1.0, prior='log-uniform'),\n        'reg_lambda': Real(1e-9, 1.0, prior='log-uniform'),\n        'n_estimators': Integer(100, 500)\n    }\n\n    # Define the model\n    xgb_estimator = xgb.XGBClassifier(tree_method='hist', device='cuda', objective='binary:logistic', eval_metric='auc')\n\n    # Define the scoring function\n    scorer = make_scorer(roc_auc_score, greater_is_better=True, needs_proba=True)\n\n    # Initialize BayesSearchCV\n    bayes_search = BayesSearchCV(estimator=xgb_estimator, search_spaces=param_dist,\n                                 scoring=scorer, cv=StratifiedKFold(n_splits=3), verbose=1, n_jobs=-1, n_iter=100)\n\n    # Perform randomized search\n    bayes_search.fit(X_train_reduced, y_train_reduced)\n\n    # Print the best parameters and the best score\n    print(f\"Best parameters found: {bayes_search.best_params_}\")\n    print(f\"Best AUC score: {bayes_search.best_score_}\")\n\n    # Store the best parameters in the dictionary\n    best_params_dict[protein] = bayes_search.best_params_\n    print(f'Best parameters for {protein} saved in best_params_dict.')","metadata":{"id":"e6tfctUF2VmL"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n\n# Then, train the best models with full datasets and predict\nsubmission_list = []\n\nfor protein in proteins:\n\n    print('------------------------------------------')\n    print(f'Training and predicting for: {protein}')\n    print('------------------------------------------')\n\n    # Load data\n    df_train = load_data(DATA_SIZE, protein, train_path)\n\n    # Calculate fingerprints\n    print('Preprocessing training data')\n    X_train = calculate_fp(df_train['molecule_smiles'], fp_name=fp_name)\n    y_train = df_train['binds']\n\n    # Split the data\n    X_train_split, X_val_split, y_train_split, y_val_split = train_test_split(X_train, y_train, test_size=0.2, random_state=42)\n\n    # Define DMatrix for training and validation\n    dtrain = xgb.DMatrix(X_train_split, label=y_train_split)\n    dval = xgb.DMatrix(X_val_split, label=y_val_split)\n\n    # Free up memory\n    del df_train\n    del X_train\n    gc.collect()\n\n    # Get the best hyperparameters for the current protein\n    best_params = best_params_dict[protein]\n    best_params['tree_method'] = 'hist'\n    best_params['device'] = 'cuda'\n    best_params['objective'] = 'binary:logistic'\n    best_params['eval_metric'] = 'auc'\n    best_params['predictor'] = 'gpu_predictor'\n    best_params['early_stopping_rounds'] = 10\n\n    # Train the model\n    print('Training the model')\n    evals = [(dtrain, 'train'), (dval, 'valid')]\n    bst = xgb.train(best_params, dtrain, num_boost_round=1000, evals=evals)\n\n    # Predict on validation set\n    y_val_pred_continuous = xgb_model.predict_proba(X_val_split)[:, 1]\n    y_val_pred_binary = [1 if pred > 0.5 else 0 for pred in y_val_pred_continuous]\n\n    # Predict on validation set\n    y_val_pred = bst.predict(dval)\n    y_val_pred_binary = [1 if pred > 0.5 else 0 for pred in y_val_pred]\n\n    # Free up memory\n    del X_train_split\n    del X_val_split\n    del y_train_split\n    del y_val_split\n    del dtrain\n    del dval\n    gc.collect()\n\n    print('Processing test data')\n    con = duckdb.connect()\n    test_path = '/content/test.parquet'\n    df_test = con.query(f\"\"\"SELECT id, molecule_smiles\n                            FROM parquet_scan('{test_path}')\n                            WHERE protein_name = '{protein}'\"\"\").df()\n    con.close()\n\n    X_test_transformed = calculate_fp(df_test['molecule_smiles'], fp_name=fp_name)\n    dtest = xgb.DMatrix(X_test_transformed)\n\n    # Predict on test set\n    test_preds = bst.predict(dtest)\n\n    df_test['binds'] = test_preds\n    df_test.drop(['molecule_smiles'], axis=1, inplace=True)\n\n    submission_list.append(df_test[['id', 'binds']])\n\nprint(\"Finished training and prediction for all proteins.\")","metadata":{"id":"1f079e7b","outputId":"f62ce11a-0e59-400d-fc40-750f37e2916b"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combine all predictions for submission continuos\nsubmission = pd.concat(submission_list)\nsubmission = submission.sort_values(by='id')\nsubmission.to_csv('/content/submission.csv', index=False)","metadata":{"id":"nqWIsqn54nAH"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Auto-submit to Kaggle\n!kaggle competitions submit -c leash-BELKA -f submission.csv -m \"Belka_prediction_XGBoostA100_2_ecfp\"","metadata":{"id":"C3VXz8csQ27H"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"id":"vHPSTTxDs9Kh"},"execution_count":null,"outputs":[]}]}