{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install duckdb\n!pip install rdkit","metadata":{"execution":{"iopub.status.busy":"2024-05-05T19:30:45.799059Z","iopub.execute_input":"2024-05-05T19:30:45.799863Z","iopub.status.idle":"2024-05-05T19:31:19.332702Z","shell.execute_reply.started":"2024-05-05T19:30:45.799810Z","shell.execute_reply":"2024-05-05T19:31:19.331498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport numpy as np\nfrom tqdm import tqdm\nimport duckdb\nfrom rdkit import Chem\nfrom rdkit.Chem import AllChem\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OneHotEncoder\nfrom xgboost import XGBClassifier\nfrom catboost import CatBoostClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.metrics import average_precision_score\n\nfrom sklearn.svm import SVC\nfrom sklearn.ensemble import StackingClassifier, VotingClassifier, BaggingClassifier, RandomForestClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.calibration import CalibratedClassifierCV\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-05T19:31:39.078877Z","iopub.execute_input":"2024-05-05T19:31:39.079389Z","iopub.status.idle":"2024-05-05T19:31:42.361231Z","shell.execute_reply.started":"2024-05-05T19:31:39.079340Z","shell.execute_reply":"2024-05-05T19:31:42.360261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_path = '/kaggle/input/leash-BELKA/train.parquet'\ntest_path = '/kaggle/input/leash-BELKA/test.parquet'\n\ncon = duckdb.connect()\n\ndf = con.query(f\"\"\"(SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0\n                        ORDER BY random()\n                        LIMIT 30000)\n                        UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1\n                        ORDER BY random()\n                        LIMIT 30000)\"\"\").df()\n\ncon.close()","metadata":{"execution":{"iopub.status.busy":"2024-05-05T19:31:42.362681Z","iopub.execute_input":"2024-05-05T19:31:42.363249Z","iopub.status.idle":"2024-05-05T19:32:33.895141Z","shell.execute_reply.started":"2024-05-05T19:31:42.363218Z","shell.execute_reply":"2024-05-05T19:32:33.894347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Convert SMILES to RDKit molecules\ndf['molecule'] = df['molecule_smiles'].apply(Chem.MolFromSmiles)\n\n# Generate ECFPs\ndef generate_ecfp(molecule, radius=2, bits=1024):\n    if molecule is None:\n        return None\n    return list(AllChem.GetMorganFingerprintAsBitVect(molecule, radius, nBits=bits))\n\ndf['ecfp'] = df['molecule'].apply(generate_ecfp)","metadata":{"execution":{"iopub.status.busy":"2024-05-05T19:32:41.444079Z","iopub.execute_input":"2024-05-05T19:32:41.444480Z","iopub.status.idle":"2024-05-05T19:34:21.847225Z","shell.execute_reply.started":"2024-05-05T19:32:41.444450Z","shell.execute_reply":"2024-05-05T19:34:21.845357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nRANDOM_STATE = 42\n\n# One-hot encode the protein_name\nonehot_encoder = OneHotEncoder(sparse_output=False)\nprotein_onehot = onehot_encoder.fit_transform(df['protein_name'].values.reshape(-1, 1))\n\n# Combine ECFPs and one-hot encoded protein_name\nX = [ecfp + protein for ecfp, protein in zip(df['ecfp'].tolist(), protein_onehot.tolist())]\ny = df['binds'].tolist()\n\n# Split the data into train, validation and test sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=RANDOM_STATE)\n\nX_train, X_val, y_train, y_val = train_test_split(X_train, y_train, test_size=0.25, random_state=RANDOM_STATE)","metadata":{"execution":{"iopub.status.busy":"2024-05-05T19:35:02.169855Z","iopub.execute_input":"2024-05-05T19:35:02.170271Z","iopub.status.idle":"2024-05-05T19:35:03.916494Z","shell.execute_reply.started":"2024-05-05T19:35:02.170232Z","shell.execute_reply":"2024-05-05T19:35:03.915597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUMBER_OF_MODELS = 3\n# Create and train the ensemble models\ncatboost_model = CatBoostClassifier(iterations=500, random_state=RANDOM_STATE)\nlgbm_model = LGBMClassifier(random_state=RANDOM_STATE)\nxgb_model = XGBClassifier(random_state=RANDOM_STATE, n_jobs=-1,)\n\nmodels = [catboost_model, lgbm_model, xgb_model]\nfor model in models:\n    model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-05-05T19:35:05.575940Z","iopub.execute_input":"2024-05-05T19:35:05.577198Z","iopub.status.idle":"2024-05-05T19:36:50.204105Z","shell.execute_reply.started":"2024-05-05T19:35:05.577149Z","shell.execute_reply":"2024-05-05T19:36:50.202151Z"},"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_test = []\n# Calibrate probabilities on validation set\nfor model in models:\n    calibrated_model = CalibratedClassifierCV(model, method='sigmoid', cv='prefit')\n    calibrated_model.fit(X_val, y_val)\n    # Evaluate calibrated model on test set\n    calibrated_probabilities = calibrated_model.predict_proba(X_test)[:, 1]\n    predictions_test.append(calibrated_probabilities)\n    \n# Ensemble predictions for the test set\nensemble_predictions_test = np.mean(predictions_test, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-05-05T19:36:50.211495Z","iopub.execute_input":"2024-05-05T19:36:50.211931Z","iopub.status.idle":"2024-05-05T19:37:17.917711Z","shell.execute_reply.started":"2024-05-05T19:36:50.211878Z","shell.execute_reply":"2024-05-05T19:37:17.916710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the mean average precision\nmap_score = average_precision_score(y_test, ensemble_predictions_test)\nprint(f\"Mean Average Precision (mAP): {map_score:.8f}\")","metadata":{"execution":{"iopub.status.busy":"2024-05-05T19:37:17.919077Z","iopub.execute_input":"2024-05-05T19:37:17.919436Z","iopub.status.idle":"2024-05-05T19:37:17.939730Z","shell.execute_reply.started":"2024-05-05T19:37:17.919405Z","shell.execute_reply":"2024-05-05T19:37:17.938621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n# Process the test.parquet file chunk by chunk sinc the file is huge\ntest_file = '/kaggle/input/leash-BELKA/test.csv'\noutput_file = 'submission.csv'  # Specify the path and filename for the output file\n\n\n\n# Read the test.csv file into a pandas DataFrame\n# Wrap the loop with tqdm to display progress\nfor df_test in tqdm(pd.read_csv(test_file, chunksize=10_000)):\n    \n    # Generate ECFPs for the molecule_smiles\n    df_test['molecule'] = df_test['molecule_smiles'].apply(Chem.MolFromSmiles)\n    df_test['ecfp'] = df_test['molecule'].apply(generate_ecfp)\n\n    # One-hot encode the protein_name\n    protein_onehot = onehot_encoder.transform(df_test['protein_name'].values.reshape(-1, 1))\n\n    # Combine ECFPs and one-hot encoded protein_name\n    X_test = [ecfp + protein for ecfp, protein in zip(df_test['ecfp'].tolist(), protein_onehot.tolist())]\n    \n    predictions_test = []\n    # Calibrate probabilities on validation set\n    for model in models:\n        calibrated_model = CalibratedClassifierCV(model, method='sigmoid', cv='prefit')\n        calibrated_model.fit(X_val, y_val)\n        # Evaluate calibrated model on test set\n        calibrated_probabilities = calibrated_model.predict_proba(X_test)[:, 1]\n        predictions_test.append(calibrated_probabilities)\n\n    # Ensemble predictions for the test set\n    ensemble_predictions_test = np.mean(predictions_test, axis=0)\n\n#     # Predict the probabilities using ensemble models\n#     probabilities_catboost = catboost_model.predict_proba(X_test)[:, 1]\n#     probabilities_lgbm = lgbm_model.predict_proba(X_test)[:, 1]\n#     probabilities_xgb = xgb_model.predict_proba(X_test)[:, 1]\n\n#     # Average the predictions from the ensemble\n#     probabilities_ensemble = (probabilities_catboost + probabilities_lgbm + probabilities_xgb) / NUMBER_OF_MODELS\n    \n     # Append the probabilities to the list\n\n# Create a DataFrame with 'id' and 'probability' columns\n    output_df = pd.DataFrame({'id': df_test['id'], 'binds': ensemble_predictions_test})\n\n# Save the output DataFrame to a CSV file\n    output_df.to_csv(output_file, index=False, mode='a', header=not os.path.exists(output_file))","metadata":{"execution":{"iopub.status.busy":"2024-05-05T19:37:17.941814Z","iopub.execute_input":"2024-05-05T19:37:17.942132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}