{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30700,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-28T11:59:46.466948Z","iopub.execute_input":"2024-04-28T11:59:46.467278Z","iopub.status.idle":"2024-04-28T11:59:47.678468Z","shell.execute_reply.started":"2024-04-28T11:59:46.467247Z","shell.execute_reply":"2024-04-28T11:59:47.677732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Abstract\nOnly the columns 'molecule_smiles' and 'protein_names' have been used for training models. The other features 'buildingblock1_smiles', 'buildingblock2_smiles', and 'buildingblock3_smiles' haven't been utilized. Here, only a portion of the data from 'molecule_smiles' and 'protein' has been included. The file is very large, and I am extracting only 50250 rows from the entire dataset—250 rows containing binds = 1 and 50000 rows with binds = 0. I am choosing these numbers to maintain their ratios as 1:200, similar to the original dataset.\n\nCurrently, I'm implementing this based on ideas gathered from others in the code section. However, I'm uncertain whether this approach is optimal. Despite having other concepts in mind, I find it challenging to implement them in code. Furthermore, handling this large dataset is proving difficult, as it takes a significant amount of time to run the code in a Kaggle notebook. I would appreciate some better ideas to ensure that I can include atleast all the data from 'molecule_smiles' and 'protein_names' in training the model.\n\nAm I understanding this code as I explained?\nPlease suggest me through whatever suggestion you think is important so that i could improve myself.","metadata":{}},{"cell_type":"markdown","source":"# Importing reqired library functions","metadata":{}},{"cell_type":"code","source":"!pip install duckdb\n!pip install rdkit\n!pip install xgboost\n!pip install catboost\n!pip install lightgbm\n\n\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport numpy as np\nfrom tqdm import tqdm\n\nimport duckdb\nfrom rdkit import Chem\nfrom rdkit.Chem import AllChem\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.svm import SVC\nfrom sklearn.ensemble import StackingClassifier, VotingClassifier, BaggingClassifier, RandomForestClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.calibration import CalibratedClassifierCV\n\nfrom xgboost import XGBClassifier\nfrom catboost import CatBoostClassifier\nfrom lightgbm import LGBMClassifier","metadata":{"execution":{"iopub.status.busy":"2024-04-28T12:04:56.557271Z","iopub.execute_input":"2024-04-28T12:04:56.557740Z","iopub.status.idle":"2024-04-28T12:05:57.563034Z","shell.execute_reply.started":"2024-04-28T12:04:56.557700Z","shell.execute_reply":"2024-04-28T12:05:57.562123Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Getting some amount of data from large data set","metadata":{}},{"cell_type":"code","source":"train_path = '/kaggle/input/leash-BELKA/train.parquet'\ntest_path = '/kaggle/input/leash-BELKA/test.parquet'\n\ncon = duckdb.connect()\n\ndf = con.query(f\"\"\"(SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0\n                        ORDER BY random()\n                        LIMIT 100000)\n                        UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1\n                        ORDER BY random()\n                        LIMIT 500)\"\"\").df()\n\ncon.close()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-28T12:07:43.868841Z","iopub.execute_input":"2024-04-28T12:07:43.869176Z","iopub.status.idle":"2024-04-28T12:07:56.694636Z","shell.execute_reply.started":"2024-04-28T12:07:43.869147Z","shell.execute_reply":"2024-04-28T12:07:56.693772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Converting molecular_smiles to ECFPs\n","metadata":{}},{"cell_type":"code","source":"# Convert SMILES to RDKit molecules\ndf['molecule'] = df['molecule_smiles'].apply(Chem.MolFromSmiles)\n\n# Generate ECFPs\ndef generate_ecfp(molecule, radius=2, bits=1024):\n    if molecule is None:\n        return None\n    return list(AllChem.GetMorganFingerprintAsBitVect(molecule, radius, nBits=bits))\n\ndf['ecfp'] = df['molecule'].apply(generate_ecfp)","metadata":{"execution":{"iopub.status.busy":"2024-04-28T12:10:19.352792Z","iopub.execute_input":"2024-04-28T12:10:19.353115Z","iopub.status.idle":"2024-04-28T12:10:57.284922Z","shell.execute_reply.started":"2024-04-28T12:10:19.353087Z","shell.execute_reply":"2024-04-28T12:10:57.284100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# One-hot encode the protein_name\n","metadata":{}},{"cell_type":"code","source":"onehot_encoder = OneHotEncoder(sparse_output=False)\nprotein_onehot = onehot_encoder.fit_transform(df['protein_name'].values.reshape(-1,1))","metadata":{"execution":{"iopub.status.busy":"2024-04-28T12:14:44.626978Z","iopub.execute_input":"2024-04-28T12:14:44.627316Z","iopub.status.idle":"2024-04-28T12:14:44.645107Z","shell.execute_reply.started":"2024-04-28T12:14:44.627275Z","shell.execute_reply":"2024-04-28T12:14:44.644380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Concatenating 2 list and making a single list X.The two lists are df[‘ecfp’] and protein_onehot\n","metadata":{}},{"cell_type":"code","source":"# Initialize an empty list to store the combined elements\nX = [ ]\n\n# Iterate over corresponding elements from 'ecfp' and 'protein_onehot' lists\nfor ecfp, protein in zip(df['ecfp'].tolist(), protein_onehot.tolist()):\n    # Combine 'ecfp' and 'protein' elements and add to the list\n    combined_element = ecfp + protein\n    X.append(combined_element)\n\ny=df[\"binds\"].tolist()","metadata":{"execution":{"iopub.status.busy":"2024-04-28T12:18:49.240116Z","iopub.execute_input":"2024-04-28T12:18:49.240530Z","iopub.status.idle":"2024-04-28T12:18:50.297652Z","shell.execute_reply.started":"2024-04-28T12:18:49.240428Z","shell.execute_reply":"2024-04-28T12:18:50.296805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Splitting the data into train, validation and test sets","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=0)\n\nX_train, X_val, y_train, y_val = train_test_split(X_train, y_train, test_size=0.25, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2024-04-28T12:19:29.833001Z","iopub.execute_input":"2024-04-28T12:19:29.833355Z","iopub.status.idle":"2024-04-28T12:19:29.869243Z","shell.execute_reply.started":"2024-04-28T12:19:29.833300Z","shell.execute_reply":"2024-04-28T12:19:29.868542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating and training models","metadata":{}},{"cell_type":"code","source":"# Create and train the ensemble models\ncatboost_model = CatBoostClassifier(iterations=500, random_state=0)\nlgbm_model = LGBMClassifier(random_state=0)\nxgb_model = XGBClassifier(random_state=0, n_jobs=-1,)\n\nmodels = [catboost_model, lgbm_model, xgb_model]\nfor model in models:\n    model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-04-28T12:19:33.248948Z","iopub.execute_input":"2024-04-28T12:19:33.249784Z","iopub.status.idle":"2024-04-28T12:20:04.139825Z","shell.execute_reply.started":"2024-04-28T12:19:33.249749Z","shell.execute_reply":"2024-04-28T12:20:04.138717Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calibrating models(I am in confusion on it)","metadata":{}},{"cell_type":"code","source":"predictions_test = [ ]\n# Calibrate probabilities on validation set\nfor model in models:\n    calibrated_model = CalibratedClassifierCV(model, method='sigmoid', cv='prefit')\n    calibrated_model.fit(X_val, y_val)\n    # Evaluate calibrated model on test set\n    calibrated_probabilities = calibrated_model.predict_proba(X_test)[:, 1]\n    predictions_test.append(calibrated_probabilities)\n    \n# Ensemble predictions for the test set\nensemble_predictions_test = np.mean(predictions_test, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-04-28T12:21:09.780467Z","iopub.execute_input":"2024-04-28T12:21:09.780833Z","iopub.status.idle":"2024-04-28T12:21:21.818377Z","shell.execute_reply.started":"2024-04-28T12:21:09.780802Z","shell.execute_reply":"2024-04-28T12:21:21.817312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction and submission","metadata":{}},{"cell_type":"code","source":"import os\n# Process the test.parquet file chunk by chunk sinc the file is huge\ntest_file = '/kaggle/input/leash-BELKA/test.csv'\noutput_file = 'submission.csv'  # Specify the path and filename for the output file\n\n\n\n# Read the test.csv file into a pandas DataFrame\n# Wrap the loop with tqdm to display progress\nfor df_test in tqdm(pd.read_csv(test_file, chunksize=10_000)):\n    \n    # Generate ECFPs for the molecule_smiles\n    df_test['molecule'] = df_test['molecule_smiles'].apply(Chem.MolFromSmiles)\n    df_test['ecfp'] = df_test['molecule'].apply(generate_ecfp)\n\n    # One-hot encode the protein_name\n    protein_onehot = onehot_encoder.transform(df_test['protein_name'].values.reshape(-1, 1))\n\n    # Combine ECFPs and one-hot encoded protein_name\n    X_test = [ecfp + protein for ecfp, protein in zip(df_test['ecfp'].tolist(), protein_onehot.tolist())]\n    \n    predictions_test = []\n    # Calibrate probabilities on validation set\n    for model in models:\n        calibrated_model = CalibratedClassifierCV(model, method='sigmoid', cv='prefit')\n        calibrated_model.fit(X_val, y_val)\n        # Evaluate calibrated model on test set\n        calibrated_probabilities = calibrated_model.predict_proba(X_test)[:, 1]\n        predictions_test.append(calibrated_probabilities)\n\n    # Ensemble predictions for the test set\n    ensemble_predictions_test = np.mean(predictions_test, axis=0)\n\n# Create a DataFrame with 'id' and 'probability' columns\n    output_df = pd.DataFrame({'id': df_test['id'], 'binds': ensemble_predictions_test})\n\n# Save the output DataFrame to a CSV file\n    output_df.to_csv(output_file, index=False, mode='a', header=not os.path.exists(output_file))","metadata":{"execution":{"iopub.status.busy":"2024-04-28T12:23:36.002802Z","iopub.execute_input":"2024-04-28T12:23:36.003128Z","iopub.status.idle":"2024-04-28T13:18:47.296937Z","shell.execute_reply.started":"2024-04-28T12:23:36.003100Z","shell.execute_reply":"2024-04-28T13:18:47.296148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}