{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30733,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-13T17:34:34.371743Z","iopub.execute_input":"2024-06-13T17:34:34.372138Z","iopub.status.idle":"2024-06-13T17:34:35.431842Z","shell.execute_reply.started":"2024-06-13T17:34:34.372105Z","shell.execute_reply":"2024-06-13T17:34:35.430673Z"},"trusted":true},"execution_count":1,"outputs":[{"name":"stdout","text":"/kaggle/input/leash-BELKA/sample_submission.csv\n/kaggle/input/leash-BELKA/train.parquet\n/kaggle/input/leash-BELKA/test.parquet\n/kaggle/input/leash-BELKA/train.csv\n/kaggle/input/leash-BELKA/test.csv\n","output_type":"stream"}]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2024-06-13T17:35:56.692396Z","iopub.execute_input":"2024-06-13T17:35:56.693011Z","iopub.status.idle":"2024-06-13T17:35:57.657562Z","shell.execute_reply.started":"2024-06-13T17:35:56.692972Z","shell.execute_reply":"2024-06-13T17:35:57.656051Z"},"trusted":true},"execution_count":2,"outputs":[]},{"cell_type":"code","source":"!pip install duckdb","metadata":{"execution":{"iopub.status.busy":"2024-06-13T17:36:03.175466Z","iopub.execute_input":"2024-06-13T17:36:03.175887Z","iopub.status.idle":"2024-06-13T17:36:21.041011Z","shell.execute_reply.started":"2024-06-13T17:36:03.175854Z","shell.execute_reply":"2024-06-13T17:36:21.039806Z"},"trusted":true},"execution_count":3,"outputs":[{"name":"stdout","text":"Collecting duckdb\n  Downloading duckdb-1.0.0-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (762 bytes)\nDownloading duckdb-1.0.0-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl (18.5 MB)\n\u001b[2K   \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m18.5/18.5 MB\u001b[0m \u001b[31m65.3 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m:00:01\u001b[0m00:01\u001b[0m\n\u001b[?25hInstalling collected packages: duckdb\nSuccessfully installed duckdb-1.0.0\n","output_type":"stream"}]},{"cell_type":"code","source":"!pip install rdkit","metadata":{"execution":{"iopub.status.busy":"2024-06-13T17:36:21.043393Z","iopub.execute_input":"2024-06-13T17:36:21.043803Z","iopub.status.idle":"2024-06-13T17:36:38.741576Z","shell.execute_reply.started":"2024-06-13T17:36:21.043768Z","shell.execute_reply":"2024-06-13T17:36:38.740361Z"},"trusted":true},"execution_count":4,"outputs":[{"name":"stdout","text":"Collecting rdkit\n  Downloading rdkit-2023.9.6-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl.metadata (3.9 kB)\nRequirement already satisfied: numpy in /opt/conda/lib/python3.10/site-packages (from rdkit) (1.26.4)\nRequirement already satisfied: Pillow in /opt/conda/lib/python3.10/site-packages (from rdkit) (9.5.0)\nDownloading rdkit-2023.9.6-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl (34.9 MB)\n\u001b[2K   \u001b[90m━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[32m34.9/34.9 MB\u001b[0m \u001b[31m40.5 MB/s\u001b[0m eta \u001b[36m0:00:00\u001b[0m:00:01\u001b[0m0:01\u001b[0mm\n\u001b[?25hInstalling collected packages: rdkit\nSuccessfully installed rdkit-2023.9.6\n","output_type":"stream"}]},{"cell_type":"code","source":"import pandas as pd\nimport duckdb\ntrain_path = '/kaggle/input/leash-BELKA/train.parquet'\ntest_path = '/kaggle/input/leash-BELKA/test.parquet'\n\ncon = duckdb.connect()\n\ndf = con.query(f\"\"\"(SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 0\n                        ORDER BY random()\n                        LIMIT 30000)\n                        UNION ALL\n                        (SELECT *\n                        FROM parquet_scan('{train_path}')\n                        WHERE binds = 1\n                        ORDER BY random()\n                        LIMIT 30000)\"\"\").df()\n\ncon.close()","metadata":{"execution":{"iopub.status.busy":"2024-06-13T17:57:59.569393Z","iopub.execute_input":"2024-06-13T17:57:59.570647Z","iopub.status.idle":"2024-06-13T17:58:53.13038Z","shell.execute_reply.started":"2024-06-13T17:57:59.570606Z","shell.execute_reply":"2024-06-13T17:58:53.129277Z"},"trusted":true},"execution_count":8,"outputs":[{"output_type":"display_data","data":{"text/plain":"FloatProgress(value=0.0, layout=Layout(width='auto'), style=ProgressStyle(bar_color='black'))","application/vnd.jupyter.widget-view+json":{"version_major":2,"version_minor":0,"model_id":"4a7ecb885f7b4b1db6333ce45481d24d"}},"metadata":{}}]},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import AllChem\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.preprocessing import OneHotEncoder\n\n# Convert SMILES to RDKit molecules\ndf['molecule'] = df['molecule_smiles'].apply(Chem.MolFromSmiles)\n\n# Generate ECFPs\ndef generate_ecfp(molecule, radius=2, bits=1024):\n    if molecule is None:\n        return None\n    return list(AllChem.GetMorganFingerprintAsBitVect(molecule, radius, nBits=bits))\n\ndf['ecfp'] = df['molecule'].apply(generate_ecfp)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nonehot_encoder = OneHotEncoder(sparse_output=False)\nprotein_onehot = onehot_encoder.fit_transform(df['protein_name'].values.reshape(-1, 1))\n\nX = [ecfp + protein for ecfp, protein in zip(df['ecfp'].tolist(), protein_onehot.tolist())]\ny = df['binds'].tolist()\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:18:29.527123Z","iopub.execute_input":"2024-06-13T08:18:29.527524Z","iopub.status.idle":"2024-06-13T08:18:32.29595Z","shell.execute_reply.started":"2024-06-13T08:18:29.527493Z","shell.execute_reply":"2024-06-13T08:18:32.294804Z"},"trusted":true},"execution_count":16,"outputs":[]},{"cell_type":"code","source":"X= np.array(X)\ny= np.array(y)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:18:34.697106Z","iopub.execute_input":"2024-06-13T08:18:34.697513Z","iopub.status.idle":"2024-06-13T08:18:40.374209Z","shell.execute_reply.started":"2024-06-13T08:18:34.697481Z","shell.execute_reply":"2024-06-13T08:18:40.373093Z"},"trusted":true},"execution_count":17,"outputs":[]},{"cell_type":"code","source":"X","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:18:40.375867Z","iopub.execute_input":"2024-06-13T08:18:40.37619Z","iopub.status.idle":"2024-06-13T08:18:40.388162Z","shell.execute_reply.started":"2024-06-13T08:18:40.376164Z","shell.execute_reply":"2024-06-13T08:18:40.38709Z"},"trusted":true},"execution_count":18,"outputs":[{"execution_count":18,"output_type":"execute_result","data":{"text/plain":"array([[1., 0., 1., ..., 0., 0., 1.],\n       [0., 1., 1., ..., 1., 0., 0.],\n       [0., 0., 0., ..., 0., 0., 1.],\n       ...,\n       [0., 0., 0., ..., 1., 0., 0.],\n       [0., 1., 0., ..., 0., 1., 0.],\n       [0., 1., 0., ..., 0., 1., 0.]])"},"metadata":{}}]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(X,y, test_size = 0.25, random_state = 42)\nprint(\"Size of training set:\", X_train.shape)\nprint(\"Size of test set:\", X_val.shape)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:20:37.716635Z","iopub.execute_input":"2024-06-13T08:20:37.717595Z","iopub.status.idle":"2024-06-13T08:20:37.937923Z","shell.execute_reply.started":"2024-06-13T08:20:37.717554Z","shell.execute_reply":"2024-06-13T08:20:37.936906Z"},"trusted":true},"execution_count":19,"outputs":[{"name":"stdout","text":"Size of training set: (45000, 1027)\nSize of test set: (15000, 1027)\n","output_type":"stream"}]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom sklearn.metrics import f1_score\nfrom catboost import CatBoostClassifier","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:20:38.545867Z","iopub.execute_input":"2024-06-13T08:20:38.546764Z","iopub.status.idle":"2024-06-13T08:20:38.55124Z","shell.execute_reply.started":"2024-06-13T08:20:38.546728Z","shell.execute_reply":"2024-06-13T08:20:38.550198Z"},"trusted":true},"execution_count":20,"outputs":[]},{"cell_type":"code","source":"import optuna\n\ndef optimize_hp(trial):\n    cb_params = {\n        'iterations': 400,\n        'learning_rate': trial.suggest_loguniform('learning_rate', 0.1, 1.0),\n        'l2_leaf_reg': trial.suggest_loguniform('l2_leaf_reg', 1, 100),\n        'bagging_temperature': trial.suggest_loguniform('bagging_temperature', 0.1, 20.0),\n        'random_strength': trial.suggest_float('random_strength', 1.0, 2.0),\n        'depth': trial.suggest_int('depth', 1, 10),\n        'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 1, 300),\n        \"use_best_model\": True,\n        \"task_type\": \"GPU\",\n        'random_seed': 42\n    }\n    \n    model = CatBoostClassifier(**cb_params)\n    model.fit(X_train, y_train, eval_set=(X_val, y_val), verbose=False)\n    y_pred = model.predict(X_val)\n    return average_precision_score(y_val, y_pred)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:20:44.988698Z","iopub.execute_input":"2024-06-13T08:20:44.989094Z","iopub.status.idle":"2024-06-13T08:20:44.996597Z","shell.execute_reply.started":"2024-06-13T08:20:44.989062Z","shell.execute_reply":"2024-06-13T08:20:44.995151Z"},"trusted":true},"execution_count":22,"outputs":[]},{"cell_type":"code","source":"study = optuna.create_study(direction=\"maximize\")\nstudy.optimize(optimize_hp, n_trials=10)\nprint('Trials:', len(study.trials))\nprint('Best parameters:', study.best_trial.params)\nprint('Best score:', study.best_value)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostClassifier","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:57:27.074501Z","iopub.execute_input":"2024-06-13T07:57:27.074859Z","iopub.status.idle":"2024-06-13T07:57:27.079024Z","shell.execute_reply.started":"2024-06-13T07:57:27.074831Z","shell.execute_reply":"2024-06-13T07:57:27.078107Z"},"trusted":true},"execution_count":34,"outputs":[]},{"cell_type":"code","source":"model","metadata":{"execution":{"iopub.status.busy":"2024-06-13T08:04:06.433503Z","iopub.execute_input":"2024-06-13T08:04:06.434091Z","iopub.status.idle":"2024-06-13T08:04:06.439844Z","shell.execute_reply.started":"2024-06-13T08:04:06.43406Z","shell.execute_reply":"2024-06-13T08:04:06.438797Z"},"trusted":true},"execution_count":36,"outputs":[{"execution_count":36,"output_type":"execute_result","data":{"text/plain":"<catboost.core.CatBoostClassifier at 0x7895b8087490>"},"metadata":{}}]},{"cell_type":"code","source":"from catboost import CatBoostClassifier\ncatb_model = CatBoostClassifier().fit(X_train, y_train)\ny_pred_proba = catb_model.predict_proba(X_val)[:, 1]  # Probability of the positive class\n\n# Calculate the mean average precision\nmap_score = average_precision_score(y_val, y_pred_proba)\nprint(f\"Mean Average Precision (mAP): {map_score:.2f}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# catb_model","metadata":{"execution":{"iopub.status.busy":"2024-06-13T07:21:04.138276Z","iopub.execute_input":"2024-06-13T07:21:04.138698Z","iopub.status.idle":"2024-06-13T07:21:04.145332Z","shell.execute_reply.started":"2024-06-13T07:21:04.138667Z","shell.execute_reply":"2024-06-13T07:21:04.144113Z"},"trusted":true},"execution_count":18,"outputs":[{"execution_count":18,"output_type":"execute_result","data":{"text/plain":"<catboost.core.CatBoostClassifier at 0x7bb1533c75b0>"},"metadata":{}}]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}