{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 0. INTRODUCTION","metadata":{}},{"cell_type":"markdown","source":"# 1. Prepare","metadata":{}},{"cell_type":"markdown","source":"## 1.1. install nescesary tools\n\n- DuckDB\n    - DuckDB is a fast in-process analytical database\n\n    \n- RDKit\n    - RDKit is a collection of cheminformatics and machine-learning software written in C++ and Python.","metadata":{}},{"cell_type":"code","source":"!pip install duckdb ## ENABLE when factory reset","metadata":{"execution":{"iopub.status.busy":"2024-05-01T04:31:09.939785Z","iopub.execute_input":"2024-05-01T04:31:09.940177Z","iopub.status.idle":"2024-05-01T04:31:25.216669Z","shell.execute_reply.started":"2024-05-01T04:31:09.940147Z","shell.execute_reply":"2024-05-01T04:31:25.215233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install rdkit ## ENABLE when factory reset","metadata":{"execution":{"iopub.status.busy":"2024-05-01T04:31:25.219957Z","iopub.execute_input":"2024-05-01T04:31:25.220312Z","iopub.status.idle":"2024-05-01T04:31:40.047110Z","shell.execute_reply.started":"2024-05-01T04:31:25.220281Z","shell.execute_reply":"2024-05-01T04:31:40.045795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.2. import libraries\n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-01T04:31:40.048780Z","iopub.execute_input":"2024-05-01T04:31:40.049143Z","iopub.status.idle":"2024-05-01T04:31:40.488883Z","shell.execute_reply.started":"2024-05-01T04:31:40.049112Z","shell.execute_reply":"2024-05-01T04:31:40.487699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import dask.dataframe as dd\nimport duckdb","metadata":{"execution":{"iopub.status.busy":"2024-05-01T04:31:40.491101Z","iopub.execute_input":"2024-05-01T04:31:40.491595Z","iopub.status.idle":"2024-05-01T04:31:41.694774Z","shell.execute_reply.started":"2024-05-01T04:31:40.491562Z","shell.execute_reply":"2024-05-01T04:31:41.693656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import AllChem","metadata":{"execution":{"iopub.status.busy":"2024-05-01T04:31:41.696308Z","iopub.execute_input":"2024-05-01T04:31:41.696833Z","iopub.status.idle":"2024-05-01T04:31:41.945572Z","shell.execute_reply.started":"2024-05-01T04:31:41.696790Z","shell.execute_reply":"2024-05-01T04:31:41.944443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2024-05-01T04:31:41.947209Z","iopub.execute_input":"2024-05-01T04:31:41.947754Z","iopub.status.idle":"2024-05-01T04:31:42.579905Z","shell.execute_reply.started":"2024-05-01T04:31:41.947722Z","shell.execute_reply":"2024-05-01T04:31:42.578751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.preprocessing import OneHotEncoder","metadata":{"execution":{"iopub.status.busy":"2024-05-01T04:31:42.581299Z","iopub.execute_input":"2024-05-01T04:31:42.581692Z","iopub.status.idle":"2024-05-01T04:31:42.589742Z","shell.execute_reply.started":"2024-05-01T04:31:42.581660Z","shell.execute_reply":"2024-05-01T04:31:42.588319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time","metadata":{"execution":{"iopub.status.busy":"2024-05-01T04:31:42.591199Z","iopub.execute_input":"2024-05-01T04:31:42.592323Z","iopub.status.idle":"2024-05-01T04:31:42.601959Z","shell.execute_reply.started":"2024-05-01T04:31:42.592272Z","shell.execute_reply":"2024-05-01T04:31:42.600689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.3. define functions and classies","metadata":{}},{"cell_type":"markdown","source":"- from below code, THANK YOU!\n    - [leash-tutorial-ecfps-and-random-forest](https://www.kaggle.com/code/andrewdblevins/leash-tutorial-ecfps-and-random-forest)","metadata":{}},{"cell_type":"code","source":"def generate_ecfp(molecule, radius=2, bits=1024):\n    if molecule is None:\n        return None\n    return list(Chem.AllChem.GetMorganFingerprintAsBitVect(molecule, radius, nBits=bits))","metadata":{"execution":{"iopub.status.busy":"2024-05-01T04:31:42.603469Z","iopub.execute_input":"2024-05-01T04:31:42.603932Z","iopub.status.idle":"2024-05-01T04:31:42.614785Z","shell.execute_reply.started":"2024-05-01T04:31:42.603891Z","shell.execute_reply":"2024-05-01T04:31:42.613132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.4. set input and output parameteres","metadata":{}},{"cell_type":"code","source":"prms_in_ = {\n    \"train\": {\n        \"file\": {\n            \"path\": \"/kaggle/input/leash-BELKA\", \n            \"name\": \"train.parquet\"\n        }\n    }, \n    \"test\": {\n        \"file\": {\n            \"path\": \"/kaggle/input/leash-BELKA\", \n            \"name\": \"test.parquet\"\n        }\n    }\n}\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-01T04:31:42.619184Z","iopub.execute_input":"2024-05-01T04:31:42.620209Z","iopub.status.idle":"2024-05-01T04:31:42.628610Z","shell.execute_reply.started":"2024-05-01T04:31:42.620170Z","shell.execute_reply":"2024-05-01T04:31:42.626809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prms_out_ = {\n    \"info_logical_name\": {\n        \"file\": {\n            \"path\": \"/kaggle/working/leash-BELKA/002_introduce_chem\", \n            \"name\": \"XXX.csv\"\n        }\n    }\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-01T04:31:42.630091Z","iopub.execute_input":"2024-05-01T04:31:42.630554Z","iopub.status.idle":"2024-05-01T04:31:42.639952Z","shell.execute_reply.started":"2024-05-01T04:31:42.630507Z","shell.execute_reply":"2024-05-01T04:31:42.638736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. read files\n\n## 2.1. with dask","metadata":{}},{"cell_type":"code","source":"dskdf_ = {}\nfor info_lgcl_nm in [\"train\", \"test\"]: \n    dskdf_[info_lgcl_nm] = dd.read_parquet(\n        \"/\".join([v for k, v in prms_in_[info_lgcl_nm][\"file\"].items()])\n    )","metadata":{"execution":{"iopub.status.busy":"2024-05-01T04:31:42.641026Z","iopub.execute_input":"2024-05-01T04:31:42.641328Z","iopub.status.idle":"2024-05-01T04:31:42.737817Z","shell.execute_reply.started":"2024-05-01T04:31:42.641304Z","shell.execute_reply":"2024-05-01T04:31:42.736754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.1. with DuckDB","metadata":{}},{"cell_type":"code","source":"query_str_ = {\n    \"train\": \"\"\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-01T04:31:42.739012Z","iopub.execute_input":"2024-05-01T04:31:42.739309Z","iopub.status.idle":"2024-05-01T04:31:42.744062Z","shell.execute_reply.started":"2024-05-01T04:31:42.739285Z","shell.execute_reply":"2024-05-01T04:31:42.742964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"query_str_[\"train\"] = \"\"\"\n(\n    SELECT * FROM parquet_scan('{train_path}') \n    WHERE binds = 0\n    ORDER BY random() \n    LIMIT 10000\n)\nUNION ALL \n(\n    SELECT * FROM parquet_scan('{train_path}') \n    WHERE binds = 1\n    ORDER BY random() \n    LIMIT 10000\n)\n\"\"\".format(\n    train_path=\"/\".join([v for k, v in prms_in_[\"train\"][\"file\"].items()])\n)\nprint(\"query: {query_str}\".format(query_str=query_str_[\"train\"]))","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-05-01T04:31:42.745654Z","iopub.execute_input":"2024-05-01T04:31:42.746295Z","iopub.status.idle":"2024-05-01T04:31:42.762558Z","shell.execute_reply.started":"2024-05-01T04:31:42.746254Z","shell.execute_reply":"2024-05-01T04:31:42.761445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dckdf_ = {}\n\ncon = duckdb.connect()\ndckdf_[\"train\"] = con.query(query_str_[\"train\"]).df()\ncon.close()","metadata":{"execution":{"iopub.status.busy":"2024-05-01T04:31:42.764115Z","iopub.execute_input":"2024-05-01T04:31:42.764546Z","iopub.status.idle":"2024-05-01T04:32:32.773895Z","shell.execute_reply.started":"2024-05-01T04:31:42.764514Z","shell.execute_reply":"2024-05-01T04:32:32.772615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. sample visualization","metadata":{}},{"cell_type":"code","source":"i1 = 0\nrow_i1 = dckdf_[\"train\"].loc[i1, :]","metadata":{"execution":{"iopub.status.busy":"2024-05-01T04:32:32.775523Z","iopub.execute_input":"2024-05-01T04:32:32.775875Z","iopub.status.idle":"2024-05-01T04:32:32.781184Z","shell.execute_reply.started":"2024-05-01T04:32:32.775844Z","shell.execute_reply":"2024-05-01T04:32:32.779964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mlcl = Chem.MolFromSmiles(row_i1[\"buildingblock1_smiles\"])\nChem.Draw.MolToImage(mlcl, size=(4*100, 3*100))","metadata":{"execution":{"iopub.status.busy":"2024-05-01T04:32:32.782540Z","iopub.execute_input":"2024-05-01T04:32:32.782851Z","iopub.status.idle":"2024-05-01T04:32:32.835855Z","shell.execute_reply.started":"2024-05-01T04:32:32.782826Z","shell.execute_reply":"2024-05-01T04:32:32.834748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# X. ref\n\n- [https://www.kaggle.com/code/yujansaya/chemensemble-molecular-binding-with-ensemble](https://www.kaggle.com/code/yujansaya/chemensemble-molecular-binding-with-ensemble)\n- [https://www.kaggle.com/code/andrewdblevins/leash-tutorial-ecfps-and-random-forest](https://www.kaggle.com/code/andrewdblevins/leash-tutorial-ecfps-and-random-forest)","metadata":{}}]}