{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 0. INTRODUCTION\n\n- make predict model \n- using ECFP (Extended-Connectivity Fingerprints) \n- using LightGBM","metadata":{}},{"cell_type":"markdown","source":"# 1. Prepare","metadata":{}},{"cell_type":"markdown","source":"## 1.1. install nescesary tools\n\n- DuckDB\n    - DuckDB is a fast in-process analytical database\n\n    \n- RDKit\n    - RDKit is a collection of cheminformatics and machine-learning software written in C++ and Python.","metadata":{}},{"cell_type":"code","source":"!pip install duckdb ## ENABLE when factory reset","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install rdkit ## ENABLE when factory reset","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.2. import libraries","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import dask.dataframe as dd\nimport duckdb","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import AllChem","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.preprocessing import OneHotEncoder","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.3. define functions and classies- from below code, THANK YOU!","metadata":{}},{"cell_type":"markdown","source":"- from below code, THANK YOU!\n    - [leash-tutorial-ecfps-and-random-forest](https://www.kaggle.com/code/andrewdblevins/leash-tutorial-ecfps-and-random-forest)\n    - [chemensemble-molecular-binding-with-ensemble](https://www.kaggle.com/code/yujansaya/chemensemble-molecular-binding-with-ensemble)","metadata":{}},{"cell_type":"code","source":"def generate_ecfp(molecule, radius=2, bits=1024):\n    if molecule is None:\n        return None\n    return list(Chem.AllChem.GetMorganFingerprintAsBitVect(molecule, radius, nBits=bits))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.4. set input and output parameteres","metadata":{}},{"cell_type":"code","source":"prms_in_ = {\n    \"train\": {\n        \"file\": {\n            \"path\": \"/kaggle/input/leash-BELKA\", \n            \"name\": \"train.parquet\"\n        }\n    }, \n    \"test\": {\n        \"file\": {\n            \"path\": \"/kaggle/input/leash-BELKA\", \n            \"name\": \"test.parquet\"\n        }\n    }\n}\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prms_out_ = {\n    \"submission\": {\n        \"file\": {\n            \"path\": \"/kaggle/working/leash-BELKA/003_LGB_with_ECFP\", \n            \"name\": \"submission.csv\"\n        }\n    }\n}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. read files\n\n## 2.1. with dask","metadata":{}},{"cell_type":"code","source":"dskdf_ = {}\nfor info_lgcl_nm in [\"train\", \"test\"]: \n    dskdf_[info_lgcl_nm] = dd.read_parquet(\n        \"/\".join([v for k, v in prms_in_[info_lgcl_nm][\"file\"].items()])\n    )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.1. with DuckDB","metadata":{}},{"cell_type":"markdown","source":"### 2.1.1. set query","metadata":{}},{"cell_type":"code","source":"query_str_ = {}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"query_str_[\"train\"] = \"\"\"\n(\n    SELECT * FROM parquet_scan('{train_path}') \n    WHERE binds = 0  AND protein_name = 'HSA'\n    ORDER BY random() \n    LIMIT {limit_row}\n)\nUNION ALL \n(\n    SELECT * FROM parquet_scan('{train_path}') \n    WHERE binds = 0  AND protein_name = 'BRD4'\n    ORDER BY random() \n    LIMIT {limit_row}\n)\nUNION ALL \n(\n    SELECT * FROM parquet_scan('{train_path}') \n    WHERE binds = 0  AND protein_name = 'sEH'\n    ORDER BY random() \n    LIMIT {limit_row}\n)\nUNION ALL \n(\n    SELECT * FROM parquet_scan('{train_path}') \n    WHERE binds = 1  AND protein_name = 'HSA'\n    ORDER BY random() \n    LIMIT {limit_row}\n)\nUNION ALL \n(\n    SELECT * FROM parquet_scan('{train_path}') \n    WHERE binds = 1  AND protein_name = 'BRD4'\n    ORDER BY random() \n    LIMIT {limit_row}\n)\nUNION ALL \n(\n    SELECT * FROM parquet_scan('{train_path}') \n    WHERE binds = 1  AND protein_name = 'sEH'\n    ORDER BY random() \n    LIMIT {limit_row}\n)\n\"\"\".format(\n    train_path=\"/\".join([v for k, v in prms_in_[\"train\"][\"file\"].items()]), \n    limit_row=10000\n)\nprint(\"query: {query_str}\".format(query_str=query_str_[\"train\"]))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"query_str_[\"test\"] = \"\"\"\nSELECT * FROM parquet_scan('{test_path}')\n\"\"\".format(\n    test_path=\"/\".join([v for k, v in prms_in_[\"test\"][\"file\"].items()]), \n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.1.2. execute query (read)","metadata":{}},{"cell_type":"code","source":"df_ = {}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"con = duckdb.connect()\ndf_[\"train\"] = con.query(query_str_[\"train\"]).df()\ncon.close()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"con = duckdb.connect()\ndf_[\"test\"] = con.query(query_str_[\"test\"]).df()\ncon.close()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. get ECFP","metadata":{}},{"cell_type":"code","source":"%%time\ndf_[\"train\"].loc[:, \"molecule\"] = df_[\"train\"].loc[:, \"molecule_smiles\"].apply(Chem.MolFromSmiles)\ndf_[\"train\"].loc[:, \"ecfp\"] = df_[\"train\"].loc[:, \"molecule\"].apply(generate_ecfp)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. make feature","metadata":{}},{"cell_type":"code","source":"onehot_encoder = OneHotEncoder(sparse_output=False)\nprotein_onehot = onehot_encoder.fit_transform(df_[\"train\"].loc[:, \"protein_name\"].values.reshape(-1, 1))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = [s+t for s, t in zip(df_[\"train\"].loc[:, \"ecfp\"].tolist(), protein_onehot.tolist())]\ny_train = df_[\"train\"].loc[:, \"binds\"].tolist()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. train","metadata":{}},{"cell_type":"code","source":"clf = lgb.LGBMClassifier()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf.fit(X_train, y_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. predict","metadata":{}},{"cell_type":"code","source":"out_probs_ = {}\n\nchunksize = int(len(df_[\"test\"])/100)\nlast_i1 = 0\n## i1 = chunksize\nfor i1 in range(chunksize, len(df_[\"test\"])+chunksize, chunksize): \n\n    t1_0 = time.time()\n\n    df_test_tmp = df_[\"test\"].loc[last_i1:i1, :].reset_index(drop=True)\n\n    df_test_tmp.loc[:, \"molecule\"] = df_test_tmp.loc[:, \"molecule_smiles\"].apply(Chem.MolFromSmiles)\n    df_test_tmp.loc[:, \"ecfp\"] = df_test_tmp.loc[:, \"molecule\"].apply(generate_ecfp)\n\n    onehot_encoder = OneHotEncoder(sparse_output=False)\n    protein_onehot = onehot_encoder.fit_transform(df_test_tmp.loc[:, \"protein_name\"].values.reshape(-1, 1))\n\n    X_test_tmp = [s+t for s, t in zip(df_test_tmp.loc[:, \"ecfp\"].tolist(), protein_onehot.tolist())]\n    y_pred_tmp = clf.predict_proba(X_test_tmp)\n\n    df_pred_tmp = pd.DataFrame(y_pred_tmp, columns=[\"complements_prob\", \"binds\"])\n    out_probs_[i1] = pd.concat(\n        [\n            df_test_tmp.loc[:, \"id\"], \n            df_pred_tmp.loc[:, \"binds\"]\n        ], axis=1\n    )\n    \n    last_i1 = i1\n    \n    t1_1 = time.time()\n\n    print(\"{} ~ {}, elapsed time: {:0.03f} sec\".format(last_i1, i1, t1_1-t1_0))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 7. output","metadata":{"execution":{"iopub.status.busy":"2024-05-01T05:18:52.547654Z","iopub.execute_input":"2024-05-01T05:18:52.548091Z","iopub.status.idle":"2024-05-01T05:18:52.561644Z","shell.execute_reply.started":"2024-05-01T05:18:52.548060Z","shell.execute_reply":"2024-05-01T05:18:52.560372Z"}}},{"cell_type":"code","source":"os.makedirs(prms_out_[\"submission\"][\"file\"][\"path\"], exist_ok=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_out = pd.concat(out_probs_).reset_index(drop=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_out.T.to_csv(\n    \"/\".join([v for k, v in prms_out_[\"submission\"][\"file\"].items()]), \n    sep=\",\",  \n    encoding=\"utf-8\", ## \"utf-8\", \"shift_jis\" \n    header=True,  \n    index=False    \n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}