{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 0. INTRODUCTION\n\n- To shoten trial time, save ECFP transformed and balansed data\n- test data only","metadata":{}},{"cell_type":"markdown","source":"# 1. Prepare","metadata":{}},{"cell_type":"markdown","source":"## 1.1. install nescesary tools\n\n- DuckDB\n    - DuckDB is a fast in-process analytical database\n\n    \n- RDKit\n    - RDKit is a collection of cheminformatics and machine-learning software written in C++ and Python.","metadata":{}},{"cell_type":"code","source":"!pip install duckdb ## ENABLE when factory reset","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install rdkit ## ENABLE when factory reset","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.2. import libraries","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import dask.dataframe as dd\nimport duckdb","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import AllChem","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.preprocessing import OneHotEncoder","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nimport datetime","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from multiprocessing import Pool","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.3. define functions and classies","metadata":{}},{"cell_type":"markdown","source":"- from below code, THANK YOU!\n    - [leash-tutorial-ecfps-and-random-forest](https://www.kaggle.com/code/andrewdblevins/leash-tutorial-ecfps-and-random-forest)\n    - [chemensemble-molecular-binding-with-ensemble](https://www.kaggle.com/code/yujansaya/chemensemble-molecular-binding-with-ensemble)","metadata":{}},{"cell_type":"code","source":"def generate_ecfp(molecule, radius=2, bits=1024):\n    if molecule is None:\n        return None\n    return list(Chem.AllChem.GetMorganFingerprintAsBitVect(molecule, radius, nBits=bits))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def from_SMILE_series_to_ecfp(smile_str_series): \n    return smile_str_series.apply(Chem.MolFromSmiles).apply(generate_ecfp)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_ECFP_transformed_df(df_in, \n          clnms_SMILES = [\"buildingblock1_smiles\", \"buildingblock2_smiles\", \"buildingblock3_smiles\", \"molecule_smiles\"], \n          clnms_identify = [\"id\", \"protein_name\", \"binds\"], \n         ): \n    \n    \n    prefix_synonym = {\n        \"buildingblock1_smiles\": \"bb1\", \n        \"buildingblock2_smiles\": \"bb2\", \n        \"buildingblock3_smiles\": \"bb3\", \n        \"molecule_smiles\": \"mlcl\"\n    }\n    \n    ecfp_ = {}\n    df_ecfp_ = {}\n    \n    t0_0 = time.time()\n    for clnm in clnms_SMILES: \n        ecfp_[clnm]= from_SMILE_series_to_ecfp(df_in.loc[:, clnm])\n\n    for clnm in clnms_SMILES: \n        df_ecfp_[clnm] = pd.DataFrame(ecfp_[clnm].tolist())\n        df_ecfp_[clnm] = df_ecfp_[clnm].rename(columns=lambda s:\"{}_{}\".format(prefix_synonym[clnm], s))\n        \n    df_train_Xy = df_in.loc[:, clnms_identify].reset_index(drop=True)\n    for k, v in df_ecfp_.items(): \n        df_train_Xy = pd.concat(\n            [\n                df_train_Xy, \n                v.reset_index(drop=True)\n            ], axis=1\n        )\n        \n        \n    t0_1 = time.time()\n    print(\"get_ECFP_transformed_df(): index {} ~ {}, elapsed time: {:0.03f} sec\".format(df_in.index.min(), df_in.index.max(), t0_1-t0_0))\n    \n    return df_train_Xy\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- to make cross product set for iterator","metadata":{}},{"cell_type":"code","source":"def get_cross_product_parameter_set(prm_dict): \n    key_list = []\n    tmp_list = [[]]\n    for k, val_vctr in prm_dict.items(): \n        tmp_list = [s+[t] for s in tmp_list for t in val_vctr]\n        key_list += [k]\n\n    ret_dict = {}\n    for i1, val_vctr in enumerate(tmp_list): \n        ret_dict[i1] = {k:v for k, v in zip(key_list, val_vctr)}\n        \n    return ret_dict\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.4. set input and output parameteres","metadata":{}},{"cell_type":"code","source":"prms_in_ = {\n    \"test\": {\n        \"file\": {\n            \"path\": \"/kaggle/input/leash-BELKA\", \n            \"name\": \"test.parquet\"\n        }\n    }\n}\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prms_out_ = {\n    \"test\": {\n        \"file\": {\n            \"path\": \"/kaggle/working\", \n            \"name\": \"test_transformed_{:04d}.parquet\"\n        }\n    }\n}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. read files","metadata":{}},{"cell_type":"markdown","source":"## 2.1. with DuckDB","metadata":{}},{"cell_type":"markdown","source":"### 2.1.1. set query","metadata":{}},{"cell_type":"code","source":"query_str_ = {}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"query_str_[\"test\"] = \"\"\"\nSELECT * FROM parquet_scan('{test_path}')\n\"\"\".format(\n    test_path=\"/\".join([v for k, v in prms_in_[\"test\"][\"file\"].items()]), \n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.1.2. execute query (read)","metadata":{}},{"cell_type":"code","source":"df_ = {}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t0_0 = time.time()\n\ncon = duckdb.connect()\ndf_[\"test\"] = con.query(query_str_[\"test\"]).df()\ncon.close()\n\nt0_1 = time.time()\nprint(\"test, elapsed time: {:0.03f} sec\".format(t0_1-t0_0))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. make feature","metadata":{}},{"cell_type":"code","source":"# rng = np.random.default_rng(99999)\n# indcs = rng.choice(df_[\"test\"].index, 16384, replace=False)\n# df_[\"test\"] = df_[\"test\"].loc[indcs, :].reset_index(drop=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chunksize = 16384 ## 2048, 16384\n\nclnms_SMILES = [\"buildingblock1_smiles\", \"buildingblock2_smiles\", \"buildingblock3_smiles\", \"molecule_smiles\"]\nclnms_id_ = {\n    \"train\": [\"id\", \"protein_name\", \"binds\"], \n    \"test\": [\"id\", \"protein_name\"]\n}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"info_lgcl_nm = \"test\"\ndf_sub = df_[info_lgcl_nm]\n## for info_lgcl_nm, df_sub in df_.items(): ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t0_0 = time.time()\nprint(\"-- {}: {}, start ---------- ---------- ----------\".format(\n    datetime.datetime.now().strftime(\"%Y-%m-%d %H:%M:%S\"), \n    info_lgcl_nm\n))\nprint(\"row : {:,d}\".format(len(df_sub)))\n\nif len(df_sub)%chunksize==0: \n    max_iterate = len(df_sub)//chunksize\nelse: \n    max_iterate = len(df_sub)//chunksize+1\n\ndf_sub_split_l = [df_sub.loc[chunksize*s:chunksize*(s+1)-1, ] for r, s in enumerate(range(0, max_iterate, 1))]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_[info_lgcl_nm], df_sub","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# p = Pool(processes=2)\n\n# get_funcs = [p.apply_async(get_ECFP_transformed_df, (s, clnms_SMILES, clnms_id_[info_lgcl_nm])) for s in df_sub_split_l]\n# for i1, f in enumerate(get_funcs): \n#     flnm_out = prms_out_[info_lgcl_nm][\"file\"][\"name\"].format(i1+1)\n#     f.get().to_parquet(\"{}/{}\".format(path_out, flnm_out))\n\n# p.close() ## これ以上プールでタスクが実行されないようにします。すべてのタスクが完了した後でワーカープロセスが終了します。\n# p.join() ## ワーカープロセスが終了するのを待ちます。 join() を使用する前に close() か terminate() を呼び出さなければなりません。\n\npath_out = prms_out_[info_lgcl_nm][\"file\"][\"path\"]\nos.makedirs(path_out, exist_ok=True)\n\nfor i1, s in enumerate(df_sub_split_l): \n    flnm_out = prms_out_[info_lgcl_nm][\"file\"][\"name\"].format(i1+1)\n    get_ECFP_transformed_df(s, clnms_SMILES, clnms_id_[info_lgcl_nm]).to_parquet(\"{}/{}\".format(path_out, flnm_out))\n\nt0_1 = time.time()\nprint(\"elapsed time: {:0.03f} sec\\n\".format(t0_1-t0_0))","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}