{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 0. INTRODUCTION\n\n- make predict model \n- using ECFP (Extended-Connectivity Fingerprints) \n- using LightGBM","metadata":{}},{"cell_type":"markdown","source":"# 1. Prepare","metadata":{}},{"cell_type":"markdown","source":"## 1.1. install nescesary tools\n\n- DuckDB\n    - DuckDB is a fast in-process analytical database\n\n    \n- RDKit\n    - RDKit is a collection of cheminformatics and machine-learning software written in C++ and Python.","metadata":{}},{"cell_type":"code","source":"!pip install duckdb ## ENABLE when factory reset","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:16:33.113486Z","iopub.execute_input":"2024-05-04T16:16:33.113954Z","iopub.status.idle":"2024-05-04T16:16:50.693575Z","shell.execute_reply.started":"2024-05-04T16:16:33.113908Z","shell.execute_reply":"2024-05-04T16:16:50.692187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install rdkit ## ENABLE when factory reset","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:16:50.696420Z","iopub.execute_input":"2024-05-04T16:16:50.696907Z","iopub.status.idle":"2024-05-04T16:17:08.279699Z","shell.execute_reply.started":"2024-05-04T16:16:50.696860Z","shell.execute_reply":"2024-05-04T16:17:08.278306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.2. import libraries","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-04T16:17:08.282069Z","iopub.execute_input":"2024-05-04T16:17:08.282518Z","iopub.status.idle":"2024-05-04T16:17:08.750102Z","shell.execute_reply.started":"2024-05-04T16:17:08.282476Z","shell.execute_reply":"2024-05-04T16:17:08.748782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import dask.dataframe as dd\nimport duckdb","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:08.753260Z","iopub.execute_input":"2024-05-04T16:17:08.754244Z","iopub.status.idle":"2024-05-04T16:17:10.195043Z","shell.execute_reply.started":"2024-05-04T16:17:08.754198Z","shell.execute_reply":"2024-05-04T16:17:10.193799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from rdkit import Chem\nfrom rdkit.Chem import AllChem","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:10.196569Z","iopub.execute_input":"2024-05-04T16:17:10.197078Z","iopub.status.idle":"2024-05-04T16:17:10.455092Z","shell.execute_reply.started":"2024-05-04T16:17:10.197049Z","shell.execute_reply":"2024-05-04T16:17:10.453735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:10.459129Z","iopub.execute_input":"2024-05-04T16:17:10.459476Z","iopub.status.idle":"2024-05-04T16:17:11.148902Z","shell.execute_reply.started":"2024-05-04T16:17:10.459448Z","shell.execute_reply":"2024-05-04T16:17:11.147542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.preprocessing import OneHotEncoder","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.150439Z","iopub.execute_input":"2024-05-04T16:17:11.150771Z","iopub.status.idle":"2024-05-04T16:17:11.156961Z","shell.execute_reply.started":"2024-05-04T16:17:11.150743Z","shell.execute_reply":"2024-05-04T16:17:11.155608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nimport datetime","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.158949Z","iopub.execute_input":"2024-05-04T16:17:11.159665Z","iopub.status.idle":"2024-05-04T16:17:11.169498Z","shell.execute_reply.started":"2024-05-04T16:17:11.159629Z","shell.execute_reply":"2024-05-04T16:17:11.168071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from multiprocessing import Pool","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.171067Z","iopub.execute_input":"2024-05-04T16:17:11.171496Z","iopub.status.idle":"2024-05-04T16:17:11.179899Z","shell.execute_reply.started":"2024-05-04T16:17:11.171456Z","shell.execute_reply":"2024-05-04T16:17:11.178804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nfrom sklearn.cluster import KMeans\n\nimport matplotlib\nimport matplotlib.pyplot as plt\n\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.185627Z","iopub.execute_input":"2024-05-04T16:17:11.186342Z","iopub.status.idle":"2024-05-04T16:17:11.619631Z","shell.execute_reply.started":"2024-05-04T16:17:11.186281Z","shell.execute_reply":"2024-05-04T16:17:11.618333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.3. define functions and classies","metadata":{}},{"cell_type":"markdown","source":"- from below code, THANK YOU!\n    - [leash-tutorial-ecfps-and-random-forest](https://www.kaggle.com/code/andrewdblevins/leash-tutorial-ecfps-and-random-forest)\n    - [chemensemble-molecular-binding-with-ensemble](https://www.kaggle.com/code/yujansaya/chemensemble-molecular-binding-with-ensemble)","metadata":{}},{"cell_type":"code","source":"def generate_ecfp(molecule, radius=2, bits=1024):\n    if molecule is None:\n        return None\n    return list(Chem.AllChem.GetMorganFingerprintAsBitVect(molecule, radius, nBits=bits))","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.621047Z","iopub.execute_input":"2024-05-04T16:17:11.621422Z","iopub.status.idle":"2024-05-04T16:17:11.626435Z","shell.execute_reply.started":"2024-05-04T16:17:11.621390Z","shell.execute_reply":"2024-05-04T16:17:11.625510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def from_SMILE_series_to_ecfp(smile_str_series): \n    return smile_str_series.apply(Chem.MolFromSmiles).apply(generate_ecfp)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.627748Z","iopub.execute_input":"2024-05-04T16:17:11.628072Z","iopub.status.idle":"2024-05-04T16:17:11.639467Z","shell.execute_reply.started":"2024-05-04T16:17:11.628039Z","shell.execute_reply":"2024-05-04T16:17:11.638587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_X(df_in): \n    t0_0 = time.time()\n    \n    ecfp_bb1  = from_SMILE_series_to_ecfp(df_in.loc[:, \"buildingblock1_smiles\"])\n    ecfp_bb2  = from_SMILE_series_to_ecfp(df_in.loc[:, \"buildingblock2_smiles\"])\n    ecfp_bb3  = from_SMILE_series_to_ecfp(df_in.loc[:, \"buildingblock3_smiles\"])\n    ecfp_mlcl = from_SMILE_series_to_ecfp(df_in.loc[:, \"molecule_smiles\"])\n    \n    X = np.array([s+t+u+v for s, t, u, v in zip(ecfp_bb1, \n                                                ecfp_bb2, \n                                                ecfp_bb3, \n                                                ecfp_mlcl)])\n    t0_1 = time.time()\n    print(\"get_X(): index {} ~ {}, elapsed time: {:0.03f} sec\".format(df_in.index.min(), df_in.index.max(), t0_1-t0_0))\n    \n    return X\n    ","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.640865Z","iopub.execute_input":"2024-05-04T16:17:11.641844Z","iopub.status.idle":"2024-05-04T16:17:11.653832Z","shell.execute_reply.started":"2024-05-04T16:17:11.641813Z","shell.execute_reply":"2024-05-04T16:17:11.652880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_submission_data(df_in, clf_in): \n    t0_0 = time.time()\n    \n    X_test = get_X(df_in)\n    y_pred = clf_in.predict_proba(X_test)\n    \n    df_pred = pd.DataFrame(y_pred, columns=[\"complements_prob\", \"binds\"], index=df_in.index)\n    \n    ret_df = pd.concat(\n        [\n            df_in.loc[:, \"id\"], \n            df_pred.loc[:, \"binds\"]\n        ], axis=1\n    )\n    \n    return ret_df\n    ","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.655857Z","iopub.execute_input":"2024-05-04T16:17:11.656330Z","iopub.status.idle":"2024-05-04T16:17:11.670725Z","shell.execute_reply.started":"2024-05-04T16:17:11.656275Z","shell.execute_reply":"2024-05-04T16:17:11.669561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- to make cross product set for iterator","metadata":{}},{"cell_type":"code","source":"def get_cross_product_of_series(val_series_1, val_series_2, series_name_1 = \"series_1\", series_name_2 = \"series_2\"): \n\n    tmp_sets = [[s, t] for s in val_series_1 for t in val_series_2]\n    ret_dict = {i1:{series_name_1:val_set[0], series_name_2:val_set[1]} for i1, val_set in enumerate(tmp_sets)}\n    \n    return ret_dict","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.672232Z","iopub.execute_input":"2024-05-04T16:17:11.672690Z","iopub.status.idle":"2024-05-04T16:17:11.682368Z","shell.execute_reply.started":"2024-05-04T16:17:11.672650Z","shell.execute_reply":"2024-05-04T16:17:11.681288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_cross_product_parameter_set(prm_dict): \n    key_list = []\n    tmp_list = [[]]\n    for k, val_vctr in prm_dict.items(): \n        tmp_list = [s+[t] for s in tmp_list for t in val_vctr]\n        key_list += [k]\n\n    ret_dict = {}\n    for i1, val_vctr in enumerate(tmp_list): \n        ret_dict[i1] = {k:v for k, v in zip(key_list, val_vctr)}\n        \n    return ret_dict\n","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.683679Z","iopub.execute_input":"2024-05-04T16:17:11.684800Z","iopub.status.idle":"2024-05-04T16:17:11.694787Z","shell.execute_reply.started":"2024-05-04T16:17:11.684758Z","shell.execute_reply":"2024-05-04T16:17:11.693401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def my_ap(y_true, y_score): \n    ap = average_precision_score(y_true, y_score, average=\"macro\", pos_label=1, sample_weight=None)\n    return \"ap\", ap, True\n    ","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.696431Z","iopub.execute_input":"2024-05-04T16:17:11.696893Z","iopub.status.idle":"2024-05-04T16:17:11.706089Z","shell.execute_reply.started":"2024-05-04T16:17:11.696862Z","shell.execute_reply":"2024-05-04T16:17:11.704914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plotScatterChart(\n    dict_data:'dictionary', \n    wholeFigSize_width=12, \n    wholeFigSize_height=8, \n    subplotNumber_width=1, \n    subplotNumber_height=1, \n    subplots_adjust={'bottom':0.3, 'top':0.85, 'left':0.1, 'right':0.85, 'wspace':0.5, 'hspace':0.5}, \n    dirPath_outPng='', \n    fileName_outPng='', \n    titleStr_suptitle='', \n    color_set = [\"#000000\", \"#999999\", \n                 \"#AA0000\", \"#FF4500\", \n                 \"#008800\", \"#98FB98\", \n                 \"#000080\", \"#ADD8E6\"], \n    isShowLegend=True\n) : \n    plt.rcParams[\"font.size\"] = 12\n    fig = plt.figure(figsize=(wholeFigSize_width, wholeFigSize_height))  # 全体枠の大きさ (w, h) inch\n\n    ## i1 = 0\n    ## dict_i1 = dict_data[i1]\n    for i1, dict_i1 in dict_data.items() : \n\n        ## df = dict_i1['data']\n\n        subplotPosition = i1+1\n\n        ax = fig.add_subplot(subplotNumber_height, subplotNumber_width, subplotPosition) # サブプロットの位置 (vertical, horizontal, number)\n        fig.subplots_adjust(\n            bottom=subplots_adjust['bottom'], \n            top=subplots_adjust['top'], \n            left=subplots_adjust['left'], \n            right=subplots_adjust['right'], \n            wspace=subplots_adjust['wspace'], \n            hspace=subplots_adjust['hspace']\n        )\n\n        ## j1=0\n        ## v = dict_i1['data'][j1]\n        for j1, v in dict_i1['data'].items() : \n            rects = ax.scatter(\n                v[\"val\"][\"x\"], \n                v[\"val\"][\"y\"], \n                color=color_set[j1 % len(color_set)], \n                s=10\n            )\n        ## end ; for j1, v in dict_i1['data'].items() : \n        del j1, v\n\n        if isShowLegend : \n            ax.legend(\n                [vv['name'] for kk,vv in dict_i1['data'].items()], \n                bbox_to_anchor=(1, 1), \n                loc='upper left', \n                borderaxespad=0\n            )\n\n        ax.set_xlim(dict_i1['axisScale']['x'])\n        ax.set_ylim(dict_i1['axisScale']['y'])\n\n        ax.set_title(dict_i1['title'])\n\n        ax.set_xlabel('{}'.format(dict_i1['axisName_01']))\n        ax.set_ylabel('{}'.format(dict_i1['axisName_02']))\n\n        ax.grid(which='major', alpha=0.5, linestyle='-')\n\n    ## end ; for i1, dict_i1 in dict_data.items() : \n\n    plt.suptitle(titleStr_suptitle)\n\n    if ((dirPath_outPng!='')&(fileName_outPng!='')) : \n        plt.savefig(dirPath_outPng+'/'+fileName_outPng, bbox_inches=\"tight\") \n    else : \n        plt.show()\n    ## end ; if ((dirPath_outPng!='')&(fileName_outPng!='')) : \n\n    plt.close()\n\n    return 0\n## end ; def plotScatterChart(\n","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.708137Z","iopub.execute_input":"2024-05-04T16:17:11.708623Z","iopub.status.idle":"2024-05-04T16:17:11.835385Z","shell.execute_reply.started":"2024-05-04T16:17:11.708561Z","shell.execute_reply":"2024-05-04T16:17:11.834113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1.4. set input and output parameteres","metadata":{}},{"cell_type":"code","source":"prms_in_ = {\n    \"train\": {\n        \"file\": {\n            \"path\": \"/kaggle/input/leash-BELKA\", \n            \"name\": \"train.parquet\"\n        }\n    }, \n    \"test\": {\n        \"file\": {\n            \"path\": \"/kaggle/input/leash-BELKA\", \n            \"name\": \"test.parquet\"\n        }\n    }\n}\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.836928Z","iopub.execute_input":"2024-05-04T16:17:11.837398Z","iopub.status.idle":"2024-05-04T16:17:11.851269Z","shell.execute_reply.started":"2024-05-04T16:17:11.837360Z","shell.execute_reply":"2024-05-04T16:17:11.849879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prms_out_ = {\n    \"submission\": {\n        \"file\": {\n            \"path\": \"/kaggle/working/leash-BELKA/005_LGB_with_ECFP_3\", \n            \"name\": \"submission.csv\"\n        }\n    }\n}","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.852573Z","iopub.execute_input":"2024-05-04T16:17:11.852956Z","iopub.status.idle":"2024-05-04T16:17:11.862044Z","shell.execute_reply.started":"2024-05-04T16:17:11.852921Z","shell.execute_reply":"2024-05-04T16:17:11.860792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. read files\n\n## 2.1. with dask","metadata":{}},{"cell_type":"code","source":"dskdf_ = {}\nfor info_lgcl_nm in [\"train\", \"test\"]: \n    dskdf_[info_lgcl_nm] = dd.read_parquet(\n        \"/\".join([v for k, v in prms_in_[info_lgcl_nm][\"file\"].items()])\n    )","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.863496Z","iopub.execute_input":"2024-05-04T16:17:11.863859Z","iopub.status.idle":"2024-05-04T16:17:11.973786Z","shell.execute_reply.started":"2024-05-04T16:17:11.863818Z","shell.execute_reply":"2024-05-04T16:17:11.972696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.1. with DuckDB","metadata":{}},{"cell_type":"markdown","source":"### 2.1.1. set query","metadata":{}},{"cell_type":"code","source":"query_str_ = {}","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.975623Z","iopub.execute_input":"2024-05-04T16:17:11.976068Z","iopub.status.idle":"2024-05-04T16:17:11.981679Z","shell.execute_reply.started":"2024-05-04T16:17:11.976026Z","shell.execute_reply":"2024-05-04T16:17:11.980383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"query_str_base = \"\"\"\n(\n    SELECT * FROM parquet_scan('{train_path}') \n    WHERE binds = {binds}  AND protein_name = '{protein_name}'\n    ORDER BY random() \n    LIMIT {limit_row}\n)\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.983487Z","iopub.execute_input":"2024-05-04T16:17:11.983969Z","iopub.status.idle":"2024-05-04T16:17:11.992158Z","shell.execute_reply.started":"2024-05-04T16:17:11.983930Z","shell.execute_reply":"2024-05-04T16:17:11.990901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# iterator_val_set_ = get_cross_product_of_series([0, 1], [\"HSA\", \"BRD4\", \"sEH\"], \"binds\", \"protein_name\")\n# display(iterator_val_set_)\n\niterator_val_set_ = get_cross_product_parameter_set(\n    {\n        \"binds\": [0, 1], \n        \"protein_name\": [\"HSA\", \"BRD4\", \"sEH\"]\n    }\n)\ndisplay(iterator_val_set_)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:11.993662Z","iopub.execute_input":"2024-05-04T16:17:11.993987Z","iopub.status.idle":"2024-05-04T16:17:12.008412Z","shell.execute_reply.started":"2024-05-04T16:17:11.993954Z","shell.execute_reply":"2024-05-04T16:17:12.007015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"query_str_[\"train\"] = \"\"\nfor i1, val_set in iterator_val_set_.items(): \n    \n    if i1>0: \n        query_str_[\"train\"] += \"UNION ALL \"\n    \n    query_str_[\"train\"] += query_str_base.format(\n        train_path=\"/\".join([v for k, v in prms_in_[\"train\"][\"file\"].items()]), \n        binds=val_set[\"binds\"], \n        protein_name=val_set[\"protein_name\"], \n        limit_row=10000\n    )","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:12.010116Z","iopub.execute_input":"2024-05-04T16:17:12.010472Z","iopub.status.idle":"2024-05-04T16:17:12.018882Z","shell.execute_reply.started":"2024-05-04T16:17:12.010439Z","shell.execute_reply":"2024-05-04T16:17:12.017786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"query: {query_str}\".format(query_str=query_str_[\"train\"]))","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2024-05-04T16:17:12.020278Z","iopub.execute_input":"2024-05-04T16:17:12.020616Z","iopub.status.idle":"2024-05-04T16:17:12.034126Z","shell.execute_reply.started":"2024-05-04T16:17:12.020589Z","shell.execute_reply":"2024-05-04T16:17:12.033263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"query_str_[\"test\"] = \"\"\"\nSELECT * FROM parquet_scan('{test_path}')\n\"\"\".format(\n    test_path=\"/\".join([v for k, v in prms_in_[\"test\"][\"file\"].items()]), \n)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:12.035430Z","iopub.execute_input":"2024-05-04T16:17:12.035738Z","iopub.status.idle":"2024-05-04T16:17:12.045474Z","shell.execute_reply.started":"2024-05-04T16:17:12.035713Z","shell.execute_reply":"2024-05-04T16:17:12.044280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2.1.2. execute query (read)","metadata":{}},{"cell_type":"code","source":"df_ = {}","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:12.053911Z","iopub.execute_input":"2024-05-04T16:17:12.054288Z","iopub.status.idle":"2024-05-04T16:17:12.059137Z","shell.execute_reply.started":"2024-05-04T16:17:12.054258Z","shell.execute_reply":"2024-05-04T16:17:12.057951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t0_0 = time.time()\n\ncon = duckdb.connect()\ndf_[\"train\"] = con.query(query_str_[\"train\"]).df()\ncon.close()\n\nt0_1 = time.time()\nprint(\"train, elapsed time: {:0.03f} sec\".format(t0_1-t0_0))","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:17:12.060639Z","iopub.execute_input":"2024-05-04T16:17:12.060962Z","iopub.status.idle":"2024-05-04T16:19:00.237510Z","shell.execute_reply.started":"2024-05-04T16:17:12.060935Z","shell.execute_reply":"2024-05-04T16:19:00.236445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t0_0 = time.time()\n\ncon = duckdb.connect()\ndf_[\"test\"] = con.query(query_str_[\"test\"]).df()\ncon.close()\n\nt0_1 = time.time()\nprint(\"test, elapsed time: {:0.03f} sec\".format(t0_1-t0_0))","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:19:00.238733Z","iopub.execute_input":"2024-05-04T16:19:00.239669Z","iopub.status.idle":"2024-05-04T16:19:02.649265Z","shell.execute_reply.started":"2024-05-04T16:19:00.239633Z","shell.execute_reply":"2024-05-04T16:19:02.648389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. EDA\n\n## 3.1. PCA","metadata":{}},{"cell_type":"markdown","source":"### 3.1.0. overview\n\n- 全 protein_name を統合したデータ\n    - buildingblock + molecule_smiles 別に PCA → binds別に色を分けて の散布図\n    - buildingblock + molecule_smiles 別に PCA → protein_name 別に色を分けて の散布図\n    - 全 buildingblock を結合して PCA → binds別に色を分けて の散布図\n    - 全 buildingblock を結合して PCA → protein_name 別に色を分けて の散布図    \n\n\n- protein_name 別データ\n    - 全 buildingblock を結合して PCA → binds別に色を分けて の散布図\n","metadata":{}},{"cell_type":"code","source":"df_keys = df_[\"train\"].loc[:, [\"id\", \"protein_name\", \"binds\"]]\n\necfp_ = {}\ndf_ecfp_ = {}\nclnms_SMILES = [\"buildingblock1_smiles\", \"buildingblock2_smiles\", \"buildingblock3_smiles\", \"molecule_smiles\"]\nfor clnm in clnms_SMILES: \n    ecfp_[clnm]= from_SMILE_series_to_ecfp(df_[\"train\"].loc[:, clnm])\n","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:19:02.650534Z","iopub.execute_input":"2024-05-04T16:19:02.651656Z","iopub.status.idle":"2024-05-04T16:19:40.153859Z","shell.execute_reply.started":"2024-05-04T16:19:02.651622Z","shell.execute_reply":"2024-05-04T16:19:40.152668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prefix_synonym = {\n    \"buildingblock1_smiles\": \"bb1\", \n    \"buildingblock2_smiles\": \"bb2\", \n    \"buildingblock3_smiles\": \"bb3\", \n    \"molecule_smiles\": \"mlcl\"\n}\nfor clnm in clnms_SMILES: \n    df_ecfp_[clnm] = pd.DataFrame(ecfp_[clnm].tolist())\n    df_ecfp_[clnm] = df_ecfp_[clnm].rename(columns=lambda s:\"{}_{}\".format(prefix_synonym[clnm], s))","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:19:40.306835Z","iopub.execute_input":"2024-05-04T16:19:40.307316Z","iopub.status.idle":"2024-05-04T16:19:54.866731Z","shell.execute_reply.started":"2024-05-04T16:19:40.307265Z","shell.execute_reply":"2024-05-04T16:19:54.865559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3.1.1. 全 protein_name を統合したデータ","metadata":{}},{"cell_type":"markdown","source":"### 3.1.1.1. binds 別","metadata":{}},{"cell_type":"code","source":"plot_data_ = {}\ni1=0\nfor clnm in clnms_SMILES: \n    df_tmp = pd.concat([df_ecfp_[clnm], df_keys], axis=1)\n\n    pca = PCA(n_components=3)\n\n    # PCAモデルにデータをフィット\n    pca.fit(df_tmp.drop(df_keys.columns, axis=1))\n\n    # データを低次元に変換\n    X_pca = pca.transform(df_tmp.drop(df_keys.columns, axis=1))\n\n    df_plot = pd.concat(\n        [\n            pd.DataFrame(X_pca), \n            df_tmp.loc[:, df_keys.columns].reset_index(drop=True)\n        ], axis=1\n    )\n\n    scale_min=min(int(1.25*df_plot.loc[:, 0:2].min().min()), 0)\n    scale_max=max(int(1.25*df_plot.loc[:, 0:2].max().max()), 0)\n    for axis_no in [0, 1, 2]: \n        plot_data_[i1]={\n            \"title\": \"{}\".format(clnm), \n            \"data\": {\n                0:{\n                    \"name\":\"data name\", \n                    \"val\":{\n                        \"x\":np.array([0, 1, 2, 3, 4, 5, 6, 7, 9, 9]), \n                        \"y\":np.array([9, 8, 7, 6, 7, 4, 3, 2, 1, 0])\n                    }\n                }\n            }, \n            \"axisName_01\": \"{}\".format((axis_no)%3), \n            \"axisName_02\": \"{}\".format((axis_no+1)%3),\n            \"axisScale\": {\n                \"x\": [scale_min, scale_max], \n                \"y\": [scale_min, scale_max]\n            }\n        }\n\n        i2=0\n        grpby_key_i2 = [\"binds\"]\n        for binds, df_plot_sub in df_plot.groupby(grpby_key_i2): \n            binds, df_plot_sub\n            plot_data_[i1][\"data\"][i2]={\n                \"name\": \"{}: {}\".format(\",\".join(grpby_key_i2), \",\".join([str(s) for s in binds])), \n                \"val\": {\n                    \"x\": df_plot_sub.loc[:, (axis_no)%3].values, \n                    \"y\": df_plot_sub.loc[:, (axis_no+1)%3].values, \n                }\n            }\n            i2+=1\n        ## end ; for binds, df_plot_sub in df_plot.groupby(grpby_key_i2): \n\n        i1+=1\n    ## end ; for axis_no in [0, 1, 2]: ","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:19:54.868218Z","iopub.execute_input":"2024-05-04T16:19:54.868619Z","iopub.status.idle":"2024-05-04T16:19:56.881604Z","shell.execute_reply.started":"2024-05-04T16:19:54.868570Z","shell.execute_reply":"2024-05-04T16:19:56.879643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subplotNumber_width=3\nsubplotNumber_height=4\n\nscale_factor=2\nwholeFigSize_width=6*scale_factor*subplotNumber_width\nwholeFigSize_height=3*scale_factor*subplotNumber_height\n\nsubplots_adjust={'bottom':0.15, 'top':0.85, 'left':0.15, 'right':0.85, 'wspace':0.5, 'hspace':0.25}\n\nplotScatterChart(\n    plot_data_, \n    wholeFigSize_width=wholeFigSize_width, \n    wholeFigSize_height=wholeFigSize_height, \n    subplotNumber_width=subplotNumber_width, \n    subplotNumber_height=subplotNumber_height,\n    subplots_adjust=subplots_adjust, \n    dirPath_outPng=\"\", \n    fileName_outPng=\"\", \n    titleStr_suptitle=\"\", \n    isShowLegend=True\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:19:56.889462Z","iopub.execute_input":"2024-05-04T16:19:56.890132Z","iopub.status.idle":"2024-05-04T16:20:01.021811Z","shell.execute_reply.started":"2024-05-04T16:19:56.890080Z","shell.execute_reply":"2024-05-04T16:20:01.020756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3.1.1.2. protein_name 別","metadata":{}},{"cell_type":"code","source":"plot_data_ = {}\ni1=0\nfor clnm in clnms_SMILES: \n    df_tmp = pd.concat([df_ecfp_[clnm], df_keys], axis=1)\n\n    pca = PCA(n_components=3)\n\n    # PCAモデルにデータをフィット\n    pca.fit(df_tmp.drop(df_keys.columns, axis=1))\n\n    # データを低次元に変換\n    X_pca = pca.transform(df_tmp.drop(df_keys.columns, axis=1))\n\n    df_plot = pd.concat(\n        [\n            pd.DataFrame(X_pca), \n            df_tmp.loc[:, df_keys.columns].reset_index(drop=True)\n        ], axis=1\n    )\n\n    scale_min=min(int(1.25*df_plot.loc[:, 0:2].min().min()), 0)\n    scale_max=max(int(1.25*df_plot.loc[:, 0:2].max().max()), 0)\n    for axis_no in [0, 1, 2]: \n        plot_data_[i1]={\n            \"title\": \"{}\".format(clnm), \n            \"data\": {\n                0:{\n                    \"name\":\"data name\", \n                    \"val\":{\n                        \"x\":np.array([0, 1, 2, 3, 4, 5, 6, 7, 9, 9]), \n                        \"y\":np.array([9, 8, 7, 6, 7, 4, 3, 2, 1, 0])\n                    }\n                }\n            }, \n            \"axisName_01\": \"{}\".format((axis_no)%3), \n            \"axisName_02\": \"{}\".format((axis_no+1)%3),\n            \"axisScale\": {\n                \"x\": [scale_min, scale_max], \n                \"y\": [scale_min, scale_max]\n            }\n        }\n\n        i2=0\n        grpby_key_i2 = [\"protein_name\"]\n        for binds, df_plot_sub in df_plot.groupby(grpby_key_i2): \n            binds, df_plot_sub\n            plot_data_[i1][\"data\"][i2]={\n                \"name\": \"{}: {}\".format(\",\".join(grpby_key_i2), \",\".join([str(s) for s in binds])), \n                \"val\": {\n                    \"x\": df_plot_sub.loc[:, (axis_no)%3].values, \n                    \"y\": df_plot_sub.loc[:, (axis_no+1)%3].values, \n                }\n            }\n            i2+=1\n        ## end ; for binds, df_plot_sub in df_plot.groupby(grpby_key_i2): \n\n        i1+=1\n    ## end ; for axis_no in [0, 1, 2]: ","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:20:01.022901Z","iopub.execute_input":"2024-05-04T16:20:01.023241Z","iopub.status.idle":"2024-05-04T16:20:03.185885Z","shell.execute_reply.started":"2024-05-04T16:20:01.023211Z","shell.execute_reply":"2024-05-04T16:20:03.184272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subplotNumber_width=3\nsubplotNumber_height=4\n\nscale_factor=2\nwholeFigSize_width=6*scale_factor*subplotNumber_width\nwholeFigSize_height=3*scale_factor*subplotNumber_height\n\nsubplots_adjust={'bottom':0.15, 'top':0.85, 'left':0.15, 'right':0.85, 'wspace':0.5, 'hspace':0.25}\n\nplotScatterChart(\n    plot_data_, \n    wholeFigSize_width=wholeFigSize_width, \n    wholeFigSize_height=wholeFigSize_height, \n    subplotNumber_width=subplotNumber_width, \n    subplotNumber_height=subplotNumber_height,\n    subplots_adjust=subplots_adjust, \n    dirPath_outPng=\"\", \n    fileName_outPng=\"\", \n    titleStr_suptitle=\"\", \n    isShowLegend=True\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:20:03.188700Z","iopub.execute_input":"2024-05-04T16:20:03.189895Z","iopub.status.idle":"2024-05-04T16:20:07.335694Z","shell.execute_reply.started":"2024-05-04T16:20:03.189834Z","shell.execute_reply":"2024-05-04T16:20:07.334662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3.1.1.3. all SMILES columns concatenated, by binds","metadata":{}},{"cell_type":"code","source":"df_all_ecfp = pd.concat([ df_ecfp_[s] for s in clnms_SMILES], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:20:07.337331Z","iopub.execute_input":"2024-05-04T16:20:07.337686Z","iopub.status.idle":"2024-05-04T16:20:07.441579Z","shell.execute_reply.started":"2024-05-04T16:20:07.337656Z","shell.execute_reply":"2024-05-04T16:20:07.439927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_data_ = {}\ni1=0\n\ndf_tmp = pd.concat([df_all_ecfp, df_keys], axis=1)\n\npca = PCA(n_components=3)\n\n# PCAモデルにデータをフィット\npca.fit(df_tmp.drop(df_keys.columns, axis=1))\n\n# データを低次元に変換\nX_pca = pca.transform(df_tmp.drop(df_keys.columns, axis=1))\n\ndf_plot = pd.concat(\n    [\n        pd.DataFrame(X_pca), \n        df_tmp.loc[:, df_keys.columns].reset_index(drop=True)\n    ], axis=1\n)\n\nscale_min=min(int(1.25*df_plot.loc[:, 0:2].min().min()), 0)\nscale_max=max(int(1.25*df_plot.loc[:, 0:2].max().max()), 0)\nfor axis_no in [0, 1, 2]: \n    plot_data_[i1]={\n        \"title\": \"\".format(), \n        \"data\": {\n            0:{\n                \"name\":\"data name\", \n                \"val\":{\n                    \"x\":np.array([0, 1, 2, 3, 4, 5, 6, 7, 9, 9]), \n                    \"y\":np.array([9, 8, 7, 6, 7, 4, 3, 2, 1, 0])\n                }\n            }\n        }, \n        \"axisName_01\": \"{}\".format((axis_no)%3), \n        \"axisName_02\": \"{}\".format((axis_no+1)%3),\n        \"axisScale\": {\n            \"x\": [scale_min, scale_max], \n            \"y\": [scale_min, scale_max]\n        }\n    }\n\n    i2=0\n    grpby_key_i2 = [\"binds\"]\n    for binds, df_plot_sub in df_plot.groupby(grpby_key_i2): \n        binds, df_plot_sub\n        plot_data_[i1][\"data\"][i2]={\n            \"name\": \"{}: {}\".format(\",\".join(grpby_key_i2), \",\".join([str(s) for s in binds])), \n            \"val\": {\n                \"x\": df_plot_sub.loc[:, (axis_no)%3].values, \n                \"y\": df_plot_sub.loc[:, (axis_no+1)%3].values, \n            }\n        }\n        i2+=1\n    ## end ; for binds, df_plot_sub in df_plot.groupby(grpby_key_i2): \n\n    i1+=1\n## end ; for axis_no in [0, 1, 2]: ","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:20:07.444350Z","iopub.execute_input":"2024-05-04T16:20:07.445358Z","iopub.status.idle":"2024-05-04T16:20:09.795473Z","shell.execute_reply.started":"2024-05-04T16:20:07.445284Z","shell.execute_reply":"2024-05-04T16:20:09.793636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subplotNumber_width=3\nsubplotNumber_height=4\n\nscale_factor=2\nwholeFigSize_width=6*scale_factor*subplotNumber_width\nwholeFigSize_height=3*scale_factor*subplotNumber_height\n\nsubplots_adjust={'bottom':0.15, 'top':0.85, 'left':0.15, 'right':0.85, 'wspace':0.5, 'hspace':0.25}\n\nplotScatterChart(\n    plot_data_, \n    wholeFigSize_width=wholeFigSize_width, \n    wholeFigSize_height=wholeFigSize_height, \n    subplotNumber_width=subplotNumber_width, \n    subplotNumber_height=subplotNumber_height,\n    subplots_adjust=subplots_adjust, \n    dirPath_outPng=\"\", \n    fileName_outPng=\"\", \n    titleStr_suptitle=\"\", \n    isShowLegend=True\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:20:09.798088Z","iopub.execute_input":"2024-05-04T16:20:09.799078Z","iopub.status.idle":"2024-05-04T16:20:10.986307Z","shell.execute_reply.started":"2024-05-04T16:20:09.799002Z","shell.execute_reply":"2024-05-04T16:20:10.985390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3.1.1.4. all SMILES columns concatenated, by protein_name","metadata":{}},{"cell_type":"code","source":"df_all_ecfp = pd.concat([ df_ecfp_[s] for s in clnms_SMILES], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:20:10.987635Z","iopub.execute_input":"2024-05-04T16:20:10.988266Z","iopub.status.idle":"2024-05-04T16:20:11.064219Z","shell.execute_reply.started":"2024-05-04T16:20:10.988234Z","shell.execute_reply":"2024-05-04T16:20:11.062823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_data_ = {}\ni1=0\n\ndf_tmp = pd.concat([df_all_ecfp, df_keys], axis=1)\n\npca = PCA(n_components=3)\n\n# PCAモデルにデータをフィット\npca.fit(df_tmp.drop(df_keys.columns, axis=1))\n\n# データを低次元に変換\nX_pca = pca.transform(df_tmp.drop(df_keys.columns, axis=1))\n\ndf_plot = pd.concat(\n    [\n        pd.DataFrame(X_pca), \n        df_tmp.loc[:, df_keys.columns].reset_index(drop=True)\n    ], axis=1\n)\n\nscale_min=min(int(1.25*df_plot.loc[:, 0:2].min().min()), 0)\nscale_max=max(int(1.25*df_plot.loc[:, 0:2].max().max()), 0)\nfor axis_no in [0, 1, 2]: \n    plot_data_[i1]={\n        \"title\": \"\".format(), \n        \"data\": {\n            0:{\n                \"name\":\"data name\", \n                \"val\":{\n                    \"x\":np.array([0, 1, 2, 3, 4, 5, 6, 7, 9, 9]), \n                    \"y\":np.array([9, 8, 7, 6, 7, 4, 3, 2, 1, 0])\n                }\n            }\n        }, \n        \"axisName_01\": \"{}\".format((axis_no)%3), \n        \"axisName_02\": \"{}\".format((axis_no+1)%3),\n        \"axisScale\": {\n            \"x\": [scale_min, scale_max], \n            \"y\": [scale_min, scale_max]\n        }\n    }\n\n    i2=0\n    grpby_key_i2 = [\"protein_name\"]\n    for binds, df_plot_sub in df_plot.groupby(grpby_key_i2): \n        binds, df_plot_sub\n        plot_data_[i1][\"data\"][i2]={\n            \"name\": \"{}: {}\".format(\",\".join(grpby_key_i2), \",\".join([str(s) for s in binds])), \n            \"val\": {\n                \"x\": df_plot_sub.loc[:, (axis_no)%3].values, \n                \"y\": df_plot_sub.loc[:, (axis_no+1)%3].values, \n            }\n        }\n        i2+=1\n    ## end ; for binds, df_plot_sub in df_plot.groupby(grpby_key_i2): \n\n    i1+=1\n## end ; for axis_no in [0, 1, 2]: ","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:20:11.066107Z","iopub.execute_input":"2024-05-04T16:20:11.066492Z","iopub.status.idle":"2024-05-04T16:20:12.716275Z","shell.execute_reply.started":"2024-05-04T16:20:11.066461Z","shell.execute_reply":"2024-05-04T16:20:12.714694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subplotNumber_width=3\nsubplotNumber_height=1\n\nscale_factor=2\nwholeFigSize_width=6*scale_factor*subplotNumber_width\nwholeFigSize_height=3*scale_factor*subplotNumber_height\n\nsubplots_adjust={'bottom':0.15, 'top':0.85, 'left':0.15, 'right':0.85, 'wspace':0.5, 'hspace':0.25}\n\nplotScatterChart(\n    plot_data_, \n    wholeFigSize_width=wholeFigSize_width, \n    wholeFigSize_height=wholeFigSize_height, \n    subplotNumber_width=subplotNumber_width, \n    subplotNumber_height=subplotNumber_height,\n    subplots_adjust=subplots_adjust, \n    dirPath_outPng=\"\", \n    fileName_outPng=\"\", \n    titleStr_suptitle=\"\", \n    isShowLegend=True\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:20:12.718993Z","iopub.execute_input":"2024-05-04T16:20:12.720114Z","iopub.status.idle":"2024-05-04T16:20:13.688880Z","shell.execute_reply.started":"2024-05-04T16:20:12.720054Z","shell.execute_reply":"2024-05-04T16:20:13.687747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3.1.2. by protain_name","metadata":{}},{"cell_type":"code","source":"df_tmp = pd.concat([df_all_ecfp, df_keys], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:20:13.690694Z","iopub.execute_input":"2024-05-04T16:20:13.691134Z","iopub.status.idle":"2024-05-04T16:20:13.911972Z","shell.execute_reply.started":"2024-05-04T16:20:13.691091Z","shell.execute_reply":"2024-05-04T16:20:13.910836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_data_ = {}\ni1=0\ngrpby_key_i1 = [\"protein_name\"]\nfor protein_name, df_sub in df_tmp.groupby(grpby_key_i1): \n\n    pca = PCA(n_components=3)\n\n    # PCAモデルにデータをフィット\n    pca.fit(df_sub.drop(df_keys.columns, axis=1))\n\n    # データを低次元に変換\n    X_pca = pca.transform(df_sub.drop(df_keys.columns, axis=1))\n\n    df_plot = pd.concat(\n        [\n            pd.DataFrame(X_pca), \n            df_sub.loc[:, df_keys.columns].reset_index(drop=True)\n        ], axis=1\n    )\n\n    scale_min=min(int(1.25*df_plot.loc[:, 0:2].min().min()), 0)\n    scale_max=max(int(1.25*df_plot.loc[:, 0:2].max().max()), 0)\n    for axis_no in [0, 1, 2]: \n        plot_data_[i1]={\n            \"title\": \"{}: {}\".format(\",\".join(grpby_key_i1), \",\".join(protein_name)), \n            \"data\": {\n                0:{\n                    \"name\":\"data name\", \n                    \"val\":{\n                        \"x\":np.array([0, 1, 2, 3, 4, 5, 6, 7, 9, 9]), \n                        \"y\":np.array([9, 8, 7, 6, 7, 4, 3, 2, 1, 0])\n                    }\n                }\n            }, \n            \"axisName_01\": \"{}\".format((axis_no)%3), \n            \"axisName_02\": \"{}\".format((axis_no+1)%3),\n            \"axisScale\": {\n                \"x\": [scale_min, scale_max], \n                \"y\": [scale_min, scale_max]\n            }\n        }\n\n        i2=0\n        grpby_key_i2 = [\"binds\"]\n        for binds, df_plot_sub in df_plot.groupby(grpby_key_i2): \n            binds, df_plot_sub\n            plot_data_[i1][\"data\"][i2]={\n                \"name\": \"{}: {}\".format(\",\".join(grpby_key_i2), \",\".join([str(s) for s in binds])), \n                \"val\": {\n                    \"x\": df_plot_sub.loc[:, (axis_no)%3].values, \n                    \"y\": df_plot_sub.loc[:, (axis_no+1)%3].values, \n                }\n            }\n            i2+=1\n        ## end ; for binds, df_plot_sub in df_plot.groupby(grpby_key_i2): \n\n        i1+=1\n    ## end ; for axis_no in [0, 1, 2]: \n## end ; for protein_name, df_sub in df_tmp.groupby(grpby_key_i1): ","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:20:13.913985Z","iopub.execute_input":"2024-05-04T16:20:13.914766Z","iopub.status.idle":"2024-05-04T16:20:16.036778Z","shell.execute_reply.started":"2024-05-04T16:20:13.914734Z","shell.execute_reply":"2024-05-04T16:20:16.035206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subplotNumber_width=3\nsubplotNumber_height=3\n\nscale_factor=2\nwholeFigSize_width=4*scale_factor*subplotNumber_width\nwholeFigSize_height=3*scale_factor*subplotNumber_height\n\nsubplots_adjust={'bottom':0.15, 'top':0.85, 'left':0.15, 'right':0.85, 'wspace':0.5, 'hspace':0.25}\n\nplotScatterChart(\n    plot_data_, \n    wholeFigSize_width=wholeFigSize_width, \n    wholeFigSize_height=wholeFigSize_height, \n    subplotNumber_width=subplotNumber_width, \n    subplotNumber_height=subplotNumber_height,\n    subplots_adjust=subplots_adjust, \n    dirPath_outPng=\"\", \n    fileName_outPng=\"\", \n    titleStr_suptitle=\"\", \n    isShowLegend=True\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:20:16.039577Z","iopub.execute_input":"2024-05-04T16:20:16.040721Z","iopub.status.idle":"2024-05-04T16:20:18.580011Z","shell.execute_reply.started":"2024-05-04T16:20:16.040659Z","shell.execute_reply":"2024-05-04T16:20:18.578791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.1. cluster","metadata":{}},{"cell_type":"code","source":"#k-means++のモデル構築\nkmeans = KMeans(n_clusters=3, init='k-means++',random_state=0)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:20:18.582038Z","iopub.execute_input":"2024-05-04T16:20:18.582431Z","iopub.status.idle":"2024-05-04T16:20:18.586971Z","shell.execute_reply.started":"2024-05-04T16:20:18.582400Z","shell.execute_reply":"2024-05-04T16:20:18.586076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_data_ = {}\ni1=0\ngrpby_key_i1 = [\"protein_name\"]\nfor protein_name, df_sub in df_tmp.groupby(grpby_key_i1): \n\n    #k-means++のモデル実行（クラスタリング）\n    clusters = kmeans.fit(df_sub.drop(df_keys.columns, axis=1))\n    df_sub.loc[:, \"cluster\"] = clusters.labels_\n    \n    grpby_key_i3 = [\"cluster\"]\n    for cluster, df_sub2 in df_sub.groupby(grpby_key_i3): \n\n        pca = PCA(n_components=3)\n\n        # PCAモデルにデータをフィット\n        pca.fit(df_sub2.drop(df_keys.columns, axis=1))\n\n        # データを低次元に変換\n        X_pca = pca.transform(df_sub2.drop(df_keys.columns, axis=1))\n\n        df_plot = pd.concat(\n            [\n                pd.DataFrame(X_pca), \n                df_sub2.loc[:, df_keys.columns].reset_index(drop=True)\n            ], axis=1\n        )\n\n        scale_min=min(int(1.25*df_plot.loc[:, 0:2].min().min()), 0)\n        scale_max=max(int(1.25*df_plot.loc[:, 0:2].max().max()), 0)\n        for axis_no in [0, 1, 2]: \n            plot_data_[i1]={\n                \"title\": \"{}: {}, {}:{}\".format(\",\".join(grpby_key_i1), \",\".join(protein_name), \n                                                \",\".join(grpby_key_i3), \",\".join([str(s) for s in cluster])), \n                \"data\": {\n                    0:{\n                        \"name\":\"data name\", \n                        \"val\":{\n                            \"x\":np.array([0, 1, 2, 3, 4, 5, 6, 7, 9, 9]), \n                            \"y\":np.array([9, 8, 7, 6, 7, 4, 3, 2, 1, 0])\n                        }\n                    }\n                }, \n                \"axisName_01\": \"{}\".format((axis_no)%3), \n                \"axisName_02\": \"{}\".format((axis_no+1)%3),\n                \"axisScale\": {\n                    \"x\": [scale_min, scale_max], \n                    \"y\": [scale_min, scale_max]\n                }\n            }\n\n            i2=0\n            grpby_key_i2 = [\"binds\"]\n            for binds, df_plot_sub in df_plot.groupby(grpby_key_i2): \n                binds, df_plot_sub\n                plot_data_[i1][\"data\"][i2]={\n                    \"name\": \"{}: {}\".format(\",\".join(grpby_key_i2), \",\".join([str(s) for s in binds])), \n                    \"val\": {\n                        \"x\": df_plot_sub.loc[:, (axis_no)%3].values, \n                        \"y\": df_plot_sub.loc[:, (axis_no+1)%3].values, \n                    }\n                }\n                i2+=1\n            ## end ; for binds, df_plot_sub in df_plot.groupby(grpby_key_i2): \n\n            i1+=1\n        ## end ; for axis_no in [0, 1, 2]: \n    ## end ; for cluster, df_sub2 in df_tmp.groupby(grpby_key_i3): \n## end ; for protein_name, df_sub in df_tmp.groupby(grpby_key_i1):     \n","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:20:18.588444Z","iopub.execute_input":"2024-05-04T16:20:18.589062Z","iopub.status.idle":"2024-05-04T16:20:41.458512Z","shell.execute_reply.started":"2024-05-04T16:20:18.589030Z","shell.execute_reply":"2024-05-04T16:20:41.457062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subplotNumber_width=3\nsubplotNumber_height=9\n\nscale_factor=2\nwholeFigSize_width=4*scale_factor*subplotNumber_width\nwholeFigSize_height=3*scale_factor*subplotNumber_height\n\nsubplots_adjust={'bottom':0.15, 'top':0.85, 'left':0.15, 'right':0.85, 'wspace':0.5, 'hspace':0.25}\n\nplotScatterChart(\n    plot_data_, \n    wholeFigSize_width=wholeFigSize_width, \n    wholeFigSize_height=wholeFigSize_height, \n    subplotNumber_width=subplotNumber_width, \n    subplotNumber_height=subplotNumber_height,\n    subplots_adjust=subplots_adjust, \n    dirPath_outPng=\"\", \n    fileName_outPng=\"\", \n    titleStr_suptitle=\"\", \n    isShowLegend=True\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T16:20:41.460148Z","iopub.execute_input":"2024-05-04T16:20:41.462971Z","iopub.status.idle":"2024-05-04T16:20:49.211867Z","shell.execute_reply.started":"2024-05-04T16:20:41.462925Z","shell.execute_reply":"2024-05-04T16:20:49.210651Z"},"trusted":true},"execution_count":null,"outputs":[]}]}