{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:39:17.748401Z","iopub.execute_input":"2025-03-12T06:39:17.748780Z","iopub.status.idle":"2025-03-12T06:39:19.172332Z","shell.execute_reply.started":"2025-03-12T06:39:17.748741Z","shell.execute_reply":"2025-03-12T06:39:19.170973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install rdkit","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:39:19.173417Z","iopub.execute_input":"2025-03-12T06:39:19.173995Z","iopub.status.idle":"2025-03-12T06:39:28.150867Z","shell.execute_reply.started":"2025-03-12T06:39:19.173963Z","shell.execute_reply":"2025-03-12T06:39:28.149477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import duckdb\nimport pandas as pd\nfrom tqdm import tqdm\nimport numpy as np # linear algebra\nfrom rdkit import Chem\nfrom rdkit.Chem import AllChem\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:39:28.152042Z","iopub.execute_input":"2025-03-12T06:39:28.152337Z","iopub.status.idle":"2025-03-12T06:39:28.925249Z","shell.execute_reply.started":"2025-03-12T06:39:28.152310Z","shell.execute_reply":"2025-03-12T06:39:28.923975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 設定檔案路徑\ntrain_path = '/kaggle/input/leash-BELKA/train.parquet'\n\n# 建立 DuckDB 連線\ncon = duckdb.connect()\n\n# 使用進度條來顯示進度\nwith tqdm(total=2, desc=\"Processing Data\") as pbar:\n    # 查詢第一部分數據\n    df_part1 = con.query(f\"\"\"SELECT *\n                              FROM parquet_scan('{train_path}')\n                              WHERE binds = 0\n                              ORDER BY random()\n                              LIMIT 150000\"\"\").df()\n    pbar.update(1)  # 更新進度條\n\n    # 查詢第二部分數據\n    df_part2 = con.query(f\"\"\"SELECT *\n                              FROM parquet_scan('{train_path}')\n                              WHERE binds = 1\n                              ORDER BY random()\n                              LIMIT 50000\"\"\").df()\n    pbar.update(1)  # 更新進度條\n\n# 合併兩部分數據\ndf = pd.concat([df_part1, df_part2], ignore_index=True)\n\n# 隨機洗牌數據（frac=1 表示保持原始大小，shuffle 整個 DataFrame）\ndf = df.sample(frac=1, random_state=42).reset_index(drop=True)\n\n# 關閉連線\ncon.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:39:28.926486Z","iopub.execute_input":"2025-03-12T06:39:28.926783Z","iopub.status.idle":"2025-03-12T06:40:29.053587Z","shell.execute_reply.started":"2025-03-12T06:39:28.926758Z","shell.execute_reply":"2025-03-12T06:40:29.052398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # 確認數據\nprint(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:40:29.054863Z","iopub.execute_input":"2025-03-12T06:40:29.055198Z","iopub.status.idle":"2025-03-12T06:40:29.073965Z","shell.execute_reply.started":"2025-03-12T06:40:29.055171Z","shell.execute_reply":"2025-03-12T06:40:29.072505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def smiles_to_topological_torsion_fingerprint(smiles, n_bits=2048):\n    \"\"\"\n    將 SMILES 分子結構轉換為拓樸扭轉指紋。\n    :param smiles: 分子的 SMILES 表示法\n    :param n_bits: 指紋的位元數 (默認為 2048)\n    :return: 一個 numpy 數組，表示拓樸扭轉指紋\n    \"\"\"\n    mol = Chem.MolFromSmiles(smiles)\n    if mol is None:\n        return np.zeros(n_bits, dtype=int)\n    else:\n        generator = AllChem.GetTopologicalTorsionGenerator(fpSize=n_bits)\n        return np.array(generator.GetFingerprint(mol), dtype=int)\n\n# 使用進度條轉換 \"molecule_smiles\" 欄位為拓樸扭轉指紋\ntqdm.pandas(desc=\"Transforming molecule_smiles to Topological Torsion Fingerprint\")\ndf[\"molecule_smiles\"] = df[\"molecule_smiles\"].progress_apply(lambda x: smiles_to_topological_torsion_fingerprint(x))\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:40:29.077525Z","iopub.execute_input":"2025-03-12T06:40:29.077857Z","iopub.status.idle":"2025-03-12T06:46:19.762275Z","shell.execute_reply.started":"2025-03-12T06:40:29.077830Z","shell.execute_reply":"2025-03-12T06:46:19.761081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 對 protein 欄位進行 One-Hot Encoding\nprotein_one_hot = pd.get_dummies(df[\"protein_name\"], prefix=\"protein\").astype(int)\n\n# 合併 One-Hot 結果\ndf_one_hot = pd.concat([df, protein_one_hot], axis=1)\n\n# 合併需要的欄位：molecule_smiles, binds, 和經過 One-Hot Encoding 的 protein\ndf_one_hot = pd.concat([df_one_hot[[\"id\", \"molecule_smiles\", \"binds\"]], protein_one_hot], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:46:19.763822Z","iopub.execute_input":"2025-03-12T06:46:19.764214Z","iopub.status.idle":"2025-03-12T06:46:20.023149Z","shell.execute_reply.started":"2025-03-12T06:46:19.764186Z","shell.execute_reply":"2025-03-12T06:46:20.021542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_one_hot","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:46:20.024405Z","iopub.execute_input":"2025-03-12T06:46:20.024742Z","iopub.status.idle":"2025-03-12T06:46:20.061579Z","shell.execute_reply.started":"2025-03-12T06:46:20.024712Z","shell.execute_reply":"2025-03-12T06:46:20.059851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # 可選：移除原始 protein 欄位\n# df1_one_hot.drop(\"protein_name\", axis=1, inplace=True)\n\n# # 檢視處理後數據\n# print(df_one_hot.head())\n\n# 僅保留需要的欄位\ncolumns_to_keep = [\"id\", \"molecule_smiles\", \"binds\"] + protein_one_hot.columns.tolist()\ndf_filtered = df_one_hot[columns_to_keep]\n\n# 儲存處理後的數據\noutput_filename = \"train_transformed_topological(150k,50k).parquet\"\ndf_filtered.to_parquet(output_filename, index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:46:20.063193Z","iopub.execute_input":"2025-03-12T06:46:20.063595Z","iopub.status.idle":"2025-03-12T06:46:48.541257Z","shell.execute_reply.started":"2025-03-12T06:46:20.063563Z","shell.execute_reply":"2025-03-12T06:46:48.538130Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"topolfile = '/kaggle/working/train_transformed_topological(150k,50k).parquet'\ntopol = pd.read_parquet(topolfile)\ntopol","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:46:48.544370Z","iopub.execute_input":"2025-03-12T06:46:48.544952Z","iopub.status.idle":"2025-03-12T06:47:15.420389Z","shell.execute_reply.started":"2025-03-12T06:46:48.544882Z","shell.execute_reply":"2025-03-12T06:47:15.419196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 設定檔案路徑\ntrain_path = '/kaggle/input/leash-BELKA/train.parquet'\n\n# 建立 DuckDB 連線\ncon = duckdb.connect()\n\n# 使用進度條來顯示進度\nwith tqdm(total=2, desc=\"Processing Data\") as pbar:\n    # 查詢第一部分數據\n    df_part1 = con.query(f\"\"\"SELECT *\n                              FROM parquet_scan('{train_path}')\n                              WHERE binds = 0\n                              ORDER BY random()\n                              LIMIT 100000\"\"\").df()\n    pbar.update(1)  # 更新進度條\n\n    # 查詢第二部分數據\n    df_part2 = con.query(f\"\"\"SELECT *\n                              FROM parquet_scan('{train_path}')\n                              WHERE binds = 1\n                              ORDER BY random()\n                              LIMIT 100000\"\"\").df()\n    pbar.update(1)  # 更新進度條\n\n# 合併兩部分數據\ndf = pd.concat([df_part1, df_part2], ignore_index=True)\n\n# 隨機洗牌數據（frac=1 表示保持原始大小，shuffle 整個 DataFrame）\ndf = df.sample(frac=1, random_state=42).reset_index(drop=True)\n\n# 關閉連線\ncon.close()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:47:15.421617Z","iopub.execute_input":"2025-03-12T06:47:15.421970Z","iopub.status.idle":"2025-03-12T06:48:08.052843Z","shell.execute_reply.started":"2025-03-12T06:47:15.421928Z","shell.execute_reply":"2025-03-12T06:48:08.051796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # 確認數據\nprint(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:48:08.054191Z","iopub.execute_input":"2025-03-12T06:48:08.054535Z","iopub.status.idle":"2025-03-12T06:48:08.063328Z","shell.execute_reply.started":"2025-03-12T06:48:08.054508Z","shell.execute_reply":"2025-03-12T06:48:08.062064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def smiles_to_topological_torsion_fingerprint(smiles, n_bits=2048):\n    \"\"\"\n    將 SMILES 分子結構轉換為拓樸扭轉指紋。\n    :param smiles: 分子的 SMILES 表示法\n    :param n_bits: 指紋的位元數 (默認為 2048)\n    :return: 一個 numpy 數組，表示拓樸扭轉指紋\n    \"\"\"\n    mol = Chem.MolFromSmiles(smiles)\n    if mol is None:\n        return np.zeros(n_bits, dtype=int)\n    else:\n        generator = AllChem.GetTopologicalTorsionGenerator(fpSize=n_bits)\n        return np.array(generator.GetFingerprint(mol), dtype=int)\n\n# 使用進度條轉換 \"molecule_smiles\" 欄位為拓樸扭轉指紋\ntqdm.pandas(desc=\"Transforming molecule_smiles to Topological Torsion Fingerprint\")\ndf[\"molecule_smiles\"] = df[\"molecule_smiles\"].progress_apply(lambda x: smiles_to_topological_torsion_fingerprint(x))\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:48:08.064691Z","iopub.execute_input":"2025-03-12T06:48:08.065170Z","iopub.status.idle":"2025-03-12T06:54:04.016425Z","shell.execute_reply.started":"2025-03-12T06:48:08.065137Z","shell.execute_reply":"2025-03-12T06:54:04.015156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 對 protein 欄位進行 One-Hot Encoding\nprotein_one_hot = pd.get_dummies(df[\"protein_name\"], prefix=\"protein\").astype(int)\n\n# 合併 One-Hot 結果\ndf_one_hot = pd.concat([df, protein_one_hot], axis=1)\n\n# 合併需要的欄位：molecule_smiles, binds, 和經過 One-Hot Encoding 的 protein\ndf_one_hot = pd.concat([df_one_hot[[\"id\", \"molecule_smiles\", \"binds\"]], protein_one_hot], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:54:04.017764Z","iopub.execute_input":"2025-03-12T06:54:04.018112Z","iopub.status.idle":"2025-03-12T06:54:04.300864Z","shell.execute_reply.started":"2025-03-12T06:54:04.018086Z","shell.execute_reply":"2025-03-12T06:54:04.299389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_one_hot","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:54:04.302209Z","iopub.execute_input":"2025-03-12T06:54:04.302633Z","iopub.status.idle":"2025-03-12T06:54:04.326002Z","shell.execute_reply.started":"2025-03-12T06:54:04.302586Z","shell.execute_reply":"2025-03-12T06:54:04.324689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # 可選：移除原始 protein 欄位\n# df1_one_hot.drop(\"protein_name\", axis=1, inplace=True)\n\n# # 檢視處理後數據\n# print(df_one_hot.head())\n\n# 僅保留需要的欄位\ncolumns_to_keep = [\"id\", \"molecule_smiles\", \"binds\"] + protein_one_hot.columns.tolist()\ndf_filtered = df_one_hot[columns_to_keep]\n\n# 儲存處理後的數據\noutput_filename = \"train_transformed_topological(100k,100k).parquet\"\ndf_filtered.to_parquet(output_filename, index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:54:04.327339Z","iopub.execute_input":"2025-03-12T06:54:04.327701Z","iopub.status.idle":"2025-03-12T06:54:25.850579Z","shell.execute_reply.started":"2025-03-12T06:54:04.327657Z","shell.execute_reply":"2025-03-12T06:54:25.849253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"topolfile = '/kaggle/working/train_transformed_topological(100k,100k).parquet'\ntopol = pd.read_parquet(topolfile)\ntopol","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:54:25.852114Z","iopub.execute_input":"2025-03-12T06:54:25.852611Z","iopub.status.idle":"2025-03-12T06:54:45.106000Z","shell.execute_reply.started":"2025-03-12T06:54:25.852500Z","shell.execute_reply":"2025-03-12T06:54:45.104724Z"}},"outputs":[],"execution_count":null}]}