{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":304117,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":259569,"modelId":280750}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-27T09:34:30.743645Z","iopub.execute_input":"2025-03-27T09:34:30.744036Z","iopub.status.idle":"2025-03-27T09:34:30.755534Z","shell.execute_reply.started":"2025-03-27T09:34:30.743983Z","shell.execute_reply":"2025-03-27T09:34:30.754080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n!pip install pandas\n!pip install tqdm\n!pip install pickle\n!pip install rdkit\n!pip install numpy\n     ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T09:34:32.536378Z","iopub.execute_input":"2025-03-27T09:34:32.536753Z","iopub.status.idle":"2025-03-27T09:34:56.661796Z","shell.execute_reply.started":"2025-03-27T09:34:32.536723Z","shell.execute_reply":"2025-03-27T09:34:56.660315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 分布統計\nimport pandas as pd\nfrom tqdm import tqdm\nimport pickle\n\n# 參數設定\nfilename = \"/kaggle/input/leash-BELKA/train.csv\"\nchunksize = 1_000_000\nbb_columns = [\"buildingblock1_smiles\", \"buildingblock2_smiles\", \"buildingblock3_smiles\"]\n\n# 計算分層分布\ndef compute_strata_counts(bb_col):\n    print(f\"🔍 統計 {bb_col} 分層分布...\")\n    strata_counts = {\"pos\": {}, \"neg\": {}}\n    for chunk in tqdm(pd.read_csv(filename, chunksize=chunksize, usecols=[bb_col, \"binds\"]), desc=f\"Counting {bb_col}\"):\n        for bb, group in chunk.groupby(bb_col):\n            pos_count = len(group[group[\"binds\"] == 1])\n            neg_count = len(group[group[\"binds\"] == 0])\n            strata_counts[\"pos\"][bb] = strata_counts[\"pos\"].get(bb, 0) + pos_count\n            strata_counts[\"neg\"][bb] = strata_counts[\"neg\"].get(bb, 0) + neg_count\n    total_pos = sum(strata_counts[\"pos\"].values())\n    total_neg = sum(strata_counts[\"neg\"].values())\n    return {\"strata_counts\": strata_counts, \"total_pos\": total_pos, \"total_neg\": total_neg}\n\n# 執行並儲存\nstrata_data = {}\nfor bb_col in bb_columns:\n    strata_data[bb_col] = compute_strata_counts(bb_col)\n    print(f\"{bb_col} - 總正類: {strata_data[bb_col]['total_pos']}, 總負類: {strata_data[bb_col]['total_neg']}\")\n\n# 儲存預處理資料\nwith open(\"strata_data.pkl\", \"wb\") as f:\n    pickle.dump(strata_data, f)\nprint(\"✅ 分布統計已儲存至 'strata_data.pkl'\")\n     \n\n# 抽樣\nimport pandas as pd\nfrom tqdm import tqdm\nimport pickle\n\n# 載入預處理資料\nwith open(\"strata_data.pkl\", \"rb\") as f:\n    strata_data = pickle.load(f)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T03:44:47.567936Z","iopub.execute_input":"2025-03-27T03:44:47.568221Z","iopub.status.idle":"2025-03-27T04:39:53.104104Z","shell.execute_reply.started":"2025-03-27T03:44:47.568199Z","shell.execute_reply":"2025-03-27T04:39:53.101890Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 參數設定 可調整特徵抽樣比例與正負類比例\nfilename = \"/kaggle/input/leash-BELKA/train.csv\"\nchunksize = 1_000_000\ntargets = [\n    {\"bb\": \"buildingblock1_smiles\", \"pos\": 80000, \"neg\": 240000}, # 特徵一\n    {\"bb\": \"buildingblock2_smiles\", \"pos\": 10000, \"neg\": 30000}, # 特徵二\n    {\"bb\": \"buildingblock3_smiles\", \"pos\": 10000, \"neg\": 30000}, # 特徵三\n]\n\n# 抽樣函數\ndef stratified_sample(bb_col, pos_target, neg_target, strata_data):\n    strata_counts = strata_data[bb_col][\"strata_counts\"]\n    total_pos = strata_data[bb_col][\"total_pos\"]\n    total_neg = strata_data[bb_col][\"total_neg\"]\n\n    print(f\"🎲 進行 {bb_col} 分層抽樣...\")\n    pos_samples = []\n    neg_samples = []\n    required_cols = [\"molecule_smiles\", \"buildingblock1_smiles\", \"buildingblock2_smiles\", \"buildingblock3_smiles\", \"protein_name\", \"binds\"]\n\n    for chunk in tqdm(pd.read_csv(filename, chunksize=chunksize, usecols=required_cols), desc=f\"Sampling {bb_col}\"):\n        for bb, group in chunk.groupby(bb_col):\n            pos_chunk = group[group[\"binds\"] == 1]\n            neg_chunk = group[group[\"binds\"] == 0]\n\n            pos_size = min(len(pos_chunk), int(pos_target * (strata_counts[\"pos\"].get(bb, 0) / total_pos)))\n            neg_size = min(len(neg_chunk), int(neg_target * (strata_counts[\"neg\"].get(bb, 0) / total_neg)))\n\n            if pos_size > 0 and len(pos_samples) < pos_target:\n                pos_sample = pos_chunk.sample(n=min(pos_size, pos_target - len(pos_samples)), random_state=42)\n                pos_samples.append(pos_sample)\n\n            if neg_size > 0 and len(neg_samples) < neg_target:\n                neg_sample = neg_chunk.sample(n=min(neg_size, neg_target - len(neg_samples)), random_state=42)\n                neg_samples.append(neg_sample)\n\n        if len(pos_samples) >= pos_target and len(neg_samples) >= neg_target:\n            break\n\n    df = pd.concat(pos_samples + neg_samples, ignore_index=True)\n    df_pos = df[df[\"binds\"] == 1].sample(n=min(pos_target, len(df[df[\"binds\"] == 1])), random_state=42)\n    df_neg = df[df[\"binds\"] == 0].sample(n=min(neg_target, len(df[df[\"binds\"] == 0])), random_state=42)\n    return pd.concat([df_pos, df_neg], ignore_index=True)\n\n# 執行分層抽樣\ntrain_dfs = []\nfor target in targets:\n    df = stratified_sample(target[\"bb\"], target[\"pos\"], target[\"neg\"], strata_data)\n    train_dfs.append(df)\n\n# 合併樣本\ntrain_df = pd.concat(train_dfs, ignore_index=True).sample(frac=1, random_state=42)\n\n# 檢查結果\nprint(\"🔎 檢查建構塊分布...\")\nbb1_unique = train_df[\"buildingblock1_smiles\"].nunique()\nbb2_unique = train_df[\"buildingblock2_smiles\"].nunique()\nbb3_unique = train_df[\"buildingblock3_smiles\"].nunique()\n\nprint(f\"總樣本數: {len(train_df)}\")\nprint(f\"正類記錄數: {len(train_df[train_df['binds'] == 1])}\")\nprint(f\"負類記錄數: {len(train_df[train_df['binds'] == 0])}\")\nprint(f\"獨特分子數: {train_df['molecule_smiles'].nunique()}\")\nprint(f\"buildingblock1_smiles相異計數: {bb1_unique}（原始271）\")\nprint(f\"buildingblock2_smiles相異計數: {bb2_unique}（原始693）\")\nprint(f\"buildingblock3_smiles相異計數: {bb3_unique}（原始872）\")\n\n# 儲存結果\ntrain_df.to_csv(\"1030_40_data.csv\", index=False)\n     ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T04:48:05.758617Z","iopub.execute_input":"2025-03-27T04:48:05.759066Z","iopub.status.idle":"2025-03-27T05:10:21.628273Z","shell.execute_reply.started":"2025-03-27T04:48:05.759034Z","shell.execute_reply":"2025-03-27T05:10:21.627129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom rdkit import Chem\nfrom rdkit.Chem import rdFingerprintGenerator\nfrom joblib import Parallel, delayed\n\n# 定義 SMILES 轉換函數\ndef smiles_to_morgan_fingerprint(smiles, n_bits=2048):\n    mol = Chem.MolFromSmiles(smiles)\n    if mol is None:\n        return np.zeros(n_bits, dtype=int)\n    else:\n        generator = rdFingerprintGenerator.GetMorganGenerator(radius=2, fpSize=n_bits)\n        return np.array(generator.GetFingerprint(mol), dtype=int)\n\n# 並行處理 SMILES 轉換\ndef parallel_smiles_conversion(smiles_series, n_jobs=4):\n    results = Parallel(n_jobs=n_jobs, backend='loky')(\n        delayed(smiles_to_morgan_fingerprint)(smiles) for smiles in smiles_series\n    )\n    return results\n\n# 載入檔案\ninput_file = '/kaggle/working/1030_40_data.csv'\ndf_test = pd.read_csv(input_file)\n\n# 分批處理參數\nbatch_size = 100_000  # 每批處理 10 萬筆，可根據記憶體情況調整\nn_batches = (len(df_test) + batch_size - 1) // batch_size\n\n# 儲存中間結果\noutput_X_dir = 'temp_X_batches'\nos.makedirs(output_X_dir, exist_ok=True)\n\nfor i in range(n_batches):\n    start_idx = i * batch_size\n    end_idx = min((i + 1) * batch_size, len(df_test))\n    batch_df = df_test.iloc[start_idx:end_idx].copy()\n\n    print(f\"Processing batch {i+1}/{n_batches} ({start_idx} to {end_idx})\")\n\n    # 對當前批次的 \"molecule_smiles\" 進行轉換\n    batch_df[\"molecule_smiles\"] = parallel_smiles_conversion(batch_df[\"molecule_smiles\"], n_jobs=4)\n\n    # 轉換為指紋數據框\n    fingerprints_df = pd.DataFrame(batch_df['molecule_smiles'].to_list())\n    protein_onehot = pd.get_dummies(batch_df[\"protein_name\"], prefix=\"protein\").astype(int).reset_index(drop=True)\n    X_batch = pd.concat([fingerprints_df, protein_onehot], axis=1)\n    X_batch.columns = X_batch.columns.astype(str)\n\n    # X 轉成 int8\n    int_cols = X_batch.select_dtypes(include=['int64']).columns\n    for col in int_cols:\n        X_batch[col] = X_batch[col].astype(np.int8)\n\n    # 儲存當前批次到臨時檔案\n    batch_file = os.path.join(output_X_dir, f'X_batch_{i}.parquet')\n    X_batch.to_parquet(batch_file)\n\n    # 清理記憶體\n    del batch_df, fingerprints_df, protein_onehot, X_batch\n\n# 合併所有 X 批次\nX_test = pd.concat([pd.read_parquet(os.path.join(output_X_dir, f))\n                   for f in os.listdir(output_X_dir) if f.endswith('.parquet')],\n                   axis=0)\n\n# 處理 y\ndf_test['binds'] = df_test['binds'].astype(np.int8)\ny_test = df_test['binds'].reset_index(drop=True)  # 重置索引為連續的 RangeIndex\n\nfrom sklearn.utils import shuffle\n\n# 打亂數據，但保持 X 和 y 的對應關係\nX_test, y_test = shuffle(X_test, y_test, random_state=42)\n\n# 再次儲存為 Parquet\nX_test.to_parquet('mg1030_X.parquet', index=False)\ny_test.to_frame().to_parquet('mg1030_y.parquet', index=False)\n\n'''\n# 儲存 X_test 為 Parquet 檔案\nX_test.to_parquet('mg1030_X.parquet', index=False)\n\n# 將 y_test 轉換為 DataFrame 並儲存為 Parquet 檔案\ny_test.to_frame().to_parquet('mg1030_y.parquet', index=False)\n'''\n\n# 可選：清理臨時目錄\nimport shutil\nshutil.rmtree('temp_X_batches')\n\n# restart kernal 釋放記憶體","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T05:11:52.608288Z","iopub.execute_input":"2025-03-27T05:11:52.608718Z","iopub.status.idle":"2025-03-27T05:42:09.931714Z","shell.execute_reply.started":"2025-03-27T05:11:52.608689Z","shell.execute_reply":"2025-03-27T05:42:09.929659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train = pd.read_parquet('/kaggle/working/mg1030_X.parquet')\ny_train = pd.read_parquet('/kaggle/working/mg1030_y.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T05:42:09.935467Z","iopub.execute_input":"2025-03-27T05:42:09.935978Z","iopub.status.idle":"2025-03-27T05:42:11.309265Z","shell.execute_reply.started":"2025-03-27T05:42:09.935950Z","shell.execute_reply":"2025-03-27T05:42:11.307234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T05:44:46.899270Z","iopub.execute_input":"2025-03-27T05:44:46.899795Z","iopub.status.idle":"2025-03-27T05:44:46.912046Z","shell.execute_reply.started":"2025-03-27T05:44:46.899751Z","shell.execute_reply":"2025-03-27T05:44:46.910428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -------------------- XGBoost --------------------\nfrom sklearn.model_selection import RandomizedSearchCV\nfrom xgboost import XGBClassifier\nfrom scipy.stats import randint\n# 最佳 param\nxgb_model = XGBClassifier(colsample_bytree=0.7, gamma=0.3, learning_rate=0.5,\n                          max_depth=11, n_estimators=258, reg_alpha=0, reg_lambda=10,\n                          subsample=1.0)\nxgb_model.fit(X_train, y_train)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T05:42:11.311227Z","iopub.execute_input":"2025-03-27T05:42:11.311583Z","iopub.status.idle":"2025-03-27T05:44:46.868197Z","shell.execute_reply.started":"2025-03-27T05:42:11.311561Z","shell.execute_reply":"2025-03-27T05:44:46.866203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#儲存模型\nimport pickle\nwith open(\"/kaggle/working/15_05_new_xgb_model.bin\", \"wb\") as f:\n    pickle.dump(xgb_model, f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T05:44:46.870054Z","iopub.execute_input":"2025-03-27T05:44:46.870429Z","iopub.status.idle":"2025-03-27T05:44:46.897631Z","shell.execute_reply.started":"2025-03-27T05:44:46.870394Z","shell.execute_reply":"2025-03-27T05:44:46.895452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.preprocessing import OneHotEncoder\n\nimport duckdb\nimport pandas as pd\nfrom tqdm import tqdm\nimport numpy as np # linear algebra\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T09:41:10.104574Z","iopub.execute_input":"2025-03-27T09:41:10.105009Z","iopub.status.idle":"2025-03-27T09:41:11.861205Z","shell.execute_reply.started":"2025-03-27T09:41:10.104978Z","shell.execute_reply":"2025-03-27T09:41:11.860169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\nxg =  open(\"/kaggle/input/new-xgb/pytorch/default/1/15_05_new_xgb_model.bin\", \"rb\")\nxgb_15_05_model =  pickle.load(xg) #載入model\nxgb_15_05_model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T09:41:11.862425Z","iopub.execute_input":"2025-03-27T09:41:11.862877Z","iopub.status.idle":"2025-03-27T09:41:12.284359Z","shell.execute_reply.started":"2025-03-27T09:41:11.862843Z","shell.execute_reply":"2025-03-27T09:41:12.283246Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def smiles_to_morgan_fingerprint(smiles, n_bits=2048):\n    mol = Chem.MolFromSmiles(smiles)\n    if mol is None:\n        return np.zeros(n_bits, dtype=int)\n    else:\n        generator = AllChem.GetMorganGenerator(radius=2, fpSize=n_bits)\n        return np.array(generator.GetFingerprint(mol), dtype=int)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T09:41:12.286847Z","iopub.execute_input":"2025-03-27T09:41:12.287283Z","iopub.status.idle":"2025-03-27T09:41:12.293796Z","shell.execute_reply.started":"2025-03-27T09:41:12.287240Z","shell.execute_reply":"2025-03-27T09:41:12.292229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Process the test.parquet file chunk by chunk\ntest_file = '/kaggle/input/leash-BELKA/test.csv' #載入檔案名稱\n\ndf_test = pd.read_csv(test_file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T09:41:12.295366Z","iopub.execute_input":"2025-03-27T09:41:12.295716Z","iopub.status.idle":"2025-03-27T09:41:18.846311Z","shell.execute_reply.started":"2025-03-27T09:41:12.295670Z","shell.execute_reply":"2025-03-27T09:41:18.845272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T09:41:18.847531Z","iopub.execute_input":"2025-03-27T09:41:18.847875Z","iopub.status.idle":"2025-03-27T09:41:18.854246Z","shell.execute_reply.started":"2025-03-27T09:41:18.847845Z","shell.execute_reply":"2025-03-27T09:41:18.853148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from rdkit.Chem import AllChem\nfrom rdkit import Chem\n\noutput_file = 'submission15_05_.csv'  # 輸出檔案名稱\n\n# Read the test.parquet file into a pandas DataFrame\nfor df_test in pd.read_csv(test_file, chunksize=104681):\n    \n    \n    # 對 \"molecule_smiles\" 欄位進行轉換並顯示進度條\n    tqdm.pandas(desc=\"Transforming molecule_smiles\")\n    df_test[\"molecule_smiles\"] = df_test[\"molecule_smiles\"].progress_apply(lambda x: smiles_to_morgan_fingerprint(x))\n    df_test.columns = df_test.columns.astype(str)\n    \n    \n    # 轉成int8\n    int_cols = df_test.select_dtypes(include=['int64']).columns\n    for col in int_cols:\n        df_test[col] = df_test[col].astype(np.int8)\n    \n    \n    \n    fingerprints_df = pd.DataFrame(df_test['molecule_smiles'].to_list())\n    print(f\"fingerprints_df shape: {fingerprints_df.shape}\")  # 應該是 (104681, 2048)\n    \n    protein_onehot = pd.get_dummies(df_test[\"protein_name\"], prefix=\"protein\").astype(int).reset_index(drop=True)\n    print(f\"protein_onehot shape: {protein_onehot.shape}\")  # 應該是 (104681, X)\n    \n    X_test = pd.concat([fingerprints_df, protein_onehot], axis=1)\n    print(f\"X_test shape: {X_test.shape}\")  # 應該是 (104681, 2048 + X)\n\n    \n    print(X_test)\n    \n    # Predict the probabilities\n    probabilities = xgb_15_05_model.predict_proba(X_test)[:, 1]\n    \n    threshold = 0.5\n    predictions = (probabilities >= threshold).astype(int)\n    \n    # 產生新的 id，範圍從 295246830 到 296921725\n    df_test['id'] = range(295246830, 295246830 + len(df_test))\n    \n    # 建立輸出 DataFrame\n    output_df = pd.DataFrame({'id': df_test['id'], 'binds': predictions})\n    \n    \n    # Save the output DataFrame to a CSV file\n    output_df.to_csv(output_file, index=False, mode='a', header=not os.path.exists(output_file))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T09:41:18.855479Z","iopub.execute_input":"2025-03-27T09:41:18.855814Z","iopub.status.idle":"2025-03-27T13:14:03.690798Z","shell.execute_reply.started":"2025-03-27T09:41:18.855779Z","shell.execute_reply":"2025-03-27T13:14:03.681159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ninput_file = '/kaggle/working/submission15_05_.csv'\noutput_file = '/kaggle/working/submission15_05_new.csv'\n# 讀取已儲存的 CSV\noutput_df = pd.read_csv(input_file)\n\nprint(len(output_df))\n\n# 修改 id 欄位\noutput_df['id'] = range(295246830, 295246830 + len(output_df))\n\nprint(output_df.shape)\n\n# 將修改後的 DataFrame 儲存回 CSV\noutput_df.to_csv(output_file, index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-27T13:27:45.840516Z","iopub.execute_input":"2025-03-27T13:27:45.841000Z","iopub.status.idle":"2025-03-27T13:27:47.464293Z","shell.execute_reply.started":"2025-03-27T13:27:45.840965Z","shell.execute_reply":"2025-03-27T13:27:47.462739Z"}},"outputs":[],"execution_count":null}]}