{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"},{"sourceId":181953426,"sourceType":"kernelVersion"},{"sourceId":184117592,"sourceType":"kernelVersion"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport pickle\nimport random, os, gc\nfrom scipy import sparse\nfrom sklearn.metrics import average_precision_score\nfrom sklearn.metrics import roc_auc_score\n\nfrom xgboost import DMatrix\nimport xgboost as xgb\nimport optuna\nfrom optuna.samplers import TPESampler","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-19T12:55:30.357456Z","iopub.execute_input":"2024-06-19T12:55:30.358411Z","iopub.status.idle":"2024-06-19T12:55:32.911051Z","shell.execute_reply.started":"2024-06-19T12:55:30.358377Z","shell.execute_reply":"2024-06-19T12:55:32.910142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed: int):\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed) \n    \nseed_everything(42)","metadata":{"execution":{"iopub.status.busy":"2024-06-19T12:55:32.913093Z","iopub.execute_input":"2024-06-19T12:55:32.913828Z","iopub.status.idle":"2024-06-19T12:55:32.919580Z","shell.execute_reply.started":"2024-06-19T12:55:32.913796Z","shell.execute_reply":"2024-06-19T12:55:32.918457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fold_properties(df):\n    print('Shape :',df.shape)\n    print('block1 :',df.buildingblock1_smiles.nunique())\n    print('block2 :',df.buildingblock2_smiles.nunique())\n    print('block3 :',df.buildingblock3_smiles.nunique())    \n    for target in TARGETS:\n        print(f'Positive Cases Ratio for {target} :', target,(df[target].sum()/df.shape[0]))","metadata":{"execution":{"iopub.status.busy":"2024-06-19T12:55:32.925235Z","iopub.execute_input":"2024-06-19T12:55:32.925561Z","iopub.status.idle":"2024-06-19T12:55:32.935879Z","shell.execute_reply.started":"2024-06-19T12:55:32.925533Z","shell.execute_reply":"2024-06-19T12:55:32.935019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGETS=['binds_BRD4','binds_HSA','binds_sEH']\ndtypes = {'buildingblock1_smiles': np.int16, 'buildingblock2_smiles': np.int16, 'buildingblock3_smiles': np.int16,\n          'binds_BRD4':np.byte, 'binds_HSA':np.byte, 'binds_sEH':np.byte}\n\ntrain = pd.read_csv('/kaggle/input/leashbio-12m-sample-data-and-ecfp/train_sample.csv', dtype = dtypes, usecols=[0,1,2,4,5,6] )\ntrain_index=train.index.tolist()\nrandom.shuffle(train_index)\nprint('Train Sample Properties')\nfold_properties(train)\n    \n#Load train_ecfp sparce matrix\ntrain_ecfp = sparse.load_npz(\"/kaggle/input/leashbio-12m-sample-data-and-ecfp/train_ecfp.npz\")\nprint('train_ecfp Shape :',train_ecfp.shape)\n\ntrain_B1_unique=train.buildingblock1_smiles.unique().tolist()\ntrain_B2_unique=train.buildingblock2_smiles.unique().tolist()\ntrain_B3_unique=train.buildingblock3_smiles.unique().tolist()\ntrain_unique_smiles=set(train_B1_unique+train_B2_unique+train_B3_unique)","metadata":{"execution":{"iopub.status.busy":"2024-06-19T12:55:32.937155Z","iopub.execute_input":"2024-06-19T12:55:32.937465Z","iopub.status.idle":"2024-06-19T12:56:47.338449Z","shell.execute_reply.started":"2024-06-19T12:55:32.937436Z","shell.execute_reply":"2024-06-19T12:56:47.337485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"iterations = 4500\nearly_stopping_rounds = 150\nverbose_eval = 100\n\n\nprint('#*****#')\nprint(f'train Data')\ntrain_fold_index=train_index[:12*10**6]\n# fold_properties(train.loc[valid_fold_index])\nX_tr=train_ecfp[(train_fold_index),:]\n\n\nprint('#*****#')\nprint('Valid Data')\nvalid_fold_index=train_index[12*10**6:]\n# fold_properties(train.loc[valid_fold_index])\nX_val=train_ecfp[(valid_fold_index),:]","metadata":{"execution":{"iopub.status.busy":"2024-06-19T12:58:55.756209Z","iopub.execute_input":"2024-06-19T12:58:55.756838Z","iopub.status.idle":"2024-06-19T12:59:06.696225Z","shell.execute_reply.started":"2024-06-19T12:58:55.756807Z","shell.execute_reply":"2024-06-19T12:59:06.695193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_ecfp\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-06-19T12:59:43.738284Z","iopub.execute_input":"2024-06-19T12:59:43.739124Z","iopub.status.idle":"2024-06-19T12:59:44.994618Z","shell.execute_reply.started":"2024-06-19T12:59:43.739094Z","shell.execute_reply":"2024-06-19T12:59:44.993606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_tr=train.loc[train_fold_index]['binds_BRD4'].values\ny_val=train.loc[valid_fold_index]['binds_BRD4'].values \nxtrain = DMatrix(data=X_tr, label=y_tr)\nxvalid = DMatrix(data=X_val, label=y_val)","metadata":{"execution":{"iopub.status.busy":"2024-06-19T12:59:52.664724Z","iopub.execute_input":"2024-06-19T12:59:52.665484Z","iopub.status.idle":"2024-06-19T13:00:13.206217Z","shell.execute_reply.started":"2024-06-19T12:59:52.665448Z","shell.execute_reply":"2024-06-19T13:00:13.205420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def objective(trial, xtrain, xvalid, X_val, y_val):\n    # Define parameters to be optimized for the LGBMClassifier\n    param = {\n#                  'boosting_type':'binary:logistic'\n                  \"objective\": \"binary:logistic\",\n                  \"eval_metric\": 'aucpr',\n#                   'early_stopping_rounds': 50,\n                  'learning_rate': 0.1, \n#                    'tree_method': 'gpu_hist',\n                    'device':'cuda',\n                  'scale_pos_weight':trial.suggest_int('scale_pos_weight',1,3),\n                  'min_child_weight': trial.suggest_int('min_child_weight', 1, 10),\n                  \"max_depth\": trial.suggest_int(\"max_depth\", 19, 27),\n                  \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.4, 1),\n                  \"subsample\": trial.suggest_float(\"subsample\", 0.4, 1),\n                   \n    }\n\n    # Create an instance of LGBMClassifier with the suggested parameters\n    \n    \n    Model = xgb.train(param, xtrain,\n                           evals=[(xvalid,'validation')],\n                           verbose_eval=verbose_eval,\n                           early_stopping_rounds=early_stopping_rounds,\n                           xgb_model=None,\n                           num_boost_round=300)\n    \n    y_pred_proba = Model.predict(xvalid)  # Probability of the positive class\n    score = average_precision_score(y_val, y_pred_proba)\n\n\n\n    return score","metadata":{"execution":{"iopub.status.busy":"2024-06-19T13:11:55.721909Z","iopub.execute_input":"2024-06-19T13:11:55.722800Z","iopub.status.idle":"2024-06-19T13:11:55.730252Z","shell.execute_reply.started":"2024-06-19T13:11:55.722766Z","shell.execute_reply":"2024-06-19T13:11:55.729357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sampler = optuna.samplers.TPESampler(seed=42)  # Using Tree-structured Parzen Estimator sampler for optimization\n\n# Create a study object for Optuna optimization\nstudy = optuna.create_study(direction=\"maximize\", sampler=sampler)\n\n# Run the optimization process\nstudy.optimize(lambda trial: objective(trial, xtrain, xvalid, X_val, y_val), n_trials=150)\n\n# Get the best parameters after optimization\nbest_params = study.best_params\n\nprint('='*50)\nprint(best_params)","metadata":{"execution":{"iopub.status.busy":"2024-06-19T13:11:58.186479Z","iopub.execute_input":"2024-06-19T13:11:58.187099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Trial 44 finished with value: 0.5885028746184565 and parameters: \n#         {'scale_pos_weight': 1, 'min_child_weight': 10, 'max_depth': 23, \n#          'colsample_bytree': 0.43174701256168335, \n#          'subsample': 0.9729991751497071}. \n#         Best is trial 44 with value: 0.5885028746184565.\n# BEST HSA","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_tr=train.loc[train_fold_index]['binds_HSA'].values\n# y_val=train.loc[valid_fold_index]['binds_HSA'].values \n# xtrain =  lgb.Dataset(X_tr, label=y_tr)\n# xvalid =  lgb.Dataset(X_val, label=y_val)\n# sampler = optuna.samplers.TPESampler(seed=42)  # Using Tree-structured Parzen Estimator sampler for optimization\n\n# # Create a study object for Optuna optimization\n# study = optuna.create_study(direction=\"maximize\", sampler=sampler)\n\n# # Run the optimization process\n# study.optimize(lambda trial: objective(trial, xtrain, xvalid, X_val, y_val), n_trials=70)\n\n# # Get the best parameters after optimization\n# best_params = study.best_params\n\n# print('='*50)\n# print(best_params)","metadata":{"execution":{"iopub.status.busy":"2024-06-19T12:56:49.713587Z","iopub.status.idle":"2024-06-19T12:56:49.714048Z","shell.execute_reply.started":"2024-06-19T12:56:49.713797Z","shell.execute_reply":"2024-06-19T12:56:49.713815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_tr=train.loc[train_fold_index]['binds_sEH'].values\n# y_val=train.loc[valid_fold_index]['binds_sEH'].values \n# xtrain =  lgb.Dataset(X_tr, label=y_tr)\n# xvalid =  lgb.Dataset(X_val, label=y_val)\n# sampler = optuna.samplers.TPESampler(seed=42)  # Using Tree-structured Parzen Estimator sampler for optimization\n\n# # Create a study object for Optuna optimization\n# study = optuna.create_study(direction=\"maximize\", sampler=sampler)\n\n# # Run the optimization process\n# study.optimize(lambda trial: objective(trial, xtrain, xvalid, X_val, y_val), n_trials=70)\n\n# # Get the best parameters after optimization\n# best_params = study.best_params\n\n# print('='*50)\n# print(best_params)","metadata":{"execution":{"iopub.status.busy":"2024-06-19T12:56:49.715232Z","iopub.status.idle":"2024-06-19T12:56:49.715577Z","shell.execute_reply.started":"2024-06-19T12:56:49.715406Z","shell.execute_reply":"2024-06-19T12:56:49.715420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for i,target in enumerate(TARGETS):\n#     print(f'training target {target}')\n#     y_tr=train.loc[train_fold_index][target].values\n#     y_val=train.loc[valid_fold_index][target].values \n        \n#     xtrain =  lgb.Dataset(X_tr, label=y_tr)\n#     xvalid =  lgb.Dataset(X_val, label=y_val)\n        \n# #     scale_pos_weight=(len(y_tr)-y_tr.sum())/y_tr.sum()\n# #     print('scale_pos_weight :',scale_pos_weight)\n#     print(f\"训练开始\")\n#     lgb_params= { \"boosting_type\": \"gbdt\",\n#                   \"objective\": \"binary\",\n#                   \"metric\": 'average_precision',\n#                  'learning_rate': 0.05, \n# #                  \"n_jobs\":12,\n#                  \"seed\": 42,\n# #                  'device':'gpu'\n#                  'colsample_bytree':0.77, \n# #                  'scale_pos_weight':scale_pos_weight,\n# #                  \"n_estimators\": 100,\n#               'verbose': -1,\n#                 'early_stopping_rounds': 100\n#                 }\n    \n    \n#     Model =  lgb.train(lgb_params, xtrain,\n#             valid_sets=xvalid,num_boost_round=2500)\n    \n    \n\n        \n\n#     y_pred_proba = Model.predict(X_val)  # Probability of the positive class\n#     map_score = average_precision_score(y_val, y_pred_proba)\n#     valid_preds_score.append(map_score)\n#     print(f\"Mean Average Precision for Valid {target}: {map_score:.2f}\")\n    \n#     lgb_models.append(Model)","metadata":{"execution":{"iopub.status.busy":"2024-06-19T12:56:49.717411Z","iopub.status.idle":"2024-06-19T12:56:49.717865Z","shell.execute_reply.started":"2024-06-19T12:56:49.717640Z","shell.execute_reply":"2024-06-19T12:56:49.717657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print('binds_BRD4 :',np.mean(valid_preds_score[0::3]))\n# print('binds_HSA :' ,np.mean(valid_preds_score[1::3]))\n# print('binds_sEH :' ,np.mean(valid_preds_score[2::3]))","metadata":{"execution":{"iopub.status.busy":"2024-06-19T12:56:49.719198Z","iopub.status.idle":"2024-06-19T12:56:49.719678Z","shell.execute_reply.started":"2024-06-19T12:56:49.719435Z","shell.execute_reply":"2024-06-19T12:56:49.719453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del train,xtrain,X_tr,xvalid,X_val,y_tr,y_val,valid_preds_score\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-06-19T12:56:49.721350Z","iopub.status.idle":"2024-06-19T12:56:49.721718Z","shell.execute_reply.started":"2024-06-19T12:56:49.721559Z","shell.execute_reply":"2024-06-19T12:56:49.721572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preds_cols=['BRD4','HSA','sEH']\n# test=pd.read_parquet('/kaggle/input/leash-BELKA/test.parquet', engine = 'pyarrow')\n# blocks_dict=np.load('/kaggle/input/leashbio-create-10m-sample-data-and-ecfp/blocks_dict.npy', allow_pickle=True)\n# blocks_dict = blocks_dict.tolist()\n# test['buildingblock1_smiles']=test['buildingblock1_smiles'].map(blocks_dict).values.astype('uint16')\n# test['buildingblock2_smiles']=test['buildingblock2_smiles'].map(blocks_dict).values.astype('uint16')\n# test['buildingblock3_smiles']=test['buildingblock3_smiles'].map(blocks_dict).values.astype('uint16')\n\n# test_preds=pd.DataFrame(test['molecule_smiles'].unique(),columns=['molecule_smiles'])\n# test_preds=test[['molecule_smiles','buildingblock1_smiles','buildingblock2_smiles','buildingblock3_smiles']].copy()\n# test_preds.drop_duplicates(inplace=True)\n# test_preds=test_preds.reset_index(drop=True) \n\n# test_ecfp=sparse.load_npz(\"/kaggle/input/leashbio-create-10m-sample-data-and-ecfp/test_ecfp.npz\")\n# test_ecfp_1=test_ecfp[:,var_thresh_index_1]\n# print('test_ecfp Shape :',train_ecfp_1.shape)\n\n\n# test_B1_unique=test_preds.buildingblock1_smiles.unique().tolist()\n# test_B2_unique=test_preds.buildingblock2_smiles.unique().tolist()\n# test_B3_unique=test_preds.buildingblock3_smiles.unique().tolist()\n# test_unique_smiles=set(test_B1_unique+test_B2_unique+test_B3_unique)\n\n# test_preds[preds_cols]=np.zeros((len(test_preds),3))\n\n# print('Shape Test :', test.shape)\n# test_no_overlap=[bb for bb in test_unique_smiles if bb not in train_unique_smiles]\n# train_no_overlap=[bb for bb in train_unique_smiles if bb not in test_unique_smiles]\n# print('Test Block Unique :', len(test_unique_smiles))\n# print('Train Block Unique :', len(train_unique_smiles))\n# print('Test Block not in train :', len(test_no_overlap))\n# print('Train Block not in test :', len(train_no_overlap))","metadata":{"execution":{"iopub.status.busy":"2024-06-19T12:56:49.723062Z","iopub.status.idle":"2024-06-19T12:56:49.723481Z","shell.execute_reply.started":"2024-06-19T12:56:49.723290Z","shell.execute_reply":"2024-06-19T12:56:49.723305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lgb.Dataset(test_ecfp_1)","metadata":{"execution":{"iopub.status.busy":"2024-06-19T12:56:49.725429Z","iopub.status.idle":"2024-06-19T12:56:49.725735Z","shell.execute_reply.started":"2024-06-19T12:56:49.725584Z","shell.execute_reply":"2024-06-19T12:56:49.725596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for i,target in enumerate(TARGETS):    \n#     test_target=target.split('_')[1]      \n#     test_preds[test_target]=lgb_models[i].predict(test_ecfp_1.astype('float32'))\n\n# test_BRD4=test_preds[['molecule_smiles','BRD4']].copy()\n# test_BRD4['protein_name']='BRD4'\n# test_BRD4=test_BRD4.rename(columns={\"BRD4\": \"binds_1\"})\n\n# test_HSA=test_preds[['molecule_smiles','HSA']].copy()\n# test_HSA['protein_name']='HSA'\n# test_HSA=test_HSA.rename(columns={\"HSA\": \"binds_1\"})\n\n# test_sEH=test_preds[['molecule_smiles','sEH']].copy()\n# test_sEH['protein_name']='sEH'\n# test_sEH=test_sEH.rename(columns={\"sEH\": \"binds_1\"})\n\n# test_preds_1=pd.concat([test_BRD4,test_HSA,test_sEH])\n# test=pd.merge(test,test_preds_1, on=['molecule_smiles','protein_name'],how = 'left')","metadata":{"execution":{"iopub.status.busy":"2024-06-19T12:56:49.726784Z","iopub.status.idle":"2024-06-19T12:56:49.727136Z","shell.execute_reply.started":"2024-06-19T12:56:49.726934Z","shell.execute_reply":"2024-06-19T12:56:49.726946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sample_submission = pd.read_csv(\"/kaggle/input/leash-BELKA/sample_submission.csv\")\n# sample_submission['binds']=test['binds_1']\n# sample_submission.to_csv('submission.csv', index=False)\n# sample_submission.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-19T12:56:49.727965Z","iopub.status.idle":"2024-06-19T12:56:49.728304Z","shell.execute_reply.started":"2024-06-19T12:56:49.728144Z","shell.execute_reply":"2024-06-19T12:56:49.728157Z"},"trusted":true},"execution_count":null,"outputs":[]}]}