{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"},{"sourceId":6783161,"sourceType":"datasetVersion","datasetId":3902994},{"sourceId":7013109,"sourceType":"datasetVersion","datasetId":4027677}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# OP2- XGBoost + Optuna\nIn this notebook, I tried to explore hyperparameters to improve the overfittings issues I am having with the model.","metadata":{}},{"cell_type":"code","source":"# Load libraries\nimport datetime\n\nimport pandas as pd\nimport numpy as np\n\nimport optuna\nimport xgboost as xgb\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OneHotEncoder\n\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-11-20T19:58:29.971817Z","iopub.execute_input":"2023-11-20T19:58:29.972460Z","iopub.status.idle":"2023-11-20T19:58:33.958208Z","shell.execute_reply.started":"2023-11-20T19:58:29.972414Z","shell.execute_reply":"2023-11-20T19:58:33.956713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read data\ndf = pd.read_parquet(\"/kaggle/input/op2-train-v4/train_v4_20231120_1328.parquet\")","metadata":{"execution":{"iopub.status.busy":"2023-11-20T19:59:19.743090Z","iopub.execute_input":"2023-11-20T19:59:19.743626Z","iopub.status.idle":"2023-11-20T19:59:22.930493Z","shell.execute_reply.started":"2023-11-20T19:59:19.743575Z","shell.execute_reply":"2023-11-20T19:59:22.929406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shuffle the data\ndf = df.sample(frac=1.0, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-11-20T19:59:28.713902Z","iopub.execute_input":"2023-11-20T19:59:28.714442Z","iopub.status.idle":"2023-11-20T19:59:28.787898Z","shell.execute_reply.started":"2023-11-20T19:59:28.714397Z","shell.execute_reply":"2023-11-20T19:59:28.786676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pre process data\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-11-20T19:59:30.893834Z","iopub.execute_input":"2023-11-20T19:59:30.894451Z","iopub.status.idle":"2023-11-20T19:59:30.935939Z","shell.execute_reply.started":"2023-11-20T19:59:30.894409Z","shell.execute_reply":"2023-11-20T19:59:30.934637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# One-hot encode 'cell_type'\nencoder = OneHotEncoder(sparse_output=False)\ncell_type_encoded = encoder.fit_transform(df[['cell_type']])\ncell_type_df = pd.DataFrame(cell_type_encoded, columns=encoder.get_feature_names_out(['cell_type']))\n\n# Drop the 'SMILES' and 'cell_type' columns\ndf.drop(['SMILES', 'cell_type'], axis=1, inplace=True)\n\n# Concatenate one-hot encoded 'cell_type' to the DataFrame\ndf = pd.concat([cell_type_df, df], axis=1)\n\n# Separate the features (X) and labels (y)\nfeature_cols = list(cell_type_df.columns) + [f'PC{i+1}' for i in range(31)]\nX = df[feature_cols]\ny = df.drop(feature_cols, axis=1)\n\n# Split data into training and test sets\nX_temp, X_test, y_temp, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\nX_train, X_val, y_train, y_val = train_test_split(X_temp, y_temp, test_size=0.25, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-11-20T19:59:36.921622Z","iopub.execute_input":"2023-11-20T19:59:36.922015Z","iopub.status.idle":"2023-11-20T19:59:37.219129Z","shell.execute_reply.started":"2023-11-20T19:59:36.921984Z","shell.execute_reply":"2023-11-20T19:59:37.217818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define custom metric (Mean Rowwise Root Mean Squared Error)\ndef mrrmse(preds, dtrain):\n    labels = dtrain.get_label()\n    n = labels.shape[1] if len(labels.shape) > 1 else 1  # Get the number of columns\n    labels_reshaped = labels.reshape(-1, n)\n    preds_reshaped = preds.reshape(-1, n)\n    rowwise_errors = np.mean(np.square(labels_reshaped - preds_reshaped), axis=1)\n    return 'MRRMSE', np.sqrt(np.mean(rowwise_errors))","metadata":{"execution":{"iopub.status.busy":"2023-10-24T11:17:13.453521Z","iopub.execute_input":"2023-10-24T11:17:13.453914Z","iopub.status.idle":"2023-10-24T11:17:13.461432Z","shell.execute_reply.started":"2023-10-24T11:17:13.453883Z","shell.execute_reply":"2023-10-24T11:17:13.459881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ensure data is in float32 format\nX_train = X_train.astype(np.float32)\nX_val = X_val.astype(np.float32)\n\n# Convert the dataset into the optimized data structure used by XGBoost\ndtrain = xgb.DMatrix(X_train, label=y_train)\ndval = xgb.DMatrix(X_val, label=y_val)","metadata":{"execution":{"iopub.status.busy":"2023-10-24T11:17:25.553127Z","iopub.execute_input":"2023-10-24T11:17:25.553543Z","iopub.status.idle":"2023-10-24T11:17:27.042514Z","shell.execute_reply.started":"2023-10-24T11:17:25.553512Z","shell.execute_reply":"2023-10-24T11:17:27.041342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define number of boosting rounds\nnum_round = 100\n\n# Optuna objective  \ndef objective(trial):\n    \n    params = {\n        # Use shallower trees\n        'max_depth': trial.suggest_int('max_depth', 1, 3),\n        \n        # Adjust regularization terms\n        'lambda': trial.suggest_uniform('lambda', 0.5, 5.0),\n        'alpha': trial.suggest_uniform('alpha', 0.5, 5.0),\n        \n        # Feature subsampling\n        'colsample_bytree': trial.suggest_uniform('colsample_bytree', 0.1, 0.9),\n        \n        # Increase this to make the algorithm more conservative\n        'min_child_weight': trial.suggest_int('min_child_weight', 2, 10),\n        \n        # Adjust the complexity control\n        'gamma': trial.suggest_uniform('gamma', 0.1, 1.0),\n        \n        # Subsampling of the training instances\n        'subsample': trial.suggest_uniform('subsample', 0.6, 0.9),\n    }\n\n    # Train model\n    bst = xgb.train(params, dtrain, num_round, evals=[(dval, 'eval')], early_stopping_rounds=10)\n    \n    # Get predictions\n    preds = bst.predict(dval)\n    \n    # Evaluate MRRMSE\n    metric = mrrmse(preds, dval)\n    return metric[1]\n\n# Create study   \nstudy = optuna.create_study(direction='minimize')\nstudy.optimize(objective, n_trials=100) \n\n# Get best hyperparameters\nbest_params = study.best_trial.params","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(best_params)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Retrain with the best parameters\nevals = [(dtrain, 'train'), (dval, 'eval')]\nevals_result = {}\n\nbst = xgb.train(best_params, dtrain, num_round, evals=evals, custom_metric=mrrmse, early_stopping_rounds=15, evals_result=evals_result)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the learning curves\ntrain_errors = evals_result['train']['MRRMSE']\nval_errors = evals_result['eval']['MRRMSE']\nplt.plot(train_errors, label='Training')\nplt.plot(val_errors, label='Testing')\nplt.xlabel('Iterations')\nplt.ylabel('Error')\nplt.legend()\nplt.title('Learning Curve')\nplt.grid(True)\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save best model\nbst.save_model('/kaggle/working/xgb_model.json')\n\n# Save best hyperparameters\nimport json\nwith open('/kaggle/working/xgb_best_params.json', 'w') as f:\n    json.dump(best_params, f)","metadata":{},"execution_count":null,"outputs":[]}]}