{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# OP2 - SMILES and Gene Ontology encoding - XGBoost Model","metadata":{}},{"cell_type":"markdown","source":"In this notebook, I tried using a dataset with SMILES and Gene Ontology encoding to train an XGBoost model. \nSo far, I have gotten strong overfitting issues on the original dataset and the one with SMILES and Gene Ontology encoding. I will run hyperparameters tunning with Optuna next.\nIn parallel, I will work on processing the ATAC-seq as it is likely where I could get interesting new features. \nGene Ontology seems to have been a miss. SMILES embedding could be helpful if combined with ATAC-seq.","metadata":{}},{"cell_type":"code","source":"# Load libraries\nimport datetime\n\nimport pandas as pd\nimport numpy as np\n\nimport xgboost as xgb\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import OneHotEncoder\n\nimport matplotlib.pyplot as plt","metadata":{"id":"kZ-H-rVshDCm","execution":{"iopub.status.busy":"2023-10-24T09:26:52.790191Z","iopub.execute_input":"2023-10-24T09:26:52.790431Z","iopub.status.idle":"2023-10-24T09:26:54.207341Z","shell.execute_reply.started":"2023-10-24T09:26:52.790408Z","shell.execute_reply":"2023-10-24T09:26:54.206340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read data\ndf = pd.read_parquet(\"/kaggle/input/op2-train-v3/train_v3_20231023_1630.parquet\")","metadata":{"execution":{"iopub.status.busy":"2023-10-24T09:30:01.653773Z","iopub.execute_input":"2023-10-24T09:30:01.654983Z","iopub.status.idle":"2023-10-24T09:30:04.425033Z","shell.execute_reply.started":"2023-10-24T09:30:01.654945Z","shell.execute_reply":"2023-10-24T09:30:04.424026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shuffle the data\ndf = df.sample(frac=1.0, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-10-24T09:30:11.651858Z","iopub.execute_input":"2023-10-24T09:30:11.652239Z","iopub.status.idle":"2023-10-24T09:30:11.699235Z","shell.execute_reply.started":"2023-10-24T09:30:11.652209Z","shell.execute_reply":"2023-10-24T09:30:11.698330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pre process data\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-24T09:30:25.481322Z","iopub.execute_input":"2023-10-24T09:30:25.481692Z","iopub.status.idle":"2023-10-24T09:30:25.514238Z","shell.execute_reply.started":"2023-10-24T09:30:25.481667Z","shell.execute_reply":"2023-10-24T09:30:25.513266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# One-hot encode 'cell_type'\nencoder = OneHotEncoder(sparse_output=False)\ncell_type_encoded = encoder.fit_transform(df[['cell_type']])\ncell_type_df = pd.DataFrame(cell_type_encoded, columns=encoder.get_feature_names_out(['cell_type']))\n\n# Drop the 'SMILES' and 'cell_type' columns\ndf.drop(['SMILES', 'cell_type'], axis=1, inplace=True)\n\n# Concatenate one-hot encoded 'cell_type' to the DataFrame\ndf = pd.concat([cell_type_df, df], axis=1)\n\n# Separate the features (X) and labels (y)\nfeature_cols = list(cell_type_df.columns) + [f'PC{i+1}' for i in range(25)]\nX = df[feature_cols]\ny = df.drop(feature_cols, axis=1)\n\n# Split data into training and test sets\nX_temp, X_test, y_temp, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\nX_train, X_val, y_train, y_val = train_test_split(X_temp, y_temp, test_size=0.25, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-10-24T09:30:53.819380Z","iopub.execute_input":"2023-10-24T09:30:53.820020Z","iopub.status.idle":"2023-10-24T09:30:54.037554Z","shell.execute_reply.started":"2023-10-24T09:30:53.819988Z","shell.execute_reply":"2023-10-24T09:30:54.036750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define custom metric (Mean Rowwise Root Mean Squared Error)\ndef mrrmse(preds, dtrain):\n    labels = dtrain.get_label()\n    n = labels.shape[1] if len(labels.shape) > 1 else 1  # Get the number of columns\n    labels_reshaped = labels.reshape(-1, n)\n    preds_reshaped = preds.reshape(-1, n)\n    rowwise_errors = np.mean(np.square(labels_reshaped - preds_reshaped), axis=1)\n    return 'MRRMSE', np.sqrt(np.mean(rowwise_errors))","metadata":{"execution":{"iopub.status.busy":"2023-10-24T09:31:29.697678Z","iopub.execute_input":"2023-10-24T09:31:29.698538Z","iopub.status.idle":"2023-10-24T09:31:29.704276Z","shell.execute_reply.started":"2023-10-24T09:31:29.698503Z","shell.execute_reply":"2023-10-24T09:31:29.703257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ensure data is in float32 format\nX_train = X_train.astype(np.float32)\nX_val = X_val.astype(np.float32)\n\n# Convert the dataset into the optimized data structure used by XGBoost\ndtrain = xgb.DMatrix(X_train, label=y_train)\ndval = xgb.DMatrix(X_val, label=y_val)\n\n# Define hyperparameters\nparams = {\n    'objective': 'reg:squarederror',\n    'eval_metric': 'rmse',\n    'tree_method': 'gpu_hist',\n    'subsample': 0.8,\n    'eta': 0.05,\n    'max_depth': 2,\n    'lambda': 1.5,\n    'alpha': 0.5,\n    'colsample_bytree': 0.8\n}\n\nevals = [(dtrain, 'train'), (dval, 'eval')]\nevals_result = {}\n\nnum_round = 100  # Number of boosting rounds\nbst = xgb.train(params, dtrain, num_round, evals=evals, custom_metric=mrrmse, early_stopping_rounds=15, evals_result=evals_result)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the learning curves\ntrain_errors = evals_result['train']['MRRMSE']\nval_errors = evals_result['eval']['MRRMSE']\nplt.plot(train_errors, label='Training')\nplt.plot(val_errors, label='Testing')\nplt.xlabel('Iterations')\nplt.ylabel('Error')\nplt.legend()\nplt.title('Learning Curve')\nplt.grid(True)\nplt.show()","metadata":{},"execution_count":null,"outputs":[]}]}