{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import gc\nimport warnings\nimport random\n\nimport pandas as pd\nimport numpy as np\n\nimport optuna as op\nimport xgboost as xgb\n\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.utils.class_weight import compute_sample_weight\n\nwarnings.filterwarnings(\"ignore\")\nwarnings.filterwarnings(action=\"ignore\", category=UserWarning)\nwarnings.filterwarnings(action=\"ignore\", message=r'.*Use subset.*of np.ndarray is not recommended')\nop.logging.set_verbosity(op.logging.INFO)\nwarnings.filterwarnings(\"ignore\", category=op.exceptions.ExperimentalWarning, module=\"optuna.*\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-26T18:43:29.646431Z","iopub.execute_input":"2023-01-26T18:43:29.647448Z","iopub.status.idle":"2023-01-26T18:43:29.654600Z","shell.execute_reply.started":"2023-01-26T18:43:29.647410Z","shell.execute_reply":"2023-01-26T18:43:29.653432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"should_be_reproducible: bool = True\nreproduce = 42 if should_be_reproducible else None\nrandom.seed(reproduce)","metadata":{"execution":{"iopub.status.busy":"2023-01-26T18:43:29.656397Z","iopub.execute_input":"2023-01-26T18:43:29.657452Z","iopub.status.idle":"2023-01-26T18:43:29.675088Z","shell.execute_reply.started":"2023-01-26T18:43:29.657411Z","shell.execute_reply":"2023-01-26T18:43:29.674189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Notebook Outline\n\nThis notebook is a basic XGBoost model with hyperparameter tuning and basic feature engineering.","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/playground-series-s3e4/train.csv')\ntest = pd.read_csv('/kaggle/input/playground-series-s3e4/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-26T18:43:29.677017Z","iopub.execute_input":"2023-01-26T18:43:29.677900Z","iopub.status.idle":"2023-01-26T18:43:32.110348Z","shell.execute_reply.started":"2023-01-26T18:43:29.677838Z","shell.execute_reply":"2023-01-26T18:43:32.109055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.Class.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-26T18:43:32.111934Z","iopub.execute_input":"2023-01-26T18:43:32.112651Z","iopub.status.idle":"2023-01-26T18:43:32.124426Z","shell.execute_reply.started":"2023-01-26T18:43:32.112612Z","shell.execute_reply":"2023-01-26T18:43:32.123266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"markdown","source":"Credits to [Sergey Saharovskiy](https://www.kaggle.com/sergiosaharovskiy) as discussed in this [thread](https://www.kaggle.com/competitions/playground-series-s3e4/discussion/381087), the additions by [Konstantin Dimitriev](https://www.kaggle.com/kdmitrie) in the said thread and the [notebook](https://www.kaggle.com/code/soupmonster/simple-lgbm-baseline-optuna) from [Soupmonster](https://www.kaggle.com/soupmonster)","metadata":{}},{"cell_type":"code","source":"def preprocessing(df):\n    \n    df['even_balance'] = (df['Amount'] % 1.0 == 0).astype(int)\n    \n    df['hour'] = df['Time'] % (24 * 3600) // 3600\n    df['day'] = (df['Time'] // (24 * 3600)) % 7\n    \n    features = [feat for feat in df.columns if 'V' in feat]\n    df['V_Sum'] = df[features].sum(axis = 1)\n    df['V_Min'] = df[features].min(axis = 1)\n    df['V_Avg'] = df[features].mean(axis = 1)\n    df['V_Std'] = df[features].std(axis = 1)\n    df['V_Neg'] = df[features].lt(0).sum(axis = 1)\n    v_max = df[features].max(axis = 1)\n    df['V_Range'] = abs(df['V_Min'] - v_max)\n    \n    df['V20_div_Amount'] = df.V20 / (df.Amount + 1e-6)\n    df['V27_div_28'] = df.V27 / (df.V28 + 5e-8)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-01-26T18:43:32.127567Z","iopub.execute_input":"2023-01-26T18:43:32.128676Z","iopub.status.idle":"2023-01-26T18:43:32.142692Z","shell.execute_reply.started":"2023-01-26T18:43:32.128640Z","shell.execute_reply":"2023-01-26T18:43:32.141763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = preprocessing(train)\ntest = preprocessing(test)","metadata":{"execution":{"iopub.status.busy":"2023-01-26T18:43:32.145176Z","iopub.execute_input":"2023-01-26T18:43:32.145806Z","iopub.status.idle":"2023-01-26T18:43:33.089972Z","shell.execute_reply.started":"2023-01-26T18:43:32.145773Z","shell.execute_reply":"2023-01-26T18:43:33.088881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weights_df = pd.DataFrame(compute_sample_weight('balanced', train['Class']), columns=['weight'])","metadata":{"execution":{"iopub.status.busy":"2023-01-26T18:43:33.091560Z","iopub.execute_input":"2023-01-26T18:43:33.092128Z","iopub.status.idle":"2023-01-26T18:43:33.141566Z","shell.execute_reply.started":"2023-01-26T18:43:33.092089Z","shell.execute_reply":"2023-01-26T18:43:33.140341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x = train.drop(['id', 'Time', 'Class'], axis=1)\ntrain_y = train['Class']","metadata":{"execution":{"iopub.status.busy":"2023-01-26T18:43:33.143199Z","iopub.execute_input":"2023-01-26T18:43:33.143821Z","iopub.status.idle":"2023-01-26T18:43:33.228447Z","shell.execute_reply.started":"2023-01-26T18:43:33.143782Z","shell.execute_reply":"2023-01-26T18:43:33.227268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XGBoost: Hyperparameter Tuning","metadata":{}},{"cell_type":"code","source":"rskf = StratifiedKFold(n_splits=15, shuffle=True, random_state=reproduce)\n\ndef objective(trial):\n    \n    scores = []\n    models = []\n    \n    params = {\n        # General Parameters\n        'objective': 'binary:logistic',\n        'eval_metric': 'auc',\n        'tree_method': 'gpu_hist',\n        'predictor': 'gpu_predictor',\n        # Tree Booster Parameters\n        'max_depth': trial.suggest_int('max_depth', 1, 16),\n        'learning_rate': trial.suggest_float('learning_rate', 1e-5, 1e-2),\n        'gamma': trial.suggest_float('gamma', 1e-8, 1.0, log=True),\n        'min_child_weight': trial.suggest_int('min_child_weight', 1, 512, log=True),\n        'subsample': trial.suggest_float('subsample', 0.05, 1.0),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.2, 1.0),\n        #'max_delta_step': trial.suggest_int('max_delta_step', 1, 10, log=True),\n        #'scale_pos_weight': trial.suggest_float('scale_pos_weight', 1e-7, 1e-3),\n        # L1 regularization\n        'lambda': trial.suggest_float('lambda', 1e-8, 10.0, log=True),\n        # L2 regularization\n        'alpha': trial.suggest_float('alpha', 1e-8, 10.0, log=True)\n    }\n    \n    for train_idx, test_idx in rskf.split(train_x, train_y):\n        \n        dtrain = xgb.DMatrix(train_x.iloc[train_idx], train_y.iloc[train_idx],\n                             weight=weights_df['weight'].iloc[train_idx])\n        \n        dval = xgb.DMatrix(train_x.iloc[test_idx], train_y.iloc[test_idx],\n                           weight=weights_df['weight'].iloc[test_idx])\n        \n        pruning_callback = op.integration.XGBoostPruningCallback(trial, \"val-auc\")\n        model = xgb.train(params=params, dtrain=dtrain, evals=[(dval, \"val\")],\n                          num_boost_round=1_000, early_stopping_rounds=10,\n                          callbacks=[pruning_callback], verbose_eval=False)\n        \n        pred = model.predict(data=dval)\n        score = roc_auc_score(train_y.iloc[test_idx], pred)\n        \n        scores.append(score)\n        models.append(model)\n        \n    trial.set_user_attr(key=\"best_booster\", value=models)\n    return np.mean(scores) ","metadata":{"execution":{"iopub.status.busy":"2023-01-26T18:43:33.233207Z","iopub.execute_input":"2023-01-26T18:43:33.233749Z","iopub.status.idle":"2023-01-26T18:43:33.251450Z","shell.execute_reply.started":"2023-01-26T18:43:33.233708Z","shell.execute_reply":"2023-01-26T18:43:33.249836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def callback(study, trial):\n    if study.best_trial.number == trial.number:\n        study.set_user_attr(key='best_booster', value=trial.user_attrs['best_booster'])","metadata":{"execution":{"iopub.status.busy":"2023-01-26T18:43:33.252976Z","iopub.execute_input":"2023-01-26T18:43:33.253569Z","iopub.status.idle":"2023-01-26T18:43:33.266403Z","shell.execute_reply.started":"2023-01-26T18:43:33.253531Z","shell.execute_reply":"2023-01-26T18:43:33.265140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pruner = op.pruners.MedianPruner(n_startup_trials=35, n_warmup_steps=50)\nsampler = op.samplers.TPESampler(multivariate=True, n_startup_trials=80, seed=reproduce)\nstudy = op.create_study(direction=\"maximize\", sampler=sampler, pruner=pruner)\n\n# Best found parameters directly enqueued.\n# For full hyperparameter seach delete this row and increase n_trials in the next line.\nstudy.enqueue_trial({'max_depth': 11, 'learning_rate': 0.0024355979768246965, 'gamma': 1.303235535781138e-05, 'min_child_weight': 305, 'subsample': 0.2590690138670975, 'colsample_bytree': 0.9635655991561479, 'lambda': 2.4200715877575878e-08, 'alpha': 0.12946143512982286})\n\nstudy.optimize(objective, n_trials=1, callbacks=[callback])\n\nbest_boosters = study.user_attrs[\"best_booster\"]","metadata":{"execution":{"iopub.status.busy":"2023-01-26T18:43:33.270177Z","iopub.execute_input":"2023-01-26T18:43:33.270622Z","iopub.status.idle":"2023-01-26T18:43:44.083400Z","shell.execute_reply.started":"2023-01-26T18:43:33.270572Z","shell.execute_reply":"2023-01-26T18:43:44.082626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#0.8117309046607052\n\nprint(\"Number of finished trials: {}\".format(len(study.trials)))\nprint(\"Best trial:\")\ntrial = study.best_trial\nprint(\"  Value: {}\".format(trial.value))\nprint(\"  Params: \")\nfor key, value in trial.params.items():\n    print(\"    {}: {}\".format(key, value))","metadata":{"execution":{"iopub.status.busy":"2023-01-26T18:43:44.085025Z","iopub.execute_input":"2023-01-26T18:43:44.085692Z","iopub.status.idle":"2023-01-26T18:43:44.149910Z","shell.execute_reply.started":"2023-01-26T18:43:44.085655Z","shell.execute_reply":"2023-01-26T18:43:44.149153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Sanity Check: Competition Training Set","metadata":{}},{"cell_type":"code","source":"#0.9050669790039232\n\npred_list = []\nfor model in best_boosters:\n    pred_list.append(model.predict(xgb.DMatrix(train_x,label=train_y)))\nroc_auc_score(train_y, np.mean(pred_list, axis=0))","metadata":{"execution":{"iopub.status.busy":"2023-01-26T18:43:44.151202Z","iopub.execute_input":"2023-01-26T18:43:44.151795Z","iopub.status.idle":"2023-01-26T18:43:49.060520Z","shell.execute_reply.started":"2023-01-26T18:43:44.151756Z","shell.execute_reply":"2023-01-26T18:43:49.059435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction & Submission","metadata":{}},{"cell_type":"code","source":"test_pred = test.drop(['id', 'Time'], axis=1)\n\npred_list = []\nfor model in best_boosters:\n    pred_list.append(model.predict(xgb.DMatrix(test_pred)))\npred = np.mean(pred_list, axis=0)","metadata":{"execution":{"iopub.status.busy":"2023-01-26T18:43:49.061781Z","iopub.execute_input":"2023-01-26T18:43:49.062690Z","iopub.status.idle":"2023-01-26T18:43:52.592341Z","shell.execute_reply.started":"2023-01-26T18:43:49.062650Z","shell.execute_reply":"2023-01-26T18:43:52.591336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\"id\": test.id, \"Class\": pred})\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-26T18:43:52.593610Z","iopub.execute_input":"2023-01-26T18:43:52.593980Z","iopub.status.idle":"2023-01-26T18:43:52.818684Z","shell.execute_reply.started":"2023-01-26T18:43:52.593934Z","shell.execute_reply":"2023-01-26T18:43:52.817673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-26T18:43:52.820357Z","iopub.execute_input":"2023-01-26T18:43:52.820858Z","iopub.status.idle":"2023-01-26T18:43:52.831218Z","shell.execute_reply.started":"2023-01-26T18:43:52.820821Z","shell.execute_reply":"2023-01-26T18:43:52.830172Z"},"trusted":true},"execution_count":null,"outputs":[]}]}