{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. | Import Required Libraries","metadata":{}},{"cell_type":"code","source":"import gc\nimport time\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.metrics import f1_score\nfrom sklearn.utils import resample\n\nimport xgboost as xgb\n\nimport optuna\nfrom optuna.pruners import HyperbandPruner\nfrom optuna.exceptions import TrialPruned","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-28T14:27:13.481878Z","iopub.execute_input":"2023-06-28T14:27:13.482247Z","iopub.status.idle":"2023-06-28T14:27:14.821388Z","shell.execute_reply.started":"2023-06-28T14:27:13.482212Z","shell.execute_reply":"2023-06-28T14:27:14.820461Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CATS = ['event_name', 'fqid', 'room_fqid', 'text']\nNUMS = ['elapsed_time', 'level', 'page', 'room_coor_x', 'room_coor_y',\n        'screen_coor_x', 'screen_coor_y', 'hover_duration']\n\n# https://www.kaggle.com/code/kimtaehun/lightgbm-baseline-with-aggregated-log-data\nEVENTS = ['navigate_click', 'person_click', 'cutscene_click', 'object_click',\n          'map_hover', 'notification_click', 'map_click', 'observation_click',\n          'checkpoint']\n\nRANDOM_SEED = 42\nPIECES = 10","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-28T14:27:14.823332Z","iopub.execute_input":"2023-06-28T14:27:14.823680Z","iopub.status.idle":"2023-06-28T14:27:14.829270Z","shell.execute_reply.started":"2023-06-28T14:27:14.823649Z","shell.execute_reply":"2023-06-28T14:27:14.828202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. |  Load Train Data and Labels","metadata":{}},{"cell_type":"markdown","source":"## 2.1. | Read Train Data","metadata":{}},{"cell_type":"code","source":"# skip the pieces preprocessing time\ndf = pd.read_csv('/kaggle/input/pspg-utils/df.csv').set_index('session_id')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T14:27:14.830813Z","iopub.execute_input":"2023-06-28T14:27:14.831132Z","iopub.status.idle":"2023-06-28T14:27:15.560509Z","shell.execute_reply.started":"2023-06-28T14:27:14.831103Z","shell.execute_reply":"2023-06-28T14:27:15.559626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.2. | Labels Data","metadata":{}},{"cell_type":"code","source":"targets = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train_labels.csv')\ntargets['session'] = targets.session_id.apply(lambda x: int(x.split('_')[0]))\ntargets['q'] = targets.session_id.apply(lambda x: int(x.split('_')[-1][1:]))","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-06-28T14:27:15.563604Z","iopub.execute_input":"2023-06-28T14:27:15.564546Z","iopub.status.idle":"2023-06-28T14:27:16.876646Z","shell.execute_reply.started":"2023-06-28T14:27:15.564495Z","shell.execute_reply":"2023-06-28T14:27:16.875457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. | Optuna Hyperparameter Tuning","metadata":{}},{"cell_type":"code","source":"FEATURES = [c for c in df.columns if c != 'level_group']\nTARGET = 'correct'\n\nN_TRIALS = 100\n\noptuna.logging.set_verbosity(optuna.logging.WARNING) # suppress evaluation output \nPRUNER = HyperbandPruner()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T14:27:16.878337Z","iopub.execute_input":"2023-06-28T14:27:16.878704Z","iopub.status.idle":"2023-06-28T14:27:16.884428Z","shell.execute_reply.started":"2023-06-28T14:27:16.878670Z","shell.execute_reply":"2023-06-28T14:27:16.883416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialization of true and true_reshaped\nALL_USERS = df.index.unique()\ntrue = pd.DataFrame(data=np.zeros((len(ALL_USERS), 18)), index=ALL_USERS)\n\nfor q_id in range(18):\n    tmp = targets.loc[targets.q == q_id+1].set_index('session').loc[ALL_USERS]\n    true[q_id] = tmp.correct.values\n\ntrue_reshaped = true.values.reshape((-1))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-28T14:27:16.885837Z","iopub.execute_input":"2023-06-28T14:27:16.886198Z","iopub.status.idle":"2023-06-28T14:27:17.000564Z","shell.execute_reply.started":"2023-06-28T14:27:16.886169Z","shell.execute_reply":"2023-06-28T14:27:16.999657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def objective(trial, df, FEATURES, TARGET, true, true_reshaped):\n    ALL_USERS = df.index.unique()\n    gkf = GroupKFold(n_splits=5)\n    oof = pd.DataFrame(data=np.zeros((len(ALL_USERS), 18)), index=ALL_USERS)\n\n    # XGBoost parameters https://docs.aws.amazon.com/sagemaker/latest/dg/xgboost-tuning.html\n    clf_params = {\n        # booster task params\n        'alpha': trial.suggest_int('alpha', 0, 1000),\n        'colsample_bylevel': trial.suggest_float('colsample_bylevel', 0.1, 1),\n        'colsample_bynode': trial.suggest_float('colsample_bynode', 0.1, 1),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1),\n        'eta': trial.suggest_float('eta', 1e-3, 0.3),\n        'gamma': trial.suggest_int('gamma', 0, 5),\n        'lambda': trial.suggest_int('lambda', 0, 1000),\n        'max_delta_step': trial.suggest_int('max_delta_step', 0, 10),\n        'max_depth': trial.suggest_int('max_depth', 3, 10),\n        'min_child_weight': trial.suggest_int('min_child_weight', 0, 120),\n        'num_round': trial.suggest_int('num_round', 1, 4000),\n        'subsample': trial.suggest_float('subsample', 0.5, 1),\n        \n        'tree_method': 'gpu_hist',\n        \n        # learning task params\n        'objective' : 'binary:logistic',\n        'eval_metric':'logloss',\n        \n        # general task params\n        'booster': trial.suggest_categorical('booster', ['gbtree', 'dart']),\n        'verbosity': 0,\n    }\n    \n    for i, (train_index, test_index) in enumerate(gkf.split(X=df, groups=df.index)):\n        print(f'fold_{i+1}', end=' ')\n        for q_id in range(1, 19):\n\n            grp = '0-4' if q_id <= 3 else '5-12' if q_id <= 13 else '13-22'\n            \n            # TRAIN DATA\n            train_x = df.iloc[train_index]\n            train_x = train_x.loc[train_x.level_group == grp]\n            train_users = train_x.index.values\n            train_y = targets.loc[targets.q == q_id].set_index('session').loc[train_users]\n\n            # VALID DATA\n            valid_x = df.iloc[test_index]\n            valid_x = valid_x.loc[valid_x.level_group == grp]\n            valid_users = valid_x.index.values\n            valid_y = targets.loc[targets.q == q_id].set_index('session').loc[valid_users]\n\n            # TRAIN MODEL\n            clf = xgb.XGBClassifier(**clf_params)\n            clf.fit(\n                train_x[FEATURES].astype('float32'), \n                train_y[TARGET],\n                eval_set=[\n                    (valid_x[FEATURES].astype('float32'), \n                     valid_y[TARGET])\n                ],\n                verbose=0\n            )\n\n            oof.loc[valid_users, q_id-1] = clf.predict_proba(valid_x[FEATURES].astype('float32'))[:, 1]\n\n    best_score = 0\n    best_threshold = 0\n\n    oof_reshaped = oof.values.reshape((-1))\n    \n    for threshold in np.arange(0.4, 0.81, 0.01):\n        preds = (oof_reshaped > threshold).astype('int')\n        m = f1_score(true_reshaped, preds, average='macro')\n        if m > best_score:\n            best_score = m\n            best_threshold = threshold\n\n    for q_id in range(18):\n        m = f1_score(true[q_id].values, (oof[q_id].values > best_threshold).astype('int'), average='macro')\n\n    m = f1_score(true_reshaped, (oof.values.reshape((-1)) > best_threshold).astype('int'), average='macro')\n\n    elapsed_time = time.time() - start_time\n    average_time_per_trial = elapsed_time / (trial.number+1)\n    total_seconds = average_time_per_trial * int(N_TRIALS - (trial.number + 1))\n    \n    print(f'\\nProgress: {(trial.number+1) * 100 / N_TRIALS:.2f}% - Trial {trial.number+1} / {N_TRIALS}\\n'\n          f'    |- Est. time: {total_seconds // 3600:.0f} h {(total_seconds % 3600) // 60:.0f} m {total_seconds % 60:.0f} s\\n'\n          f'    |- Best Threshold: {best_threshold:.2f}\\n'\n          f'    |- F1-score: {m:.4f}\\n')\n\n    trial.report(m, 0)\n    if trial.should_prune():\n        raise optuna.exceptions.TrialPruned()\n\n    return m","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-28T14:28:37.306681Z","iopub.execute_input":"2023-06-28T14:28:37.307045Z","iopub.status.idle":"2023-06-28T14:28:37.326894Z","shell.execute_reply.started":"2023-06-28T14:28:37.307016Z","shell.execute_reply":"2023-06-28T14:28:37.325980Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Optimization\nstart_time = time.time()\nclf_study = optuna.create_study(direction=\"maximize\", pruner=PRUNER)\nclf_study.optimize(lambda trial: objective(trial, df, FEATURES, TARGET, true, true_reshaped), n_trials=N_TRIALS)","metadata":{"_kg_hide-input":false,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-28T14:28:37.328770Z","iopub.execute_input":"2023-06-28T14:28:37.329323Z","iopub.status.idle":"2023-06-28T20:44:08.248020Z","shell.execute_reply.started":"2023-06-28T14:28:37.329292Z","shell.execute_reply":"2023-06-28T20:44:08.246907Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. | Hyperparameter Tuning Result","metadata":{}},{"cell_type":"code","source":"print(\"Classifier Best trial:\")\nclf_trial = clf_study.best_trial\nprint(f\"  Best value (score): {clf_trial.value}\")\nprint(\"  Params: \")\nfor key, value in clf_trial.params.items():\n    print(f\"    {key}: {value}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-28T20:44:08.249673Z","iopub.execute_input":"2023-06-28T20:44:08.250062Z","iopub.status.idle":"2023-06-28T20:44:08.257056Z","shell.execute_reply.started":"2023-06-28T20:44:08.250026Z","shell.execute_reply":"2023-06-28T20:44:08.255970Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.visualization.plot_param_importances(clf_study)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-06-28T20:44:08.259413Z","iopub.execute_input":"2023-06-28T20:44:08.259844Z","iopub.status.idle":"2023-06-28T20:44:12.104739Z","shell.execute_reply.started":"2023-06-28T20:44:08.259811Z","shell.execute_reply":"2023-06-28T20:44:12.103886Z"},"trusted":true},"execution_count":null,"outputs":[]}]}