{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport optuna\n\nfrom catboost import CatBoostRegressor\nfrom lightgbm import LGBMRegressor\nfrom optuna.samplers import TPESampler\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom sklearn.base import clone, BaseEstimator, TransformerMixin\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score, make_scorer\nfrom xgboost import XGBRegressor","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-21T13:27:01.253569Z","iopub.execute_input":"2024-10-21T13:27:01.254082Z","iopub.status.idle":"2024-10-21T13:27:07.153842Z","shell.execute_reply.started":"2024-10-21T13:27:01.254033Z","shell.execute_reply":"2024-10-21T13:27:07.152547Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CMIData:\n    def __init__(self):\n        self.directory = Path('/kaggle/input/child-mind-institute-problematic-internet-use')\n        self._load_data()\n        self._process_data()\n\n    def _process_actigraphy(self, mode):\n        out = pd.DataFrame()\n        for path in tqdm(\n            (self.directory / f\"series_{mode}.parquet\").rglob('*.parquet')\n        ):\n            id_ = path.parent.name.split('=')[1]\n            df = pd.read_parquet(path)\n            df = df.drop(columns='step')\n            stats = df.describe().T.stack()\n            stats.index = stats.index.map(lambda x: '_'.join(x))\n            stats = stats.to_frame().T\n            stats['id'] = id_\n            out = pd.concat([out, stats], axis=0, ignore_index=True)\n        return out\n\n    def _load_data(self):\n        train = pd.read_csv(self.directory / 'train.csv')\n        test = pd.read_csv(self.directory / 'test.csv')\n        actigraphy_train = self._process_actigraphy('train')\n        actigraphy_test = self._process_actigraphy('test')\n        self.train = pd.merge(train, actigraphy_train, how='left', on='id')\n        self.test = pd.merge(test, actigraphy_test, how='left', on='id')\n\n    def _process_data(self):\n        self.train.index = self.train['id']\n        self.train = self.train.drop(columns='id')\n        self.test.index = self.test['id']\n        self.test = self.test.drop(columns='id')\n\n        cols_to_rm = [\n            col for col in self.train.columns if col not in self.test.columns\n        ]\n        cols_to_rm.remove('sii')\n        self.train = self.train.drop(columns=cols_to_rm)\n\n        self.train = self.train.loc[~self.train['sii'].isna(), :]\n        # na_prop = self.train.isna().mean()\n        # cols_to_rm = na_prop[na_prop > 0.8].index\n        # self.train = self.train.drop(columns=cols_to_rm)\n        # self.test = self.test.drop(columns=cols_to_rm)\n\n        cat_cols = self.train.select_dtypes('object').columns.to_list()\n        cat_cols.append('sii')\n        self.train[cat_cols] = self.train[cat_cols].astype('category')\n        cat_cols = self.test.select_dtypes('object').columns.to_list()\n        self.test[cat_cols] = self.test[cat_cols].astype('category')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T13:27:07.155965Z","iopub.execute_input":"2024-10-21T13:27:07.156620Z","iopub.status.idle":"2024-10-21T13:27:07.176299Z","shell.execute_reply.started":"2024-10-21T13:27:07.156576Z","shell.execute_reply":"2024-10-21T13:27:07.174524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42\n\ndef cohen_kappa_metric(y_true, y_pred):\n    return cohen_kappa_score(y_true, np.round(y_pred), weights=\"quadratic\")\n\ndef convert_to_ordinal(prediction, thresholds):\n    thresholds = np.sort(thresholds)\n    return np.where(prediction < thresholds[0], 0, np.where(prediction < thresholds[1], 1, np.where(prediction < thresholds[2], 2, 3)))","metadata":{"execution":{"iopub.status.busy":"2024-10-21T13:27:07.179218Z","iopub.execute_input":"2024-10-21T13:27:07.179811Z","iopub.status.idle":"2024-10-21T13:27:07.205348Z","shell.execute_reply.started":"2024-10-21T13:27:07.179748Z","shell.execute_reply":"2024-10-21T13:27:07.204040Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def objective(trial):\n    xgb_params = {\n        \"tree_method\": trial.suggest_categorical(\n            \"xgb__tree_method\", [\"hist\", \"approx\", \"exact\"]\n        ),\n        \"n_estimators\": trial.suggest_int(\"xgb__n_estimators\", 100, 1000),\n        \"max_depth\": trial.suggest_int(\"xgb__max_depth\", 3, 10, log=True),\n        \"learning_rate\": trial.suggest_float(\"xgb__learning_rate\", 1e-3, 1e-1),\n        \"subsample\": trial.suggest_float(\"xgb__subsample\", 0.5, 1.0),\n        \"colsample_bytree\": trial.suggest_float(\"xgb__colsample_bytree\", 0.5, 1.0),\n        \"reg_alpha\": trial.suggest_float(\"xgb__reg_alpha\", 1e-5, 1.0, log=True),\n        \"reg_lambda\": trial.suggest_float(\"xgb__reg_lambda\", 1e-5, 1.0, log=True),\n        \"min_child_weight\": trial.suggest_int(\"xgb__min_child_weight\", 1, 10),\n    }\n\n    lgb_params = {\n        \"n_jobs\": -1,\n        \"verbose\": -1,\n        \"n_estimators\": trial.suggest_int(\"lgb__n_estimators\", 100, 1000),\n        \"max_depth\": trial.suggest_int(\"lgb__max_depth\", -1, 15),\n        \"learning_rate\": trial.suggest_float(\"lgb__learning_rate\", 1e-3, 1e-1, log=True),\n        \"num_leaves\": trial.suggest_int(\"lgb__num_leaves\", 31, 256),\n        \"subsample\": trial.suggest_float(\"lgb__subsample\", 0.5, 1.0),\n        \"colsample_bytree\": trial.suggest_float(\"lgb__colsample_bytree\", 0.5, 1.0),\n        \"reg_alpha\": trial.suggest_float(\"lgb__reg_alpha\", 1e-5, 10.0, log=True),\n        \"reg_lambda\": trial.suggest_float(\"lgb__reg_lambda\", 1e-5, 10.0, log=True),\n        \"min_child_samples\": trial.suggest_int(\"lgb__min_child_samples\", 5, 100),\n    }\n\n    cat_params = {\n        \"iterations\": trial.suggest_int(\"cat__iterations\", 100, 1000),\n        \"depth\": trial.suggest_int(\"cat__depth\", 4, 10),\n        \"learning_rate\": trial.suggest_float(\"cat__learning_rate\", 1e-3, 0.3, log=True),\n        \"l2_leaf_reg\": trial.suggest_float(\"cat__l2_leaf_reg\", 1e-3, 10.0, log=True),\n        \"random_strength\": trial.suggest_float(\"cat__random_strength\", 1e-3, 10.0),\n        \"bagging_temperature\": trial.suggest_float(\n            \"cat__bagging_temperature\", 0.0, 1.0\n        ),\n        \"border_count\": trial.suggest_int(\"cat__border_count\", 1, 255),\n    }\n\n    xgb_model = XGBRegressor(**xgb_params, random_state=SEED)\n    lgb_model = LGBMRegressor(**lgb_params, random_state=SEED)\n    cat_model = CatBoostRegressor(**cat_params, random_seed=SEED, verbose=0)\n    voting_model = VotingRegressor([(\"xgb\", xgb_model), (\"lgb\", lgb_model), (\"cat\", cat_model)])\n    clf_pipeline = Pipeline(\n        [(\"preprocessor\", preprocessor), (\"voting_model\", voting_model)]\n    )\n\n    skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=SEED)\n    kappa_vals = []\n    for train_idx, val_idx in skf.split(X, y):\n        X_train, y_train = X.iloc[train_idx], y.iloc[train_idx]\n        X_val, y_val = X.iloc[val_idx], y.iloc[val_idx]\n\n        clf = clone(clf_pipeline)\n        clf.fit(X_train, y_train)\n\n        y_hat_val = clf.predict(X_val)\n        label_val = convert_to_ordinal(\n            y_hat_val,\n            thresholds=[\n                trial.suggest_float(\"threshold_1\", 0, 1),\n                trial.suggest_float(\"threshold_2\", 1, 2),\n                trial.suggest_float(\"threshold_3\", 2, 3),\n            ],\n        )\n\n        kappa_val = cohen_kappa_metric(y_val, label_val)\n        kappa_vals.append(kappa_val)\n\n        return np.mean(kappa_vals)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T13:27:07.209039Z","iopub.execute_input":"2024-10-21T13:27:07.209574Z","iopub.status.idle":"2024-10-21T13:27:07.231735Z","shell.execute_reply.started":"2024-10-21T13:27:07.209519Z","shell.execute_reply":"2024-10-21T13:27:07.230267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# if __name__ == \"__main__\":\n\ndata = CMIData()\n\nX = data.train.drop(columns=\"sii\")\ny = data.train[\"sii\"]\n\nnumerical_features = X.select_dtypes(include=np.number).columns\nnumerical_pipeline = Pipeline(steps=[(\"imputer\", SimpleImputer(strategy=\"median\"))])\n\ncategorical_features = X.select_dtypes(exclude=np.number).columns\ncategorical_pipeline = Pipeline(\n    steps=[\n        (\"imputer\", SimpleImputer(strategy=\"constant\", fill_value=\"Missing\")),\n        (\"encoder\", OneHotEncoder(drop=\"if_binary\")),\n    ]\n)\n\npreprocessor = ColumnTransformer(\n    transformers=[\n        (\"num\", numerical_pipeline, numerical_features),\n        (\"cat\", categorical_pipeline, categorical_features),\n    ]\n)\n\n# study = optuna.create_study(direction=\"maximize\", sampler=TPESampler(seed=SEED))\n# study.optimize(objective, n_trials=1000, show_progress_bar=True)\n\n# print(f\"Best trial score: {study.best_trial.value}\")\n# print(f\"Best parameters: {study.best_params}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T13:27:07.233470Z","iopub.execute_input":"2024-10-21T13:27:07.233941Z","iopub.status.idle":"2024-10-21T13:27:15.036470Z","shell.execute_reply.started":"2024-10-21T13:27:07.233890Z","shell.execute_reply":"2024-10-21T13:27:15.033140Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_params = {'xgb__tree_method': 'approx', 'xgb__n_estimators': 446, 'xgb__max_depth': 6, 'xgb__learning_rate': 0.007824167219506793, 'xgb__subsample': 0.9134697062744027, 'xgb__colsample_bytree': 0.8837311563592708, 'xgb__reg_alpha': 0.00024287014767847108, 'xgb__reg_lambda': 0.01063764922947431, 'xgb__min_child_weight': 9, 'lgb__n_estimators': 679, 'lgb__max_depth': 9, 'lgb__learning_rate': 0.013872424592100306, 'lgb__num_leaves': 245, 'lgb__subsample': 0.9765991065011734, 'lgb__colsample_bytree': 0.6892488245273761, 'lgb__reg_alpha': 0.6627468890448425, 'lgb__reg_lambda': 1.0485030660629364, 'lgb__min_child_samples': 63, 'cat__iterations': 439, 'cat__depth': 4, 'cat__learning_rate': 0.0147007696942144, 'cat__l2_leaf_reg': 0.003196608743452647, 'cat__random_strength': 1.784622624506519, 'cat__bagging_temperature': 0.5104402682415026, 'cat__border_count': 133, 'threshold_1': 0.7219292985664101, 'threshold_2': 1.0214951704445223, 'threshold_3': 2.8748369593796625}\n# best_params = study.best_params\nxgb_best_params = {k.replace('xgb__', ''): v for k, v in best_params.items() if k.startswith('xgb')}\nlgb_best_params = {k.replace('lgb__', ''): v for k, v in best_params.items() if k.startswith('lgb')}\ncat_best_params = {k.replace('cat__', ''): v for k, v in best_params.items() if k.startswith('cat')}\nbest_threshold = {k: v for k, v in best_params.items() if k.startswith('threshold')}\nbest_threshold = list(best_threshold.values())\n\n\nxgb_model = XGBRegressor(**xgb_best_params, random_state=SEED)\nlgb_model = LGBMRegressor(**lgb_best_params, random_state=SEED)\ncat_model = CatBoostRegressor(**cat_best_params, verbose=0, random_state=SEED)\nvoting_model = VotingRegressor([(\"xgb\", xgb_model), (\"lgb\", lgb_model), (\"cat\", cat_model)])\nfinal_pipeline = Pipeline([(\"preprocessor\", preprocessor), (\"voting_model\", voting_model)])\nfinal_pipeline.fit(X, y)\n\ny_hat_test = final_pipeline.predict(data.test)\nlabel_test = convert_to_ordinal(y_hat_test, best_threshold)\n\nsubmission = pd.DataFrame({'id': data.test.index, 'sii': label_test})\nsubmission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T13:27:15.037702Z","iopub.status.idle":"2024-10-21T13:27:15.038354Z","shell.execute_reply.started":"2024-10-21T13:27:15.038019Z","shell.execute_reply":"2024-10-21T13:27:15.038050Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T13:27:15.040526Z","iopub.status.idle":"2024-10-21T13:27:15.041211Z","shell.execute_reply.started":"2024-10-21T13:27:15.040853Z","shell.execute_reply":"2024-10-21T13:27:15.040886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def objective(trial):\n\n#     params = {\n#         \"random_state\": SEED,\n#         \"tree_method\": trial.suggest_categorical(\n#             \"tree_method\", [\"hist\", \"approx\", \"exact\"]\n#         ),\n#         \"n_estimators\": trial.suggest_int(\"n_estimators\", 100, 1000),\n#         \"max_depth\": trial.suggest_int(\"max_depth\", 3, 10, log=True),\n#         \"learning_rate\": trial.suggest_float(\"learning_rate\", 1e-3, 1e-1),\n#         \"subsample\": trial.suggest_float(\"subsample\", 0.5, 1.0),\n#         \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.5, 1.0),\n#         \"reg_alpha\": trial.suggest_float(\"reg_alpha\", 1e-5, 1.0, log=True),\n#         \"reg_lambda\": trial.suggest_float(\"reg_lambda\", 1e-5, 1.0, log=True),\n#         \"min_child_weight\": trial.suggest_int(\"min_child_weight\", 1, 10),\n#     }\n\n#     model = XGBRegressor(**params)\n\n#     clf_pipeline = Pipeline(steps=[(\"preprocessor\", preprocessor), (\"model\", model)])\n\n#     skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=SEED)\n\n#     kappa_vals = []\n\n#     for fold, (train_idx, val_idx) in enumerate(skf.split(X, y)):\n#         X_train, y_train = X.iloc[train_idx], y.iloc[train_idx]\n#         X_val, y_val = X.iloc[val_idx], y.iloc[val_idx]\n\n#         clf_pipeline.fit(X_train, y_train)\n#         # y_hat_train = model.predict(X_train)\n#         y_hat_val = clf_pipeline.predict(X_val)\n#         label_val = convert_to_ordinal(\n#             y_hat_val,\n#             thresholds=[\n#                 trial.suggest_float(\"threshold_1\", 0, 0.99),\n#                 trial.suggest_float(\"threshold_2\", 1, 1.99),\n#                 trial.suggest_float(\"threshold_3\", 2, 3),\n#             ],\n#         )\n\n#         kappa_val = cohen_kappa_metric(y_val, label_val)\n#         kappa_vals.append(kappa_val)\n\n#     return np.mean(kappa_vals)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T13:27:15.044386Z","iopub.status.idle":"2024-10-21T13:27:15.045114Z","shell.execute_reply.started":"2024-10-21T13:27:15.044775Z","shell.execute_reply":"2024-10-21T13:27:15.044811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# X = data.train.drop(columns='sii')\n# y = data.train['sii']\n\n# numerical_features = X.select_dtypes(include=np.number).columns\n# numerical_pipeline = Pipeline(steps=[(\"imputer\", SimpleImputer(strategy=\"median\"))])\n\n# categorical_features = X.select_dtypes(exclude=np.number).columns\n# categorical_pipeline = Pipeline(\n#     steps=[\n#         (\"imputer\", SimpleImputer(strategy=\"constant\", fill_value=\"Missing\")),\n#         (\"encoder\", OneHotEncoder(drop=\"if_binary\")),\n#     ]\n# )\n\n# preprocessor = ColumnTransformer(\n#     transformers=[\n#         (\"num\", numerical_pipeline, numerical_features),\n#         (\"cat\", categorical_pipeline, categorical_features),\n#     ]\n# )","metadata":{"execution":{"iopub.status.busy":"2024-10-21T13:27:15.047997Z","iopub.status.idle":"2024-10-21T13:27:15.048845Z","shell.execute_reply.started":"2024-10-21T13:27:15.048336Z","shell.execute_reply":"2024-10-21T13:27:15.048379Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def objective(trial):\n#     params = {\n#         \"tree_method\": trial.suggest_categorical(\"tree_method\", [\"hist\", \"approx\", \"exact\"]),\n#         \"n_estimators\": trial.suggest_int(\"n_estimators\", 100, 1000),\n#         \"max_depth\": trial.suggest_int(\"max_depth\", 3, 10, log=True),\n#         \"learning_rate\": trial.suggest_float(\"learning_rate\", 1e-3, 1e-1),\n#         \"subsample\": trial.suggest_float(\"subsample\", 0.5, 1.0),\n#         \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.5, 1.0),\n#         \"reg_alpha\": trial.suggest_float(\"reg_alpha\", 1e-5, 1.0, log=True),\n#         \"reg_lambda\": trial.suggest_float(\"reg_lambda\", 1e-5, 1.0, log=True),\n#         \"min_child_weight\": trial.suggest_int(\"min_child_weight\", 1, 10),\n#     }\n\n#     model = XGBRegressor(**params, random_state=SEED)\n\n#     clf_pipeline = Pipeline(steps=[('preprocessor', preprocessor), ('model', model)])\n\n#     skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=SEED)\n\n#     score = cross_val_score(\n#         clf_pipeline,\n#         X=X,\n#         y=y,\n#         cv=skf,\n#         scoring=QUADRATIC_KAPPA_SCORER\n#     )\n\n#     return score.mean()","metadata":{"execution":{"iopub.status.busy":"2024-10-21T13:27:15.050789Z","iopub.status.idle":"2024-10-21T13:27:15.051475Z","shell.execute_reply.started":"2024-10-21T13:27:15.051112Z","shell.execute_reply":"2024-10-21T13:27:15.051144Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# study = optuna.create_study(direction='maximize')\n# study.optimize(objective, n_trials=100, show_progress_bar=True)\n\n# print(f\"Best trial score: {study.best_trial.value}\")\n# print(f\"Best parameters: {study.best_params}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-21T13:27:15.054763Z","iopub.status.idle":"2024-10-21T13:27:15.055469Z","shell.execute_reply.started":"2024-10-21T13:27:15.055107Z","shell.execute_reply":"2024-10-21T13:27:15.055139Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # best_params = study.best_params\n\n# best_params = {\n#     'tree_method': 'exact',\n#     'n_estimators': 239,\n#     'max_depth': 7,\n#     'learning_rate': 0.035867839001371374,\n#     'subsample': 0.8088521019506825,\n#     'colsample_bytree': 0.8547542759648129,\n#     'reg_alpha': 0.03142718155558386,\n#     'reg_lambda': 0.002875666456277365,\n#     'min_child_weight': 5\n# }\n\n# best_model = XGBRegressor(\n#     **best_params,\n#     random_state=SEED\n# )\n\n# model = Pipeline(steps=[('preprocessor', preprocessor), ('best_model', best_model)])\n# model.fit(X, y)\n# y_hat = np.round(model.predict(data.test))\n# submission = pd.DataFrame({'id': data.test.index, 'sii': y_hat})\n# submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-21T13:27:15.057619Z","iopub.status.idle":"2024-10-21T13:27:15.058182Z","shell.execute_reply.started":"2024-10-21T13:27:15.057942Z","shell.execute_reply":"2024-10-21T13:27:15.057966Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission = pd.DataFrame({'id': data.test.index, 'sii': y_hat})\n# submission","metadata":{"execution":{"iopub.status.busy":"2024-10-21T13:27:15.060483Z","iopub.status.idle":"2024-10-21T13:27:15.061016Z","shell.execute_reply.started":"2024-10-21T13:27:15.060777Z","shell.execute_reply":"2024-10-21T13:27:15.060801Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-21T13:27:15.062566Z","iopub.status.idle":"2024-10-21T13:27:15.063035Z","shell.execute_reply.started":"2024-10-21T13:27:15.062815Z","shell.execute_reply":"2024-10-21T13:27:15.062838Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# num_cols = data.train.select_dtypes('number').columns.to_list()\n# cat_cols = data.train.select_dtypes('category').columns.to_list()\n# cat_cols.remove('sii')\n\n# preprocessor = ColumnTransformer(\n#     [\n#         (\n#             'cat_imp', SimpleImputer(strategy='constant',\n#                                      fill_value='NA'), cat_cols\n#         )\n#     ],\n#     remainder='passthrough'\n# )\n# preprocessor.set_output(transform='pandas')\n\n# xgb_params = {\n#     'max_depth': 5,\n#     'learning_rate': 0.05,\n#     'n_estimators': 100,\n#     'tree_method': 'hist'\n# }\n\n# clf = XGBRegressor(enable_categorical=True, **xgb_params)\n# X = data.train.drop(columns='sii')\n# y = data.train['sii']\n\n# skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=SEED)\n\n# acc_train_S = []\n# acc_val_S = []\n\n# kappa_train_S = []\n# kappa_val_S = []\n\n# for train_idx, val_idx in tqdm(skf.split(X, y)):\n#     X_train, y_train = X.iloc[train_idx], y.iloc[train_idx]\n#     X_val, y_val = X.iloc[val_idx], y.iloc[val_idx]\n\n#     X_train = clone(preprocessor).fit_transform(X_train)\n#     cols_to_cast = X_train.select_dtypes('object').columns.to_list()\n#     X_train[cols_to_cast] = X_train[cols_to_cast].astype('category')\n\n#     X_val = clone(preprocessor).fit_transform(X_val)\n#     cols_to_cast = X_val.select_dtypes('object').columns.to_list()\n#     X_val[cols_to_cast] = X_val[cols_to_cast].astype('category')\n\n#     model = clone(clf)\n#     model.fit(X_train, y_train)\n\n#     y_hat_train = np.round(model.predict(X_train))\n#     y_hat_val = np.round(model.predict(X_val))\n\n#     acc_train = accuracy_score(y_train, y_hat_train)\n#     acc_train_S.append(acc_train)\n#     acc_val = accuracy_score(y_val, y_hat_val)\n#     acc_val_S.append(acc_val)\n\n#     kappa_train = cohen_kappa_score(y_train, y_hat_train, weights='quadratic')\n#     kappa_train_S.append(kappa_train)\n#     kappa_val = cohen_kappa_score(y_val, y_hat_val, weights='quadratic')\n#     kappa_val_S.append(kappa_val)\n\n# print(acc_train_S)\n# print(kappa_train_S)\n# print(np.mean(acc_val_S))\n# print(np.mean(kappa_val_S))\n# y_train = y\n# X_train = clone(preprocessor).fit_transform(X)\n# cols_to_cast = X_train.select_dtypes('object').columns.to_list()\n# X_train[cols_to_cast] = X_train[cols_to_cast].astype('category')\n# model = clone(clf)\n# model.fit(X_train, y_train)\n\n# X_test = clone(preprocessor).fit_transform(data.test)\n# cols_to_cast = X_test.select_dtypes('object').columns.to_list()\n# X_test[cols_to_cast] = X_test[cols_to_cast].astype('category')\n# sii_hat = np.round(model.predict(X_test))\n# submission = pd.DataFrame(\n#     {\n#         'id': X_test.index,\n#         'sii': sii_hat\n#     }\n# )\n# submission\n# submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-21T13:27:15.065548Z","iopub.status.idle":"2024-10-21T13:27:15.066002Z","shell.execute_reply.started":"2024-10-21T13:27:15.065784Z","shell.execute_reply":"2024-10-21T13:27:15.065806Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}