{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30805,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"ON_KAGGLE = True\nTUNING = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:12:02.818654Z","iopub.execute_input":"2024-12-11T09:12:02.818954Z","iopub.status.idle":"2024-12-11T09:12:02.823101Z","shell.execute_reply.started":"2024-12-11T09:12:02.818929Z","shell.execute_reply":"2024-12-11T09:12:02.822150Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport optuna\nimport os\nimport pandas as pd\npd.options.display.max_columns = None\nimport random\nfrom concurrent.futures import ThreadPoolExecutor\nfrom colorama import Fore, Style # 出力に色をつける\nfrom IPython.display import clear_output\nfrom scipy.optimize import minimize\nfrom tqdm import tqdm\n\n# 機械学習アルゴリズム関連\nimport lightgbm as lgb\nimport xgboost as xgb\nimport catboost as ctb\n\n# scikit-learn関連\nfrom sklearn.base import clone\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.impute import SimpleImputer # KNNImputer \nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.preprocessing import OrdinalEncoder #LabelEncoder\nfrom sklearn.preprocessing import StandardScaler\n\n# torch関連\n# import torch\n# import torch.nn as nn\n# import torch.optim as optim\n\n# Warningを非表示にする\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:12:04.243470Z","iopub.execute_input":"2024-12-11T09:12:04.243765Z","iopub.status.idle":"2024-12-11T09:12:08.805306Z","shell.execute_reply.started":"2024-12-11T09:12:04.243739Z","shell.execute_reply":"2024-12-11T09:12:08.804434Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Random Seed","metadata":{}},{"cell_type":"code","source":"SEED = 42\ndef seed_everything(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    # torch.manual_seed(seed)\n    # torch.cuda.manual_seed(seed)\n    # torch.backends.cudnn.deterministic = True\n    # torch.backends.cudnn.benchmark = True\nseed_everything(SEED)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:12:23.023774Z","iopub.execute_input":"2024-12-11T09:12:23.024414Z","iopub.status.idle":"2024-12-11T09:12:23.029450Z","shell.execute_reply.started":"2024-12-11T09:12:23.024380Z","shell.execute_reply":"2024-12-11T09:12:23.028365Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Constant","metadata":{}},{"cell_type":"code","source":"# データ読み込み先\nif ON_KAGGLE:\n    PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/' # kaggle用\nelse:\n    PATH = '../ignore_dir/input/' # ローカル用\n\n# 目的変数たち\nCOL_PCIAT_SEASON = ['PCIAT-Season']\nCOL_PCIAT_N = [\n    'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05',  \n    'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', \n    'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15', \n    'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20', \n]\nCOL_PCIAT_TOTAL = ['PCIAT-PCIAT_Total']\nCOL_SII = ['sii']\nCOL_TARGETS = COL_PCIAT_SEASON + COL_PCIAT_N + COL_PCIAT_TOTAL + COL_SII\n\n# 説明変数\nCOL_FEATURES_DEMO = [\n    'Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n    'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n    'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n    'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n    'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n    'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n    'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n    'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n    'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n    'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n    'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n    'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n    'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n    'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n    'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n    'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n    'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n    'PreInt_EduHx-computerinternet_hoursday',\n]\n\n# カテゴリ変数\nCOL_CATEGORY = [\n    'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n    'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', 'PAQ_A-Season', \n    'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season'\n]\n\n# 他\nN_SPLITS = 5 # 交差検証の分割数","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:12:24.388409Z","iopub.execute_input":"2024-12-11T09:12:24.388751Z","iopub.status.idle":"2024-12-11T09:12:24.396129Z","shell.execute_reply.started":"2024-12-11T09:12:24.388721Z","shell.execute_reply":"2024-12-11T09:12:24.395388Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load","metadata":{}},{"cell_type":"code","source":"# demograhic data\ntrain_demo = pd.read_csv(f'{PATH}train.csv')\ntest_demo  = pd.read_csv(f'{PATH}test.csv')\nsample_demo = pd.read_csv(f'{PATH}sample_submission.csv')\n\nprint(train_demo.shape)\nprint(test_demo.shape)\nprint(sample_demo.shape)\ndisplay(train_demo.head(3))\ndisplay(test_demo.head(3))\ndisplay(sample_demo.head(3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:12:25.859044Z","iopub.execute_input":"2024-12-11T09:12:25.859747Z","iopub.status.idle":"2024-12-11T09:12:26.059063Z","shell.execute_reply.started":"2024-12-11T09:12:25.859713Z","shell.execute_reply":"2024-12-11T09:12:26.058223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# actigraphy data\ndef process_file(filename, dirname):\n    data_ft = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    data_ft.drop('step', axis=1, inplace=True)\n    return data_ft.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    data_ft = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n    data_ft['id'] = indexes\n    \n    return data_ft\n\ntrain_ts = load_time_series(f'{PATH}series_train.parquet')\ntest_ts = load_time_series(f'{PATH}series_test.parquet')\nprint(train_ts.shape)\nprint(test_ts.shape)\ndisplay(train_ts.head(3))\ndisplay(test_ts.head(3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:12:26.548079Z","iopub.execute_input":"2024-12-11T09:12:26.548646Z","iopub.status.idle":"2024-12-11T09:13:37.565416Z","shell.execute_reply.started":"2024-12-11T09:12:26.548612Z","shell.execute_reply":"2024-12-11T09:13:37.564587Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Preprocess","metadata":{}},{"cell_type":"code","source":"train_merged = pd.merge(train_demo, train_ts, on='id', how='left')\ntest_merged  = pd.merge(test_demo, test_ts, on='id', how='left')\n\ntrain_merged = train_merged.set_index('id')\ntest_merged = test_merged.set_index('id')\n\nprint(train_merged.shape)\nprint(test_merged.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:13:37.567030Z","iopub.execute_input":"2024-12-11T09:13:37.567618Z","iopub.status.idle":"2024-12-11T09:13:37.601072Z","shell.execute_reply.started":"2024-12-11T09:13:37.567577Z","shell.execute_reply":"2024-12-11T09:13:37.600303Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display(train_merged.head())\ndisplay(test_merged.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:13:37.602233Z","iopub.execute_input":"2024-12-11T09:13:37.602804Z","iopub.status.idle":"2024-12-11T09:13:37.788630Z","shell.execute_reply.started":"2024-12-11T09:13:37.602764Z","shell.execute_reply":"2024-12-11T09:13:37.787863Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Global Variables","metadata":{}},{"cell_type":"code","source":"COL_NUM = ['Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-CGAS_Score', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference', 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-computerinternet_hoursday']\nCOL_TABULAR = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference', 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season', 'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-Season', 'PreInt_EduHx-computerinternet_hoursday']\nCOL_TARGET = 'sii'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:13:37.790993Z","iopub.execute_input":"2024-12-11T09:13:37.791386Z","iopub.status.idle":"2024-12-11T09:13:37.797567Z","shell.execute_reply.started":"2024-12-11T09:13:37.791351Z","shell.execute_reply":"2024-12-11T09:13:37.796625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"COL_TS = train_ts.columns.tolist()\nCOL_TS.remove('id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:13:37.798725Z","iopub.execute_input":"2024-12-11T09:13:37.798980Z","iopub.status.idle":"2024-12-11T09:13:37.812410Z","shell.execute_reply.started":"2024-12-11T09:13:37.798956Z","shell.execute_reply":"2024-12-11T09:13:37.811658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"COL_FEATURE = COL_TABULAR + COL_TS\nCOL_NUM = COL_NUM + COL_TS\nprint(len(COL_FEATURE))\nprint(len(COL_NUM))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:13:37.813281Z","iopub.execute_input":"2024-12-11T09:13:37.813514Z","iopub.status.idle":"2024-12-11T09:13:37.823936Z","shell.execute_reply.started":"2024-12-11T09:13:37.813491Z","shell.execute_reply":"2024-12-11T09:13:37.823095Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Drop Rows with Missing Targets","metadata":{}},{"cell_type":"code","source":"train_merged = train_merged.dropna(subset=[COL_TARGET])\nprint(train_merged.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:13:37.824789Z","iopub.execute_input":"2024-12-11T09:13:37.825022Z","iopub.status.idle":"2024-12-11T09:13:37.839249Z","shell.execute_reply.started":"2024-12-11T09:13:37.825000Z","shell.execute_reply":"2024-12-11T09:13:37.838318Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Numeric Value Imputing","metadata":{}},{"cell_type":"code","source":"imputer = SimpleImputer(\n    strategy='mean',\n)\n\ntrain_merged[COL_NUM] = imputer.fit_transform(train_merged[COL_NUM])\ntest_merged[COL_NUM] = imputer.transform(test_merged[COL_NUM])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:13:37.840381Z","iopub.execute_input":"2024-12-11T09:13:37.840695Z","iopub.status.idle":"2024-12-11T09:13:37.889524Z","shell.execute_reply.started":"2024-12-11T09:13:37.840671Z","shell.execute_reply":"2024-12-11T09:13:37.888884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_merged.shape)\nprint(test_merged.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:13:37.890471Z","iopub.execute_input":"2024-12-11T09:13:37.890708Z","iopub.status.idle":"2024-12-11T09:13:37.894818Z","shell.execute_reply.started":"2024-12-11T09:13:37.890685Z","shell.execute_reply":"2024-12-11T09:13:37.893959Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Category Encoding","metadata":{}},{"cell_type":"code","source":"encoder = OrdinalEncoder(\n    dtype=np.int32,\n    handle_unknown='use_encoded_value',\n    unknown_value=-1,\n    encoded_missing_value=-2,\n)\n\ntrain_merged[COL_CATEGORY] = encoder.fit_transform(train_merged[COL_CATEGORY])\ntrain_merged[COL_CATEGORY] = train_merged[COL_CATEGORY].astype(float)\n\ntest_merged[COL_CATEGORY] = encoder.transform(test_merged[COL_CATEGORY])\ntest_merged[COL_CATEGORY] = test_merged[COL_CATEGORY].astype(float)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:13:37.896649Z","iopub.execute_input":"2024-12-11T09:13:37.896878Z","iopub.status.idle":"2024-12-11T09:13:37.920868Z","shell.execute_reply.started":"2024-12-11T09:13:37.896856Z","shell.execute_reply":"2024-12-11T09:13:37.920066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_merged.shape)\ndisplay(train_merged.head())\nprint(test_merged.shape)\ndisplay(test_merged.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:13:37.921844Z","iopub.execute_input":"2024-12-11T09:13:37.922160Z","iopub.status.idle":"2024-12-11T09:13:38.096626Z","shell.execute_reply.started":"2024-12-11T09:13:37.922124Z","shell.execute_reply":"2024-12-11T09:13:38.095776Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Split","metadata":{}},{"cell_type":"code","source":"X_train = train_merged[COL_FEATURE]\ny_train = train_merged[COL_TARGET]\nX_test = test_merged[COL_FEATURE]\n\nprint(X_train.shape)\nprint(y_train.shape)\nprint(X_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:13:38.097925Z","iopub.execute_input":"2024-12-11T09:13:38.098348Z","iopub.status.idle":"2024-12-11T09:13:38.111506Z","shell.execute_reply.started":"2024-12-11T09:13:38.098300Z","shell.execute_reply":"2024-12-11T09:13:38.110614Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Hyperparameter Tuning","metadata":{}},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=N_SPLITS, shuffle=True, random_state=SEED)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:13:38.112458Z","iopub.execute_input":"2024-12-11T09:13:38.112830Z","iopub.status.idle":"2024-12-11T09:13:38.117196Z","shell.execute_reply.started":"2024-12-11T09:13:38.112788Z","shell.execute_reply":"2024-12-11T09:13:38.116423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set Models\n\ndef threshold_rounder(y_pred, thresholds):\n    return np.where(y_pred < thresholds[0], 0,\n                    np.where(y_pred < thresholds[1], 1,\n                             np.where(y_pred < thresholds[2], 2, 3)))\n\ndef eval_preds(thresholds, y_true, y_pred):\n    y_pred = threshold_rounder(y_pred, thresholds)\n    score = cohen_kappa_score(y_true, y_pred, weights='quadratic')\n    return -score\n\nclass CustomLGBMRegressor(lgb.LGBMRegressor):\n    '''\n    Custom LightGBM Regressor\n    \n    It optimizes threshold values during fitting.\n    Main goal is preventing overfit on validation data.\n    '''\n    def fit(self, X, y, **kwargs):\n        super().fit(X, y, **kwargs)\n        y_pred = super().predict(X, **kwargs)\n        \n        self.optimizer = minimize(\n            eval_preds, \n            x0=[0.5, 1.5, 2.5], \n            args=(y, y_pred), \n            method='Nelder-Mead',\n        )\n        \n    def predict(self, X, **kwargs):\n        y_pred = super().predict(X, **kwargs)\n        y_pred = threshold_rounder(y_pred, self.optimizer.x)\n        return y_pred\n    \nclass CustomXGBRegressor(xgb.XGBRegressor):\n    '''\n    Custom XGBoost Regressor\n    \n    It optimizes threshold values during fitting.\n    Main goal is preventing overfit on validation data.\n    '''\n    def fit(self, X, y, **kwargs):\n        super().fit(X, y, **kwargs)\n        y_pred = super().predict(X, **kwargs)\n        \n        self.optimizer = minimize(\n            eval_preds, \n            x0=[0.5, 1.5, 2.5], \n            args=(y, y_pred), \n            method='Nelder-Mead',\n        )\n        \n    def predict(self, X, **kwargs):\n        y_pred = super().predict(X, **kwargs)\n        y_pred = threshold_rounder(y_pred, self.optimizer.x)\n        return y_pred\n    \nclass CustomCTBRegressor(ctb.CatBoostRegressor):\n    '''\n    Custom CatBoost Regressor\n    \n    It optimizes threshold values during fitting.\n    Main goal is preventing overfit on validation data.\n    '''\n    def fit(self, X, y, **kwargs):\n        super().fit(X, y, **kwargs)\n        y_pred = super().predict(X, **kwargs)\n        \n        self.optimizer = minimize(\n            eval_preds, \n            x0=[0.5, 1.5, 2.5], \n            args=(y, y_pred), \n            method='Nelder-Mead',\n        )\n        \n    def predict(self, X, **kwargs):\n        y_pred = super().predict(X, **kwargs)\n        y_pred = threshold_rounder(y_pred, self.optimizer.x)\n        return y_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:13:38.118256Z","iopub.execute_input":"2024-12-11T09:13:38.118547Z","iopub.status.idle":"2024-12-11T09:13:38.128579Z","shell.execute_reply.started":"2024-12-11T09:13:38.118521Z","shell.execute_reply":"2024-12-11T09:13:38.127657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#SEED = 21\n\ndef lgb_objective(trial):\n    params = {\n        'objective':         'l2',\n        'verbosity':         -1,\n        'n_iter':            200,\n        #'random_state':      SEED,\n        'boosting_type':     'gbdt',\n        'lambda_l1':         trial.suggest_float('lambda_l1', 1e-3, 10.0, log=True), # 0.005116829730239727(1e-3, 10.0)\n        'lambda_l2':         trial.suggest_float('lambda_l2', 1e-3, 10.0, log=True), # 0.0011520776712645852(1e-3, 10.0)\n        'learning_rate':     trial.suggest_float('learning_rate', 1e-2, 1e-1, log=True), # 0.02376367323636638(1e-2, 1e-1)\n        'max_depth':         trial.suggest_int('max_depth', 2, 32), # 5(4,8)\n        'num_leaves':        trial.suggest_int('num_leaves', 16, 512), # 207(16, 256)\n        'colsample_bytree':  trial.suggest_float('colsample_bytree', 0.4, 1.0), # 0.7759862336963801(0.4, 1.0)\n        'colsample_bynode':  trial.suggest_float('colsample_bynode', 0.4, 1.0), # 0.5110355095943208(0.4, 1.0)\n        'bagging_fraction':  trial.suggest_float('bagging_fraction', 0.4, 1.0), # 0.5485770314992224(0.4, 1.0)\n        'bagging_freq':      trial.suggest_int('bagging_freq', 1, 16), # 7(1, 7)\n        'min_data_in_leaf':  trial.suggest_int('min_data_in_leaf', 2, 256), # 78(5, 100)\n    }\n    \n    scores = []\n    y_pred_oof = np.zeros(X_train.shape[0])\n\n    # 交差検証を実行\n    for fold_id, (tr_index, val_index) in enumerate(skf.split(X_train, y_train)):\n        # Split\n        X_tr = X_train.iloc[tr_index]\n        y_tr = y_train.iloc[tr_index]\n        X_val = X_train.iloc[val_index]\n        y_val = y_train.iloc[val_index]\n        # Fit\n        model = CustomLGBMRegressor(**params, random_state=SEED)\n        model.fit(X_tr, y_tr)\n        # Predict\n        y_pred_val = model.predict(X_val)\n        y_pred_val_round = np.round(y_pred_val).astype(int)\n        # Save\n        y_pred_oof[val_index] = y_pred_val\n        # score: QWK\n        qwk = cohen_kappa_score(y_val, y_pred_val_round, weights='quadratic')\n        scores.append(qwk)\n        # Print\n        print(f'QWK: {qwk:.5f}')\n        # Clear output\n        clear_output(wait=True)\n    return np.mean(scores)\n\ndef xgb_objective(trial):\n    params = {\n        'objective': 'reg:squarederror',\n        'verbosity': 0,\n        'booster': 'gbtree', # trial.suggest_categorical('booster', ['gbtree', 'gblinear', 'dart']),\n        'lambda': trial.suggest_loguniform('lambda', 1e-6, 1.0),\n        'alpha': trial.suggest_loguniform('alpha', 1e-6, 1e-2),\n        'subsample': trial.suggest_uniform('subsample', 0.7, 1.0),\n        'colsample_bytree': trial.suggest_uniform('colsample_bytree', 0.6, 1.0),\n        'max_depth': trial.suggest_int('max_depth', 3, 9),\n        'eta': trial.suggest_loguniform('eta', 1e-6, 0.5),\n        'gamma': trial.suggest_loguniform('gamma', 1e-8, 1e-2),\n        'grow_policy': 'lossguide', #trial.suggest_categorical('grow_policy', ['depthwise', 'lossguide']),\n        'seed': SEED,\n    }\n    \n    scores = []\n    y_pred_oof = np.zeros(X_train.shape[0])\n\n    # 交差検証を実行\n    for fold_id, (tr_index, val_index) in enumerate(skf.split(X_train, y_train)):\n        # Split\n        X_tr = X_train.iloc[tr_index]\n        y_tr = y_train.iloc[tr_index]\n        X_val = X_train.iloc[val_index]\n        y_val = y_train.iloc[val_index]\n        # Fit\n        model = CustomXGBRegressor(**params, random_state=SEED)\n        model.fit(X_tr, y_tr)\n        # Predict\n        y_pred_val = model.predict(X_val)\n        y_pred_val_round = np.round(y_pred_val).astype(int)\n        # Save\n        y_pred_oof[val_index] = y_pred_val\n        # score: QWK\n        qwk = cohen_kappa_score(y_val, y_pred_val_round, weights='quadratic')\n        scores.append(qwk)\n        # Print\n        print(f'QWK: {qwk:.5f}')\n        # Clear output\n        clear_output(wait=True)\n    return np.mean(scores)\n\ndef ctb_objective(trial):\n    params = {\n        #'cat_features': COL_CATEGORY,\n        'objective': 'RMSE',\n        'verbose': 0,\n        'random_seed': SEED,\n        'bootstrap_type': 'Bernoulli', # trial.suggest_categorical('bootstrap_type', ['Bayesian', 'Bernoulli', 'MVS']),\n        'subsample': trial.suggest_uniform('subsample', 0.7, 1.0),\n        'learning_rate': trial.suggest_loguniform('learning_rate', 1e-3, 1e-1),\n        'max_depth': trial.suggest_int('max_depth', 3, 9),\n        'l2_leaf_reg': trial.suggest_loguniform('l2_leaf_reg', 1e-6, 1.0),\n        'colsample_bylevel': trial.suggest_uniform('colsample_bylevel', 0.6, 1.0),\n        'min_child_samples': trial.suggest_int('min_child_samples', 1, 16),\n        #'task_type': 'GPU',\n    }\n    \n    scores = []\n    y_pred_oof = np.zeros(X_train.shape[0])\n\n    # 交差検証を実行\n    for fold_id, (tr_index, val_index) in enumerate(skf.split(X_train, y_train)):\n        # Split\n        X_tr = X_train.iloc[tr_index]\n        y_tr = y_train.iloc[tr_index]\n        X_val = X_train.iloc[val_index]\n        y_val = y_train.iloc[val_index]\n        # Fit\n        model = CustomCTBRegressor(**params)\n        model.fit(X_tr, y_tr)\n        # Predict\n        y_pred_val = model.predict(X_val)\n        y_pred_val_round = np.round(y_pred_val).astype(int)\n        # Save\n        y_pred_oof[val_index] = y_pred_val\n        # score: QWK\n        qwk = cohen_kappa_score(y_val, y_pred_val_round, weights='quadratic')\n        scores.append(qwk)\n        # Print\n        print(f'QWK: {qwk:.5f}')\n        # Clear output\n        clear_output(wait=True)\n    return np.mean(scores)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:13:54.513227Z","iopub.execute_input":"2024-12-11T09:13:54.513582Z","iopub.status.idle":"2024-12-11T09:13:54.530011Z","shell.execute_reply.started":"2024-12-11T09:13:54.513551Z","shell.execute_reply":"2024-12-11T09:13:54.529151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if ON_KAGGLE:\n    pass\nelse:\n    study = optuna.create_study(direction='maximize', study_name='Regressor')\n    study.optimize(lgb_objective, n_trials=1000, show_progress_bar=True)\n    #study.optimize(xgb_objective, n_trials=1000, show_progress_bar=True)\n    #study.optimize(ctb_objective, n_trials=10, show_progress_bar=True)\n    print('========================================')\n    print('Best parameters:', study.best_params)\n    print('========================================')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:14:03.479853Z","iopub.execute_input":"2024-12-11T09:14:03.480553Z","iopub.status.idle":"2024-12-11T09:14:03.485085Z","shell.execute_reply.started":"2024-12-11T09:14:03.480516Z","shell.execute_reply":"2024-12-11T09:14:03.484111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print('========================================')\n# print('Best parameters:', study.best_params)\n# print('========================================')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:14:10.471668Z","iopub.execute_input":"2024-12-11T09:14:10.472002Z","iopub.status.idle":"2024-12-11T09:14:10.475955Z","shell.execute_reply.started":"2024-12-11T09:14:10.471972Z","shell.execute_reply":"2024-12-11T09:14:10.474976Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train","metadata":{}},{"cell_type":"code","source":"# Set Hyperparameters\nparams_lgb_21 = {\n    'objective'       : 'l2',\n    'verbosity'       : -1,\n    'n_iter'          : 200,\n    'lambda_l1'       : 1.1036838852846345, #\n    'lambda_l2'       : 0.0018727057435177147, # \n    'learning_rate'   : 0.013593221762698716, #\n    'max_depth'       : 18, # \n    'num_leaves'      : 97, # \n    'colsample_bytree': 0.9191575323682373, # \n    'colsample_bynode': 0.6600892388188884, # \n    'bagging_fraction': 0.9558619896854538, # \n    'bagging_freq'    : 8, # \n    'min_data_in_leaf': 114, # \n}\n\nparams_lgb = {\n    'objective'       : 'l2',\n    'verbosity'       : -1,\n    'n_iter'          : 200,\n    'lambda_l1'       : 1.4058736888672292, # 2.7276207582583747, 0.005116829730239727,\n    'lambda_l2'       : 0.01746155322499161, # 0.25376531773233724, 0.0011520776712645852,\n    'learning_rate'   : 0.030811264356272017, # 0.020593132574393384, 0.02376367323636638,\n    'max_depth'       : 3, # 5, 5,\n    'num_leaves'      : 294, # 222, 207,\n    'colsample_bytree': 0.5531503443149224, # 0.907617122772157, 0.7759862336963801,\n    'colsample_bynode': 0.7872499757233671, # 0.49236906446384354, 0.5110355095943208,\n    'bagging_fraction': 0.9782746298736651, # 0.968399248398697, 0.5485770314992224,\n    'bagging_freq'    : 1, # 5, 7,\n    'min_data_in_leaf': 131, # 140, 78,\n}\n\nparams_lgb_11 = {\n    'objective'       : 'l2',\n    'verbosity'       : -1,\n    'n_iter'          : 200,\n    'lambda_l1'       : 8.345449534651427, #\n    'lambda_l2'       : 0.005798251572619511, # \n    'learning_rate'   : 0.02934724501865707, #\n    'max_depth'       : 28, # \n    'num_leaves'      : 228, # \n    'colsample_bytree': 0.907721237675025, # \n    'colsample_bynode': 0.7424472424102492, # \n    'bagging_fraction': 0.9206292952137408, # \n    'bagging_freq'    : 14, # \n    'min_data_in_leaf': 161, # \n}\n\nparams_xgb = {\n    # 'objective': 'reg:squarederror',\n    # 'verbosity': 0,\n    # 'booster': 'gbtree', # trial.suggest_categorical('booster', ['gbtree', 'gblinear', 'dart']),\n    # 'lambda': study.best_params['lambda'], # trial.suggest_loguniform('lambda', 1e-8, 1.0),\n    # 'alpha': study.best_params['alpha'], # trial.suggest_loguniform('alpha', 1e-8, 1.0),\n    # 'subsample': study.best_params['subsample'], # trial.suggest_uniform('subsample', 0.5, 1.0),\n    # 'colsample_bytree': study.best_params['colsample_bytree'], # trial.suggest_uniform('colsample_bytree', 0.5, 1.0),\n    # 'max_depth': study.best_params['max_depth'], # trial.suggest_int('max_depth', 3, 9),\n    # 'eta': study.best_params['eta'], # trial.suggest_loguniform('eta', 1e-8, 0.5),\n    # 'gamma': study.best_params['gamma'], # trial.suggest_loguniform('gamma', 1e-8, 1.0),\n    # 'grow_policy': 'lossguide', #trial.suggest_categorical('grow_policy', ['depthwise', 'lossguide']),\n    # 'seed': SEED,\n\n    'objective': 'reg:squarederror',\n    'verbosity': 0,\n    'booster': 'gbtree', # trial.suggest_categorical('booster', ['gbtree', 'gblinear', 'dart']),\n    'lambda': 0.23277540043552367,\n    'alpha': 0.00021032110409257415, \n    'subsample': 0.9326347773099516, \n    'colsample_bytree': 0.6196517511131725, \n    'max_depth': 3, \n    'eta': 0.02978737477427037, \n    'gamma': 0.005039018976280037,\n    'grow_policy': 'lossguide', #trial.suggest_categorical('grow_policy', ['depthwise', 'lossguide']),\n    'seed': SEED,\n\n    # 'learning_rate': 0.05,\n    # 'max_depth': 6,\n    # 'n_estimators': 200,\n    # 'subsample': 0.8,\n    # 'colsample_bytree': 0.8,\n    # 'reg_alpha': 1,  # Increased from 0.1\n    # 'reg_lambda': 5,  # Increased from 1\n    # 'random_state': SEED\n}\n\nparams_ctb = {\n    # 'objective': 'RMSE',\n    # 'verbose': 0,\n    # 'random_seed': SEED,\n    # 'bootstrap_type': 'Bernoulli', # trial.suggest_categorical('bootstrap_type', ['Bayesian', 'Bernoulli', 'MVS']),\n    # 'subsample': study.best_params['subsample'], # trial.suggest_uniform('subsample', 0.7, 1.0),\n    # 'learning_rate': study.best_params['learning_rate'], # trial.suggest_loguniform('learning_rate', 1e-3, 1e-1),\n    # 'max_depth': study.best_params['max_depth'], # trial.suggest_int('max_depth', 3, 9),\n    # 'l2_leaf_reg': study.best_params['l2_leaf_reg'], # trial.suggest_loguniform('l2_leaf_reg', 1e-6, 1.0),\n    # 'colsample_bylevel': study.best_params['colsample_bylevel'], # trial.suggest_uniform('colsample_bylevel', 0.6, 1.0),\n    # 'min_child_samples': study.best_params['learning_rate'], # trial.suggest_int('min_child_samples', 1, 16),\n\n    'objective': 'RMSE',\n    'verbose': 0,\n    'random_seed': SEED,\n    'bootstrap_type': 'Bernoulli', \n    'subsample': 0.9151999957916496,\n    'learning_rate': 0.010592594983420255,\n    'max_depth': 3,\n    'l2_leaf_reg': 1.4551328911786104e-05,\n    'colsample_bylevel': 0.617073809019603,\n    'min_child_samples': 10,\n\n    # 'learning_rate': 0.05,\n    # 'depth': 6,\n    # 'iterations': 200,\n    # 'random_seed': SEED,\n    # 'verbose': 0,\n    # 'l2_leaf_reg': 10,  # Increase this value\n    # 'task_type': 'GPU',\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:14:16.500627Z","iopub.execute_input":"2024-12-11T09:14:16.500930Z","iopub.status.idle":"2024-12-11T09:14:16.510608Z","shell.execute_reply.started":"2024-12-11T09:14:16.500906Z","shell.execute_reply":"2024-12-11T09:14:16.509540Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_voting = VotingRegressor([\n    #('lgb_0',  CustomLGBMRegressor(**params_lgb, random_state=42)),\n    #('lgb_11', CustomLGBMRegressor(**params_lgb_11, random_state=11)),\n    ('lgb_21', CustomLGBMRegressor(**params_lgb_21, random_state=21)),\n    #('xgb_0', CustomXGBRegressor(**params_xgb, random_state=SEED)),\n    #('ctb_0', CustomCTBRegressor(**params_ctb)),\n]) #, weights=[1, 0, 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:14:18.988426Z","iopub.execute_input":"2024-12-11T09:14:18.988845Z","iopub.status.idle":"2024-12-11T09:14:18.993907Z","shell.execute_reply.started":"2024-12-11T09:14:18.988815Z","shell.execute_reply":"2024-12-11T09:14:18.992259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"qwks_no_optimized = []\ny_pred_oof = np.zeros(y_train.shape[0])\ny_preds_test = pd.DataFrame(index=X_test.index)\n\nfor fold_id, (tr_index, val_index) in enumerate(skf.split(X_train, y_train)):\n    print('========================')\n    print(f'Fold {fold_id}')\n    print('========================')\n    X_tr = X_train.iloc[tr_index]\n    y_tr = y_train.iloc[tr_index]\n    X_val = X_train.iloc[val_index]\n    # pciat_n_tr = pciat_n_train.iloc[tr_index]\n    # pciat_n_val = pciat_n_train.iloc[val_index]\n    y_val = y_train.iloc[val_index]\n    # Fit\n    model = clone(model_voting) # model_lgb\n    model.fit(X_tr, y_tr)\n    # Predict\n    y_pred_val = model.predict(X_val)\n    y_pred_val_round = np.round(y_pred_val).astype(int)\n    y_pred_test = model.predict(X_test)\n    # Save\n    y_pred_oof[val_index] = y_pred_val\n    y_preds_test[f\"fold_{fold_id}\"] = y_pred_test\n    # QWK\n    qwk_no_optimized = cohen_kappa_score(y_val, y_pred_val_round, weights='quadratic')\n    qwks_no_optimized.append(qwk_no_optimized)\n    # Print\n    print(f'QWK_no_optimized: {qwk_no_optimized:.5f}')\n    # Clear output\n    clear_output(wait=True)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:14:20.535932Z","iopub.execute_input":"2024-12-11T09:14:20.536283Z","iopub.status.idle":"2024-12-11T09:14:23.750555Z","shell.execute_reply.started":"2024-12-11T09:14:20.536250Z","shell.execute_reply":"2024-12-11T09:14:23.749706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(qwks_no_optimized)\nprint('=== CV score(QWK no optimized) ===')\nprint(f'{sum(qwks_no_optimized) / len(qwks_no_optimized):.6f}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:14:24.193236Z","iopub.execute_input":"2024-12-11T09:14:24.193945Z","iopub.status.idle":"2024-12-11T09:14:24.198216Z","shell.execute_reply.started":"2024-12-11T09:14:24.193914Z","shell.execute_reply":"2024-12-11T09:14:24.197312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\" === QWK: y_pred_oof\")\nprint(f\"{cohen_kappa_score(y_train, y_pred_oof.astype(int), weights='quadratic'):5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:14:27.479512Z","iopub.execute_input":"2024-12-11T09:14:27.479845Z","iopub.status.idle":"2024-12-11T09:14:27.487885Z","shell.execute_reply.started":"2024-12-11T09:14:27.479814Z","shell.execute_reply":"2024-12-11T09:14:27.486896Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Submit","metadata":{}},{"cell_type":"code","source":"y_sub = np.round(y_preds_test.mean(axis=1).values).astype(int)\nsub = sample_demo.copy()\nsub['sii'] = y_sub\ndisplay(sub)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:14:35.235103Z","iopub.execute_input":"2024-12-11T09:14:35.235497Z","iopub.status.idle":"2024-12-11T09:14:35.245781Z","shell.execute_reply.started":"2024-12-11T09:14:35.235468Z","shell.execute_reply":"2024-12-11T09:14:35.244929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T09:14:36.839975Z","iopub.execute_input":"2024-12-11T09:14:36.840635Z","iopub.status.idle":"2024-12-11T09:14:36.847236Z","shell.execute_reply.started":"2024-12-11T09:14:36.840600Z","shell.execute_reply":"2024-12-11T09:14:36.846563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}