{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Copied From https://www.kaggle.com/code/greysky/cmi-single-lgbm-cv-0-471-lb-0-460","metadata":{}},{"cell_type":"code","source":"import os\nimport random\nfrom tqdm import tqdm\nfrom pathlib import Path\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.metrics import make_scorer\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.impute import SimpleImputer\n\nfrom scipy.optimize import minimize\nimport optuna\n\nimport lightgbm as lgb\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\n\nSEED = 42\n\nKAPPA_SCORER = make_scorer(\n    cohen_kappa_score, \n    greater_is_better=True, \n    weights='quadratic',\n)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:30:49.339947Z","iopub.execute_input":"2024-10-17T14:30:49.340686Z","iopub.status.idle":"2024-10-17T14:30:52.346753Z","shell.execute_reply.started":"2024-10-17T14:30:49.340642Z","shell.execute_reply":"2024-10-17T14:30:52.345571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    \n    return df.describe().values.reshape(-1), filename.split('=')[1]","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:30:52.348793Z","iopub.execute_input":"2024-10-17T14:30:52.349508Z","iopub.status.idle":"2024-10-17T14:30:52.358035Z","shell.execute_reply.started":"2024-10-17T14:30:52.349457Z","shell.execute_reply":"2024-10-17T14:30:52.356891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lgb_objective(trial):\n    params = {\n        'objective':         'l2',\n        'verbosity':         -1,\n        'n_iter':            200,\n        'random_state':      SEED,\n        'boosting_type':     'gbdt',\n        'lambda_l1':         trial.suggest_float('lambda_l1', 1e-3, 10.0, log=True),\n        'lambda_l2':         trial.suggest_float('lambda_l2', 1e-3, 10.0, log=True),\n        'learning_rate':     trial.suggest_float('learning_rate', 1e-2, 1e-1, log=True),\n        'max_depth':         trial.suggest_int('max_depth', 4, 8),\n        'num_leaves':        trial.suggest_int('num_leaves', 16, 256),\n        'colsample_bytree':  trial.suggest_float('colsample_bytree', 0.4, 1.0),\n        'colsample_bynode':  trial.suggest_float('colsample_bynode', 0.4, 1.0),\n        'bagging_fraction':  trial.suggest_float('bagging_fraction', 0.4, 1.0),\n        'bagging_freq':      trial.suggest_int('bagging_freq', 1, 7),\n        'min_data_in_leaf':  trial.suggest_int('min_data_in_leaf', 5, 100),\n    }\n    \n    X = df_train[feature_cols]\n    y = df_train[target_col]\n    cv = StratifiedKFold(5, shuffle=True, random_state=SEED)\n    estimator = CustomLGBMRegressor(**params)\n\n    val_scores = cross_val_score(\n        estimator=estimator, \n        X=X, y=y, \n        cv=cv, \n        scoring=KAPPA_SCORER,\n    )\n\n    return np.mean(val_scores)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:30:52.359446Z","iopub.execute_input":"2024-10-17T14:30:52.359801Z","iopub.status.idle":"2024-10-17T14:30:52.369863Z","shell.execute_reply.started":"2024-10-17T14:30:52.359766Z","shell.execute_reply":"2024-10-17T14:30:52.368857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_time_series(dirname):\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:30:52.372362Z","iopub.execute_input":"2024-10-17T14:30:52.372741Z","iopub.status.idle":"2024-10-17T14:30:52.385784Z","shell.execute_reply.started":"2024-10-17T14:30:52.372704Z","shell.execute_reply":"2024-10-17T14:30:52.383944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def quadratic_weighted_kappa(estimator, X, y_true):\n    y_pred = estimator.predict(X).round()\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:30:52.387687Z","iopub.execute_input":"2024-10-17T14:30:52.388304Z","iopub.status.idle":"2024-10-17T14:30:52.396604Z","shell.execute_reply.started":"2024-10-17T14:30:52.388254Z","shell.execute_reply":"2024-10-17T14:30:52.394994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def threshold_rounder(y_pred, thresholds):\n    return np.where(y_pred < thresholds[0], 0,\n                    np.where(y_pred < thresholds[1], 1,\n                             np.where(y_pred < thresholds[2], 2, 3)))","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:30:52.398478Z","iopub.execute_input":"2024-10-17T14:30:52.399144Z","iopub.status.idle":"2024-10-17T14:30:52.408864Z","shell.execute_reply.started":"2024-10-17T14:30:52.399088Z","shell.execute_reply":"2024-10-17T14:30:52.407390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def eval_preds(thresholds, y_true, y_pred):\n    y_pred = threshold_rounder(y_pred, thresholds)\n    score = cohen_kappa_score(y_true, y_pred, weights='quadratic')\n    return -score","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:30:52.411016Z","iopub.execute_input":"2024-10-17T14:30:52.411479Z","iopub.status.idle":"2024-10-17T14:30:52.421470Z","shell.execute_reply.started":"2024-10-17T14:30:52.411431Z","shell.execute_reply":"2024-10-17T14:30:52.419886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomLGBMRegressor(lgb.LGBMRegressor):\n    '''\n    Custom LightGBM Regressor\n    \n    It optimizes threshold values during fitting.\n    Main goal is preventing overfit on validation data.\n    '''\n    def fit(self, X, y, **kwargs):\n        super().fit(X, y, **kwargs)\n        y_pred = super().predict(X, **kwargs)\n        \n        self.optimizer = minimize(\n            eval_preds, \n            x0=[0.5, 1.5, 2.5], \n            args=(y, y_pred), \n            method='Nelder-Mead',\n        )\n        \n    def predict(self, X, **kwargs):\n        y_pred = super().predict(X, **kwargs)\n        y_pred = threshold_rounder(y_pred, self.optimizer.x)\n        return y_pred","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:30:52.422981Z","iopub.execute_input":"2024-10-17T14:30:52.423426Z","iopub.status.idle":"2024-10-17T14:30:52.432802Z","shell.execute_reply.started":"2024-10-17T14:30:52.423379Z","shell.execute_reply":"2024-10-17T14:30:52.431576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root = Path('/kaggle/input/child-mind-institute-problematic-internet-use')","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:30:52.434268Z","iopub.execute_input":"2024-10-17T14:30:52.434616Z","iopub.status.idle":"2024-10-17T14:30:52.444708Z","shell.execute_reply.started":"2024-10-17T14:30:52.434580Z","shell.execute_reply":"2024-10-17T14:30:52.442998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Tabular Data","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(root / 'train.csv')\ndf_test = pd.read_csv(root / 'test.csv')\ndf_subm = pd.read_csv(root / 'sample_submission.csv', index_col='id')","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:30:52.448682Z","iopub.execute_input":"2024-10-17T14:30:52.449040Z","iopub.status.idle":"2024-10-17T14:30:52.536784Z","shell.execute_reply.started":"2024-10-17T14:30:52.448994Z","shell.execute_reply":"2024-10-17T14:30:52.535663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:30:52.540314Z","iopub.execute_input":"2024-10-17T14:30:52.540685Z","iopub.status.idle":"2024-10-17T14:30:52.581059Z","shell.execute_reply.started":"2024-10-17T14:30:52.540648Z","shell.execute_reply":"2024-10-17T14:30:52.579361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Time Series Data","metadata":{}},{"cell_type":"code","source":"ts_train = load_time_series(root / \"series_train.parquet\")\nts_test = load_time_series(root / \"series_test.parquet\")\n\ntime_series_cols = ts_train.columns.tolist()\ntime_series_cols.remove(\"id\")","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:30:52.583288Z","iopub.execute_input":"2024-10-17T14:30:52.583740Z","iopub.status.idle":"2024-10-17T14:32:32.246390Z","shell.execute_reply.started":"2024-10-17T14:30:52.583691Z","shell.execute_reply":"2024-10-17T14:32:32.245313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ts_train","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:32.247721Z","iopub.execute_input":"2024-10-17T14:32:32.248069Z","iopub.status.idle":"2024-10-17T14:32:32.283291Z","shell.execute_reply.started":"2024-10-17T14:32:32.248027Z","shell.execute_reply":"2024-10-17T14:32:32.282300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Merge Operation","metadata":{}},{"cell_type":"code","source":"df_train = pd.merge(df_train, ts_train, how=\"left\", on='id')\ndf_test = pd.merge(df_test, ts_test, how=\"left\", on='id')\n\ndf_train = df_train.set_index('id')\ndf_test = df_test.set_index('id')","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:32.284420Z","iopub.execute_input":"2024-10-17T14:32:32.284789Z","iopub.status.idle":"2024-10-17T14:32:32.326786Z","shell.execute_reply.started":"2024-10-17T14:32:32.284750Z","shell.execute_reply":"2024-10-17T14:32:32.325851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Global Variables","metadata":{}},{"cell_type":"code","source":"cat_cols = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\nnum_cols = ['Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-CGAS_Score', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference', 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-computerinternet_hoursday']\ntabular_cols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference', 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season', 'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-Season', 'PreInt_EduHx-computerinternet_hoursday']\ntarget_col = 'sii'\n\nfeature_cols = tabular_cols + time_series_cols\nnum_cols = num_cols + time_series_cols\n\nunused_features = ['stat_10', 'stat_11', 'stat_3', 'stat_4', 'stat_41', 'stat_42', 'stat_44', 'stat_5', 'stat_58', 'stat_6', 'stat_69', 'stat_7', 'stat_70', 'stat_77', 'stat_79', 'stat_8', 'stat_88', 'stat_89', 'stat_9', 'stat_93', 'stat_46', 'stat_56', 'stat_57', 'stat_64', 'stat_68', 'stat_92', 'stat_94', 'stat_27', 'stat_28', 'stat_39', 'stat_82', 'stat_2', 'stat_37', 'stat_43', 'stat_45', 'stat_53', 'stat_60', 'stat_1', 'stat_22', 'stat_72', 'stat_18', 'stat_91']\n\n# for f in unused_features:\n#     feature_cols.del(f)\n#     num_cols.del(f)\nprint(len(feature_cols), len(num_cols), len(cat_cols))\nfeature_cols = [f for f in feature_cols if f not in unused_features]\nnum_cols = [f for f in num_cols if f not in unused_features]\ncat_cols = [f for f in cat_cols if f not in unused_features]\nprint(len(feature_cols), len(num_cols), len(cat_cols))","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:32.328218Z","iopub.execute_input":"2024-10-17T14:32:32.328585Z","iopub.status.idle":"2024-10-17T14:32:32.341246Z","shell.execute_reply.started":"2024-10-17T14:32:32.328543Z","shell.execute_reply":"2024-10-17T14:32:32.340276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for col in num_cols:\n#     df_train[col]","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:32.342636Z","iopub.execute_input":"2024-10-17T14:32:32.342951Z","iopub.status.idle":"2024-10-17T14:32:32.355742Z","shell.execute_reply.started":"2024-10-17T14:32:32.342918Z","shell.execute_reply":"2024-10-17T14:32:32.354456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Drop Rows with Missing Targets","metadata":{}},{"cell_type":"code","source":"df_train = df_train.dropna(subset=[target_col])","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:32.357329Z","iopub.execute_input":"2024-10-17T14:32:32.358118Z","iopub.status.idle":"2024-10-17T14:32:32.369643Z","shell.execute_reply.started":"2024-10-17T14:32:32.358069Z","shell.execute_reply":"2024-10-17T14:32:32.368745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Numeric Value Imputing","metadata":{}},{"cell_type":"code","source":"imputer = SimpleImputer(\n    strategy='mean',\n)\n\ndf_train[num_cols] = imputer.fit_transform(df_train[num_cols])\ndf_test[num_cols] = imputer.transform(df_test[num_cols])","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:32.370989Z","iopub.execute_input":"2024-10-17T14:32:32.371642Z","iopub.status.idle":"2024-10-17T14:32:32.418358Z","shell.execute_reply.started":"2024-10-17T14:32:32.371595Z","shell.execute_reply":"2024-10-17T14:32:32.417405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Category Encoding","metadata":{}},{"cell_type":"code","source":"encoder = OrdinalEncoder(\n    dtype=np.int32,\n    handle_unknown='use_encoded_value',\n    unknown_value=-1,\n    encoded_missing_value=-2,\n)\n\ndf_train[cat_cols] = encoder.fit_transform(df_train[cat_cols])\ndf_train[cat_cols] = df_train[cat_cols].astype('category')\n\ndf_test[cat_cols] = encoder.transform(df_test[cat_cols])\ndf_test[cat_cols] = df_test[cat_cols].astype('category')","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:32.419599Z","iopub.execute_input":"2024-10-17T14:32:32.419909Z","iopub.status.idle":"2024-10-17T14:32:32.457280Z","shell.execute_reply.started":"2024-10-17T14:32:32.419876Z","shell.execute_reply":"2024-10-17T14:32:32.456199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Optuna - Hyperparameter Tuning","metadata":{}},{"cell_type":"code","source":"# study = optuna.create_study(direction='maximize', study_name='Regressor')\n# study.optimize(lgb_objective, n_trials=30, show_progress_bar=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:32.458765Z","iopub.execute_input":"2024-10-17T14:32:32.459645Z","iopub.status.idle":"2024-10-17T14:32:32.463855Z","shell.execute_reply.started":"2024-10-17T14:32:32.459596Z","shell.execute_reply":"2024-10-17T14:32:32.462833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Tuned Hyperparameters","metadata":{}},{"cell_type":"code","source":"params = {\n    'objective'       : 'l2',\n    'verbosity'       : -1,\n    'n_iter'          : 200,\n    'lambda_l1'       : 0.005116829730239727,\n    'lambda_l2'       : 0.0011520776712645852,\n    'learning_rate'   : 0.02376367323636638,\n    'max_depth'       : 5,\n    'num_leaves'      : 207,\n    'colsample_bytree': 0.7759862336963801,\n    'colsample_bynode': 0.5110355095943208,\n    'bagging_fraction': 0.5485770314992224,\n    'bagging_freq'    : 7,\n    'min_data_in_leaf': 78,\n#     'lambda_l1': 0.0017491527148460124, \n#     'lambda_l2': 1.6284499076302, \n#     'learning_rate': 0.01987141965013073, \n#     'max_depth': 6, \n#     'num_leaves': 31, \n#     'colsample_bytree': 0.6946529271381489, \n#     'colsample_bynode': 0.8023106405158018, \n#     'bagging_fraction': 0.5395450118145588, \n#     'bagging_freq': 5, \n#     'min_data_in_leaf': 52\n}\n\nmodel = CustomLGBMRegressor(**params, random_state=SEED)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:32.465223Z","iopub.execute_input":"2024-10-17T14:32:32.465650Z","iopub.status.idle":"2024-10-17T14:32:32.474153Z","shell.execute_reply.started":"2024-10-17T14:32:32.465604Z","shell.execute_reply":"2024-10-17T14:32:32.473244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.feature_importances_","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:32.475354Z","iopub.execute_input":"2024-10-17T14:32:32.475719Z","iopub.status.idle":"2024-10-17T14:32:32.487080Z","shell.execute_reply.started":"2024-10-17T14:32:32.475685Z","shell.execute_reply":"2024-10-17T14:32:32.485970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cross Validation","metadata":{}},{"cell_type":"code","source":"X = df_train[feature_cols]\ny = df_train[target_col]\ncv = StratifiedKFold(5, shuffle=True, random_state=SEED)\n\nval_scores = cross_val_score(\n    model, X, y, cv=cv, \n    scoring=KAPPA_SCORER,\n)\n\nprint(f'kappa score: {np.mean(val_scores):.4f}')","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:32.488300Z","iopub.execute_input":"2024-10-17T14:32:32.488664Z","iopub.status.idle":"2024-10-17T14:32:36.567985Z","shell.execute_reply.started":"2024-10-17T14:32:32.488630Z","shell.execute_reply":"2024-10-17T14:32:36.566939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Seed Ensembling","metadata":{}},{"cell_type":"code","source":"model = CustomLGBMRegressor(**params, random_state=SEED)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:36.569458Z","iopub.execute_input":"2024-10-17T14:32:36.569914Z","iopub.status.idle":"2024-10-17T14:32:36.575131Z","shell.execute_reply.started":"2024-10-17T14:32:36.569866Z","shell.execute_reply":"2024-10-17T14:32:36.573979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training","metadata":{}},{"cell_type":"code","source":"X = df_train[feature_cols]\ny = df_train[target_col]\n\nmodel.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:36.576313Z","iopub.execute_input":"2024-10-17T14:32:36.576673Z","iopub.status.idle":"2024-10-17T14:32:37.518251Z","shell.execute_reply.started":"2024-10-17T14:32:36.576638Z","shell.execute_reply":"2024-10-17T14:32:37.517183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(feature_cols)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:37.519585Z","iopub.execute_input":"2024-10-17T14:32:37.519924Z","iopub.status.idle":"2024-10-17T14:32:37.526115Z","shell.execute_reply.started":"2024-10-17T14:32:37.519890Z","shell.execute_reply":"2024-10-17T14:32:37.525165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nwarnings.simplefilter(action=\"ignore\", category=FutureWarning)\n\nfeature_imp = pd.DataFrame(sorted(zip(model.feature_importances_, X)), columns=['Value', 'Feature'])\n\nfrom lightgbm import plot_importance\nfig,ax = plt.subplots(figsize=(10,8))\nplot_importance(model,max_num_features=200,ax=ax)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:37.527202Z","iopub.execute_input":"2024-10-17T14:32:37.527508Z","iopub.status.idle":"2024-10-17T14:32:39.014281Z","shell.execute_reply.started":"2024-10-17T14:32:37.527475Z","shell.execute_reply":"2024-10-17T14:32:39.013142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# unused_feature = []\n# for v, f in zip(feature_imp['Value'], feature_imp['Feature']):\n#     if v < 1:\n#         unused_feature.append(f)\n#         print(str(f), v)\n# print(unused_feature)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:39.018429Z","iopub.execute_input":"2024-10-17T14:32:39.018790Z","iopub.status.idle":"2024-10-17T14:32:39.023309Z","shell.execute_reply.started":"2024-10-17T14:32:39.018754Z","shell.execute_reply":"2024-10-17T14:32:39.022280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prediction","metadata":{}},{"cell_type":"code","source":"df_subm[target_col] = model.predict(df_test[feature_cols])\ndf_subm[target_col] = df_subm[target_col].round()\n\ndf_subm.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:39.024648Z","iopub.execute_input":"2024-10-17T14:32:39.024972Z","iopub.status.idle":"2024-10-17T14:32:39.052411Z","shell.execute_reply.started":"2024-10-17T14:32:39.024936Z","shell.execute_reply":"2024-10-17T14:32:39.051383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm","metadata":{"execution":{"iopub.status.busy":"2024-10-17T14:32:39.053619Z","iopub.execute_input":"2024-10-17T14:32:39.053937Z","iopub.status.idle":"2024-10-17T14:32:39.065566Z","shell.execute_reply.started":"2024-10-17T14:32:39.053903Z","shell.execute_reply":"2024-10-17T14:32:39.064504Z"},"trusted":true},"execution_count":null,"outputs":[]}]}