{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport random\nfrom tqdm import tqdm\nfrom pathlib import Path\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.metrics import make_scorer\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.impute import SimpleImputer\n\nfrom scipy.optimize import minimize\nimport optuna\n\nimport lightgbm as lgb\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\n\nSEED = 42\n\nKAPPA_SCORER = make_scorer(\n    cohen_kappa_score, \n    greater_is_better=True, \n    weights='quadratic',\n)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:27:15.230438Z","iopub.execute_input":"2024-10-16T07:27:15.231025Z","iopub.status.idle":"2024-10-16T07:27:18.786995Z","shell.execute_reply.started":"2024-10-16T07:27:15.230954Z","shell.execute_reply":"2024-10-16T07:27:18.785673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lgb_objective(trial):\n    params = {\n        'objective':         'l2',\n        'verbosity':         -1,\n        'n_iter':            200,\n        'random_state':      SEED,\n        'boosting_type':     'gbdt',\n        'lambda_l1':         trial.suggest_float('lambda_l1', 1e-3, 10.0, log=True),\n        'lambda_l2':         trial.suggest_float('lambda_l2', 1e-3, 10.0, log=True),\n        'learning_rate':     trial.suggest_float('learning_rate', 1e-2, 1e-1, log=True),\n        'max_depth':         trial.suggest_int('max_depth', 4, 8),\n        'num_leaves':        trial.suggest_int('num_leaves', 16, 256),\n        'colsample_bytree':  trial.suggest_float('colsample_bytree', 0.4, 1.0),\n        'colsample_bynode':  trial.suggest_float('colsample_bynode', 0.4, 1.0),\n        'bagging_fraction':  trial.suggest_float('bagging_fraction', 0.4, 1.0),\n        'bagging_freq':      trial.suggest_int('bagging_freq', 1, 7),\n        'min_data_in_leaf':  trial.suggest_int('min_data_in_leaf', 5, 100),\n    }\n    \n    X = df_train[feature_cols]\n    y = df_train[target_col]\n    cv = StratifiedKFold(5, shuffle=True, random_state=SEED)\n    estimator = CustomLGBMRegressor(**params)\n\n    val_scores = cross_val_score(\n        estimator=estimator, \n        X=X, y=y, \n        cv=cv, \n        scoring=KAPPA_SCORER,\n    )\n\n    return np.mean(val_scores)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:27:18.789439Z","iopub.execute_input":"2024-10-16T07:27:18.790136Z","iopub.status.idle":"2024-10-16T07:27:18.802890Z","shell.execute_reply.started":"2024-10-16T07:27:18.790084Z","shell.execute_reply":"2024-10-16T07:27:18.801618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    \n    return df.describe().values.reshape(-1), filename.split('=')[1]","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:27:18.804694Z","iopub.execute_input":"2024-10-16T07:27:18.805184Z","iopub.status.idle":"2024-10-16T07:27:18.823352Z","shell.execute_reply.started":"2024-10-16T07:27:18.805121Z","shell.execute_reply":"2024-10-16T07:27:18.821815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_time_series(dirname):\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:27:18.826531Z","iopub.execute_input":"2024-10-16T07:27:18.827188Z","iopub.status.idle":"2024-10-16T07:27:18.836300Z","shell.execute_reply.started":"2024-10-16T07:27:18.827119Z","shell.execute_reply":"2024-10-16T07:27:18.835041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def quadratic_weighted_kappa(estimator, X, y_true):\n    y_pred = estimator.predict(X).round()\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:27:18.838070Z","iopub.execute_input":"2024-10-16T07:27:18.838662Z","iopub.status.idle":"2024-10-16T07:27:18.848942Z","shell.execute_reply.started":"2024-10-16T07:27:18.838601Z","shell.execute_reply":"2024-10-16T07:27:18.847381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def threshold_rounder(y_pred, thresholds):\n    return np.where(y_pred < thresholds[0], 0,\n                    np.where(y_pred < thresholds[1], 1,\n                             np.where(y_pred < thresholds[2], 2, 3)))","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:27:18.850783Z","iopub.execute_input":"2024-10-16T07:27:18.851330Z","iopub.status.idle":"2024-10-16T07:27:18.865977Z","shell.execute_reply.started":"2024-10-16T07:27:18.851269Z","shell.execute_reply":"2024-10-16T07:27:18.864645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def eval_preds(thresholds, y_true, y_pred):\n    y_pred = threshold_rounder(y_pred, thresholds)\n    score = cohen_kappa_score(y_true, y_pred, weights='quadratic')\n    return -score","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:27:18.867610Z","iopub.execute_input":"2024-10-16T07:27:18.868167Z","iopub.status.idle":"2024-10-16T07:27:18.877972Z","shell.execute_reply.started":"2024-10-16T07:27:18.868107Z","shell.execute_reply":"2024-10-16T07:27:18.876599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomLGBMRegressor(lgb.LGBMRegressor):\n    '''\n    Custom LightGBM Regressor\n    \n    It optimizes threshold values during fitting.\n    Main goal is preventing overfit on validation data.\n    '''\n    def fit(self, X, y, **kwargs):\n        super().fit(X, y, **kwargs)\n        y_pred = super().predict(X, **kwargs)\n        \n        self.optimizer = minimize(\n            eval_preds, \n            x0=[0.5, 1.5, 2.5], \n            args=(y, y_pred), \n            method='Nelder-Mead',\n        )\n        \n    def predict(self, X, **kwargs):\n        y_pred = super().predict(X, **kwargs)\n        y_pred = threshold_rounder(y_pred, self.optimizer.x)\n        return y_pred","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:27:18.879633Z","iopub.execute_input":"2024-10-16T07:27:18.880181Z","iopub.status.idle":"2024-10-16T07:27:18.890269Z","shell.execute_reply.started":"2024-10-16T07:27:18.880127Z","shell.execute_reply":"2024-10-16T07:27:18.888902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root = Path('/kaggle/input/child-mind-institute-problematic-internet-use')","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:27:18.891892Z","iopub.execute_input":"2024-10-16T07:27:18.892354Z","iopub.status.idle":"2024-10-16T07:27:18.903937Z","shell.execute_reply.started":"2024-10-16T07:27:18.892308Z","shell.execute_reply":"2024-10-16T07:27:18.902369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Tabular Data","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(root / 'train.csv')\ndf_test = pd.read_csv(root / 'test.csv')\ndf_subm = pd.read_csv(root / 'sample_submission.csv', index_col='id')","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:27:18.907863Z","iopub.execute_input":"2024-10-16T07:27:18.908996Z","iopub.status.idle":"2024-10-16T07:27:19.026421Z","shell.execute_reply.started":"2024-10-16T07:27:18.908933Z","shell.execute_reply":"2024-10-16T07:27:19.024777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Time Series Data","metadata":{}},{"cell_type":"code","source":"ts_train = load_time_series(root / \"series_train.parquet\")\nts_test = load_time_series(root / \"series_test.parquet\")\n\ntime_series_cols = ts_train.columns.tolist()\ntime_series_cols.remove(\"id\")","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:27:19.028453Z","iopub.execute_input":"2024-10-16T07:27:19.029161Z","iopub.status.idle":"2024-10-16T07:29:17.587287Z","shell.execute_reply.started":"2024-10-16T07:27:19.028982Z","shell.execute_reply":"2024-10-16T07:29:17.585922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Merge Operation","metadata":{}},{"cell_type":"code","source":"df_train = pd.merge(df_train, ts_train, how=\"left\", on='id')\ndf_test = pd.merge(df_test, ts_test, how=\"left\", on='id')\n\ndf_train = df_train.set_index('id')\ndf_test = df_test.set_index('id')","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:29:17.589148Z","iopub.execute_input":"2024-10-16T07:29:17.589584Z","iopub.status.idle":"2024-10-16T07:29:17.656195Z","shell.execute_reply.started":"2024-10-16T07:29:17.589533Z","shell.execute_reply":"2024-10-16T07:29:17.654640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Global Variables","metadata":{}},{"cell_type":"code","source":"cat_cols = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\nnum_cols = ['Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-CGAS_Score', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference', 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-computerinternet_hoursday']\ntabular_cols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference', 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season', 'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-Season', 'PreInt_EduHx-computerinternet_hoursday']\ntarget_col = 'sii'\n\nfeature_cols = tabular_cols + time_series_cols\nnum_cols = num_cols + time_series_cols","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:29:17.657860Z","iopub.execute_input":"2024-10-16T07:29:17.658304Z","iopub.status.idle":"2024-10-16T07:29:17.671294Z","shell.execute_reply.started":"2024-10-16T07:29:17.658262Z","shell.execute_reply":"2024-10-16T07:29:17.669909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Drop Rows with Missing Targets","metadata":{}},{"cell_type":"code","source":"df_train = df_train.dropna(subset=[target_col])","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:29:17.673406Z","iopub.execute_input":"2024-10-16T07:29:17.673991Z","iopub.status.idle":"2024-10-16T07:29:17.696760Z","shell.execute_reply.started":"2024-10-16T07:29:17.673937Z","shell.execute_reply":"2024-10-16T07:29:17.695279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Numeric Value Imputing","metadata":{}},{"cell_type":"code","source":"imputer = SimpleImputer(\n    strategy='mean',\n)\n\ndf_train[num_cols] = imputer.fit_transform(df_train[num_cols])\ndf_test[num_cols] = imputer.transform(df_test[num_cols])","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:29:17.701822Z","iopub.execute_input":"2024-10-16T07:29:17.702977Z","iopub.status.idle":"2024-10-16T07:29:17.796727Z","shell.execute_reply.started":"2024-10-16T07:29:17.702920Z","shell.execute_reply":"2024-10-16T07:29:17.795314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Category Encoding","metadata":{}},{"cell_type":"code","source":"encoder = OrdinalEncoder(\n    dtype=np.int32,\n    handle_unknown='use_encoded_value',\n    unknown_value=-1,\n    encoded_missing_value=-2,\n)\n\ndf_train[cat_cols] = encoder.fit_transform(df_train[cat_cols])\ndf_train[cat_cols] = df_train[cat_cols].astype('category')\n\ndf_test[cat_cols] = encoder.transform(df_test[cat_cols])\ndf_test[cat_cols] = df_test[cat_cols].astype('category')","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:29:17.798451Z","iopub.execute_input":"2024-10-16T07:29:17.798938Z","iopub.status.idle":"2024-10-16T07:29:17.851371Z","shell.execute_reply.started":"2024-10-16T07:29:17.798890Z","shell.execute_reply":"2024-10-16T07:29:17.850179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Optuna - Hyperparameter Tuning","metadata":{}},{"cell_type":"code","source":"# study = optuna.create_study(direction='maximize', study_name='Regressor')\n# study.optimize(lgb_objective, n_trials=30, show_progress_bar=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:29:17.852851Z","iopub.execute_input":"2024-10-16T07:29:17.853251Z","iopub.status.idle":"2024-10-16T07:29:17.858245Z","shell.execute_reply.started":"2024-10-16T07:29:17.853199Z","shell.execute_reply":"2024-10-16T07:29:17.857080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Tuned Hyperparameters","metadata":{}},{"cell_type":"code","source":"params = {\n    'objective'       : 'l2',\n    'verbosity'       : -1,\n    'n_iter'          : 200,\n    'lambda_l1'       : 0.005116829730239727,\n    'lambda_l2'       : 0.0011520776712645852,\n    'learning_rate'   : 0.02376367323636638,\n    'max_depth'       : 5,\n    'num_leaves'      : 207,\n    'colsample_bytree': 0.7759862336963801,\n    'colsample_bynode': 0.5110355095943208,\n    'bagging_fraction': 0.5485770314992224,\n    'bagging_freq'    : 7,\n    'min_data_in_leaf': 78,\n}\n\nmodel = CustomLGBMRegressor(**params, random_state=SEED)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:29:17.860323Z","iopub.execute_input":"2024-10-16T07:29:17.860779Z","iopub.status.idle":"2024-10-16T07:29:17.877074Z","shell.execute_reply.started":"2024-10-16T07:29:17.860735Z","shell.execute_reply":"2024-10-16T07:29:17.875615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cross Validation","metadata":{}},{"cell_type":"code","source":"X = df_train[feature_cols]\ny = df_train[target_col]\ncv = StratifiedKFold(5, shuffle=True, random_state=SEED)\n\nval_scores = cross_val_score(\n    model, X, y, cv=cv, \n    scoring=KAPPA_SCORER,\n)\n\nprint(f'kappa score: {np.mean(val_scores):.4f}')","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:29:17.878697Z","iopub.execute_input":"2024-10-16T07:29:17.879183Z","iopub.status.idle":"2024-10-16T07:29:26.731207Z","shell.execute_reply.started":"2024-10-16T07:29:17.879138Z","shell.execute_reply":"2024-10-16T07:29:26.729824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Seed Ensembling","metadata":{}},{"cell_type":"code","source":"model = VotingRegressor([\n    ('lgb_0', CustomLGBMRegressor(**params, random_state=12)),\n    ('lgb_1', CustomLGBMRegressor(**params, random_state=22)),\n    ('lgb_2', CustomLGBMRegressor(**params, random_state=32)),\n    ('lgb_3', CustomLGBMRegressor(**params, random_state=42)),\n    ('lgb_4', CustomLGBMRegressor(**params, random_state=52)),\n    ('lgb_5', CustomLGBMRegressor(**params, random_state=62)),\n    ('lgb_6', CustomLGBMRegressor(**params, random_state=72)),\n    ('lgb_7', CustomLGBMRegressor(**params, random_state=82)),\n    ('lgb_8', CustomLGBMRegressor(**params, random_state=92)),\n    ('lgb_9', CustomLGBMRegressor(**params, random_state=102)),\n])","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:30:00.474873Z","iopub.execute_input":"2024-10-16T07:30:00.477158Z","iopub.status.idle":"2024-10-16T07:30:00.495811Z","shell.execute_reply.started":"2024-10-16T07:30:00.477069Z","shell.execute_reply":"2024-10-16T07:30:00.494150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training","metadata":{}},{"cell_type":"code","source":"X = df_train[feature_cols]\ny = df_train[target_col]\n\nmodel.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:30:02.422498Z","iopub.execute_input":"2024-10-16T07:30:02.423097Z","iopub.status.idle":"2024-10-16T07:30:17.612383Z","shell.execute_reply.started":"2024-10-16T07:30:02.423047Z","shell.execute_reply":"2024-10-16T07:30:17.611098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Prediction","metadata":{}},{"cell_type":"code","source":"df_subm[target_col] = model.predict(df_test[feature_cols])\ndf_subm[target_col] = df_subm[target_col].round()\n\ndf_subm.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-16T07:30:17.614590Z","iopub.execute_input":"2024-10-16T07:30:17.615084Z","iopub.status.idle":"2024-10-16T07:30:17.769415Z","shell.execute_reply.started":"2024-10-16T07:30:17.615021Z","shell.execute_reply":"2024-10-16T07:30:17.768054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}