{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport polars as pl\nimport re\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\nimport lightgbm\nfrom sklearn.metrics import cohen_kappa_score\nfrom colorama import Fore, Style\nfrom scipy.optimize import minimize\nfrom sklearn.metrics import ConfusionMatrixDisplay  \nfrom sklearn.ensemble import VotingClassifier\nfrom scipy.stats import mode\nimport scipy.special\nfrom catboost import CatBoostRegressor \nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport seaborn as sns\nfrom IPython.display import clear_output","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:10:35.158462Z","iopub.execute_input":"2024-12-16T13:10:35.158976Z","iopub.status.idle":"2024-12-16T13:10:38.515094Z","shell.execute_reply.started":"2024-12-16T13:10:35.158931Z","shell.execute_reply":"2024-12-16T13:10:38.513883Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Preparation","metadata":{}},{"cell_type":"code","source":"train_csv = pl.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_csv = pl.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\n# One-Hot Encoding \nseason_columns = [col for col in train_csv.columns if re.search(r'Season$', col)]\n\ntrain_csv = train_csv.to_dummies(columns=season_columns)\ntest_csv = test_csv.to_dummies(columns=season_columns)\n\nall_seasons = ['Spring', 'Summer', 'Fall', 'Winter']\n\nfor season_col in season_columns:\n    for season in all_seasons:\n        column_name = f'{season_col}_{season}'\n        if column_name not in train_csv.columns:\n            train_csv = train_csv.with_columns([pl.Series(name=column_name, values=[0] * train_csv.height).cast(pl.Int8())])\n        if column_name not in test_csv.columns and season_col != 'PCIAT-Season':\n            test_csv = test_csv.with_columns([pl.Series(name=column_name, values=[0] * test_csv.height).cast(pl.Int8())])\n            \ntrain_csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:10:38.517233Z","iopub.execute_input":"2024-12-16T13:10:38.517940Z","iopub.status.idle":"2024-12-16T13:10:38.702479Z","shell.execute_reply.started":"2024-12-16T13:10:38.517894Z","shell.execute_reply":"2024-12-16T13:10:38.701349Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Parquet Data","metadata":{}},{"cell_type":"markdown","source":"Reference: https://www.kaggle.com/code/taimour/h2o-automated-machine-learning-piu/notebook ","metadata":{}},{"cell_type":"code","source":"# Function to process individual file\ndef process_file(filename, dirname):\n    df = pl.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df = df.drop('step')\n    stats_values = df.describe().drop('statistic').to_numpy().reshape(-1)\n    return stats_values, filename.split('=')[1]\n\n# Function to load time series from a directory\ndef load_time_series(dirname) -> pl.DataFrame:\n    ids = os.listdir(dirname)\n    \n    feature_name = ['X', 'Y', 'Z', 'enmo', 'anglez', 'non-wear_flag', 'light', 'battery_voltage', 'time_of_day', 'weekday', 'quarter', 'relative_date_PCIAT']\n    stats_name = ['count', 'null_count', 'mean', 'std', 'min', '25%', '50%', '75%', 'max']\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    # Create a DataFrame with more meaningful column names based on describe output\n    column_names = []\n    for fea in feature_name:\n        column_names.extend([f\"{fea}_{stat}\" for stat in stats_name])\n    \n    df = pl.DataFrame({column_names[i]: [stat[i] for stat in stats] for i in range(len(stats[0]))})\n    df = df.with_columns(pl.Series('id', indexes))\n    \n    return df\n\n# Load train and test time series data using polars\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\n# Get time series columns and remove 'id'\ntime_series_cols = train_ts.columns\nif \"id\" in time_series_cols:\n    time_series_cols.remove(\"id\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:10:38.703925Z","iopub.execute_input":"2024-12-16T13:10:38.704371Z","iopub.status.idle":"2024-12-16T13:11:57.587130Z","shell.execute_reply.started":"2024-12-16T13:10:38.704319Z","shell.execute_reply":"2024-12-16T13:11:57.585838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Trim Parquet files\n# enmo_cols = ['enmo_mean', 'enmo_std', 'enmo_min', 'enmo_25%', 'enmo_50%', 'enmo_75%', 'enmo_max']\n# light_cols = ['light_mean', 'light_std', 'light_min', 'light_25%', 'light_50%', 'light_75%', 'light_max']\n\nenmo_cols = ['enmo_mean', 'enmo_std', 'enmo_max']\nlight_cols = ['light_mean', 'light_std', 'light_max']\n\ntrain_ts_final = train_ts.select(enmo_cols + light_cols + ['id'])\ntest_ts_final = test_ts.select(enmo_cols + light_cols + ['id'])\n\ntrain_ts_final","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:11:57.589349Z","iopub.execute_input":"2024-12-16T13:11:57.589734Z","iopub.status.idle":"2024-12-16T13:11:57.600356Z","shell.execute_reply.started":"2024-12-16T13:11:57.589696Z","shell.execute_reply":"2024-12-16T13:11:57.599056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Merge CSV & Parquet\ntrain = train_csv.join(train_ts_final, on=\"id\", how=\"left\")\ntest = test_csv.join(test_ts_final, on=\"id\", how=\"left\")\n\n# pl.Config.restore_defaults()\ntrain","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:11:57.602015Z","iopub.execute_input":"2024-12-16T13:11:57.602412Z","iopub.status.idle":"2024-12-16T13:11:57.648179Z","shell.execute_reply.started":"2024-12-16T13:11:57.602373Z","shell.execute_reply":"2024-12-16T13:11:57.647042Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Advanced: Data Cleaning","metadata":{}},{"cell_type":"code","source":"def preprocess_data(df, train_data=False):\n    # Fill missing values with median\n    df = df.with_columns(pl.all().exclude(\"id\").fill_null(pl.all().exclude(\"id\").median())) \n    \n    return df\n    \n# train = preprocess_data(train)\n# test = preprocess_data(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:11:57.650136Z","iopub.execute_input":"2024-12-16T13:11:57.650631Z","iopub.status.idle":"2024-12-16T13:11:57.656870Z","shell.execute_reply.started":"2024-12-16T13:11:57.650573Z","shell.execute_reply":"2024-12-16T13:11:57.655589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:11:57.658126Z","iopub.execute_input":"2024-12-16T13:11:57.658453Z","iopub.status.idle":"2024-12-16T13:11:57.672648Z","shell.execute_reply.started":"2024-12-16T13:11:57.658420Z","shell.execute_reply":"2024-12-16T13:11:57.671441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def optimized_feature_engineering(df):\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n\n    df['LST_TBW'] = df['BIA-BIA_LST'] / (df['BIA-BIA_TBW'] + 1e-8)\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / (df['Physical-Weight'] + 1e-8)\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / (df['Physical-Weight'] + 1e-8)\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / (df['Physical-Height'] + 1e-8)\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / (df['BIA-BIA_FMI'] + 1e-8)\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / (df['Physical-Weight'] + 1e-8)\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / (df['BIA-BIA_TBW'] + 1e-8)\n    df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:11:57.674039Z","iopub.execute_input":"2024-12-16T13:11:57.674415Z","iopub.status.idle":"2024-12-16T13:11:57.690735Z","shell.execute_reply.started":"2024-12-16T13:11:57.674375Z","shell.execute_reply":"2024-12-16T13:11:57.689597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.to_pandas()\ntest = test.to_pandas()\n\ntrain = feature_engineering(train)\ntest = feature_engineering(test)\n\ntrain","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:11:57.692488Z","iopub.execute_input":"2024-12-16T13:11:57.693349Z","iopub.status.idle":"2024-12-16T13:11:57.792982Z","shell.execute_reply.started":"2024-12-16T13:11:57.693292Z","shell.execute_reply":"2024-12-16T13:11:57.791910Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.replace([np.inf, -np.inf], np.nan)\ntest = test.replace([np.inf, -np.inf], np.nan)\n\ntrain","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:11:57.796074Z","iopub.execute_input":"2024-12-16T13:11:57.796435Z","iopub.status.idle":"2024-12-16T13:11:57.850566Z","shell.execute_reply.started":"2024-12-16T13:11:57.796396Z","shell.execute_reply":"2024-12-16T13:11:57.849393Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model: LGBMRegressor","metadata":{}},{"cell_type":"markdown","source":"#### Int Metrics","metadata":{}},{"cell_type":"code","source":"# Convert PCIAT-PCIAT_Total to Severity Impairment Index (sii) based on predefined thresholds\ndef pciat_to_sii(pciat_total):\n    if 80 <= pciat_total:\n        return 3  # Severe\n    elif 50 <= pciat_total < 80:\n        return 2  # Moderate\n    elif 30 < pciat_total < 50:\n        return 1  # Mild\n    else:   # pciat_total <= 30:\n        return 0  # None\n    \n# sii_values = np.vectorize(pciat_to_sii)(pciat_totals)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:11:57.852148Z","iopub.execute_input":"2024-12-16T13:11:57.852615Z","iopub.status.idle":"2024-12-16T13:11:57.859375Z","shell.execute_reply.started":"2024-12-16T13:11:57.852563Z","shell.execute_reply":"2024-12-16T13:11:57.858081Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Float Metrics","metadata":{}},{"cell_type":"code","source":"# Convert PCIAT-PCIAT_Total to Severity Impairment Index (sii) with fractional part based on predefined thresholds\ndef pciat_to_sii_fractional(pciat_total):\n    if 80 <= pciat_total:\n        return 3 + ((pciat_total - 80) / 20)  # Severe\n    elif 50 <= pciat_total < 80:\n        return 2 + ((pciat_total - 50) / 30) # Moderate\n    elif 30 < pciat_total < 50:\n        return 1 + ((pciat_total - 30) / 20) # Mild\n    else:   # pciat_total <= 30:\n        return 0 + (pciat_total / 30) # None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:11:57.861080Z","iopub.execute_input":"2024-12-16T13:11:57.861450Z","iopub.status.idle":"2024-12-16T13:11:57.874887Z","shell.execute_reply.started":"2024-12-16T13:11:57.861410Z","shell.execute_reply":"2024-12-16T13:11:57.873368Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Target: PCIAT-PCIAT_01 ~ 20","metadata":{}},{"cell_type":"code","source":"# supervised_usable_new_total = (\n#     train\n#     .filter(\n#         (pl.col('PCIAT-PCIAT_Total').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_01').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_02').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_03').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_04').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_05').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_06').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_07').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_08').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_09').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_10').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_11').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_12').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_13').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_14').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_15').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_16').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_17').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_18').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_19').is_not_null()) &\n#         (pl.col('PCIAT-PCIAT_20').is_not_null())\n#     )\n# )\n\n# ids_in_total = supervised_usable_new_total.select(\"id\")\n\n# # Initialize arrays to store out-of-foldnd models for each PCIAT-PCIAT_01 to PCIAT-PCIAT_20\n# oof_raw_total = np.zeros(len(supervised_usable_new_total), dtype=float)\n# models_total = []\n\n# # Loop over each PCIAT-PCIAT_01 to PCIAT-PCIAT_20\n# for i in range(1, 21):\n#     pciat_column = f'PCIAT-PCIAT_{i:02}'\n\n#     supervised_usable_new = (\n#         train\n#         .filter(\n#             (pl.col(pciat_column).is_not_null())\n#         )\n#     )\n\n#     # Define target and features for the specific PCIAT column\n#     y = supervised_usable_new.get_column(pciat_column)\n#     X = supervised_usable_new.drop('id', 'sii', '^PCIAT.*$').to_pandas()\n\n#     # Initialize out-of-fold predictions for the current PCIAT column\n#     id_column = supervised_usable_new.select(\"id\").to_series()\n#     y_column = np.zeros(len(y), dtype=float)\n#     models = []\n\n#     # Perform Stratified K-Fold cross-validation\n#     kf = StratifiedKFold(shuffle=True, random_state=1)\n#     for fold, (idx_tr, idx_va) in enumerate(kf.split(X, y)):\n#         X_tr = X.iloc[idx_tr]\n#         X_va = X.iloc[idx_va]\n#         y_tr = y[idx_tr]\n#         y_va = y[idx_va]\n\n#         model = lightgbm.LGBMRegressor(verbose=-1)\n#         # model = CatBoostRegressor() # better performance, but took much longer time\n#         model.fit(X_tr, y_tr.to_numpy())\n#         models.append(model)\n\n#         # Predict and store out-of-fold predictions\n#         y_pred = model.predict(X_va)\n#         y_column[idx_va] = y_pred\n \n#         # Calculate score for the current fold\n#         y_pred_rounded = y_pred.round(0).astype(int)\n#         score = cohen_kappa_score(y_va, y_pred_rounded, weights='quadratic')\n#         print(f\"# Fold {fold} for {pciat_column}: {score=:.3f}\")\n\n#     # Store the models and out-of-fold predictions\n#     models_total.append(models)\n#     oof_raw = pl.DataFrame({\n#         \"id\": id_column,\n#         \"y\": y_column\n#     })\n#     oof_raw = oof_raw.filter(\n#         pl.col(\"id\").is_in(ids_in_total[\"id\"])\n#     )\n#     oof_raw = oof_raw.select(\"y\").to_numpy().flatten()\n#     oof_raw_total += oof_raw\n#     print(oof_raw_total)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:11:57.876452Z","iopub.execute_input":"2024-12-16T13:11:57.876907Z","iopub.status.idle":"2024-12-16T13:11:57.896084Z","shell.execute_reply.started":"2024-12-16T13:11:57.876866Z","shell.execute_reply":"2024-12-16T13:11:57.894804Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filter out rows where PCIAT-PCIAT_Total and all PCIAT-PCIAT_01 to PCIAT-PCIAT_20 are not null\nPCIAT_Score_Cols = ['PCIAT-PCIAT_Total'] + [f'PCIAT-PCIAT_{str(i).zfill(2)}' for i in range(1, 21)]\ntrain_df = train.dropna(subset=PCIAT_Score_Cols)\n\n# Initialize arrays to store out-of-fold predictions and models for each PCIAT-PCIAT_01 to PCIAT-PCIAT_20\noof_raw_total = np.zeros(len(train_df), dtype=float)\nmodels_total = []\n\n# Loop over each PCIAT-PCIAT_01 to PCIAT-PCIAT_20\nfor i in range(1, 21):\n    pciat_column = f'PCIAT-PCIAT_{i:02}'\n\n    # Define target and features for the specific PCIAT column\n    y = train_df[pciat_column]\n    X = train_df.drop(columns=['id', 'sii'] + PCIAT_Score_Cols)\n    # X = train_df.drop(columns=['id', 'sii', 'PCIAT-Season_Spring', 'PCIAT-Season_Fall', \\\n    #                            'PCIAT-Season_Summer' ,'PCIAT-Season_Winter', 'PCIAT-Season_null'] + PCIAT_Score_Cols)\n\n    # Initialize out-of-fold predictions for the current PCIAT column\n    oof_raw = np.zeros(len(y), dtype=float)\n    models = []\n\n    # Perform Stratified K-Fold cross-validation\n    kf = StratifiedKFold(shuffle=True, random_state=1)\n    for fold, (idx_tr, idx_va) in enumerate(kf.split(X, y)):\n        X_tr = X.iloc[idx_tr]\n        X_va = X.iloc[idx_va]\n        y_tr = y.iloc[idx_tr]\n        y_va = y.iloc[idx_va]\n        \n        model = lightgbm.LGBMRegressor(verbose=-1)\n        # model = CatBoostRegressor(cat_features=season_columns) # better performance, but took much longer time\n        model.fit(X_tr, y_tr.to_numpy())\n        models.append(model)\n\n        # Predict and store out-of-fold predictions\n        y_pred = model.predict(X_va)\n        oof_raw[idx_va] = y_pred\n\n        # Calculate score for the current fold\n        y_pred_rounded = y_pred.round(0).astype(int)\n        score = cohen_kappa_score(y_va, y_pred_rounded, weights='quadratic')\n\n        print(f\"# Fold {fold} for {pciat_column}: {score=:.3f}\")\n\n    # Store the models and out-of-fold predictions\n    models_total.append(models)\n    oof_raw_total += oof_raw","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:13:09.960394Z","iopub.execute_input":"2024-12-16T13:13:09.960925Z","iopub.status.idle":"2024-12-16T13:14:20.127551Z","shell.execute_reply.started":"2024-12-16T13:13:09.960879Z","shell.execute_reply":"2024-12-16T13:14:20.126201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compare oof_raw_total with the original PCIAT-PCIAT_Total and calculate the Cohen Kappa Score\ny_true = train_df['PCIAT-PCIAT_Total'].to_numpy()\ny_pred_rounded = oof_raw_total.round(0).astype(int)\n\nscore = cohen_kappa_score(y_true, y_pred_rounded, weights='quadratic')\nprint(f\"{Fore.GREEN}{Style.BRIGHT}# Overall Cohen Kappa Score for PCIAT-PCIAT_Total: {score=:.3f}{Style.RESET_ALL}\")\n\ny_true_sii = train_df['sii'].to_numpy()\ny_pred_to_sii = np.vectorize(pciat_to_sii)(oof_raw_total)\n\nscore_sii = cohen_kappa_score(y_true_sii, y_pred_to_sii, weights='quadratic')\nprint(f\"{Fore.GREEN}{Style.BRIGHT}# Overall Cohen Kappa Score for sii: {score_sii=:.3f}{Style.RESET_ALL}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:14:20.129546Z","iopub.execute_input":"2024-12-16T13:14:20.129912Z","iopub.status.idle":"2024-12-16T13:14:20.149429Z","shell.execute_reply.started":"2024-12-16T13:14:20.129875Z","shell.execute_reply":"2024-12-16T13:14:20.148256Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Tuning Thresholds","metadata":{}},{"cell_type":"code","source":"def round_with_thresholds(raw_preds, thresholds):\n    return np.where(raw_preds < thresholds[0], 0,\n                    np.where(raw_preds < thresholds[1], 1,\n                             np.where(raw_preds < thresholds[2], 2, 3)))\n\ndef fun(thresholds, y_true, raw_preds):\n    rounded_preds = round_with_thresholds(raw_preds, thresholds)\n    return - cohen_kappa_score(y_true, rounded_preds, weights='quadratic')\n\n# Determine the thresholds which give the highest quadratic kappa score\noof_raw_total_new = np.vectorize(pciat_to_sii_fractional)(oof_raw_total)\nres = minimize(fun, x0=[0.5, 1.5, 2.5], args=(y_true_sii, oof_raw_total_new), method='Nelder-Mead')\nassert res.success\noof_tuned = round_with_thresholds(oof_raw_total_new, res.x)\nprint(f\"# Optimized thresholds: {res.x.round(2)}\")\nprint(f\"# Score with default rounding:     {cohen_kappa_score(y_true_sii, y_pred_to_sii, weights='quadratic'):.3f}\")\nprint(f\"# Score with optimized thresholds: {cohen_kappa_score(y_true_sii, oof_tuned, weights='quadratic'):.3f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:14:20.150492Z","iopub.execute_input":"2024-12-16T13:14:20.150832Z","iopub.status.idle":"2024-12-16T13:14:20.502933Z","shell.execute_reply.started":"2024-12-16T13:14:20.150800Z","shell.execute_reply":"2024-12-16T13:14:20.501598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_labels = ['None', 'Mild', 'Moderate', 'Severe']\nregression_labels = [f\"{i} = {target_labels[i]}\" for i in range(4)]\n\nConfusionMatrixDisplay.from_predictions(y_true_sii, oof_tuned)\nplt.title('Confusion matrix with tuned thresholds')\nplt.xticks(np.arange(4), regression_labels)\nplt.yticks(np.arange(4), regression_labels)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:14:20.504425Z","iopub.execute_input":"2024-12-16T13:14:20.504891Z","iopub.status.idle":"2024-12-16T13:14:20.861669Z","shell.execute_reply.started":"2024-12-16T13:14:20.504840Z","shell.execute_reply":"2024-12-16T13:14:20.860466Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Test: Hard-Voting Ensemble","metadata":{}},{"cell_type":"code","source":"# Initialize arrays to store out-of-foldnd models for each PCIAT-PCIAT_01 to PCIAT-PCIAT_20\nmodels_total = []\n\nPCIAT_Score_Cols = ['PCIAT-PCIAT_Total'] + [f'PCIAT-PCIAT_{str(i).zfill(2)}' for i in range(1, 21)]\n\n# Loop over each PCIAT-PCIAT_01 to PCIAT-PCIAT_20\nfor i in range(1, 21):\n    pciat_column = f'PCIAT-PCIAT_{i:02}'\n\n    train_df = train.dropna(subset=[pciat_column])\n\n    # Define target and features for the specific PCIAT column\n    y = train_df[pciat_column]\n    X = train_df.drop(columns=['id', 'sii'] + PCIAT_Score_Cols)\n    # X = train_df.drop(columns=['id', 'sii', 'PCIAT-Season_Spring', 'PCIAT-Season_Fall', \\\n    #                            'PCIAT-Season_Summer' ,'PCIAT-Season_Winter', 'PCIAT-Season_null'] + PCIAT_Score_Cols)\n\n    # Initialize out-of-fold predictions for the current PCIAT column\n    oof_raw = np.zeros(len(y), dtype=float)\n    models = []\n\n    # Perform Stratified K-Fold cross-validation\n    kf = StratifiedKFold(shuffle=True, random_state=1)\n    for fold, (idx_tr, idx_va) in enumerate(kf.split(X, y)):\n        X_tr = X.iloc[idx_tr]\n        X_va = X.iloc[idx_va]\n        y_tr = y.iloc[idx_tr]\n        y_va = y.iloc[idx_va]\n\n        model = lightgbm.LGBMRegressor(verbose=-1)\n        # model = CatBoostRegressor(cat_features=season_columns) # better performance, but took much longer time\n        model.fit(X_tr, y_tr.to_numpy())\n        models.append(model)\n\n        # Predict and store out-of-fold predictions\n        y_pred = model.predict(X_va)\n        oof_raw[idx_va] = y_pred\n\n        # Calculate score for the current fold\n        y_pred_rounded = y_pred.round(0).astype(int)\n        score = cohen_kappa_score(y_va, y_pred_rounded, weights='quadratic')\n        \n        print(f\"# Fold {fold} for {pciat_column}: {score=:.3f}\")\n\n    # Store the models and out-of-fold predictions\n    models_total.append(models)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:14:20.864295Z","iopub.execute_input":"2024-12-16T13:14:20.864680Z","iopub.status.idle":"2024-12-16T13:15:31.583570Z","shell.execute_reply.started":"2024-12-16T13:14:20.864642Z","shell.execute_reply":"2024-12-16T13:15:31.582248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# lightgbm.plot_importance(models_total[0][0], importance_type=\"split\", figsize=(10, 20), title=\"LightGBM Feature Importance (Split)\")\n# plt.savefig('fi_split_model[0][0].png', dpi=300, bbox_inches='tight')\n\n# lightgbm.plot_importance(models_total[0][0], importance_type=\"gain\", figsize=(10, 20), title=\"LightGBM Feature Importance (Gain)\")\n# plt.savefig('fi_gain_model[0][0].png', dpi=300, bbox_inches='tight')\n\n# lightgbm.plot_importance(models_total[7][2], importance_type=\"split\", figsize=(10, 20), title=\"LightGBM Feature Importance (Split)\")\n# plt.savefig('fi_split_model[7][2].png', dpi=300, bbox_inches='tight')\n\n# lightgbm.plot_importance(models_total[7][2], importance_type=\"gain\", figsize=(10, 20), title=\"LightGBM Feature Importance (Gain)\")\n# plt.savefig('fi_gain_model[7][2].png', dpi=300, bbox_inches='tight')\n\n# lightgbm.plot_importance(models_total[15][1], importance_type=\"split\", figsize=(10, 20), title=\"LightGBM Feature Importance (Split)\")\n# plt.savefig('fi_split_model[15][1].png', dpi=300, bbox_inches='tight')\n\n# lightgbm.plot_importance(models_total[15][1], importance_type=\"gain\", figsize=(10, 20), title=\"LightGBM Feature Importance (Gain)\")\n# plt.savefig('fi_gain_model[15][1].png', dpi=300, bbox_inches='tight')\n\n# plt.show() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-16T13:25:01.054842Z","iopub.execute_input":"2024-12-16T13:25:01.055580Z","iopub.status.idle":"2024-12-16T13:25:19.423763Z","shell.execute_reply.started":"2024-12-16T13:25:01.055492Z","shell.execute_reply":"2024-12-16T13:25:19.422657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Testing Step (+ k-fold)\n\nX_test = test.drop(columns=['id'])\n\n# Predict using all K models for each PCIAT-PCIAT_01 to PCIAT-PCIAT_20 and take majority vote among the K summed models \ntest_preds = np.zeros((len(X_test), len(models_total[0])))\n\nfor i, models in enumerate(models_total):\n    fold_preds = np.zeros((len(X_test), len(models)))\n    for j, model in enumerate(models):\n        fold_preds[:, j] = model.predict(X_test)\n    test_preds[:, :] += fold_preds \n\n# Convert the summed prediction to Severity Impairment Index (sii)\nfor i in range(len(models_total[0])):\n    test_preds[:, i] = np.vectorize(pciat_to_sii_fractional)(test_preds[:, i])\n    test_preds[:, i] = round_with_thresholds(test_preds[:, i], res.x)\n\ny_pred = mode(test_preds, axis=1)[0].flatten().astype(int)\n\n# Print or return the final predictions\nprint(\"Final Test Predictions:\", y_pred)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.DataFrame({\n    'id': test['id'],\n    'sii': y_pred\n})\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission file 'submission.csv' has been created.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}