{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":13594915,"sourceType":"datasetVersion","datasetId":5850835}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">🎒Import Libraries</span>","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars as pl\nfrom sklearn.base import clone\nfrom scipy.optimize import minimize\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom matplotlib.ticker import MaxNLocator\n\nimport re\nfrom colorama import Fore, Style\n\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport warnings\nwarnings.filterwarnings('ignore')\n# pd.options.display.max_columns = None\n\nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import RobustScaler\nfrom mlxtend.regressor import StackingCVRegressor\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\nfrom xgboost import XGBRegressor\nimport xgboost as xgb\nimport lightgbm as lgb\nfrom lightgbm import early_stopping\n\nfrom datetime import datetime\nfrom sklearn.model_selection import KFold\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import PowerTransformer\nimport scipy.stats as stats\n\nfrom sklearn.decomposition import SparsePCA\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer \n\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score\n\n# n_splits = 6","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2025-11-27T09:13:17.06493Z","iopub.execute_input":"2025-11-27T09:13:17.065559Z","iopub.status.idle":"2025-11-27T09:13:17.074644Z","shell.execute_reply.started":"2025-11-27T09:13:17.065522Z","shell.execute_reply":"2025-11-27T09:13:17.0735Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style=\"background-color:#79a5ed;padding:10px;border-radius:40px;\">✨Preprocessing</span>","metadata":{}},{"cell_type":"code","source":"# def process_file(filename, dirname):\n#     df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n#     df = df[df['non-wear_flag'] != 1]\n#     df.drop('step', axis=1, inplace=True)\n#     df.drop('time_of_day', axis=1, inplace=True)\n#     df.drop('weekday', axis=1, inplace=True)\n#     df.drop('quarter', axis=1, inplace=True)\n#     df.drop('relative_date_PCIAT', axis=1, inplace=True)\n    \n#     return df.describe().values.reshape(-1), filename.split('=')[1]\n\n# def load_time_series(dirname) -> pd.DataFrame:\n#     ids = os.listdir(dirname)\n    \n#     with ThreadPoolExecutor() as executor:\n#         results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n#     stats, indexes = zip(*results)\n    \n#     df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n#     df['id'] = indexes\n    \n#     return df\n\n# train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n\n# train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\n\n# time_series_cols = train_ts.columns.tolist()\n# time_series_cols.remove(\"id\")\n\n# train = pd.merge(train, train_ts, how=\"left\", on='id')\n\n# train = train.drop('id', axis=1)\n\n# featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n#                 'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n#                 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n#                 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n#                 'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n#                 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n#                 'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n#                 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n#                 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n#                 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n#                 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n#                 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n#                 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n#                 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n#                 'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n#                 'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n#                 'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n#                 'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\n# featuresCols += time_series_cols\n\n# train = train[featuresCols]\n# train = train.dropna(subset='sii')\n\n# ######### SEASONS ONLY\n# # cat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n# #           'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\n# # def update(df):\n# #     for c in cat_c: \n# #         df[c] = df[c].fillna('Missing')\n# #         df[c] = df[c].astype('category')\n# #     return df\n        \n# # train = update(train)\n\n# # def create_mapping(column, dataset):\n# #     unique_values = dataset[column].unique()\n# #     return {value: idx for idx, value in enumerate(unique_values)}\n\n# # \"\"\"This Mapping Works Fine For me I also Check Each Values in Train and test Using Logic. There no Data Lekage.\"\"\"\n\n# # for col in cat_c:\n# #     mapping_train = create_mapping(col, train)    \n# #     train[col] = train[col].replace(mapping_train).astype(int)\n\n# # DROP_COLS = [\n# #     \"Basic_Demos-Enroll_Season\", \"CGAS-Season\", \"Physical-Season\",\n# #     \"Fitness_Endurance-Season\", \"FGC-Season\", \"BIA-Season\",\n# #     \"PAQ_A-Season\", \"PAQ_C-Season\", \"SDS-Season\", \"PreInt_EduHx-Season\"\n# # ]\n\n# # train = train.drop(DROP_COLS,axis=1)\n\n# # print(f'Train Shape : {train.shape}')\n\n# # train.to_csv(\"processed_train.csv\", index=False)\n\n","metadata":{"execution":{"iopub.status.busy":"2025-11-27T09:13:17.076562Z","iopub.execute_input":"2025-11-27T09:13:17.076867Z","iopub.status.idle":"2025-11-27T09:13:17.095343Z","shell.execute_reply.started":"2025-11-27T09:13:17.076838Z","shell.execute_reply":"2025-11-27T09:13:17.094231Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/cmi-problematic-internet-usage-combined-data/ColDrop_base_train.csv')\nprint(f'Train Shape : {train.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T09:13:17.097353Z","iopub.execute_input":"2025-11-27T09:13:17.097707Z","iopub.status.idle":"2025-11-27T09:13:17.153268Z","shell.execute_reply.started":"2025-11-27T09:13:17.097668Z","shell.execute_reply":"2025-11-27T09:13:17.152261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def analyze_columns(df: pd.DataFrame, verbose: bool = True) -> pd.DataFrame:\n#     \"\"\"\n#     Analyzes column completeness with clinical research standards\n    \n#     Args:\n#         df: Input DataFrame\n#         verbose: If True, prints formatted report\n        \n#     Returns:\n#         DataFrame with column completeness statistics\n#     \"\"\"\n#     # Calculate metrics\n#     total_rows = len(df)\n#     missing = df.isna().sum()\n#     present = total_rows - missing\n#     completeness = (present / total_rows) * 100\n    \n#     # Create report\n#     report = pd.DataFrame({\n#         'Total Values': total_rows,\n#         'Non-Missing': present,\n#         'Missing': missing,\n#         'Completeness %': completeness.round(1)\n#     })\n    \n#     # Add status classification\n#     report['Status'] = pd.cut(\n#         report['Completeness %'],\n#         bins=[-1, 50, 80, 95, 100],\n#         labels=['Critical Missingness (>50%)', \n#                 'Moderate Missingness (20-50%)',\n#                 'Acceptable Missingness (5-20%)',\n#                 'High Completeness (<5%)']\n#     )\n    \n#     if verbose:\n#         print(\"=\"*60)\n#         print(f\"DATA COMPLETENESS ANALYSIS ({len(df)} samples, {len(df.columns)} features)\")\n#         print(\"=\"*60)\n        \n#         # Summary statistics\n#         empty_cols = report[report['Completeness %'] == 0].index.tolist()\n#         high_completeness = len(report[report['Completeness %'] >= 95])\n        \n#         print(f\"- Completely empty columns: {len(empty_cols)}\")\n#         print(f\"- Columns with >95% completeness: {high_completeness}\")\n#         print(f\"- Overall dataset completeness: {report['Completeness %'].mean():.1f}%\")\n#         print(\"-\"*60)\n        \n#         # Show empty columns if any\n#         if empty_cols:\n#             print(f\"EMPTY COLUMNS (0% data):\")\n#             for col in empty_cols:\n#                 print(f\"  • {col}\")\n#             print(\"-\"*60)\n        \n#         # Show completeness distribution\n#         status_counts = report['Status'].value_counts().sort_index()\n#         for status, count in status_counts.items():\n#             print(f\"- {status}: {count} columns\")\n    \n#     return report\n\n# # USAGE\n# completeness_report = analyze_columns(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T09:13:17.280571Z","iopub.execute_input":"2025-11-27T09:13:17.280977Z","iopub.status.idle":"2025-11-27T09:13:17.287288Z","shell.execute_reply.started":"2025-11-27T09:13:17.280944Z","shell.execute_reply":"2025-11-27T09:13:17.286197Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">🔎View Data</span>","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2025-11-27T09:13:17.289009Z","iopub.execute_input":"2025-11-27T09:13:17.29046Z","iopub.status.idle":"2025-11-27T09:13:17.316965Z","shell.execute_reply.started":"2025-11-27T09:13:17.290403Z","shell.execute_reply":"2025-11-27T09:13:17.315958Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"actigraphy = pl.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=00115b9f/part-0.parquet')\nactigraphy","metadata":{"execution":{"iopub.status.busy":"2025-11-27T09:13:17.318352Z","iopub.execute_input":"2025-11-27T09:13:17.318788Z","iopub.status.idle":"2025-11-27T09:13:17.339707Z","shell.execute_reply.started":"2025-11-27T09:13:17.318743Z","shell.execute_reply":"2025-11-27T09:13:17.338735Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">📊Visualization</span>","metadata":{}},{"cell_type":"code","source":"vc = train['sii'].value_counts()\n\n# Map labels to sii \nsii_map = {0: 'None', 1: 'Mild', 2: 'Moderate', 3: 'Severe'}\n\n# Create labels using the map (Unchanged logic)\nlabels = [sii_map[label] for label in vc.index]\n\n# 1. Set Figure Size and Theme\nplt.figure(figsize=(10, 10)) # Increased figure size for a larger chart\nsns.set_context(\"notebook\", font_scale=1.4) # Globally increase font scale\n\n# 2. Define Custom Colors\ncolors = ['#1f77b4', '#ff7f0e', '#2ca02c', '#d62728'] \n\n# 3. Define Autopct Text Properties (for percentages)\n# Ensures the percentage text is large and clear\nautopct_format = lambda p: '{:.1f}%'.format(p) if p > 0 else '' # Only show percentages > 0\n\n# 4. Plot the Pie Chart\nwedges, texts, autotexts = plt.pie(\n    vc.values, \n    labels=labels, \n    autopct=autopct_format,\n    colors=colors,\n    startangle=90, \n    wedgeprops={'linewidth': 1, 'edgecolor': 'white'}, # Add white borders for separation\n    textprops={'fontsize': 22} # Set font size for labels\n)\n\n# 5. Increase Font Size for the Percentage Text (Autopct)\nplt.setp(autotexts, size=22)\n\n\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T09:13:17.340816Z","iopub.execute_input":"2025-11-27T09:13:17.341134Z","iopub.status.idle":"2025-11-27T09:13:17.49039Z","shell.execute_reply.started":"2025-11-27T09:13:17.341105Z","shell.execute_reply":"2025-11-27T09:13:17.489477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tmp_train = train.copy()\ntmp = train[['sii','PreInt_EduHx-computerinternet_hoursday']].value_counts().reset_index()\ntmp[['sii','PreInt_EduHx-computerinternet_hoursday']] = tmp[['sii','PreInt_EduHx-computerinternet_hoursday']].astype(int)\n\ntmp = tmp.pivot(\n    columns=\"PreInt_EduHx-computerinternet_hoursday\",\n    index=\"sii\",\n    values=\"count\"\n).fillna(0)\n\n# Set a plotting context to automatically scale fonts up\nsns.set_context(\"notebook\", font_scale=1.2) \n\nplt.figure(figsize=(14, 10)) # Increased figure size slightly more for the larger color bar\n\n# Create Heatmap with styling arguments\nax = sns.heatmap(\n    tmp, \n    annot=True, \n    fmt=\".0f\", \n    cmap=\"YlGnBu\", \n    cbar=True, # Make sure color bar is active\n    linewidths=1, \n    linecolor='white', \n    annot_kws={\"size\": 22}, \n    square=True,\n)\n\n# 1. Increase the size of the Color Bar Label (Title)\nax.figure.axes[-1].yaxis.label.set_size(22) \nax.figure.axes[-1].yaxis.label.set_weight('bold')\n\n# 2. Increase the size of the Color Bar Ticks (Numbers on the bar)\nax.figure.axes[-1].tick_params(labelsize=20) \n\n# --- Existing Axis/Label Font Adjustments (for completeness) ---\nplt.ylabel('SII', fontsize=22, labelpad=15)\nplt.xlabel('Internet Hours per Day', fontsize=22, labelpad=15)\nplt.xticks(fontsize=22)\nplt.yticks(fontsize=22, rotation=0) \n\nplt.tight_layout() \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T09:13:17.492991Z","iopub.execute_input":"2025-11-27T09:13:17.494322Z","iopub.status.idle":"2025-11-27T09:13:17.809105Z","shell.execute_reply.started":"2025-11-27T09:13:17.494271Z","shell.execute_reply":"2025-11-27T09:13:17.807862Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Bio-electric Impedance Analysis (BIA) are skewed so, lets apply Power Transformer with Yeo-Jhonson. \n\n# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">📈YEO-JHONSON</span>","metadata":{}},{"cell_type":"code","source":"# Identify columns to include\ninclude_cols = ['BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW']\n# Create copies of the train and test DataFrames to avoid modifying the originals\ntrain_transformed = train.copy()\ny_train = train['sii']\nX_train = train_transformed.drop('sii', axis=1)\n\n# Apply the Yeo-Johnson transformation to the numerical columns\npt = PowerTransformer()\n# Fit and transform the specified columns in the train DataFrame\nX_train[include_cols] = pt.fit_transform(X_train[include_cols])","metadata":{"execution":{"iopub.status.busy":"2025-11-27T09:13:17.810566Z","iopub.execute_input":"2025-11-27T09:13:17.811025Z","iopub.status.idle":"2025-11-27T09:13:17.88735Z","shell.execute_reply.started":"2025-11-27T09:13:17.810982Z","shell.execute_reply":"2025-11-27T09:13:17.886497Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"After applying Yeo-Jhonson, BIA columns have been normalised.\n\n# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">✳️Standardize and Impute</span>","metadata":{}},{"cell_type":"code","source":"# Standardize features\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\n\nn_components = 35  # Number of components to extract\nalpha = 0.01  # Sparsity parameter (higher alpha means sparser components)\n\n# Create a pipeline for imputation and Sparse PCA\npipeline = Pipeline([\n    ('imputer', SimpleImputer(strategy='mean')),  # Replace missing values with mean\n])\n\n# Fit the pipeline to the training data\npipeline.fit(X_train_scaled)\n\n# Transform both training and test data\nX_train_sparse = pipeline.transform(X_train_scaled)\n\nX_train = pd.DataFrame(X_train_sparse)\nX_train = X_train.reset_index(drop=True)\ny_train = y_train.reset_index(drop=True)\n\n# Concatenate X_train and y_train\ntrain_data_pca = pd.concat([pd.DataFrame(X_train), y_train], axis=1)\n\nprint(train_data_pca.head(1))\n\ntrain = train_data_pca","metadata":{"execution":{"iopub.status.busy":"2025-11-27T09:13:17.888832Z","iopub.execute_input":"2025-11-27T09:13:17.889388Z","iopub.status.idle":"2025-11-27T09:13:17.934137Z","shell.execute_reply.started":"2025-11-27T09:13:17.889341Z","shell.execute_reply":"2025-11-27T09:13:17.9331Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2025-11-27T09:13:17.935656Z","iopub.execute_input":"2025-11-27T09:13:17.936393Z","iopub.status.idle":"2025-11-27T09:13:17.942766Z","shell.execute_reply.started":"2025-11-27T09:13:17.936347Z","shell.execute_reply":"2025-11-27T09:13:17.941729Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">⚖️Quadratic Weighted Kappa</span>","metadata":{}},{"cell_type":"code","source":"def TrainML(model_class, n_splits=5, SEED=5):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n    \n    # 3. Use proper stratification for ordinal target\n    y_binned = pd.cut(y, bins=4, labels=False)  # Convert to 4 bins for stratification\n    \n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    oof_non_rounded = np.zeros(len(y), dtype=float)\n    oof_rounded = np.zeros(len(y), dtype=int)\n    train_kappas = []\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y_binned), \n                                                    desc=\"Training Folds\", \n                                                    total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        # 4. ADD VALIDATION SET FOR EARLY STOPPING (10% of training data)\n        X_train_fit, X_early_stop, y_train_fit, y_early_stop = train_test_split(\n            X_train, y_train, test_size=0.1, random_state=SEED\n        )\n\n        # 5. TRAIN EACH MODEL WITH EARLY STOPPING\n        for name, model in model_class.items():\n            if \"LightGBM\" in name:\n                model.fit(\n                    X_train_fit, y_train_fit,\n                    eval_set=[(X_early_stop, y_early_stop)],\n                    callbacks=[lgb.early_stopping(stopping_rounds=100, verbose=False)]\n                )\n            elif \"XGB\" in name:\n                model.fit(\n                    X_train_fit, y_train_fit,\n                    eval_set=[(X_early_stop, y_early_stop)],\n                    early_stopping_rounds=100,\n                    verbose=False\n                )\n            elif \"Cat\" in name:\n                model.fit(\n                    X_train_fit, y_train_fit,\n                    eval_set=[(X_early_stop, y_early_stop)],\n                    early_stopping_rounds=100,\n                    verbose=False\n                )\n\n        # 6. PREDICT ON VALIDATION SET (using models trained with early stopping)\n        y_val_pred = blend_predict(X_val, model_class)\n        oof_non_rounded[test_idx] = y_val_pred\n        oof_rounded[test_idx] = np.round(y_val_pred).astype(int)\n\n        # 7. CORRECT METRICS: Use EARLY STOPPING SET for training metrics\n        y_train_pred = blend_predict(X_train_fit, model_class)\n        train_kappa = cohen_kappa_score(\n            y_train_fit, \n            np.round(y_train_pred).astype(int),\n            weights='quadratic'\n        )\n        val_kappa = cohen_kappa_score(y_val, oof_rounded[test_idx], weights='quadratic')\n        \n        train_kappas.append(train_kappa)\n        print(f\"Fold {fold+1} | Train QWK: {train_kappa:.4f} | Validation QWK: {val_kappa:.4f}\")\n\n    print(f\"\\nMean Train QWK: {np.mean(train_kappas):.4f}\")\n\n    # 8. THRESHOLD OPTIMIZATION (unchanged)\n    def evaluate_predictions(thresholds, y_true, oof):\n        rounded = np.select(\n            [oof < thresholds[0], oof < thresholds[1], oof < thresholds[2]],\n            [0, 1, 2],\n            default=3\n        )\n        return -cohen_kappa_score(y_true, rounded, weights='quadratic')\n\n    res = minimize(\n        evaluate_predictions,\n        x0=[0.5, 1.5, 2.5],\n        args=(y, oof_non_rounded),\n        method='Nelder-Mead',\n        options={'xatol': 0.01, 'maxiter': 100}\n    )\n    \n    assert res.success, \"Threshold optimization failed\"\n    oof_tuned = np.select(\n        [oof_non_rounded < res.x[0], oof_non_rounded < res.x[1], oof_non_rounded < res.x[2]],\n        [0, 1, 2],\n        default=3\n    )\n    tKappa = cohen_kappa_score(y, oof_tuned, weights='quadratic')\n    \n    print(f\"\\nOptimized QWK: {Fore.CYAN}{Style.BRIGHT}{tKappa:.3f}{Style.RESET_ALL}\")\n    return tKappa, res.x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-27T09:13:17.944438Z","iopub.execute_input":"2025-11-27T09:13:17.944878Z","iopub.status.idle":"2025-11-27T09:13:17.961827Z","shell.execute_reply.started":"2025-11-27T09:13:17.944828Z","shell.execute_reply":"2025-11-27T09:13:17.960716Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">🖥️Pipelines and Models</span>","metadata":{}},{"cell_type":"code","source":"SEED = 5\n# 1. FIX MODEL DEFINITIONS: Set high n_estimators + remove fixed values\nParams_lgb = {\n    'learning_rate': 0.01, \n    'max_depth': 12, \n    'num_leaves': 413, \n    'min_data_in_leaf': 14,\n    'feature_fraction': 0.79, \n    'bagging_fraction': 0.76, \n    'bagging_freq': 2, \n    'lambda_l1': 5.5, \n    'lambda_l2': 5.5e-06,\n    # 'device': 'gpu',\n    'n_estimators': 10000  # High value for early stopping\n}\nLight = lgb.LGBMRegressor(**Params_lgb,random_state=SEED, verbose=-1)\n\nParams_xgb = {\n    'learning_rate': 0.01,\n    'max_depth': 12,\n    'max_leaves': 413,\n    'min_child_weight': 14,\n    'colsample_bytree': 0.79,\n    'subsample': 0.76,\n    'reg_alpha': 5.5,\n    'reg_lambda': 5.5e-06,\n    # 'device': 'cuda',\n    'n_estimators': 10000  # High value\n}\nXGBoost = xgb.XGBRegressor(**Params_xgb, random_state=SEED, verbosity=0)\n\n# Remove n_estimators from CatBoost params (set in constructor)\nParams_cat =  {\n    'learning_rate': 0.01,\n    'max_depth': 12,\n    'min_child_samples': 14,  # Equivalent to min_data_in_leaf\n    'rsm': 0.79,  # Equivalent to feature_fraction\n    'subsample': 0.76,  # Equivalent to bagging_fraction\n    'bootstrap_type': 'Bernoulli',\n    'l2_leaf_reg': 5.5,  # Equivalent to lambda_l2/lambda_l1\n    'random_seed': SEED,  # Use same seed for reproducibility\n    'n_estimators': 10000,  # Number of trees\n    'task_type': 'GPU',\n    'verbose': 0  # Suppresses output\n}\nCatBoost = CatBoostRegressor(**Params_cat)\n","metadata":{"execution":{"iopub.status.busy":"2025-11-27T09:13:17.963535Z","iopub.execute_input":"2025-11-27T09:13:17.963981Z","iopub.status.idle":"2025-11-27T09:13:17.975705Z","shell.execute_reply.started":"2025-11-27T09:13:17.963933Z","shell.execute_reply":"2025-11-27T09:13:17.974592Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">📕Create Dictionaries and Predict</span>","metadata":{}},{"cell_type":"code","source":"# Create a dictionary to store the models\nmodels = {'LightGBMRegressor': Light,\n         'XGBoostRegressor': XGBoost}\n\ndef blend_predict(X, models):\n    return 0.5 * models['LightGBMRegressor'].predict(X) + 0.5 * models['XGBoostRegressor'].predict(X)","metadata":{"execution":{"iopub.status.busy":"2025-11-27T09:13:17.977159Z","iopub.execute_input":"2025-11-27T09:13:17.977589Z","iopub.status.idle":"2025-11-27T09:13:17.992309Z","shell.execute_reply.started":"2025-11-27T09:13:17.977545Z","shell.execute_reply":"2025-11-27T09:13:17.991275Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <span style=\"background-color:#bbebfa;color:black;padding:10px;border-radius:40px;\">🚅Train Model</span>","metadata":{}},{"cell_type":"code","source":"# Train models and get predictions\nTrainML(models)","metadata":{"_kg_hide-output":false,"execution":{"iopub.status.busy":"2025-11-27T09:13:17.993736Z","iopub.execute_input":"2025-11-27T09:13:17.994708Z","iopub.status.idle":"2025-11-27T09:13:18.002255Z","shell.execute_reply.started":"2025-11-27T09:13:17.994643Z","shell.execute_reply":"2025-11-27T09:13:18.001238Z"},"trusted":true},"outputs":[],"execution_count":null}]}