{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"colab":{"provenance":[]},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# defining necessary imports\nimport numpy as np\nimport pandas as pd\nimport os\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score, make_scorer, confusion_matrix\nfrom sklearn.model_selection import StratifiedKFold, KFold, train_test_split\nfrom sklearn.preprocessing import StandardScaler, QuantileTransformer # Added QuantileTransformer\nfrom sklearn.decomposition import PCA\nfrom scipy.optimize import minimize\nfrom scipy import stats\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport warnings\nfrom sklearn.linear_model import ElasticNetCV, LassoCV, Lasso, LinearRegression\nfrom sklearn.ensemble import ExtraTreesRegressor, RandomForestRegressor\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nimport optuna\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport random\nimport mord\nfrom sklearn.impute import KNNImputer","metadata":{"id":"7de75204","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:02:37.100564Z","iopub.execute_input":"2025-04-24T06:02:37.100907Z","iopub.status.idle":"2025-04-24T06:02:40.142573Z","shell.execute_reply.started":"2025-04-24T06:02:37.100876Z","shell.execute_reply":"2025-04-24T06:02:40.141846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ignoring warnings\nwarnings.filterwarnings('ignore')","metadata":{"id":"fa827df4","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:02:40.143399Z","iopub.execute_input":"2025-04-24T06:02:40.143905Z","iopub.status.idle":"2025-04-24T06:02:40.148817Z","shell.execute_reply.started":"2025-04-24T06:02:40.143882Z","shell.execute_reply":"2025-04-24T06:02:40.147520Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# defining constants\nSEED = 3\nn_splits = 10\noptimize_params = True # change depending on if you want params changed or not\nn_optuna_trials = 25 # n_trials for optuna\nbase_thresholds = [30, 50, 80] # Initial guess for thresholds","metadata":{"id":"926cc4a1","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:02:40.149904Z","iopub.execute_input":"2025-04-24T06:02:40.150162Z","iopub.status.idle":"2025-04-24T06:02:40.173125Z","shell.execute_reply.started":"2025-04-24T06:02:40.150133Z","shell.execute_reply":"2025-04-24T06:02:40.172197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# defining paths\nTABULAR_TRAIN_PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\nTABULAR_TEST_PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'\nACTIGRAPHY_TRAIN_PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet'\nACTIGRAPHY_TEST_PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet'\nSUBMISSION_PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv'\nOUTPUT_PATH = '/kaggle/working/'","metadata":{"id":"15742c98","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:02:40.174094Z","iopub.execute_input":"2025-04-24T06:02:40.174337Z","iopub.status.idle":"2025-04-24T06:02:40.198067Z","shell.execute_reply.started":"2025-04-24T06:02:40.174319Z","shell.execute_reply":"2025-04-24T06:02:40.196721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_full = pd.read_csv(TABULAR_TRAIN_PATH)\ntest = pd.read_csv(TABULAR_TEST_PATH)","metadata":{"id":"7194ff60","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:02:40.199167Z","iopub.execute_input":"2025-04-24T06:02:40.199487Z","iopub.status.idle":"2025-04-24T06:02:40.323517Z","shell.execute_reply.started":"2025-04-24T06:02:40.199459Z","shell.execute_reply":"2025-04-24T06:02:40.322486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_weights(series):\n    # Create bins for the target variable and assign weights based on frequency\n    bins = pd.cut(series, bins=10, labels=False)\n    weights = bins.value_counts().reset_index()\n    weights.columns = ['target_bins', 'count']\n    weights['count'] = 1 / weights['count']\n    weight_map = weights.set_index('target_bins')['count'].to_dict()\n    weights = bins.map(weight_map)\n    return weights / weights.mean()","metadata":{"id":"e0d6e6d1","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:02:40.326190Z","iopub.execute_input":"2025-04-24T06:02:40.326459Z","iopub.status.idle":"2025-04-24T06:02:40.331819Z","shell.execute_reply.started":"2025-04-24T06:02:40.326438Z","shell.execute_reply":"2025-04-24T06:02:40.330899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def round_with_thresholds(raw_preds, thresholds):\n    return np.where(raw_preds < thresholds[0], int(0),\n                    np.where(raw_preds < thresholds[1], int(1),\n                             np.where(raw_preds < thresholds[2], int(2), int(3))))","metadata":{"id":"78ebf890","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:02:40.332838Z","iopub.execute_input":"2025-04-24T06:02:40.333185Z","iopub.status.idle":"2025-04-24T06:02:40.349785Z","shell.execute_reply.started":"2025-04-24T06:02:40.333158Z","shell.execute_reply":"2025-04-24T06:02:40.348941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def perform_pca(train, test, n_components=None, random_state=33):\n    \"\"\"\n    Performs PCA on train, describes the Explained Variance Ratio and transforms train and test\n\n    Returns: transformed train, transformed test, pca object\n\n    \"\"\"\n    pca = PCA(n_components=n_components, random_state=random_state)\n    train_pca = pca.fit_transform(train)\n    test_pca = pca.transform(test)\n\n    explained_variance_ratio = pca.explained_variance_ratio_\n    print(f\"Explained variance ratio of the components:\\n {explained_variance_ratio}\")\n    print(f\"Cumulative explained variance: {np.sum(explained_variance_ratio)}\") # Added cumulative variance\n\n    train_pca_df = pd.DataFrame(train_pca, columns=[f'PC_{i+1}' for i in range(train_pca.shape[1])], index=train.index) # Preserve index\n    test_pca_df = pd.DataFrame(test_pca, columns=[f'PC_{i+1}' for i in range(test_pca.shape[1])], index=test.index) # Preserve index\n\n    return train_pca_df, test_pca_df, pca","metadata":{"id":"a05b1a84","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:02:40.350988Z","iopub.execute_input":"2025-04-24T06:02:40.351330Z","iopub.status.idle":"2025-04-24T06:02:40.367235Z","shell.execute_reply.started":"2025-04-24T06:02:40.351295Z","shell.execute_reply":"2025-04-24T06:02:40.366382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def time_features(df):\n    \"\"\"Function extracting Features from ActiGraph data of an individual\"\"\"\n    # Convert time_of_day to hours\n    df[\"hours\"] = df[\"time_of_day\"] // (3_600 * 1_000_000_000)\n    # Basic features\n    features = [\n        df[\"non-wear_flag\"].mean(),\n        df[\"enmo\"][df[\"enmo\"] >= 0.05].sum(),\n    ]\n\n    # Define conditions for night, day, and no mask (full data)\n    night = ((df[\"hours\"] >= 22) | (df[\"hours\"] <= 5))\n    day = ((df[\"hours\"] <= 20) & (df[\"hours\"] >= 7))\n    no_mask = np.ones(len(df), dtype=bool)\n\n    # List of columns of interest and masks\n    keys = [\"enmo\", \"anglez\", \"light\", \"battery_voltage\"]\n    masks = [no_mask, night, day]\n\n    # Helper function for feature extraction\n    def extract_stats(data):\n        return [\n            data.mean(),\n            data.std(),\n            data.max(),\n            data.min(),\n            data.diff().mean(),\n            data.diff().std()\n        ]\n\n    # Iterate over keys and masks to generate the statistics\n    for key in keys:\n        for mask in masks:\n            filtered_data = df.loc[mask, key]\n            features.extend(extract_stats(filtered_data))\n\n    return features","metadata":{"id":"d7023f17","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:02:40.368184Z","iopub.execute_input":"2025-04-24T06:02:40.368499Z","iopub.status.idle":"2025-04-24T06:02:40.388176Z","shell.execute_reply.started":"2025-04-24T06:02:40.368474Z","shell.execute_reply":"2025-04-24T06:02:40.387125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    # Process file and extract time features\n    try:\n        df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n        df.drop('step', axis=1, inplace=True)\n        return time_features(df), filename.split('=')[1]\n    except Exception as e:\n        print(f\"Error processing {filename}: {e}\")\n        # Return placeholder data or handle error appropriately\n        return [0.0] * 74, filename.split('=')[1] # Adjust size based on expected features","metadata":{"id":"e6b706e4","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:02:40.389138Z","iopub.execute_input":"2025-04-24T06:02:40.389398Z","iopub.status.idle":"2025-04-24T06:02:40.407350Z","shell.execute_reply.started":"2025-04-24T06:02:40.389378Z","shell.execute_reply":"2025-04-24T06:02:40.406400Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def load_time_series(dirname) -> pd.DataFrame:\n#     # Load time series from directory in parallel\n#     ids = os.listdir(dirname)\n\n#     with ThreadPoolExecutor() as executor:\n#         results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids), desc=\"Processing Time Series\"))\n\n#     stats_list, indexes = zip(*results)\n\n#     # Check consistency of feature lengths\n#     feature_length = len(stats_list[0])\n#     if not all(len(s) == feature_length for s in stats_list):\n#         print(\"Warning: Inconsistent feature lengths detected in time series processing.\")\n#         # Add logic here to handle inconsistent lengths if necessary, e.g., padding or error reporting\n\n#     df = pd.DataFrame(stats_list, columns=[f\"stat_{i}\" for i in range(feature_length)])\n#     df['id'] = indexes\n\n#     return df","metadata":{"id":"1ec5795a","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:02:40.408303Z","iopub.execute_input":"2025-04-24T06:02:40.409132Z","iopub.status.idle":"2025-04-24T06:02:40.424086Z","shell.execute_reply.started":"2025-04-24T06:02:40.409104Z","shell.execute_reply":"2025-04-24T06:02:40.423064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_time_series(dirname) -> pd.DataFrame:\n    # Only process subdirectories\n    ids = [d for d in os.listdir(dirname) if os.path.isdir(os.path.join(dirname, d))]\n\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(\n            executor.map(lambda fname: process_file(fname, dirname), ids),\n            total=len(ids),\n            desc=\"Processing Time Series\"\n        ))\n\n    stats_list, indexes = zip(*results)\n\n    feature_length = len(stats_list[0])\n    if not all(len(s) == feature_length for s in stats_list):\n        print(\"Warning: Inconsistent feature lengths detected.\")\n\n    df = pd.DataFrame(stats_list, columns=[f\"stat_{i}\" for i in range(feature_length)])\n    df['id'] = indexes\n\n    return df\n","metadata":{"id":"a2efa54c","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:02:40.425000Z","iopub.execute_input":"2025-04-24T06:02:40.425254Z","iopub.status.idle":"2025-04-24T06:02:40.441012Z","shell.execute_reply.started":"2025-04-24T06:02:40.425235Z","shell.execute_reply":"2025-04-24T06:02:40.440126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts = load_time_series(ACTIGRAPHY_TRAIN_PATH)\ntest_ts = load_time_series(ACTIGRAPHY_TEST_PATH)","metadata":{"id":"9dc812e7","outputId":"83f0fa11-3911-4da4-fb5c-724c954b3a66","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:02:40.441975Z","iopub.execute_input":"2025-04-24T06:02:40.442309Z","iopub.status.idle":"2025-04-24T06:03:50.600625Z","shell.execute_reply.started":"2025-04-24T06:02:40.442281Z","shell.execute_reply":"2025-04-24T06:03:50.599716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train_ts = train_ts.drop('id', axis=1).set_index(train_ts['id']) # Set ID as index\ndf_test_ts = test_ts.drop('id', axis=1).set_index(test_ts['id'])","metadata":{"id":"9a01ff82","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:50.601588Z","iopub.execute_input":"2025-04-24T06:03:50.601933Z","iopub.status.idle":"2025-04-24T06:03:50.609510Z","shell.execute_reply.started":"2025-04-24T06:03:50.601903Z","shell.execute_reply":"2025-04-24T06:03:50.608601Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# scaling and imputation prior to PCA\n# using a quantile transformer and KNN Imputer\nqt = QuantileTransformer(output_distribution='normal', random_state=SEED)\nknn_imputer_ts = KNNImputer(n_neighbors=5) # Impute based on neighbors\n\ndf_train_ts_qt = qt.fit_transform(df_train_ts)\ndf_test_ts_qt = qt.transform(df_test_ts)\n\ndf_train_ts_imputed = knn_imputer_ts.fit_transform(df_train_ts_qt)\ndf_test_ts_imputed = knn_imputer_ts.transform(df_test_ts_qt)\n\ndf_train_ts = pd.DataFrame(df_train_ts_imputed, columns=df_train_ts.columns, index=df_train_ts.index)\ndf_test_ts = pd.DataFrame(df_test_ts_imputed, columns=df_test_ts.columns, index=df_test_ts.index)","metadata":{"id":"74750baf","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:50.610395Z","iopub.execute_input":"2025-04-24T06:03:50.610639Z","iopub.status.idle":"2025-04-24T06:03:50.923616Z","shell.execute_reply.started":"2025-04-24T06:03:50.610621Z","shell.execute_reply":"2025-04-24T06:03:50.922709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"TS Train shape after imputation: {df_train_ts.shape}\")","metadata":{"id":"e88ce539","outputId":"ccf5a9ad-3a1c-43f7-874d-6d81ddf29f67","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:50.924369Z","iopub.execute_input":"2025-04-24T06:03:50.924601Z","iopub.status.idle":"2025-04-24T06:03:50.930306Z","shell.execute_reply.started":"2025-04-24T06:03:50.924583Z","shell.execute_reply":"2025-04-24T06:03:50.929287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# perform PCA\nprint(\"Performing PCA on time series features...\")\ndf_train_pca, df_test_pca, pca = perform_pca(df_train_ts, df_test_ts, n_components=20, random_state=SEED) # Increased components slightly","metadata":{"id":"ab406938","outputId":"3d798844-8c9e-4fc1-ae0b-f8f7717fcef3","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:50.931229Z","iopub.execute_input":"2025-04-24T06:03:50.931532Z","iopub.status.idle":"2025-04-24T06:03:51.021678Z","shell.execute_reply.started":"2025-04-24T06:03:50.931503Z","shell.execute_reply":"2025-04-24T06:03:51.019448Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Merge PCA features back\ntrain_full = pd.merge(train_full, df_train_pca, how=\"left\", left_on='id', right_index=True)\ntest = pd.merge(test, df_test_pca, how=\"left\", left_on='id', right_index=True)\nprint(f\"Train shape after merging PCA: {train_full.shape}\")","metadata":{"id":"064b732f","outputId":"a1346a67-a4bb-4341-9bab-8c70f9547de8","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:51.023156Z","iopub.execute_input":"2025-04-24T06:03:51.024356Z","iopub.status.idle":"2025-04-24T06:03:51.070596Z","shell.execute_reply.started":"2025-04-24T06:03:51.024330Z","shell.execute_reply":"2025-04-24T06:03:51.069390Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# feature cleaning\ndef clean_features(df):\n    # Remove highly implausible values\n\n    # Clip Grip\n    df[['FGC-FGC_GSND', 'FGC-FGC_GSD']] = df[['FGC-FGC_GSND', 'FGC-FGC_GSD']].clip(lower=9, upper=60)\n    # Remove implausible body-fat\n    df[\"BIA-BIA_Fat\"] = np.where(df[\"BIA-BIA_Fat\"] < 5, np.nan, df[\"BIA-BIA_Fat\"])\n    df[\"BIA-BIA_Fat\"] = np.where(df[\"BIA-BIA_Fat\"] > 60, np.nan, df[\"BIA-BIA_Fat\"])\n    # Basal Metabolic Rate\n    df[\"BIA-BIA_BMR\"] = np.where(df[\"BIA-BIA_BMR\"] > 4000, np.nan, df[\"BIA-BIA_BMR\"])\n    # Daily Energy Expenditure\n    df[\"BIA-BIA_DEE\"] = np.where(df[\"BIA-BIA_DEE\"] > 8000, np.nan, df[\"BIA-BIA_DEE\"])\n    # Bone Mineral Content\n    df[\"BIA-BIA_BMC\"] = np.where(df[\"BIA-BIA_BMC\"] <= 0, np.nan, df[\"BIA-BIA_BMC\"])\n    df[\"BIA-BIA_BMC\"] = np.where(df[\"BIA-BIA_BMC\"] > 10, np.nan, df[\"BIA-BIA_BMC\"])\n    # Fat Free Mass Index - Corrected column name assuming it's BIA-BIA_FFM\n    df[\"BIA-BIA_FFM\"] = np.where(df[\"BIA-BIA_FFM\"] <= 0, np.nan, df[\"BIA-BIA_FFM\"])\n    df[\"BIA-BIA_FFM\"] = np.where(df[\"BIA-BIA_FFM\"] > 300, np.nan, df[\"BIA-BIA_FFM\"])\n    # Fat Mass Index\n    df[\"BIA-BIA_FMI\"] = np.where(df[\"BIA-BIA_FMI\"] < 0, np.nan, df[\"BIA-BIA_FMI\"])\n    # Extra Cellular Water\n    df[\"BIA-BIA_ECW\"] = np.where(df[\"BIA-BIA_ECW\"] > 100, np.nan, df[\"BIA-BIA_ECW\"])\n    # Intra Cellular Water - commented out in original\n    # df[\"BIA-BIA_ICW\"] = np.where(df[\"BIA-BIA_ICW\"] > 100, np.nan, df[\"BIA-BIA_ICW\"])\n    # Lean Dry Mass\n    df[\"BIA-BIA_LDM\"] = np.where(df[\"BIA-BIA_LDM\"] > 100, np.nan, df[\"BIA-BIA_LDM\"])\n    # Lean Soft Tissue\n    df[\"BIA-BIA_LST\"] = np.where(df[\"BIA-BIA_LST\"] > 300, np.nan, df[\"BIA-BIA_LST\"])\n    # Skeletal Muscle Mass\n    df[\"BIA-BIA_SMM\"] = np.where(df[\"BIA-BIA_SMM\"] > 300, np.nan, df[\"BIA-BIA_SMM\"])\n    # Total Body Water\n    df[\"BIA-BIA_TBW\"] = np.where(df[\"BIA-BIA_TBW\"] > 300, np.nan, df[\"BIA-BIA_TBW\"])\n\n    return df","metadata":{"id":"ac4d5c1b","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:51.072835Z","iopub.execute_input":"2025-04-24T06:03:51.073624Z","iopub.status.idle":"2025-04-24T06:03:51.093114Z","shell.execute_reply.started":"2025-04-24T06:03:51.073597Z","shell.execute_reply":"2025-04-24T06:03:51.091834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_full = clean_features(train_full)\ntest = clean_features(test)","metadata":{"id":"7c2d6209","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:51.094841Z","iopub.execute_input":"2025-04-24T06:03:51.096315Z","iopub.status.idle":"2025-04-24T06:03:51.138211Z","shell.execute_reply.started":"2025-04-24T06:03:51.096284Z","shell.execute_reply":"2025-04-24T06:03:51.137410Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# feature engineering\ndef feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1, errors='ignore') # Use errors='ignore'\n\n    # From here on own features\n    def assign_group(age):\n        thresholds = [5, 6, 7, 8, 10, 12, 14, 17, 22]\n        for i, j in enumerate(thresholds):\n            if age <= j:\n                return i\n        return np.nan # Return NaN if age is outside defined ranges\n\n    # Age groups\n    df[\"group\"] = df['Basic_Demos-Age'].apply(assign_group)\n\n    # BMI\n    BMI_map = {0: 16.3,1: 15.9,2: 16.1,3: 16.8,4: 17.3,5: 19.2,6: 20.2,7: 22.3, 8: 23.6}\n    # Handle potential NaN in group map result\n    df['BMI_mean'] = df[['Physical-BMI', 'BIA-BIA_BMI']].mean(axis=1)\n    df['group_BMI_mean'] = df[\"group\"].map(BMI_map)\n    df['BMI_mean_norm'] = df['BMI_mean'] / df['group_BMI_mean']\n    df.drop(['BMI_mean', 'group_BMI_mean'], axis=1, inplace=True)\n\n\n    # FGC zone aggregate\n    zones = ['FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone',\n             'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone',\n             'FGC-FGC_TL_Zone']\n\n    # Ensure zones exist before calculating stats\n    existing_zones = [z for z in zones if z in df.columns]\n    if existing_zones:\n        df['FGC_Zones_mean'] = df[existing_zones].mean(axis=1)\n        df['FGC_Zones_min'] = df[existing_zones].min(axis=1)\n        df['FGC_Zones_max'] = df[existing_zones].max(axis=1)\n\n    # Grip\n    GSD_max_map = {0: 9, 1: 9, 2: 9, 3: 9, 4: 16.2, 5: 19.9, 6: 26.1, 7: 31.3, 8: 35.4}\n    GSD_min_map = {0: 9, 1: 9, 2: 9, 3: 9, 4: 14.4, 5: 17.8, 6: 23.4, 7: 27.8, 8: 31.1}\n\n    df['GS_max_val'] = df[['FGC-FGC_GSND', 'FGC-FGC_GSD']].max(axis=1)\n    df['GS_min_val'] = df[['FGC-FGC_GSND', 'FGC-FGC_GSD']].min(axis=1)\n    df['group_GSD_max'] = df[\"group\"].map(GSD_max_map)\n    df['group_GSD_min'] = df[\"group\"].map(GSD_min_map)\n\n    df['GS_max'] = df['GS_max_val'] / df['group_GSD_max']\n    df['GS_min'] = df['GS_min_val'] / df['group_GSD_min']\n    df.drop(['GS_max_val', 'GS_min_val', 'group_GSD_max', 'group_GSD_min'], axis=1, inplace=True)\n\n\n    # Curl-ups, push-ups, trunk-lifts... normalized based on age-group\n    cu_map = {0: 1.0, 1: 3.0, 2: 5.0, 3: 7.0, 4: 10.0, 5: 14.0, 6: 20.0, 7: 20.0, 8: 20.0}\n    pu_map = {0: 1.0, 1: 2.0, 2: 3.0, 3: 4.0, 4: 5.0, 5: 7.0, 6: 8.0, 7: 10.0, 8: 14.0}\n    tl_map = {0: 8.0, 1: 8.0, 2: 8.0, 3: 9.0, 4: 9.0, 5: 10.0, 6: 10.0, 7: 10.0, 8: 10.0}\n\n    df[\"CU_norm\"] = df['FGC-FGC_CU'] / df['group'].map(cu_map)\n    df[\"PU_norm\"] = df['FGC-FGC_PU'] / df['group'].map(pu_map)\n    df[\"TL_norm\"] = df['FGC-FGC_TL'] / df['group'].map(tl_map)\n\n    # Reach\n    df[\"SR_min\"] = df[['FGC-FGC_SRL', 'FGC-FGC_SRR']].min(axis=1)\n    df[\"SR_max\"] = df[['FGC-FGC_SRL', 'FGC-FGC_SRR']].max(axis=1)\n\n    # BIA Features\n    # Energy Expenditure\n    bmr_map = {0: 934.0, 1: 941.0, 2: 999.0, 3: 1048.0, 4: 1283.0, 5: 1255.0, 6: 1481.0, 7: 1519.0, 8: 1650.0}\n    dee_map = {0: 1471.0, 1: 1508.0, 2: 1640.0, 3: 1735.0, 4: 2132.0, 5: 2121.0, 6: 2528.0, 7: 2566.0, 8: 2793.0}\n    df[\"BMR_norm\"] = df[\"BIA-BIA_BMR\"] / df[\"group\"].map(bmr_map)\n    df[\"DEE_norm\"] = df[\"BIA-BIA_DEE\"] / df[\"group\"].map(dee_map)\n    df[\"DEE_BMR\"] = df[\"BIA-BIA_DEE\"] - df[\"BIA-BIA_BMR\"] # Consider potential division by zero or NaNs\n\n    # FMM\n    ffm_map = {0: 42.0, 1: 43.0, 2: 49.0, 3: 54.0, 4: 60.0, 5: 76.0, 6: 94.0, 7: 104.0, 8: 111.0}\n    # Handle potential NaN division\n    df['group_ffm_map'] = df[\"group\"].map(ffm_map)\n    df[\"FFM_norm\"] = df[\"BIA-BIA_FFM\"] / df['group_ffm_map']\n    df.drop(['group_ffm_map'], axis=1, inplace=True)\n\n    # ECW ICW\n    # Handle potential division by zero or NaN\n    df[\"ICW_ECW\"] = df[\"BIA-BIA_ECW\"] / df[\"BIA-BIA_ICW\"].replace(0, np.nan) # Replace 0 with NaN before division\n\n    # Drop original/intermediate features\n    drop_feats = ['FGC-FGC_GSND', 'FGC-FGC_GSD', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone',\n                  'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL_Zone',\n                  'Physical-BMI', 'BIA-BIA_BMI', 'FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_TL', 'FGC-FGC_SRL', 'FGC-FGC_SRR',\n                 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_Frame_num', \"BIA-BIA_FFM\", \"BIA-BIA_ICW\", \"BIA-BIA_ECW\", # Added ICW, ECW\n                 'group' # Drop the group feature itself after use\n                 ]\n    # Drop only columns that exist in the dataframe\n    existing_drop_feats = [feat for feat in drop_feats if feat in df.columns]\n    df = df.drop(existing_drop_feats, axis=1)\n    return df","metadata":{"id":"bc45f7ce","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:51.144335Z","iopub.execute_input":"2025-04-24T06:03:51.144731Z","iopub.status.idle":"2025-04-24T06:03:51.167076Z","shell.execute_reply.started":"2025-04-24T06:03:51.144696Z","shell.execute_reply":"2025-04-24T06:03:51.165950Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_full = feature_engineering(train_full)\ntest = feature_engineering(test)","metadata":{"id":"2656463f","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:51.168108Z","iopub.execute_input":"2025-04-24T06:03:51.168460Z","iopub.status.idle":"2025-04-24T06:03:51.277134Z","shell.execute_reply.started":"2025-04-24T06:03:51.168433Z","shell.execute_reply":"2025-04-24T06:03:51.276116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Binning (OPTIONAL, test both with and without it to see which performs better)\n# def bin_data(train, test, columns, n_bins=10):\n#     # Combine train and test for consistent bin edges\n#     combined = pd.concat([train, test], axis=0)\n\n#     bin_edges = {}\n#     for col in columns:\n#         # Compute quantile bin edges\n#         edges = pd.qcut(combined[col], n_bins, retbins=True, labels=range(n_bins), duplicates=\"drop\")[1]\n#         bin_edges[col] = edges\n\n#     # Apply the same bin edges to both train and test\n#     for col, edges in bin_edges.items():\n#         train[col] = pd.cut(\n#             train[col], bins=edges, labels=range(len(edges) - 1), include_lowest=True\n#         ).astype(float)\n#         test[col] = pd.cut(\n#             test[col], bins=edges, labels=range(len(edges) - 1), include_lowest=True\n#         ).astype(float)\n\n#     return train, test\n\n# columns_to_bin = [\n#     \"PAQ_A-PAQ_A_Total\", \"BMR_norm\", \"DEE_norm\", \"GS_min\", \"GS_max\", \"BIA-BIA_FFMI\",\n#     \"BIA-BIA_BMC\", \"Physical-HeartRate\", \"BIA-BIA_ICW\", \"Fitness_Endurance-Time_Sec\",\n#     \"BIA-BIA_LDM\", \"BIA-BIA_SMM\", \"BIA-BIA_TBW\", \"DEE_BMR\", \"ICW_ECW\"\n# ]\n# # Bin specified columns in train and test\n# train, test = bin_data(train, test, columns_to_bin, n_bins=10)","metadata":{"id":"73899546","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:51.278225Z","iopub.execute_input":"2025-04-24T06:03:51.278490Z","iopub.status.idle":"2025-04-24T06:03:51.283314Z","shell.execute_reply.started":"2025-04-24T06:03:51.278472Z","shell.execute_reply":"2025-04-24T06:03:51.282342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# define features\npciat_cols = [col for col in train_full.columns if 'PCIAT-' in col and col != 'PCIAT-Season'] # PCIAT score is needed later\ny_model_col = \"PCIAT-PCIAT_Total\" # Intermediate score target\ny_comp_col = \"sii\" # Final competition target","metadata":{"id":"b5b1d159","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:51.284324Z","iopub.execute_input":"2025-04-24T06:03:51.284680Z","iopub.status.idle":"2025-04-24T06:03:51.302078Z","shell.execute_reply.started":"2025-04-24T06:03:51.284632Z","shell.execute_reply":"2025-04-24T06:03:51.301067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Features available in both train and test after processing\nbase_features = [f for f in test.columns if f not in ['id']]\n# Ensure all base_features are also in train\nfeatures = [f for f in base_features if f in train_full.columns]\n# Make sure no target/ID columns slipped through\nfeatures = [f for f in features if f not in [y_comp_col, y_model_col]]","metadata":{"id":"0bda338a","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:51.302983Z","iopub.execute_input":"2025-04-24T06:03:51.303202Z","iopub.status.idle":"2025-04-24T06:03:51.328075Z","shell.execute_reply.started":"2025-04-24T06:03:51.303185Z","shell.execute_reply":"2025-04-24T06:03:51.326809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Number of features: {len(features)}\")\nprint(f\"Missing features in test compared to train_full (should be only targets/PCIAT): {set(train_full.columns) - set(test.columns) - set(['id'])}\")","metadata":{"id":"f0a61daa","outputId":"76cc269f-163d-4296-e5bf-0cc9b9f8da44","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:51.329205Z","iopub.execute_input":"2025-04-24T06:03:51.329599Z","iopub.status.idle":"2025-04-24T06:03:51.348612Z","shell.execute_reply.started":"2025-04-24T06:03:51.329565Z","shell.execute_reply":"2025-04-24T06:03:51.347547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# separate labelled and unlabelled data\ntrain_labeled = train_full[train_full[y_comp_col].notna()].copy()\ntrain_unlabeled = train_full[train_full[y_comp_col].isna()].copy()\nprint(f\"Labeled samples: {len(train_labeled)}, Unlabeled samples: {len(train_unlabeled)}\")","metadata":{"id":"a5b8e2fa","outputId":"df8e196d-caf5-46ad-9d0e-1960f32def53","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:51.349593Z","iopub.execute_input":"2025-04-24T06:03:51.349884Z","iopub.status.idle":"2025-04-24T06:03:51.373617Z","shell.execute_reply.started":"2025-04-24T06:03:51.349864Z","shell.execute_reply":"2025-04-24T06:03:51.372694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# impute using KNN Imputer (instead of Impute_With_Model)\nimputer = KNNImputer(n_neighbors=7) # Adjust n_neighbors as needed","metadata":{"id":"d193d737","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:51.374631Z","iopub.execute_input":"2025-04-24T06:03:51.374949Z","iopub.status.idle":"2025-04-24T06:03:51.381243Z","shell.execute_reply.started":"2025-04-24T06:03:51.374929Z","shell.execute_reply":"2025-04-24T06:03:51.380311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#fit on labelled data only\nimputer.fit(train_labeled[features])","metadata":{"id":"7f78f710","outputId":"d4e412bd-5c70-479e-880a-e143bf132c27","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:51.382223Z","iopub.execute_input":"2025-04-24T06:03:51.382524Z","iopub.status.idle":"2025-04-24T06:03:51.421627Z","shell.execute_reply.started":"2025-04-24T06:03:51.382495Z","shell.execute_reply":"2025-04-24T06:03:51.420578Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labeled[features] = imputer.transform(train_labeled[features])\ntest[features] = imputer.transform(test[features])","metadata":{"id":"ac735d0b","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:51.422643Z","iopub.execute_input":"2025-04-24T06:03:51.423588Z","iopub.status.idle":"2025-04-24T06:03:55.012662Z","shell.execute_reply.started":"2025-04-24T06:03:51.423567Z","shell.execute_reply":"2025-04-24T06:03:55.011810Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_params = {\n    'objective': 'poisson', # Good for scores\n    'metric': 'rmse', # Monitor RMSE during potential early stopping\n    'n_estimators': 500, # Increase estimators, use early stopping\n    'learning_rate': 0.03,\n    'feature_fraction': 0.8, # Equivalent to colsample_bytree\n    'bagging_fraction': 0.8, # Equivalent to subsample\n    'bagging_freq': 1,\n    'lambda_l1': 0.1,\n    'lambda_l2': 0.1,\n    'num_leaves': 31, # Default is 31, adjust based on max_depth\n    'verbose': -1,\n    'n_jobs': -1,\n    'seed': SEED,\n    'boosting_type': 'gbdt',\n    'max_depth': 5, # Limit depth\n    'min_child_samples': 20, # Equivalent to min_data_in_leaf\n}","metadata":{"id":"d730cee7","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:55.013581Z","iopub.execute_input":"2025-04-24T06:03:55.013858Z","iopub.status.idle":"2025-04-24T06:03:55.019123Z","shell.execute_reply.started":"2025-04-24T06:03:55.013839Z","shell.execute_reply":"2025-04-24T06:03:55.018021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# impute data for pseudo-labeling\nif not train_unlabeled.empty:\n     # Check if unlabeled data has missing values in features\n    if train_unlabeled[features].isnull().sum().sum() > 0:\n        train_unlabeled[features] = imputer.transform(train_unlabeled[features])\n    else:\n        print(\"No NaNs to impute in unlabeled feature set.\")","metadata":{"id":"cffa6936","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:55.020112Z","iopub.execute_input":"2025-04-24T06:03:55.020439Z","iopub.status.idle":"2025-04-24T06:03:58.000071Z","shell.execute_reply.started":"2025-04-24T06:03:55.020411Z","shell.execute_reply":"2025-04-24T06:03:57.999099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# psuedo labelling\nif not train_unlabeled.empty:\n    print(\"Performing pseudo-labeling...\")\n    # 1. Train an initial model on labeled data (e.g., LGBM)\n    # Using pre-defined params for simplicity, tune if needed\n    pseudo_label_model = LGBMRegressor(**lgb_params, random_state=SEED+1, verbosity=-1) # Use different seed\n\n    # Optional: Use PCIAT score as target if it's cleaner\n    if y_model_col in train_labeled.columns and not train_labeled[y_model_col].isnull().all():\n         print(f\"Using {y_model_col} for initial pseudo-label model training.\")\n         weights_pseudo = calculate_weights(train_labeled[y_model_col]) # Reuse weight function\n         pseudo_label_model.fit(train_labeled[features], train_labeled[y_model_col], sample_weight=weights_pseudo)\n         pseudo_scores = pseudo_label_model.predict(train_unlabeled[features])\n         # Use consistent thresholds (e.g., average from previous CV or base)\n         # Find average thresholds from a preliminary CV run if possible, otherwise use base\n         # avg_thresholds_pseudo = np.mean(np.array(preliminary_thresholds), axis=0)\n         avg_thresholds_pseudo = base_thresholds # Fallback\n         pseudo_labels = round_with_thresholds(pseudo_scores, avg_thresholds_pseudo)\n    else:\n        # Fallback: Train directly on 'sii' if PCIAT score isn't reliable/available\n        # This might be less accurate as input 'sii' is ordinal\n         print(f\"Warning: Falling back to training pseudo-label model directly on {y_comp_col}.\")\n         pseudo_label_model.fit(train_labeled[features], train_labeled[y_comp_col])\n         pseudo_labels = pseudo_label_model.predict(train_unlabeled[features]).round().astype(int).clip(0, 3) # Predict sii directly\n\n    # 2. Add pseudo-labels to unlabeled data\n    train_unlabeled[y_comp_col] = pseudo_labels\n    # Optional: Add PCIAT score prediction if that was used\n    if y_model_col in train_labeled.columns and not train_labeled[y_model_col].isnull().all():\n        train_unlabeled[y_model_col] = pseudo_scores # Add the predicted score too\n\n    # 3. Combine labeled and pseudo-labeled data\n    train_combined = pd.concat([train_labeled, train_unlabeled], ignore_index=True)\n    print(f\"Combined training data size: {len(train_combined)}\")\n\n    # Optional: Assign lower weight to pseudo-labeled samples during final training\n    sample_indices = np.arange(len(train_combined))\n    pseudo_labeled_indices = sample_indices[len(train_labeled):]\n    # Create a weight array (e.g., 1 for original, 0.5 for pseudo)\n    # sample_weight_pseudo = np.ones(len(train_combined))\n    # sample_weight_pseudo[pseudo_labeled_indices] = 0.5 # Adjust weight factor\n\nelse:\n    print(\"No unlabeled data found, skipping pseudo-labeling.\")\n    train_combined = train_labeled.copy()\n    # sample_weight_pseudo = np.ones(len(train_combined)) # All weights are 1\n\n# Use train_combined for subsequent training\ntrain = train_combined # Rename for consistency with the rest of the script","metadata":{"id":"797ed06a","outputId":"e128aa29-3e70-4c5d-b8a8-6164d8d67973","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:03:58.001113Z","iopub.execute_input":"2025-04-24T06:03:58.001428Z","iopub.status.idle":"2025-04-24T06:04:00.296172Z","shell.execute_reply.started":"2025-04-24T06:03:58.001402Z","shell.execute_reply":"2025-04-24T06:04:00.294719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# threshold optimization\ndef round_with_thresholds(raw_preds, thresholds):\n    # Ensure thresholds are sorted\n    thresholds = np.sort(thresholds)\n    return np.where(raw_preds < thresholds[0], int(0),\n                    np.where(raw_preds < thresholds[1], int(1),\n                             np.where(raw_preds < thresholds[2], int(2), int(3))))","metadata":{"id":"4317e45b","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:04:00.298716Z","iopub.execute_input":"2025-04-24T06:04:00.299221Z","iopub.status.idle":"2025-04-24T06:04:00.306792Z","shell.execute_reply.started":"2025-04-24T06:04:00.299182Z","shell.execute_reply":"2025-04-24T06:04:00.305830Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def optimize_thresholds(y_true, raw_preds, start_vals=[0.5, 1.5, 2.5]):\n    # Ensure y_true and raw_preds have the same length\n    if len(y_true) != len(raw_preds):\n         raise ValueError(f\"Length mismatch: y_true ({len(y_true)}) != raw_preds ({len(raw_preds)})\")\n\n    # Check for NaNs\n    if np.isnan(raw_preds).any():\n        print(\"Warning: NaNs found in raw_preds during threshold optimization. Replacing with mean.\")\n        raw_preds = np.nan_to_num(raw_preds, nan=np.nanmean(raw_preds))\n\n    def fun(thresholds, y_true, raw_preds):\n        # Ensure thresholds are sorted within the function\n        sorted_thresholds = np.sort(thresholds)\n        rounded_preds = round_with_thresholds(raw_preds, sorted_thresholds)\n        # Return negative kappa score for minimization\n        return -cohen_kappa_score(y_true, rounded_preds, weights='quadratic')\n\n    # Use bounds to ensure thresholds remain ordered and reasonable\n    bnds = [(min(raw_preds)-1, max(raw_preds)+1) for _ in range(len(start_vals))]\n    # Add constraints to keep thresholds ordered t0 < t1 < t2\n    constraints = ({'type': 'ineq', 'fun': lambda x: x[1] - x[0] - 1e-6}, # t1 > t0\n                   {'type': 'ineq', 'fun': lambda x: x[2] - x[1] - 1e-6}) # t2 > t1\n\n    res = minimize(fun, x0=np.sort(start_vals), args=(y_true, raw_preds), method='SLSQP', bounds=bnds, constraints=constraints) # Changed method to SLSQP for bounds/constraints\n\n    if not res.success:\n         print(f\"Threshold optimization failed: {res.message}. Returning start_vals.\")\n         return np.sort(start_vals) # Return sorted start values on failure\n\n    return np.sort(res.x) # Return sorted thresholds","metadata":{"id":"d8db8fe8","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:04:00.307887Z","iopub.execute_input":"2025-04-24T06:04:00.308113Z","iopub.status.idle":"2025-04-24T06:04:00.324473Z","shell.execute_reply.started":"2025-04-24T06:04:00.308096Z","shell.execute_reply":"2025-04-24T06:04:00.323405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_weights(series):\n    # Create bins for the target variable and assign weights based on frequency\n    # Handle potential NaNs in the series\n    if series.isnull().any():\n        print(\"Warning: NaNs found in target series for weight calculation. Dropping NaNs.\")\n        series = series.dropna()\n    if series.empty:\n        print(\"Warning: Empty series provided for weight calculation. Returning uniform weights.\")\n        return pd.Series(1.0, index=series.index) # Or handle as error\n\n    # Increase robustness for series with few unique values or skewed distributions\n    try:\n        # Use pd.qcut for potentially better binning with skewed data\n        bins = pd.qcut(series, q=min(10, series.nunique()), labels=False, duplicates='drop')\n    except ValueError:\n        # Fallback to cut if qcut fails (e.g., too few unique values)\n        bins = pd.cut(series, bins=min(10, series.nunique()), labels=False, include_lowest=True)\n\n    # If all values fall into one bin, return uniform weights\n    if bins.nunique() <= 1:\n        print(\"Warning: Target series has low variance. Returning uniform weights.\")\n        return pd.Series(1.0, index=series.index)\n\n    weights_df = bins.value_counts().reset_index()\n    weights_df.columns = ['target_bins', 'count']\n\n    # Prevent division by zero if a bin somehow has zero count (shouldn't happen with value_counts)\n    weights_df['count'] = weights_df['count'].replace(0, 1)\n\n    weights_df['weight'] = 1 / weights_df['count']\n    weight_map = weights_df.set_index('target_bins')['weight'].to_dict()\n\n    # Map weights back, handle potential missing bins if qcut/cut produced fewer than expected\n    final_weights = bins.map(weight_map).fillna(1.0) # Fill potential NaNs with 1\n\n    # Normalize weights\n    mean_weight = final_weights.mean()\n    if mean_weight == 0: # Prevent division by zero\n        print(\"Warning: Mean weight is zero. Returning uniform weights.\")\n        return pd.Series(1.0, index=series.index)\n\n    normalized_weights = final_weights / mean_weight\n    return normalized_weights","metadata":{"id":"5b4f35ec","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:04:00.325664Z","iopub.execute_input":"2025-04-24T06:04:00.326180Z","iopub.status.idle":"2025-04-24T06:04:00.344210Z","shell.execute_reply.started":"2025-04-24T06:04:00.326156Z","shell.execute_reply":"2025-04-24T06:04:00.343151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cross validation\ndef cross_validate_regressor(model_, data, features, score_col, index_col, cv, sample_weights_series=None, verbose=False):\n    \"\"\"\n    Perform cross-validation with a regressor model.\n    Predicts the score_col, optimizes thresholds, and calculates QWK on index_col.\n\n    Returns:\n    float: Mean Kappa score across all folds.\n    array: Out-of-fold score predictions.\n    array: Out-of-fold index predictions (after thresholding).\n    list: List of optimized thresholds per fold.\n    \"\"\"\n    kappa_scores = []\n    oof_score_predictions = np.zeros(len(data))\n    oof_index_predictions = np.zeros(len(data), dtype=int)\n    fold_thresholds = []\n    models = [] # Store models from each fold\n\n    for fold_idx, (train_idx, val_idx) in enumerate(cv.split(data, data[index_col])):\n        X_train, X_val = data[features].iloc[train_idx], data[features].iloc[val_idx]\n        y_train_score = data[score_col].iloc[train_idx]\n        y_train_index = data[index_col].iloc[train_idx]\n        # y_val_score = data[score_col].iloc[val_idx] # Not strictly needed for score calculation here\n        y_val_index = data[index_col].iloc[val_idx]\n\n        current_model = clone(model_) # Clone model for each fold\n\n        # Handle sample weights\n        fit_params = {}\n        if sample_weights_series is not None:\n            # Ensure weights are aligned with training indices\n            weights_train = sample_weights_series.iloc[train_idx].values\n            # Check for NaNs or zeros in weights\n            if np.isnan(weights_train).any() or np.isinf(weights_train).any() or (weights_train <= 0).any():\n                 print(f\"Warning: Invalid weights found in fold {fold_idx}. Using uniform weights for this fold.\")\n            else:\n                # CatBoost uses 'sample_weight', LGBM/XGB use 'sample_weight' in fit\n                fit_params['sample_weight'] = weights_train\n\n\n        # Train model\n        try:\n             # Special handling for CatBoost verbosity if needed\n            if isinstance(current_model, CatBoostRegressor):\n                 current_model.fit(X_train, y_train_score, eval_set=[(X_val, data[score_col].iloc[val_idx])], early_stopping_rounds=50, verbose=0, **fit_params)\n            else:\n                 current_model.fit(X_train, y_train_score, **fit_params)\n        except Exception as e:\n             print(f\"Error fitting model in fold {fold_idx}: {e}\")\n             # Handle error, maybe skip fold or use default predictions?\n             continue # Skip this fold on error\n\n        # Predict scores on validation set\n        y_pred_val_score = current_model.predict(X_val)\n        oof_score_predictions[val_idx] = y_pred_val_score\n\n        # Optimize thresholds using validation scores and true validation index\n        # Use train predictions and train index for optimizing thresholds to avoid overfitting thresholds to val set?\n        # This is debatable. Original code used train set predictions. Let's stick to that for consistency.\n        y_pred_train_score = current_model.predict(X_train)\n        # Ensure no NaNs in prediction used for thresholding\n        y_pred_train_score_clean = np.nan_to_num(y_pred_train_score, nan=np.nanmean(y_pred_train_score))\n\n        # Use a robust starting point for thresholds based on score distribution\n        start_thresholds = np.percentile(y_train_score.dropna(), [25, 50, 75]) # Use percentiles as start\n\n        # Check if start_thresholds are distinct\n        if len(np.unique(start_thresholds)) < 3:\n            start_thresholds = base_thresholds # Fallback to base if percentiles are degenerate\n\n\n        # Optimize thresholds on TRAIN predictions vs TRAIN index\n        try:\n             t_opt = optimize_thresholds(y_train_index, y_pred_train_score_clean, start_vals=start_thresholds)\n             fold_thresholds.append(t_opt)\n        except ValueError as e:\n             print(f\"Error optimizing thresholds in fold {fold_idx}: {e}. Using base thresholds.\")\n             t_opt = base_thresholds # Use base thresholds if optimization fails\n             fold_thresholds.append(t_opt)\n\n\n        # Apply optimized thresholds to validation score predictions\n        y_pred_val_index = round_with_thresholds(y_pred_val_score, t_opt)\n        oof_index_predictions[val_idx] = y_pred_val_index\n\n        # Calculate Kappa score for the fold\n        kappa_score = cohen_kappa_score(y_val_index, y_pred_val_index, weights='quadratic')\n        kappa_scores.append(kappa_score)\n        models.append(current_model) # Store the trained model\n\n        if verbose:\n            print(f\"Fold {fold_idx}: Optimized Kappa = {kappa_score:.4f}, Thresholds = {np.round(t_opt, 2)}\")\n\n    mean_kappa = np.mean(kappa_scores)\n    std_kappa = np.std(kappa_scores)\n    if verbose:\n        print(f\"\\n## Mean CV Kappa Score: {mean_kappa:.4f} ##\")\n        print(f\"## Std CV Kappa Score: {std_kappa:.4f} ##\\n\")\n\n    # Calculate overall OOF score using the index predictions from each fold\n    overall_oof_kappa = cohen_kappa_score(data[index_col], oof_index_predictions, weights='quadratic')\n    if verbose:\n        print(f\"## Overall OOF Kappa Score: {overall_oof_kappa:.4f} ##\\n\")\n\n\n    return mean_kappa, oof_score_predictions, oof_index_predictions, fold_thresholds, models, overall_oof_kappa","metadata":{"id":"944f64d0","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:04:00.345405Z","iopub.execute_input":"2025-04-24T06:04:00.346329Z","iopub.status.idle":"2025-04-24T06:04:00.372082Z","shell.execute_reply.started":"2025-04-24T06:04:00.346296Z","shell.execute_reply":"2025-04-24T06:04:00.370945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cross val for ordinal\ndef cross_validate_ordinal(model_, data, features, index_col, cv, verbose=False):\n    \"\"\"\n    Perform cross-validation with an ordinal classification model (from mord).\n    Directly predicts the index_col.\n\n    Returns:\n    float: Mean Kappa score across all folds.\n    array: Out-of-fold index predictions.\n    list: List of trained models per fold.\n    float: Overall OOF Kappa score.\n    \"\"\"\n    kappa_scores = []\n    oof_index_predictions = np.zeros(len(data), dtype=int)\n    models = []\n\n    for fold_idx, (train_idx, val_idx) in enumerate(cv.split(data, data[index_col])):\n        X_train, X_val = data[features].iloc[train_idx], data[features].iloc[val_idx]\n        y_train_index = data[index_col].iloc[train_idx]\n        y_val_index = data[index_col].iloc[val_idx]\n\n        current_model = clone(model_)\n        current_model.fit(X_train, y_train_index)\n\n        y_pred_val_index = current_model.predict(X_val)\n        oof_index_predictions[val_idx] = y_pred_val_index\n\n        kappa_score = cohen_kappa_score(y_val_index, y_pred_val_index, weights='quadratic')\n        kappa_scores.append(kappa_score)\n        models.append(current_model)\n\n        if verbose:\n            print(f\"Fold {fold_idx}: Ordinal Kappa = {kappa_score:.4f}\")\n\n    mean_kappa = np.mean(kappa_scores)\n    std_kappa = np.std(kappa_scores)\n\n    if verbose:\n        print(f\"\\n## Mean CV Kappa Score (Ordinal): {mean_kappa:.4f} ##\")\n        print(f\"## Std CV Kappa Score (Ordinal): {std_kappa:.4f} ##\\n\")\n\n    # Calculate overall OOF score\n    overall_oof_kappa = cohen_kappa_score(data[index_col], oof_index_predictions, weights='quadratic')\n    if verbose:\n        print(f\"## Overall OOF Kappa Score (Ordinal): {overall_oof_kappa:.4f} ##\\n\")\n\n\n    return mean_kappa, oof_index_predictions, models, overall_oof_kappa","metadata":{"id":"1172cb7c","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:04:00.373130Z","iopub.execute_input":"2025-04-24T06:04:00.373387Z","iopub.status.idle":"2025-04-24T06:04:00.397265Z","shell.execute_reply.started":"2025-04-24T06:04:00.373369Z","shell.execute_reply":"2025-04-24T06:04:00.396145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- MODIFIED Optuna Objective Function ---\ndef objective(trial, model_type, X, features, score_col, index_col, cv, sample_weights_series=None):\n    \"\"\"\n    Optuna objective function using the new cross-validation functions.\n    Performs ONE cross-validation run per trial.\n    \"\"\"\n    is_ordinal = False # Flag to check if it's an ordinal model trial\n\n    # --- Parameter Space Definitions ---\n\n    # LightGBM\n    if model_type == 'lightgbm':\n        params = {\n            'objective': trial.suggest_categorical('objective', ['poisson', 'regression_l1', 'rmse']),\n            'metric': 'rmse',\n            'random_state': SEED,\n            'n_estimators': trial.suggest_int('n_estimators', 100, 1000),\n            'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.1, log=True),\n            'num_leaves': trial.suggest_int('num_leaves', 20, 60),\n            'max_depth': trial.suggest_int('max_depth', 3, 7),\n            'subsample': trial.suggest_float('subsample', 0.6, 0.9),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.6, 0.9),\n            'reg_alpha': trial.suggest_float('reg_alpha', 1e-3, 1.0, log=True),\n            'reg_lambda': trial.suggest_float('reg_lambda', 1e-3, 1.0, log=True),\n            'min_child_samples': trial.suggest_int('min_child_samples', 5, 50),\n            'n_jobs': -1,\n            'verbosity': -1,\n        }\n        model = LGBMRegressor(**params)\n        is_ordinal = False\n\n    # --- NEW: XGBoost ---\n    elif model_type == 'xgboost':\n        params = {\n            'objective': trial.suggest_categorical('objective', ['reg:squarederror', 'reg:pseudohubererror']), # Could add reg:tweedie if desired\n            'eval_metric': 'rmse',\n            'eta': trial.suggest_float('eta', 0.01, 0.1, log=True),  # learning_rate\n            'max_depth': trial.suggest_int('max_depth', 3, 7),\n            'subsample': trial.suggest_float('subsample', 0.6, 0.9),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.6, 0.9),\n            'min_child_weight': trial.suggest_int('min_child_weight', 1, 10),\n            'gamma': trial.suggest_float('gamma', 0.0, 0.5), # min_split_loss\n            'lambda': trial.suggest_float('lambda', 1e-3, 1.0, log=True),  # L2 reg\n            'alpha': trial.suggest_float('alpha', 1e-3, 1.0, log=True),  # L1 reg\n            'seed': SEED,\n            'n_jobs': -1,\n            # Add 'tree_method': 'hist' if using GPU or want faster histogram method\n        }\n        # Optional: Tweedie specific parameter\n        # if params['objective'] == 'reg:tweedie':\n        #     params['tweedie_variance_power'] = trial.suggest_float('tweedie_variance_power', 1.0, 2.0)\n\n        model = XGBRegressor(**params)\n        is_ordinal = False\n\n    # --- NEW: CatBoost ---\n    elif model_type == 'catboost':\n        params = {\n            'loss_function': trial.suggest_categorical('loss_function', ['RMSE', 'MAE', 'Poisson']), # Could add Tweedie\n            'iterations': trial.suggest_int('iterations', 100, 1000), # n_estimators\n            'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.1, log=True),\n            'depth': trial.suggest_int('depth', 4, 8),\n            'l2_leaf_reg': trial.suggest_float('l2_leaf_reg', 1.0, 10.0, log=True), # Lambda L2 reg\n            'subsample': trial.suggest_float('subsample', 0.6, 0.9), # Only if bootstrapping_type is Bayesian/Bernoulli\n            'colsample_bylevel': trial.suggest_float('colsample_bylevel', 0.6, 0.9), # Feature fraction\n            'random_seed': SEED,\n            'verbose': 0, # Keep quiet during optuna\n            'early_stopping_rounds': 50, # Use early stopping during CV folds\n            'border_count': trial.suggest_int('border_count', 32, 255), # Controls discretization\n             #'boosting_type': 'Plain', # Could try 'Ordered' but often slower\n             #'bootstrap_type': trial.suggest_categorical('bootstrap_type', ['Bayesian', 'Bernoulli', 'MVS']) # If using subsample\n        }\n        # Optional: Tweedie specific parameter\n        # if 'Tweedie' in params['loss_function']:\n        #     params['loss_function'] = f\"Tweedie:variance_power={trial.suggest_float('tweedie_variance_power', 1.0, 2.0)}\"\n        # Optional: Subsample requires bootstrap_type\n        # if params.get('bootstrap_type') in ['Bayesian', 'Bernoulli']:\n        #     params['subsample'] = trial.suggest_float('subsample', 0.6, 0.95)\n\n\n        model = CatBoostRegressor(**params)\n        is_ordinal = False\n\n    # Ordinal Ridge\n    elif model_type == 'ordinal_ridge' and mord is not None:\n        params = {\n            'alpha': trial.suggest_float('alpha', 0.1, 10.0, log=True),\n            'fit_intercept': trial.suggest_categorical('fit_intercept', [True, False]),\n        }\n        model = mord.OrdinalRidge(**params)\n        is_ordinal = True\n\n    # Unsupported type\n    else:\n        if model_type == 'ordinal_ridge' and mord is None:\n             print(\"Skipping ordinal_ridge trial as 'mord' library is not installed.\")\n             return -1.0\n        print(f\"Warning: model_type '{model_type}' not recognized in objective function or dependencies missing.\")\n        return -1.0 # Return poor score for unrecognized types\n\n\n    # --- Perform Cross-Validation (Single Run) ---\n    try:\n        if is_ordinal:\n            mean_fold_kappa, _, _, overall_oof_kappa = cross_validate_ordinal(\n                model, X, features, index_col, cv, verbose=False\n            )\n        else:\n            mean_fold_kappa, _, _, _, _, overall_oof_kappa = cross_validate_regressor(\n                model, X, features, score_col, index_col, cv,\n                sample_weights_series=sample_weights_series, verbose=False\n            )\n        # Return the score Optuna should optimize\n        return overall_oof_kappa\n\n    except Exception as e:\n        print(f\"Error during cross-validation for trial {trial.number} ({model_type}): {e}\")\n        # Important: Return a poor score if CV fails, so Optuna doesn't favor failing parameters\n        return -1.0 # Or appropriate value indicating failure","metadata":{"id":"e5df062c","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:04:00.398366Z","iopub.execute_input":"2025-04-24T06:04:00.398770Z","iopub.status.idle":"2025-04-24T06:04:00.419205Z","shell.execute_reply.started":"2025-04-24T06:04:00.398746Z","shell.execute_reply":"2025-04-24T06:04:00.418194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#def run_optimization\ndef run_optimization(X, features, score_col, index_col, model_type, n_trials=30, cv=None, sample_weights_series=None): # Accepts Series or None\n    \"\"\"\n    Runs Optuna optimization using the objective function.\n\n    Args:\n        X (pd.DataFrame): Training data.\n        features (list): List of feature names.\n        score_col (str): Name of the intermediate score column (for regressors).\n        index_col (str): Name of the final target index column (sii).\n        model_type (str): Type of model to optimize ('lightgbm', 'ordinal_ridge', etc.).\n        n_trials (int): Number of Optuna trials.\n        cv (cross-validation generator): The CV object (e.g., StratifiedKFold).\n        sample_weights_series (pd.Series, optional): Series containing sample weights. Defaults to None.\n    \"\"\"\n    study = optuna.create_study(direction=\"maximize\")\n\n    # Pass the actual sample_weights_series (or None) to the objective function\n    study.optimize(lambda trial: objective(trial, model_type, X, features, score_col, index_col, cv, sample_weights_series),\n                   n_trials=n_trials)\n\n    print(f\"\\nOptuna finished for {model_type}\")\n    print(f\"Best params found: {study.best_params}\")\n    # study.best_value holds the best overall_oof_kappa found during optimization\n    print(f\"Best Overall OOF Kappa score achieved: {study.best_value:.4f}\")\n    return study.best_params","metadata":{"id":"ff118151","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:04:00.420304Z","iopub.execute_input":"2025-04-24T06:04:00.420623Z","iopub.status.idle":"2025-04-24T06:04:00.442437Z","shell.execute_reply.started":"2025-04-24T06:04:00.420604Z","shell.execute_reply":"2025-04-24T06:04:00.441367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# feature subsets\n# Replace if subsets for features have been selected\n\n# List of manually selected features NOT to include\nexclude = [\n    \"PC_9\", \"PC_12\", \"Fitness_Endurance-Max_Stage\", \"Basic_Demos-Sex\", \"BMI_mean_norm\", \"PC_11\",\n    \"PC_8\", \"FGC_Zones_min\", \"Physical-Systolic_BP\", \"PC_4\", \"BIA-BIA_FMI\", \"BIA-BIA_LST\", \"Physical-Diastolic_BP\",\n    \"BIA-BIA_ECW\", \"Fitness_Endurance-Time_Mins\", \"PAQ_C-PAQ_C_Total\", \"PC_10\", \"BIA-BIA_Fat\", \"FFM_norm\", \"PC_14\", \"PC_7\"\n]\n\nreduced_features = [f for f in features if f not in exclude]\n\nlgb_features = reduced_features\nxgb_features = reduced_features\ncat_features = reduced_features\nprint(len(reduced_features))","metadata":{"id":"eb933e53","outputId":"72c0368e-6bf9-4e2f-8177-719c7ba6b4c6","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:04:00.443603Z","iopub.execute_input":"2025-04-24T06:04:00.444170Z","iopub.status.idle":"2025-04-24T06:04:00.463129Z","shell.execute_reply.started":"2025-04-24T06:04:00.444146Z","shell.execute_reply":"2025-04-24T06:04:00.462169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_features = features # Start with all features after cleaning/eng\nprint(f\"Using {len(selected_features)} features for modeling.\")","metadata":{"id":"ac33284f","outputId":"914c9a2b-61fd-421b-8dcc-7074efc4850b","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:04:00.464092Z","iopub.execute_input":"2025-04-24T06:04:00.465028Z","iopub.status.idle":"2025-04-24T06:04:00.483786Z","shell.execute_reply.started":"2025-04-24T06:04:00.465006Z","shell.execute_reply":"2025-04-24T06:04:00.482425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model Parameters\n\n# lgb_params = {\n#     'objective': 'poisson', # Good for scores\n#     'metric': 'rmse', # Monitor RMSE during potential early stopping\n#     'n_estimators': 500, # Increase estimators, use early stopping\n#     'learning_rate': 0.03,\n#     'feature_fraction': 0.8, # Equivalent to colsample_bytree\n#     'bagging_fraction': 0.8, # Equivalent to subsample\n#     'bagging_freq': 1,\n#     'lambda_l1': 0.1,\n#     'lambda_l2': 0.1,\n#     'num_leaves': 31, # Default is 31, adjust based on max_depth\n#     'verbose': -1,\n#     'n_jobs': -1,\n#     'seed': SEED,\n#     'boosting_type': 'gbdt',\n#     'max_depth': 5, # Limit depth\n#     'min_child_samples': 20, # Equivalent to min_data_in_leaf\n# }\n\nxgb_params = {\n    'objective': 'reg:squarederror', # Simpler objective, rely on thresholding\n    'eval_metric': 'rmse',\n    'eta': 0.03, # learning_rate\n    'max_depth': 4,\n    'subsample': 0.7,\n    'colsample_bytree': 0.7,\n    'min_child_weight': 1,\n    'gamma': 0.0,\n    'lambda': 1, # L2 reg\n    'alpha': 0, # L1 reg\n    'seed': SEED\n}\n\n\ncat_params = {\n    'loss_function': 'RMSE',\n    'eval_metric': 'RMSE',\n    'iterations': 500, # Increase estimators, use early stopping\n    'learning_rate': 0.04,\n    'depth': 5,\n    'l2_leaf_reg': 3,\n    'subsample': 0.7,\n    'colsample_bylevel': 0.7, # CatBoost specific feature sampling\n    'random_seed': SEED,\n    'verbose': 0, # Suppress verbose output during CV\n    'early_stopping_rounds': 50 # Use early stopping\n}\n\nxtrees_params = {\n    'n_estimators': 300, # Slightly reduced\n    'max_depth': 12, # Control complexity\n    'min_samples_leaf': 15,\n    'min_samples_split': 10,\n    'random_state': SEED,\n    'n_jobs': -1\n}\n\nordinal_params = {\n    'alpha': 1.0,\n    'fit_intercept': True\n}","metadata":{"id":"50631866","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:04:00.485342Z","iopub.execute_input":"2025-04-24T06:04:00.485981Z","iopub.status.idle":"2025-04-24T06:04:00.502069Z","shell.execute_reply.started":"2025-04-24T06:04:00.485938Z","shell.execute_reply":"2025-04-24T06:04:00.500710Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if optimize_params:\n    print(\"\\n--- Running Optuna Hyperparameter Tuning ---\")\n    # Define a CV split specifically for Optuna (can be fewer splits for speed)\n    kf_for_optuna = StratifiedKFold(n_splits=5, shuffle=True, random_state=SEED)\n\n    # Calculate weights based on the target score distribution IN THE FINAL 'train' DATAFRAME\n    if y_model_col in train.columns and not train[y_model_col].isnull().all():\n        weights_for_optuna = calculate_weights(train[y_model_col])\n    else:\n        weights_for_optuna = calculate_weights(train[y_comp_col]) # Fallback\n\n    # Tune LightGBM\n    print(\"\\nOptimizing LightGBM...\")\n    best_lgb_params = run_optimization(\n        train, selected_features, y_model_col, y_comp_col,\n        model_type='lightgbm',\n        n_trials=n_optuna_trials,\n        cv=kf_for_optuna,\n        sample_weights_series=weights_for_optuna\n    )\n    lgb_params.update(best_lgb_params) # Update the main dict with best params found\n    print(\"Updated LGBM Params:\", lgb_params) # Print updated params\n\n    print(\"\\nOptimizing XGBoost...\")\n    best_xgb_params = run_optimization(\n        train, selected_features, y_model_col, y_comp_col, # Pass necessary args\n        model_type='xgboost',                             # Specify model type\n        n_trials=n_optuna_trials,\n        cv=kf_for_optuna,\n        sample_weights_series=weights_for_optuna         # Pass weights\n    )\n    xgb_params.update(best_xgb_params)                     # Update the dict\n    print(\"Updated XGBoost Params:\", xgb_params)\n\n    print(\"\\nOptimizing CatBoost...\")\n    best_cat_params = run_optimization(\n        train, selected_features, y_model_col, y_comp_col, # Pass necessary args\n        model_type='catboost',                            # Specify model type\n        n_trials=n_optuna_trials,\n        cv=kf_for_optuna,\n        sample_weights_series=weights_for_optuna         # Pass weights\n    )\n    cat_params.update(best_cat_params)                     # Update the dict\n    print(\"Updated CatBoost Params:\", cat_params)\n\n    # Tune Ordinal Ridge (if mord installed and objective defined)\n    if mord is not None:\n         print(\"\\nOptimizing Ordinal Ridge...\")\n         # Make sure 'ordinal_ridge' is handled in the objective function\n         best_ord_params = run_optimization(\n             train, selected_features, y_model_col, y_comp_col, # score_col not used but passed\n             model_type='ordinal_ridge',\n             n_trials=n_optuna_trials // 2, # Maybe fewer trials for simpler model\n             cv=kf_for_optuna,\n             sample_weights_series=None # OrdinalRidge doesn't use sample_weight in fit\n         )\n         # If you had an ordinal_params dict, update it here. Otherwise, store separately.\n         ordinal_params.update(best_ord_params) # Example\n         print(f\"Best OrdinalRidge params: {best_ord_params}\") # Store/use as needed\n\n\n    print(\"\\n--- Optuna Tuning Finished ---\")\n    print(\"Updated LGBM Params:\", lgb_params)\n    # Print other updated params...\n\nelse:\n    print(\"\\nSkipping Optuna tuning, using predefined parameters.\")","metadata":{"id":"361cbb0f","outputId":"1468457e-a557-42e9-a7cf-05dcfe4f659f","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:04:00.503228Z","iopub.execute_input":"2025-04-24T06:04:00.503564Z","iopub.status.idle":"2025-04-24T06:13:51.751858Z","shell.execute_reply.started":"2025-04-24T06:04:00.503538Z","shell.execute_reply":"2025-04-24T06:13:51.751033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define models\nlgb_model = LGBMRegressor(**lgb_params)\nxgb_model = XGBRegressor(**xgb_params)\ncat_model = CatBoostRegressor(**cat_params)\nxtrees_model = ExtraTreesRegressor(**xtrees_params)\n\n# !! NEW: Add Ordinal Model !!\nordinal_model = None\nif mord is not None:\n    try:\n        # Initialize using the potentially updated ordinal_params dict\n        ordinal_model = mord.OrdinalRidge(**ordinal_params)\n        print(f\"Initializing OrdinalRidge with parameters: {ordinal_params}\")\n    except Exception as e:\n        # Fallback if initialization fails for some reason\n        print(f\"Error initializing OrdinalRidge with params {ordinal_params}: {e}\")\n        print(\"Falling back to basic OrdinalRidge initialization.\")\n        ordinal_model = mord.OrdinalRidge() # Simplest fallback\n\n\n# --- Cross-Validation Execution ---\nkf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n# Calculate weights based on the target score distribution in the combined training data\n# Use PCIAT score if reliable, otherwise fallback to sii (though less ideal for regression)\nif y_model_col in train.columns and not train[y_model_col].isnull().all():\n    print(f\"Calculating sample weights based on {y_model_col}\")\n    weights = calculate_weights(train[y_model_col])\nelse:\n    print(f\"Warning: {y_model_col} not suitable for weights. Calculating based on {y_comp_col}.\")\n    weights = calculate_weights(train[y_comp_col]) # Less ideal but fallback","metadata":{"id":"36b43cf7","outputId":"c29fffca-ac1c-4c27-8888-baa5da537257","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:13:51.752907Z","iopub.execute_input":"2025-04-24T06:13:51.753307Z","iopub.status.idle":"2025-04-24T06:13:51.775641Z","shell.execute_reply.started":"2025-04-24T06:13:51.753272Z","shell.execute_reply":"2025-04-24T06:13:51.774430Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n--- Cross-Validating Regressor Models ---\")\noof_preds_regressors = {}\noof_indices_regressors = {}\nthresholds_regressors = {}\nmodels_regressors = {}\noverall_kappa_regressors = {}\n\n\n# LGBM\nprint(\"Cross-validating LGBM...\")\nscore_lgb, oof_preds_regressors['lgb'], oof_indices_regressors['lgb'], thresholds_regressors['lgb'], models_regressors['lgb'], overall_kappa_regressors['lgb'] = cross_validate_regressor(\n    lgb_model, train, selected_features, y_model_col, y_comp_col, kf, sample_weights_series=weights, verbose=True\n)\n\n# XGB\nprint(\"\\nCross-validating XGBoost...\")\nscore_xgb, oof_preds_regressors['xgb'], oof_indices_regressors['xgb'], thresholds_regressors['xgb'], models_regressors['xgb'], overall_kappa_regressors['xgb'] = cross_validate_regressor(\n    xgb_model, train, selected_features, y_model_col, y_comp_col, kf, sample_weights_series=weights, verbose=True\n)\n\n# CatBoost\nprint(\"\\nCross-validating CatBoost...\")\nscore_cat, oof_preds_regressors['cat'], oof_indices_regressors['cat'], thresholds_regressors['cat'], models_regressors['cat'], overall_kappa_regressors['cat'] = cross_validate_regressor(\n    cat_model, train, selected_features, y_model_col, y_comp_col, kf, sample_weights_series=weights, verbose=True\n)\n\n# ExtraTrees\nprint(\"\\nCross-validating ExtraTrees...\")\n# ExtraTrees doesn't natively support sample weights in fit, so we omit them here\nscore_xtrees, oof_preds_regressors['xtrees'], oof_indices_regressors['xtrees'], thresholds_regressors['xtrees'], models_regressors['xtrees'], overall_kappa_regressors['xtrees'] = cross_validate_regressor(\n    xtrees_model, train, selected_features, y_model_col, y_comp_col, kf, sample_weights_series=None, verbose=True # No weights for ET\n)","metadata":{"id":"d3f2a0b4","outputId":"002e5d41-08a4-46fa-e3fc-cc572cdaaad7","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:13:51.777316Z","iopub.execute_input":"2025-04-24T06:13:51.777870Z","iopub.status.idle":"2025-04-24T06:15:12.130147Z","shell.execute_reply.started":"2025-04-24T06:13:51.777831Z","shell.execute_reply":"2025-04-24T06:15:12.129082Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cross val ordinal model\noof_indices_ordinal = {}\nmodels_ordinal = {}\noverall_kappa_ordinal = {}\n\nif ordinal_model is not None:\n    print(\"\\n--- Cross-Validating Ordinal Model ---\")\n    score_ord, oof_indices_ordinal['ord'], models_ordinal['ord'], overall_kappa_ordinal['ord'] = cross_validate_ordinal(\n        ordinal_model, train, selected_features, y_comp_col, kf, verbose=True\n    )\nelse:\n    print(\"\\nSkipping Ordinal Model cross-validation (mord library not found or model not defined).\")","metadata":{"id":"fa14e587","outputId":"d8391484-9ccd-4517-cd70-6bfd3b2fe212","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:15:12.131144Z","iopub.execute_input":"2025-04-24T06:15:12.131450Z","iopub.status.idle":"2025-04-24T06:15:12.336119Z","shell.execute_reply.started":"2025-04-24T06:15:12.131428Z","shell.execute_reply":"2025-04-24T06:15:12.335163Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ensemble prep\navg_thresholds = {}\nfor model_name, fold_thresholds in thresholds_regressors.items():\n    if fold_thresholds: # Check if list is not empty\n         avg_thresholds[model_name] = np.mean(np.array(fold_thresholds), axis=0)\n    else:\n         print(f\"Warning: No thresholds found for {model_name}. Using base thresholds.\")\n         avg_thresholds[model_name] = base_thresholds\n\n\n# Prepare OOF predictions for stacking\n# Use the thresholded index predictions from regressors and direct predictions from ordinal\noof_stacking_features = pd.DataFrame(oof_indices_regressors)\nif 'ord' in oof_indices_ordinal:\n    oof_stacking_features['ord'] = oof_indices_ordinal['ord']\n\nprint(\"\\nOOF Predictions for Stacking:\")\nprint(oof_stacking_features.head())\nprint(f\"Correlation of OOF index predictions:\\n{oof_stacking_features.corr()}\")","metadata":{"id":"c1830b8e","outputId":"62f7773b-a403-4243-94b9-5fb42c3a7ff6","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:15:12.337110Z","iopub.execute_input":"2025-04-24T06:15:12.337430Z","iopub.status.idle":"2025-04-24T06:15:12.358619Z","shell.execute_reply.started":"2025-04-24T06:15:12.337398Z","shell.execute_reply":"2025-04-24T06:15:12.357726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# stacking ensemble\nfrom sklearn.linear_model import LogisticRegression\n\nmeta_model = LogisticRegression(random_state=SEED, C=1.0) # Adjust C (regularization) if needed","metadata":{"id":"f52ac694","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:15:12.359758Z","iopub.execute_input":"2025-04-24T06:15:12.360071Z","iopub.status.idle":"2025-04-24T06:15:12.365043Z","shell.execute_reply.started":"2025-04-24T06:15:12.360051Z","shell.execute_reply":"2025-04-24T06:15:12.364202Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Fitting meta-model on OOF features...\")\nmeta_model.fit(oof_stacking_features, train[y_comp_col])\nprint(\"Meta-model fitting complete.\")","metadata":{"id":"40cfbb00","outputId":"b5e5978f-630c-42b9-85ce-c2dde19b97fa","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:15:12.365897Z","iopub.execute_input":"2025-04-24T06:15:12.366187Z","iopub.status.idle":"2025-04-24T06:15:12.787504Z","shell.execute_reply.started":"2025-04-24T06:15:12.366168Z","shell.execute_reply":"2025-04-24T06:15:12.786706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate meta-model on OOF predictions\noof_stacking_preds = meta_model.predict(oof_stacking_features)\nstacking_oof_kappa = cohen_kappa_score(train[y_comp_col], oof_stacking_preds, weights='quadratic')\nprint(f\"\\nStacking Ensemble OOF Kappa: {stacking_oof_kappa:.4f}\")","metadata":{"id":"6dface86","outputId":"84c2d30d-8b18-4d67-b791-a4f9fe6c0f9e","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:15:12.788170Z","iopub.execute_input":"2025-04-24T06:15:12.788396Z","iopub.status.idle":"2025-04-24T06:15:12.814897Z","shell.execute_reply.started":"2025-04-24T06:15:12.788379Z","shell.execute_reply":"2025-04-24T06:15:12.813719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot Confusion Matrix for Stacking Ensemble\nprint(\"\\nPlotting Stacking Ensemble Confusion Matrix...\")\nconf_matrix_stacking = confusion_matrix(train[y_comp_col], oof_stacking_preds)\nplt.figure(figsize=(8, 6))\nsns.heatmap(conf_matrix_stacking, annot=True, fmt=\"d\", cmap=\"YlGnBu\", cbar=False, linewidths=0.5, linecolor='black',\n            xticklabels=range(4), yticklabels=range(4))\nplt.title('Stacking Ensemble Confusion Matrix (OOF)', fontsize=16)\nplt.xlabel('Predicted', fontsize=12)\nplt.ylabel('True', fontsize=12)\nplt.show()","metadata":{"id":"1fd8506e","outputId":"5aaf84a0-3b46-4667-b7de-0554c78038e2","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:15:12.816107Z","iopub.execute_input":"2025-04-24T06:15:12.817854Z","iopub.status.idle":"2025-04-24T06:15:13.066945Z","shell.execute_reply.started":"2025-04-24T06:15:12.817826Z","shell.execute_reply":"2025-04-24T06:15:13.066011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# final model trg\n\nprint(\"\\n--- Training Final Models on Full Data ---\")\nfinal_models = {}\n\n# Regressors\nprint(\"Training LGBM...\")\nfinal_models['lgb'] = clone(lgb_model).fit(train[selected_features], train[y_model_col], sample_weight=weights)\nprint(\"Training XGBoost...\")\nfinal_models['xgb'] = clone(xgb_model).fit(train[selected_features], train[y_model_col], sample_weight=weights)\nprint(\"Training CatBoost...\")\n# Use early stopping with a validation set split from train for CatBoost final fit\nX_train_final, X_val_final, y_train_final, y_val_final, w_train_final, _ = train_test_split(\n    train[selected_features], train[y_model_col], weights, test_size=0.1, random_state=SEED, stratify=train[y_comp_col] # Stratify by sii\n)\nfinal_models['cat'] = clone(cat_model).fit(X_train_final, y_train_final, eval_set=[(X_val_final, y_val_final)], verbose=0, sample_weight=w_train_final)\nprint(\"Training ExtraTrees...\")\nfinal_models['xtrees'] = clone(xtrees_model).fit(train[selected_features], train[y_model_col]) # No weights\n\n# Ordinal\nif ordinal_model is not None:\n    print(\"Training OrdinalRidge...\")\n    final_models['ord'] = clone(ordinal_model).fit(train[selected_features], train[y_comp_col])","metadata":{"id":"0cc63d0c","outputId":"d897695e-7501-4591-e143-b96add18eb64","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:15:13.067982Z","iopub.execute_input":"2025-04-24T06:15:13.068261Z","iopub.status.idle":"2025-04-24T06:15:19.634206Z","shell.execute_reply.started":"2025-04-24T06:15:13.068241Z","shell.execute_reply":"2025-04-24T06:15:19.633362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on test set with base models\ntest_preds_regressors = {}\ntest_indices_regressors = {}\ntest_indices_ordinal = {}\n\nprint(\"Predicting on test set with base models...\")\nfor name, model in final_models.items():\n    if name in avg_thresholds: # Regressor models\n        test_scores = model.predict(test[selected_features])\n        test_preds_regressors[name] = test_scores\n        test_indices_regressors[name] = round_with_thresholds(test_scores, avg_thresholds[name])\n    elif name == 'ord': # Ordinal model\n        test_indices_ordinal[name] = model.predict(test[selected_features])","metadata":{"id":"c9be87db","outputId":"0ea1724c-9710-46e2-dc5a-324b25d0f969","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:15:19.635172Z","iopub.execute_input":"2025-04-24T06:15:19.635440Z","iopub.status.idle":"2025-04-24T06:15:19.735405Z","shell.execute_reply.started":"2025-04-24T06:15:19.635419Z","shell.execute_reply":"2025-04-24T06:15:19.734642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare test features for stacking meta-model\ntest_stacking_features = pd.DataFrame(test_indices_regressors)\nif 'ord' in test_indices_ordinal:\n    test_stacking_features['ord'] = test_indices_ordinal['ord']","metadata":{"id":"638dc9a7","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:15:19.736366Z","iopub.execute_input":"2025-04-24T06:15:19.736624Z","iopub.status.idle":"2025-04-24T06:15:19.741975Z","shell.execute_reply.started":"2025-04-24T06:15:19.736596Z","shell.execute_reply":"2025-04-24T06:15:19.741131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ensure column order matches OOF features used for training meta-model\ntest_stacking_features = test_stacking_features[oof_stacking_features.columns]","metadata":{"id":"e7e940aa","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:15:19.742903Z","iopub.execute_input":"2025-04-24T06:15:19.743124Z","iopub.status.idle":"2025-04-24T06:15:19.760038Z","shell.execute_reply.started":"2025-04-24T06:15:19.743109Z","shell.execute_reply":"2025-04-24T06:15:19.759166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# final preds using stacking ensemble\nprint(\"Predicting final labels using Stacking Ensemble...\")\nfinal_test_predictions = meta_model.predict(test_stacking_features)","metadata":{"id":"8538df2e","outputId":"9b8f773c-6c98-4c39-e29f-0a66e279b2b9","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:15:19.761000Z","iopub.execute_input":"2025-04-24T06:15:19.761258Z","iopub.status.idle":"2025-04-24T06:15:19.779749Z","shell.execute_reply.started":"2025-04-24T06:15:19.761238Z","shell.execute_reply":"2025-04-24T06:15:19.778772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Creating submission file...\")\nsubmission = pd.read_csv(SUBMISSION_PATH)\nsubmission['sii'] = final_test_predictions.astype(int) # Ensure integer type\n\n# Define the output path for the submission file\nsubmission_path = os.path.join(OUTPUT_PATH, \"submission.csv\")","metadata":{"id":"d3e75e87","outputId":"1d8b7247-4eea-402f-f4d9-c0c9dfe4135e","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:15:19.780880Z","iopub.execute_input":"2025-04-24T06:15:19.781701Z","iopub.status.idle":"2025-04-24T06:15:19.827228Z","shell.execute_reply.started":"2025-04-24T06:15:19.781645Z","shell.execute_reply":"2025-04-24T06:15:19.826389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv(submission_path, index=False)\nprint(f\"Submission file saved to: {submission_path}\")\nprint(\"Value Counts in Submission:\")\nprint(submission['sii'].value_counts())\n\nprint(\"\\nScript finished.\")","metadata":{"id":"e5017aaa","outputId":"67b2343a-a306-43ff-d888-e914b2371184","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:15:19.828287Z","iopub.execute_input":"2025-04-24T06:15:19.828615Z","iopub.status.idle":"2025-04-24T06:15:19.840447Z","shell.execute_reply.started":"2025-04-24T06:15:19.828588Z","shell.execute_reply":"2025-04-24T06:15:19.839735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # --- !! ADD THIS SECTION TO PRINT FINAL OOF SCORES !! ---\n\n# print(\"\\n--- Final OOF QWK Scores ---\")\n\n# # Calculate and print QWK for each base regressor model\n# print(\"Base Regressor Models (OOF):\")\n# for name, oof_preds in oof_indices_regressors.items():\n#     if len(oof_preds) == len(train): # Ensure prediction array has correct length\n#         qwk = cohen_kappa_score(train[y_comp_col], oof_preds, weights='quadratic')\n#         print(f\"  - {name.upper()}: {qwk:.4f}\")\n#     else:\n#         print(f\"  - {name.upper()}: Error calculating score (length mismatch)\")\n\n# # Calculate and print QWK for the ordinal model (if used)\n# if 'ord' in oof_indices_ordinal:\n#     print(\"\\nBase Ordinal Model (OOF):\")\n#     if len(oof_indices_ordinal['ord']) == len(train):\n#         qwk_ord = cohen_kappa_score(train[y_comp_col], oof_indices_ordinal['ord'], weights='quadratic')\n#         print(f\"  - OrdinalRidge: {qwk_ord:.4f}\")\n#     else:\n#          print(f\"  - OrdinalRidge: Error calculating score (length mismatch)\")\n\n\n# # Print the stacking ensemble QWK (already calculated)\n# print(\"\\nStacking Ensemble Model (OOF):\")\n# # Ensure stacking_oof_kappa was calculated previously\n# try:\n#     print(f\"  - Stacking Ensemble: {stacking_oof_kappa:.4f}\")\n# except NameError:\n#     # If stacking_oof_kappa wasn't calculated or stored, recalculate it\n#     if 'oof_stacking_preds' in locals() and len(oof_stacking_preds) == len(train):\n#          stacking_oof_kappa = cohen_kappa_score(train[y_comp_col], oof_stacking_preds, weights='quadratic')\n#          print(f\"  - Stacking Ensemble: {stacking_oof_kappa:.4f}\")\n#     else:\n#          print(f\"  - Stacking Ensemble: Error calculating score (predictions unavailable or length mismatch)\")\n\n\n# # --- (Continue with final test set prediction and submission generation...) ---","metadata":{"id":"dd0412a5","outputId":"5ff4492c-d06e-48c0-f66b-5049f0da6185","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:15:19.841549Z","iopub.execute_input":"2025-04-24T06:15:19.841946Z","iopub.status.idle":"2025-04-24T06:15:19.848786Z","shell.execute_reply.started":"2025-04-24T06:15:19.841918Z","shell.execute_reply":"2025-04-24T06:15:19.847924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import joblib","metadata":{"id":"30324e11","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:15:19.849799Z","iopub.execute_input":"2025-04-24T06:15:19.850178Z","iopub.status.idle":"2025-04-24T06:15:19.866698Z","shell.execute_reply.started":"2025-04-24T06:15:19.850150Z","shell.execute_reply":"2025-04-24T06:15:19.865911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # --- (Previous code: Final model training, including final_models dict and meta_model.fit()) ---\n\n# # --- !! ADD THIS SECTION TO SAVE FINAL MODELS !! ---\n\n# print(\"\\n--- Saving Final Trained Models ---\")\n\n# # Define a directory to save the models\n# MODEL_SAVE_DIR = os.path.join(OUTPUT_PATH, \"final_models\") # Using OUTPUT_PATH defined earlier\n# os.makedirs(MODEL_SAVE_DIR, exist_ok=True) # Create directory if it doesn't exist\n# print(f\"Models will be saved in: {MODEL_SAVE_DIR}\")\n\n# # Save the base models stored in the final_models dictionary\n# for name, model in final_models.items():\n#     save_path = os.path.join(MODEL_SAVE_DIR, f\"{name}_model\") # Base path without extension\n\n#     try:\n#         print(f\"Saving model: {name}...\")\n#         if isinstance(model, LGBMRegressor):\n#             model.booster_.save_model(f\"{save_path}.lgbm\") # Use LightGBM's method\n#         elif isinstance(model, XGBRegressor):\n#             model.save_model(f\"{save_path}.xgb\") # Use XGBoost's method\n#         elif isinstance(model, CatBoostRegressor):\n#             model.save_model(f\"{save_path}.cbm\") # Use CatBoost's method\n#         elif isinstance(model, (ExtraTreesRegressor, mord.OrdinalRidge)): # Models compatible with joblib\n#              joblib.dump(model, f\"{save_path}.joblib\")\n#         else:\n#             print(f\"  - Warning: Unknown model type '{type(model)}' for '{name}'. Attempting joblib dump.\")\n#             joblib.dump(model, f\"{save_path}.joblib\") # Fallback attempt\n#         print(f\"  - Saved {name} successfully.\")\n\n#     except Exception as e:\n#         print(f\"  - Error saving model {name}: {e}\")\n\n\n# # Save the stacking meta-model (Logistic Regression)\n# try:\n#     print(\"Saving stacking meta-model...\")\n#     meta_model_save_path = os.path.join(MODEL_SAVE_DIR, \"stacking_meta_model.joblib\")\n#     joblib.dump(meta_model, meta_model_save_path)\n#     print(\"  - Saved stacking meta-model successfully.\")\n# except Exception as e:\n#      print(f\"  - Error saving stacking meta-model: {e}\")\n\n# print(\"--- Model Saving Complete ---\")\n\n# # --- (Continue with test set prediction, submission generation, etc.) ---","metadata":{"id":"890cd7a8","outputId":"3029018d-7bf3-4355-cd52-0823f55aec09","trusted":true,"execution":{"iopub.status.busy":"2025-04-24T06:15:19.867718Z","iopub.execute_input":"2025-04-24T06:15:19.867998Z","iopub.status.idle":"2025-04-24T06:15:19.882477Z","shell.execute_reply.started":"2025-04-24T06:15:19.867978Z","shell.execute_reply":"2025-04-24T06:15:19.881447Z"}},"outputs":[],"execution_count":null}]}