{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score, make_scorer, confusion_matrix\nfrom sklearn.model_selection import StratifiedKFold, KFold, train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom scipy.optimize import minimize\nfrom scipy import stats\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport warnings\nfrom sklearn.linear_model import ElasticNetCV, LassoCV, Lasso, LinearRegression\nfrom sklearn.ensemble import ExtraTreesRegressor, RandomForestRegressor\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nimport optuna\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nimport random\nimport shap\n# 设置中文字体\nplt.rcParams['font.sans-serif'] = ['SimHei']\n# 解决负号显示问题\nplt.rcParams['axes.unicode_minus'] = False\n\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:08:53.698127Z","iopub.execute_input":"2025-04-04T11:08:53.699102Z","iopub.status.idle":"2025-04-04T11:08:53.709603Z","shell.execute_reply.started":"2025-04-04T11:08:53.699057Z","shell.execute_reply":"2025-04-04T11:08:53.708315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(1+1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:08:56.464337Z","iopub.execute_input":"2025-04-04T11:08:56.464720Z","iopub.status.idle":"2025-04-04T11:08:56.470717Z","shell.execute_reply.started":"2025-04-04T11:08:56.464687Z","shell.execute_reply":"2025-04-04T11:08:56.469549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 643\nn_splits = 10\noptimize_params = False\nn_trials = 25 \nvoting = True\nbase_thresholds = [30, 50, 80]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:08:58.201747Z","iopub.execute_input":"2025-04-04T11:08:58.202250Z","iopub.status.idle":"2025-04-04T11:08:58.208047Z","shell.execute_reply.started":"2025-04-04T11:08:58.202203Z","shell.execute_reply":"2025-04-04T11:08:58.206965Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\nTEST_PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'\nTRAIN_TS_PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet'\nTEST_TS_PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:18:02.178658Z","iopub.execute_input":"2025-04-04T11:18:02.179120Z","iopub.status.idle":"2025-04-04T11:18:02.183515Z","shell.execute_reply.started":"2025-04-04T11:18:02.179087Z","shell.execute_reply":"2025-04-04T11:18:02.182379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 加载数据\ntrain = pd.read_csv(TRAIN_PATH)\ntest = pd.read_csv(TEST_PATH)\ntrain.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:18:04.203565Z","iopub.execute_input":"2025-04-04T11:18:04.203909Z","iopub.status.idle":"2025-04-04T11:18:04.292159Z","shell.execute_reply.started":"2025-04-04T11:18:04.203883Z","shell.execute_reply":"2025-04-04T11:18:04.291096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:21:28.180456Z","iopub.execute_input":"2025-04-04T11:21:28.180856Z","iopub.status.idle":"2025-04-04T11:21:28.187603Z","shell.execute_reply.started":"2025-04-04T11:21:28.180825Z","shell.execute_reply":"2025-04-04T11:21:28.186368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def describe_x(df):\n    X = df['X']\n    return [\n        X.std(),\n    ]\n\ndef describe_y(df):\n    Y = df['Y']\n    return [\n        Y.std(),\n    ]\n\ndef describe_z(df):\n    Z = df['Z']\n    return [\n        Z.std(),  \n    ]\n\ndef describe_enmo(df):\n    enmo = df['enmo']\n    return [\n        enmo.mean(),  \n    ]\n\ndef describe_anglez(df):\n    anglez = df['anglez']\n    return [\n        anglez.std(),\n    ]\n    \n# Light level thresholds (in lux)\nlight_bins = [\n    (0, 5, 'Twilight'),\n    (5, 10, 'Minimal Street Lighting'),\n    (10, 50, 'Sunset'),\n    (50, 80, 'Family Living Room'),\n    (80, 100, 'Hallway'),\n    (100, 320, 'Very Dark Overcast Day'),\n    (320, 500, 'Office Lighting'),\n    (500, 1000, 'Sunrise/Sunset'),\n    (1000, 10000, 'Overcast Day'),\n    (10000, 25000, 'Full Daylight'),\n    (25000, 130000, 'Direct Sunlight')\n]\n\n\ndef categorize_light(light_value):\n    for low, high, label in light_bins:\n        if low <= light_value < high:\n            return label\n    return 'Unknown'\n\ndef describe_light(df):\n    df['light_category'] = df['light'].apply(categorize_light)\n    light_categories = df['light_category'].value_counts(normalize=True).to_dict()\n    \n    features = [light_categories.get(label, 0) for _, _, label in light_bins]\n    return features\n\ndef longest_inactivity_streaks(df, window_size=100, threshold=10, top_n=5):\n    rolling_cumsum = df['enmo'].rolling(window=window_size).sum()\n    inactive = rolling_cumsum <= threshold\n    \n    # Calculate streaks\n    streak_lengths = []\n    current_streak = 0\n    for is_inactive in inactive:\n        if is_inactive:\n            current_streak += 1\n        else:\n            if current_streak > 0:\n                streak_lengths.append(current_streak)\n            current_streak = 0\n    \n    # If the last streak is still active, add it\n    if current_streak > 0:\n        streak_lengths.append(current_streak)\n    \n    # Sort streaks in descending order and pick top N\n    streak_lengths = sorted(streak_lengths, reverse=True)[:top_n]\n    \n    # Pad with zeros if there are fewer than N streaks\n    streak_lengths += [0] * (top_n - len(streak_lengths))\n    return streak_lengths\n\n\ndef longest_activity_streaks(df, window_size=100, threshold=1, top_n=5):\n    # Calculate cumsum of enmo in the defined window\n    rolling_cumsum = df['enmo'].rolling(window=window_size).sum()\n    \n    # Identify active windows (cumsum > threshold)\n    active = rolling_cumsum > threshold\n    \n    # Calculate streaks\n    streak_lengths = []\n    current_streak = 0\n    for is_active in active:\n        if is_active:\n            current_streak += 1\n        else:\n            if current_streak > 0:\n                streak_lengths.append(current_streak)\n            current_streak = 0\n    \n    # If the last streak is still active, add it\n    if current_streak > 0:\n        streak_lengths.append(current_streak)\n    \n    # Sort streaks in descending order and pick top N\n    streak_lengths = sorted(streak_lengths, reverse=True)[:top_n]\n    \n    # Pad with zeros if there are fewer than N streaks\n    streak_lengths += [0] * (top_n - len(streak_lengths))\n    return streak_lengths\n\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop(['step'], axis=1, inplace=True)\n   \n    features = []\n    features.extend(describe_x(df))\n    features.extend(describe_y(df))\n    features.extend(describe_z(df))\n    features.extend(describe_enmo(df))\n    features.extend(describe_anglez(df))\n    features.extend(describe_light(df))  \n    \n    enmo_active_ratio = (df['enmo'] > 0).mean()\n    features.append(enmo_active_ratio)\n    features.extend(longest_inactivity_streaks(df, threshold=1))\n    features.extend(longest_activity_streaks(df, threshold=5))\n   \n    return np.array(features), filename.split('=')[1]\n\n\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:18:22.072231Z","iopub.execute_input":"2025-04-04T11:18:22.072617Z","iopub.status.idle":"2025-04-04T11:18:22.091388Z","shell.execute_reply.started":"2025-04-04T11:18:22.072585Z","shell.execute_reply":"2025-04-04T11:18:22.090329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:09:54.012300Z","iopub.execute_input":"2025-04-04T11:09:54.012701Z","iopub.status.idle":"2025-04-04T11:14:34.299391Z","shell.execute_reply.started":"2025-04-04T11:09:54.012671Z","shell.execute_reply":"2025-04-04T11:14:34.298212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = train_ts.drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:18:46.187451Z","iopub.execute_input":"2025-04-04T11:18:46.187905Z","iopub.status.idle":"2025-04-04T11:18:46.193664Z","shell.execute_reply.started":"2025-04-04T11:18:46.187871Z","shell.execute_reply":"2025-04-04T11:18:46.192678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = StandardScaler()\ndf_train = pd.DataFrame(scaler.fit_transform(df_train), columns=df_train.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:18:51.423271Z","iopub.execute_input":"2025-04-04T11:18:51.423591Z","iopub.status.idle":"2025-04-04T11:18:51.432540Z","shell.execute_reply.started":"2025-04-04T11:18:51.423567Z","shell.execute_reply":"2025-04-04T11:18:51.431518Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:21:41.458425Z","iopub.execute_input":"2025-04-04T11:21:41.458816Z","iopub.status.idle":"2025-04-04T11:21:41.481569Z","shell.execute_reply.started":"2025-04-04T11:21:41.458790Z","shell.execute_reply":"2025-04-04T11:21:41.480075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test = test_ts.drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:18:57.549754Z","iopub.execute_input":"2025-04-04T11:18:57.550151Z","iopub.status.idle":"2025-04-04T11:18:57.555688Z","shell.execute_reply.started":"2025-04-04T11:18:57.550119Z","shell.execute_reply":"2025-04-04T11:18:57.554505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test = pd.DataFrame(scaler.transform(df_test), columns=df_test.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:18:59.750556Z","iopub.execute_input":"2025-04-04T11:18:59.750978Z","iopub.status.idle":"2025-04-04T11:18:59.757260Z","shell.execute_reply.started":"2025-04-04T11:18:59.750913Z","shell.execute_reply":"2025-04-04T11:18:59.756131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for c in df_train.columns:\n    m = np.mean(df_train[c])\n    df_train[c].fillna(m, inplace=True)\n    df_test[c].fillna(m, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:19:02.544508Z","iopub.execute_input":"2025-04-04T11:19:02.544885Z","iopub.status.idle":"2025-04-04T11:19:02.565403Z","shell.execute_reply.started":"2025-04-04T11:19:02.544855Z","shell.execute_reply":"2025-04-04T11:19:02.564278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df_train.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:21:50.886483Z","iopub.execute_input":"2025-04-04T11:21:50.886819Z","iopub.status.idle":"2025-04-04T11:21:50.892037Z","shell.execute_reply.started":"2025-04-04T11:21:50.886793Z","shell.execute_reply":"2025-04-04T11:21:50.890860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndf_train['id'] = train_ts['id']\ndf_test['id'] = test_ts['id']\n\ntrain = pd.merge(train, df_train, how=\"left\", on='id')\ntest = pd.merge(test, df_test, how=\"left\", on='id')\ntrain.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:19:13.387862Z","iopub.execute_input":"2025-04-04T11:19:13.388316Z","iopub.status.idle":"2025-04-04T11:19:13.407146Z","shell.execute_reply.started":"2025-04-04T11:19:13.388281Z","shell.execute_reply":"2025-04-04T11:19:13.406036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:22:09.660974Z","iopub.execute_input":"2025-04-04T11:22:09.661357Z","iopub.status.idle":"2025-04-04T11:22:09.668396Z","shell.execute_reply.started":"2025-04-04T11:22:09.661326Z","shell.execute_reply":"2025-04-04T11:22:09.667332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def clean_features(df):\n    # Remove highly implausible values\n    # Clip Grip\n    df[['FGC-FGC_GSND', 'FGC-FGC_GSD']] = df[['FGC-FGC_GSND', 'FGC-FGC_GSD']].clip(lower=9, upper=60)\n    # Remove implausible body-fat(Body Fat Percentage)\n    df[\"BIA-BIA_Fat\"] = np.where(df[\"BIA-BIA_Fat\"] <= 0, np.nan, df[\"BIA-BIA_Fat\"])\n    # Basal Metabolic Rate(基础代谢率)\n    df[\"BIA-BIA_BMR\"] = np.where(df[\"BIA-BIA_BMR\"] > 4000, np.nan, df[\"BIA-BIA_BMR\"])\n    # Daily Energy Expenditure(每日能耗)\n    df[\"BIA-BIA_DEE\"] = np.where(df[\"BIA-BIA_DEE\"] > 8000, np.nan, df[\"BIA-BIA_DEE\"])\n    # Bone Mineral Content（骨矿物含量）\n    df[\"BIA-BIA_BMC\"] = np.where(df[\"BIA-BIA_BMC\"] <= 0, np.nan, df[\"BIA-BIA_BMC\"])\n    df[\"BIA-BIA_BMC\"] = np.where(df[\"BIA-BIA_BMC\"] > 30, np.nan, df[\"BIA-BIA_BMC\"])\n    # Fat Free Mass Index（肌肉质量指数，去脂体重与身高平方的比值）\n    df[\"BIA-BIA_FFM\"] = np.where(df[\"BIA-BIA_FFM\"] > 300, np.nan, df[\"BIA-BIA_FFM\"])\n    # Fat Mass Index(脂肪质量指数)\n    df[\"BIA-BIA_FMI\"] = np.where(df[\"BIA-BIA_FMI\"] <= 0, np.nan, df[\"BIA-BIA_FMI\"])\n    # Extra Cellular Water（细胞外液）\n    df[\"BIA-BIA_ECW\"] = np.where(df[\"BIA-BIA_ECW\"] > 100, np.nan, df[\"BIA-BIA_ECW\"])\n    # Intra Cellular Water（细胞内液）\n    df[\"BIA-BIA_ICW\"] = np.where(df[\"BIA-BIA_ICW\"] > 100, np.nan, df[\"BIA-BIA_ICW\"])\n    # Lean Dry Mass（去脂肪体重）s\n    df[\"BIA-BIA_LDM\"] = np.where(df[\"BIA-BIA_LDM\"] > 100, np.nan, df[\"BIA-BIA_LDM\"])\n    # Lean Soft Tissue（瘦软组织是指人体中除了脂肪组织以外的软组织部分）\n    df[\"BIA-BIA_LST\"] = np.where(df[\"BIA-BIA_LST\"] > 400, np.nan, df[\"BIA-BIA_LST\"])\n    # Skeletal Muscle Mass（骨骼肌）\n    df[\"BIA-BIA_SMM\"] = np.where(df[\"BIA-BIA_SMM\"] > 300, np.nan, df[\"BIA-BIA_SMM\"])\n    # Total Body Water（水份）\n    df[\"BIA-BIA_TBW\"] = np.where(df[\"BIA-BIA_TBW\"] > 300, np.nan, df[\"BIA-BIA_TBW\"])\n\n    df[\"BIA-BIA_BMI\"] = np.where(df[\"BIA-BIA_BMI\"] <10, np.nan, df[\"BIA-BIA_BMI\"])\n\n    df[\"BIA-BIA_FFMI\"] = np.where(df[\"BIA-BIA_FFMI\"] > 100, np.nan, df[\"BIA-BIA_FFMI\"])\n    \n    return df\n\ntrain = clean_features(train)\ntest = clean_features(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:22:12.760897Z","iopub.execute_input":"2025-04-04T11:22:12.761292Z","iopub.status.idle":"2025-04-04T11:22:12.793236Z","shell.execute_reply.started":"2025-04-04T11:22:12.761264Z","shell.execute_reply":"2025-04-04T11:22:12.791860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:44:28.592536Z","iopub.execute_input":"2025-03-25T08:44:28.592962Z","iopub.status.idle":"2025-03-25T08:44:28.618885Z","shell.execute_reply.started":"2025-03-25T08:44:28.592923Z","shell.execute_reply":"2025-03-25T08:44:28.617434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:22:19.056144Z","iopub.execute_input":"2025-04-04T11:22:19.056620Z","iopub.status.idle":"2025-04-04T11:22:19.064704Z","shell.execute_reply.started":"2025-04-04T11:22:19.056575Z","shell.execute_reply":"2025-04-04T11:22:19.063253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    \n    # From here on own features\n    def assign_group(age):\n        thresholds = [5, 6, 7, 8, 10, 12, 14, 17, 22]\n        for i, j in enumerate(thresholds):\n            if age <= j:\n                return i\n        return np.nan\n    \n    # Age groups\n    df[\"group\"] = df['Basic_Demos-Age'].apply(assign_group)\n    \n    # BMI (body mass index)体质指数,衡量健康水平\n    BMI_map = {0: 16.3,1: 15.9,2: 16.1,3: 16.8,4: 17.3,5: 19.2,6: 20.2,7: 22.3, 8: 23.6}\n    df['BMI_mean_norm'] = df[['Physical-BMI', 'BIA-BIA_BMI']].mean(axis=1) / df[\"group\"].map(BMI_map)\n    \n    # FGC zone aggregate （aerobic capacity, muscular strength, muscular endurance, flexibility, and body composition.）（可改）\n    zones = ['FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone',\n            'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone',\n            'FGC-FGC_TL_Zone']\n    \n    df['FGC_Zones_mean'] = df[zones].mean(axis=1)\n    df['FGC_Zones_min'] = df[zones].min(axis=1)\n    df['FGC_Zones_max'] = df[zones].max(axis=1)\n    \n    # Grip（抓力）\n    GSD_max_map = {0: 9, 1: 9, 2: 9, 3: 9, 4: 16.2, 5: 19.9, 6: 26.1, 7: 31.3, 8: 35.4}\n    GSD_min_map = {0: 9, 1: 9, 2: 9, 3: 9, 4: 14.4, 5: 17.8, 6: 23.4, 7: 27.8, 8: 31.1}\n    \n    df['GS_max'] = df[['FGC-FGC_GSND', 'FGC-FGC_GSD']].max(axis=1) / df[\"group\"].map(GSD_max_map)\n    df['GS_min'] = df[['FGC-FGC_GSND', 'FGC-FGC_GSD']].min(axis=1) / df[\"group\"].map(GSD_min_map)\n    \n    # Curl-ups（仰卧起坐）, push-ups（俯卧撑）, trunk-lifts（起身式）... normalized based on age-group\n    cu_map = {0: 1.0, 1: 3.0, 2: 5.0, 3: 7.0, 4: 10.0, 5: 14.0, 6: 20.0, 7: 20.0, 8: 20.0}\n    pu_map = {0: 1.0, 1: 2.0, 2: 3.0, 3: 4.0, 4: 5.0, 5: 7.0, 6: 8.0, 7: 10.0, 8: 14.0}\n    tl_map = {0: 8.0, 1: 8.0, 2: 8.0, 3: 9.0, 4: 9.0, 5: 10.0, 6: 10.0, 7: 10.0, 8: 10.0}\n    \n    df[\"CU_norm\"] = df['FGC-FGC_CU'] / df['group'].map(cu_map)\n    df[\"PU_norm\"] = df['FGC-FGC_PU'] / df['group'].map(pu_map)\n    df[\"TL_norm\"] = df['FGC-FGC_TL'] / df['group'].map(tl_map)\n    \n    # Reach （坐位体前屈）\n    df[\"SR_min\"] = df[['FGC-FGC_SRL', 'FGC-FGC_SRR']].min(axis=1)\n    df[\"SR_max\"] = df[['FGC-FGC_SRL', 'FGC-FGC_SRR']].max(axis=1)\n\n    # BIA Features\n    # Energy Expenditure    \n    bmr_map = {0: 934.0, 1: 941.0, 2: 999.0, 3: 1048.0, 4: 1283.0, 5: 1255.0, 6: 1481.0, 7: 1519.0, 8: 1650.0}\n    dee_map = {0: 1471.0, 1: 1508.0, 2: 1640.0, 3: 1735.0, 4: 2132.0, 5: 2121.0, 6: 2528.0, 7: 2566.0, 8: 2793.0}\n    df[\"BMR_norm\"] = df[\"BIA-BIA_BMR\"] / df[\"group\"].map(bmr_map)  # 基础代谢率\n    df[\"DEE_norm\"] = df[\"BIA-BIA_DEE\"] / df[\"group\"].map(dee_map) # 每日能耗\n    df[\"DEE_BMR\"] = df[\"BIA-BIA_DEE\"] - df[\"BIA-BIA_BMR\"]\n\n    # FMM （去脂体重）\n    ffm_map = {0: 42.0, 1: 43.0, 2: 49.0, 3: 54.0, 4: 60.0, 5: 76.0, 6: 94.0, 7: 104.0, 8: 111.0}\n    df[\"FFM_norm\"] = df[\"BIA-BIA_FFM\"] / df[\"group\"].map(ffm_map)\n\n    # ECW ICW\n    df[\"ECW_to_ICW\"] = df[\"BIA-BIA_ECW\"] / df[\"BIA-BIA_ICW\"]\n    \n    # \n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n\n    # \n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n\n    #\n    df['ECW_to_SMM'] = df[\"BIA-BIA_ECW\"] / df['BIA-BIA_SMM']\n    \n    # Fitness_Endurance\n    df[\"Fitness_Endurance-Time\"] = df[\"Fitness_Endurance-Time_Mins\"]*60+df[\"Fitness_Endurance-Time_Sec\"]\n    \n    drop_feats = ['FGC-FGC_GSND', 'FGC-FGC_GSD', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL_Zone',\n                'Physical-BMI', 'BIA-BIA_BMI', 'FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_TL', 'FGC-FGC_SRL', 'FGC-FGC_SRR',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', \"BIA-BIA_FFM\",\"Fitness_Endurance-Time_Mins\",\"Fitness_Endurance-Time_Sec\"]\n    df = df.drop(drop_feats, axis=1) \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:22:23.265890Z","iopub.execute_input":"2025-04-04T11:22:23.266311Z","iopub.status.idle":"2025-04-04T11:22:23.283126Z","shell.execute_reply.started":"2025-04-04T11:22:23.266280Z","shell.execute_reply":"2025-04-04T11:22:23.281889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Perform feature_engineering on train and test\ntrain = feature_engineering(train)\ntest = feature_engineering(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:22:48.116485Z","iopub.execute_input":"2025-04-04T11:22:48.116863Z","iopub.status.idle":"2025-04-04T11:22:48.182592Z","shell.execute_reply.started":"2025-04-04T11:22:48.116831Z","shell.execute_reply":"2025-04-04T11:22:48.181407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:40:29.785895Z","iopub.execute_input":"2025-04-04T11:40:29.786277Z","iopub.status.idle":"2025-04-04T11:40:29.793312Z","shell.execute_reply.started":"2025-04-04T11:40:29.786249Z","shell.execute_reply":"2025-04-04T11:40:29.791984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:23:05.211402Z","iopub.execute_input":"2025-04-04T11:23:05.211805Z","iopub.status.idle":"2025-04-04T11:23:05.217832Z","shell.execute_reply.started":"2025-04-04T11:23:05.211770Z","shell.execute_reply":"2025-04-04T11:23:05.216714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:23:36.567468Z","iopub.execute_input":"2025-04-04T11:23:36.567856Z","iopub.status.idle":"2025-04-04T11:23:36.575205Z","shell.execute_reply.started":"2025-04-04T11:23:36.567824Z","shell.execute_reply":"2025-04-04T11:23:36.573896Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndef bin_data(train, test, columns, n_bins=10):\n    # 检查输入是否为DataFrame\n    if not isinstance(train, pd.DataFrame) or not isinstance(test, pd.DataFrame):\n        raise ValueError(\"输入的train和test必须是pandas DataFrame类型。\")\n    \n    # Combine train and test for consistent bin edges\n    combined = pd.concat([train, test], axis=0)\n    \n    bin_edges = {}\n    for col in columns:\n        try:\n            # Compute quantile bin edges\n            edges = pd.qcut(combined[col], n_bins, retbins=True, labels=range(n_bins), duplicates=\"drop\")[1]\n            # 检查分桶边界是否单调递增\n            if not pd.Series(edges).is_monotonic_increasing:\n                raise ValueError(f\"列 {col} 的分桶边界不是单调递增的。\")\n            bin_edges[col] = edges\n        except ValueError as e:\n            print(f\"处理列 {col} 时出错: {e}，跳过该列。\")\n    \n    # Apply the same bin edges to both train and test\n    for col, edges in bin_edges.items():\n        train[col] = pd.cut(\n            train[col], bins=edges, labels=range(len(edges) - 1), include_lowest=True\n        ).astype(float)\n        test[col] = pd.cut(\n            test[col], bins=edges, labels=range(len(edges) - 1), include_lowest=True\n        ).astype(float)\n    \n    return train, test\n\n# 假设train和test是已经定义好的DataFrame\n# 示例代码中没有给出train和test的定义，你需要确保这两个变量已经正确定义\n# train = ...\n# test = ...\n\ncolumns_to_bin = [\n    \"PAQ_A-PAQ_A_Total\", \"PAQ_C-PAQ_C_Total\",\"BMR_norm\", \"DEE_norm\", \"GS_min\", \"GS_max\", \"BIA-BIA_FFMI\", \n    \"BIA-BIA_BMC\", \"Physical-HeartRate\", \"BIA-BIA_ICW\", \"BIA-BIA_ECW\",\n    \"BIA-BIA_LDM\", \"BIA-BIA_LST\",\"BIA-BIA_SMM\", \"BIA-BIA_TBW\", \"DEE_BMR\", \"ECW_to_ICW\",'Hydration_Status',\n    'Muscle_to_Fat','ECW_to_SMM'\n]\n# Bin specified columns in train and test\ntry:\n    train, test = bin_data(train, test, columns_to_bin, n_bins=10)\nexcept ValueError as e:\n    print(f\"出现错误: {e}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:42:19.457857Z","iopub.execute_input":"2025-04-04T11:42:19.458280Z","iopub.status.idle":"2025-04-04T11:42:19.556617Z","shell.execute_reply.started":"2025-04-04T11:42:19.458249Z","shell.execute_reply":"2025-04-04T11:42:19.555529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:50:51.039214Z","iopub.execute_input":"2025-04-04T11:50:51.039750Z","iopub.status.idle":"2025-04-04T11:50:51.069123Z","shell.execute_reply.started":"2025-04-04T11:50:51.039676Z","shell.execute_reply":"2025-04-04T11:50:51.067923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:51:01.913851Z","iopub.execute_input":"2025-04-04T11:51:01.914369Z","iopub.status.idle":"2025-04-04T11:51:01.922500Z","shell.execute_reply.started":"2025-04-04T11:51:01.914331Z","shell.execute_reply":"2025-04-04T11:51:01.921444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Features to exclude, because they're not in test\nexclude = ['PCIAT-Season', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03',\n        'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07',\n        'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11',\n        'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15',\n        'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19',\n        'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total', 'sii', 'id']\n\ny_model = \"PCIAT-PCIAT_Total\" # Score, target for the model\ny_comp = \"sii\" # Index, target of the competition\nfeatures = [f for f in train.columns if f not in exclude]    \ntrain = train[train[\"sii\"].notna()] # Keep rows where target is available\ntrain.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:51:17.899632Z","iopub.execute_input":"2025-04-04T11:51:17.900122Z","iopub.status.idle":"2025-04-04T11:51:17.913379Z","shell.execute_reply.started":"2025-04-04T11:51:17.900084Z","shell.execute_reply":"2025-04-04T11:51:17.912059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Impute_With_Model:\n    def __init__(self, na_frac=0.5, min_samples=0):\n        self.model_dict = {} # Dictionary storing models for imputation\n        self.mean_dict = {} # Dictionary storing mean of feature\n        self.features = None\n        self.na_frac = na_frac # Maximum fraction of missing values allowed for features to be used\n        self.min_samples = min_samples # Minimum number of samples required for model fitting\n        \n    def find_features(self, data, feature, tmp_features):\n        # Finds valid features where the fraction of missing values is <= na_frac\n        missing_rows = data[feature].isna()\n        na_fraction = data[missing_rows][tmp_features].isna().mean(axis=0) # 对于缺失行看其他特征是否完整\n        valid_features = np.array(tmp_features)[na_fraction <= self.na_frac]\n        return valid_features\n\n    def fit_models(self, model, data, features):\n        self.features = features\n        n_data = data.shape[0]\n        for feature in features:\n            self.mean_dict[feature] = np.mean(data[feature])\n        # Iterate over all features\n        for feature in tqdm(features):\n            # Impute if there are missing values in the data\n            if data[feature].isna().sum() > 0:\n                model_clone = clone(model)\n                X = data[data[feature].notna()].copy() # Select data where target values are available as trainings data\n                tmp_features = [f for f in features if f != feature]\n                tmp_features = self.find_features(data, feature, tmp_features)\n                if len(tmp_features) >= 1 and X.shape[0] > self.min_samples:\n                    # Fit model if enough features and sufficient samples\n                    for f in tmp_features:\n                        X[f] = X[f].fillna(self.mean_dict[f])\n                    model_clone.fit(X[tmp_features], X[feature])\n                    # Add model and features to dictionary\n                    self.model_dict[feature] = (model_clone, tmp_features.copy())\n                else:\n                    # Revert to mean imputation if too few \n                    self.model_dict[feature] = (\"mean\", np.mean(data[feature]))\n            \n    def impute(self, data):\n        imputed_data = data.copy()\n        # Iterate over models\n        for feature, model in self.model_dict.items():\n            missing_rows = imputed_data[feature].isna() # Identify rows to be imputed\n            if missing_rows.any():\n                if model[0] == \"mean\":\n                    # Mean imputation if \"mean\"\n                    imputed_data[feature].fillna(model[1], inplace=True)\n                else:\n                    # Prepare data for imputation and predict\n                    tmp_features = [f for f in self.features if f != feature]\n                    X_missing = data.loc[missing_rows, tmp_features].copy()\n                    for f in tmp_features:\n                        X_missing[f] = X_missing[f].fillna(self.mean_dict[f])\n                    imputed_data.loc[missing_rows, feature] = model[0].predict(X_missing[model[1]])\n        return imputed_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:51:20.960199Z","iopub.execute_input":"2025-04-04T11:51:20.960585Z","iopub.status.idle":"2025-04-04T11:51:20.973407Z","shell.execute_reply.started":"2025-04-04T11:51:20.960552Z","shell.execute_reply":"2025-04-04T11:51:20.972092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = LassoCV(cv=5, random_state=SEED)\nimputer = Impute_With_Model(na_frac=0.4) \nimputer.fit_models(model, train, features)\ntrain = imputer.impute(train)\ntest = imputer.impute(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:51:24.050858Z","iopub.execute_input":"2025-04-04T11:51:24.051287Z","iopub.status.idle":"2025-04-04T11:51:40.787242Z","shell.execute_reply.started":"2025-04-04T11:51:24.051259Z","shell.execute_reply":"2025-04-04T11:51:40.785986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def round_with_thresholds(raw_preds, thresholds):\n    return np.where(raw_preds < thresholds[0], int(0),\n                    np.where(raw_preds < thresholds[1], int(1),\n                            np.where(raw_preds < thresholds[2], int(2), int(3))))\n\ndef optimize_thresholds(y_true, raw_preds, start_vals=[0.5, 1.5, 2.5]):\n    def fun(thresholds, y_true, raw_preds):\n        rounded_preds = round_with_thresholds(raw_preds, thresholds)\n        return -cohen_kappa_score(y_true, rounded_preds, weights='quadratic')\n\n    res = minimize(fun, x0=start_vals, args=(y_true, raw_preds), method='Powell')\n    assert res.success\n    return res.x # 返回优化后的阈值","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:51:43.977799Z","iopub.execute_input":"2025-04-04T11:51:43.978306Z","iopub.status.idle":"2025-04-04T11:51:43.987015Z","shell.execute_reply.started":"2025-04-04T11:51:43.978267Z","shell.execute_reply":"2025-04-04T11:51:43.985672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_weights(series):\n    # Create bins for the target variable and assign weights based on frequency\n    bins = pd.cut(series, bins=10, labels=False)\n    weights = bins.value_counts().reset_index()\n    weights.columns = ['target_bins', 'count']\n    weights['count'] = 1 / weights['count']\n    weight_map = weights.set_index('target_bins')['count'].to_dict()\n    weights = bins.map(weight_map)\n    return weights / weights.mean() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:51:48.067068Z","iopub.execute_input":"2025-04-04T11:51:48.067466Z","iopub.status.idle":"2025-04-04T11:51:48.073079Z","shell.execute_reply.started":"2025-04-04T11:51:48.067433Z","shell.execute_reply":"2025-04-04T11:51:48.071746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def cross_validate(model_, data, features, score_col, index_col, cv, sample_weights=False, verbose=False):\n    \"\"\"\n    Perform cross-validation with a given model and compute the out-of-fold \n    predictions and Cohen's Kappa score for each fold.\n\n    Returns:\n    float: Mean Kappa score across all folds.\n    array: Out-of-fold score predictions for the entire dataset.\n    \"\"\"\n    kappa_scores = [] \n    oof_score_predictions = np.zeros(len(data))  \n\n    score_to_index_thresholds = base_thresholds  \n    thresholds = []\n    for fold_idx, (train_idx, val_idx) in enumerate(cv.split(data, data[index_col])):\n        X_train, X_val = data[features].iloc[train_idx], data[features].iloc[val_idx]  #划分训练集与测试集\n        y_train_score = data[score_col].iloc[train_idx]  # 训练集分数\n        y_train_index = data[index_col].iloc[train_idx]  # 训练集分类\n        y_val_score = data[score_col].iloc[val_idx]  # 测试集分数\n        y_val_index = data[index_col].iloc[val_idx]  # 测试集分类 \n        \n        # Train model with sample weights if provided\n        if sample_weights:\n            weights = calculate_weights(y_train_score)\n            model_.fit(X_train, y_train_score, sample_weight=weights)\n        else:\n            model_.fit(X_train, y_train_score)\n\n        y_pred_train_score = model_.predict(X_train)\n        y_pred_val_score = model_.predict(X_val)\n        \n        oof_score_predictions[val_idx] = y_pred_val_score \n\n        # Find optimal threshold in sample \n        t_1 = optimize_thresholds(y_train_index, y_pred_train_score, start_vals=base_thresholds) # 返回优化后的阈值\n        thresholds.append(t_1)\n\n        y_pred_val_index = round_with_thresholds(y_pred_val_score, t_1)\n\n        kappa_score = cohen_kappa_score(y_val_index, y_pred_val_index, weights='quadratic')\n        kappa_scores.append(kappa_score)\n        \n        if verbose:\n            print(f\"Fold {fold_idx}: Optimized Kappa Score = {kappa_score}\")\n    \n    if verbose:\n        print(f\"## Mean CV Kappa Score: {np.mean(kappa_scores)} ##\")\n        print(f\"## Std CV: {np.std(kappa_scores)}\")\n    \n    return np.mean(kappa_scores), oof_score_predictions, thresholds\n\ndef n_cross_validate(model_, data, features, score_col, index_col, cv, seeds, sample_weights=False, verbose=False):\n    # Performs repeated cross-validation by reseeding the cv object\n    scores = []\n    for seed in seeds:\n        cv.random_state=seed\n        score, oof, _ = cross_validate(model_, data, features, score_col, index_col, cv, sample_weights=True, verbose=False)\n        scores.append(score)\n    return np.mean(score), oof # score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:51:50.628793Z","iopub.execute_input":"2025-04-04T11:51:50.629214Z","iopub.status.idle":"2025-04-04T11:51:50.639449Z","shell.execute_reply.started":"2025-04-04T11:51:50.629186Z","shell.execute_reply":"2025-04-04T11:51:50.638283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# for optuna\ndef objective(trial, model_type, X, features, score_col, index_col, cv, sample_weights=False):\n    # Parameter space to explore if model is xgboost\n    if model_type == 'xgboost':\n        params = {\n            'objective': trial.suggest_categorical('objective', ['reg:tweedie', 'reg:pseudohubererror']),\n            'random_state': SEED,\n            'num_parallel_tree': trial.suggest_int('num_parallel_tree', 2, 30),\n            'n_estimators': trial.suggest_int('n_estimators', 100, 300),\n            'max_depth': trial.suggest_int('max_depth', 2, 4),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 0.02, 0.05),\n            'subsample': trial.suggest_float('subsample', 0.5, 0.8),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 0.8),\n            'reg_alpha': trial.suggest_loguniform('reg_alpha', 1e-5, 1e-1),\n            'reg_lambda': trial.suggest_loguniform('reg_lambda', 1e-5, 1e-1),\n        }\n        if params['objective'] == 'reg:tweedie':\n            params['tweedie_variance_power'] = trial.suggest_float('tweedie_variance_power', 1, 2)\n        model = XGBRegressor(**params, use_label_encoder=False)\n    \n    # Parameter space to explore if model is lightgbm\n    elif model_type == 'lightgbm':\n        params = {\n            'objective': trial.suggest_categorical('objective', ['poisson', 'tweedie', 'regression']),\n            'random_state': SEED,\n            'verbosity': -1,\n            'n_estimators': trial.suggest_int('n_estimators', 100, 300),\n            'max_depth': trial.suggest_int('max_depth', 2, 4),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.05),\n            'subsample': trial.suggest_float('subsample', 0.5, 0.8),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 0.8),\n            'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 20, 100)\n        }\n        if params['objective'] == 'tweedie':\n            params['tweedie_variance_power'] = trial.suggest_float('tweedie_variance_power', 1, 2)\n        model = LGBMRegressor(**params)\n    \n    # Parameter space to explore if model is catboost\n    elif model_type == 'catboost':\n        params = {\n            'loss_function': trial.suggest_categorical('objective', ['Tweedie:variance_power=1.5', \n                                                                    'Poisson', 'RMSE']),\n            'random_state': SEED,\n            'iterations': trial.suggest_int('iterations', 100, 300),\n            'depth': trial.suggest_int('depth', 2, 4),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.05),\n            'l2_leaf_reg': trial.suggest_loguniform('l2_leaf_reg', 1e-3, 1e-1),\n            'subsample': trial.suggest_float('subsample', 0.5, 0.7),\n            'bagging_temperature': trial.suggest_float('bagging_temperature', 0.0, 1.0),\n            'random_strength': trial.suggest_float('random_strength', 1e-3, 10.0),\n            'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 20, 60),\n        }\n        model = CatBoostRegressor(**params, verbose=0)\n    \n    else:\n        raise ValueError(f\"Unsupported model_type: {model_type}\")\n        \n    seeds = [random.randint(1, 10000) for _ in range(20)] # Seeds for repeated KFold\n\n    score, _ = n_cross_validate(model, X, features, score_col, index_col, cv, seeds, sample_weights=True, verbose=True)\n\n    return score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:51:56.973130Z","iopub.execute_input":"2025-04-04T11:51:56.973460Z","iopub.status.idle":"2025-04-04T11:51:56.985239Z","shell.execute_reply.started":"2025-04-04T11:51:56.973435Z","shell.execute_reply":"2025-04-04T11:51:56.984014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def run_optimization(X, features, score_col, index_col, model_type, n_trials=30, cv=None, sample_weights=False):\n    study = optuna.create_study(direction=\"maximize\")\n    study.optimize(lambda trial: objective(trial, model_type, X, features, score_col, index_col, cv, sample_weights), \n                n_trials=n_trials)\n    \n    print(f\"Best params for {model_type}: {study.best_params}\")\n    print(f\"Best score: {study.best_value}\")\n    return study.best_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:52:01.079893Z","iopub.execute_input":"2025-04-04T11:52:01.080330Z","iopub.status.idle":"2025-04-04T11:52:01.086046Z","shell.execute_reply.started":"2025-04-04T11:52:01.080299Z","shell.execute_reply":"2025-04-04T11:52:01.084805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:52:06.445766Z","iopub.execute_input":"2025-04-04T11:52:06.446164Z","iopub.status.idle":"2025-04-04T11:52:06.453220Z","shell.execute_reply.started":"2025-04-04T11:52:06.446132Z","shell.execute_reply":"2025-04-04T11:52:06.452001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_features = features\nxgb_features = features\ncat_features = features\nprint(len(features))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:52:10.223652Z","iopub.execute_input":"2025-04-04T11:52:10.224117Z","iopub.status.idle":"2025-04-04T11:52:10.231524Z","shell.execute_reply.started":"2025-04-04T11:52:10.224083Z","shell.execute_reply":"2025-04-04T11:52:10.230100Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 用于在交叉验证过程中生成折叠（folds），同时保证每个折叠中的各类别的比例与整个数据集中的相同\nkf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:52:15.386607Z","iopub.execute_input":"2025-04-04T11:52:15.387016Z","iopub.status.idle":"2025-04-04T11:52:15.391801Z","shell.execute_reply.started":"2025-04-04T11:52:15.386982Z","shell.execute_reply":"2025-04-04T11:52:15.390394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:52:17.181060Z","iopub.execute_input":"2025-04-04T11:52:17.181454Z","iopub.status.idle":"2025-04-04T11:52:17.205727Z","shell.execute_reply.started":"2025-04-04T11:52:17.181424Z","shell.execute_reply":"2025-04-04T11:52:17.204409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# lgb_params = run_optimization(train, lgb_features, 'PCIAT-PCIAT_Total', 'sii', 'lightgbm', n_trials=n_trials, cv=kf, sample_weights=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:44:38.168432Z","iopub.execute_input":"2025-03-25T08:44:38.168808Z","iopub.status.idle":"2025-03-25T08:44:38.184883Z","shell.execute_reply.started":"2025-03-25T08:44:38.168779Z","shell.execute_reply":"2025-03-25T08:44:38.183380Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_params = {\n    'objective': 'tweedie', \n    'n_estimators': 104, \n    'max_depth': 4, \n    'learning_rate':0.03803102883584067, \n    'subsample':0.5018987434259187, \n    'colsample_bytree': 0.7796775245856102, \n    'min_data_in_leaf': 26,\n    'tweedie_variance_power': 1.1151854443935258\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:52:21.590694Z","iopub.execute_input":"2025-04-04T11:52:21.591131Z","iopub.status.idle":"2025-04-04T11:52:21.596241Z","shell.execute_reply.started":"2025-04-04T11:52:21.591092Z","shell.execute_reply":"2025-04-04T11:52:21.595107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# xgb_params = run_optimization(train, xgb_features, 'PCIAT-PCIAT_Total', 'sii', 'xgboost', n_trials=n_trials, cv=kf, sample_weights=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:44:38.206268Z","iopub.execute_input":"2025-03-25T08:44:38.206743Z","iopub.status.idle":"2025-03-25T08:44:38.226037Z","shell.execute_reply.started":"2025-03-25T08:44:38.206690Z","shell.execute_reply":"2025-03-25T08:44:38.224579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_params = {\n    'objective': 'reg:tweedie', \n    'num_parallel_tree': 10, \n    'n_estimators': 103, \n    'max_depth': 4, \n    'learning_rate': 0.041516820857180226, \n    'subsample':  0.5584143457265243, \n    'colsample_bytree': 0.5452085572379025, \n    'reg_alpha': 7.342903529712683e-05,\n    'reg_lambda': 0.0005153495971550512, \n    'tweedie_variance_power':  1.2404020613161162\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:53:47.483749Z","iopub.execute_input":"2025-04-04T11:53:47.484225Z","iopub.status.idle":"2025-04-04T11:53:47.489509Z","shell.execute_reply.started":"2025-04-04T11:53:47.484188Z","shell.execute_reply":"2025-04-04T11:53:47.488234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cat_params = run_optimization(train, cat_features, 'PCIAT-PCIAT_Total', 'sii', 'catboost', n_trials=n_trials, cv=kf, sample_weights=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:44:38.245710Z","iopub.execute_input":"2025-03-25T08:44:38.246091Z","iopub.status.idle":"2025-03-25T08:44:38.264090Z","shell.execute_reply.started":"2025-03-25T08:44:38.246060Z","shell.execute_reply":"2025-03-25T08:44:38.262744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_params = {\n    'objective': 'RMSE', \n    'iterations': 300, \n    'depth': 3, \n    'learning_rate': 0.026355664467056426, \n    'l2_leaf_reg':0.03478233218557398, \n    'subsample': 0.6211868950201196, \n    'bagging_temperature': 0.6743370890806787, \n    'random_strength': 0.07766429401725408, \n    'min_data_in_leaf': 33\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:44:38.265262Z","iopub.execute_input":"2025-03-25T08:44:38.265660Z","iopub.status.idle":"2025-03-25T08:44:38.284172Z","shell.execute_reply.started":"2025-03-25T08:44:38.265620Z","shell.execute_reply":"2025-03-25T08:44:38.282819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_model = LGBMRegressor(**lgb_params, random_state=SEED, verbosity=-1)\nweights = calculate_weights(train['PCIAT-PCIAT_Total'])\n# Cross-validate LGBM model\nscore_lgb, oof_lgb, lgb_thresholds = cross_validate(\n    lgb_model, train, lgb_features, 'PCIAT-PCIAT_Total', 'sii', kf, verbose=True, sample_weights=True\n)\n# Fit final model and predict test samples\nlgb_model.fit(train[lgb_features], train['PCIAT-PCIAT_Total'], sample_weight=weights)\ntest_lgb = lgb_model.predict(test[lgb_features])\nprint(score_lgb)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:52:28.306418Z","iopub.execute_input":"2025-04-04T11:52:28.306755Z","iopub.status.idle":"2025-04-04T11:52:38.337485Z","shell.execute_reply.started":"2025-04-04T11:52:28.306729Z","shell.execute_reply":"2025-04-04T11:52:38.336347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 创建 SHAP 解释器\nexplainer = shap.Explainer(lgb_model)\n\n# 计算 SHAP 值\nshap_values = explainer(train[lgb_features])\n\n# 可视化 SHAP 总结图\nshap.summary_plot(shap_values.values, train[lgb_features], feature_names=xgb_features)    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:52:54.476398Z","iopub.execute_input":"2025-04-04T11:52:54.477212Z","iopub.status.idle":"2025-04-04T11:52:56.478673Z","shell.execute_reply.started":"2025-04-04T11:52:54.477149Z","shell.execute_reply":"2025-04-04T11:52:56.477244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_thresholds_ens = np.mean(np.array(lgb_thresholds), axis=0)\nlgb_thresholds_ens","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:44:48.184277Z","iopub.execute_input":"2025-03-25T08:44:48.184802Z","iopub.status.idle":"2025-03-25T08:44:48.192365Z","shell.execute_reply.started":"2025-03-25T08:44:48.184741Z","shell.execute_reply":"2025-03-25T08:44:48.191324Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n# 绘制散点图\nscatter1 = plt.scatter(train['PCIAT-PCIAT_Total'], oof_lgb, c=train[\"sii\"], cmap=\"autumn\", alpha=0.5)\n\n# 设置坐标轴标签\nplt.xlabel(\"True Score\")\nplt.ylabel(\"OOF Predictions - LGBM\")\n\n# 设置坐标轴范围\nplt.ylim(0, np.max(train['PCIAT-PCIAT_Total']))\nplt.xlim(0, np.max(train['PCIAT-PCIAT_Total']))\n\n# 设置坐标轴比例\nplt.gca().set_aspect('equal', adjustable='box')\n\n# 绘制阈值线\nthresholds = [30, 50, 80]\nfor threshold in thresholds:\n    plt.axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\nfor threshold in lgb_thresholds_ens:\n    plt.axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n\n# 显示图形\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:44:48.193401Z","iopub.execute_input":"2025-03-25T08:44:48.193869Z","iopub.status.idle":"2025-03-25T08:44:48.520668Z","shell.execute_reply.started":"2025-03-25T08:44:48.193828Z","shell.execute_reply":"2025-03-25T08:44:48.519286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_model = XGBRegressor(**xgb_params, random_state=SEED, verbosity=0)\n# Cross-validate XGBoost model\nscore_xgb, oof_xgb, xgb_thresholds = cross_validate(\n    xgb_model, train, xgb_features, 'PCIAT-PCIAT_Total', 'sii', kf, verbose=True, sample_weights=True\n)\n# Fit final model and predict test samples\nxgb_model.fit(train[xgb_features], train['PCIAT-PCIAT_Total'], sample_weight=weights)\ntest_xgb = xgb_model.predict(test[xgb_features])\nprint(score_xgb)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:54:02.411485Z","iopub.execute_input":"2025-04-04T11:54:02.411870Z","iopub.status.idle":"2025-04-04T11:54:35.849505Z","shell.execute_reply.started":"2025-04-04T11:54:02.411836Z","shell.execute_reply":"2025-04-04T11:54:35.848249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 创建 SHAP 解释器\nexplainer = shap.Explainer(xgb_model)\n\n# 计算 SHAP 值\nshap_values = explainer(train[xgb_features])\n\n# 可视化 SHAP 总结图\nshap.summary_plot(shap_values.values, train[xgb_features], feature_names=xgb_features)    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-04T11:54:42.344575Z","iopub.execute_input":"2025-04-04T11:54:42.345015Z","iopub.status.idle":"2025-04-04T11:54:49.848495Z","shell.execute_reply.started":"2025-04-04T11:54:42.344972Z","shell.execute_reply":"2025-04-04T11:54:49.847085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_thresholds_ens = np.mean(np.array(xgb_thresholds), axis=0)\nxgb_thresholds_ens","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:18.957823Z","iopub.execute_input":"2025-03-25T08:45:18.958218Z","iopub.status.idle":"2025-03-25T08:45:18.966496Z","shell.execute_reply.started":"2025-03-25T08:45:18.958175Z","shell.execute_reply":"2025-03-25T08:45:18.965267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# 绘制散点图\nscatter1 = plt.scatter(train['PCIAT-PCIAT_Total'], oof_xgb, c=train[\"sii\"], cmap=\"autumn\", alpha=0.5)\n\n# 设置坐标轴标签\nplt.xlabel(\"True Score\")\nplt.ylabel(\"OOF Predictions - XGB\")\n\n# 设置坐标轴范围\nplt.ylim(0, np.max(train['PCIAT-PCIAT_Total']))\nplt.xlim(0, np.max(train['PCIAT-PCIAT_Total']))\n\n# 设置坐标轴比例\nplt.gca().set_aspect('equal', adjustable='box')\n\n# 绘制阈值线\nthresholds = [30, 50, 80]\nfor threshold in thresholds:\n    plt.axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\nfor threshold in xgb_thresholds_ens:\n    plt.axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n\n# 显示图形\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:18.967881Z","iopub.execute_input":"2025-03-25T08:45:18.968304Z","iopub.status.idle":"2025-03-25T08:45:19.299002Z","shell.execute_reply.started":"2025-03-25T08:45:18.968262Z","shell.execute_reply":"2025-03-25T08:45:19.297377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_model = CatBoostRegressor(**cat_params, random_state=SEED, verbose=0)\n# Cross-validate CatBoost model\nscore_cat, oof_cat, cat_thresholds = cross_validate(\n    cat_model, train, cat_features, 'PCIAT-PCIAT_Total', 'sii', kf, verbose=True, sample_weights=True\n)\n# Fit final model and predict test samples\ncat_model.fit(train[cat_features], train['PCIAT-PCIAT_Total'], sample_weight=weights)\ntest_cat = cat_model.predict(test[cat_features])\n\nprint(score_cat)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:19.300250Z","iopub.execute_input":"2025-03-25T08:45:19.300625Z","iopub.status.idle":"2025-03-25T08:45:34.972233Z","shell.execute_reply.started":"2025-03-25T08:45:19.300594Z","shell.execute_reply":"2025-03-25T08:45:34.970759Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 创建 SHAP 解释器\nexplainer = shap.Explainer(cat_model)\n\n# 计算 SHAP 值\nshap_values = explainer(train[cat_features])\n\n# 打印第一个样本的 SHAP 值\nprint(\"第一个样本的 SHAP 值：\")\nprint(shap_values.values[0])\n\n# 可视化 SHAP 总结图\nshap.summary_plot(shap_values.values, train[cat_features], feature_names=cat_features)    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:34.973569Z","iopub.execute_input":"2025-03-25T08:45:34.974020Z","iopub.status.idle":"2025-03-25T08:45:36.305922Z","shell.execute_reply.started":"2025-03-25T08:45:34.973985Z","shell.execute_reply":"2025-03-25T08:45:36.304660Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_thresholds_ens = np.mean(np.array(cat_thresholds), axis=0)\ncat_thresholds_ens","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:36.307169Z","iopub.execute_input":"2025-03-25T08:45:36.307597Z","iopub.status.idle":"2025-03-25T08:45:36.315810Z","shell.execute_reply.started":"2025-03-25T08:45:36.307553Z","shell.execute_reply":"2025-03-25T08:45:36.314599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 绘制散点图\nscatter1 = plt.scatter(train['PCIAT-PCIAT_Total'], oof_cat, c=train[\"sii\"], cmap=\"autumn\", alpha=0.5)\n\n# 设置坐标轴标签\nplt.xlabel(\"True Score\")\nplt.ylabel(\"OOF Predictions - XCat\")\n\n# 设置坐标轴范围\nplt.ylim(0, np.max(train['PCIAT-PCIAT_Total']))\nplt.xlim(0, np.max(train['PCIAT-PCIAT_Total']))\n\n# 设置坐标轴比例\nplt.gca().set_aspect('equal', adjustable='box')\n\n# 绘制阈值线\nthresholds = [30, 50, 80]\nfor threshold in thresholds:\n    plt.axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\nfor threshold in cat_thresholds_ens:\n    plt.axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n\n# 显示图形\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:36.317127Z","iopub.execute_input":"2025-03-25T08:45:36.317549Z","iopub.status.idle":"2025-03-25T08:45:36.653667Z","shell.execute_reply.started":"2025-03-25T08:45:36.317445Z","shell.execute_reply":"2025-03-25T08:45:36.652288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.set_theme(style=\"white\")\nfig, axes = plt.subplots(1,3, figsize=(14, 6))\n\nscatter1 = axes[0].scatter(train['PCIAT-PCIAT_Total'], oof_xgb, c=train[\"sii\"], cmap=\"autumn\", alpha=0.5)\naxes[0].set_xlabel(\"True Score\")\naxes[0].set_ylabel(\"OOF Predictions - XGB\")\naxes[0].set_ylim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[0].set_xlim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[0].set_aspect('equal', adjustable='box')\nthresholds = [30, 50, 80]\nfor threshold in thresholds:\n    axes[0].axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\nfor threshold in xgb_thresholds_ens:\n    axes[0].axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    \nscatter2 = axes[1].scatter(train['PCIAT-PCIAT_Total'], oof_lgb, c=train[\"sii\"], cmap=\"autumn\", alpha=0.5)\naxes[1].set_xlabel(\"True Score\")\naxes[1].set_ylabel(\"OOF Predictions - LGBM\")\naxes[1].set_ylim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[1].set_xlim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[1].set_aspect('equal', adjustable='box')\n\nfor threshold in thresholds:\n    axes[1].axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\nfor threshold in lgb_thresholds_ens:\n    axes[1].axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    \nscatter3 = axes[2].scatter(train['PCIAT-PCIAT_Total'], oof_cat, c=train[\"sii\"], cmap=\"autumn\", alpha=0.5)\naxes[2].set_xlabel(\"True Score\")\naxes[2].set_ylabel(\"OOF Predictions - Cat\")\naxes[2].set_ylim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[2].set_xlim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[2].set_aspect('equal', adjustable='box')\n\nfor threshold in thresholds:\n    axes[2].axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\nfor threshold in cat_thresholds_ens:\n    axes[2].axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:36.654759Z","iopub.execute_input":"2025-03-25T08:45:36.655080Z","iopub.status.idle":"2025-03-25T08:45:37.658352Z","shell.execute_reply.started":"2025-03-25T08:45:36.655051Z","shell.execute_reply":"2025-03-25T08:45:37.656914Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1,2, figsize=(20,6))\n\nmodel_preds = pd.DataFrame({\n    'lgb': oof_lgb,\n    'xgb': oof_xgb,\n    'cat': oof_cat\n})\n\ncorr_df = model_preds.corr()\nsns.heatmap(corr_df, annot=True, cmap=\"autumn\", cbar=False, linewidths=0.5, linecolor='black', ax=axes[0])\naxes[0].set_title(\"Correlation Between Models\")\n\nlgb_thresholds_ens = np.mean(np.array(lgb_thresholds), axis=0)\nxgb_thresholds_ens = np.mean(np.array(xgb_thresholds), axis=0)\ncat_thresholds_ens = np.mean(np.array(cat_thresholds), axis=0)\nthresholds_df = pd.DataFrame({\n    \"LGB Thresholds\": lgb_thresholds_ens,\n    \"XGB Thresholds\": xgb_thresholds_ens,\n    \"Cat Thresholds\": cat_thresholds_ens\n})\n\nsns.heatmap(thresholds_df, annot=True, cmap=\"autumn\", cbar=False, linewidths=0.5, linecolor='black', ax=axes[1])\naxes[1].set_title(\"Optimal Thresholds Derived from CV\")\naxes[1].set_xticklabels(thresholds_df.columns, rotation=45)  \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:37.660068Z","iopub.execute_input":"2025-03-25T08:45:37.660518Z","iopub.status.idle":"2025-03-25T08:45:38.526668Z","shell.execute_reply.started":"2025-03-25T08:45:37.660445Z","shell.execute_reply":"2025-03-25T08:45:38.525411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"oof_lgb_index = round_with_thresholds(oof_lgb, lgb_thresholds_ens)\nprint(f\"LGBM optimized Kappa: {cohen_kappa_score(train['sii'], oof_lgb_index , weights='quadratic')}\")\n\n\noof_xgb_index  = round_with_thresholds(oof_xgb, xgb_thresholds_ens)\nprint(f\"XGB optimized Kappa: {cohen_kappa_score(train['sii'], oof_xgb_index , weights='quadratic')}\")\n\noof_cat_index  = round_with_thresholds(oof_cat, cat_thresholds_ens)\nprint(f\"CAT optimized Kappa: {cohen_kappa_score(train['sii'], oof_cat_index , weights='quadratic')}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:38.527801Z","iopub.execute_input":"2025-03-25T08:45:38.528114Z","iopub.status.idle":"2025-03-25T08:45:38.551731Z","shell.execute_reply.started":"2025-03-25T08:45:38.528087Z","shell.execute_reply":"2025-03-25T08:45:38.550286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Showing how sensitive QWK is with changes in threshold\nscores = []\nts = []\nm = 30\nfor i in tqdm(np.linspace(-2,10, 100)):\n    thresholds = [m+i, 50, 80]\n    pred = round_with_thresholds(oof_xgb, thresholds)\n    score = cohen_kappa_score(train[\"sii\"], pred, weights='quadratic')\n    ts.append(m+i)\n    scores.append(score)\n    \nplt.plot(ts, scores,color='orange')\nplt.title(\"Demonstration of QWK sensitivity to changes in threshold 0\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:38.553423Z","iopub.execute_input":"2025-03-25T08:45:38.553864Z","iopub.status.idle":"2025-03-25T08:45:39.181122Z","shell.execute_reply.started":"2025-03-25T08:45:38.553832Z","shell.execute_reply":"2025-03-25T08:45:39.179561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_importances = pd.DataFrame({\n    'Feature': lgb_features,\n    'Importance': lgb_model.feature_importances_\n}).sort_values(by='Importance', ascending=False)\n\nxgb_importances = pd.DataFrame({\n    'Feature': xgb_features,\n    'Importance': xgb_model.feature_importances_\n}).sort_values(by='Importance', ascending=False)\n\ncat_importances = pd.DataFrame({\n    'Feature': cat_features,\n    'Importance': cat_model.feature_importances_\n}).sort_values(by='Importance', ascending=False)\n\n# Set the number of features to display\nn_top_features = 40\n\nfig, axes = plt.subplots(1, 3, figsize=(18, 8))\nsns.set_theme(style=\"whitegrid\")\n\nsns.barplot(ax=axes[0], data=lgb_importances.head(n_top_features),\n            x='Importance', y='Feature', palette=\"autumn\")\naxes[0].set_title('LightGBM Top Feature Importances')\n\nsns.barplot(ax=axes[1], data=xgb_importances.head(n_top_features),\n            x='Importance', y='Feature', palette=\"autumn\")\naxes[1].set_title('XGBoost Top Feature Importances')\n\nsns.barplot(ax=axes[2], data=cat_importances.head(n_top_features),\n            x='Importance', y='Feature', palette=\"autumn\")\naxes[2].set_title('CatBoost Top Feature Importances')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:39.182336Z","iopub.execute_input":"2025-03-25T08:45:39.182760Z","iopub.status.idle":"2025-03-25T08:45:41.027069Z","shell.execute_reply.started":"2025-03-25T08:45:39.182726Z","shell.execute_reply":"2025-03-25T08:45:41.025558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"(set(lgb_importances[:10][\"Feature\"]) & set(xgb_importances[:10][\"Feature\"]) & set(cat_importances[:10][\"Feature\"]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:41.028125Z","iopub.execute_input":"2025-03-25T08:45:41.028415Z","iopub.status.idle":"2025-03-25T08:45:41.037782Z","shell.execute_reply.started":"2025-03-25T08:45:41.028391Z","shell.execute_reply":"2025-03-25T08:45:41.036553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"weights = [0.3,0.5,0.2]\noof_preds = np.array([oof_lgb_index, oof_xgb_index, oof_cat_index])\nweighted_oof = np.average(oof_preds, axis=0, weights=weights)\nfinal_oof = np.round(weighted_oof).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:41.039207Z","iopub.execute_input":"2025-03-25T08:45:41.039677Z","iopub.status.idle":"2025-03-25T08:45:41.059164Z","shell.execute_reply.started":"2025-03-25T08:45:41.039632Z","shell.execute_reply":"2025-03-25T08:45:41.057265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_oof","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:41.060341Z","iopub.execute_input":"2025-03-25T08:45:41.060680Z","iopub.status.idle":"2025-03-25T08:45:41.087401Z","shell.execute_reply.started":"2025-03-25T08:45:41.060651Z","shell.execute_reply":"2025-03-25T08:45:41.085953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.DataFrame(final_oof).to_csv(\"final_oof.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:41.088980Z","iopub.execute_input":"2025-03-25T08:45:41.089510Z","iopub.status.idle":"2025-03-25T08:45:41.112154Z","shell.execute_reply.started":"2025-03-25T08:45:41.089438Z","shell.execute_reply":"2025-03-25T08:45:41.110814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate Kappa score for voted OOF predictions\nkappa_score = cohen_kappa_score(train[\"sii\"], final_oof, weights='quadratic')\nprint(f\"Ensemble Kappa score: {kappa_score}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:41.113639Z","iopub.execute_input":"2025-03-25T08:45:41.114071Z","iopub.status.idle":"2025-03-25T08:45:41.138660Z","shell.execute_reply.started":"2025-03-25T08:45:41.114033Z","shell.execute_reply":"2025-03-25T08:45:41.137378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot confusion matrix\nconf_matrix = confusion_matrix(train[\"sii\"], final_oof)\nsns.set_theme(style=\"whitegrid\")\nplt.figure(figsize=(8, 6))\nsns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"autumn\", cbar=False, linewidths=0.5, linecolor='black')\nplt.title('Confusion Matrix', fontsize=16)\nplt.xlabel('Predicted', fontsize=12)\nplt.ylabel('True', fontsize=12)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:41.140086Z","iopub.execute_input":"2025-03-25T08:45:41.140533Z","iopub.status.idle":"2025-03-25T08:45:41.368328Z","shell.execute_reply.started":"2025-03-25T08:45:41.140463Z","shell.execute_reply":"2025-03-25T08:45:41.367065Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"conf_matrix = confusion_matrix(train[\"sii\"], oof_lgb_index)\nsns.set_theme(style=\"whitegrid\")\nplt.figure(figsize=(8, 6))\nsns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"autumn\", cbar=False, linewidths=0.5, linecolor='black')\nplt.title('Confusion Matrix', fontsize=16)\nplt.xlabel('Predicted', fontsize=12)\nplt.ylabel('True', fontsize=12)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:41.369869Z","iopub.execute_input":"2025-03-25T08:45:41.370371Z","iopub.status.idle":"2025-03-25T08:45:41.589379Z","shell.execute_reply.started":"2025-03-25T08:45:41.370329Z","shell.execute_reply":"2025-03-25T08:45:41.588091Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"conf_matrix = confusion_matrix(train[\"sii\"], oof_xgb_index)\nsns.set_theme(style=\"whitegrid\")\nplt.figure(figsize=(8, 6))\nsns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"autumn\", cbar=False, linewidths=0.5, linecolor='black')\nplt.title('Confusion Matrix', fontsize=16)\nplt.xlabel('Predicted', fontsize=12)\nplt.ylabel('True', fontsize=12)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:41.590647Z","iopub.execute_input":"2025-03-25T08:45:41.591004Z","iopub.status.idle":"2025-03-25T08:45:41.813681Z","shell.execute_reply.started":"2025-03-25T08:45:41.590977Z","shell.execute_reply":"2025-03-25T08:45:41.811900Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"conf_matrix = confusion_matrix(train[\"sii\"], oof_cat_index)\nsns.set_theme(style=\"whitegrid\")\nplt.figure(figsize=(8, 6))\nsns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"autumn\", cbar=False, linewidths=0.5, linecolor='black')\nplt.title('Confusion Matrix', fontsize=16)\nplt.xlabel('Predicted', fontsize=12)\nplt.ylabel('True', fontsize=12)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:41.814991Z","iopub.execute_input":"2025-03-25T08:45:41.815409Z","iopub.status.idle":"2025-03-25T08:45:42.032799Z","shell.execute_reply.started":"2025-03-25T08:45:41.815369Z","shell.execute_reply":"2025-03-25T08:45:42.031671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 设置绘图主题\nsns.set_theme(style=\"whitegrid\")\n# 创建包含 1 行 3 列子图的图形\nfig, axes = plt.subplots(1, 3, figsize=(14, 6))\n\n# 绘制 XGBoost 模型的混淆矩阵\nconf_matrix = confusion_matrix(train[\"sii\"], oof_xgb_index)\nsns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"autumn\", cbar=False, linewidths=0.5, linecolor='black', ax=axes[0])\naxes[0].set_title('Confusion Matrix (XGBoost)', fontsize=16)\naxes[0].set_xlabel('Predicted', fontsize=12)\naxes[0].set_ylabel('True', fontsize=12)\n\n# 绘制 LightGBM 模型的混淆矩阵\nconf_matrix = confusion_matrix(train[\"sii\"], oof_lgb_index)\nsns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"autumn\", cbar=False, linewidths=0.5, linecolor='black', ax=axes[1])\naxes[1].set_title('Confusion Matrix (LightGBM)', fontsize=16)\naxes[1].set_xlabel('Predicted', fontsize=12)\naxes[1].set_ylabel('True', fontsize=12)\n\n# 绘制 CatBoost 模型的混淆矩阵\nconf_matrix = confusion_matrix(train[\"sii\"], oof_cat_index)\nsns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"autumn\", cbar=False, linewidths=0.5, linecolor='black', ax=axes[2])\naxes[2].set_title('Confusion Matrix (CatBoost)', fontsize=16)\naxes[2].set_xlabel('Predicted', fontsize=12)\naxes[2].set_ylabel('True', fontsize=12)\n\n# 自动调整子图布局\nplt.tight_layout()\n# 显示图形\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:42.034141Z","iopub.execute_input":"2025-03-25T08:45:42.034586Z","iopub.status.idle":"2025-03-25T08:45:43.043292Z","shell.execute_reply.started":"2025-03-25T08:45:42.034545Z","shell.execute_reply":"2025-03-25T08:45:43.041670Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_lgb_index = round_with_thresholds(test_lgb, lgb_thresholds_ens)\ntest_xgb_index = round_with_thresholds(test_xgb, xgb_thresholds_ens)\ntest_cat_index = round_with_thresholds(test_cat, cat_thresholds_ens)\nif voting:\n    test_preds = np.array([test_lgb_index, test_xgb_index, test_cat_index])\n    voted_test = stats.mode(test_preds, axis=0).mode.flatten().astype(int)\n    final_test = np.round(voted_test).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:43.044530Z","iopub.execute_input":"2025-03-25T08:45:43.044835Z","iopub.status.idle":"2025-03-25T08:45:43.056656Z","shell.execute_reply.started":"2025-03-25T08:45:43.044809Z","shell.execute_reply":"2025-03-25T08:45:43.055195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:43.058091Z","iopub.execute_input":"2025-03-25T08:45:43.058515Z","iopub.status.idle":"2025-03-25T08:45:43.082049Z","shell.execute_reply.started":"2025-03-25T08:45:43.058447Z","shell.execute_reply":"2025-03-25T08:45:43.080840Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission['sii'] = final_test\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-25T08:45:43.082872Z","iopub.execute_input":"2025-03-25T08:45:43.083170Z","iopub.status.idle":"2025-03-25T08:45:43.094643Z","shell.execute_reply.started":"2025-03-25T08:45:43.083143Z","shell.execute_reply":"2025-03-25T08:45:43.093568Z"}},"outputs":[],"execution_count":null}]}