{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":11134441,"sourceType":"datasetVersion","datasetId":6944543}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd\nimport os\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score, make_scorer, confusion_matrix\nfrom sklearn.model_selection import StratifiedKFold, KFold, train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom scipy.optimize import minimize\nfrom scipy import stats\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport warnings\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.linear_model import ElasticNetCV, LassoCV, Lasso, LinearRegression\nfrom sklearn.ensemble import ExtraTreesRegressor, RandomForestRegressor\nfrom pytorch_tabnet.tab_model import TabNetRegressor\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nimport optuna\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator\nfrom matplotlib import font_manager\nimport seaborn as sns\nimport numpy as np\nimport random\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:31:45.382316Z","iopub.execute_input":"2025-06-15T08:31:45.382686Z","iopub.status.idle":"2025-06-15T08:31:54.748977Z","shell.execute_reply.started":"2025-06-15T08:31:45.382653Z","shell.execute_reply":"2025-06-15T08:31:54.748170Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 643\nn_splits = 10\noptimize_params = False\nn_trials = 25 # n_trials for optuna \nvoting = True\nbase_thresholds = [30, 50, 80]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:31:54.750213Z","iopub.execute_input":"2025-06-15T08:31:54.750781Z","iopub.status.idle":"2025-06-15T08:31:54.754925Z","shell.execute_reply.started":"2025-06-15T08:31:54.750754Z","shell.execute_reply":"2025-06-15T08:31:54.754166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\nTEST_PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'\nTRAIN_TS_PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet'\nTEST_TS_PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet'\nSUBMISSION_PATH = '/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv'\nOUTPUT_PATH = '/kaggle/working/'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:31:54.755734Z","iopub.execute_input":"2025-06-15T08:31:54.756108Z","iopub.status.idle":"2025-06-15T08:31:54.783156Z","shell.execute_reply.started":"2025-06-15T08:31:54.756084Z","shell.execute_reply":"2025-06-15T08:31:54.782439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def time_features(df):\n    \"\"\"从个人的ActiGraph数据中提取特征的函数\"\"\"\n    # 将一天中的时间转换为小时\n    df[\"hours\"] = df[\"time_of_day\"] // (3_600 * 1_000_000_000)\n    # 基本特征\n    features = [\n        df[\"non-wear_flag\"].mean(),\n        df[\"enmo\"][df[\"enmo\"] >= 0.05].sum(),\n    ]\n    \n    # 定义夜间、白天和无掩码（完整数据）的条件\n    night = ((df[\"hours\"] >= 22) | (df[\"hours\"] <= 5))\n    day = ((df[\"hours\"] <= 20) & (df[\"hours\"] >= 7))\n    no_mask = np.ones(len(df), dtype=bool)\n    \n    # 感兴趣的列和掩码列表\n    keys = [\"enmo\", \"anglez\", \"light\", \"battery_voltage\"]\n    masks = [no_mask, night, day]\n    \n    # 特征提取的辅助函数\n    def extract_stats(data):\n        return [\n            data.mean(), \n            data.std(), \n            data.max(), \n            data.min(), \n            data.diff().mean(), \n            data.diff().std()\n        ]\n    \n    # 遍历键和掩码以生成统计数据\n    for key in keys:\n        for mask in masks:\n            filtered_data = df.loc[mask, key]\n            features.extend(extract_stats(filtered_data))\n\n    return features\n\ndef process_file(filename, dirname):\n    # 处理文件并提取时间特征\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return time_features(df), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    # 从目录并行加载时间序列\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:31:54.784607Z","iopub.execute_input":"2025-06-15T08:31:54.784791Z","iopub.status.idle":"2025-06-15T08:31:54.800493Z","shell.execute_reply.started":"2025-06-15T08:31:54.784776Z","shell.execute_reply":"2025-06-15T08:31:54.799781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 加载表格数据\ntrain = pd.read_csv(TRAIN_PATH)\ntest = pd.read_csv(TEST_PATH)\n\n# 加载时序数据\ntrain_ts = load_time_series(TRAIN_TS_PATH)\ntest_ts = load_time_series(TEST_TS_PATH)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:31:54.801277Z","iopub.execute_input":"2025-06-15T08:31:54.801525Z","iopub.status.idle":"2025-06-15T08:32:51.121714Z","shell.execute_reply.started":"2025-06-15T08:31:54.801508Z","shell.execute_reply":"2025-06-15T08:32:51.120891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.shape)\nprint(train_ts.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:32:51.122621Z","iopub.execute_input":"2025-06-15T08:32:51.122996Z","iopub.status.idle":"2025-06-15T08:32:51.127334Z","shell.execute_reply.started":"2025-06-15T08:32:51.122970Z","shell.execute_reply":"2025-06-15T08:32:51.126792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:32:51.128113Z","iopub.execute_input":"2025-06-15T08:32:51.128369Z","iopub.status.idle":"2025-06-15T08:32:51.170404Z","shell.execute_reply.started":"2025-06-15T08:32:51.128344Z","shell.execute_reply":"2025-06-15T08:32:51.169677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 定义输出文件路径\nsubmission_path = os.path.join(OUTPUT_PATH, \"train_data.csv\")\ntrain.T.reset_index().to_csv(submission_path, index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:32:51.171362Z","iopub.execute_input":"2025-06-15T08:32:51.171661Z","iopub.status.idle":"2025-06-15T08:32:51.321713Z","shell.execute_reply.started":"2025-06-15T08:32:51.171637Z","shell.execute_reply":"2025-06-15T08:32:51.321016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 定义输出文件路径\nsubmission_path = os.path.join(OUTPUT_PATH, \"train_ts_data.csv\")\ntrain_ts.T.reset_index().to_csv(submission_path, index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:32:51.322609Z","iopub.execute_input":"2025-06-15T08:32:51.322899Z","iopub.status.idle":"2025-06-15T08:32:51.424448Z","shell.execute_reply.started":"2025-06-15T08:32:51.322852Z","shell.execute_reply":"2025-06-15T08:32:51.423465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:32:51.427597Z","iopub.execute_input":"2025-06-15T08:32:51.427807Z","iopub.status.idle":"2025-06-15T08:32:51.446266Z","shell.execute_reply.started":"2025-06-15T08:32:51.427792Z","shell.execute_reply":"2025-06-15T08:32:51.445682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#设置中文字体\nfont_path = \"/kaggle/input/simchn/SimplifiedChinese.ttf/SimplifiedChinese.ttf\"\nfont_manager.fontManager.addfont(font_path)\nfont_name = font_manager.FontProperties(fname=font_path).get_name()\nplt.rcParams[\"font.family\"] = font_name\nplt.rcParams[\"axes.unicode_minus\"] = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:46:45.412478Z","iopub.execute_input":"2025-06-15T08:46:45.413043Z","iopub.status.idle":"2025-06-15T08:46:45.418524Z","shell.execute_reply.started":"2025-06-15T08:46:45.413019Z","shell.execute_reply":"2025-06-15T08:46:45.417854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 生成缺失值条形图\n# 过滤 'sii' 非缺失的行\ntrain_filtered = train.dropna(subset=['sii'])\n\n# 计算缺失值比例\nmissing_ratio = train_filtered.isna().mean().sort_values(ascending=False)\n\n# 绘制缺失值条形图\nplt.figure(figsize=(20,16))\nsns.barplot(x=missing_ratio, y=missing_ratio.index, palette=\"autumn\")\nplt.xlabel(\"缺失值比例\")\nplt.ylabel(\"特征\")\nplt.title(\"仅包含 'sii' 非缺失行的缺失值分布\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:46:48.888799Z","iopub.execute_input":"2025-06-15T08:46:48.889073Z","iopub.status.idle":"2025-06-15T08:46:50.678322Z","shell.execute_reply.started":"2025-06-15T08:46:48.889052Z","shell.execute_reply":"2025-06-15T08:46:50.677461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('测试集里没有的列:')\nprint([f for f in train.columns if f not in test.columns])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:32:53.660850Z","iopub.execute_input":"2025-06-15T08:32:53.661270Z","iopub.status.idle":"2025-06-15T08:32:53.665801Z","shell.execute_reply.started":"2025-06-15T08:32:53.661251Z","shell.execute_reply":"2025-06-15T08:32:53.665071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 获取时序数据的列名，并去除 'id' 列\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\n# 将时序数据 train_ts 和 test_ts 通过 'id' 合并到主表train和test, 给合并后的数据添加后缀t以区分原数据。\ntrain_t = pd.merge(train, train_ts, how=\"left\", on='id')\ntest_t = pd.merge(test, test_ts, how=\"left\", on='id')\n\n# 删除 'id' 列，因为它不再需要\ntrain_t = train_t.drop('id', axis=1)\ntest_t = test_t.drop('id', axis=1)\n\n# 选择用于训练的特征列，包括基本人口统计信息、身体测量数据、体能测试数据、互联网使用情况等\nfeaturesCols = [\n    'Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n    'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n    'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n    'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n    'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n    'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n    'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n    'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n    'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n    'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n    'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n    'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n    'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n    'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n    'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n    'PAQ_C-PAQ_C_Total', 'PCIAT-PCIAT_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n    'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n    'PreInt_EduHx-computerinternet_hoursday', 'sii'  \n] # sii 为目标变量\n\n# 将时序特征列添加到特征列表中\nfeaturesCols += time_series_cols\n\n# 仅保留选定的特征列\ntrain_t = train_t[featuresCols]\n\n# 删除目标变量 'sii' 为空的行\ntrain_t = train_t.dropna(subset=['sii'])\n\n# 定义类别变量\ncat_c = [\n    'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n    'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season'\n]\n\n# 处理类别变量：填充缺失值并转换为类别类型\ndef update(df):\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')  # 用 'Missing' 填充缺失值\n        df[c] = df[c].astype('category')  # 转换为类别类型\n    return df\n        \ntrain_t = update(train_t)\ntest_t = update(test_t)\n\n# 创建类别变量的映射，将类别转换为整数索引\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\n\n# 应用类别映射，将类别变量转换为整数\nfor col in cat_c:\n    mapping_train = create_mapping(col, train_t)\n    mapping_test = create_mapping(col, test_t)\n    \n    train_t[col] = train_t[col].replace(mapping_train).astype(int)\n    test_t[col] = test_t[col].replace(mapping_test).astype(int)\n\n# 打印最终数据集的形状\nprint(f'Train Shape : {train_t.shape} || Test Shape : {test_t.shape}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:32:53.666826Z","iopub.execute_input":"2025-06-15T08:32:53.667038Z","iopub.status.idle":"2025-06-15T08:32:53.748493Z","shell.execute_reply.started":"2025-06-15T08:32:53.667020Z","shell.execute_reply":"2025-06-15T08:32:53.747701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(18, 6))\n\n# 1. 目标变量 sii 的分布（柱状图）\nsns.countplot(x=\"sii\", data=train, palette=\"autumn\", ax=axes[0])\naxes[0].set_title(\"SII各类别分布\")\naxes[0].set_xlabel(\"SII等级 (0: 无, 1: 轻度, 2: 中度, 3: 重度)\")\n\n# 2. 目标变量 sii 的分布（饼图）\nvc = train['sii'].value_counts()\n\n# 映射 sii 级别到中文标签\nsii_map = {0: '无', 1: '轻度', 2: '中度', 3: '重度'}\nlabels = [sii_map[label] for label in vc.index]\n\n# 绘制饼图\naxes[1].pie(vc.values, labels=labels, autopct=\"%1.1f%%\", colors=[\"#ff9999\", \"#66b3ff\", \"#99ff99\", \"#ffcc99\"])\naxes[1].set_title(\"SII各类别分布（饼图）\")\n\n# 3. 连续型指标 PCIAT-PCIAT_Total的分布（直方图+核密度估计）\nsns.histplot(train[\"PCIAT-PCIAT_Total\"], bins=30, kde=True, color='skyblue', ax=axes[2])\naxes[2].set_title(\"PCIAT-PCIAT_Total分布\")\naxes[2].set_xlabel(\"PCIAT-PCIAT_Total\")\n\n# 调整布局，使图像不重叠\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:32:53.749429Z","iopub.execute_input":"2025-06-15T08:32:53.750230Z","iopub.status.idle":"2025-06-15T08:32:54.559037Z","shell.execute_reply.started":"2025-06-15T08:32:53.750209Z","shell.execute_reply":"2025-06-15T08:32:54.558155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 计算 \"Basic_Demos-Enroll_Season\" 变量的各类别计数\nvc = train['Basic_Demos-Enroll_Season'].value_counts()\n\n# 映射季节名称到中文\nseason_map = {'Spring': '春季', 'Summer': '夏季', 'Fall': '秋季', 'Winter': '冬季'}\n\n# 使用映射创建中文标签\nlabels = [season_map[label] for label in vc.index]\n\nplt.pie(vc, labels=labels)\n\nplt.pie(vc.values, labels=labels, autopct=\"%1.1f%%\")  # 显示每个类别的占比\nplt.title('季节注册人数分布')  \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:32:54.559956Z","iopub.execute_input":"2025-06-15T08:32:54.560204Z","iopub.status.idle":"2025-06-15T08:32:54.716350Z","shell.execute_reply.started":"2025-06-15T08:32:54.560178Z","shell.execute_reply":"2025-06-15T08:32:54.715586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_stats(data, columns):\n    if isinstance(columns, str):\n        columns = [columns]\n\n    stats = []\n    for col in columns:\n        if data[col].dtype in ['object', 'category']:\n            counts = data[col].value_counts(dropna=False, sort=False)\n            percents = data[col].value_counts(normalize=True, dropna=False, sort=False) * 100\n            formatted = counts.astype(str) + ' (' + percents.round(2).astype(str) + '%)'\n            stats_col = pd.DataFrame({'count (%)': formatted})\n            stats.append(stats_col)\n        else:\n            stats_col = data[col].describe().to_frame().transpose()\n            stats_col['missing'] = data[col].isnull().sum()\n            stats_col.index.name = col\n            stats.append(stats_col)\n\n    return pd.concat(stats, axis=0)\n    \n# 按年龄给对象进行分组\ntrain_t['Age Group'] = pd.cut(\n    train_t['Basic_Demos-Age'],\n    bins=[4, 12, 18, 22],\n    labels=['儿童(5-12)', '青少年 (13-18)', '成年人 (19-22)']\n)\ncalculate_stats(train_t, 'Age Group')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:32:54.717274Z","iopub.execute_input":"2025-06-15T08:32:54.717583Z","iopub.status.idle":"2025-06-15T08:32:54.739495Z","shell.execute_reply.started":"2025-06-15T08:32:54.717561Z","shell.execute_reply":"2025-06-15T08:32:54.738764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 创建 1×2 的画布\nfig, axs = plt.subplots(1, 2, figsize=(14, 6))\n\n# 1. 绘制性别比例饼图\nvc = train['Basic_Demos-Sex'].value_counts()  # 计算性别分布\ncounts = vc.values  # 获取各性别的样本数量\nlabels = ['男孩', '女孩']  # 'Male' 对应 '男孩'，'Female' 对应 '女孩'\n\naxs[0].pie(counts, labels=labels, autopct=\"%1.1f%%\", colors=['lightblue', 'coral'])  # 绘制饼图\naxs[0].set_title('性别比例')  \n\n# 2. 绘制性别年龄分布柱状图\nfor sex in range(2):\n    ax = axs[1]  \n\n    # 通过布尔索引筛选数据\n    sex_filter = train['Basic_Demos-Sex'] == sex\n    vc = train[sex_filter]['Basic_Demos-Age'].value_counts()  # 计算不同年龄的样本数量\n\n    # 绘制柱状图\n    ax.bar(vc.index, vc.values, color=['lightblue', 'coral'][sex], label=['男孩', '女孩'][sex])\n    ax.xaxis.set_major_locator(MaxNLocator(integer=True)) \n    ax.set_ylabel('样本数量')\n    ax.legend() \n\n\nplt.suptitle('性别与年龄分布', fontsize=16)\naxs[1].set_xlabel('年龄')  \n\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:32:54.740322Z","iopub.execute_input":"2025-06-15T08:32:54.740667Z","iopub.status.idle":"2025-06-15T08:32:55.155547Z","shell.execute_reply.started":"2025-06-15T08:32:54.740643Z","shell.execute_reply":"2025-06-15T08:32:55.154870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 创建 1×3 的画布\nfig, axes = plt.subplots(1, 3, figsize=(18, 5))\n\n# 1. SII 与年龄的关系（箱型图）\nsns.boxplot(y=train['Basic_Demos-Age'], x=train_t['sii'], ax=axes[0], palette=\"Set3\")\naxes[0].set_title('SII 与年龄的关系')  # 设置标题\naxes[0].set_ylabel('年龄')  # 设置 y 轴标签\naxes[0].set_xlabel('SII 等级')  # 设置 x 轴标签\n\n# 2. 不同年龄组的完整 PCIAT 评分（箱型图）\nsns.boxplot(\n    x='Age Group', y='PCIAT-PCIAT_Total',\n    data=train_t, palette=\"Set3\", ax=axes[1]\n)\naxes[1].set_title('不同年龄组的完整 PCIAT 评分')  # 设置标题\naxes[1].set_ylabel('完整答题的 PCIAT 总分')  # 设置 y 轴标签\naxes[1].set_xlabel('年龄组')  # 设置 x 轴标签\n\n# 3. 不同性别的 PCIAT 总分分布（直方图）\nsns.histplot(\n    data=train_t, x='PCIAT-PCIAT_Total',\n    hue='Basic_Demos-Sex', multiple='stack',\n    palette=\"Set3\", bins=20, ax=axes[2]\n)\naxes[2].set_title('不同性别的 PCIAT 总分分布')  # 设置标题\naxes[2].set_xlabel('完整答题的 PCIAT 总分')  # 设置 x 轴标签\naxes[2].set_ylabel('频率')  # 设置 y 轴标签\n\n# 调整布局，避免重叠\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:32:55.156584Z","iopub.execute_input":"2025-06-15T08:32:55.157384Z","iopub.status.idle":"2025-06-15T08:32:56.121020Z","shell.execute_reply.started":"2025-06-15T08:32:55.157357Z","shell.execute_reply":"2025-06-15T08:32:56.120246Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 计算不同年龄组的 SII 分布\nstats = train_t.groupby(['Age Group', 'sii']).size().unstack(fill_value=0)\n\n# 创建 1×N 的画布，每个年龄组一个饼图\nfig, axes = plt.subplots(1, len(stats), figsize=(18, 5))\n\n# 遍历每个年龄组，绘制 SII 分布饼图\nfor i, age_group in enumerate(stats.index):\n    group_counts = stats.loc[age_group] / stats.loc[age_group].sum()  # 计算各 SII 级别的占比\n    axes[i].pie(\n        group_counts, labels=group_counts.index, autopct='%1.1f%%',\n        startangle=90, colors=sns.color_palette(\"Set3\"),\n        labeldistance=1.05, pctdistance=0.80\n    )\n    axes[i].set_title(f'{age_group} 的 SII 分布') \n    axes[i].axis('equal') \n\nplt.tight_layout()\nplt.show()\n\n# 计算不同性别的 SII 分布\nstats = train_t.groupby(['Basic_Demos-Sex', 'sii']).size().unstack(fill_value=0)\n\n# 创建 1×2 的画布，每个性别一个饼图\nfig, axes = plt.subplots(1, len(stats), figsize=(18, 5))\n\n# 映射性别 0 和 1 为 \"男性\" 和 \"女性\"\nsex_map = {0: \"男性\", 1: \"女性\"}\n\n# 遍历每个性别，绘制 SII 分布饼图\nfor i, sex in enumerate(stats.index):\n    group_counts = stats.loc[sex] / stats.loc[sex].sum()  # 计算各 SII 级别的占比\n    axes[i].pie(\n        group_counts, labels=group_counts.index, autopct='%1.1f%%',\n        startangle=90, colors=sns.color_palette(\"Set3\"),\n        labeldistance=1.05, pctdistance=0.80\n    )\n    axes[i].set_title(f'{sex_map[sex]} 的 SII 分布')\n    axes[i].axis('equal')  \n\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:32:56.121951Z","iopub.execute_input":"2025-06-15T08:32:56.122214Z","iopub.status.idle":"2025-06-15T08:32:56.795715Z","shell.execute_reply.started":"2025-06-15T08:32:56.122187Z","shell.execute_reply":"2025-06-15T08:32:56.794894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 选取连续性变量，计算相关矩阵\ncontinuous_columns = [\n    'Basic_Demos-Age', 'Physical-BMI', 'Physical-Height', 'Physical-Weight',\n    'FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_TL',\n    'FGC-FGC_GSND', 'FGC-FGC_GSD', 'FGC-FGC_SRL', 'FGC-FGC_SRR',\n    'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE',\n    'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI',\n    'BIA-BIA_Fat', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST',\n    'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total', 'sii'\n]\n\ncorrelation_matrix = train_t[continuous_columns].corr()\n\n# 绘制相关性热图\nplt.figure(figsize=(18, 15))  \nsns.heatmap(\n    correlation_matrix, \n    annot=True,  \n    fmt=\".2f\",  \n    cmap=\"coolwarm\", \n    center=0,  \n    annot_kws={\"size\": 8},  \n    cbar_kws={\"shrink\": 0.8} \n)\n\nplt.xticks(rotation=90, fontsize=10)  \nplt.yticks(rotation=0, fontsize=10)   \nplt.title(\"连续变量与 SII 的相关性热图\", fontsize=18, color=\"#004080\") \nplt.tight_layout() \n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:32:56.796483Z","iopub.execute_input":"2025-06-15T08:32:56.796696Z","iopub.status.idle":"2025-06-15T08:33:00.300646Z","shell.execute_reply.started":"2025-06-15T08:32:56.796680Z","shell.execute_reply":"2025-06-15T08:33:00.299940Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(18, 5))\n\n# 1. 年龄与体重的关系（散点图）\nplt.subplot(1, 3, 1)  \nsns.scatterplot(x='Basic_Demos-Age', y='Physical-Weight', data=train)  \nplt.title('年龄与体重的关系')  \nplt.xlabel('年龄')\nplt.ylabel('体重 (kg)') \n\n# 2. 年龄与身高的关系（散点图）\nplt.subplot(1, 3, 2)  \nsns.scatterplot(x='Basic_Demos-Age', y='Physical-Height', data=train)  \nplt.title('年龄与身高的关系')  \nplt.xlabel('年龄')  \nplt.ylabel('身高 (cm)')  \n\nplt.tight_layout()  \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:00.301618Z","iopub.execute_input":"2025-06-15T08:33:00.301849Z","iopub.status.idle":"2025-06-15T08:33:00.766650Z","shell.execute_reply.started":"2025-06-15T08:33:00.301831Z","shell.execute_reply":"2025-06-15T08:33:00.765919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 定义 FGC 类相关变量\nFGC_columns = ['FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_TL',\n    'FGC-FGC_GSND', 'FGC-FGC_GSD', 'FGC-FGC_SRL', 'FGC-FGC_SRR']\n\n# 获取所有年龄组\nage_groups = train_t['Age Group'].unique()\n\n# 创建 1×3 的画布，设置共享 y 轴\nfig, axes = plt.subplots(1, 3, figsize=(18, 6), sharey=True)\n\n# 遍历每个年龄组，绘制相关性热力图\nfor i, age_group in enumerate(age_groups):\n    group_data = train_t[train_t['Age Group'] == age_group]  # 筛选当前年龄组的数据\n    corr_matrix = group_data[FGC_columns + ['PCIAT-PCIAT_Total', 'Basic_Demos-Age']].corr()  # 计算相关矩阵\n    \n    # 绘制相关性热力图\n    sns.heatmap(corr_matrix, annot=True, cmap='coolwarm', fmt='.1f',\n                vmin=-1, vmax=1, ax=axes[i], cbar=i == 0)  \n    \n    axes[i].set_title(f'{age_group} 组的相关性热力图') \n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:00.767283Z","iopub.execute_input":"2025-06-15T08:33:00.767460Z","iopub.status.idle":"2025-06-15T08:33:02.801047Z","shell.execute_reply.started":"2025-06-15T08:33:00.767446Z","shell.execute_reply":"2025-06-15T08:33:02.800219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 定义 BIA 类相关变量\nBIA_columns = ['BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE',\n    'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI',\n    'BIA-BIA_Fat', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST',\n    'BIA-BIA_SMM', 'BIA-BIA_TBW']\n\nplt.figure(figsize=(24, 20))\n\n# 遍历每个 BIA 变量，绘制直方图\nfor idx, col in enumerate(BIA_columns):\n    plt.subplot(4, 4, idx + 1)  # 创建 4×4 的子图布局\n    sns.histplot(train_t[col].dropna(), bins=20, kde=True)  # 绘制直方图，去除缺失值\n    plt.title(f'{col} 的分布')  \n    plt.xlabel('数值')  \n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:02.801810Z","iopub.execute_input":"2025-06-15T08:33:02.802049Z","iopub.status.idle":"2025-06-15T08:33:06.703404Z","shell.execute_reply.started":"2025-06-15T08:33:02.802031Z","shell.execute_reply":"2025-06-15T08:33:06.702581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 读取数据字典文件\ndata_dictionary = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\n\n# 筛选出数据类型为整数（int），但不包含类别型（categorical）的变量\ninteger_columns = list(data_dictionary[\n    data_dictionary['Type'].str.contains('int') & ~data_dictionary['Type'].str.contains('categorical')\n][\"Field\"].unique())\n\n# 计算每个整数变量的统计信息\ncolumn_stats = {\n    col: {\n        '缺失值比例': train_t[col].isna().mean() * 100,  # 计算缺失值占比（百分比）\n        '唯一值数量': train_t[col].nunique(),  # 计算该列的唯一值数量\n        '均值': train_t[col].mean(),  # 计算均值\n        '标准差': train_t[col].std(),  # 计算标准差\n        '最小值': train_t[col].min(),  # 计算最小值\n        '25%分位数': train_t[col].quantile(0.25),  # 计算 25% 分位数（第一四分位数）\n        '中位数': train_t[col].median(),  # 计算中位数\n        '75%分位数': train_t[col].quantile(0.75),  # 计算 75% 分位数（第三四分位数）\n        '最大值': train_t[col].max()  # 计算最大值\n    } \n    for col in integer_columns\n}\n\n# 将统计信息转换为 DataFrame 以便展示\nstats_df = pd.DataFrame.from_dict(column_stats, orient='index').reset_index()\nstats_df.columns = [\n    '变量名', '缺失值比例', '唯一值数量', \n    '均值', '标准差', '最小值', '25%分位数', '中位数', '75%分位数', '最大值'\n]\n\n# 按缺失值比例降序排序\nstats_df = stats_df.sort_values(by='缺失值比例', ascending=False).reset_index(drop=True)\n\n# 显示统计信息 DataFrame\nstats_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:06.704139Z","iopub.execute_input":"2025-06-15T08:33:06.704324Z","iopub.status.idle":"2025-06-15T08:33:06.753903Z","shell.execute_reply.started":"2025-06-15T08:33:06.704310Z","shell.execute_reply":"2025-06-15T08:33:06.753288Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 创建 1×2 的画布\nfig, axes = plt.subplots(1, 2, figsize=(16, 6))\n\n# 1. SII 与睡眠障碍评分的散点图\naxes[0].scatter(train['sii'], train['SDS-SDS_Total_Raw'], color='royalblue', alpha=0.7)  # 绘制散点图\naxes[0].set_xlabel('SII 等级') \naxes[0].set_ylabel('睡眠障碍评分')  \naxes[0].set_title('SII 与睡眠障碍评分的散点图')  \naxes[0].grid(True)  \n\n# 2. 睡眠障碍评分的小提琴图\nsns.violinplot(x='sii', y='SDS-SDS_Total_Raw', data=train, color='royalblue', ax=axes[1])  # 绘制小提琴图\naxes[1].set_xlabel('SII 等级')  \naxes[1].set_ylabel('睡眠障碍评分')  \naxes[1].set_title('睡眠障碍评分的分布（小提琴图）')  \n\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:06.754610Z","iopub.execute_input":"2025-06-15T08:33:06.754817Z","iopub.status.idle":"2025-06-15T08:33:07.286617Z","shell.execute_reply.started":"2025-06-15T08:33:06.754802Z","shell.execute_reply":"2025-06-15T08:33:07.285767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def clean_features(df):\n    # 移除不合理的数值\n\n    # 限制握力测试数据范围（FGC-FGC_GSND 和 FGC-FGC_GSD）\n    df[['FGC-FGC_GSND', 'FGC-FGC_GSD']] = df[['FGC-FGC_GSND', 'FGC-FGC_GSD']].clip(lower=9, upper=60)\n\n    # 处理不合理的体脂率数据（BIA-BIA_Fat）\n    df[\"BIA-BIA_Fat\"] = np.where(df[\"BIA-BIA_Fat\"] < 5, np.nan, df[\"BIA-BIA_Fat\"])  # 低于 5% 设为 NaN\n    df[\"BIA-BIA_Fat\"] = np.where(df[\"BIA-BIA_Fat\"] > 60, np.nan, df[\"BIA-BIA_Fat\"])  # 高于 60% 设为 NaN\n\n    # 处理基础代谢率（BIA-BIA_BMR）\n    df[\"BIA-BIA_BMR\"] = np.where(df[\"BIA-BIA_BMR\"] > 4000, np.nan, df[\"BIA-BIA_BMR\"])  # 超过 4000 设为 NaN\n\n    # 处理每日能量消耗（BIA-BIA_DEE）\n    df[\"BIA-BIA_DEE\"] = np.where(df[\"BIA-BIA_DEE\"] > 8000, np.nan, df[\"BIA-BIA_DEE\"])  # 超过 8000 设为 NaN\n\n    # 处理骨矿物质含量（BIA-BIA_BMC）\n    df[\"BIA-BIA_BMC\"] = np.where(df[\"BIA-BIA_BMC\"] <= 0, np.nan, df[\"BIA-BIA_BMC\"])  # 小于等于 0 设为 NaN\n    df[\"BIA-BIA_BMC\"] = np.where(df[\"BIA-BIA_BMC\"] > 10, np.nan, df[\"BIA-BIA_BMC\"])  # 超过 10 设为 NaN\n\n    # 处理去脂体重（BIA-BIA_FFM）\n    df[\"BIA-BIA_FFM\"] = np.where(df[\"BIA-BIA_FFM\"] <= 0, np.nan, df[\"BIA-BIA_FFM\"])  # 小于等于 0 设为 NaN\n    df[\"BIA-BIA_FFM\"] = np.where(df[\"BIA-BIA_FFM\"] > 300, np.nan, df[\"BIA-BIA_FFM\"])  # 超过 300 设为 NaN\n\n    # 处理脂肪质量指数（BIA-BIA_FMI）\n    df[\"BIA-BIA_FMI\"] = np.where(df[\"BIA-BIA_FMI\"] < 0, np.nan, df[\"BIA-BIA_FMI\"])  # 小于 0 设为 NaN\n\n    # 处理细胞外水分（BIA-BIA_ECW）\n    df[\"BIA-BIA_ECW\"] = np.where(df[\"BIA-BIA_ECW\"] > 100, np.nan, df[\"BIA-BIA_ECW\"])  # 超过 100 设为 NaN\n\n    # 处理瘦干质量（BIA-BIA_LDM）\n    df[\"BIA-BIA_LDM\"] = np.where(df[\"BIA-BIA_LDM\"] > 100, np.nan, df[\"BIA-BIA_LDM\"])  # 超过 100 设为 NaN\n\n    # 处理瘦软组织（BIA-BIA_LST）\n    df[\"BIA-BIA_LST\"] = np.where(df[\"BIA-BIA_LST\"] > 300, np.nan, df[\"BIA-BIA_LST\"])  # 超过 300 设为 NaN\n\n    # 处理骨骼肌质量（BIA-BIA_SMM）\n    df[\"BIA-BIA_SMM\"] = np.where(df[\"BIA-BIA_SMM\"] > 300, np.nan, df[\"BIA-BIA_SMM\"])  # 超过 300 设为 NaN\n\n    # 处理总身体水分（BIA-BIA_TBW）\n    df[\"BIA-BIA_TBW\"] = np.where(df[\"BIA-BIA_TBW\"] > 300, np.nan, df[\"BIA-BIA_TBW\"])  # 超过 300 设为 NaN\n\n    return df\n\n# 清理训练集和测试集数据\ntrain = clean_features(train)\ntest = clean_features(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:07.287329Z","iopub.execute_input":"2025-06-15T08:33:07.287557Z","iopub.status.idle":"2025-06-15T08:33:07.312501Z","shell.execute_reply.started":"2025-06-15T08:33:07.287517Z","shell.execute_reply":"2025-06-15T08:33:07.311929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def perform_pca(train, test, n_components=None, random_state=42):\n    # 初始化 PCA 模型\n    pca = PCA(n_components=n_components, random_state=random_state)\n    \n    # 在训练数据上拟合 PCA 并转换数据\n    train_pca = pca.fit_transform(train)\n    \n    # 在测试数据上应用相同的 PCA 变换\n    test_pca = pca.transform(test)\n    \n    # 获取各主成分的解释方差比\n    explained_variance_ratio = pca.explained_variance_ratio_\n    print(f\"各主成分的解释方差比:\\n {explained_variance_ratio}\")\n    print(f\"总解释方差: {np.sum(explained_variance_ratio)}\")\n    \n    # 将转换后的数据转换为 DataFrame，并为主成分命名\n    train_pca_df = pd.DataFrame(train_pca, columns=[f'PC_{i+1}' for i in range(train_pca.shape[1])])\n    test_pca_df = pd.DataFrame(test_pca, columns=[f'PC_{i+1}' for i in range(test_pca.shape[1])])\n    \n    return train_pca_df, test_pca_df, pca","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:07.316343Z","iopub.execute_input":"2025-06-15T08:33:07.317141Z","iopub.status.idle":"2025-06-15T08:33:07.322881Z","shell.execute_reply.started":"2025-06-15T08:33:07.317115Z","shell.execute_reply":"2025-06-15T08:33:07.322036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 移除 'id' 列，仅保留时序数据\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\n# 在执行PCA之前进行标准化处理\nscaler = StandardScaler()\ndf_train = pd.DataFrame(scaler.fit_transform(df_train), columns=df_train.columns)  # 训练数据标准化\ndf_test = pd.DataFrame(scaler.transform(df_test), columns=df_test.columns)  # 测试数据标准化（使用相同的缩放参数）\n\n# 在PCA之前对时序数据进行均值填充（处理缺失值）\nfor c in df_train.columns:\n    m = np.mean(df_train[c])  # 计算该列的均值\n    df_train[c].fillna(m, inplace=True)  # 用均值填充训练数据中的缺失值\n    df_test[c].fillna(m, inplace=True)  # 用训练数据的均值填充测试数据中的缺失值\n\n# 输出训练数据的形状\nprint(df_train.shape)\n\n# 执行 PCA 降维，选择15个主成分\ndf_train_pca, df_test_pca, pca = perform_pca(df_train, df_test, n_components=15, random_state=SEED)\n\n# 重新添加 'id' 列，以便后续合并\ndf_train_pca['id'] = train_ts['id']\ndf_test_pca['id'] = test_ts['id']\n\n# 将 PCA 处理后的数据合并回原始训练集和测试集\ntrain = pd.merge(train, df_train_pca, how=\"left\", on='id')\ntest = pd.merge(test, df_test_pca, how=\"left\", on='id')\n\n# 输出最终训练数据的形状\ntrain.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:07.323708Z","iopub.execute_input":"2025-06-15T08:33:07.323972Z","iopub.status.idle":"2025-06-15T08:33:07.446141Z","shell.execute_reply.started":"2025-06-15T08:33:07.323950Z","shell.execute_reply":"2025-06-15T08:33:07.445323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    # 1. 删除季节相关变量\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1)  # 删除所有包含 \"Season\" 的列\n    \n    # 2. 创建年龄组（Age Group）\n    def assign_group(age):\n        thresholds = [5, 6, 7, 8, 10, 12, 14, 17, 22]  # 年龄分组阈值\n        for i, j in enumerate(thresholds):\n            if age <= j:\n                return i  # 根据年龄分配组别\n        return np.nan\n    \n    df[\"group\"] = df['Basic_Demos-Age'].apply(assign_group)  # 应用年龄分组函数\n    \n    # 3. 归一化 BMI\n    BMI_map = {0: 16.3, 1: 15.9, 2: 16.1, 3: 16.8, 4: 17.3, 5: 19.2, 6: 20.2, 7: 22.3, 8: 23.6}\n    df['BMI_mean_norm'] = df[['Physical-BMI', 'BIA-BIA_BMI']].mean(axis=1) / df[\"group\"].map(BMI_map)\n    \n    # 4. FGC 区域特征聚合\n    zones = ['FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone',\n             'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone',\n             'FGC-FGC_TL_Zone']\n    \n    df['FGC_Zones_mean'] = df[zones].mean(axis=1)  # 计算 FGC 区域均值\n    df['FGC_Zones_min'] = df[zones].min(axis=1)  # 计算 FGC 区域最小值\n    df['FGC_Zones_max'] = df[zones].max(axis=1)  # 计算 FGC 区域最大值\n    \n    # 5. 握力测试归一化\n    GSD_max_map = {0: 9, 1: 9, 2: 9, 3: 9, 4: 16.2, 5: 19.9, 6: 26.1, 7: 31.3, 8: 35.4}\n    GSD_min_map = {0: 9, 1: 9, 2: 9, 3: 9, 4: 14.4, 5: 17.8, 6: 23.4, 7: 27.8, 8: 31.1}\n    \n    df['GS_max'] = df[['FGC-FGC_GSND', 'FGC-FGC_GSD']].max(axis=1) / df[\"group\"].map(GSD_max_map)\n    df['GS_min'] = df[['FGC-FGC_GSND', 'FGC-FGC_GSD']].min(axis=1) / df[\"group\"].map(GSD_min_map)\n    \n    # 6. 仰卧起坐、俯卧撑、躯干抬升归一化\n    cu_map = {0: 1.0, 1: 3.0, 2: 5.0, 3: 7.0, 4: 10.0, 5: 14.0, 6: 20.0, 7: 20.0, 8: 20.0}\n    pu_map = {0: 1.0, 1: 2.0, 2: 3.0, 3: 4.0, 4: 5.0, 5: 7.0, 6: 8.0, 7: 10.0, 8: 14.0}\n    tl_map = {0: 8.0, 1: 8.0, 2: 8.0, 3: 9.0, 4: 9.0, 5: 10.0, 6: 10.0, 7: 10.0, 8: 10.0}\n    \n    df[\"CU_norm\"] = df['FGC-FGC_CU'] / df['group'].map(cu_map)\n    df[\"PU_norm\"] = df['FGC-FGC_PU'] / df['group'].map(pu_map)\n    df[\"TL_norm\"] = df['FGC-FGC_TL'] / df['group'].map(tl_map)\n    \n    # 7. 伸展测试\n    df[\"SR_min\"] = df[['FGC-FGC_SRL', 'FGC-FGC_SRR']].min(axis=1)  # 计算最小值\n    df[\"SR_max\"] = df[['FGC-FGC_SRL', 'FGC-FGC_SRR']].max(axis=1)  # 计算最大值\n\n    # 8. BIA 特征归一化\n    # 能量消耗归一化\n    bmr_map = {0: 934.0, 1: 941.0, 2: 999.0, 3: 1048.0, 4: 1283.0, 5: 1255.0, 6: 1481.0, 7: 1519.0, 8: 1650.0}\n    dee_map = {0: 1471.0, 1: 1508.0, 2: 1640.0, 3: 1735.0, 4: 2132.0, 5: 2121.0, 6: 2528.0, 7: 2566.0, 8: 2793.0}\n    \n    df[\"BMR_norm\"] = df[\"BIA-BIA_BMR\"] / df[\"group\"].map(bmr_map)\n    df[\"DEE_norm\"] = df[\"BIA-BIA_DEE\"] / df[\"group\"].map(dee_map)\n    df[\"DEE_BMR\"] = df[\"BIA-BIA_DEE\"] - df[\"BIA-BIA_BMR\"]  # 计算每日能量消耗与基础代谢率的差值\n\n    # 9. 去脂体重归一化\n    ffm_map = {0: 42.0, 1: 43.0, 2: 49.0, 3: 54.0, 4: 60.0, 5: 76.0, 6: 94.0, 7: 104.0, 8: 111.0}\n    df[\"FFM_norm\"] = df[\"BIA-BIA_FFM\"] / df[\"group\"].map(ffm_map)\n\n    # 10. 细胞外水分与细胞内水分比值\n    df[\"ICW_ECW\"] = df[\"BIA-BIA_ECW\"] / df[\"BIA-BIA_ICW\"]\n    \n    # 11. 删除冗余特征\n    drop_feats = ['FGC-FGC_GSND', 'FGC-FGC_GSD', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone',\n                  'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL_Zone',\n                  'Physical-BMI', 'BIA-BIA_BMI', 'FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_TL', 'FGC-FGC_SRL', 'FGC-FGC_SRR',\n                 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_Frame_num', \"BIA-BIA_FFM\"]\n    \n    df = df.drop(drop_feats, axis=1)  # 删除不需要的特征\n    \n    return df\n# 实施特征工程\ntrain = feature_engineering(train)\ntest = feature_engineering(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:07.447264Z","iopub.execute_input":"2025-06-15T08:33:07.447497Z","iopub.status.idle":"2025-06-15T08:33:07.525427Z","shell.execute_reply.started":"2025-06-15T08:33:07.447478Z","shell.execute_reply":"2025-06-15T08:33:07.524703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def bin_data(train, test, columns, n_bins=10):\n   \n    # 1. 合并训练集和测试集，以确保分箱边界一致**\n    combined = pd.concat([train, test], axis=0)\n    \n    # 2. 计算每个特征的分箱边界**\n    bin_edges = {}\n    for col in columns:\n        # 使用 `qcut` 计算分位数分箱边界\n        edges = pd.qcut(combined[col], n_bins, retbins=True, labels=range(n_bins), duplicates=\"drop\")[1]\n        bin_edges[col] = edges  # 存储分箱边界\n    \n    # 3. 在训练集和测试集中应用相同的分箱边界\n    for col, edges in bin_edges.items():\n        train[col] = pd.cut(\n            train[col], bins=edges, labels=range(len(edges) - 1), include_lowest=True\n        ).astype(float)  # 训练集分箱\n        test[col] = pd.cut(\n            test[col], bins=edges, labels=range(len(edges) - 1), include_lowest=True\n        ).astype(float)  # 测试集分箱\n    \n    return train, test\n\n# 需要进行分箱的特征列表\ncolumns_to_bin = [\n    \"PAQ_A-PAQ_A_Total\", \"BMR_norm\", \"DEE_norm\", \"GS_min\", \"GS_max\", \"BIA-BIA_FFMI\", \n    \"BIA-BIA_BMC\", \"Physical-HeartRate\", \"BIA-BIA_ICW\", \"Fitness_Endurance-Time_Sec\", \n    \"BIA-BIA_LDM\", \"BIA-BIA_SMM\", \"BIA-BIA_TBW\", \"DEE_BMR\", \"ICW_ECW\"\n]\n\n# 对训练集和测试集的指定特征进行分箱\ntrain, test = bin_data(train, test, columns_to_bin, n_bins=10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:07.527957Z","iopub.execute_input":"2025-06-15T08:33:07.528497Z","iopub.status.idle":"2025-06-15T08:33:07.593898Z","shell.execute_reply.started":"2025-06-15T08:33:07.528473Z","shell.execute_reply":"2025-06-15T08:33:07.593232Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. 需要排除的特征（因为测试集中没有这些变量）\nexclude = ['PCIAT-Season', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03',\n           'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07',\n           'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11',\n           'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15',\n           'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19',\n           'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total', 'sii', 'id']\n\n# 2. 目标变量\ny_model = \"PCIAT-PCIAT_Total\"  # 预测模型的目标变量（PCIAT 总分）\ny_comp = \"sii\"  # 竞赛目标变量（SII 指数）\n\n# 3. 选取用于训练的特征\nfeatures = [f for f in train.columns if f not in exclude]  # 仅保留未被排除的特征\n\n# 4. 处理类别型特征\ncat_c = []  # 目前没有需要处理的类别型特征\n\n# 5. 映射类别型特征（如果有的话）\nfor col in cat_c:\n    a_map = {}  # 创建映射字典\n    all_unique = set(train[col].unique()) | set(test[col].unique())  # 获取训练集和测试集中的所有唯一值\n    for i, value in enumerate(all_unique):\n        a_map[value] = i  # 为每个类别分配一个数值\n\n    train[col] = train[col].map(a_map)  # 在训练集中应用映射\n    test[col] = test[col].map(a_map)  # 在测试集中应用映射\n\n# 6. 仅保留 `sii` 目标变量非空的行\ntrain = train[train[\"sii\"].notna()]  # 过滤掉 `sii` 为空的样本\n\n# 7. 输出训练集的形状\ntrain.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:07.594608Z","iopub.execute_input":"2025-06-15T08:33:07.594791Z","iopub.status.idle":"2025-06-15T08:33:07.605613Z","shell.execute_reply.started":"2025-06-15T08:33:07.594778Z","shell.execute_reply":"2025-06-15T08:33:07.604853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 使用机器学习模型进行缺失值填充\nclass Impute_With_Model:\n    def __init__(self, na_frac=0.5, min_samples=0):\n        self.model_dict = {}  # 存储用于填充的模型\n        self.mean_dict = {}  # 存储每个特征的均值（用于均值填充）\n        self.features = None  # 需要填充的特征列表\n        self.na_frac = na_frac  # 允许的最大缺失值比例\n        self.min_samples = min_samples  # 训练模型所需的最小样本数\n    \n    # 选择用于填充的特征（仅保留缺失比例 <= na_frac 的特征）    \n    def find_features(self, data, feature, tmp_features):\n        missing_rows = data[feature].isna()  # 找出目标特征缺失的行\n        na_fraction = data[missing_rows][tmp_features].isna().mean(axis=0)  # 计算候选特征的缺失比例\n        valid_features = np.array(tmp_features)[na_fraction <= self.na_frac]  # 仅保留缺失比例 <= na_frac 的特征\n        return valid_features\n    \n    # 训练填充模型\n    def fit_models(self, model, data, features):\n        self.features = features\n        n_data = data.shape[0]\n        \n        # 计算每个特征的均值（用于均值填充）\n        for feature in features:\n            self.mean_dict[feature] = np.mean(data[feature])\n        \n        # 遍历所有特征，训练填充模型\n        for feature in tqdm(features):\n            # 仅对存在缺失值的特征进行填充\n            if data[feature].isna().sum() > 0:\n                model_clone = clone(model)  # 复制模型\n                X = data[data[feature].notna()].copy()  # 仅使用非缺失值的数据进行训练\n                \n                # 选择用于填充的特征\n                tmp_features = [f for f in features if f != feature]\n                tmp_features = self.find_features(data, feature, tmp_features)\n                \n                if len(tmp_features) >= 1 and X.shape[0] > self.min_samples:\n                    # 仅在特征足够多且样本数足够时训练模型\n                    for f in tmp_features:\n                        X[f] = X[f].fillna(self.mean_dict[f])  # 用均值填充缺失值\n                    model_clone.fit(X[tmp_features], X[feature])  # 训练模型\n                    \n                    # 存储模型及其使用的特征\n                    self.model_dict[feature] = (model_clone, tmp_features.copy())\n                else:\n                    # 如果特征或样本数不足，则使用均值填充\n                    self.model_dict[feature] = (\"mean\", np.mean(data[feature]))\n            \n    # 使用训练好的模型填充缺失值\n    def impute(self, data):\n        \n        imputed_data = data.copy()\n        \n        # 遍历所有填充模型\n        for feature, model in self.model_dict.items():\n            missing_rows = imputed_data[feature].isna()  # 找出需要填充的行\n            \n            if missing_rows.any():\n                if model[0] == \"mean\":\n                    # 使用均值填充\n                    imputed_data[feature].fillna(model[1], inplace=True)\n                else:\n                    # 使用机器学习模型填充\n                    tmp_features = [f for f in self.features if f != feature]\n                    X_missing = data.loc[missing_rows, tmp_features].copy()\n                    \n                    # 用均值填充缺失值\n                    for f in tmp_features:\n                        X_missing[f] = X_missing[f].fillna(self.mean_dict[f])\n                    \n                    # 预测缺失值并填充\n                    imputed_data.loc[missing_rows, feature] = model[0].predict(X_missing[model[1]])\n        \n        return imputed_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:07.606492Z","iopub.execute_input":"2025-06-15T08:33:07.606998Z","iopub.status.idle":"2025-06-15T08:33:07.623949Z","shell.execute_reply.started":"2025-06-15T08:33:07.606970Z","shell.execute_reply":"2025-06-15T08:33:07.623239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 显示缺失比例 > 30% 的前 60 个特征\nmissing = pd.DataFrame(train.isna().sum() / len(train))\nmissing[missing[0] > 0.3][:60]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:07.624874Z","iopub.execute_input":"2025-06-15T08:33:07.625098Z","iopub.status.idle":"2025-06-15T08:33:07.651995Z","shell.execute_reply.started":"2025-06-15T08:33:07.625081Z","shell.execute_reply":"2025-06-15T08:33:07.651117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 训练填充模型\nmodel = LassoCV(cv=5, random_state=SEED)  # 使用 LassoCV 作为填充模型\nimputer = Impute_With_Model(na_frac=0.4)  # 允许的最大缺失比例为 40%\nimputer.fit_models(model, train, features)  # 训练填充模型\n\n# 使用填充模型填充训练集和测试集\ntrain = imputer.impute(train)\ntest = imputer.impute(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:07.652704Z","iopub.execute_input":"2025-06-15T08:33:07.652952Z","iopub.status.idle":"2025-06-15T08:33:16.066177Z","shell.execute_reply.started":"2025-06-15T08:33:07.652935Z","shell.execute_reply":"2025-06-15T08:33:16.065365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_weights(series):\n    # 1. 对目标变量进行分箱（将数据划分为 10 个区间）\n    bins = pd.cut(series, bins=10, labels=False)  # 使用 `pd.cut` 将数据分成 10 个区间\n    \n    # 2. 计算每个区间的样本数量\n    weights = bins.value_counts().reset_index()  # 统计每个区间的样本数量\n    weights.columns = ['target_bins', 'count']  # 重命名列\n    \n    # 3. 计算权重（样本数量的倒数）\n    weights['count'] = 1 / weights['count']  # 计算每个区间的权重（样本数量越少，权重越大）\n    \n    # 4. 创建权重映射表\n    weight_map = weights.set_index('target_bins')['count'].to_dict()  # 将权重映射为字典\n    \n    # 5. 将权重映射到原始数据\n    weights = bins.map(weight_map)  # 根据分箱结果为每个样本分配权重\n    \n    # 6. 归一化权重（确保均值为 1）\n    return weights / weights.mean()  # 归一化权重，使其均值为 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:16.067074Z","iopub.execute_input":"2025-06-15T08:33:16.067308Z","iopub.status.idle":"2025-06-15T08:33:16.072690Z","shell.execute_reply.started":"2025-06-15T08:33:16.067284Z","shell.execute_reply":"2025-06-15T08:33:16.071847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def round_with_thresholds(raw_preds, thresholds):\n    # 根据给定的阈值，将连续预测值转换为离散类别\n    return np.where(raw_preds < thresholds[0], int(0),\n                    np.where(raw_preds < thresholds[1], int(1),\n                             np.where(raw_preds < thresholds[2], int(2), int(3))))\n\ndef optimize_thresholds(y_true, raw_preds, start_vals=[0.5, 1.5, 2.5]):\n    # 通过优化寻找最佳阈值，使 Cohen's Kappa 分数最大化\n    def fun(thresholds, y_true, raw_preds):\n        # 计算当前阈值下的 Cohen's Kappa 分数（取负值用于最小化）\n        rounded_preds = round_with_thresholds(raw_preds, thresholds)  # 根据当前阈值转换预测值\n        return -cohen_kappa_score(y_true, rounded_preds, weights='quadratic')  # 计算 Kappa 分数（取负值用于优化）\n\n    # 使用 Powell 方法优化阈值\n    res = minimize(fun, x0=start_vals, args=(y_true, raw_preds), method='Powell')\n    \n    assert res.success  # 确保优化成功\n    return res.x  # 返回优化后的阈值","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:16.073474Z","iopub.execute_input":"2025-06-15T08:33:16.073700Z","iopub.status.idle":"2025-06-15T08:33:16.089789Z","shell.execute_reply.started":"2025-06-15T08:33:16.073684Z","shell.execute_reply":"2025-06-15T08:33:16.089059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def cross_validate(model_, data, features, score_col, index_col, cv, sample_weights=False, verbose=False):\n    # 使用交叉验证评估模型，并计算每折的 Cohen's Kappa 分数，同时获取全数据集的预测值（Out-of-Fold 预测）。\n    kappa_scores = [] \n    oof_score_predictions = np.zeros(len(data))  # 存储 Out-of-Fold 预测值\n\n    score_to_index_thresholds = base_thresholds  \n    thresholds = []\n    for fold_idx, (train_idx, val_idx) in enumerate(cv.split(data, data[index_col])):\n        X_train, X_val = data[features].iloc[train_idx], data[features].iloc[val_idx]\n        y_train_score = data[score_col].iloc[train_idx] \n        y_train_index = data[index_col].iloc[train_idx]\n        y_val_score = data[score_col].iloc[val_idx]      \n        y_val_index = data[index_col].iloc[val_idx]     \n        \n         # 训练模型\n        if sample_weights:\n            weights = calculate_weights(y_train_score)\n            model_.fit(X_train, y_train_score, sample_weight=weights)\n        else:\n            model_.fit(X_train, y_train_score)\n\n        y_pred_train_score = model_.predict(X_train)\n        y_pred_val_score = model_.predict(X_val)\n        \n        oof_score_predictions[val_idx] = y_pred_val_score \n\n         # 优化阈值\n        t_1 = optimize_thresholds(y_train_index, y_pred_train_score, start_vals=base_thresholds)\n        thresholds.append(t_1) # 存储当前折的最优阈值\n\n        y_pred_val_index = round_with_thresholds(y_pred_val_score, t_1)\n\n        kappa_score = cohen_kappa_score(y_val_index, y_pred_val_index, weights='quadratic')\n        kappa_scores.append(kappa_score)\n        \n        if verbose:\n            print(f\"Fold {fold_idx}: Optimized Kappa Score = {kappa_score}\")\n    \n    if verbose:\n        print(f\"## Mean CV Kappa Score: {np.mean(kappa_scores)} ##\")\n        print(f\"## Std CV: {np.std(kappa_scores)}\")\n    \n    return np.mean(kappa_scores), oof_score_predictions, thresholds\n\ndef n_cross_validate(model_, data, features, score_col, index_col, cv, seeds, sample_weights=False, verbose=False):\n    \n    # 进行多次交叉验证，每次使用不同的随机种子，以提高模型评估的稳定性。\n    scores = []  # 存储每次交叉验证的 Kappa 分数\n    \n    for seed in seeds:\n        cv.random_state = seed  # 设置交叉验证的随机种子\n        score, oof, _ = cross_validate(model_, data, features, score_col, index_col, cv, sample_weights=True, verbose=False)\n        scores.append(score)\n    \n    return score, oof","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:16.090524Z","iopub.execute_input":"2025-06-15T08:33:16.091168Z","iopub.status.idle":"2025-06-15T08:33:16.115324Z","shell.execute_reply.started":"2025-06-15T08:33:16.091148Z","shell.execute_reply":"2025-06-15T08:33:16.114571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def cross_validate_tabnet(model, X, y_cont, y_disc, cv, max_epochs=100, patience=10, \n                          batch_size=1024, virtual_batch_size=128,verbose=True):\n   \n    kappa_list = []\n    oof_preds = np.zeros(len(y_cont))\n    thresholds = []\n\n    for fold_idx, (train_idx, val_idx) in enumerate(cv.split(X, y_disc)):\n        X_tr, X_val = X[train_idx], X[val_idx]\n        y_tr_cont, y_val_cont = y_cont[train_idx], y_cont[val_idx]\n        y_tr_disc, y_val_disc = y_disc[train_idx], y_disc[val_idx]\n        \n        # TabNet 需要目标为二维数组 (n_samples, 1)\n        y_tr_cont = y_tr_cont.reshape(-1, 1)\n        y_val_cont = y_val_cont.reshape(-1, 1)\n        \n        # 在本折上训练模型\n        model.fit(\n            X_tr, y_tr_cont,\n            eval_set=[(X_val, y_val_cont)],\n            max_epochs=max_epochs,\n            patience=patience,\n            batch_size=batch_size,\n            virtual_batch_size=virtual_batch_size,\n        )\n        \n        # 在训练集上预测连续输出，并优化阈值\n        preds_train = model.predict(X_tr)\n        best_thr = optimize_thresholds(y_tr_disc, preds_train, start_vals=base_thresholds)\n        thresholds.append(best_thr)\n        \n        # 在验证集上预测连续输出，经过阈值映射得到离散预测\n        preds_val = model.predict(X_val)\n        preds_val_disc = round_with_thresholds(preds_val, best_thr)\n        \n        kappa = cohen_kappa_score(y_val_disc, preds_val_disc, weights=\"quadratic\")\n        kappa_list.append(kappa)\n        \n        oof_preds[val_idx] = preds_val.flatten()\n        \n        if verbose:\n            print(f\"Fold {fold_idx}: Optimized Kappa Score = {kappa:.6f}\")\n            \n    mean_kappa = np.mean(kappa_list)\n    if verbose:\n        print(f\"## Mean CV Kappa Score for TabNet: {mean_kappa:.6f} ##\")\n    return mean_kappa, oof_preds, thresholds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:16.116253Z","iopub.execute_input":"2025-06-15T08:33:16.116558Z","iopub.status.idle":"2025-06-15T08:33:16.137112Z","shell.execute_reply.started":"2025-06-15T08:33:16.116516Z","shell.execute_reply":"2025-06-15T08:33:16.136402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def objective(trial, model_type, X, features, score_col, index_col, cv, sample_weights=False):\n    # xgboost\n    if model_type == 'xgboost':\n        params = {\n            'objective': trial.suggest_categorical('objective', ['reg:tweedie', 'reg:pseudohubererror']),\n            'random_state': SEED,\n            'num_parallel_tree': trial.suggest_int('num_parallel_tree', 2, 30),\n            'n_estimators': trial.suggest_int('n_estimators', 100, 300),\n            'max_depth': trial.suggest_int('max_depth', 2, 4),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 0.02, 0.05),\n            'subsample': trial.suggest_float('subsample', 0.5, 0.8),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 0.8),\n            'reg_alpha': trial.suggest_loguniform('reg_alpha', 1e-5, 1e-1),\n            'reg_lambda': trial.suggest_loguniform('reg_lambda', 1e-5, 1e-1),\n        }\n        if params['objective'] == 'reg:tweedie':\n            params['tweedie_variance_power'] = trial.suggest_float('tweedie_variance_power', 1, 2)\n        model = XGBRegressor(**params, use_label_encoder=False)\n    \n    # lightgbm\n    elif model_type == 'lightgbm':\n        params = {\n            'objective': trial.suggest_categorical('objective', ['poisson', 'tweedie', 'regression']),\n            'random_state': SEED,\n            'verbosity': -1,\n            'n_estimators': trial.suggest_int('n_estimators', 100, 300),\n            'max_depth': trial.suggest_int('max_depth', 2, 4),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.05),\n            'subsample': trial.suggest_float('subsample', 0.5, 0.8),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 0.8),\n            'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 20, 100)\n        }\n        if params['objective'] == 'tweedie':\n            params['tweedie_variance_power'] = trial.suggest_float('tweedie_variance_power', 1, 2)\n        model = LGBMRegressor(**params)\n    \n    # catboost\n    elif model_type == 'catboost':\n        params = {\n            'loss_function': trial.suggest_categorical('objective', ['Tweedie:variance_power=1.5', \n                                                                     'Poisson', 'RMSE']),\n            'random_state': SEED,\n            'iterations': trial.suggest_int('iterations', 100, 300),\n            'depth': trial.suggest_int('depth', 2, 4),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.05),\n            'l2_leaf_reg': trial.suggest_loguniform('l2_leaf_reg', 1e-3, 1e-1),\n            'subsample': trial.suggest_float('subsample', 0.5, 0.7),\n            'bagging_temperature': trial.suggest_float('bagging_temperature', 0.0, 1.0),\n            'random_strength': trial.suggest_float('random_strength', 1e-3, 10.0),\n            'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 20, 60),\n        }\n        model = CatBoostRegressor(**params, verbose=0)\n    \n    else:\n        raise ValueError(f\"Unsupported model_type: {model_type}\")\n        \n    seeds = [random.randint(1, 10000) for _ in range(20)] # Seeds for repeated KFold\n\n    score, _ = n_cross_validate(model, X, features, score_col, index_col, cv, seeds, sample_weights=True, verbose=True)\n\n    return score\n\ndef run_optimization(X, features, score_col, index_col, model_type, n_trials=30, cv=None, sample_weights=False):\n    study = optuna.create_study(direction=\"maximize\")\n    study.optimize(lambda trial: objective(trial, model_type, X, features, score_col, index_col, cv, sample_weights), \n                   n_trials=n_trials)\n    \n    print(f\"Best params for {model_type}: {study.best_params}\")\n    print(f\"Best score: {study.best_value}\")\n    return study.best_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:16.137883Z","iopub.execute_input":"2025-06-15T08:33:16.138066Z","iopub.status.idle":"2025-06-15T08:33:16.154145Z","shell.execute_reply.started":"2025-06-15T08:33:16.138051Z","shell.execute_reply":"2025-06-15T08:33:16.153580Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom pytorch_tabnet.callbacks import Callback\nimport os\nimport torch\nfrom pytorch_tabnet.callbacks import Callback\n# 目标函数\ndef objective_tabnet_kappa(trial):\n    # 超参数建议\n    n_d = trial.suggest_int(\"n_d\", 8, 32, step=8)\n    n_a = n_d  # 通常 n_a 设置为和 n_d 相同\n    n_steps = trial.suggest_int(\"n_steps\", 3, 10)\n    gamma = trial.suggest_float(\"gamma\", 1.0, 2.0)\n    lambda_sparse = trial.suggest_loguniform(\"lambda_sparse\", 1e-6, 1e-3)\n    lr = trial.suggest_loguniform(\"lr\", 1e-3, 1e-1)\n    beta1 = trial.suggest_float(\"beta1\", 0.5, 0.9)\n    \n    # 构建 TabNetRegressor 模型\n    model = TabNetRegressor(\n        n_d=n_d,\n        n_a=n_a,\n        n_steps=n_steps,\n        gamma=gamma,\n        lambda_sparse=lambda_sparse,\n        optimizer_params=dict(lr=lr, betas=(beta1, 0.999), weight_decay=1e-5),\n        mask_type=entmax,\n        scheduler_params= dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n        scheduler_fn=torch.optim.lr_scheduler.ReduceLROnPlateau,\n        verbose=0,\n        seed=SEED,\n    )\n    \n    # 使用 StratifiedKFold 进行 10 折交叉验证\n    cv = StratifiedKFold(n_splits=10, shuffle=True, random_state=SEED)\n    mean_kappa, _, _ = cross_validate_tabnet(model, X_tab, y_cont, y_disc, cv, \n                                               max_epochs=100, patience=10,\n                                               batch_size=1024, virtual_batch_size=128,\n                                               verbose=False)\n    \n    return mean_kappa","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:16.154934Z","iopub.execute_input":"2025-06-15T08:33:16.155231Z","iopub.status.idle":"2025-06-15T08:33:16.188775Z","shell.execute_reply.started":"2025-06-15T08:33:16.155203Z","shell.execute_reply":"2025-06-15T08:33:16.187992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 选择需要排除的特征列表\nexclude = [\n    \"PC_9\", \"PC_12\", \"Fitness_Endurance-Max_Stage\", \"Basic_Demos-Sex\", \"BMI_mean_norm\", \"PC_11\", \n    \"PC_8\", \"FGC_Zones_min\", \"Physical-Systolic_BP\", \"PC_4\", \"BIA-BIA_FMI\", \"BIA-BIA_LST\", \"Physical-Diastolic_BP\", \n    \"BIA-BIA_ECW\", \"Fitness_Endurance-Time_Mins\", \"PAQ_C-PAQ_C_Total\", \"PC_10\", \"BIA-BIA_Fat\", \"FFM_norm\", \"PC_14\", \"PC_7\"\n]\n\n# 从原始特征列表中移除排除的特征\nreduced_features = [f for f in features if f not in exclude]\n\n# 打印最终选定的特征数量\nprint(len(reduced_features))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:16.189451Z","iopub.execute_input":"2025-06-15T08:33:16.189681Z","iopub.status.idle":"2025-06-15T08:33:16.194821Z","shell.execute_reply.started":"2025-06-15T08:33:16.189664Z","shell.execute_reply":"2025-06-15T08:33:16.194249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:16.195509Z","iopub.execute_input":"2025-06-15T08:33:16.195714Z","iopub.status.idle":"2025-06-15T08:33:16.214676Z","shell.execute_reply.started":"2025-06-15T08:33:16.195700Z","shell.execute_reply":"2025-06-15T08:33:16.214010Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 由下面超参数搜索得到的超参数\nlgb_params = {\n    'objective': 'poisson', \n    'n_estimators': 295, \n    'max_depth': 4, \n    'learning_rate': 0.04505693066482616, \n    'subsample': 0.6042489155604022, \n    'colsample_bytree': 0.5021876720502726, \n    'min_data_in_leaf': 100\n}\n\nxgb_params = {\n    'objective': 'reg:tweedie', \n    'num_parallel_tree': 12, \n    'n_estimators': 236, \n    'max_depth': 3, \n    'learning_rate': 0.04223740904479563, \n    'subsample': 0.7157264603586825, \n    'colsample_bytree': 0.7897918901977528, \n    'reg_alpha': 0.005335705058190553, \n    'reg_lambda': 0.0001897435318347022, \n    'tweedie_variance_power': 1.1393958601390142\n}\n\ncat_params = {\n    'objective': 'RMSE', \n    'iterations': 238, \n    'depth': 4, \n    'learning_rate': 0.044523361750173816, \n    'l2_leaf_reg': 0.09301285673435761, \n    'subsample': 0.6902492783438681, \n    'bagging_temperature': 0.3007304771330199, \n    'random_strength': 3.562201626987314, \n    'min_data_in_leaf': 60\n}\n\nxtrees_params = {\n    'n_estimators': 500, \n    'max_depth': 15, \n    'min_samples_leaf': 20, \n    'bootstrap': False\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:16.215494Z","iopub.execute_input":"2025-06-15T08:33:16.215755Z","iopub.status.idle":"2025-06-15T08:33:16.229179Z","shell.execute_reply.started":"2025-06-15T08:33:16.215731Z","shell.execute_reply":"2025-06-15T08:33:16.228581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if optimize_params:\n    # LightGBM Optimization\n    lgb_params = run_optimization(train, lgb_features, 'PCIAT-PCIAT_Total', 'sii', 'lightgbm', n_trials=n_trials, cv=kf, sample_weights=True)\n\n    # XGBoost Optimization\n    xgb_params = run_optimization(train, xgb_features, 'PCIAT-PCIAT_Total', 'sii', 'xgboost', n_trials=n_trials, cv=kf, sample_weights=True)\n\n    # CatBoost Optimization\n    cat_params = run_optimization(train, cat_features, 'PCIAT-PCIAT_Total', 'sii', 'catboost', n_trials=n_trials, cv=kf, sample_weights=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:16.229900Z","iopub.execute_input":"2025-06-15T08:33:16.230616Z","iopub.status.idle":"2025-06-15T08:33:16.244924Z","shell.execute_reply.started":"2025-06-15T08:33:16.230594Z","shell.execute_reply":"2025-06-15T08:33:16.244218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 构建模型\nlgb_model = LGBMRegressor(**lgb_params, random_state=SEED, verbosity=-1)\nxgb_model = XGBRegressor(**xgb_params, random_state=SEED, verbosity=0)\ncat_model = CatBoostRegressor(**cat_params, random_state=SEED, verbose=0)\nxtrees_model = ExtraTreesRegressor(**xtrees_params, random_state=SEED)\n\nweights = calculate_weights(train['PCIAT-PCIAT_Total'])\n# ------------------------------\n# 交叉验证评估\n# 用 cross_validate 来得到连续预测与最佳阈值\n\n# lightgbm\nscore_lgb, oof_lgb, lgb_thresholds = cross_validate(\n    lgb_model, train, reduced_features, 'PCIAT-PCIAT_Total', 'sii', kf, verbose=True, sample_weights=True\n)\n\n# xgboost\nscore_xgb, oof_xgb, xgb_thresholds = cross_validate(\n    xgb_model, train, reduced_features, 'PCIAT-PCIAT_Total', 'sii', kf, verbose=True, sample_weights=True\n)\n\n# catboost\nscore_cat, oof_cat, cat_thresholds = cross_validate(\n    cat_model, train, reduced_features, 'PCIAT-PCIAT_Total', 'sii', kf, verbose=True, sample_weights=True\n)\n\n# ExtraTree\nscore_xtrees, oof_xtrees, xtrees_thresholds = cross_validate(\n    xtrees_model, train, reduced_features, 'PCIAT-PCIAT_Total', 'sii', kf, verbose=True, sample_weights=True\n)\n\n# 所有模型的平均Kappa得分\nprint(f'Overall Mean Kappa: {np.mean([score_lgb, score_xgb, score_cat, score_xtrees])}') # Ensemble score likely higher","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:33:16.245662Z","iopub.execute_input":"2025-06-15T08:33:16.245863Z","iopub.status.idle":"2025-06-15T08:34:28.569348Z","shell.execute_reply.started":"2025-06-15T08:33:16.245848Z","shell.execute_reply":"2025-06-15T08:34:28.568573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 由下面的超参数搜索得到\nTabNet_Params = {'n_d': 8,\n 'n_steps': 7,\n 'gamma': 1.199858033402975,\n 'lambda_sparse': 1.3135195995947674e-0,\n 'optimizer_params':dict(lr=0.004546098269759013, betas=(0.8631657893341091,0.999), weight_decay=1e-5)          \n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:35:02.016834Z","iopub.execute_input":"2025-06-15T08:35:02.017443Z","iopub.status.idle":"2025-06-15T08:35:02.021304Z","shell.execute_reply.started":"2025-06-15T08:35:02.017421Z","shell.execute_reply":"2025-06-15T08:35:02.020491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 构造用于 TabNet 的输入数据\nX_tab = train[reduced_features].values            # 特征矩阵\ny_cont = train[y_model].values            # 连续目标（回归值）\ny_disc = train[y_comp].values.astype(int)   # 离散目标，用于阈值优化和分层\n\nbest_params = TabNet_Params\nif optimize_params:\n    # 使用 Optuna 调优，目标方向为 maximize Kappa 分数\n    study = optuna.create_study(direction=\"maximize\")\n    study.optimize(objective_tabnet_kappa, n_trials=30)\n    \n    print(\"Best TabNet Kappa:\", study.best_value)\n    print(\"Best hyperparameters:\", study.best_params)\n    # 利用最佳超参数构建最终模型并进行交叉验证\n    best_params = study.best_params\n\nfinal_tabnet_model = TabNetRegressor(**best_params,verbose=0,seed=SEED,)\n\n# 这里用10折交叉验证保存 OOF 预测及各折阈值以供后续绘图\ncv_full = StratifiedKFold(n_splits=10, shuffle=True, random_state=SEED)\ntabnet_kappa_full, oof_tabnet, tabnet_thresholds = cross_validate_tabnet(final_tabnet_model, X_tab, y_cont, y_disc, cv_full, verbose=True)\nprint(\"Final Mean CV Kappa for TabNet:\", tabnet_kappa_full)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:35:09.191060Z","iopub.execute_input":"2025-06-15T08:35:09.191595Z","iopub.status.idle":"2025-06-15T08:37:47.584579Z","shell.execute_reply.started":"2025-06-15T08:35:09.191549Z","shell.execute_reply":"2025-06-15T08:37:47.583710Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_thresholds_ens = np.mean(np.array(lgb_thresholds), axis=0)\nxgb_thresholds_ens = np.mean(np.array(xgb_thresholds), axis=0)\ncat_thresholds_ens = np.mean(np.array(cat_thresholds), axis=0)\nxtrees_thresholds_ens = np.mean(np.array(xtrees_thresholds), axis=0)\ntabnet_thresholds_ens = np.mean(np.array(tabnet_thresholds), axis=0)\n\n# Apply the optimized thresholds to OOF predictions\noof_lgb_t = round_with_thresholds(oof_lgb, lgb_thresholds_ens)\nprint(f\"LGBM optimized Kappa: {cohen_kappa_score(train['sii'], oof_lgb_t, weights='quadratic')}\")\noof_xgb_t = round_with_thresholds(oof_xgb, xgb_thresholds_ens)\nprint(f\"XGB optimized Kappa: {cohen_kappa_score(train['sii'], oof_xgb_t, weights='quadratic')}\")\noof_cat_t = round_with_thresholds(oof_cat, cat_thresholds_ens)\nprint(f\"CAT optimized Kappa: {cohen_kappa_score(train['sii'], oof_cat_t, weights='quadratic')}\")\noof_xtrees_t = round_with_thresholds(oof_xtrees, xtrees_thresholds_ens)\nprint(f\"ExtraTrees optimized Kappa: {cohen_kappa_score(train['sii'], oof_xtrees_t, weights='quadratic')}\")\noof_tabnet_t = round_with_thresholds(oof_tabnet, tabnet_thresholds_ens)\nprint(f\"TabNet optimized Kappa: {cohen_kappa_score(train['sii'], oof_tabnet_t, weights='quadratic')}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T08:42:54.562614Z","iopub.execute_input":"2025-06-15T08:42:54.562928Z","iopub.status.idle":"2025-06-15T08:42:54.589893Z","shell.execute_reply.started":"2025-06-15T08:42:54.562909Z","shell.execute_reply":"2025-06-15T08:42:54.589276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"oof_preds1 = np.array([oof_lgb, oof_xgb, oof_cat])\nweighted_oof1 = np.average(oof_preds1, axis=0, weights= [0.3,0.3,0.2])\nfinal_oof1 = np.round(weighted_oof1).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T09:10:12.192052Z","iopub.execute_input":"2025-06-15T09:10:12.192556Z","iopub.status.idle":"2025-06-15T09:10:12.197209Z","shell.execute_reply.started":"2025-06-15T09:10:12.192517Z","shell.execute_reply":"2025-06-15T09:10:12.196569Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 离散型预测值的散点图\nsns.set_theme(style=\"white\")\nfig, axes = plt.subplots(1, 4, figsize=(16, 12))\n# LightGBM\nscatter1 = axes[0].scatter(train['PCIAT-PCIAT_Total'], oof_lgb, c=train[\"sii\"], cmap=\"autumn\", alpha=0.5)\naxes[0].set_xlabel(\"True Score\")\naxes[0].set_ylabel(\"OOF Predictions - LGBM\")\naxes[0].set_ylim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[0].set_xlim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[0].set_aspect('equal', adjustable='box')\n\nthresholds = [30, 50, 80]\nfor threshold in thresholds:\n    axes[0].axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    axes[0].axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n# XGBoost\nscatter2 = axes[1].scatter(train['PCIAT-PCIAT_Total'], oof_xgb, c=train[\"sii\"], cmap=\"autumn\", alpha=0.5)\naxes[1].set_xlabel(\"True Score\")\naxes[1].set_ylabel(\"OOF Predictions - XGB\")\naxes[1].set_ylim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[1].set_xlim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[1].set_aspect('equal', adjustable='box')\n\nfor threshold in thresholds:\n    axes[1].axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    axes[1].axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n\n# CatBoost    \nscatter2 = axes[2].scatter(train['PCIAT-PCIAT_Total'], oof_cat, c=train[\"sii\"], cmap=\"autumn\", alpha=0.5)\naxes[2].set_xlabel(\"True Score\")\naxes[2].set_ylabel(\"OOF Predictions - Cat\")\naxes[2].set_ylim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[2].set_xlim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[2].set_aspect('equal', adjustable='box')\n\nfor threshold in thresholds:\n    axes[2].axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    axes[2].axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n\n# 集成模型1\n# CatBoost    \nscatter3 = axes[3].scatter(train['PCIAT-PCIAT_Total'], final_oof1, c=train[\"sii\"], cmap=\"autumn\", alpha=0.5)\naxes[3].set_xlabel(\"True Score\")\naxes[3].set_ylabel(\"OOF Predictions - Ensemble-1\")\naxes[3].set_ylim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[3].set_xlim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[3].set_aspect('equal', adjustable='box')\n\nfor threshold in thresholds:\n    axes[3].axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    axes[3].axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T09:10:13.590075Z","iopub.execute_input":"2025-06-15T09:10:13.590798Z","iopub.status.idle":"2025-06-15T09:10:14.557656Z","shell.execute_reply.started":"2025-06-15T09:10:13.590778Z","shell.execute_reply":"2025-06-15T09:10:14.556938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 集成模型1 ： lgb, xgb, cat\noof_preds1 = np.array([oof_lgb_t, oof_xgb_t, oof_cat_t])\nweighted_oof1 = np.average(oof_preds1, axis=0, weights= [0.3,0.3,0.2])\nfinal_oof1 = np.round(weighted_oof1).astype(int)\nkappa_score1 = cohen_kappa_score(train[\"sii\"], final_oof1, weights='quadratic')\nprint(f\"Ensemble Kappa score: {kappa_score1}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T09:29:20.342985Z","iopub.execute_input":"2025-06-15T09:29:20.343448Z","iopub.status.idle":"2025-06-15T09:29:20.352728Z","shell.execute_reply.started":"2025-06-15T09:29:20.343426Z","shell.execute_reply":"2025-06-15T09:29:20.351910Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 定义 FGC 类相关变量\nFGC_columns = ['FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_TL',\n    'FGC-FGC_GSND', 'FGC-FGC_GSD', 'FGC-FGC_SRL', 'FGC-FGC_SRR']\n\n# 获取所有年龄组\nage_groups = train_t['Age Group'].unique()\n\n# 创建 1×3 的画布，设置共享 y 轴\nfig, axes = plt.subplots(1, 3, figsize=(18, 6), sharey=True)\n\n# 遍历每个年龄组，绘制相关性热力图\nfor i, age_group in enumerate(age_groups):\n    group_data = train_t[train_t['Age Group'] == age_group]  # 筛选当前年龄组的数据\n    corr_matrix = group_data[FGC_columns + ['PCIAT-PCIAT_Total', 'Basic_Demos-Age']].corr()  # 计算相关矩阵\n    \n    # 绘制相关性热力图\n    sns.heatmap(corr_matrix, annot=True, cmap='coolwarm', fmt='.1f',\n                vmin=-1, vmax=1, ax=axes[i], cbar=i == 0)  \n    \n    axes[i].set_title(f'{age_group} 组的相关性热力图') \n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"oof_list = [oof_lgb_t, oof_xgb_t, oof_cat_t, final_oof1]\nmodelname = ['lgb', 'xgb', 'cat', 'ensemble-1']\n# 分类型预测值热力图\nfig, axes = plt.subplots(1, 4, figsize=(18, 8), sharey=True)\nfor i in range(4):\n    conf_matrix = confusion_matrix(train[\"sii\"], oof_list[i])\n    sns.set_theme(style=\"whitegrid\")\n    sns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"autumn\", cbar=False, linewidths=0.5, linecolor='black', ax=axes[i])\n    axes[i].set_title('Confusion Matrix-{}'.format(modelname[i]), fontsize=16)\n    axes[i].set_xlabel('Predicted', fontsize=12)\n    axes[i].set_ylabel('True', fontsize=12)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T09:30:25.035660Z","iopub.execute_input":"2025-06-15T09:30:25.036382Z","iopub.status.idle":"2025-06-15T09:30:26.064095Z","shell.execute_reply.started":"2025-06-15T09:30:25.036357Z","shell.execute_reply":"2025-06-15T09:30:26.063290Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"oof_preds2 = np.array([oof_lgb, oof_xgb, oof_cat, oof_xtrees])\nweighted_oof2 = np.average(oof_preds2, axis=0, weights= [0.3,0.3,0.2,0.2])\nfinal_oof2 = np.round(weighted_oof2).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T09:00:10.753810Z","iopub.execute_input":"2025-06-15T09:00:10.754638Z","iopub.status.idle":"2025-06-15T09:00:10.759292Z","shell.execute_reply.started":"2025-06-15T09:00:10.754606Z","shell.execute_reply":"2025-06-15T09:00:10.758489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 离散型预测值的散点图\nsns.set_theme(style=\"white\")\nfig, axes = plt.subplots(1, 2, figsize=(8, 6))\n# 极端随机树   \nscatter1 = axes[0].scatter(train['PCIAT-PCIAT_Total'], oof_xtrees, c=train[\"sii\"], cmap=\"autumn\", alpha=0.5)\naxes[0].set_xlabel(\"True Score\")\naxes[0].set_ylabel(\"OOF Predictions - ExtraTrees\")\naxes[0].set_ylim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[0].set_xlim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[0].set_aspect('equal', adjustable='box')\n\nfor threshold in thresholds:\n    axes[0].axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    axes[0].axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    \n# 集成模型2 \nscatter2 = axes[1].scatter(train['PCIAT-PCIAT_Total'], final_oof2, c=train[\"sii\"], cmap=\"autumn\", alpha=0.5)\naxes[1].set_xlabel(\"True Score\")\naxes[1].set_ylabel(\"OOF Predictions - Ensemble-2\")\naxes[1].set_ylim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[1].set_xlim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[1].set_aspect('equal', adjustable='box')\n\nfor threshold in thresholds:\n    axes[1].axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    axes[1].axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T09:00:11.908436Z","iopub.execute_input":"2025-06-15T09:00:11.908931Z","iopub.status.idle":"2025-06-15T09:00:12.415212Z","shell.execute_reply.started":"2025-06-15T09:00:11.908902Z","shell.execute_reply":"2025-06-15T09:00:12.414402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 集成模型2 ： lgb, xgb, cat, extratrees \noof_preds2 = np.array([oof_lgb_t, oof_xgb_t, oof_cat_t, oof_xtrees_t])\nweighted_oof2 = np.average(oof_preds2, axis=0, weights= [0.3,0.3,0.2,0.2])\nfinal_oof2 = np.round(weighted_oof2).astype(int)\nkappa_score2 = cohen_kappa_score(train[\"sii\"], final_oof2, weights='quadratic')\nprint(f\"Ensemble Kappa score: {kappa_score2}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T09:47:37.951704Z","iopub.execute_input":"2025-06-15T09:47:37.952335Z","iopub.status.idle":"2025-06-15T09:47:37.961316Z","shell.execute_reply.started":"2025-06-15T09:47:37.952315Z","shell.execute_reply":"2025-06-15T09:47:37.960706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"oof_list = [oof_xtrees_t, final_oof2]\nmodelname = ['ExtraTrees', 'ensemble-2']\n# 分类型预测值热力图\nfig, axes = plt.subplots(1, 2, figsize=(16, 8), sharey=True)\nfor i in range(2):\n    conf_matrix = confusion_matrix(train[\"sii\"], oof_list[i])\n    sns.set_theme(style=\"whitegrid\")\n    sns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"autumn\", cbar=False, linewidths=0.5, linecolor='black', ax=axes[i])\n    axes[i].set_title('Confusion Matrix-{}'.format(modelname[i]), fontsize=16)\n    axes[i].set_xlabel('Predicted', fontsize=12)\n    axes[i].set_ylabel('True', fontsize=12)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T09:55:02.484967Z","iopub.execute_input":"2025-06-15T09:55:02.485252Z","iopub.status.idle":"2025-06-15T09:55:02.930995Z","shell.execute_reply.started":"2025-06-15T09:55:02.485233Z","shell.execute_reply":"2025-06-15T09:55:02.930309Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 集成模型3 ： lgb, xgb, cat, tabnet\noof_preds3 = np.array([oof_lgb, oof_xgb, oof_cat, oof_tabnet])\nweighted_oof3 = np.average(oof_preds3, axis=0, weights= [0.3,0.3,0.2,0.1])\nfinal_oof3 = np.round(weighted_oof3).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T09:48:08.784210Z","iopub.execute_input":"2025-06-15T09:48:08.784894Z","iopub.status.idle":"2025-06-15T09:48:08.790463Z","shell.execute_reply.started":"2025-06-15T09:48:08.784866Z","shell.execute_reply":"2025-06-15T09:48:08.789461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 离散型预测值的散点图\nsns.set_theme(style=\"white\")\nfig, axes = plt.subplots(1, 2, figsize=(8, 6))\n\n# TabNet\nscatter4 =axes[0].scatter(train['PCIAT-PCIAT_Total'], oof_tabnet, c=train[\"sii\"], cmap=\"autumn\", alpha=0.5)\naxes[0].set_xlabel(\"True Score\")\naxes[0].set_ylabel(\"OOF Predictions - TabNet\")\naxes[0].set_ylim(0, np.max(train['PCIAT-PCIAT_Total']))\naxes[0].set_xlim(0, np.max(train['PCIAT-PCIAT_Total']))\naxes[0].set_aspect('equal', adjustable='box')\nfor thr in thresholds:\n    axes[0].axhline(thr, color=\"blue\", linestyle=\"--\", lw=1)\n    axes[0].axvline(thr, color=\"blue\", linestyle=\"--\", lw=1)\n    \n# 集成模型2 \nscatter2 = axes[1].scatter(train['PCIAT-PCIAT_Total'], final_oof3, c=train[\"sii\"], cmap=\"autumn\", alpha=0.5)\naxes[1].set_xlabel(\"True Score\")\naxes[1].set_ylabel(\"OOF Predictions - Ensemble-3\")\naxes[1].set_ylim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[1].set_xlim(0,np.max(train['PCIAT-PCIAT_Total']))\naxes[1].set_aspect('equal', adjustable='box')\n\nfor threshold in thresholds:\n    axes[1].axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    axes[1].axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T09:48:13.315932Z","iopub.execute_input":"2025-06-15T09:48:13.316657Z","iopub.status.idle":"2025-06-15T09:48:13.800142Z","shell.execute_reply.started":"2025-06-15T09:48:13.316628Z","shell.execute_reply":"2025-06-15T09:48:13.799318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 集成模型3 ： lgb, xgb, cat, tabnet\noof_preds3 = np.array([oof_lgb_t, oof_xgb_t, oof_cat_t, oof_tabnet_t])\nweighted_oof3 = np.average(oof_preds3, axis=0, weights= [0.3,0.3,0.2,0.1])\nfinal_oof3 = np.round(weighted_oof3).astype(int)\nkappa_score3 = cohen_kappa_score(train[\"sii\"], final_oof3, weights='quadratic')\nprint(f\"Ensemble Kappa score: {kappa_score3}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T09:48:19.214193Z","iopub.execute_input":"2025-06-15T09:48:19.214944Z","iopub.status.idle":"2025-06-15T09:48:19.223527Z","shell.execute_reply.started":"2025-06-15T09:48:19.214919Z","shell.execute_reply":"2025-06-15T09:48:19.222951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"oof_list = [oof_tabnet_t, final_oof3]\nmodelname = ['TabNet', 'ensemble-3']\n# 分类型预测值热力图\nfig, axes = plt.subplots(1, 2, figsize=(16, 8), sharey=True)\nfor i in range(2):\n    conf_matrix = confusion_matrix(train[\"sii\"], oof_list[i])\n    sns.set_theme(style=\"whitegrid\")\n    sns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"autumn\", cbar=False, linewidths=0.5, linecolor='black', ax=axes[i])\n    axes[i].set_title('Confusion Matrix-{}'.format(modelname[i]), fontsize=16)\n    axes[i].set_xlabel('Predicted', fontsize=12)\n    axes[i].set_ylabel('True', fontsize=12)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T10:10:40.261049Z","iopub.execute_input":"2025-06-15T10:10:40.261641Z","iopub.status.idle":"2025-06-15T10:10:40.677966Z","shell.execute_reply.started":"2025-06-15T10:10:40.261620Z","shell.execute_reply":"2025-06-15T10:10:40.677124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 集成模型3各模型的相关度\nplt.figure(figsize=(8, 6))  \nmodel_preds = pd.DataFrame({\n    'lgb': oof_lgb,\n    'xgb': oof_xgb,\n    'cat': oof_cat,\n    'tabnet': oof_tabnet_t\n})\ncorr_df = model_preds.corr()\nsns.heatmap(corr_df, annot=True, cmap=\"autumn\", cbar=False, linewidths=0.5, linecolor='black', )\nplt.title(\"Correlation Between Models\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-15T10:23:14.867406Z","iopub.execute_input":"2025-06-15T10:23:14.868089Z","iopub.status.idle":"2025-06-15T10:23:15.054667Z","shell.execute_reply.started":"2025-06-15T10:23:14.868058Z","shell.execute_reply":"2025-06-15T10:23:15.053890Z"}},"outputs":[],"execution_count":null}]}