{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Initialization","metadata":{}},{"cell_type":"markdown","source":"## a. Import libraries","metadata":{}},{"cell_type":"code","source":"# Import libraries\nimport os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.impute import KNNImputer\nfrom IPython.display import display\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:59:00.381343Z","iopub.execute_input":"2025-03-29T04:59:00.381718Z","iopub.status.idle":"2025-03-29T04:59:03.740259Z","shell.execute_reply.started":"2025-03-29T04:59:00.381689Z","shell.execute_reply":"2025-03-29T04:59:03.739319Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## b. Check if input dataset is loaded from Kaggle\n* If this cell has no output, means input data is not loaded.\n* Solution: Unload and reload input dataset from sidebar\n","metadata":{}},{"cell_type":"markdown","source":"## c. Import datasets","metadata":{"execution":{"iopub.status.busy":"2025-02-24T16:56:22.484808Z","iopub.execute_input":"2025-02-24T16:56:22.485191Z","iopub.status.idle":"2025-02-24T16:56:22.489862Z","shell.execute_reply.started":"2025-02-24T16:56:22.485161Z","shell.execute_reply":"2025-02-24T16:56:22.488509Z"}}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample_submission = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:59:18.101975Z","iopub.execute_input":"2025-03-29T04:59:18.102535Z","iopub.status.idle":"2025-03-29T04:59:18.197032Z","shell.execute_reply.started":"2025-03-29T04:59:18.102452Z","shell.execute_reply":"2025-03-29T04:59:18.196023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_features = [\n    'Basic_Demos-Age', 'CGAS-CGAS_Score', 'Physical-BMI', 'Physical-Weight', \n    'Physical-Waist_Circumference', 'Physical-Diastolic_BP', \n    'Physical-HeartRate', 'Physical-Systolic_BP','Physical-Height',\n    'Fitness_Endurance-Max_Stage','Fitness_Endurance-Time_Mins',\n    'Fitness_Endurance-Time_Sec','FGC-FGC_CU','FGC-FGC_GSND',\n    'FGC-FGC_GSD','FGC-FGC_PU','FGC-FGC_SRL','FGC-FGC_SRR',\n    'FGC-FGC_TL','BIA-BIA_BMC','BIA-BIA_BMI','BIA-BIA_BMR',\n    'BIA-BIA_DEE','BIA-BIA_ECW','BIA-BIA_FFM','BIA-BIA_FFMI',\n    'BIA-BIA_FMI','BIA-BIA_Fat','PAQ_A-PAQ_A_Total','BIA-BIA_ICW',\n    'BIA-BIA_LDM','BIA-BIA_LST','BIA-BIA_SMM','BIA-BIA_TBW',\n    'PAQ_C-PAQ_C_Total','PCIAT-PCIAT_01','PCIAT-PCIAT_02',\n    'PCIAT-PCIAT_03','PCIAT-PCIAT_04','PCIAT-PCIAT_05','PCIAT-PCIAT_06',\n    'PCIAT-PCIAT_07','PCIAT-PCIAT_08','PCIAT-PCIAT_09','PCIAT-PCIAT_10',\n    'PCIAT-PCIAT_11','PCIAT-PCIAT_12','PCIAT-PCIAT_13','PCIAT-PCIAT_14',\n    'PCIAT-PCIAT_15','PCIAT-PCIAT_16','PCIAT-PCIAT_17','PCIAT-PCIAT_18',\n    'PCIAT-PCIAT_19','PCIAT-PCIAT_20','PCIAT-PCIAT_Total','SDS-SDS_Total_Raw',\n    'SDS-SDS_Total_T','PreInt_EduHx-computerinternet_hoursday'\n]\ncategorical_features = [\n    'Basic_Demos-Enroll_Season', 'Basic_Demos-Sex', 'CGAS-Season', \n    'Physical-Season', 'Fitness_Endurance-Season', 'FGC-Season',\n    'FGC-FGC_CU_Zone','FGC-FGC_GSND_Zone','FGC-FGC_GSD_Zone',\n    'FGC-FGC_PU_Zone','FGC-FGC_SRL_Zone','FGC-FGC_SRR_Zone',\n    'FGC-FGC_TL_Zone','BIA-Season','BIA-BIA_Activity_Level_num',\n    'BIA-BIA_Frame_num','PAQ_A-Season','PAQ_C-Season','PCIAT-Season',\n    'SDS-Season','PreInt_EduHx-Season'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:59:20.054133Z","iopub.execute_input":"2025-03-29T04:59:20.054572Z","iopub.status.idle":"2025-03-29T04:59:20.061067Z","shell.execute_reply.started":"2025-03-29T04:59:20.054537Z","shell.execute_reply":"2025-03-29T04:59:20.059945Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Data Visualisation","metadata":{}},{"cell_type":"markdown","source":"## 2a. Tabular data","metadata":{}},{"cell_type":"code","source":"sii_counts = train['sii'].value_counts().sort_index()\n\n# Plot pie chart\nplt.figure(figsize=(6, 6))\nplt.pie(sii_counts, labels=sii_counts.index, autopct='%1.1f%%', startangle=90, counterclock=False)\nplt.title('sii Class Distribution')\nplt.axis('equal')  # Make pie chart a circle\nplt.tight_layout()\nplt.show()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:59:22.831034Z","iopub.execute_input":"2025-03-29T04:59:22.831368Z","iopub.status.idle":"2025-03-29T04:59:23.105367Z","shell.execute_reply.started":"2025-03-29T04:59:22.831341Z","shell.execute_reply":"2025-03-29T04:59:23.104214Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As the sii classes are imbalanced, we will apply techniques to mitigate the issue.","metadata":{}},{"cell_type":"markdown","source":"Add more later!","metadata":{}},{"cell_type":"markdown","source":"## 2b. Actigraphy data","metadata":{}},{"cell_type":"code","source":"# Path to one Parquet file\nfile_path = \"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=00115b9f/part-0.parquet\"\n\n# Read the Parquet file\ndf = pd.read_parquet(file_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:59:27.051769Z","iopub.execute_input":"2025-03-29T04:59:27.052149Z","iopub.status.idle":"2025-03-29T04:59:27.359052Z","shell.execute_reply.started":"2025-03-29T04:59:27.052120Z","shell.execute_reply":"2025-03-29T04:59:27.358034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Show basic info\nprint(df.info())  # Check column types","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T03:31:21.802909Z","iopub.execute_input":"2025-03-29T03:31:21.803233Z","iopub.status.idle":"2025-03-29T03:31:21.825398Z","shell.execute_reply.started":"2025-03-29T03:31:21.803205Z","shell.execute_reply":"2025-03-29T03:31:21.824273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot histograms for all numeric columns\ndf.hist(figsize=(12, 8), bins=30, edgecolor='black')\nplt.suptitle(\"Distribution of Numeric Features\", fontsize=14)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T03:31:49.730337Z","iopub.execute_input":"2025-03-29T03:31:49.730705Z","iopub.status.idle":"2025-03-29T03:31:51.792497Z","shell.execute_reply.started":"2025-03-29T03:31:49.730676Z","shell.execute_reply":"2025-03-29T03:31:51.791460Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* **relative_date_PCIAT**: # days since PCIAT test was administered (negative data mean data collected before test was administered)\n* **enmo**: motion in terms of x, y, and z axis. 0 means period of no motion.\n* **non-wear_flag:** 0: watch is being worn, 1: watch is not worn\n* **light**: ambient light in > lux\n\nWe chose to use **enmo_mean** as it describes the average amount of motion by the wearer, **non-wear_flag_sum** as it shows us how many times the activity tracker was not worn, and **light_mean** as it indicates the average light exposure the wearer is exposed to.\n","metadata":{}},{"cell_type":"code","source":"# Aggregate features by relative_date_PCIAT\ndf_agg = df.groupby(\"relative_date_PCIAT\").agg({\n    \"enmo\": [\"mean\"],\n    \"light\": [\"mean\"],\n    \"non-wear_flag\": [\"sum\"], \n})\n\n\n# Flatten column names\ndf_agg.columns = ['_'.join(col).strip() for col in df_agg.columns.values]\n\n# Reset index if needed\ndf_agg = df_agg.reset_index()\n\n# View aggregated dataset\ndf_agg.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:59:34.460012Z","iopub.execute_input":"2025-03-29T04:59:34.460348Z","iopub.status.idle":"2025-03-29T04:59:34.503557Z","shell.execute_reply.started":"2025-03-29T04:59:34.460321Z","shell.execute_reply":"2025-03-29T04:59:34.502530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\nplt.plot(df.iloc[:, 0], df.iloc[:, 1], marker='o')  # Adjust columns as needed\nplt.xlabel(\"X-Axis Feature (Time or Index)\")\nplt.ylabel(\"Y-Axis Feature\")\nplt.title(\"Line Plot of Sequential Data\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:33:27.106576Z","iopub.execute_input":"2025-03-29T04:33:27.106959Z","iopub.status.idle":"2025-03-29T04:33:27.494115Z","shell.execute_reply.started":"2025-03-29T04:33:27.106911Z","shell.execute_reply":"2025-03-29T04:33:27.492980Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Encoding actigraphy data","metadata":{}},{"cell_type":"code","source":"from concurrent.futures import ThreadPoolExecutor\nfrom sklearn.neural_network import MLPRegressor\nfrom sklearn.preprocessing import StandardScaler\nimport numpy as np\nimport pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:59:38.311705Z","iopub.execute_input":"2025-03-29T04:59:38.312044Z","iopub.status.idle":"2025-03-29T04:59:38.336520Z","shell.execute_reply.started":"2025-03-29T04:59:38.312017Z","shell.execute_reply":"2025-03-29T04:59:38.335502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def time_features(df):\n    # Convert time_of_day to hours\n    df[\"hours\"] = df[\"time_of_day\"] // (3_600 * 1_000_000_000)\n    # Basic features\n    features = [\n        df[\"non-wear_flag\"].mean(),\n        df[\"enmo\"][df[\"enmo\"] >= 0.05].sum(),\n    ]\n\n    # Define conditions for night, day, and no mask (full data)\n    night = ((df[\"hours\"] >= 22) | (df[\"hours\"] <= 5))\n    day = ((df[\"hours\"] <= 20) & (df[\"hours\"] >= 7))\n    no_mask = np.ones(len(df), dtype=bool)\n\n    # List of columns of interest and masks\n    keys = [\"enmo\", \"anglez\", \"light\", \"battery_voltage\"]\n    masks = [no_mask, night, day]\n\n    # Helper function for feature extraction\n    def extract_stats(data):\n        return [\n            data.mean(),\n            data.std(),\n            data.max(),\n            data.min(),\n            data.diff().mean(),\n            data.diff().std()\n        ]\n\n    # Iterate over keys and masks to generate the statistics\n    for key in keys:\n        for mask in masks:\n            filtered_data = df.loc[mask, key]\n            features.extend(extract_stats(filtered_data))\n\n    return features\n\ndef build_feature_matrix_modified_v2(root):\n    def process_time_features(folder):\n        path = os.path.join(root, folder, 'part-0.parquet')\n        try:\n            df = pd.read_parquet(path).drop(columns='step', errors='ignore')\n            features = time_features(df)\n            subject_id = folder.split('=')[-1]\n            return features, subject_id\n        except FileNotFoundError:\n            print(f\"File not found: {path}\")\n            return None, folder.split('=')[-1]\n\n    dirs = os.listdir(root)\n    with ThreadPoolExecutor() as ex:\n        stats_and_ids = list(tqdm(ex.map(process_time_features, dirs), total=len(dirs)))\n\n    features_list = []\n    ids = []\n    for result in stats_and_ids:\n        if result[0] is not None:\n            features_list.append(result[0])\n            ids.append(result[1])\n\n    if not features_list:\n        return pd.DataFrame()\n\n    df = pd.DataFrame(features_list, columns=[f'time_feat_{i}' for i in range(len(features_list[0]))])\n    df['id'] = ids\n    return df\n\n# 示例调用 (假设你的数据根目录是 '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet' 或类似的)\ntrain_series_raw = build_feature_matrix_modified_v2(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_series_raw = build_feature_matrix_modified_v2(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\nprint(\"Processed train data shape:\", train_series_raw.shape)\nprint(\"Processed test data shape:\", test_series_raw.shape)\nprint(\"First few rows of processed train data:\")\nprint(train_series_raw.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:07:33.279688Z","iopub.execute_input":"2025-03-29T05:07:33.280022Z","iopub.status.idle":"2025-03-29T05:08:37.865071Z","shell.execute_reply.started":"2025-03-29T05:07:33.279997Z","shell.execute_reply":"2025-03-29T05:08:37.864104Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"index 0 -- non-wear_flag.mean(), index 1 -- sum(enmo >= 0.05), index 2–7 -- stats for enmo (no mask), index 8–13 -- stats for enmo (night), 14–19 -- stats for enmo (day) etc\n\n","metadata":{}},{"cell_type":"code","source":"train_ids = train_series_raw['id'].copy()\ntest_ids = test_series_raw['id'].copy()\n\ntrain_pca = train_series_raw.drop(columns='id')\ntest_pca = test_series_raw.drop(columns='id')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:08:43.677984Z","iopub.execute_input":"2025-03-29T05:08:43.678361Z","iopub.status.idle":"2025-03-29T05:08:43.686040Z","shell.execute_reply.started":"2025-03-29T05:08:43.678324Z","shell.execute_reply":"2025-03-29T05:08:43.684859Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_pca=train_series_raw.copy()\n# test_pca=train_series_raw.copy()\n# train_pca=train_pca.drop(columns='id')\n# test_pca=test_pca.drop(columns='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T03:31:22.894905Z","iopub.status.idle":"2025-03-29T03:31:22.895307Z","shell.execute_reply":"2025-03-29T03:31:22.895127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from sklearn.impute import KNNImputer\n\n# train_ids = train_series_raw['id'].copy()\n# test_ids = test_series_raw['id'].copy()\n\n# train_pca = train_series_raw.drop(columns='id')\n# test_pca = test_series_raw.drop(columns='id')\n\n# # Initialize the KNN Imputer\n# imputer = KNNImputer(n_neighbors=5)  # You can tune n_neighbors as needed\n\n# # Fit on train and transform both train and test\n# train_knn_array = imputer.fit_transform(train_pca)\n# test_knn_array = imputer.transform(test_pca)  # Important: Use only .transform() here\n\n# # Convert arrays back to DataFrames\n# train_knn_imputed = pd.DataFrame(train_knn_array, columns=train_pca.columns)\n# test_knn_imputed = pd.DataFrame(test_knn_array, columns=test_pca.columns)\n\n# # Reattach 'id' columns\n# train_knn_imputed['id'] = train_ids\n# test_knn_imputed['id'] = test_ids\n\n# print(\" First few rows of train set after KNN imputation:\")\n# print(train_knn_imputed.head())\n\n# print(\"\\n Missing values in train set:\")\n# print(train_knn_imputed.isna().sum().sum())\n\n# print(\"\\n\" + \"=\"*40 + \"\\n\")\n\n# print(\" First few rows of test set after KNN imputation:\")\n# print(test_knn_imputed.head())\n\n# print(\"\\n Missing values in test set:\")\n# print(test_knn_imputed.isna().sum().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T03:31:22.896371Z","iopub.status.idle":"2025-03-29T03:31:22.896704Z","shell.execute_reply":"2025-03-29T03:31:22.896580Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nfrom sklearn.linear_model import BayesianRidge\nimport pandas as pd\n\nimputer = IterativeImputer(estimator=BayesianRidge(), max_iter=10, random_state=0)\n\n# Fit and transform train, transform test\ntrain_pca_imputed_array = imputer.fit_transform(train_pca)\ntest_pca_imputed_array = imputer.transform(test_pca)\n\n# Convert back to DataFrame\ntrain_pca_imputed = pd.DataFrame(train_pca_imputed_array, columns=train_pca.columns)\ntest_pca_imputed = pd.DataFrame(test_pca_imputed_array, columns=test_pca.columns)\n\nprint(\"Train (MICE) Imputed Sample:\")\nprint(train_pca_imputed.head())\n\nprint(\"\\n Missing values in Train (MICE):\")\nprint(train_pca_imputed.isna().sum().sum())\n\nprint(\"\\n Test (MICE) Imputed Sample:\")\nprint(test_pca_imputed.head())\n\nprint(\"\\n Missing values in Test (MICE):\")\nprint(test_pca_imputed.isna().sum().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:08:46.321890Z","iopub.execute_input":"2025-03-29T05:08:46.322219Z","iopub.status.idle":"2025-03-29T05:08:47.712499Z","shell.execute_reply.started":"2025-03-29T05:08:46.322193Z","shell.execute_reply":"2025-03-29T05:08:47.711588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_pca=train_pca_imputed\ntest_pca=test_pca_imputed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:08:52.662804Z","iopub.execute_input":"2025-03-29T05:08:52.663138Z","iopub.status.idle":"2025-03-29T05:08:52.667314Z","shell.execute_reply.started":"2025-03-29T05:08:52.663112Z","shell.execute_reply":"2025-03-29T05:08:52.666333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.decomposition import PCA\n\n# 定义要测试的保留主成分数量列表\nn_components_to_test = [15]  # 你可以根据需要修改这些值\nexplained_variance_ratios = {}\n\nfor n_components in n_components_to_test:\n    # 初始化 PCA 模型\n    pca = PCA(n_components=n_components, random_state=42)\n\n    # 在训练集上拟合 PCA 模型\n    pca.fit(train_pca)\n\n    # 获取解释的方差比例\n    explained_variance_ratio = np.sum(pca.explained_variance_ratio_)\n    explained_variance_ratios[n_components] = explained_variance_ratio\n\n    print(f\"保留 {n_components} 个主成分时，解释的方差比例为: {explained_variance_ratio:.4f}\")\n\n# 找出保留信息最多的主成分数量\nbest_n_components = max(explained_variance_ratios, key=explained_variance_ratios.get)\nbest_explained_variance = explained_variance_ratios[best_n_components]\n\nprint(f\"\\n保留信息最多的主成分数量是 {best_n_components}，解释的方差比例为: {best_explained_variance:.4f}\")\n\n# 你可以选择使用 best_n_components 来进行最终的降维\nbest_pca = PCA(n_components=best_n_components, random_state=42)\nprincipal_components_train_best = best_pca.fit_transform(train_pca)\ncolumn_names_train_best = [f'PC_{i+1}' for i in range(best_n_components)]\npca_df_train_best = pd.DataFrame(data=principal_components_train_best, columns=column_names_train_best)\n\n# 同样的方法应用于测试集 (使用在训练集上学习到的 best_pca)\nprincipal_components_test = best_pca.transform(test_pca)\ncolumn_names_test = [f'PC_{i+1}' for i in range(best_n_components)]\npca_df_test = pd.DataFrame(data=principal_components_test, columns=column_names_test)\n\nprint(\"\\n使用最佳主成分数量压缩后的训练集维度:\", pca_df_train_best.shape)\nprint(\"使用最佳主成分数量压缩后的测试集维度:\", pca_df_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:11:49.821988Z","iopub.execute_input":"2025-03-29T05:11:49.822467Z","iopub.status.idle":"2025-03-29T05:11:49.870735Z","shell.execute_reply.started":"2025-03-29T05:11:49.822418Z","shell.execute_reply":"2025-03-29T05:11:49.869543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pca_df_train_best['id']=train_series_raw['id']\npca_df_test['id']=test_series_raw['id']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:11:52.122434Z","iopub.execute_input":"2025-03-29T05:11:52.122874Z","iopub.status.idle":"2025-03-29T05:11:52.128825Z","shell.execute_reply.started":"2025-03-29T05:11:52.122838Z","shell.execute_reply":"2025-03-29T05:11:52.127751Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# merge (combing tabular with time data)\ntrain_df = pd.merge(train, pca_df_train_best, on='id', how='left')\ntest_df = pd.merge(test, pca_df_test, on='id', how='left')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:11:53.229335Z","iopub.execute_input":"2025-03-29T05:11:53.229755Z","iopub.status.idle":"2025-03-29T05:11:53.243524Z","shell.execute_reply.started":"2025-03-29T05:11:53.229722Z","shell.execute_reply":"2025-03-29T05:11:53.242337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:11:55.773735Z","iopub.execute_input":"2025-03-29T05:11:55.774073Z","iopub.status.idle":"2025-03-29T05:11:55.804566Z","shell.execute_reply.started":"2025-03-29T05:11:55.774047Z","shell.execute_reply":"2025-03-29T05:11:55.803379Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Fill Missing Values","metadata":{}},{"cell_type":"markdown","source":"## a. Some preprocessing","metadata":{}},{"cell_type":"code","source":"# checkpoint: able to rerun from here onwards\ntrain_unfilled = train_df.copy()\ntest_unfilled = test_df.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:11:58.472304Z","iopub.execute_input":"2025-03-29T05:11:58.472709Z","iopub.status.idle":"2025-03-29T05:11:58.480564Z","shell.execute_reply.started":"2025-03-29T05:11:58.472677Z","shell.execute_reply":"2025-03-29T05:11:58.479467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop rows with empty sii\ntrain_unfilled = train_unfilled.dropna(subset='sii')\n\n# Replace 0s with NaN in 'Physical-Weight'\ntrain_unfilled.loc[train_unfilled[\"Physical-Weight\"] == 0, \"Physical-Weight\"] = np.nan\ntest_unfilled.loc[test_unfilled[\"Physical-Weight\"] == 0, \"Physical-Weight\"] = np.nan","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:11:59.571674Z","iopub.execute_input":"2025-03-29T05:11:59.572077Z","iopub.status.idle":"2025-03-29T05:11:59.582292Z","shell.execute_reply.started":"2025-03-29T05:11:59.572036Z","shell.execute_reply":"2025-03-29T05:11:59.581275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def check_missing_values(df):\n    # Check for missing (NaN) values\n    print(\"Missing values in each column:\\n\", df.isna().sum())\n    \n    # Check for infinite values\n    print(\"\\nInfinite values in each column:\\n\", (df == np.inf).sum() + (df == -np.inf).sum())\n    \n    # # Display dataset information\n    # df.info()\n\n    # Calculate the percentage of missing values per column\n    missing_percentage = (df.isna().sum() / len(df)) * 100\n    \n    # Sort the columns by missing percentage in descending order\n    missing_percentage = missing_percentage.sort_values(ascending=False)\n    \n    # Plot the missing values as a horizontal bar chart (axes inverted)\n    plt.figure(figsize=(8, 15))\n    bars = missing_percentage.plot(kind='barh', color='skyblue', edgecolor='black')\n    \n    # Increase bar thickness\n    for bar in bars.patches:\n        bar.set_height(0.6)  # Adjust the thickness of bars\n    \n    # Formatting the plot\n    plt.ylabel(\"Columns\", fontsize=10)\n    plt.xlabel(\"Percentage of Missing Values (%)\", fontsize=12)\n    plt.title(\"Percentage of Missing Values Per Column (Sorted)\", fontsize=14)\n    plt.grid(axis='x', linestyle='--', alpha=0.7)\n    plt.yticks(fontsize=8)  # Reduce the size of y-axis labels\n    \n    # Show the plot\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:12:01.173787Z","iopub.execute_input":"2025-03-29T05:12:01.174126Z","iopub.status.idle":"2025-03-29T05:12:01.181280Z","shell.execute_reply.started":"2025-03-29T05:12:01.174098Z","shell.execute_reply":"2025-03-29T05:12:01.180031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"check_missing_values(train_unfilled)\ncheck_missing_values(test_unfilled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:12:03.363840Z","iopub.execute_input":"2025-03-29T05:12:03.364181Z","iopub.status.idle":"2025-03-29T05:12:05.134280Z","shell.execute_reply.started":"2025-03-29T05:12:03.364154Z","shell.execute_reply":"2025-03-29T05:12:05.133181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def clean_features(df):\n    # Remove highly implausible values\n\n    # Clip Grip\n    df[['FGC-FGC_GSND', 'FGC-FGC_GSD']] = df[['FGC-FGC_GSND', 'FGC-FGC_GSD']].clip(lower=9, upper=60)\n    # Remove implausible body-fat\n    df[\"BIA-BIA_Fat\"] = np.where(df[\"BIA-BIA_Fat\"] < 5, np.nan, df[\"BIA-BIA_Fat\"])\n    df[\"BIA-BIA_Fat\"] = np.where(df[\"BIA-BIA_Fat\"] > 60, np.nan, df[\"BIA-BIA_Fat\"])\n    # Basal Metabolic Rate\n    df[\"BIA-BIA_BMR\"] = np.where(df[\"BIA-BIA_BMR\"] > 4000, np.nan, df[\"BIA-BIA_BMR\"])\n    # Daily Energy Expenditure\n    df[\"BIA-BIA_DEE\"] = np.where(df[\"BIA-BIA_DEE\"] > 8000, np.nan, df[\"BIA-BIA_DEE\"])\n    # Bone Mineral Content\n    df[\"BIA-BIA_BMC\"] = np.where(df[\"BIA-BIA_BMC\"] <= 0, np.nan, df[\"BIA-BIA_BMC\"])\n    df[\"BIA-BIA_BMC\"] = np.where(df[\"BIA-BIA_BMC\"] > 10, np.nan, df[\"BIA-BIA_BMC\"])\n    # Fat Free Mass Index\n    df[\"BIA-BIA_FFM\"] = np.where(df[\"BIA-BIA_FFM\"] <= 0, np.nan, df[\"BIA-BIA_FFM\"])\n    df[\"BIA-BIA_FFM\"] = np.where(df[\"BIA-BIA_FFM\"] > 300, np.nan, df[\"BIA-BIA_FFM\"])\n    # Fat Mass Index\n    df[\"BIA-BIA_FMI\"] = np.where(df[\"BIA-BIA_FMI\"] < 0, np.nan, df[\"BIA-BIA_FMI\"])\n    # Extra Cellular Water\n    df[\"BIA-BIA_ECW\"] = np.where(df[\"BIA-BIA_ECW\"] > 100, np.nan, df[\"BIA-BIA_ECW\"])\n    # Intra Cellular Water\n    # df[\"BIA-BIA_ICW\"] = np.where(df[\"BIA-BIA_ICW\"] > 100, np.nan, df[\"BIA-BIA_ICW\"])\n    # Lean Dry Mass\n    df[\"BIA-BIA_LDM\"] = np.where(df[\"BIA-BIA_LDM\"] > 100, np.nan, df[\"BIA-BIA_LDM\"])\n    # Lean Soft Tissue\n    df[\"BIA-BIA_LST\"] = np.where(df[\"BIA-BIA_LST\"] > 300, np.nan, df[\"BIA-BIA_LST\"])\n    # Skeletal Muscle Mass\n    df[\"BIA-BIA_SMM\"] = np.where(df[\"BIA-BIA_SMM\"] > 300, np.nan, df[\"BIA-BIA_SMM\"])\n    # Total Body Water\n    df[\"BIA-BIA_TBW\"] = np.where(df[\"BIA-BIA_TBW\"] > 300, np.nan, df[\"BIA-BIA_TBW\"])\n    \n    return df\n\ntrain_unfilled  = clean_features(train_unfilled)\ntest_unfilled = clean_features(test_unfilled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:12:08.821189Z","iopub.execute_input":"2025-03-29T05:12:08.821613Z","iopub.status.idle":"2025-03-29T05:12:08.859459Z","shell.execute_reply.started":"2025-03-29T05:12:08.821577Z","shell.execute_reply":"2025-03-29T05:12:08.858180Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_unfilled.shape)\nprint(test_unfilled.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:12:11.342802Z","iopub.execute_input":"2025-03-29T05:12:11.343188Z","iopub.status.idle":"2025-03-29T05:12:11.348599Z","shell.execute_reply.started":"2025-03-29T05:12:11.343152Z","shell.execute_reply":"2025-03-29T05:12:11.347566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"exclude = ['PCIAT-Season', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03',\n           'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07',\n           'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11',\n           'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15',\n           'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19',\n           'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total', 'sii', 'id']\n\ny_model = \"PCIAT-PCIAT_Total\" # Score, target for the model\ny_comp = \"sii\" # Index, target of the competition\nfeatures = [f for f in train_unfilled.columns if f not in exclude]\n\nnumerical_cols = train_unfilled.select_dtypes(include=['number'])\nprint(numerical_cols)\n\n# 获取数值型列的列名\nnumerical_col_names = numerical_cols.columns.tolist()\n\nprint(len(numerical_col_names))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:12:13.039278Z","iopub.execute_input":"2025-03-29T05:12:13.039824Z","iopub.status.idle":"2025-03-29T05:12:13.063740Z","shell.execute_reply.started":"2025-03-29T05:12:13.039768Z","shell.execute_reply":"2025-03-29T05:12:13.062450Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_features=list(set(numerical_col_names) & set(features))\ntrain_to_fill=train_unfilled.copy()\ntest_to_fill=test_unfilled.copy()\n\ntrain_to_fill=train_to_fill[numerical_features]\ntest_to_fill=test_to_fill[numerical_features]\nprint(train_to_fill.shape,test_to_fill.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:12:16.228547Z","iopub.execute_input":"2025-03-29T05:12:16.228937Z","iopub.status.idle":"2025-03-29T05:12:16.240304Z","shell.execute_reply.started":"2025-03-29T05:12:16.228910Z","shell.execute_reply":"2025-03-29T05:12:16.239186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LassoCV\nfrom sklearn.impute import KNNImputer\nfrom sklearn.base import clone\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\n# 假设你已经有了 train_to_fill 和 test_to_fill 这两个 DataFrame\n# 并且你已经定义了 features 列表 (包含要使用的特征列名)\n# 以及 SEED (随机种子)\n\n# 如果 features 列表还没有定义，你需要根据你的数据确定要使用的特征列\n# 示例:\n# features = [col for col in train_to_fill.columns if col != 'target_column']\n\nSEED = 42 # 或者你之前定义的 SEED 值\n\nclass Impute_With_Model:\n\n    def __init__(self, na_frac=0.2, min_samples=0):\n        self.model_dict = {}\n        self.mean_dict = {}\n        self.features = None\n        self.na_frac = na_frac\n        self.min_samples = min_samples\n\n    def find_features(self, data, feature, tmp_features):\n        missing_rows = data[feature].isna()\n        na_fraction = data[missing_rows][tmp_features].isna().mean(axis=0)\n        valid_features = np.array(tmp_features)[na_fraction <= self.na_frac]\n        return valid_features\n\n    def fit_models(self, model, data, features):\n        self.features = features\n        n_data = data.shape[0]\n        for feature in features:\n            self.mean_dict[feature] = np.mean(data[feature])\n        for feature in tqdm(features):\n            if data[feature].isna().sum() > 0:\n                model_clone = clone(model)\n                X = data[data[feature].notna()].copy()\n                tmp_features = [f for f in features if f != feature]\n                tmp_features = self.find_features(data, feature, tmp_features)\n                if len(tmp_features) >= 1 and X.shape[0] > self.min_samples:\n                    for f in tmp_features:\n                        X[f] = X[f].fillna(self.mean_dict[f])\n                    model_clone.fit(X[tmp_features], X[feature])\n                    self.model_dict[feature] = (model_clone, tmp_features.copy())\n                else:\n                    self.model_dict[feature] = (\"mean\", np.mean(data[feature]))\n\n    def impute(self, data):\n        imputed_data = data.copy()\n        for feature, model in self.model_dict.items():\n            missing_rows = imputed_data[feature].isna()\n            if missing_rows.any():\n                if model[0] == \"mean\":\n                    imputed_data[feature].fillna(model[1], inplace=True)\n                else:\n                    tmp_features = [f for f in self.features if f != feature]\n                    X_missing = data.loc[missing_rows, tmp_features].copy()\n                    for f in tmp_features:\n                        X_missing[f] = X_missing[f].fillna(self.mean_dict[f])\n                    imputed_data.loc[missing_rows, feature] = model[0].predict(X_missing[model[1]])\n        return imputed_data\n\n# 初始化模型 (使用 LassoCV 作为示例，你可以替换为你想要使用的模型)\nmodel = model = LassoCV(cv=5, random_state=SEED, max_iter=10000) # 将默认的迭代次数增加到 1000\n\n# 初始化 Impute_With_Model 类\nimputer = Impute_With_Model(na_frac=0.2) # 设置 na_frac 参数\n\n# 拟合 Imputation 模型 (仅在 train_to_fill 上进行拟合)\nimputer.fit_models(model, train_to_fill, numerical_features)\n\n# 使用拟合好的模型填充 train_to_fill 和 test_to_fill 中的缺失值\ntrain_filled = imputer.impute(train_to_fill)\ntest_filled = imputer.impute(test_to_fill)\n\n# 现在 train_filled 和 test_filled 是填充了缺失值后的 DataFrame\nprint(\"train_to_fill 填充后的形状:\", train_filled.shape)\nprint(\"test_to_fill 填充后的形状:\", test_filled.shape)\nprint(\"\\ntrain_to_fill 填充后缺失值总数:\", train_filled.isna().sum().sum())\nprint(\"test_to_fill 填充后缺失值总数:\", test_filled.isna().sum().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:12:21.432968Z","iopub.execute_input":"2025-03-29T05:12:21.433305Z","iopub.status.idle":"2025-03-29T05:12:30.479816Z","shell.execute_reply.started":"2025-03-29T05:12:21.433277Z","shell.execute_reply":"2025-03-29T05:12:30.478724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # 初始化模型 (使用 LassoCV 作为示例，你可以替换为你想要使用的模型)\n# model = model = LassoCV(cv=5, random_state=SEED, max_iter=10000) # 将默认的迭代次数增加到 1000\n\n# # 初始化 Impute_With_Model 类\n# imputer = Impute_With_Model(na_frac=0.4) # 设置 na_frac 参数\n\n# # 拟合 Imputation 模型 (仅在 train_to_fill 上进行拟合)\n# imputer.fit_models(model, train_to_fill, numerical_features)\n\n# # 使用拟合好的模型填充 train_to_fill 和 test_to_fill 中的缺失值\n# train_filled = imputer.impute(train_to_fill)\n# test_filled = imputer.impute(test_to_fill)\n\n# # 现在 train_filled 和 test_filled 是填充了缺失值后的 DataFrame\n# print(\"train_to_fill 填充后的形状:\", train_filled.shape)\n# print(\"test_to_fill 填充后的形状:\", test_filled.shape)\n# print(\"\\ntrain_to_fill 填充后缺失值总数:\", train_filled.isna().sum().sum())\n# print(\"test_to_fill 填充后缺失值总数:\", test_filled.isna().sum().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:38:14.300073Z","iopub.execute_input":"2025-03-29T04:38:14.300415Z","iopub.status.idle":"2025-03-29T04:38:14.316627Z","shell.execute_reply.started":"2025-03-29T04:38:14.300391Z","shell.execute_reply":"2025-03-29T04:38:14.315218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # adjust na_frac\n\n# base_model = LassoCV(cv=5, random_state=42, max_iter=10000)\n\n# # Try different na_frac values\n# na_frac_values = [0.2, 0.3, 0.4, 0.5, 0.6]\n# filled_results = {}\n\n# # Loop over each na_frac and record how many NaNs are filled\n# for na_frac in na_frac_values:\n#     print(f\"\\n Testing na_frac = {na_frac}\")\n    \n#     # Create a fresh imputer instance each time\n#     imputer = Impute_With_Model(na_frac=na_frac)\n    \n#     # Fit the model on the training data\n#     imputer.fit_models(clone(base_model), train_to_fill.copy(), numerical_features)\n    \n#     # Apply imputation\n#     filled = imputer.impute(train_to_fill.copy())\n    \n#     # Count remaining NaNs after filling\n#     remaining_nans = filled[numerical_features].isna().sum().sum()\n#     total_filled = train_to_fill[numerical_features].isna().sum().sum() - remaining_nans\n    \n#     filled_results[na_frac] = total_filled\n#     print(f\" NaNs filled: {total_filled}, Remaining: {remaining_nans}\")\n\n# # Show comparison\n# print(\"\\n NaNs filled per na_frac:\")\n# for frac, count in filled_results.items():\n#     print(f\"na_frac = {frac}: filled {count} values\")\n\n# # Find best na_frac\n# best_frac = max(filled_results, key=filled_results.get)\n# print(f\"\\n Best na_frac = {best_frac}, filled {filled_results[best_frac]} missing values\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T03:31:22.919001Z","iopub.status.idle":"2025-03-29T03:31:22.919420Z","shell.execute_reply":"2025-03-29T03:31:22.919209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from sklearn.ensemble import HistGradientBoostingRegressor\n# from sklearn.base import clone\n\n# hgb_model = HistGradientBoostingRegressor(random_state=42)\n\n# # Test different na_frac values if you want\n# na_frac = 0.2 \n\n# # Create new imputer instance\n# imputer = Impute_With_Model(na_frac=na_frac)\n\n# # Fit the imputer model\n# imputer.fit_models(hgb_model, train_to_fill, numerical_features)\n\n# # Apply the imputer\n# train_filled = imputer.impute(train_to_fill)\n# test_filled = imputer.impute(test_to_fill)\n\n# # Confirm result\n# print(\"Imputation complete\")\n# print(\"Remaining NaNs in train:\", train_filled.isna().sum().sum())\n# print(\"Remaining NaNs in test:\", test_filled.isna().sum().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:03:20.016079Z","iopub.execute_input":"2025-03-29T04:03:20.016425Z","iopub.status.idle":"2025-03-29T04:03:47.637875Z","shell.execute_reply.started":"2025-03-29T04:03:20.016399Z","shell.execute_reply":"2025-03-29T04:03:47.636984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_filled['id']=train_unfilled['id']\ntest_filled['id']=test_unfilled['id']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:39:32.554758Z","iopub.execute_input":"2025-03-29T04:39:32.555201Z","iopub.status.idle":"2025-03-29T04:39:32.562119Z","shell.execute_reply.started":"2025-03-29T04:39:32.555170Z","shell.execute_reply":"2025-03-29T04:39:32.561171Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## b. Fill missing values for categorical columns using Mode\nMode represents the majority category in the data, minimising distortion by maintaining the most common class in the variable.","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"categorical_features=list(set(features) - set(numerical_col_names))\nprint(train_unfilled[categorical_features].dtypes)\ntrain_unfilled_cat=train_unfilled.copy()\ntest_unfilled_cat=test_unfilled.copy()\ntrain_unfilled_cat=train_unfilled_cat[categorical_features]\ntest_unfilled_cat=test_unfilled_cat[categorical_features]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:13:06.467232Z","iopub.execute_input":"2025-03-29T05:13:06.467604Z","iopub.status.idle":"2025-03-29T05:13:06.481042Z","shell.execute_reply.started":"2025-03-29T05:13:06.467574Z","shell.execute_reply":"2025-03-29T05:13:06.479874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # 假设你已经有了 train_unfilled_cat 和 test_unfilled_cat 这两个 DataFrame\n\n# # 遍历 train_unfilled_cat 的每一列\n# for col in train_unfilled_cat.columns:\n#     # 计算训练数据该列的众数\n#     mode_value = train_unfilled_cat[col].mode()\n\n#     # 检查是否计算出众数 (mode() 返回一个 Series，可能为空)\n#     if not mode_value.empty:\n#         # 取第一个众数（即使有多个众数，也只取一个来填充）\n#         first_mode = mode_value[0]\n\n#         # 使用训练数据的众数填充 train_unfilled_cat 中该列的 NaN (修改后)\n#         train_unfilled_cat[col] = train_unfilled_cat[col].fillna(first_mode)\n\n#         # 使用训练数据的众数填充 test_unfilled_cat 中对应列的 NaN (如果该列存在) (修改后)\n#         if col in test_unfilled_cat.columns:\n#             test_unfilled_cat[col] = test_unfilled_cat[col].fillna(first_mode)\n#             print(f\"列 '{col}' 的 NaN 已使用训练数据众数 '{first_mode}' 填充。\")\n#         else:\n#             print(f\"警告: 测试集中不存在列 '{col}'。\")\n#     else:\n#         print(f\"警告: 训练集列 '{col}' 没有众数 (可能所有值都是 NaN 或唯一)。\")\n\n# # 验证填充结果 (可选)\n# print(\"\\ntrain_unfilled_cat 填充后 NaN 总数:\", train_unfilled_cat.isnull().sum().sum())\n# print(\"test_unfilled_cat 填充后 NaN 总数:\", test_unfilled_cat.isnull().sum().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:13:12.065158Z","iopub.execute_input":"2025-03-29T05:13:12.065572Z","iopub.status.idle":"2025-03-29T05:13:12.108884Z","shell.execute_reply.started":"2025-03-29T05:13:12.065536Z","shell.execute_reply":"2025-03-29T05:13:12.107612Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# from sklearn.preprocessing import LabelEncoder\n# import pandas as pd\n\n# # 假设你已经有了 train_unfilled_cat 和 test_unfilled_cat 这两个 DataFrame\n\n# # 创建 LabelEncoder 的实例\n# label_encoder = LabelEncoder()\n\n# # 处理 train_unfilled_cat\n# for col in train_unfilled_cat.columns:\n#     if train_unfilled_cat[col].dtype == 'object':\n#         # 将列中的 NaN 替换为某个占位符，例如 'missing'，以便 LabelEncoder 可以处理\n#         train_unfilled_cat[col] = train_unfilled_cat[col].fillna('missing')\n#         # 使用 LabelEncoder 对列进行编码\n#         train_unfilled_cat[col] = label_encoder.fit_transform(train_unfilled_cat[col])\n#         print(f\"训练集列 '{col}' 已转换为数值型。\")\n\n# # 处理 test_unfilled_cat\n# for col in test_unfilled_cat.columns:\n#     if col in train_unfilled_cat.columns and test_unfilled_cat[col].dtype == 'object':\n#         # 将列中的 NaN 替换为与训练集相同的占位符\n#         test_unfilled_cat[col] = test_unfilled_cat[col].fillna('missing')\n#         # 注意：这里我们使用在训练集上 fit 过的 LabelEncoder 来 transform 测试集\n#         # 这样可以保证相同的类别映射到相同的数字\n#         test_unfilled_cat[col] = label_encoder.transform(test_unfilled_cat[col])\n#         print(f\"测试集列 '{col}' 已转换为数值型。\")\n#     elif test_unfilled_cat[col].dtype == 'object':\n#         print(f\"警告: 测试集列 '{col}' 在训练集中不存在，无法进行一致的 Label Encoding。\")\n\n# # 检查转换后的数据类型 (可选)\n# print(\"\\n训练集转换后数据类型:\")\n# print(train_unfilled_cat.dtypes)\n# print(\"\\n测试集转换后数据类型:\")\n# print(test_unfilled_cat.dtypes)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:39:54.110716Z","iopub.execute_input":"2025-03-29T04:39:54.111159Z","iopub.status.idle":"2025-03-29T04:39:54.155471Z","shell.execute_reply.started":"2025-03-29T04:39:54.111125Z","shell.execute_reply":"2025-03-29T04:39:54.154498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_unfilled_cat","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T03:31:22.926762Z","iopub.status.idle":"2025-03-29T03:31:22.927200Z","shell.execute_reply":"2025-03-29T03:31:22.927060Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_merged\n# test_merged","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T03:31:22.927866Z","iopub.status.idle":"2025-03-29T03:31:22.928197Z","shell.execute_reply":"2025-03-29T03:31:22.928070Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_merged = pd.concat([train_filled, train_unfilled_cat], axis=1)\n# test_merged = pd.concat([test_filled, test_unfilled_cat], axis=1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T03:31:22.928815Z","iopub.status.idle":"2025-03-29T03:31:22.929356Z","shell.execute_reply":"2025-03-29T03:31:22.929190Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## using HistGradientBoostingRegressor","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.ensemble import HistGradientBoostingClassifier\n\n# Encode season categories\ncat_encoder = OrdinalEncoder(handle_unknown=\"use_encoded_value\", unknown_value=-1)\ntrain_encoded = train_unfilled_cat.copy()\ntest_encoded = test_unfilled_cat.copy()\n\n# Fit on train, transform both\ntrain_encoded[categorical_features] = cat_encoder.fit_transform(train_encoded[categorical_features])\ntest_encoded[categorical_features] = cat_encoder.transform(test_encoded[categorical_features])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:13:24.076082Z","iopub.execute_input":"2025-03-29T05:13:24.076439Z","iopub.status.idle":"2025-03-29T05:13:24.310138Z","shell.execute_reply.started":"2025-03-29T05:13:24.076409Z","shell.execute_reply":"2025-03-29T05:13:24.308848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Impute\ncat_imputer = Impute_With_Model(na_frac=0.3, min_samples=20)\nhgb = HistGradientBoostingClassifier(random_state=42)\n\ncat_imputer.fit_models(hgb, train_encoded, categorical_features)\ntrain_cat_filled = cat_imputer.impute(train_encoded)\ntest_cat_filled = cat_imputer.impute(test_encoded)\n\n\nprint(\"All season columns filled accurately with classification model.\")\nprint(\"Remaining NaNs in train:\", train_cat_filled.isna().sum().sum())\nprint(\"Remaining NaNs in test:\", test_cat_filled.isna().sum().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:40:16.679282Z","iopub.execute_input":"2025-03-29T04:40:16.679641Z","iopub.status.idle":"2025-03-29T04:40:16.698675Z","shell.execute_reply.started":"2025-03-29T04:40:16.679603Z","shell.execute_reply":"2025-03-29T04:40:16.697066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Re-encode categorical features\n# train_encoded = train_cat_filled.copy()\n# test_encoded = test_cat_filled.copy()\n\n# cat_encoder = OrdinalEncoder(handle_unknown='use_encoded_value', unknown_value=-1)\n# train_encoded[categorical_features] = cat_encoder.fit_transform(train_encoded[categorical_features])\n# test_encoded[categorical_features] = cat_encoder.transform(test_encoded[categorical_features])\n\n# Final merge for model input\ntrain_merged = pd.concat([train_filled, train_encoded], axis=1)\ntest_merged = pd.concat([test_filled, test_encoded], axis=1)\ntrain_merged","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:13:36.080805Z","iopub.execute_input":"2025-03-29T05:13:36.081197Z","iopub.status.idle":"2025-03-29T05:13:36.120131Z","shell.execute_reply.started":"2025-03-29T05:13:36.081163Z","shell.execute_reply":"2025-03-29T05:13:36.119065Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## c. Check filled dataset","metadata":{}},{"cell_type":"code","source":"check_missing_values(train_merged)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:14:05.538023Z","iopub.execute_input":"2025-03-29T05:14:05.538373Z","iopub.status.idle":"2025-03-29T05:14:06.489615Z","shell.execute_reply.started":"2025-03-29T05:14:05.538344Z","shell.execute_reply":"2025-03-29T05:14:06.488442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"check_missing_values(test_merged)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T03:31:22.934738Z","iopub.status.idle":"2025-03-29T03:31:22.935045Z","shell.execute_reply":"2025-03-29T03:31:22.934911Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Feature Engineering","metadata":{}},{"cell_type":"code","source":"# checkpoint: able to rerun from here onwards\ntrain_fe = train_unfilled.copy()\ntest_fe = test_unfilled.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:14:10.191358Z","iopub.execute_input":"2025-03-29T05:14:10.191805Z","iopub.status.idle":"2025-03-29T05:14:10.200312Z","shell.execute_reply.started":"2025-03-29T05:14:10.191772Z","shell.execute_reply":"2025-03-29T05:14:10.199171Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## a. Adding new features","metadata":{}},{"cell_type":"code","source":"def feature_engineering(df):\n    new_features = pd.DataFrame({\n        \"BP_HeartRate\": df[\"Physical-Systolic_BP\"] * df[\"Physical-HeartRate\"],  # BP & Heart Rate Interaction\n        \"BFP_BMI\": df[\"BIA-BIA_Fat\"] / df[\"BIA-BIA_BMI\"],  # Body Fat Percentage to BMI\n        \"LST_TBW\": df[\"BIA-BIA_LST\"] / df[\"BIA-BIA_BMI\"], # Lean Mass to Total Body Water\n        \"DEE_Weight\": df[\"BIA-BIA_DEE\"] / df[\"Physical-Weight\"],  # Daily Energy Expenditure per Weight\n        \"Fitness_Index\": (df[\"Fitness_Endurance-Max_Stage\"] + df[\"Fitness_Endurance-Time_Mins\"]) / 2,  # Fitness Performance\n        \"Hydration_Status\": df[\"BIA-BIA_TBW\"] / df[\"Physical-Weight\"],  # Hydration Relative to Weight\n        \"Internet_to_Activity_Ratio\": df[\"PreInt_EduHx-computerinternet_hoursday\"] / (df[\"PAQ_C-PAQ_C_Total\"] + 1),  # Internet vs Activity\n        \"Screen_Sleep_Impact\": df[\"PreInt_EduHx-computerinternet_hoursday\"] / (df[\"SDS-SDS_Total_T\"] + 1),  # Screen Time vs Sleep\n        \"BP_Variability\": df[\"Physical-Systolic_BP\"] - df[\"Physical-Diastolic_BP\"],  # BP Variability\n        \"HeartRate_BMI\": df[\"Physical-HeartRate\"] / (df[\"Physical-BMI\"] + 1),  # Heart Rate to BMI\n        \"BP_Weight_Ratio\": (df[\"Physical-Systolic_BP\"] + df[\"Physical-Diastolic_BP\"]) / df[\"Physical-Weight\"],  # BP to Weight Ratio\n        \"BMR_Age_Adjusted\": df[\"BIA-BIA_BMR\"] / (df[\"Basic_Demos-Age\"] + 1),  # BMR Efficiency Adjusted for Age\n        \"Muscle_to_Fat_Index\": df[\"BIA-BIA_FFMI\"] / (df[\"BIA-BIA_FMI\"] + 1),  # FFMI to FMI Ratio\n        \"Water_to_Muscle\": df[\"BIA-BIA_TBW\"] / (df[\"BIA-BIA_SMM\"] + 1),  # Water-to-Muscle Ratio\n        \"Hydration_Index\": df[\"BIA-BIA_ICW\"] / (df[\"BIA-BIA_ECW\"] + 1),  # Intracellular vs Extracellular Water\n        \"Winter_Enroll\": (df[\"Basic_Demos-Enroll_Season\"] == \"Winter\").astype(int),  # Winter Enrollment Indicator\n        \"Seasonal_Physical_Activity\": df[\"PAQ_C-PAQ_C_Total\"] * (df[\"Basic_Demos-Enroll_Season\"] == \"Winter\").astype(int)  # Seasonal Activity Impact\n    })\n\n    # Concatenate the new features in a single operation\n    df = pd.concat([df, new_features], axis=1)\n\n    # Add new feature names to the numerical_features list only if they don't already exist\n    global numerical_features\n    for feature in new_features.columns:\n        if feature not in numerical_features:\n            numerical_features.append(feature)\n\n    return df\n\n# Apply feature engineering\ntrain_fe = feature_engineering(train_merged)\ntest_fe = feature_engineering(test_merged)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:14:14.894236Z","iopub.execute_input":"2025-03-29T05:14:14.894737Z","iopub.status.idle":"2025-03-29T05:14:14.924273Z","shell.execute_reply.started":"2025-03-29T05:14:14.894693Z","shell.execute_reply":"2025-03-29T05:14:14.922690Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## b. Check that all engineered features are valid (no inf or NaN)","metadata":{}},{"cell_type":"code","source":"def print_first_10_rows_with_missing_or_inf(df):\n    \"\"\"\n    Prints only the first 10 rows where any column contains NaN or Inf values.\n    Ensures all columns are displayed in a properly formatted table.\n    \n    Parameters:\n    - df (pd.DataFrame): DataFrame to check for NaN or Inf values.\n    \"\"\"\n    # Enable full display of all columns\n    pd.set_option('display.max_columns', None)  # Show all columns\n    pd.set_option('display.width', 1000)  # Prevent truncation\n    pd.set_option('display.float_format', '{:.3f}'.format)  # Format float values nicely\n\n    # Filter rows where any column has NaN or Inf values\n    rows_with_issues = df[df.isna().any(axis=1) | df.isin([np.inf, -np.inf]).any(axis=1)]\n\n    # Check if there are any problematic rows\n    if rows_with_issues.empty:\n        print(\"✅ No NaN or Inf values found in the DataFrame.\")\n        return\n\n    # Limit to first 10 rows\n    print(f\"⚠️ Found {len(rows_with_issues)} rows with NaN or Inf values (showing first 10):\")\n    display(rows_with_issues.head(10))  # Show only first 10 rows","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:14:17.593806Z","iopub.execute_input":"2025-03-29T05:14:17.594165Z","iopub.status.idle":"2025-03-29T05:14:17.600796Z","shell.execute_reply.started":"2025-03-29T05:14:17.594137Z","shell.execute_reply":"2025-03-29T05:14:17.599628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print_first_10_rows_with_missing_or_inf(train_fe)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:14:20.384602Z","iopub.execute_input":"2025-03-29T05:14:20.385001Z","iopub.status.idle":"2025-03-29T05:14:20.420919Z","shell.execute_reply.started":"2025-03-29T05:14:20.384967Z","shell.execute_reply":"2025-03-29T05:14:20.419578Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print_first_10_rows_with_missing_or_inf(test_fe)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:07:30.234432Z","iopub.execute_input":"2025-03-29T04:07:30.234791Z","iopub.status.idle":"2025-03-29T04:07:30.243242Z","shell.execute_reply.started":"2025-03-29T04:07:30.234762Z","shell.execute_reply":"2025-03-29T04:07:30.241605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_fe.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T03:31:23.179443Z","iopub.status.idle":"2025-03-29T03:31:23.180081Z","shell.execute_reply":"2025-03-29T03:31:23.179801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test_fe.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T03:31:23.181097Z","iopub.status.idle":"2025-03-29T03:31:23.181570Z","shell.execute_reply":"2025-03-29T03:31:23.181368Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Model Training (lgbm)\nGoal: classify test scores for ids","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\nimport pandas as pd\n\n# 假设你已经有了 train_fe 和 test_fe 这两个 DataFrame (预测特征)\n# 并且你已经有了 train_unfilled DataFrame，其中包含目标变量 \"PCIAT-PCIAT_Total\"\n\n# 确保目标变量在 train_unfilled 中存在\nif \"PCIAT-PCIAT_Total\" not in train_unfilled.columns:\n    raise ValueError(\"目标变量 'PCIAT-PCIAT_Total' 不存在于 train_unfilled DataFrame 中。\")\n\n# 定义特征和目标变量\nX_train = train_fe.drop(columns='id')\ny_train = train_unfilled[\"PCIAT-PCIAT_Total\"]\nX_test = test_fe.drop(columns='id')\n\n# 初始化 LightGBM 回归模型\nlgbm = lgb.LGBMRegressor(random_state=42) # 你可以根据需要调整模型的参数\n\n# 训练模型，并添加评估\nlgbm.fit(X_train, y_train,\n         eval_set=[(X_train, y_train)],  # 使用训练集作为评估集\n         eval_metric='rmse',             # 使用均方根误差作为评估指标 (你可以选择其他回归指标)\n         )                     # 设置为大于 0 可以显示评估结果\n\n# 在测试集上进行预测\npredictions = lgbm.predict(X_test)\n\n# 将预测结果转换为 DataFrame\npredictions_df = pd.DataFrame({'PCIAT-PCIAT_Total_Predicted': predictions})\n\n# 打印预测结果 DataFrame\nprint(\"\\n测试集 'PCIAT-PCIAT_Total' 的预测结果:\")\nprint(predictions_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:40:59.581973Z","iopub.execute_input":"2025-03-29T04:40:59.582350Z","iopub.status.idle":"2025-03-29T04:41:00.058135Z","shell.execute_reply.started":"2025-03-29T04:40:59.582324Z","shell.execute_reply":"2025-03-29T04:41:00.057105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"thresholds = [30, 50, 80]\n\ndef convert_score_to_sii(score, thresholds):\n    \"\"\"根据阈值将 PCIAT-PCIAT_Total 分数转换为 sii 值。\"\"\"\n    if score < thresholds[0]:\n        return 0\n    elif score < thresholds[1]:\n        return 1\n    elif score < thresholds[2]:\n        return 2\n    else:\n        return 3\n\n# 将预测的 PCIAT-PCIAT_Total 分数转换为 sii 值\npredictions_df['sii_Predicted'] = predictions_df['PCIAT-PCIAT_Total_Predicted'].apply(lambda x: convert_score_to_sii(x, thresholds))\n\n# 打印包含预测的 PCIAT-PCIAT_Total 和 sii 的 DataFrame\nprint(\"\\n包含预测的 PCIAT-PCIAT_Total 和 sii 的 DataFrame (前 5 行):\")\nprint(predictions_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:41:02.194712Z","iopub.execute_input":"2025-03-29T04:41:02.195127Z","iopub.status.idle":"2025-03-29T04:41:02.205150Z","shell.execute_reply.started":"2025-03-29T04:41:02.195093Z","shell.execute_reply":"2025-03-29T04:41:02.203870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def convert_to_sii_categories(scores, thresholds=[30, 50, 80]):\n    scores = np.array(scores)\n    return np.digitize(scores, thresholds)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:17:24.886156Z","iopub.execute_input":"2025-03-29T05:17:24.886685Z","iopub.status.idle":"2025-03-29T05:17:24.891519Z","shell.execute_reply.started":"2025-03-29T05:17:24.886648Z","shell.execute_reply":"2025-03-29T05:17:24.890300Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_xgb_classifier(train_X, test_X, train_y_raw, thresholds=[30, 50, 80]):\n    \"\"\"Train XGBoost with 5-fold CV and return predictions and feature importances.\"\"\"\n    y = convert_to_sii_categories(train_y_raw, thresholds)\n    X = train_X.drop(columns='id', errors='ignore')\n    test_X_ = test_X.drop(columns='id', errors='ignore')\n\n    print(f\"Training on {X.shape[0]} samples, Testing on {test_X_.shape[0]} samples.\")\n    print(\"SII category distribution:\")\n    for i, count in enumerate(np.bincount(y)):\n        print(f\"  Category {i}: {count} samples ({100 * count / len(y):.1f}%)\")\n\n    params = {\n        'objective': 'multi:softprob', 'num_class': 4, 'learning_rate': 0.02,\n        'max_depth': 6, 'min_child_weight': 3, 'subsample': 0.8, 'colsample_bytree': 0.8,\n        'gamma': 0.1, 'reg_alpha': 0.1, 'reg_lambda': 0.1, 'n_estimators': 500,\n        'random_state': 42, 'verbosity': 0\n    }\n\n    kf = KFold(n_splits=5, shuffle=True, random_state=42)\n    test_preds = np.zeros((test_X_.shape[0], 4))\n    importances = np.zeros(X.shape[1])\n    acc_scores = []\n\n    for i, (tr_idx, val_idx) in enumerate(kf.split(X)):\n        print(f\"\\nFold {i+1}\")\n        model = xgb.XGBClassifier(**params)\n        model.fit(X.iloc[tr_idx], y[tr_idx])\n\n        preds_val = model.predict(X.iloc[val_idx])\n        acc = accuracy_score(y[val_idx], preds_val)\n        acc_scores.append(acc)\n        print(f\"  Accuracy: {acc:.4f}\")\n        if i == 0:\n            print(classification_report(y[val_idx], preds_val))\n\n        test_preds += model.predict_proba(test_X_) / kf.n_splits\n        importances += model.feature_importances_\n\n    final_preds = np.argmax(test_preds, axis=1)\n    feature_df = pd.DataFrame({\n        'Feature': X.columns,\n        'Importance': importances / kf.n_splits\n    }).sort_values('Importance', ascending=False)\n\n    print(\"\\nTop features:\")\n    print(feature_df.head(10))\n\n    print(f\"\\nAverage CV Accuracy: {np.mean(acc_scores):.4f}\")\n    print(\"Test prediction distribution:\")\n    for i, c in enumerate(np.bincount(final_preds)):\n        print(f\"  Category {i}: {c} samples ({100 * c / len(final_preds):.1f}%)\")\n\n    submission = pd.DataFrame({\n        'sii_Predicted': final_preds\n    })\n    if 'id' in test_X.columns:\n        submission.insert(0, 'id', test_X['id'].values)\n    else:\n        submission.insert(0, 'id', np.arange(len(final_preds)))  # fallback\n\n    return submission, np.mean(acc_scores), feature_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:22:51.789143Z","iopub.execute_input":"2025-03-29T05:22:51.789582Z","iopub.status.idle":"2025-03-29T05:22:51.803314Z","shell.execute_reply.started":"2025-03-29T05:22:51.789544Z","shell.execute_reply":"2025-03-29T05:22:51.802115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def find_best_thresholds(train_fe, train_unfilled, num_combinations=5):\n    \"\"\"Test multiple thresholds and return the best one.\"\"\"\n    sample_idx = np.random.choice(train_fe.index, size=int(0.2 * len(train_fe)), replace=False)\n    sample_X = train_fe.loc[sample_idx]\n    sample_y = train_unfilled.loc[sample_idx][\"PCIAT-PCIAT_Total\"]\n\n    candidate_thresholds = [\n        [30, 50, 80], [35, 55, 75], [25, 45, 70],\n        [28, 48, 78], [33, 53, 83]\n    ]\n    best_acc, best_t = 0, candidate_thresholds[0]\n    for t in candidate_thresholds:\n        print(f\"\\nTrying thresholds: {t}\")\n        _, acc, _ = train_xgb_classifier(sample_X, sample_X, sample_y, thresholds=t)\n        if acc > best_acc:\n            best_acc, best_t = acc, t\n            print(f\"  New best accuracy: {acc:.4f}\")\n    print(f\"\\nBest thresholds: {best_t} with accuracy: {best_acc:.4f}\")\n    return best_t, best_acc\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:18:43.764524Z","iopub.execute_input":"2025-03-29T05:18:43.764899Z","iopub.status.idle":"2025-03-29T05:18:43.771578Z","shell.execute_reply.started":"2025-03-29T05:18:43.764869Z","shell.execute_reply":"2025-03-29T05:18:43.770292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict_sii_with_xgboost(train_fe, test_fe, train_unfilled, find_thresholds=False):\n    \"\"\"\n    Predict SII using XGBoost, optionally tuning thresholds.\n    \"\"\"\n    if \"PCIAT-PCIAT_Total\" not in train_unfilled.columns:\n        raise ValueError(\"Missing 'PCIAT-PCIAT_Total' in training target DataFrame.\")\n    \n    if find_thresholds:\n        thresholds, _ = find_best_thresholds(train_fe, train_unfilled)\n    else:\n        thresholds = [30, 50, 80]\n\n    print(f\"\\nTraining final model using thresholds: {thresholds}\")\n    submission_df, accuracy, importance_df = train_xgb_classifier(\n        train_fe, test_fe, train_unfilled[\"PCIAT-PCIAT_Total\"], thresholds\n    )\n    return submission_df, accuracy, importance_df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:18:59.344019Z","iopub.execute_input":"2025-03-29T05:18:59.344468Z","iopub.status.idle":"2025-03-29T05:18:59.351748Z","shell.execute_reply.started":"2025-03-29T05:18:59.344431Z","shell.execute_reply":"2025-03-29T05:18:59.350505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom sklearn.metrics import accuracy_score, classification_report\nimport xgboost as xgb\n\nsubmission, accuracy, importance = predict_sii_with_xgboost(\n    train_fe=train_fe,\n    test_fe=test_fe,\n    train_unfilled=train_unfilled,\n    find_thresholds=True  # or False if you want to skip threshold tuning\n)\n\nsubmission.columns = ['id', 'sii']\nprint(\"Submission columns:\", submission.columns.tolist())\n\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)\nprint(\"Submission saved to /kaggle/working/submission.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T05:22:59.497223Z","iopub.execute_input":"2025-03-29T05:22:59.497606Z","iopub.status.idle":"2025-03-29T05:25:40.988997Z","shell.execute_reply.started":"2025-03-29T05:22:59.497573Z","shell.execute_reply":"2025-03-29T05:25:40.988069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:17:31.620414Z","iopub.execute_input":"2025-03-29T04:17:31.620743Z","iopub.status.idle":"2025-03-29T04:17:31.630509Z","shell.execute_reply.started":"2025-03-29T04:17:31.620719Z","shell.execute_reply":"2025-03-29T04:17:31.629199Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. Submission","metadata":{}},{"cell_type":"markdown","source":"Submission sample code:","metadata":{}},{"cell_type":"code","source":"# my_submission = pd.DataFrame({'Id': test.Id, 'SalePrice': predicted_prices})\n# # you could use any filename. We choose submission here\n# my_submission.to_csv('submission-SVR_SukulAdisak-4features.csv', index=False)\n\n\n\nsubmission_df = pd.concat([test_unfilled['id'], predictions_df['sii_Predicted']], axis=1)\n\n# 9. Save the submission file\nsubmission_path = '/kaggle/working/submission.csv'\nsubmission_df.to_csv(submission_path, index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-29T04:41:07.383081Z","iopub.execute_input":"2025-03-29T04:41:07.383464Z","iopub.status.idle":"2025-03-29T04:41:07.392079Z","shell.execute_reply.started":"2025-03-29T04:41:07.383435Z","shell.execute_reply":"2025-03-29T04:41:07.391079Z"}},"outputs":[],"execution_count":null}]}