{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Add the following feature engineering steps:\n- recorded features by time periods\n- rank ordering features stratified by age\n- woe binning features & EDA\n","metadata":{}},{"cell_type":"markdown","source":"# FEATURE ENGINEERING(TIME SERIES)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T05:01:53.517967Z","iopub.execute_input":"2024-10-07T05:01:53.518972Z","iopub.status.idle":"2024-10-07T05:01:53.550222Z","shell.execute_reply.started":"2024-10-07T05:01:53.518910Z","shell.execute_reply":"2024-10-07T05:01:53.549012Z"}}},{"cell_type":"markdown","source":"## - import","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd\nfrom sklearn.base import clone\nfrom copy import deepcopy\nimport optuna\nfrom scipy.optimize import minimize\nimport matplotlib.pyplot as plt\nimport missingno as msno\nimport re\nimport os\nfrom colorama import Fore, Style\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\nimport lightgbm as lgb\nfrom catboost import CatBoostRegressor, CatBoostClassifier\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nimport xgboost as xgb\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator\nSEED = 42\nn_splits = 5","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-08T10:18:10.296745Z","iopub.execute_input":"2024-10-08T10:18:10.297893Z","iopub.status.idle":"2024-10-08T10:18:15.476683Z","shell.execute_reply.started":"2024-10-08T10:18:10.297837Z","shell.execute_reply":"2024-10-08T10:18:15.475223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndescriptionlist=['count','mean','std','min','25%','50%','75%','max']\nvarlist = ['X', 'Y', 'Z', 'enmo', 'anglez', 'non-wear_flag', 'light', 'battery_voltage', 'weekday', 'quarter', 'relative_date_PCIAT', 'time_of_day_hours']\ndaynightlist=['day','night']\nperiodlist=['1','2','3','4','5','6']\ncolumnslist1 = [f\"{variable}_{stat}\" for stat in descriptionlist for variable in varlist]\ncolumnslist2 = [f\"{variable}_{stat}_{day_night}\" for variable in varlist for stat in descriptionlist for day_night in daynightlist]\ncolumnslist3 = [f\"{variable}_{stat}_{period}\" for variable in varlist for stat in descriptionlist for period in periodlist]\ncolumnslist=np.concatenate((columnslist1,columnslist2,columnslist3))\n    \ndef process_file(filename, dirname):\n    \n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    \n    # convert to understandable scale\n    df['time_of_day_hours'] = (df['time_of_day'] / 1e9 / 3600)\n    df.drop('time_of_day', axis=1, inplace=True)\n    output=df.describe().unstack()\n    values=df.describe().values.reshape(-1)\n    formatted_names = [f\"{level[0]}_{level[1]}\" for level in output.index]\n\n    # 'day'：1;'night':0\n    df['day_period'] = np.where(\n        ( df['time_of_day_hours'] >= 8) &\n        ( df['time_of_day_hours'] < 21),\n        'day', 'night'\n    )\n    output1=df.groupby('day_period').describe().unstack()\n    values1=output1.values\n    formatted_names1 = [f\"{level[0]}_{level[1]}_{level[2]}\" for level in output1.index]\n    \n    # 4-hour periods\n    df.drop('day_period', axis=1, inplace=True)\n    df['hours_period'] = np.ceil(df['time_of_day_hours'] / 4)\n    output2=df.groupby('hours_period').describe().unstack()\n    values2=output2.values\n    formatted_names2 = [f\"{level[0]}_{level[1]}_{int(level[2])}\" for level in output2.index]\n\n    #combine results\n    value_combine=np.concatenate((values,values1,values2))\n    formatted_names_combine=np.concatenate((formatted_names,formatted_names1,formatted_names2))\n    \n    #mapping results\n    value_dict = dict(zip(formatted_names_combine, value_combine))\n    value_combine_mapped = [value_dict.get(name, np.nan) if name in value_dict else np.nan for name in columnslist]\n    \n    return value_combine_mapped,filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    ids = ids\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats,indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=columnslist)\n    df['id'] = indexes\n    \n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id',axis=1)\ntest = test.drop('id',axis=1)\n\ngeneral_cols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n       'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n       'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n       'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n       'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n       'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n       'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n       'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n       'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n       'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n       'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n       'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n       'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n       'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n       'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n       'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n       'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n       'PreInt_EduHx-computerinternet_hoursday']\n\ntarget_cols=['sii']\n\nfeaturesCols = general_cols+ time_series_cols+target_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season','Fitness_Endurance-Season','FGC-Season',\n 'BIA-Season','PAQ_A-Season','PAQ_C-Season','SDS-Season','PreInt_EduHx-Season']\n\ndef update(df):\n    \n    global cat_c\n    for c in cat_c : \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n        \n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:18:15.479660Z","iopub.execute_input":"2024-10-08T10:18:15.480585Z","iopub.status.idle":"2024-10-08T10:31:07.529492Z","shell.execute_reply.started":"2024-10-08T10:18:15.480516Z","shell.execute_reply":"2024-10-08T10:31:07.528163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:31:07.530786Z","iopub.execute_input":"2024-10-08T10:31:07.531229Z","iopub.status.idle":"2024-10-08T10:31:08.662694Z","shell.execute_reply.started":"2024-10-08T10:31:07.531186Z","shell.execute_reply":"2024-10-08T10:31:08.661210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"series_{train|test}.parquet/id={id} - Series to be used as training data, partitioned by id. Each series is a continuous recording of accelerometer data for a single subject spanning many days.\n\nid - The patient identifier corresponding to the id field in train/test.csv.\nstep - An integer timestep for each observation within a series.\nCorresponds to sequential data collection points.\n\nX, Y, Z - Measure of acceleration, in g, experienced by the wrist-worn watch along each standard axis.\nThe three-dimensional accelerometer readings, describe the physical movement of the person wearing the device\n\nenmo - As calculated and described by the wristpy package, ENMO is the Euclidean Norm Minus One of all accelerometer signals (along each of the x-, y-, and z-axis, measured in g-force) with negative values rounded to zero. Zero values are indicative of periods of no motion. While no standard measure of acceleration exists in this space, this is one of the several commonly computed features.\nA scalar value for the level of motion, where higher values indicate more activity, and zeros - of no motion\n\nanglez - As calculated and described by the wristpy package, Angle-Z is a metric derived from individual accelerometer components and refers to the angle of the arm relative to the horizontal plane.\nProvide information about posture or how the arm is positioned during the activity\n\nnon-wear_flag - A flag (0: watch is being worn, 1: the watch is not worn) to help determine periods when the watch has been removed, based on the GGIR definition, which uses the standard deviation and range of the accelerometer data.\nWould be great to know how it was calculated (I can see the detect_nonwear function in the wristpy package, but was it used? what were the parameters at least?)\n\nlight - Measure of ambient light in lux. See here for details.\nMay help distinguish between different times of day (day vs. night) or types of activities (e.g., being indoors or outdoors)\n\nbattery_voltage - A measure of the battery voltage in mV.\nUnlikely to be important for modeling, but may help in identifying issues related to the data collection process (low battery voltage might align with gaps, inconsistencies, or interruptions in the data)\n\ntime_of_day - Time of day representing the start of a 5s window that the data has been sampled over, with format %H:%M:%S.%9f.\nAgain as @AmbrosM have been already said, this description is not true. Time of the day was measured in nanoseconds since midnight.\n\nweekday - The day of the week, coded as an integer with 1 being Monday and 7 being Sunday.\nThis could provide insights into behavior patterns during weekdays versus weekends\n\nquarter - The quarter of the year, an integer from 1 to 4.\nCould be relevant if there are seasonal patterns in activity (e.g., more outdoor activity in summer)\n\nrelative_date_PCIAT - The number of days (integer) since the PCIAT test was administered (negative days indicate that the actigraphy data has been collected before the test was administered).","metadata":{}},{"cell_type":"code","source":"%%time\n\nX = train.drop(['sii'], axis=1)\n# 'sii_prob_sii_1','sii_prob_sii_2','sii_prob_sii_3'\ny = train['sii']\n# モデルを学習した後のコード\nXGBoost = xgb.XGBRegressor(random_state=SEED,enable_categorical=True)\nXGBoost.fit(X, y)\n\n# 特徴量重要度を取得\nimportance = XGBoost.feature_importances_\n\n# 特徴量の名前を取得\nfeatures = X.columns\n\n# データフレームとして整理\nimportance_df = pd.DataFrame({'Feature': features, 'Importance': importance})\n\n# 特徴量重要度を降順に並び替え\nimportance_df = importance_df.sort_values(by='Importance', ascending=False)\n# importance_df_plot= importance_df.head(50)\n# plt.figure(figsize=(10, 25))\n# plt.barh(importance_df_plot['Feature'], importance_df_plot['Importance'])\n# plt.xlabel('Feature Importance')\n# plt.ylabel('Features')\n# plt.title('Feature Importance in XGBoost')\n# plt.gca().invert_yaxis()  # 重要度が高いものを上に\n# plt.show()\n\nimportance_df","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:31:08.665553Z","iopub.execute_input":"2024-10-08T10:31:08.666027Z","iopub.status.idle":"2024-10-08T10:31:29.754264Z","shell.execute_reply.started":"2024-10-08T10:31:08.665979Z","shell.execute_reply":"2024-10-08T10:31:29.753021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 特徴量の名前を取得\nfeatures = X.columns\nmissing_percent_train = train.isnull().mean() * 100\nmissing_percent_test = test.isnull().mean() * 100\n# データフレームとして整理（特徴量重要度と欠損値の割合を結合）\nimportance_df = pd.DataFrame({'Feature': features, 'Importance': importance})\nmissing_df = pd.DataFrame({'Feature': missing_percent_train.index, 'MissingPercent': missing_percent_train.values})\ncombined_df = pd.merge(importance_df, missing_df, on='Feature')\n\n# 散布図を作成\nplt.figure(figsize=(10, 6))\nplt.scatter(combined_df['MissingPercent'], combined_df['Importance'], alpha=0.7)\nplt.xlabel('Missing Percentage (%)')\nplt.ylabel('Feature Importance')\nplt.title('Feature Importance vs Missing Percentage')\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:31:29.755785Z","iopub.execute_input":"2024-10-08T10:31:29.756174Z","iopub.status.idle":"2024-10-08T10:31:30.128569Z","shell.execute_reply.started":"2024-10-08T10:31:29.756133Z","shell.execute_reply":"2024-10-08T10:31:30.127237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMPOTTANCE_TH = 0\ndrop_fes = importance_df.loc[importance_df[\"Importance\"]<=IMPOTTANCE_TH,\"Feature\"].to_list()\ntrain = train.drop(columns=drop_fes)\ntest = test.drop(columns=drop_fes)\nprint(f\"---> time series data, # {len([feature for feature in time_series_cols if feature in drop_fes])}\")\nprint([feature for feature in time_series_cols if feature in drop_fes])\nprint(f\"---> general data, # {len([feature for feature in drop_fes if feature not in time_series_cols])}\")\nprint([feature for feature in drop_fes if feature not in time_series_cols])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:31:30.130405Z","iopub.execute_input":"2024-10-08T10:31:30.131084Z","iopub.status.idle":"2024-10-08T10:31:30.167298Z","shell.execute_reply.started":"2024-10-08T10:31:30.131017Z","shell.execute_reply":"2024-10-08T10:31:30.165985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"topN=100\ntop_features = importance_df.sort_values(by=\"Importance\", ascending=False).head(topN)[\"Feature\"].to_list()\nprint(f\"---> time series data, # {len([feature for feature in time_series_cols if feature in top_features])}\")\nprint([feature for feature in time_series_cols if feature in top_features])\nprint(f\"---> general data, # {len([feature for feature in top_features if feature not in time_series_cols])}\")\nprint([feature for feature in top_features if feature not in time_series_cols])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:31:30.168682Z","iopub.execute_input":"2024-10-08T10:31:30.169144Z","iopub.status.idle":"2024-10-08T10:31:30.185795Z","shell.execute_reply.started":"2024-10-08T10:31:30.169101Z","shell.execute_reply":"2024-10-08T10:31:30.184064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"topN=200\ntop_features = importance_df.sort_values(by=\"Importance\", ascending=False).head(topN)[\"Feature\"].to_list()\nprint(f\"---> time series data, # {len([feature for feature in time_series_cols if feature in top_features])}\")\nprint([feature for feature in time_series_cols if feature in top_features])\nprint(f\"---> general data, # {len([feature for feature in top_features if feature not in time_series_cols])}\")\nprint([feature for feature in top_features if feature not in time_series_cols])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:31:30.187607Z","iopub.execute_input":"2024-10-08T10:31:30.188505Z","iopub.status.idle":"2024-10-08T10:31:30.206771Z","shell.execute_reply.started":"2024-10-08T10:31:30.188457Z","shell.execute_reply":"2024-10-08T10:31:30.205572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#仅保留topN的time series feature\ntopN=50\ntop_features = importance_df.sort_values(by=\"Importance\", ascending=False).head(topN)[\"Feature\"].to_list()\n\nkeep_features = importance_df.loc[importance_df[\"Importance\"]>IMPOTTANCE_TH,\"Feature\"].to_list()\ndrop_fes=[feature for feature in keep_features if (feature not in top_features) and (feature in time_series_cols)]\n\ntrain = train.drop(columns=drop_fes)\ntest = test.drop(columns=drop_fes)\n\nprint(f\"---> time series data, # {len([feature for feature in time_series_cols if feature in drop_fes])}\")\nprint([feature for feature in time_series_cols if feature in drop_fes])\nprint(f\"---> general data, # {len([feature for feature in drop_fes if feature not in time_series_cols])}\")\nprint([feature for feature in drop_fes if feature not in time_series_cols])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:31:30.208668Z","iopub.execute_input":"2024-10-08T10:31:30.209211Z","iopub.status.idle":"2024-10-08T10:31:30.241666Z","shell.execute_reply.started":"2024-10-08T10:31:30.209159Z","shell.execute_reply":"2024-10-08T10:31:30.240391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntrain","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:31:30.247089Z","iopub.execute_input":"2024-10-08T10:31:30.247516Z","iopub.status.idle":"2024-10-08T10:31:30.399572Z","shell.execute_reply.started":"2024-10-08T10:31:30.247475Z","shell.execute_reply":"2024-10-08T10:31:30.398386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# FEATURE ENGINEERING(NON-WOE)","metadata":{}},{"cell_type":"markdown","source":"## - histgrams","metadata":{}},{"cell_type":"code","source":"%%time\n\nimport pandas as pd\nimport numpy as np\nfrom lightgbm import LGBMRegressor, LGBMClassifier\n\ndef fill_missing_with_lgbm(train, test, target_column, n_estimators=1000, random_state=42):\n    \"\"\"\n    使用LightGBM模型填补train和test的某个特征中的缺失值：\n    1. 先合并train和test，\n    2. 训练模型补全缺失值，\n    3. 再拆分出补全后的train和test。\n    \n    参数:\n    train (pd.DataFrame): 训练集数据框\n    test (pd.DataFrame): 测试集数据框\n    target_column (str): 需要填补缺失值的目标特征（列名称）\n    model_type (str): 'regression' 或 'classification' 来指定模型类型\n    n_estimators (int): LightGBM基学习器数量\n    random_state (int): 随机种子\n    \n    返回:\n    train_filled, test_filled: 两个DataFrame（train和test的缺失值已被填充）\n    \"\"\"\n    global cat_c\n    if target_column in cat_c:\n        model_type = 'classification'\n    else:\n        model_type = 'regression'\n    \n    # 1. 添加标记列来标识 train 和 test\n    train['is_train'] = 1\n    test['is_train'] = 0\n\n    # 合并train和test数据集\n    df = pd.concat([train, test], ignore_index=True)\n\n    # 2. 找出不包含目标列的特征列\n    features_columns = df.columns[df.columns != target_column].tolist()\n\n    # 提取缺失值的行和不缺失的行\n    df_missing = df[df[target_column].isnull()]  # 缺失值部分\n    df_not_missing = df[~df[target_column].isnull()]  # 不缺失部分\n\n    if df_missing.empty or df_not_missing.empty:\n        print(f\"No missing data in '{target_column}' column or all data are missing.\")\n        return train, test\n\n    # 3. 准备训练集和特征\n    X_train = df_not_missing[features_columns]  # 非缺失值行的特征\n    y_train = df_not_missing[target_column]     # 非缺失值行的目标列\n\n    # 4. 初始化 LGBM 模型\n    if model_type == 'regression':\n        model = LGBMRegressor(n_estimators=n_estimators, random_state=random_state)\n    elif model_type == 'classification':\n        model = LGBMClassifier(n_estimators=n_estimators, random_state=random_state)\n    else:\n        raise ValueError(\"model_type should be either 'regression' or 'classification'.\")\n\n    # 5. 训练模型\n    model.fit(X_train, y_train)\n\n    # 6. 使用模型预测缺失值\n    X_pred = df_missing[features_columns]\n    y_pred = model.predict(X_pred)\n\n    # 7. 用预测值填充缺失值部分\n    df.loc[df[target_column].isnull(), target_column] = y_pred\n\n    # 8. 将数据集重新拆分为原来的 train 和 test\n    train_filled = df[df['is_train'] == 1].drop(columns=['is_train'])\n    test_filled = df[df['is_train'] == 0].drop(columns=['is_train'])\n\n    return train_filled, test_filled\n\n\ntianbu_cols = ['CGAS-CGAS_Score','Physical-BMI','Physical-Height','Physical-Weight', \\\n               'Physical-Diastolic_BP','Physical-HeartRate','Physical-Systolic_BP', \\\n               'SDS-SDS_Total_Raw','SDS-SDS_Total_T','PreInt_EduHx-computerinternet_hoursday']\n\nfor col in tianbu_cols:\n    print(\"开始填补特征：\"+ col)\n    train, test = fill_missing_with_lgbm(train, test, col)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:31:30.401125Z","iopub.execute_input":"2024-10-08T10:31:30.401610Z","iopub.status.idle":"2024-10-08T10:33:35.460598Z","shell.execute_reply.started":"2024-10-08T10:31:30.401567Z","shell.execute_reply":"2024-10-08T10:33:35.459039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nFGC_cols = [\n  'FGC-FGC_CU',\n  'FGC-FGC_CU_Zone',\n  'FGC-FGC_GSND',\n  'FGC-FGC_GSND_Zone',\n  'FGC-FGC_GSD',\n  'FGC-FGC_GSD_Zone',\n  'FGC-FGC_PU',\n  'FGC-FGC_PU_Zone',\n  'FGC-FGC_SRL',\n  'FGC-FGC_SRL_Zone',\n  'FGC-FGC_SRR',\n  'FGC-FGC_SRR_Zone',\n  'FGC-FGC_TL',\n  'FGC-FGC_TL_Zone'\n]\nfor col in FGC_cols:\n    print(\"开始填补特征：\"+ col)\n    train, test = fill_missing_with_lgbm(train, test, col)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:33:35.462191Z","iopub.execute_input":"2024-10-08T10:33:35.462588Z","iopub.status.idle":"2024-10-08T10:35:54.366737Z","shell.execute_reply.started":"2024-10-08T10:33:35.462546Z","shell.execute_reply":"2024-10-08T10:35:54.364963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nBIA = [\"BIA-BIA_TBW\",\"BIA-BIA_TBW\",\"BIA-BIA_DEE\",\"BIA-BIA_BMC\",\"BIA-BIA_Fat\", \"BIA-BIA_BMI\",]\nfor col in BIA:\n    print(\"开始填补特征：\"+ col)\n    train, test = fill_missing_with_lgbm(train, test, col)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:35:54.368962Z","iopub.execute_input":"2024-10-08T10:35:54.369420Z","iopub.status.idle":"2024-10-08T10:36:55.178644Z","shell.execute_reply.started":"2024-10-08T10:35:54.369375Z","shell.execute_reply":"2024-10-08T10:36:55.177498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntrain = update(train)\ntest = update(test)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:36:55.180556Z","iopub.execute_input":"2024-10-08T10:36:55.181344Z","iopub.status.idle":"2024-10-08T10:36:55.206784Z","shell.execute_reply.started":"2024-10-08T10:36:55.181299Z","shell.execute_reply":"2024-10-08T10:36:55.205701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntrain.hist(figsize=(15, 25), bins=20, xlabelsize=8, ylabelsize=8)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:36:55.208374Z","iopub.execute_input":"2024-10-08T10:36:55.208811Z","iopub.status.idle":"2024-10-08T10:37:17.377565Z","shell.execute_reply.started":"2024-10-08T10:36:55.208748Z","shell.execute_reply":"2024-10-08T10:37:17.376275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.dropna(subset='sii')\ntrain[\"sii\"].hist()","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:37:17.379211Z","iopub.execute_input":"2024-10-08T10:37:17.379636Z","iopub.status.idle":"2024-10-08T10:37:17.720353Z","shell.execute_reply.started":"2024-10-08T10:37:17.379594Z","shell.execute_reply":"2024-10-08T10:37:17.719121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## - missing values","metadata":{}},{"cell_type":"code","source":"%%time\n\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import PercentFormatter\nimport polars as pl\n\n# 假设 train 和 test 是你的数据集，top_features 是你的特征列表\n# 这里我们使用 pl.from_pandas 来创建 Polars DataFrame\nsupervised_usable_train = pl.from_pandas(train[top_features])\nsupervised_usable_test = pl.from_pandas(test[top_features])\n\n# 计算训练集的缺失值比例\nmissing_count_train = (\n    supervised_usable_train\n    .null_count()\n    .transpose(include_header=True,\n               header_name='feature',\n               column_names=['null_count'])\n    .with_columns((pl.col('null_count') / len(supervised_usable_train)).alias('null_ratio'))\n)\n\n# 计算测试集的缺失值比例\nmissing_count_test = (\n    supervised_usable_test\n    .null_count()\n    .transpose(include_header=True,\n               header_name='feature',\n               column_names=['null_count'])\n    .with_columns((pl.col('null_count') / len(supervised_usable_test)).alias('null_ratio'))\n)\n\n# 创建一个图形和两个子图\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 30))\n\n# 第一个子图\nax1.barh(np.arange(len(missing_count_train)), missing_count_train.get_column('null_ratio'), color='coral', label='missing')\nax1.barh(np.arange(len(missing_count_train)), \n         1 - missing_count_train.get_column('null_ratio'),\n         left=missing_count_train.get_column('null_ratio'),\n         color='darkseagreen', label='available')\nax1.set_yticks(np.arange(len(missing_count_train)))\nax1.set_yticklabels(missing_count_train.get_column('feature'))\nax1.xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nax1.set_xlim(0, 1)\nax1.set_title('Training Set')\nax1.legend()\n\n# 第二个子图\nax2.barh(np.arange(len(missing_count_test)), missing_count_test.get_column('null_ratio'), color='coral', label='missing')\nax2.barh(np.arange(len(missing_count_test)), \n         1 - missing_count_test.get_column('null_ratio'),\n         left=missing_count_test.get_column('null_ratio'),\n         color='darkseagreen', label='available')\nax2.set_yticks(np.arange(len(missing_count_test)))\nax2.set_yticklabels(missing_count_test.get_column('feature'))\nax2.xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nax2.set_xlim(0, 1)\nax2.set_title('Test Set')\nax2.legend()\n\n# 调整布局\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:37:17.722388Z","iopub.execute_input":"2024-10-08T10:37:17.722932Z","iopub.status.idle":"2024-10-08T10:37:20.595467Z","shell.execute_reply.started":"2024-10-08T10:37:17.722881Z","shell.execute_reply":"2024-10-08T10:37:20.594200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.matrix(train)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:37:20.597372Z","iopub.execute_input":"2024-10-08T10:37:20.598417Z","iopub.status.idle":"2024-10-08T10:37:21.334794Z","shell.execute_reply.started":"2024-10-08T10:37:20.598368Z","shell.execute_reply":"2024-10-08T10:37:21.333463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"msno.matrix(test)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:37:21.336378Z","iopub.execute_input":"2024-10-08T10:37:21.336808Z","iopub.status.idle":"2024-10-08T10:37:21.972093Z","shell.execute_reply.started":"2024-10-08T10:37:21.336748Z","shell.execute_reply":"2024-10-08T10:37:21.970774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## - interaction feature","metadata":{}},{"cell_type":"code","source":"def create_interaction_features(df, feature_pairs):\n    global cat_c\n    for feature1, feature2 in feature_pairs:\n        if(feature1 not in cat_c or feature2 not in cat_c):\n            print(\"feature1:\" + feature1 + \",feature2:\" + feature2)\n            new_feature_name = f\"{feature1}_x_{feature2}\"\n            df[new_feature_name] = df[feature1] * df[feature2]\n    return df\n\ndef create_chufa_features(df, feature_pairs):\n    global cat_c\n    for feature1, feature2 in feature_pairs:\n        if feature1 not in cat_c or feature2 not in cat_c:\n            print(f\"feature1: {feature1}, feature2: {feature2}\")\n            \n            # 构造 A/B 特征，先进行除法运算，但在除法之前控制出现0或NaN的情况\n            new_feature_name1 = f\"{feature1}_div_{feature2}\"\n            df[new_feature_name1] = df[feature1] / df[feature2]\n            \n            # 当A或B中有0或NaN时保留NaN\n            df[new_feature_name1] = df[new_feature_name1].mask((df[feature1] == 0) | (df[feature2] == 0) | \n                                                               df[feature1].isna() | df[feature2].isna())\n            \n            # 构造 B/A 特征，类似方式处理\n            new_feature_name2 = f\"{feature2}_div_{feature1}\"\n            df[new_feature_name2] = df[feature2] / df[feature1]\n            df[new_feature_name2] = df[new_feature_name2].mask((df[feature1] == 0) | (df[feature2] == 0) | \n                                                               df[feature1].isna() | df[feature2].isna())\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:37:21.974483Z","iopub.execute_input":"2024-10-08T10:37:21.974949Z","iopub.status.idle":"2024-10-08T10:37:21.987194Z","shell.execute_reply.started":"2024-10-08T10:37:21.974901Z","shell.execute_reply":"2024-10-08T10:37:21.985904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['Basic_Demos-Sex']=train['Basic_Demos-Sex']+1\ntest['Basic_Demos-Sex']=test['Basic_Demos-Sex']+1\n\nfeature_pairs = [\n    ('PreInt_EduHx-computerinternet_hoursday', 'Basic_Demos-Age'),\n    ('Basic_Demos-Age', 'SDS-SDS_Total_T'),\n    ('FGC-FGC_SRR_Zone', 'SDS-SDS_Total_T'),\n    ('BIA-BIA_BMC', 'Physical-HeartRate'),\n    #('Fitness_Endurance-Season', 'Physical-Waist_Circumference'),\n    ('BIA-BIA_Fat', 'Physical-BMI'),\n    ('PreInt_EduHx-Season', 'Fitness_Endurance-Season'),\n    ('SDS-SDS_Total_T', 'Physical-Systolic_BP'),\n    ('Basic_Demos-Sex', 'FGC-FGC_PU_Zone')\n]\n\ntrain = create_interaction_features(train, feature_pairs)\ntest = create_interaction_features(test, feature_pairs)\ntrain = create_chufa_features(train, feature_pairs)\ntest = create_chufa_features(test, feature_pairs)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:37:21.988709Z","iopub.execute_input":"2024-10-08T10:37:21.989188Z","iopub.status.idle":"2024-10-08T10:37:22.068595Z","shell.execute_reply.started":"2024-10-08T10:37:21.989127Z","shell.execute_reply":"2024-10-08T10:37:22.067494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:37:22.069889Z","iopub.execute_input":"2024-10-08T10:37:22.070299Z","iopub.status.idle":"2024-10-08T10:37:22.235547Z","shell.execute_reply.started":"2024-10-08T10:37:22.070258Z","shell.execute_reply":"2024-10-08T10:37:22.234252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## - rank ordering stratified by age","metadata":{}},{"cell_type":"code","source":"def rank_ordering_rules(datain,quantileout,stratified_var,original_var):\n    \n    import numpy as np\n    import seaborn as sns\n    import matplotlib.pyplot as plt\n    from mpl_toolkits.mplot3d import Axes3D\n    import copy\n    \n    datain=copy.deepcopy(datain)\n\n    # 筛选出'original_var'非空的行\n    non_empty_rows = datain[datain[original_var].notnull()]\n    frequency_table = non_empty_rows[stratified_var].value_counts()\n    frequency_table = frequency_table.reset_index()\n    frequency_table.columns = ['variable', 'frequency']\n    frequency_table = frequency_table.sort_values(by='variable')\n#     print(f\"{original_var} stratified by {stratified_var}\")\n#     print(frequency_table)\n\n    #calculate quantiles\n    quantiles = datain.groupby(stratified_var)[original_var].quantile([0.1, 0.2, 0.3, 0.4, 0.5, 0.6, 0.7, 0.8, 0.9])\n    quantile_data={}\n    quantile_data['stratified_var'] = quantiles.index.get_level_values(0).tolist()\n    quantile_data['quantile'] = quantiles.index.get_level_values(1).tolist()\n    quantile_data['quantile_value'] = quantiles.tolist()\n    quantile_data['stratified_var_name'] =stratified_var\n    quantile_data['original_var_name'] =original_var\n    quantile_data = pd.DataFrame(quantile_data)\n    \n    # plot 2D\n#     plt.figure(figsize=(8, 6))\n#     quantiles = quantiles.unstack(level=-1)\n#     heatmap = sns.heatmap(quantiles, annot=True, fmt=\".2f\", cmap=\"YlGnBu\")\n#     plt.show()\n\n    if quantileout.empty:\n        return quantile_data.reset_index(drop=True)\n    else:\n        return pd.concat([quantileout, quantile_data], axis=0)\n\ndef rank_ordering_rules_loop(datain,quantileout,stratified_var_group,original_var_group):\n        \n    for feature1 in stratified_var_group:\n        for feature2 in original_var_group:\n            quantileout = rank_ordering_rules(datain=datain,quantileout=quantileout,stratified_var=feature1,original_var=feature2)\n    return quantileout\n\ndef rank_ordering_scoring(scoring_data,quantile_data,stratified_var,original_var):\n    \n    # scoring\n    import copy\n    dataout=copy.deepcopy(scoring_data)\n    def safe_quantile(x):\n        matching_rows = quantile_data[(quantile_data['stratified_var'] == x[stratified_var])&(quantile_data['stratified_var_name'] ==stratified_var)&(quantile_data['original_var_name'] ==original_var)]\n        if matching_rows.empty:\n            return np.nan  # 或者某个默认值，如0\n        quantile_values = matching_rows['quantile_value']\n        if quantile_values.empty:\n            return np.nan  # 或者某个默认值，如0\n        index = np.searchsorted(quantile_values, x[original_var], side='left')\n        if index == 0:\n            return 0  # 或者某个默认值，如0\n        return matching_rows['quantile'].iloc[index - 1]\n\n    dataout[f\"{original_var}_by_{stratified_var}_quantile\"] = dataout.apply(safe_quantile, axis=1)\n    \n    dataout.loc[dataout[original_var].isna(), f\"{original_var}_by_{stratified_var}_quantile\"] = np.nan\n#     print(dataout[[stratified_var,original_var,f\"{original_var}_by_{stratified_var}_quantile\"]].reset_index(drop=True).head(10).to_string(index=False))     #data check\n    \n    return dataout\n\ndef rank_ordering_scoring_loop(scoring_data,quantile_data,stratified_var_group,original_var_group):\n        \n    for feature1 in stratified_var_group:\n        for feature2 in original_var_group:\n            scoring_data = rank_ordering_scoring(scoring_data=scoring_data,quantile_data=quantile_data,stratified_var=feature1,original_var=feature2)\n    return scoring_data","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:37:22.237867Z","iopub.execute_input":"2024-10-08T10:37:22.238376Z","iopub.status.idle":"2024-10-08T10:37:22.258559Z","shell.execute_reply.started":"2024-10-08T10:37:22.238330Z","shell.execute_reply":"2024-10-08T10:37:22.257167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time \n\n#stratified by the original variable\nstratified_var='Basic_Demos-Age'\noriginal_var='Physical-BMI'\n\nfrequency_table = train[stratified_var].value_counts()\nfrequency_table = frequency_table.reset_index()\nfrequency_table.columns = ['variable', 'frequency']\nfrequency_table = frequency_table.sort_values(by='variable')\nprint(frequency_table)\n\ndef assign_variable_bin(row):\n    if row['Basic_Demos-Age'] <= 6:\n        return 6\n    elif row['Basic_Demos-Age'] >= 17:\n        return 17\n    else:\n        return row['Basic_Demos-Age']\n\ntrain['Basic_Demos-Age_bin'] = train.apply(assign_variable_bin, axis=1)\ntest['Basic_Demos-Age_bin'] = test.apply(assign_variable_bin, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:37:22.261134Z","iopub.execute_input":"2024-10-08T10:37:22.261616Z","iopub.status.idle":"2024-10-08T10:37:22.353635Z","shell.execute_reply.started":"2024-10-08T10:37:22.261571Z","shell.execute_reply":"2024-10-08T10:37:22.352323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 获取数值类型的列\nnumeric_cols = train.select_dtypes(include=['number']).columns.tolist()\n\n# 要删除的列名列表\ncolumns_to_remove = ['random_numbers', 'sample_size', 'group_rank', 'i_sample','sii','Basic_Demos-Age_bin','Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season']\n\n# 使用列表推导式过滤掉这些列\nnumeric_cols = [col for col in numeric_cols if col not in columns_to_remove]\n\n# 获取分类类型的列\ncategory_cols = train.select_dtypes(include=['category']).columns.tolist()\n\n# 打印结果\nprint(\"Numeric columns:\", numeric_cols)\nprint(\"Category columns:\", category_cols)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:37:22.355237Z","iopub.execute_input":"2024-10-08T10:37:22.355647Z","iopub.status.idle":"2024-10-08T10:37:22.369360Z","shell.execute_reply.started":"2024-10-08T10:37:22.355604Z","shell.execute_reply":"2024-10-08T10:37:22.368116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time\nquantile_data = pd.DataFrame()\nstratified_var_group=['Basic_Demos-Age_bin']\n# original_var_group=numeric_cols\noriginal_var_group=['Physical-BMI', 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference', 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-computerinternet_hoursday', 'anglez_mean', 'non-wear_flag_75%', 'X_max_day', 'Y_mean_night', 'Z_min_day', 'enmo_50%_night', 'enmo_75%_day', 'anglez_std_night', 'non-wear_flag_std_day', 'light_25%_night', 'time_of_day_hours_25%_day', 'time_of_day_hours_50%_night', 'X_mean_1', 'X_25%_3', 'X_max_6', 'Y_mean_2', 'Y_std_6', 'Y_min_2', 'Y_min_5', 'Z_std_3', 'Z_25%_5', 'Z_50%_6', 'Z_75%_3', 'Z_max_5', 'enmo_mean_6', 'enmo_25%_5', 'enmo_75%_3', 'enmo_75%_4', 'enmo_75%_5', 'non-wear_flag_std_5', 'light_mean_4', 'light_std_4', 'light_25%_4', 'light_25%_6', 'light_50%_2', 'light_max_3', 'battery_voltage_mean_1', 'battery_voltage_mean_2', 'battery_voltage_min_5', 'battery_voltage_25%_4', 'battery_voltage_50%_5', 'battery_voltage_75%_4', 'weekday_mean_3', 'weekday_75%_1', 'quarter_mean_1', 'relative_date_PCIAT_mean_1', 'relative_date_PCIAT_25%_2', 'relative_date_PCIAT_max_4', 'time_of_day_hours_75%_5']\nquantile_data=rank_ordering_rules_loop(datain=train,quantileout=quantile_data,stratified_var_group=stratified_var_group,original_var_group=original_var_group)\nquantile_data","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:37:22.370821Z","iopub.execute_input":"2024-10-08T10:37:22.371230Z","iopub.status.idle":"2024-10-08T10:37:23.197083Z","shell.execute_reply.started":"2024-10-08T10:37:22.371188Z","shell.execute_reply":"2024-10-08T10:37:23.195854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time\ntrain=rank_ordering_scoring_loop(scoring_data=train,quantile_data=quantile_data,stratified_var_group=stratified_var_group,original_var_group=original_var_group)\ntest=rank_ordering_scoring_loop(scoring_data=test,quantile_data=quantile_data,stratified_var_group=stratified_var_group,original_var_group=original_var_group)\ntrain","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:37:23.198648Z","iopub.execute_input":"2024-10-08T10:37:23.199131Z","iopub.status.idle":"2024-10-08T10:57:31.336708Z","shell.execute_reply.started":"2024-10-08T10:37:23.199083Z","shell.execute_reply":"2024-10-08T10:57:31.335350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## -feature selection","metadata":{}},{"cell_type":"code","source":"numeric_cols = train.select_dtypes(include=['number']).columns.tolist()\nrank_ordering_features = [col for col in numeric_cols if '_quantile' in col]\n# rank_ordering_features_timeseries = [col for col in rank_ordering_features if '_count' in col or '_mean' in col or '_std' in col or '_min' in col or '_max' in col or '_25%' in col or '_50%' in col or '_75%' in col]\nsuffixes = {'_count', '_mean', '_std', '_min', '_max', '_25%', '_50%', '_75%'}\nrank_ordering_features_timeseries = [col for col in rank_ordering_features if any(suffix in col for suffix in suffixes)]\nrank_ordering_features_general = [col for col in rank_ordering_features if col not in rank_ordering_features_timeseries]","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:31.346615Z","iopub.execute_input":"2024-10-08T10:57:31.347548Z","iopub.status.idle":"2024-10-08T10:57:31.360624Z","shell.execute_reply.started":"2024-10-08T10:57:31.347497Z","shell.execute_reply":"2024-10-08T10:57:31.359205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nX = train.drop(['sii'], axis=1)\ny = train['sii']\n# モデルを学習した後のコード\nXGBoost = xgb.XGBRegressor(random_state=SEED,enable_categorical=True)\nXGBoost.fit(X, y)\n\n# 特徴量重要度を取得\nimportance = XGBoost.feature_importances_\n\n# 特徴量の名前を取得\nfeatures = X.columns\n\n# データフレームとして整理\nimportance_df = pd.DataFrame({'Feature': features, 'Importance': importance})\n\n# 特徴量重要度を降順に並び替え\nimportance_df = importance_df.sort_values(by='Importance', ascending=False)\nimportance_df_plot= importance_df.head(50)\n\nplt.figure(figsize=(10, 25))\nplt.barh(importance_df_plot['Feature'], importance_df_plot['Importance'])\nplt.xlabel('Feature Importance')\nplt.ylabel('Features')\nplt.title('Feature Importance in XGBoost')\nplt.gca().invert_yaxis()  # 重要度が高いものを上に\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:31.362525Z","iopub.execute_input":"2024-10-08T10:57:31.363087Z","iopub.status.idle":"2024-10-08T10:57:36.861857Z","shell.execute_reply.started":"2024-10-08T10:57:31.363019Z","shell.execute_reply":"2024-10-08T10:57:36.860291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMPOTTANCE_TH = 0\ndrop_fes = importance_df.loc[importance_df[\"Importance\"]<=IMPOTTANCE_TH,\"Feature\"].to_list()\ntrain = train.drop(columns=drop_fes)\ntest = test.drop(columns=drop_fes)\nprint(f\"-> all, # {len(drop_fes)}\")\nprint(drop_fes)\nprint(f\"---> time series data, # {len([feature for feature in time_series_cols if feature in drop_fes])}\")\nprint([feature for feature in time_series_cols if feature in drop_fes])\nprint(f\"---> general data, # {len([feature for feature in drop_fes if feature in general_cols])}\")\nprint([feature for feature in drop_fes if feature in general_cols])\nprint(f\"---> rank ordering data, # {len([feature for feature in drop_fes if feature in rank_ordering_features])}\")\nprint([feature for feature in drop_fes if feature in rank_ordering_features])\nprint(f\"---> others, # {len([feature for feature in drop_fes if feature not in time_series_cols+general_cols+rank_ordering_features])}\")\nprint([feature for feature in drop_fes if feature not in time_series_cols+general_cols+rank_ordering_features])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:36.863848Z","iopub.execute_input":"2024-10-08T10:57:36.864305Z","iopub.status.idle":"2024-10-08T10:57:36.884626Z","shell.execute_reply.started":"2024-10-08T10:57:36.864248Z","shell.execute_reply":"2024-10-08T10:57:36.882991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"topN=50\ntop_features = importance_df.sort_values(by=\"Importance\", ascending=False).head(topN)[\"Feature\"].to_list()\nprint(f\"---> time series data, # {len([feature for feature in time_series_cols if feature in top_features])}\")\nprint([feature for feature in time_series_cols if feature in top_features])\nprint(f\"---> rank ordering data (time series), # {len([feature for feature in top_features if feature in rank_ordering_features_timeseries])}\")\nprint([feature for feature in top_features if feature in rank_ordering_features_timeseries])\nprint(f\"---> general data, # {len([feature for feature in top_features if feature in general_cols])}\")\nprint([feature for feature in top_features if feature in general_cols])\nprint(f\"---> rank ordering data (general), # {len([feature for feature in top_features if feature in rank_ordering_features_general])}\")\nprint([feature for feature in top_features if feature in rank_ordering_features_general])\nprint(f\"---> others, # {len([feature for feature in top_features if feature not in time_series_cols+general_cols+rank_ordering_features])}\")\nprint([feature for feature in top_features if feature not in time_series_cols+general_cols+rank_ordering_features])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:36.888373Z","iopub.execute_input":"2024-10-08T10:57:36.888878Z","iopub.status.idle":"2024-10-08T10:57:36.913029Z","shell.execute_reply.started":"2024-10-08T10:57:36.888828Z","shell.execute_reply":"2024-10-08T10:57:36.911473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"topN=100\ntop_features = importance_df.sort_values(by=\"Importance\", ascending=False).head(topN)[\"Feature\"].to_list()\nprint(f\"---> time series data, # {len([feature for feature in time_series_cols if feature in top_features])}\")\nprint([feature for feature in time_series_cols if feature in top_features])\nprint(f\"---> rank ordering data (time series), # {len([feature for feature in top_features if feature in rank_ordering_features_timeseries])}\")\nprint([feature for feature in top_features if feature in rank_ordering_features_timeseries])\nprint(f\"---> general data, # {len([feature for feature in top_features if feature in general_cols])}\")\nprint([feature for feature in top_features if feature in general_cols])\nprint(f\"---> rank ordering data (general), # {len([feature for feature in top_features if feature in rank_ordering_features_general])}\")\nprint([feature for feature in top_features if feature in rank_ordering_features_general])\nprint(f\"---> others, # {len([feature for feature in top_features if feature not in time_series_cols+general_cols+rank_ordering_features])}\")\nprint([feature for feature in top_features if feature not in time_series_cols+general_cols+rank_ordering_features])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:36.914657Z","iopub.execute_input":"2024-10-08T10:57:36.915109Z","iopub.status.idle":"2024-10-08T10:57:36.940261Z","shell.execute_reply.started":"2024-10-08T10:57:36.915056Z","shell.execute_reply":"2024-10-08T10:57:36.938659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"topN=150\ntop_features = importance_df.sort_values(by=\"Importance\", ascending=False).head(topN)[\"Feature\"].to_list()\nprint(f\"---> time series data, # {len([feature for feature in time_series_cols if feature in top_features])}\")\nprint([feature for feature in time_series_cols if feature in top_features])\nprint(f\"---> rank ordering data (time series), # {len([feature for feature in top_features if feature in rank_ordering_features_timeseries])}\")\nprint([feature for feature in top_features if feature in rank_ordering_features_timeseries])\nprint(f\"---> general data, # {len([feature for feature in top_features if feature in general_cols])}\")\nprint([feature for feature in top_features if feature in general_cols])\nprint(f\"---> rank ordering data (general), # {len([feature for feature in top_features if feature in rank_ordering_features_general])}\")\nprint([feature for feature in top_features if feature in rank_ordering_features_general])\nprint(f\"---> others, # {len([feature for feature in top_features if feature not in time_series_cols+general_cols+rank_ordering_features])}\")\nprint([feature for feature in top_features if feature not in time_series_cols+general_cols+rank_ordering_features])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:36.942118Z","iopub.execute_input":"2024-10-08T10:57:36.942599Z","iopub.status.idle":"2024-10-08T10:57:36.971781Z","shell.execute_reply.started":"2024-10-08T10:57:36.942552Z","shell.execute_reply":"2024-10-08T10:57:36.970274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:36.973644Z","iopub.execute_input":"2024-10-08T10:57:36.974189Z","iopub.status.idle":"2024-10-08T10:57:37.252307Z","shell.execute_reply.started":"2024-10-08T10:57:36.974130Z","shell.execute_reply":"2024-10-08T10:57:37.251070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"importance_df=importance_df.loc[importance_df[\"Importance\"]>IMPOTTANCE_TH,]\nimportant_feature=importance_df.sort_values(by=\"Importance\", ascending=False)[\"Feature\"].to_list()\n# important_feature","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:37.253924Z","iopub.execute_input":"2024-10-08T10:57:37.254424Z","iopub.status.idle":"2024-10-08T10:57:37.267941Z","shell.execute_reply.started":"2024-10-08T10:57:37.254366Z","shell.execute_reply":"2024-10-08T10:57:37.266569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# FEATURE ENGINEERING(WOE)","metadata":{}},{"cell_type":"markdown","source":"## - split train dataset/test dataset","metadata":{}},{"cell_type":"code","source":"def stratified_sampling(data_in, sample_pct, stratified_var, count_var):\n    \n    data_in['random_numbers'] = np.random.rand(data_in.shape[0]) \n    \n    #define total sample size desired\n    N = data_in.shape[0]*sample_pct\n    \n    #sample size calculation\n    data_in['sample_size'] = round(data_in.groupby(stratified_var)  \n                                 .transform('count')[count_var]  \n                                 .div(data_in.shape[0])\n                                 .mul(N))\n    #sample selection\n    data_in['group_rank'] = data_in.groupby(stratified_var)['random_numbers'].rank(method='first', ascending=True)\n    data_in['i_sample'] = np.where(train['sample_size'] >= data_in['group_rank'], 1, 0) \n    data_in\n\n    data_out_train = data_in[data_in['i_sample'] != 1].drop(['random_numbers','group_rank','sample_size','i_sample'], axis=1)\n    data_out_test = data_in[data_in['i_sample'] == 1].drop(['random_numbers','group_rank','sample_size','i_sample'], axis=1)\n\n    #check sampled result\n    print(f\"----> || sampled size :: {data_out_test.shape[0]:.0f} & percentage :: {data_out_test.shape[0]/data_in.shape[0]:.2f}\")\n    print(f\"--------> || original sample size ::\")\n    print(data_in.groupby(stratified_var).size().unstack(fill_value=0))\n    print(f\"--------> || sampled sample size ::\")\n    print(data_out_test.groupby(stratified_var).size().unstack(fill_value=0))\n\n    return data_out_train,data_out_test","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:37.270175Z","iopub.execute_input":"2024-10-08T10:57:37.270564Z","iopub.status.idle":"2024-10-08T10:57:37.284190Z","shell.execute_reply.started":"2024-10-08T10:57:37.270525Z","shell.execute_reply":"2024-10-08T10:57:37.282579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" train_train,train_test = stratified_sampling(data_in=train, sample_pct=0.5,stratified_var=['Basic_Demos-Sex', 'sii'],count_var='Basic_Demos-Enroll_Season')","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:37.286121Z","iopub.execute_input":"2024-10-08T10:57:37.286656Z","iopub.status.idle":"2024-10-08T10:57:37.358365Z","shell.execute_reply.started":"2024-10-08T10:57:37.286592Z","shell.execute_reply":"2024-10-08T10:57:37.357123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_train","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:37.360481Z","iopub.execute_input":"2024-10-08T10:57:37.361164Z","iopub.status.idle":"2024-10-08T10:57:37.716445Z","shell.execute_reply.started":"2024-10-08T10:57:37.361103Z","shell.execute_reply":"2024-10-08T10:57:37.715014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sample_consistency_exam(data_1, data_2, exam_var, target_var):\n    \n    import seaborn as sns\n    plt.figure(figsize=(18, 5))\n    plt.subplot(1, 2, 1)\n    unstacked = data_1.groupby([exam_var,target_var]).size().unstack(fill_value=0) \n    row_percentages = unstacked.divide(unstacked.sum(axis=1), axis=0)\n    print(unstacked)\n    sns.set(style=\"whitegrid\")  \n    ax = sns.heatmap(row_percentages, annot=True, fmt=\".2%\", cmap=\"YlGnBu\")  \n    plt.title('Percentages')  \n    plt.xlabel(target_var)  \n    plt.ylabel(exam_var)  \n    \n    plt.subplot(1, 2, 2)\n    unstacked = data_2.groupby([exam_var,target_var]).size().unstack(fill_value=0) \n    row_percentages = unstacked.divide(unstacked.sum(axis=1), axis=0)\n    print(unstacked)\n    sns.set(style=\"whitegrid\")  \n    ax = sns.heatmap(row_percentages, annot=True, fmt=\".2%\", cmap=\"YlGnBu\")  \n    plt.title('Percentages')  \n    plt.xlabel(target_var)  \n    plt.ylabel(exam_var)  \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:37.718605Z","iopub.execute_input":"2024-10-08T10:57:37.719183Z","iopub.status.idle":"2024-10-08T10:57:37.735117Z","shell.execute_reply.started":"2024-10-08T10:57:37.719123Z","shell.execute_reply":"2024-10-08T10:57:37.733530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_consistency_exam(data_1=train,data_2=train_test,exam_var='Basic_Demos-Sex', target_var='sii')\nsample_consistency_exam(data_1=train,data_2=train_test,exam_var='Basic_Demos-Age', target_var='sii')","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:37.736866Z","iopub.execute_input":"2024-10-08T10:57:37.739870Z","iopub.status.idle":"2024-10-08T10:57:40.148273Z","shell.execute_reply.started":"2024-10-08T10:57:37.739794Z","shell.execute_reply":"2024-10-08T10:57:40.146938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## - binning feature","metadata":{"execution":{"iopub.status.busy":"2024-09-28T08:37:11.215575Z","iopub.execute_input":"2024-09-28T08:37:11.216543Z","iopub.status.idle":"2024-09-28T08:37:11.221368Z","shell.execute_reply.started":"2024-09-28T08:37:11.216492Z","shell.execute_reply":"2024-09-28T08:37:11.220281Z"}}},{"cell_type":"code","source":"# train=train.drop(['SDS-SDS_Total_T_bin_sii_1','Basic_Demos-Age_bin_sii_1'], axis=1)\n# train","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:40.150386Z","iopub.execute_input":"2024-10-08T10:57:40.150898Z","iopub.status.idle":"2024-10-08T10:57:40.156480Z","shell.execute_reply.started":"2024-10-08T10:57:40.150842Z","shell.execute_reply":"2024-10-08T10:57:40.155092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dummies of ordinal targets\ntrain_train['sii_1'] = train_train['sii'].apply(lambda x: 1 if x >= 1 else 0)  \ntrain_train['sii_2'] = train_train['sii'].apply(lambda x: 1 if x >= 2 else 0)  \ntrain_train['sii_3'] = train_train['sii'].apply(lambda x: 1 if x >= 3 else 0) \ntrain_train","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:40.158182Z","iopub.execute_input":"2024-10-08T10:57:40.158684Z","iopub.status.idle":"2024-10-08T10:57:40.446181Z","shell.execute_reply.started":"2024-10-08T10:57:40.158625Z","shell.execute_reply":"2024-10-08T10:57:40.444901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def smoothed_binning(datain,num_bins,bin_feature,target_feature,smooth_frac,dataout):\n    \n    import statsmodels.api as sm\n    from copy import copy\n\n    # 使用pandas的qcut进行等频分箱  \n    new_feature_name = f\"{bin_feature}_bin_{target_feature}\"\n    data=copy(datain)\n    data[new_feature_name] = pd.qcut(data[bin_feature], q=num_bins, duplicates='drop')  \n\n    # 计算每个箱的权重（这里简单以目标变量1的比例作为权重）  \n    # 先计算每个箱的统计信息  \n    bin_stats = data.groupby(new_feature_name)[target_feature].agg(['count', 'sum'])  \n    # 计算目标变量为1的比例 \n    bin_stats['prob'] = bin_stats['sum'] / bin_stats['count'].replace(0, 1)\n    bin_stats['variable']=bin_feature\n    bin_stats['target']=target_feature\n    bin_stats['bin']=bin_stats.index\n    bin_stats['id']=range(1,len(bin_stats['bin'].tolist())+1)\n    bin_stats['upper_bound']= bin_stats['bin'].astype(str).str.extract(r'(\\d+\\.\\d+)]').astype(float)\n    bin_stats.loc[bin_stats['id'] ==len(bin_stats['bin'].tolist()), 'upper_bound'] = np.inf\n\n    # 将平滑后的结果添加到原始DataFrame中\n    smoothed_probs = sm.nonparametric.lowess(bin_stats['prob'], bin_stats['id'], frac=smooth_frac, it=0)\n    bin_stats['prob_smoothed'] = smoothed_probs[:, 1]\n    \n    print(f\"--------->{bin_feature} over {target_feature}\")\n    print(bin_stats.reset_index(drop=True).to_string(index=False))\n\n    # 画图\n    df = pd.DataFrame(bin_stats)  \n\n    # 设置图表的宽度和高度  \n    # plt.figure(figsize=(12, 6))  \n\n    # 创建一个子图用于直方图  \n    ax1 = plt.subplot(111)  \n\n    # 绘制count和sum的直方图（这里以堆叠的方式展示，也可以根据需要调整）  \n    ax1.bar(range(1,len(df['bin'].tolist())+1),df['count'].tolist(),label='count', width=0.9, align='center', alpha=0.7)  \n    ax1.bar(range(1,len(df['bin'].tolist())+1),df['sum'].tolist(), bottom=df['count'].tolist(), label='sum', width=0.9, align='center', alpha=0.7)  \n\n    # 设置x轴的刻度标签\n    ax1.set_xticks(range(1,len(df['bin'].tolist())+1))\n    ax1.set_xticklabels(df['bin'].tolist(), rotation=90)\n\n    # 设置直方图的标签和标题  \n#     ax1.set_xlabel('Bin')  \n    ax1.set_ylabel('Count')  \n#     ax1.set_title(new_feature_name)\n\n    # 创建一个共享x轴的第二个y轴用于折线图  \n    ax2 = ax1.twinx()  \n    # 绘制SDS-SDS_Total_T_prob的折线图  \n    ax2.plot(range(1,len(df['bin'].tolist())+1),df['prob'].tolist(), label='prob', color='green', marker='o')  \n    ax2.plot(range(1,len(df['bin'].tolist())+1),df['prob_smoothed'].tolist(), label='prob', color='red', marker='o')  \n    # 设置折线图的标签  \n    ax2.set_ylabel(f\"prob_{target_feature}\")  \n\n    # 添加图例  \n    plt.legend(handles=[ax1.get_children()[0], ax1.get_children()[1], ax2.get_lines()[0], ax2.get_lines()[1]],labels=['# total', '# target=1', 'probability','smoothed probability'])  \n\n    # 显示图表  \n    plt.show()\n    \n    if dataout.empty:\n        return bin_stats.reset_index(drop=True)\n    else:\n        return pd.concat([dataout, bin_stats], axis=0)\n\ndef smoothed_binning_loop(datain,num_bins,bin_feature,target_feature,smooth_frac,dataout):\n    for feature1 in bin_feature:\n        for feature2 in target_feature:\n            dataout = smoothed_binning(datain=datain,num_bins=num_bins,bin_feature=feature1,target_feature=feature2,smooth_frac=smooth_frac,dataout=dataout)\n    return dataout","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:40.447700Z","iopub.execute_input":"2024-10-08T10:57:40.448121Z","iopub.status.idle":"2024-10-08T10:57:40.472191Z","shell.execute_reply.started":"2024-10-08T10:57:40.448076Z","shell.execute_reply":"2024-10-08T10:57:40.470937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 获取数值类型的列\nnumeric_cols = train.select_dtypes(include=['number']).columns.tolist()\nimportant_feature_numeric=[feature for feature in important_feature if feature in numeric_cols]\n\n# 获取分类类型的列\ncategory_cols = train.select_dtypes(include=['category']).columns.tolist()\n\n# 打印结果\nprint(\"Numeric columns:\", important_feature_numeric)\nprint(\"Category columns:\", category_cols)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:40.473931Z","iopub.execute_input":"2024-10-08T10:57:40.474434Z","iopub.status.idle":"2024-10-08T10:57:40.498110Z","shell.execute_reply.started":"2024-10-08T10:57:40.474389Z","shell.execute_reply":"2024-10-08T10:57:40.496857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bin_stats_out = pd.DataFrame()\ntarget_feature=['sii_1']\nbin_stats_out = smoothed_binning_loop(datain=train_train,num_bins=8,bin_feature=important_feature_numeric,target_feature=target_feature,smooth_frac=0.8,dataout=bin_stats_out)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:57:40.499605Z","iopub.execute_input":"2024-10-08T10:57:40.500120Z","iopub.status.idle":"2024-10-08T10:59:45.666751Z","shell.execute_reply.started":"2024-10-08T10:57:40.500055Z","shell.execute_reply":"2024-10-08T10:59:45.665597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 获取数值类型的列\nnumeric_cols = train.select_dtypes(include=['number']).columns.tolist()\nimportant_feature_numeric2=[feature for feature in important_feature if feature in numeric_cols and feature in general_cols+rank_ordering_features_general]\n\n# 获取分类类型的列\ncategory_cols = train.select_dtypes(include=['category']).columns.tolist()\n\n# 打印结果\nprint(\"Numeric columns:\", important_feature_numeric)\nprint(\"Category columns:\", category_cols)\n\n# bin_feature=['CGAS-CGAS_Score', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference', 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-computerinternet_hoursday', 'PreInt_EduHx-computerinternet_hoursday_x_Basic_Demos-Age', 'Basic_Demos-Age_x_SDS-SDS_Total_T', 'FGC-FGC_SRR_Zone_x_SDS-SDS_Total_T', 'BIA-BIA_BMC_x_Physical-HeartRate', 'BIA-BIA_Fat_x_Physical-BMI', 'SDS-SDS_Total_T_x_Physical-Systolic_BP', 'Basic_Demos-Sex_x_FGC-FGC_PU_Zone', 'PreInt_EduHx-computerinternet_hoursday_div_Basic_Demos-Age', 'Basic_Demos-Age_div_PreInt_EduHx-computerinternet_hoursday', 'Basic_Demos-Age_div_SDS-SDS_Total_T', 'SDS-SDS_Total_T_div_Basic_Demos-Age', 'FGC-FGC_SRR_Zone_div_SDS-SDS_Total_T', 'SDS-SDS_Total_T_div_FGC-FGC_SRR_Zone', 'BIA-BIA_BMC_div_Physical-HeartRate', 'Physical-HeartRate_div_BIA-BIA_BMC', 'BIA-BIA_Fat_div_Physical-BMI', 'Physical-BMI_div_BIA-BIA_Fat', 'SDS-SDS_Total_T_div_Physical-Systolic_BP', 'Physical-Systolic_BP_div_SDS-SDS_Total_T', 'Basic_Demos-Sex_div_FGC-FGC_PU_Zone', 'FGC-FGC_PU_Zone_div_Basic_Demos-Sex']\ntarget_feature=['sii_2']\nbin_stats_out = smoothed_binning_loop(datain=train_train,num_bins=8,bin_feature=important_feature_numeric2,target_feature=target_feature,smooth_frac=0.8,dataout=bin_stats_out)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T10:59:45.668218Z","iopub.execute_input":"2024-10-08T10:59:45.668852Z","iopub.status.idle":"2024-10-08T11:00:43.519039Z","shell.execute_reply.started":"2024-10-08T10:59:45.668807Z","shell.execute_reply":"2024-10-08T11:00:43.517623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# bin_stats_out = pd.DataFrame()\n# bin_stats_out = smoothed_binning(datain=train_train,num_bins=8,bin_feature='FGC-FGC_PU',target_feature='sii_1',smooth_frac=0.8,dataout=bin_stats_out)\n# bin_stats_out = smoothed_binning(datain=train,num_bins=30,bin_feature='Basic_Demos-Age',target_feature='sii_1',smooth_frac=0.5,dataout=bin_stats_out)\n# bin_stats_out = smoothed_binning(datain=train,num_bins=30,bin_feature='SDS-SDS_Total_T',target_feature='sii_2',smooth_frac=0.5,dataout=bin_stats_out)\n# bin_stats_out = smoothed_binning(datain=train,num_bins=30,bin_feature='Basic_Demos-Age',target_feature='sii_2',smooth_frac=0.5,dataout=bin_stats_out)\n# bin_stats_out = smoothed_binning(datain=train,num_bins=30,bin_feature='SDS-SDS_Total_T',target_feature='sii_3',smooth_frac=0.5,dataout=bin_stats_out)\n# bin_stats_out = smoothed_binning(datain=train,num_bins=30,bin_feature='Basic_Demos-Age',target_feature='sii_3',smooth_frac=0.5,dataout=bin_stats_out)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:00:43.521240Z","iopub.execute_input":"2024-10-08T11:00:43.521889Z","iopub.status.idle":"2024-10-08T11:00:43.530259Z","shell.execute_reply.started":"2024-10-08T11:00:43.521822Z","shell.execute_reply":"2024-10-08T11:00:43.528668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def binning_transformation(datain,bin_stats_out,original_var,target_var):\n    \n    import pandas as pd\n    import numpy as np\n    import copy\n\n    df=deepcopy(pd.DataFrame(datain))\n\n    # 假设这是您的截断值和对应的转换值\n    cutoffs = bin_stats_out.loc[(bin_stats_out['variable'] == original_var) & (bin_stats_out['target'] == target_var) & (bin_stats_out['upper_bound'] != np.inf),'upper_bound'].tolist() # 这些是截断值，包括末端的inf\n    probs = bin_stats_out.loc[(bin_stats_out['variable'] == original_var) & (bin_stats_out['target'] == target_var),'prob_smoothed'].tolist()\n\n    # 自动创建 conditions 列表\n    conditions = []\n    for i in range(len(cutoffs)):\n        if i == 0:\n            # 第一个条件是小于第一个截断值\n            conditions.append(df[original_var] <= cutoffs[i])\n        else:\n            # 中间的条件是大于前一个截断值且小于或等于当前截断值\n            conditions.append((df[original_var] > cutoffs[i - 1]) & (df[original_var] <= cutoffs[i]))\n    # 添加最后一个条件，是大于最后的截断值\n    if cutoffs:\n        conditions.append(df[original_var] > cutoffs[-1])\n\n    # 使用 np.select 根据条件选择相应的值\n    df[f\"{original_var}_prob_{target_var}\"] = np.select(conditions, probs, default=np.nan)\n\n    return df\n\ndef create_binning_features(datain,bin_stats_out,original_var,target_var):\n    for feature1 in original_var:\n        for feature2 in target_var:\n#             print(f\"{feature1}+{feature2}\")\n            datain=binning_transformation(datain=datain,bin_stats_out=bin_stats_out,original_var=feature1,target_var=feature2)\n    return datain","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:00:43.534424Z","iopub.execute_input":"2024-10-08T11:00:43.534897Z","iopub.status.idle":"2024-10-08T11:00:43.554482Z","shell.execute_reply.started":"2024-10-08T11:00:43.534848Z","shell.execute_reply":"2024-10-08T11:00:43.553058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bin_stats_out","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:00:43.557061Z","iopub.execute_input":"2024-10-08T11:00:43.558918Z","iopub.status.idle":"2024-10-08T11:00:43.600169Z","shell.execute_reply.started":"2024-10-08T11:00:43.558844Z","shell.execute_reply":"2024-10-08T11:00:43.598796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"important_feature_numeric_up = [feature for feature in important_feature_numeric if feature != 'Basic_Demos-Sex']\ntrain=create_binning_features(datain=train,bin_stats_out=bin_stats_out,original_var=important_feature_numeric_up,target_var=['sii_1'])\ntrain_train=create_binning_features(datain=train_train,bin_stats_out=bin_stats_out,original_var=important_feature_numeric_up,target_var=['sii_1'])\ntrain_test=create_binning_features(datain=train_test,bin_stats_out=bin_stats_out,original_var=important_feature_numeric_up,target_var=['sii_1'])\ntest=create_binning_features(datain=test,bin_stats_out=bin_stats_out,original_var=important_feature_numeric_up,target_var=['sii_1'])\n\nimportant_feature_numeric_up2 = [feature for feature in important_feature_numeric2 if feature != 'Basic_Demos-Sex']\ntrain=create_binning_features(datain=train,bin_stats_out=bin_stats_out,original_var=important_feature_numeric_up2,target_var=['sii_2'])\ntrain_train=create_binning_features(datain=train_train,bin_stats_out=bin_stats_out,original_var=important_feature_numeric_up2,target_var=['sii_2'])\ntrain_test=create_binning_features(datain=train_test,bin_stats_out=bin_stats_out,original_var=important_feature_numeric_up2,target_var=['sii_2'])\ntest=create_binning_features(datain=test,bin_stats_out=bin_stats_out,original_var=important_feature_numeric_up2,target_var=['sii_2'])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:00:43.601945Z","iopub.execute_input":"2024-10-08T11:00:43.602394Z","iopub.status.idle":"2024-10-08T11:00:55.567595Z","shell.execute_reply.started":"2024-10-08T11:00:43.602347Z","shell.execute_reply":"2024-10-08T11:00:55.566199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:00:55.569321Z","iopub.execute_input":"2024-10-08T11:00:55.569708Z","iopub.status.idle":"2024-10-08T11:00:56.147662Z","shell.execute_reply.started":"2024-10-08T11:00:55.569667Z","shell.execute_reply":"2024-10-08T11:00:56.146417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## -  feature selection","metadata":{}},{"cell_type":"code","source":"X = train_test.drop(['sii'], axis=1)\n# 'sii_prob_sii_1','sii_prob_sii_2','sii_prob_sii_3'\ny = train_test['sii']\n# モデルを学習した後のコード\nXGBoost = xgb.XGBRegressor(random_state=SEED,enable_categorical=True)\nXGBoost.fit(X, y)\n\n# 特徴量重要度を取得\nimportance = XGBoost.feature_importances_\n\n# 特徴量の名前を取得\nfeatures = X.columns\n\n# データフレームとして整理\nimportance_df = pd.DataFrame({'Feature': features, 'Importance': importance})\n\n# 特徴量重要度を降順に並び替え\nimportance_df = importance_df.sort_values(by='Importance', ascending=False)\n\n# プロット\n# plt.figure(figsize=(10, 90))\n# plt.barh(importance_df['Feature'], importance_df['Importance'])\n# plt.xlabel('Feature Importance')\n# plt.ylabel('Features')\n# plt.title('Feature Importance in XGBoost')\n# plt.gca().invert_yaxis()  # 重要度が高いものを上に\n# plt.show()\n\nimportance_df\n","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:00:56.149208Z","iopub.execute_input":"2024-10-08T11:00:56.149604Z","iopub.status.idle":"2024-10-08T11:01:00.600071Z","shell.execute_reply.started":"2024-10-08T11:00:56.149562Z","shell.execute_reply":"2024-10-08T11:01:00.598817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This plot shows the feature importance for an XGBoost model, ranked from the most important feature to the least important. Key insights from this visualization include:\n\nTop Features:\n\nThe most important feature is PreInt_EduHx-computerinternet_hoursday, indicating that the time spent on computer/internet usage per day plays the largest role in the model's predictions.\nOther highly ranked features include Basic_Demos-Age, FGC-FGC_SRR_Zone, and SDS-SDS_Total_T, which suggest that age and specific health or fitness-related scores are also crucial for the model's decisions.\n\nFeature Distribution:\n\nThere is a wide range of feature importances, but most of them contribute relatively equally beyond the top few features.\nThe long tail of features implies that while the top few features dominate, a large number of other features still contribute in small amounts to the predictions.\n","metadata":{}},{"cell_type":"code","source":"# 特徴量の名前を取得\nfeatures = X.columns\n\n# データフレームとして整理（特徴量重要度と欠損値の割合を結合）\nimportance_df = pd.DataFrame({'Feature': features, 'Importance': importance})\nmissing_df = pd.DataFrame({'Feature': missing_percent_train.index, 'MissingPercent': missing_percent_train.values})\ncombined_df = pd.merge(importance_df, missing_df, on='Feature')\n\n# 散布図を作成\nplt.figure(figsize=(10, 6))\nplt.scatter(combined_df['MissingPercent'], combined_df['Importance'], alpha=0.7)\nplt.xlabel('Missing Percentage (%)')\nplt.ylabel('Feature Importance')\nplt.title('Feature Importance vs Missing Percentage')\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:01:00.601987Z","iopub.execute_input":"2024-10-08T11:01:00.602547Z","iopub.status.idle":"2024-10-08T11:01:01.037894Z","shell.execute_reply.started":"2024-10-08T11:01:00.602485Z","shell.execute_reply":"2024-10-08T11:01:01.036557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMPOTTANCE_TH = 0\ndrop_fes = importance_df.loc[importance_df[\"Importance\"]<=IMPOTTANCE_TH,\"Feature\"].to_list()\ntrain = train.drop(columns=drop_fes)\ntest = test.drop(columns=drop_fes)\ntrain_train = train_train.drop(columns=drop_fes)\ntrain_test = train_test.drop(columns=drop_fes)\nprint(f\"-> all, # {len(drop_fes)}\")\nprint(drop_fes)\nprint(f\"---> time series data, # {len([feature for feature in time_series_cols if feature in drop_fes])}\")\nprint([feature for feature in time_series_cols if feature in drop_fes])\nprint(f\"---> general data, # {len([feature for feature in drop_fes if feature in general_cols])}\")\nprint([feature for feature in drop_fes if feature in general_cols])\nprint(f\"---> rank ordering data, # {len([feature for feature in drop_fes if feature in rank_ordering_features])}\")\nprint([feature for feature in drop_fes if feature in rank_ordering_features])\nprint(f\"---> others, # {len([feature for feature in drop_fes if feature not in time_series_cols+general_cols+rank_ordering_features])}\")\nprint([feature for feature in drop_fes if feature not in time_series_cols+general_cols+rank_ordering_features])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:01:01.039387Z","iopub.execute_input":"2024-10-08T11:01:01.039753Z","iopub.status.idle":"2024-10-08T11:01:01.075880Z","shell.execute_reply.started":"2024-10-08T11:01:01.039714Z","shell.execute_reply":"2024-10-08T11:01:01.074566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"topN=50\ntop_features = importance_df.sort_values(by=\"Importance\", ascending=False).head(topN)[\"Feature\"].to_list()\nprint(f\"---> time series data, # {len([feature for feature in time_series_cols if feature in top_features])}\")\nprint([feature for feature in time_series_cols if feature in top_features])\nprint(f\"---> rank ordering data (time series), # {len([feature for feature in top_features if feature in rank_ordering_features_timeseries])}\")\nprint([feature for feature in top_features if feature in rank_ordering_features_timeseries])\nprint(f\"---> general data, # {len([feature for feature in top_features if feature in general_cols])}\")\nprint([feature for feature in top_features if feature in general_cols])\nprint(f\"---> rank ordering data (general), # {len([feature for feature in top_features if feature in rank_ordering_features_general])}\")\nprint([feature for feature in top_features if feature in rank_ordering_features_general])\nprint(f\"---> others, # {len([feature for feature in top_features if feature not in time_series_cols+general_cols+rank_ordering_features])}\")\nprint([feature for feature in top_features if feature not in time_series_cols+general_cols+rank_ordering_features])","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:01:01.077415Z","iopub.execute_input":"2024-10-08T11:01:01.077833Z","iopub.status.idle":"2024-10-08T11:01:01.096454Z","shell.execute_reply.started":"2024-10-08T11:01:01.077771Z","shell.execute_reply":"2024-10-08T11:01:01.095034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:01:01.098480Z","iopub.execute_input":"2024-10-08T11:01:01.099033Z","iopub.status.idle":"2024-10-08T11:01:01.501901Z","shell.execute_reply.started":"2024-10-08T11:01:01.098972Z","shell.execute_reply":"2024-10-08T11:01:01.500353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 获取各自列的集合\ntrain_columns = set(train_test.columns)\ntest_columns = set(test.columns)\n\n# 查找只在 train 中但不在 test 中的列\ntrain_only = train_columns - test_columns\n\n# 查找只在 test 中但不在 train 中的列\ntest_only = test_columns - train_columns\n\n# 输出差异列\nprint(f\"Columns only in train: {train_only}\")\nprint(f\"Columns only in test: {test_only}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:01:01.503823Z","iopub.execute_input":"2024-10-08T11:01:01.504363Z","iopub.status.idle":"2024-10-08T11:01:01.513545Z","shell.execute_reply.started":"2024-10-08T11:01:01.504292Z","shell.execute_reply":"2024-10-08T11:01:01.511910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 获取各自列的集合\ntrain_columns = set(train.columns)\ntest_columns = set(train_test.columns)\n\n# 查找只在 train 中但不在 test 中的列\ntrain_only = train_columns - test_columns\n\n# 查找只在 test 中但不在 train 中的列\ntest_only = test_columns - train_columns\n\n# 输出差异列\nprint(f\"Columns only in train: {train_only}\")\nprint(f\"Columns only in test: {test_only}\")\n\ntrain=train.drop(['random_numbers', 'i_sample', 'sample_size', 'group_rank'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:01:01.515108Z","iopub.execute_input":"2024-10-08T11:01:01.515512Z","iopub.status.idle":"2024-10-08T11:01:01.531260Z","shell.execute_reply.started":"2024-10-08T11:01:01.515467Z","shell.execute_reply":"2024-10-08T11:01:01.529893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 获取各自列的集合\ntrain_columns = set(train_train.columns)\ntest_columns = set(test.columns)\n\n# 查找只在 train 中但不在 test 中的列\ntrain_only = train_columns - test_columns\n\n# 查找只在 test 中但不在 train 中的列\ntest_only = test_columns - train_columns\n\n# 输出差异列\nprint(f\"Columns only in train: {train_only}\")\nprint(f\"Columns only in test: {test_only}\")\n\ntrain_train=train_train.drop(['sii_3', 'sii_1', 'sii_2'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:01:01.533209Z","iopub.execute_input":"2024-10-08T11:01:01.533739Z","iopub.status.idle":"2024-10-08T11:01:01.551785Z","shell.execute_reply.started":"2024-10-08T11:01:01.533682Z","shell.execute_reply":"2024-10-08T11:01:01.550384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# MODEL","metadata":{}},{"cell_type":"markdown","source":"## -Training & Tunning","metadata":{}},{"cell_type":"code","source":"%%time\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, train, test_data):\n    \n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n    \n    test_data = test_data.drop(['sii'], axis=1)\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n        \n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead') # Nelder-Mead | # Powell\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission, tKappa\n\ndef TrainML_with_plot(model_class, train, test_data):\n    \n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n    test_data = test_data.drop(['sii'], axis=1)\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    results_iter_all = pd.DataFrame()  \n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n\n        # define the datasets to evaluate each iteration\n        evalset = [(X_train, y_train), (X_val,y_val)]\n        model.fit(X_train, y_train,eval_set=evalset)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        # retrieve performance metrics\n        results = model.evals_result()\n        results_iter = pd.DataFrame(np.column_stack((results['validation_0']['rmse'], results['validation_1']['rmse'],np.full(len(results['validation_1']['rmse']), fold+1),range(1, len(results['validation_1']['rmse']) + 1) )), columns=['train_rmse', 'validation_rmse','fold','iter'])\n        results_iter_all = pd.concat([results_iter_all, results_iter], axis=0) \n        \n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        \n    clear_output(wait=True)\n    \n    # plot learning curves\n    from matplotlib import pyplot\n    for fold in range(1,n_splits+1):\n        print(f\"Fold --> {fold:.0f}\")\n        plot_data=results_iter_all[results_iter_all['fold'] == fold]\n        pyplot.plot(plot_data['train_rmse'], label='train')\n        pyplot.plot(plot_data['validation_rmse'], label='test')\n        # show the legend\n        pyplot.legend()\n        # show the plot\n        pyplot.show()\n    \n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead') # Nelder-Mead | # Powell\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission, tKappa, results_iter_all","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:01:01.556851Z","iopub.execute_input":"2024-10-08T11:01:01.557462Z","iopub.status.idle":"2024-10-08T11:01:01.594604Z","shell.execute_reply.started":"2024-10-08T11:01:01.557407Z","shell.execute_reply":"2024-10-08T11:01:01.593074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#调参函数\ndef param_tunning(max_evals,type,train_data,scoring_data):\n    \n    from hyperopt import fmin, tpe, hp, Trials, STATUS_OK  \n    from sklearn.metrics import mean_squared_error  \n    from sklearn.model_selection import train_test_split  \n    import xgboost as xgb  \n\n\n    def objective(params):  \n        # 从params字典中提取超参数  \n        learning_rate = params['learning_rate']  \n        n_estimators = int(params['n_estimators'])  \n        iterations = int(params['iterations'])  \n        subsample = params['subsample']  \n        colsample_bytree = params['colsample_bytree']  \n        max_depth = int(params['max_depth'])  \n        l2_leaf_reg=params['l2_leaf_reg']   \n        random_strength=params['random_strength']   \n        bagging_temperature=params['bagging_temperature']   \n        border_count=params['border_count']\n        reg_alpha=params['reg_alpha']\n        reg_lambda=params['reg_lambda']\n        min_child_weight=params['min_child_weight']\n        \n\n        # 定义模型 \n        if type=='LGBM':\n            Light = lgb.LGBMRegressor(\n                learning_rate=learning_rate,  \n                n_estimators=n_estimators,  \n                subsample=subsample,  \n                colsample_bytree=colsample_bytree,  \n                max_depth=max_depth,\n                min_child_weight=min_child_weight,\n                reg_alpha=reg_alpha,\n                reg_lambda=reg_lambda,\n                verbosity=0,\n                random_state=SEED, \n                verbose=-1\n            )    \n            Submission_XGB, k_lgbm = TrainML(Light,train_data,scoring_data)\n            return {'loss': -k_lgbm, 'status': STATUS_OK}\n        elif type=='XGB':\n            XGBoost = xgb.XGBRegressor(  \n                learning_rate=learning_rate,  \n                n_estimators=n_estimators,  \n                subsample=subsample,  \n                colsample_bytree=colsample_bytree,  \n                max_depth=max_depth,  \n                objective='reg:squarederror',  \n                verbosity=0,  \n                eval_metric='rmse',  \n                random_state=SEED,  \n                enable_categorical=True  \n            )  \n            Submission_XGB, k_xgb = TrainML(XGBoost,train_data,scoring_data)\n            return {'loss': -k_xgb, 'status': STATUS_OK}\n\n        elif type=='CAT':\n            CatBoost = CatBoostRegressor(  \n                learning_rate=learning_rate,  \n                iterations=iterations,  \n                subsample=subsample,  \n                max_depth=max_depth,\n                l2_leaf_reg=l2_leaf_reg,\n                random_strength=random_strength,\n                bagging_temperature=bagging_temperature,\n                border_count=border_count,\n                random_state=SEED, verbose=0,cat_features=cat_c  \n            )  \n            Submission_CatBoost , k_cat= TrainML(CatBoost,train_data,scoring_data)\n            return {'loss': -k_cat, 'status': STATUS_OK}\n\n\n    space = {  \n        'learning_rate': hp.loguniform('learning_rate', -3, 0),  \n         'n_estimators': hp.quniform('n_estimators', 100,500,100),  \n        'iterations': hp.quniform('iterations', 100,500,100),\n        'subsample': hp.quniform('subsample', 0.6,0.9,0.1),  \n        'colsample_bytree': hp.quniform('colsample_bytree', 0.6,0.9,0.1),  \n        'max_depth': hp.quniform('max_depth', 2, 4, 1),\n        'l2_leaf_reg': hp.loguniform('l2_leaf_reg', np.log(1e-3), np.log(10)),\n        'random_strength': hp.uniform('random_strength', 0.1, 2),\n        'bagging_temperature': hp.uniform('bagging_temperature', 0.01, 1),\n        'border_count': hp.quniform('border_count', 8, 255, 1),\n        'reg_alpha': hp.loguniform('reg_alpha', 1e-3, 10),\n        'reg_lambda': hp.loguniform('reg_lambda', 1e-3, 10),\n        'min_child_weight':hp.loguniform('min_child_weight', 1e-3, 10)\n    }\n\n    trials = Trials()  \n    \n    best = fmin(fn=objective,  \n                space=space,  \n                algo=tpe.suggest,  \n                max_evals=max_evals,\n                trials=trials)  \n\n#     for trial in trials.trials:  \n#         print(\"Trial result:\")  \n#         print(\"Params:\", trial['misc']['vals'])  \n#         print(\"Loss:\", trial['result']['loss'])  \n#         print(\"Status:\", trial['result']['status'])  \n#         print(\"---\")  \n#     clear_output(wait=True) \n    \n    best['max_depth'] = int(best['max_depth'])\n    if type=='CAT':\n        best['iterations'] = int(best['iterations'])\n    else:\n        best['n_estimators'] = int(best['n_estimators'])\n        \n    if type=='LGBM':\n        best_filtered = {k: v for k, v in best.items() if k in ['learning_rate', 'n_estimators','subsample', 'colsample_bytree','max_depth', 'min_child_weight','reg_alpha', 'reg_lambda']}\n    elif type=='XGB':\n        best_filtered = {k: v for k, v in best.items() if k in ['learning_rate', 'n_estimators','subsample', 'colsample_bytree','max_depth']}\n    elif type=='CAT':\n        best_filtered = {k: v for k, v in best.items() if k in ['learning_rate', 'iterations','subsample','max_depth','l2_leaf_reg','random_strength','bagging_temperature']}\n                \n    print(f\"--->{type} best parameters:{best_filtered}\")\n    \n    return best_filtered","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:01:01.596864Z","iopub.execute_input":"2024-10-08T11:01:01.597352Z","iopub.status.idle":"2024-10-08T11:01:01.627471Z","shell.execute_reply.started":"2024-10-08T11:01:01.597303Z","shell.execute_reply":"2024-10-08T11:01:01.625906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_params_lgbm=param_tunning(max_evals=50,type='LGBM',train_data=train_test,scoring_data=test)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:01:01.629202Z","iopub.execute_input":"2024-10-08T11:01:01.629616Z","iopub.status.idle":"2024-10-08T11:01:01.646850Z","shell.execute_reply.started":"2024-10-08T11:01:01.629573Z","shell.execute_reply":"2024-10-08T11:01:01.645559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# best_params_xgb=param_tunning(max_evals=50,type='XGB',train_data=train_test,scoring_data=test)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:01:01.648519Z","iopub.execute_input":"2024-10-08T11:01:01.648961Z","iopub.status.idle":"2024-10-08T11:01:01.663893Z","shell.execute_reply.started":"2024-10-08T11:01:01.648917Z","shell.execute_reply":"2024-10-08T11:01:01.662179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# best_params_catboost=param_tunning(max_evals=50,type='CAT',train_data=train_test,scoring_data=test)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:01:01.665468Z","iopub.execute_input":"2024-10-08T11:01:01.666019Z","iopub.status.idle":"2024-10-08T11:01:01.677800Z","shell.execute_reply.started":"2024-10-08T11:01:01.665950Z","shell.execute_reply":"2024-10-08T11:01:01.676493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_splits=10\nbest_params_lgbm = {'learning_rate': 0.011747572224219955, 'n_estimators': 180,  'min_child_weight': 0.0025036281384857462, 'colsample_bytree': 0.6648896193058901, 'reg_alpha': 0.9, 'reg_lambda': 0.12158717311465662}\nLight = lgb.LGBMRegressor(**best_params_lgbm, random_state=SEED, verbose=-1)\nSubmission_LGBM, k_lgbm = TrainML(Light, train, test)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T13:48:43.048295Z","iopub.execute_input":"2024-10-08T13:48:43.048866Z","iopub.status.idle":"2024-10-08T13:49:24.277537Z","shell.execute_reply.started":"2024-10-08T13:48:43.048808Z","shell.execute_reply":"2024-10-08T13:49:24.276244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_splits=10\n# best_params_xgb = {'colsample_bytree': 0.6, 'learning_rate': 0.07354863493566943, 'max_depth': 3, 'n_estimators': 200, 'subsample': 0.8}\nbest_params_xgb = {'colsample_bytree': 0.7000000000000001, 'learning_rate': 0.05266814142163941, 'max_depth': 3, 'n_estimators': 100, 'subsample': 0.9}\nXGBoost = xgb.XGBRegressor(**best_params_xgb, random_state=SEED,enable_categorical=True)\nSubmission_XGB, k_xgb = TrainML(XGBoost,train,test)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T13:45:19.988207Z","iopub.execute_input":"2024-10-08T13:45:19.988751Z","iopub.status.idle":"2024-10-08T13:45:33.494716Z","shell.execute_reply.started":"2024-10-08T13:45:19.988702Z","shell.execute_reply":"2024-10-08T13:45:33.493182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_splits=10\n# best_params_catboost = {'bagging_temperature': 0.32649664918447163, 'iterations': 200, 'l2_leaf_reg': 0.018903935374679576, 'learning_rate': 0.11646016498659839, 'max_depth': 3, 'random_strength': 1.075785966806238, 'subsample': 0.6000000000000001}\nbest_params_catboost = {'bagging_temperature': 0.05562451310592907, 'iterations': 100, 'l2_leaf_reg': 2.3813787042100976, 'learning_rate': 0.055256285805877665, 'max_depth': 4, 'random_strength': 0.2727373280481198, 'subsample': 0.8}\nCatBoost = CatBoostRegressor(**best_params_catboost, random_state=SEED, verbose=0,cat_features=cat_c)\nSubmission_CatBoost , k_cat= TrainML(CatBoost,train,test)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T13:46:59.194227Z","iopub.execute_input":"2024-10-08T13:46:59.194702Z","iopub.status.idle":"2024-10-08T13:47:12.516796Z","shell.execute_reply.started":"2024-10-08T13:46:59.194660Z","shell.execute_reply":"2024-10-08T13:47:12.515563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# best_params_lgbm = {'learning_rate': 0.011747572224219955, 'n_estimators': 993,  'min_child_weight': 0.0025036281384857462, 'colsample_bytree': 0.6648896193058901, 'reg_alpha': 0.7153672744430527, 'reg_lambda': 0.12158717311465662}\n# best_params_xgb = {'learning_rate': 0.007356059931165658, 'n_estimators': 957, 'subsample': 0.6555266544650088, 'colsample_bytree': 0.7712019245727745}\n# best_params_catboost = {'iterations': 804, 'learning_rate': 0.007849710402582562, 'l2_leaf_reg': 7.31183636902306, 'subsample': 0.5630297785016092, 'random_strength': 1.7097065892440113, 'bagging_temperature': 0.026593521316435192, 'border_count': 12}","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:02:19.108073Z","iopub.execute_input":"2024-10-08T11:02:19.108611Z","iopub.status.idle":"2024-10-08T11:02:19.115013Z","shell.execute_reply.started":"2024-10-08T11:02:19.108553Z","shell.execute_reply":"2024-10-08T11:02:19.113556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import copy\ntrain_columns = set(train.columns)\nsuffixes = {'_count', '_mean', '_std', '_min', '_max', '_25%', '_50%', '_75%'}\nkeep_columns = [col for col in train_columns if not any(suffix in col for suffix in suffixes)]\ntrain2= copy.deepcopy(train[keep_columns])\ntest2= copy.deepcopy(test[keep_columns])\ntrain2","metadata":{"execution":{"iopub.status.busy":"2024-10-08T12:26:18.930475Z","iopub.execute_input":"2024-10-08T12:26:18.931098Z","iopub.status.idle":"2024-10-08T12:26:19.337916Z","shell.execute_reply.started":"2024-10-08T12:26:18.931032Z","shell.execute_reply":"2024-10-08T12:26:19.336532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_splits=10\nbest_params_lgbm = {'learning_rate': 0.011747572224219955, 'n_estimators': 150,  'min_child_weight': 0.0025036281384857462, 'colsample_bytree': 0.6648896193058901, 'reg_alpha': 0.9, 'reg_lambda': 0.12158717311465662}\nLight = lgb.LGBMRegressor(**best_params_lgbm, random_state=SEED, verbose=-1)\nSubmission_LGBM2, k_lgbm2 = TrainML(Light, train2, test2)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T13:38:05.178753Z","iopub.execute_input":"2024-10-08T13:38:05.179343Z","iopub.status.idle":"2024-10-08T13:38:26.099705Z","shell.execute_reply.started":"2024-10-08T13:38:05.179288Z","shell.execute_reply":"2024-10-08T13:38:26.098250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_splits=10\n# best_params_xgb = {'colsample_bytree': 0.6, 'learning_rate': 0.07354863493566943, 'max_depth': 3, 'n_estimators': 200, 'subsample': 0.8}\nbest_params_xgb = {'colsample_bytree': 0.7000000000000001, 'learning_rate': 0.05266814142163941, 'max_depth': 3, 'n_estimators': 100, 'subsample': 0.9}\nXGBoost = xgb.XGBRegressor(**best_params_xgb, random_state=SEED,enable_categorical=True)\nSubmission_XGB2, k_xgb2 = TrainML(XGBoost,train2,test2)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T13:35:58.519694Z","iopub.execute_input":"2024-10-08T13:35:58.520816Z","iopub.status.idle":"2024-10-08T13:36:07.301352Z","shell.execute_reply.started":"2024-10-08T13:35:58.520747Z","shell.execute_reply":"2024-10-08T13:36:07.300132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_splits=10\n# best_params_catboost = {'bagging_temperature': 0.32649664918447163, 'iterations': 200, 'l2_leaf_reg': 0.018903935374679576, 'learning_rate': 0.11646016498659839, 'max_depth': 3, 'random_strength': 1.075785966806238, 'subsample': 0.6000000000000001}\nbest_params_catboost = {'bagging_temperature': 0.05562451310592907, 'iterations': 100, 'l2_leaf_reg': 2.3813787042100976, 'learning_rate': 0.055256285805877665, 'max_depth': 4, 'random_strength': 0.2727373280481198, 'subsample': 0.8}\nCatBoost = CatBoostRegressor(**best_params_catboost, random_state=SEED, verbose=0,cat_features=cat_c)\nSubmission_CatBoost2 , k_cat2= TrainML(CatBoost,train2,test2)","metadata":{"execution":{"iopub.status.busy":"2024-10-08T13:46:22.553258Z","iopub.execute_input":"2024-10-08T13:46:22.553723Z","iopub.status.idle":"2024-10-08T13:46:32.045413Z","shell.execute_reply.started":"2024-10-08T13:46:22.553678Z","shell.execute_reply":"2024-10-08T13:46:32.044169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SUBMIT","metadata":{}},{"cell_type":"code","source":"print(Submission_LGBM['sii'].value_counts())\nprint(Submission_XGB['sii'].value_counts())\nprint(Submission_CatBoost['sii'].value_counts())\nprint(Submission_LGBM2['sii'].value_counts())\nprint(Submission_XGB2['sii'].value_counts())\nprint(Submission_CatBoost2['sii'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2024-10-08T13:41:23.154284Z","iopub.execute_input":"2024-10-08T13:41:23.154829Z","iopub.status.idle":"2024-10-08T13:41:23.169232Z","shell.execute_reply.started":"2024-10-08T13:41:23.154752Z","shell.execute_reply":"2024-10-08T13:41:23.167973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 通过求和 k值计算每个模型的权重\ntotal_k = k_cat + k_xgb + k_lgbm + k_cat2 + k_xgb2 + k_lgbm2\n\nweight_cat = k_cat / total_k    # CatBoost权重\nweight_xgb = k_xgb / total_k    # XGBoost权重\nweight_lgbm = k_lgbm / total_k  # LightGBM权重\nweight_cat2 = k_cat2 / total_k    # CatBoost权重\nweight_xgb2 = k_xgb2 / total_k    # XGBoost权重\nweight_lgbm2 = k_lgbm2 / total_k  # LightGBM权重\n\n# 为每个模型准备预测结果（提交）\n# 假设 “sii ”是类别标签\nensemble_df = pd.DataFrame({\n    'id': Submission_LGBM['id'],\n    'cat': Submission_CatBoost[\"sii\"],\n    'xgb': Submission_XGB[\"sii\"],\n    'lgbm': Submission_LGBM[\"sii\"],\n    'cat2': Submission_CatBoost2[\"sii\"],\n    'xgb2': Submission_XGB2[\"sii\"],\n    'lgbm2': Submission_LGBM2[\"sii\"]\n})\n\n# 将预测结果转换为较长的格式\nmelted = ensemble_df.melt(id_vars='id', value_vars=['cat','xgb','lgbm','cat2','xgb2','lgbm2'], \n                          var_name='model', value_name='sii')\n\n# 为每个模型分配相应权重\nmelted['weight'] = melted['model'].map({\n    'cat': weight_cat,\n    'xgb': weight_xgb,\n    'lgbm': weight_lgbm,\n    'cat2': weight_cat2,\n    'xgb2': weight_xgb2,\n    'lgbm2': weight_lgbm2,\n})\n\n#weights revision\nids = ['00115b9f', '001f3379']\nmodels = ['cat', 'xgb', 'lgbm']\nmodels2 = ['cat2', 'xgb2', 'lgbm2']\n\nmelted['weight_revised'] = melted.apply(lambda row: row['weight'] * 0.7 if (row['id'] in ids and row['model'] in models) else\n                             row['weight'] * 0.3 if (row['id'] in ids and row['model'] not in models2) else\n                             row['weight'] * 0.7 if (row['id'] not in ids and row['model'] in models2) else\n                             row['weight'] * 0.3, axis=1)\n\n# 每个 ID 和每个 sii 的总权重\ngrouped = melted.groupby(['id', 'sii'])['weight_revised'].sum().reset_index()\n\n# 为每个 ID 选择权重最大的 sii\nbest_submission = grouped.loc[grouped.groupby('id')['weight_revised'].idxmax()][['id', 'sii']]","metadata":{"execution":{"iopub.status.busy":"2024-10-08T13:51:38.223040Z","iopub.execute_input":"2024-10-08T13:51:38.223619Z","iopub.status.idle":"2024-10-08T13:51:38.264205Z","shell.execute_reply.started":"2024-10-08T13:51:38.223569Z","shell.execute_reply":"2024-10-08T13:51:38.262660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # 通过求和 k 值计算每个模型的权重\n# total_k = k_cat + k_xgb + k_lgbm\n\n# weight_cat = k_cat / total_k    # CatBoost权重\n# weight_xgb = k_xgb / total_k    # XGBoost权重\n# weight_lgbm = k_lgbm / total_k  # LightGBM权重\n\n# # 为每个模型准备预测结果（提交）\n# # 假设 “sii ”是类别标签\n# ensemble_df = pd.DataFrame({\n#     'id': Submission_LGBM['id'],\n#     'cat': Submission_CatBoost[\"sii\"],\n#     'xgb': Submission_XGB[\"sii\"],\n#     'lgbm': Submission_LGBM[\"sii\"]\n# })\n\n# # 将预测结果转换为较长的格式\n# melted = ensemble_df.melt(id_vars='id', value_vars=['cat', 'xgb', 'lgbm'], \n#                           var_name='model', value_name='sii')\n\n# # 为每个模型分配相应权重\n# melted['weight'] = melted['model'].map({\n#     'cat': weight_cat,\n#     'xgb': weight_xgb,\n#     'lgbm': weight_lgbm\n# })\n\n# # 每个 ID 和每个 sii 的总权重\n# grouped = melted.groupby(['id', 'sii'])['weight'].sum().reset_index()\n\n# # 为每个 ID 选择权重最大的 sii\n# best_submission = grouped.loc[grouped.groupby('id')['weight'].idxmax()][['id', 'sii']]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"comparison_df = best_submission.merge(ensemble_df, on='id', how='left')\ncomparison_df","metadata":{"execution":{"iopub.status.busy":"2024-10-08T13:51:46.180371Z","iopub.execute_input":"2024-10-08T13:51:46.180833Z","iopub.status.idle":"2024-10-08T13:51:46.202717Z","shell.execute_reply.started":"2024-10-08T13:51:46.180790Z","shell.execute_reply":"2024-10-08T13:51:46.201449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nbest_submission.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-08T11:14:13.381438Z","iopub.execute_input":"2024-10-08T11:14:13.381930Z","iopub.status.idle":"2024-10-08T11:14:13.391503Z","shell.execute_reply.started":"2024-10-08T11:14:13.381882Z","shell.execute_reply":"2024-10-08T11:14:13.390189Z"},"trusted":true},"execution_count":null,"outputs":[]}]}