{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7074842,"sourceType":"datasetVersion","datasetId":4074593},{"sourceId":7570020,"sourceType":"datasetVersion","datasetId":4021289},{"sourceId":10212925,"sourceType":"datasetVersion","datasetId":6312307},{"sourceId":10221130,"sourceType":"datasetVersion","datasetId":6318540},{"sourceId":203746103,"sourceType":"kernelVersion"},{"sourceId":212791578,"sourceType":"kernelVersion"}],"dockerImageVersionId":30805,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install --no-index -U --find-links=/kaggle/input/tensorflow-2-15/tensorflow tensorflow==2.15.0\n!pip install --no-index -U --find-links=/kaggle/input/deeptables-v0-2-5/deeptables-0.2.5 deeptables==0.2.5\n!pip install --no-index -U --find-links=/kaggle/input/fix-deeptables/deeptables-0.2.6 deeptables==0.2.6","metadata":{"trusted":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-12-17T08:02:55.032036Z","iopub.execute_input":"2024-12-17T08:02:55.032794Z","iopub.status.idle":"2024-12-17T08:04:13.743124Z","shell.execute_reply.started":"2024-12-17T08:02:55.032744Z","shell.execute_reply":"2024-12-17T08:04:13.742014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport math\nimport random, copy\nimport warnings\nfrom pathlib import Path\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom colorama import Fore, Style\nfrom numpy.typing import ArrayLike\nfrom scipy.optimize import minimize\nfrom sklearn.base import BaseEstimator\nfrom sklearn.impute import SimpleImputer\nfrom polars.testing import assert_frame_equal\nfrom sklearn.metrics import cohen_kappa_score\nimport numpy as np, pandas as pd, polars as pl\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.model_selection import StratifiedKFold\n\nimport tensorflow as tf, deeptables as dt\nfrom tensorflow.keras import backend as K\nfrom tensorflow.keras.utils import plot_model\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom tensorflow.keras.optimizers.legacy import Adam\nfrom deeptables.models import DeepTable, ModelConfig, deepnets\nfrom tensorflow.keras.models import load_model\nfrom IPython.display import clear_output\n\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import make_scorer\nfrom sklearn.base import clone\nfrom tqdm import tqdm\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\n\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.pipeline import Pipeline\n\nwarnings.filterwarnings(\"ignore\")\nprint('TensorFlow version:',tf.__version__+',',\n      'GPU =',tf.test.is_gpu_available())\nprint('DeepTables version:',dt.__version__)\n\nimport sys\nsys.path.append('/kaggle/input/futils7/')\nfrom fea_utils import load_time_series, dfinto_csvdf, feature_engineering, concat_feature_engineering, \\\nget_optimized_kappa_score, threshold_Rounder, create_mapping, TrainML,\\\nget_optimized_kappa_score_global, train_deep, imputerDf, find_optimal_t, quadratic_weighted_kappa, evaluate_predictions\n\ndef seed_everything(seed):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\nseed_everything(seed=42)","metadata":{"trusted":true,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2024-12-17T08:08:24.634007Z","iopub.execute_input":"2024-12-17T08:08:24.634387Z","iopub.status.idle":"2024-12-17T08:08:24.674767Z","shell.execute_reply.started":"2024-12-17T08:08:24.634355Z","shell.execute_reply":"2024-12-17T08:08:24.674008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"above_dir = \"/kaggle/input/child-mind-institute-problematic-internet-use/\"\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n# train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntrain = pd.read_csv('/kaggle/input/feature/train.csv')\ntrain_ts = pd.read_csv('/kaggle/input/feature/train_ts.csv')\ntest = dfinto_csvdf(test, 'test', above_dir)\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T08:08:26.034598Z","iopub.execute_input":"2024-12-17T08:08:26.035430Z","iopub.status.idle":"2024-12-17T08:08:31.356873Z","shell.execute_reply.started":"2024-12-17T08:08:26.035393Z","shell.execute_reply":"2024-12-17T08:08:31.355970Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train, test = concat_feature_engineering(train, test)\n\nfeaturesCols = list(train.loc[:, ~train.columns.str.contains('PCIAT-PCIAT')].columns )\nfeaturesCols = [i for i in featuresCols if i!='PCIAT-Season']\ntrain = train[featuresCols]\ntest = test[featuresCols]\ntest = test.drop(columns=['sii'], axis=1)\ntrain = train.dropna(subset=[\"sii\"]).reset_index(drop=True)\n\n###TS#########\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\nfeaturesCols += time_series_cols\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\n#############\n\ninner_test_id = list(test['id'])\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)  \n\nnumerical_columns = train.select_dtypes(include=['int64', 'float64']).columns\nnumerical_columns = numerical_columns.drop('sii') \ntrain = feature_engineering(train,numerical_columns)\ntest  = feature_engineering(test,numerical_columns)\n\n\nnumerical_columns = train.select_dtypes(include=['int64', 'float64']).columns\nnumerical_columns = numerical_columns.drop('sii') \n\nprint(len(train.columns)), print(len(test.columns))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T08:08:31.358830Z","iopub.execute_input":"2024-12-17T08:08:31.359136Z","iopub.status.idle":"2024-12-17T08:08:31.735259Z","shell.execute_reply.started":"2024-12-17T08:08:31.359107Z","shell.execute_reply":"2024-12-17T08:08:31.734378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = ['Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-CGAS_Score', 'Physical-Height', 'Physical-Weight', \n              'Physical-Waist_Circumference', 'Physical-HeartRate', 'Physical-Systolic_BP', 'FGC-FGC_CU', \n              'FGC-FGC_GSND', 'FGC-FGC_GSD', 'FGC-FGC_PU', 'FGC-FGC_SRL', 'FGC-FGC_SRR', 'BIA-BIA_LDM', \n              'BIA-BIA_SMM', 'PAQ_C-Season', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', \n              'PreInt_EduHx-computerinternet_hoursday', 'CGAS-CGAS_Score_Age_sex_count', \n              'Basic_Demos-SexCGAS-CGAS_Score_mean', 'Basic_Demos-SexCGAS-CGAS_Score_std', \n              'Basic_Demos-SexCGAS-CGAS_Score_max', 'Basic_Demos-AgePreInt_EduHx-computerinternet_hoursday_std', \n              'Basic_Demos-Age_Basic_Demos-SexPreInt_EduHx-computerinternet_hoursday_mean', \n              'Basic_Demos-AgePhysical-Weight_std', 'Basic_Demos-Age_Basic_Demos-SexPhysical-Weight_mean', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Weight_median', 'Basic_Demos-Age_Basic_Demos-SexPhysical-Weight_std',\n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Weight_max', 'Basic_Demos-AgePhysical-BMI_min', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-BMI_mean', 'Basic_Demos-Age_Basic_Demos-SexPhysical-BMI_median', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-BMI_std', 'Basic_Demos-AgePhysical-Height_median', \n              'Basic_Demos-AgePhysical-Height_std', 'Basic_Demos-AgePhysical-Height_max', \n              'Basic_Demos-AgePhysical-Height_min', 'Basic_Demos-Age_Basic_Demos-SexPhysical-Height_mean', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Height_median', 'Basic_Demos-Age_Basic_Demos-SexPhysical-Height_max',\n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Height_min', 'Basic_Demos-AgePhysical-Waist_Circumference_median', \n              'Basic_Demos-AgePhysical-Waist_Circumference_std', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Waist_Circumference_mean', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Waist_Circumference_median', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Waist_Circumference_min', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Diastolic_BP_max', \n              'Basic_Demos-AgePhysical-HeartRate_mean', 'Basic_Demos-AgePhysical-HeartRate_median', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-HeartRate_mean', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-HeartRate_median', \n              'Basic_Demos-AgePhysical-Systolic_BP_mean', 'Basic_Demos-Age_Basic_Demos-SexPhysical-Systolic_BP_mean', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Systolic_BP_median', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Systolic_BP_max', 'Internet_Hours_Age', \n              'BMI_Internet_Hours', 'SDS_T_Age']\n\nfeatures90 = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-CGAS_Score', \n              'Physical-Season', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', \n              'Physical-Waist_Circumference', 'Physical-HeartRate', 'Physical-Systolic_BP', \n              'Fitness_Endurance-Season', 'FGC-FGC_CU', 'FGC-FGC_GSD', 'FGC-FGC_PU', 'FGC-FGC_SRR', \n              'FGC-FGC_TL', 'BIA-Season', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_FFMI', 'BIA-BIA_LDM', \n              'BIA-BIA_SMM', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', \n              'PreInt_EduHx-computerinternet_hoursday', 'CGAS-CGAS_Score_Age_sex_count', \n              'Basic_Demos-AgeCGAS-CGAS_Score_mean_diff', 'Basic_Demos-AgeCGAS-CGAS_Score_std', \n              'Basic_Demos-SexCGAS-CGAS_Score_mean', 'Basic_Demos-SexCGAS-CGAS_Score_mean_diff', \n              'Basic_Demos-AgePreInt_EduHx-computerinternet_hoursday_mean', \n              'Basic_Demos-Age_Basic_Demos-SexPreInt_EduHx-computerinternet_hoursday_mean', \n              'Basic_Demos-AgePhysical-Weight_std', 'Basic_Demos-Age_Basic_Demos-SexPhysical-Weight_mean',\n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Weight_median', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Weight_std', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Weight_max', 'Basic_Demos-AgePhysical-BMI_min', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-BMI_mean', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-BMI_median', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-BMI_std', 'Basic_Demos-AgePhysical-Height_std', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Height_max', \n              'Basic_Demos-AgePhysical-Waist_Circumference_std', \n              'Basic_Demos-AgePhysical-Waist_Circumference_max', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Waist_Circumference_mean', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Waist_Circumference_median', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Waist_Circumference_max', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Waist_Circumference_min', \n              'Basic_Demos-AgePhysical-Diastolic_BP_median', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Diastolic_BP_max', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Diastolic_BP_min', \n              'Basic_Demos-AgePhysical-HeartRate_mean', 'Basic_Demos-AgePhysical-HeartRate_median', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-HeartRate_mean', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-HeartRate_median', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Systolic_BP_mean', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Systolic_BP_median', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Systolic_BP_max', \n              'Basic_Demos-AgeSDS-SDS_Total_T_mean', 'Basic_Demos-AgeFGC-FGC_CU_median', \n              'Basic_Demos-AgeFGC-FGC_CU_std', 'Basic_Demos-Age_Basic_Demos-SexFGC-FGC_CU_mean', \n              'Basic_Demos-Age_Basic_Demos-SexFGC-FGC_CU_median', \n              'Basic_Demos-Age_Basic_Demos-SexFGC-FGC_CU_std', \n              'Basic_Demos-AgeFGC-FGC_PU_median', 'Basic_Demos-Age_Basic_Demos-SexFGC-FGC_PU_mean', \n              'Basic_Demos-Age_Basic_Demos-SexFGC-FGC_PU_median', \n              'Basic_Demos-Age_Basic_Demos-SexFGC-FGC_PU_std', \n              'Basic_Demos-Age_Basic_Demos-SexBIA-BIA_BMI_mean', 'BMI_Age', \n              'Internet_Hours_Age', 'BMI_Internet_Hours', 'SMM_Height', 'Muscle_to_Fat', \n              'SDS_T_Age', 'Mean_Arterial_Pressure', 'stat_8', 'stat_10', 'stat_12', 'stat_57', \n              'stat_65', 'stat_73', 'stat_77']\n\nfeatures80 = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-Season', \n              'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI', 'Physical-Height', 'Physical-Weight',\n              'Physical-Waist_Circumference', 'Physical-HeartRate', 'Physical-Systolic_BP', \n              'Fitness_Endurance-Season', 'FGC-FGC_CU', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', \n              'FGC-FGC_GSD', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRR', \n              'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'BIA-Season', 'BIA-BIA_Activity_Level_num', \n              'BIA-BIA_DEE', 'BIA-BIA_FFMI', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', \n              'PAQ_A-Season', 'PAQ_C-Season', 'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw', \n              'SDS-SDS_Total_T', 'PreInt_EduHx-Season', 'PreInt_EduHx-computerinternet_hoursday', \n              'Wrist-wkend_enmoXlight', 'CGAS-CGAS_Score_Age_sex_count', \n              'Basic_Demos-AgeCGAS-CGAS_Score_mean_diff', 'Basic_Demos-AgeCGAS-CGAS_Score_std', \n              'Basic_Demos-SexCGAS-CGAS_Score_mean', 'Basic_Demos-SexCGAS-CGAS_Score_mean_diff',\n              'Basic_Demos-AgePreInt_EduHx-computerinternet_hoursday_mean', \n              'Basic_Demos-AgePreInt_EduHx-computerinternet_hoursday_std', \n              'Basic_Demos-Age_Basic_Demos-SexPreInt_EduHx-computerinternet_hoursday_mean', \n              'Basic_Demos-AgePhysical-Weight_std', 'Basic_Demos-Age_Basic_Demos-SexPhysical-Weight_mean',\n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Weight_median', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Weight_std', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Weight_max', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Weight_min', \n              'Basic_Demos-AgePhysical-BMI_mean', 'Basic_Demos-AgePhysical-BMI_min', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-BMI_mean', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-BMI_median', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-BMI_std', \n              'Basic_Demos-AgePhysical-Height_std', 'Basic_Demos-Age_Basic_Demos-SexPhysical-Height_max',\n              'Basic_Demos-AgePhysical-Waist_Circumference_std', \n              'Basic_Demos-AgePhysical-Waist_Circumference_max', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Waist_Circumference_mean', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Waist_Circumference_median', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Waist_Circumference_max', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Waist_Circumference_min', \n              'Basic_Demos-AgePhysical-Diastolic_BP_median', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Diastolic_BP_max', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Diastolic_BP_min',\n              'Basic_Demos-AgePhysical-HeartRate_mean', 'Basic_Demos-AgePhysical-HeartRate_median', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-HeartRate_mean',\n              'Basic_Demos-Age_Basic_Demos-SexPhysical-HeartRate_median', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-HeartRate_min', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Systolic_BP_mean', \n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Systolic_BP_median',\n              'Basic_Demos-Age_Basic_Demos-SexPhysical-Systolic_BP_max', \n              'Basic_Demos-AgeSDS-SDS_Total_T_mean', 'Basic_Demos-AgeFGC-FGC_CU_median', \n              'Basic_Demos-AgeFGC-FGC_CU_std', 'Basic_Demos-AgeFGC-FGC_CU_max', \n              'Basic_Demos-Age_Basic_Demos-SexFGC-FGC_CU_mean', \n              'Basic_Demos-Age_Basic_Demos-SexFGC-FGC_CU_median', \n              'Basic_Demos-Age_Basic_Demos-SexFGC-FGC_CU_std', \n              'Basic_Demos-Age_Basic_Demos-SexFGC-FGC_CU_max', \n              'Basic_Demos-AgeFGC-FGC_PU_median', 'Basic_Demos-AgeFGC-FGC_PU_max', \n              'Basic_Demos-Age_Basic_Demos-SexFGC-FGC_PU_mean', \n              'Basic_Demos-Age_Basic_Demos-SexFGC-FGC_PU_median', \n              'Basic_Demos-Age_Basic_Demos-SexFGC-FGC_PU_std', \n              'Basic_Demos-Age_Basic_Demos-SexBIA-BIA_BMI_mean', \n              'Basic_Demos-Age_Basic_Demos-SexBIA-BIA_BMI_median', \n              'Basic_Demos-Age_Basic_Demos-SexBIA-BIA_BMI_std', 'BMI_Age', \n              'Internet_Hours_Age', 'BMI_Internet_Hours', 'BFP_BMI', 'SMM_Height', \n              'Muscle_to_Fat', 'SDS_T_Age', 'PAQ_C_Age', 'Pulse_Pressure', 'Mean_Arterial_Pressure', \n              'Waist-Height', 'stat_8', 'stat_10', 'stat_12', 'stat_41', 'stat_57', 'stat_60', \n              'stat_65', 'stat_73', 'stat_77', 'stat_84']\nff = [\"Wrist-idle_mode\",\"Wrist-battery_days\",\n        \"Wrist-pct_signal\",\"Wrist-ave_PCIAT_date\",\n        \"Wrist-bedtime\",\"Wrist-sleep_hours\",\n        \"Wrist-pct_active\",\n        \"Wrist-enmoMLSmean\",\"Wrist-enmoMLSstd\",\n        \"Wrist-lightLMSmean\",\"Wrist-lightLMSstd\",\n        \"Wrist-enmoXlight\", \"Wrist-wkend_enmoXlight\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T08:08:31.736659Z","iopub.execute_input":"2024-12-17T08:08:31.736958Z","iopub.status.idle":"2024-12-17T08:08:31.751528Z","shell.execute_reply.started":"2024-12-17T08:08:31.736909Z","shell.execute_reply":"2024-12-17T08:08:31.750774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def imputerDf(df1, df2, numerical_columns, threshold ,if_imputer, mapping):\n    # 筛选出缺失值占比低于 threshold 的列\n    if threshold >= 1:\n        low_nan_cols = [col for col in df1.columns if col!='sii']\n    else:\n        low_nan_cols = [col for col in df1.columns if df1[col].isnull().mean() < threshold and col!='sii']\n    \n    # 选择这些列\n    X = df1[low_nan_cols]\n    X_test = df2[low_nan_cols]\n    y_sii = df1['sii'].values\n    \n    cat_cols = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n                'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n    cat_cols = [i for i in cat_cols if i in low_nan_cols]\n    def updateCats(df):\n        for c in cat_cols: \n            df[c] = df[c].fillna('Missing')\n            df[c] = df[c].astype('category')\n        return df\n\n    num_features = [i for i in low_nan_cols if i in numerical_columns]\n\n    X = updateCats(X)\n    X_test  = updateCats(X_test) \n\n    if if_imputer:\n        # Preprocessing numerical\n        imputer = SimpleImputer(strategy=\"median\")\n        imputer = IterativeImputer(max_iter=19, random_state=0)\n        # imputer = SimpleImputer(strategy=\"constant\",fill_value = -1)\n        X[num_features] = imputer.fit_transform(X[num_features])\n        X_test[num_features] = imputer.transform(X_test[num_features])\n    \n    scaler = MinMaxScaler()\n    X[num_features] = scaler.fit_transform(X[num_features])\n    X_test[num_features] = scaler.transform(X_test[num_features])\n    \n    if mapping:\n        for col in cat_cols:\n            mapping = create_mapping(col, X)\n            # mappingTe = create_mapping(col, test)\n            \n            X[col] = X[col].replace(mapping).astype(int)\n            X_test[col] = X_test[col].replace(mapping).astype(int)\n    \n    scaler = StandardScaler()\n    \n    X[num_features] = scaler.fit_transform(X[num_features])\n    X_test[num_features] = scaler.transform(X_test[num_features])\n    return X, X_test, y_sii","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T08:08:31.753718Z","iopub.execute_input":"2024-12-17T08:08:31.754133Z","iopub.status.idle":"2024-12-17T08:08:31.768303Z","shell.execute_reply.started":"2024-12-17T08:08:31.754094Z","shell.execute_reply":"2024-12-17T08:08:31.767583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_impute, X_test_impute, y_sii = imputerDf(train, test, numerical_columns, threshold=0.3 ,if_imputer=True, mapping=False)\n\nX, X_test, y_sii = imputerDf(train, test, numerical_columns, threshold=1.0 ,if_imputer=False, mapping=True)\n\nfeatures_impute = [i for i in features if i in X_impute.columns]\nfeatures90_impute = [i for i in features90 if i in X_impute.columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T08:08:31.769257Z","iopub.execute_input":"2024-12-17T08:08:31.769566Z","iopub.status.idle":"2024-12-17T08:09:47.415488Z","shell.execute_reply.started":"2024-12-17T08:08:31.769541Z","shell.execute_reply":"2024-12-17T08:09:47.414356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LR_START = 1e-3\nLR_MAX = 5e-3\nLR_MIN = 1e-3\nLR_RAMPUP_EPOCHS = 0\nLR_SUSTAIN_EPOCHS = 0\nEPOCHS = 10\n\ndef lrfn(epoch):\n    if epoch < LR_RAMPUP_EPOCHS:\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        lr = LR_MAX\n    else:\n        decay_total_epochs = EPOCHS - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS - 1\n        decay_epoch_index = epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS\n        phase = math.pi * decay_epoch_index / decay_total_epochs\n        cosine_decay = 0.5 * (1 + math.cos(phase))\n        lr = (LR_MAX - LR_MIN) * cosine_decay + LR_MIN    \n    return lr\n\nrng = [i for i in range(EPOCHS)]\nlr_y = [lrfn(x) for x in rng]\nLR_Scheduler = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=True)\n\nclass CFG:\n    folds = 5\n    epochs = 10\n    batch_size = 32\n    LR_Scheduler = [LR_Scheduler]\n    optimizer = Adam(learning_rate=1e-3)\n\n    conf = ModelConfig(auto_imputation=True,\n                       auto_discrete=False,\n                       categorical_columns='auto',\n                       fixed_embedding_dim=True,\n                       embeddings_output_dim=4,\n                       embedding_dropout=0.3,\n                       nets=deepnets.DeepFM,\n                       # nets =['linear','cin_nets','dnn_nets'],\n                       dnn_params={\n                           'hidden_units': ((512, 0.5, True),\n                                            (256, 0.5, True),\n                                            (128, 0.3, True),\n                                            (64, 0.2, True)),\n                           'dnn_activation': 'relu',\n                           },\n                       stacking_op='concat',\n                       # stacking_op = 'add',\n                       output_use_bias=False,\n                       optimizer=optimizer,\n                       task='regression',\n                       loss='auto',\n                       metrics=[\"RootMeanSquaredError\"],\n                       earlystopping_patience=1,\n                       )","metadata":{"trusted":true,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2024-12-17T08:09:47.417902Z","iopub.execute_input":"2024-12-17T08:09:47.418770Z","iopub.status.idle":"2024-12-17T08:09:47.428849Z","shell.execute_reply.started":"2024-12-17T08:09:47.418721Z","shell.execute_reply":"2024-12-17T08:09:47.427885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_deep(CFG, y_sii, X, test_data):\n    skf = StratifiedKFold(n_splits=CFG.folds, shuffle=True, random_state=42)\n    oof_non_rounded = np.zeros(len(y_sii), dtype=float) \n    valid_scores = []\n    models = []\n    test_preds = np.zeros((len(test_data), CFG.folds))\n    for fi, (train_idx, valid_idx) in enumerate(skf.split(X, y_sii)):\n        print(\"#\"*25)\n        print(f\"### Fold {fi+1}/{CFG.folds} ...\")\n        print(\"#\"*25)\n\n        os.makedirs(f\"/kaggle/working/models/fold{fi}\", exist_ok=True)\n        os.makedirs(f\"/tmp/workdir/kaggle/working/models/fold{fi}\", exist_ok=True)\n\n        X_train, y_train = X.iloc[train_idx], y_sii[train_idx]\n        X_valid, y_valid = X.iloc[valid_idx], y_sii[valid_idx]\n\n        K.clear_session()\n        model = DeepTable(config=CFG.conf)\n        checkpoint = ModelCheckpoint(\n            filepath=f'./fold{fi}.h5',  \n            monitor='val_loss',        \n            mode='min',                \n            save_best_only=True,       \n            verbose=1                  \n        )\n        model.fit(X_train, y_train,\n                validation_data=(X_valid, y_valid),\n                callbacks=[CFG.LR_Scheduler],\n                batch_size=CFG.batch_size, epochs=CFG.epochs, \n                verbose=0,\n                    )\n        # from deeptables.models.layers import MultiColumnEmbedding, FM\n        # model = load_model(f'./fold{fi}.h5', \n        #                    custom_objects = {\"MultiColumnEmbedding\":MultiColumnEmbedding,\n        #                                     \"FM\":FM})\n        model.save(f'/kaggle/working/models/fold{fi}')\n        os.system(f'cp -r /kaggle/working/models/fold{fi}/* /tmp/workdir/kaggle/working/models/fold{fi}/')\n        # model.keras_model.load_weights(f'./fold{fi}.h5')\n        # model = DeepTable.load('/kaggle/working/', 'fold{fi}.h5')#load_model(f'./fold{fi}.h5')\n        models.append(model)\n        # Avoid some errors\n        with K.name_scope(CFG.optimizer.__class__.__name__):\n            for j, var in enumerate(CFG.optimizer.weights):\n                name = 'variable{}'.format(j)\n                CFG.optimizer.weights[j] = tf.Variable(var, name=name)\n        CFG.conf = CFG.conf._replace(optimizer=CFG.optimizer)\n        y_valid_pred = model.predict(X_valid, verbose=1, batch_size=512).flatten()\n        oof_non_rounded[valid_idx] = y_valid_pred\n        y_valid_pred_rounded = y_valid_pred.round(0).astype(int)\n\n        test_preds[:, fi] = model.predict(test_data).reshape(-1)\n\n        valid_kappa = quadratic_weighted_kappa(y_valid, y_valid_pred_rounded)\n        valid_scores.append(valid_kappa)\n        print(f\"\\n{Fore.GREEN}{Style.BRIGHT}Fold {fi+1} | valid qwk: {valid_kappa:.4f}\\n\")\n    print(f\"\\n{Fore.BLUE}{Style.BRIGHT}OOF mean valid qwk ---> {np.mean(valid_scores):.4f}\\n\")\n\n    kappa_optimizer = minimize(\n        evaluate_predictions,\n        x0=[0.5, 1.5, 2.5],\n        args=(y_sii, oof_non_rounded),\n        method=\"Nelder-Mead\",\n    )\n    assert kappa_optimizer.success, \"Optimization did not converge.\"\n    best_thresholds = kappa_optimizer.x\n    print(\"Best thresholds:\", best_thresholds)\n    oof_tuned = threshold_Rounder(oof_non_rounded, best_thresholds)\n    tuned_kappa = quadratic_weighted_kappa(y_sii, oof_tuned)\n    print(f\"\\n----> || Optimized qwk: {Fore.BLUE}{Style.BRIGHT}{tuned_kappa:.4f}\\n\")\n    kappa_score, best_thresholds_new = get_optimized_kappa_score_global(oof_non_rounded, y_sii)\n    print(\"New Best thresholds:\", best_thresholds_new)\n    print(f\"\\n----> ||New Optimized qwk: {Fore.BLUE}{Style.BRIGHT}{kappa_score:.4f}\\n\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, best_thresholds_new)\n    \n    return tpm, oof_non_rounded, best_thresholds_new","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T08:11:21.168899Z","iopub.execute_input":"2024-12-17T08:11:21.169732Z","iopub.status.idle":"2024-12-17T08:11:21.181629Z","shell.execute_reply.started":"2024-12-17T08:11:21.169698Z","shell.execute_reply":"2024-12-17T08:11:21.180732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"deep_tmp1, impute_oof_res1 , best_thresholds = train_deep(CFG, y_sii, X_impute[features_impute], X_test_impute[features_impute])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T08:12:38.734690Z","iopub.execute_input":"2024-12-17T08:12:38.735084Z","iopub.status.idle":"2024-12-17T08:13:17.411497Z","shell.execute_reply.started":"2024-12-17T08:12:38.735049Z","shell.execute_reply":"2024-12-17T08:13:17.410453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 1024\nLGB_Params =  {'bagging_fraction': 0.10020936844855222, \n          'bagging_freq': 5, \n          'colsample_bynode': 0.6422622659705792, \n          'colsample_bytree': 0.33969056304653567, \n          'lambda_l1': 0.17858570227702397, \n          'lambda_l2': 9.797526622837422, \n          'learning_rate': 0.028603937677519488, \n          'max_depth': 5, \n          'min_data_in_leaf': 40, \n          'n_estimators': 280, \n          'num_leaves': 32}\n\nLGB_Params2 = {'bagging_fraction': 0.41789344973183123, \n        'bagging_freq': 2, \n        'colsample_bynode': 0.8791911255399735, \n        'colsample_bytree': 0.8522150275529529, \n        'lambda_l1': 0.9877103349738448, \n        'lambda_l2': 4.047718818185633, \n        'learning_rate': 0.05038484526006823,\n        'max_depth': 8, \n        'min_data_in_leaf': 74, \n        'n_estimators': 120, \n        'num_leaves': 32\n        }\n\n# XGBoost parameters\nXGB_Params = {'colsample_bytree': 0.7357932429633182, \n 'learning_rate': 0.0203085464429225, \n 'max_depth': 3, \n 'n_estimators': 220, \n 'reg_alpha': 0.7555093840035503, \n 'reg_lambda': 10.4629379082598666, \n 'subsample': 0.9501364590835391\n}\n\n\n\nCatBoost_Params = {\n    'learning_rate': 0.021977221360647805,\n    'depth': 5,\n    # 'use_best_model': True,\n    'iterations': 280,\n    'random_seed': SEED,\n    \"loss_function\":'RMSE',\n    'eval_metric': 'RMSE',\n    \"min_data_in_leaf\":10,\n    # 'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 8.28129485921427  # Increase this value\n}\n\n\n# Create model instances\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n# Create model instances\nLight = LGBMRegressor(**LGB_Params, random_state=SEED, verbose=-1)\nLight2 = LGBMRegressor(**LGB_Params2, random_state=SEED, verbose=-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T08:10:20.630125Z","iopub.execute_input":"2024-12-17T08:10:20.630481Z","iopub.status.idle":"2024-12-17T08:10:20.638551Z","shell.execute_reply.started":"2024-12-17T08:10:20.630449Z","shell.execute_reply":"2024-12-17T08:10:20.637570Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"voting_model = VotingRegressor(estimators=[\n    ('lightgbm2', Light2),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ],\n    weights = [5, 5, 5]\n)\n\nvoting_model2 = VotingRegressor(estimators=[\n    ('lightgbm2', Light2),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ],\n    weights = [5, 4, 3]\n)\n\nvoting_model3 = VotingRegressor(estimators=[\n    ('lightgbm2', Light2),\n     ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ],\n    weights = [3, 2, 3, 2]\n)\n\nml_tmp1, oof_res1 , best_thresholds = TrainML(voting_model, 5, X, X_test, y_sii, features)\nml_tmp2, oof_res2 , best_thresholds = TrainML(voting_model3, 5, X, X_test, y_sii, features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T08:10:21.974878Z","iopub.execute_input":"2024-12-17T08:10:21.975247Z","iopub.status.idle":"2024-12-17T08:10:37.926323Z","shell.execute_reply.started":"2024-12-17T08:10:21.975217Z","shell.execute_reply":"2024-12-17T08:10:37.925513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ml_tmp3, oof_res3 , best_thresholds = TrainML(voting_model2, 5, X, X_test, y_sii, features80)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T08:10:37.928033Z","iopub.execute_input":"2024-12-17T08:10:37.928325Z","iopub.status.idle":"2024-12-17T08:10:49.303540Z","shell.execute_reply.started":"2024-12-17T08:10:37.928297Z","shell.execute_reply":"2024-12-17T08:10:49.302803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nml_tmp4, oof_res4 , best_thresholds = TrainML(voting_model3, 5, X, X_test, y_sii, features90)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T08:10:49.304597Z","iopub.execute_input":"2024-12-17T08:10:49.304865Z","iopub.status.idle":"2024-12-17T08:10:59.834273Z","shell.execute_reply.started":"2024-12-17T08:10:49.304838Z","shell.execute_reply":"2024-12-17T08:10:59.833535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# deep_tmp1, impute_oof_res1\n\noof_res = impute_oof_res1 * 0.1 + oof_res1 * 0.26 + oof_res2 * 0.13 + oof_res3 * 0.3 + oof_res4 * 0.21\nkappa, coefficients = get_optimized_kappa_score_global(oof_res, y_sii)\nprint(f\"ensemble kappa {kappa}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T08:14:28.156895Z","iopub.execute_input":"2024-12-17T08:14:28.157770Z","iopub.status.idle":"2024-12-17T08:14:30.063619Z","shell.execute_reply.started":"2024-12-17T08:14:28.157734Z","shell.execute_reply":"2024-12-17T08:14:30.062755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# res = []\n# for i1 in np.arange(0.16, 0.36, 0.04):\n#     for i2 in np.arange(0.1, 0.4, 0.02):\n#         for i3 in np.arange(0.1, 0.4, 0.02):\n#             for i4 in np.arange(0.1, 0.4, 0.02):\n#                 for i5 in np.arange(0.1, 0.4, 0.02):\n#                     if i1 + i2 + i3 + i4 + i5 == 1.0:\n#                         oof_res = impute_oof_res1 * i1 + oof_res1 * i2 + oof_res2 * i3 + oof_res3 * i4 + oof_res4 * i5\n#                         kappa, coefficients = get_optimized_kappa_score_global(oof_res, y_sii)\n#                         res.append((kappa, [i1, i2, i3, i4,i5]))\n#                         print((kappa, [i1 , i2 , i3 , i4 , i5]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T08:05:40.220365Z","iopub.execute_input":"2024-12-17T08:05:40.220647Z","iopub.status.idle":"2024-12-17T08:05:40.224961Z","shell.execute_reply.started":"2024-12-17T08:05:40.220620Z","shell.execute_reply":"2024-12-17T08:05:40.223993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tpm_res = deep_tmp1 * 0.1 + ml_tmp1 * 0.26 + ml_tmp2 * 0.13 + ml_tmp3 * 0.3 + ml_tmp4 * 0.21\ntest_pred_rounded = threshold_Rounder(tpm_res, best_thresholds)\n\nsubmission= pd.DataFrame({\n    'id': sample['id'],\n    'sii': test_pred_rounded \n})\n\n# submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T08:14:14.824248Z","iopub.execute_input":"2024-12-17T08:14:14.824855Z","iopub.status.idle":"2024-12-17T08:14:14.830158Z","shell.execute_reply.started":"2024-12-17T08:14:14.824819Z","shell.execute_reply":"2024-12-17T08:14:14.829218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def TrainStack(model_class, n_splits, train, test_data, y_sii):\n    sum_coefficients = 0\n    X = train#.drop(['sii'], axis=1)\n    y = y_sii#train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n    \n    imp_df = pd.DataFrame()\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X[train_idx], X[test_idx]\n        y_train, y_val = y[train_idx], y[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n\n        test_preds[:, fold] = model.predict(test_data) # model.predict(test_data)\n        \n\n        valid_kappa, valid_coefficients = get_optimized_kappa_score(y_val_pred, y_val)\n        train_kappa, train_coefficients = get_optimized_kappa_score(y_train_pred, y_train)\n\n        train_S.append(train_kappa)\n        test_S.append(valid_kappa)\n\n        # imp_df[\"feature\"] = list(features)\n        # imp_df[\"importance_gain\"] = model.feature_importances_\n\n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f} - Valid QWK: {valid_kappa:.4f}, valid_coefficients: {valid_coefficients}\")\n        clear_output(wait=True)\n        sum_coefficients += np.array(valid_coefficients)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}, {test_S}\")\n    \n    kappa, coefficients = get_optimized_kappa_score(oof_non_rounded, y)\n    print(f\"----> || Opitmizer Rounder SCORE :: {kappa}, {coefficients},{sum_coefficients/5}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, coefficients)\n    \n    # submission = pd.DataFrame({\n    #     'id': sample['id'],\n    #     'sii': tpTuned\n    # })\n\n    return tpm, oof_non_rounded, coefficients","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T08:05:40.237491Z","iopub.execute_input":"2024-12-17T08:05:40.238242Z","iopub.status.idle":"2024-12-17T08:05:40.249165Z","shell.execute_reply.started":"2024-12-17T08:05:40.238196Z","shell.execute_reply":"2024-12-17T08:05:40.248448Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression, Ridge\nfrom sklearn.ensemble import BaggingClassifier, BaggingRegressor\n\ntrain_stack = np.concatenate([impute_oof_res1.reshape(-1, 1), oof_res1.reshape(-1, 1), oof_res2.reshape(-1, 1) ,\\\n                oof_res3.reshape(-1, 1) , oof_res4.reshape(-1, 1)], axis = 1)\n\ntest_stack = np.concatenate([deep_tmp1.reshape(-1, 1), ml_tmp1.reshape(-1, 1), ml_tmp2.reshape(-1, 1) ,\\\n                ml_tmp3.reshape(-1, 1) , ml_tmp4.reshape(-1, 1)], axis = 1)\n\nridge = Ridge(random_state=SEED)\nbagging = BaggingRegressor(estimator=ridge, n_estimators=50, n_jobs= -1, random_state=42)\n\ntpm_stack, oof_stack, coefficients = TrainStack(bagging, 5, train_stack, test_stack, y_sii)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T08:05:40.250232Z","iopub.execute_input":"2024-12-17T08:05:40.251032Z","iopub.status.idle":"2024-12-17T08:05:44.154905Z","shell.execute_reply.started":"2024-12-17T08:05:40.250984Z","shell.execute_reply":"2024-12-17T08:05:44.153890Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kappa, coefficients = get_optimized_kappa_score_global(oof_stack, y_sii)\nprint(f\"ensemble kappa {kappa}\")\n\ntpm_res = tpm_stack\ntest_pred_rounded = threshold_Rounder(tpm_res, best_thresholds)\n\nsubmission= pd.DataFrame({\n    'id': sample['id'],\n    'sii': test_pred_rounded \n})\n\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T08:05:44.156620Z","iopub.execute_input":"2024-12-17T08:05:44.157029Z","iopub.status.idle":"2024-12-17T08:05:45.943888Z","shell.execute_reply.started":"2024-12-17T08:05:44.156984Z","shell.execute_reply":"2024-12-17T08:05:45.943102Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}