{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":135.706304,"end_time":"2024-12-19T21:03:55.705420","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-12-19T21:01:39.999116","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"b8891f9c","cell_type":"code","source":"# import lib\nimport numpy as np\nimport pandas as pd\nimport os\nimport re\n\nfrom colorama import Fore, Style\nfrom sklearn.base import clone, BaseEstimator, RegressorMixin\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score, mean_squared_error\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\nfrom sklearn.decomposition import PCA\nfrom sklearn.datasets import make_classification\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\n\nimport random","metadata":{"execution":{"iopub.status.busy":"2024-12-19T22:32:33.443512Z","iopub.execute_input":"2024-12-19T22:32:33.443761Z","iopub.status.idle":"2024-12-19T22:32:54.401494Z","shell.execute_reply.started":"2024-12-19T22:32:33.443736Z","shell.execute_reply":"2024-12-19T22:32:54.400339Z"},"papermill":{"duration":18.326293,"end_time":"2024-12-19T21:02:00.630450","exception":false,"start_time":"2024-12-19T21:01:42.304157","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"16debd43","cell_type":"code","source":"# init + setup model\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\nSEED = 42\nn_splits = 5\n\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\nseed_everything(2024)\n\ntrain_featuresCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW', 'BMI_PHR']\n\ntest_featuresCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW', 'BMI_PHR']\n\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n    'device': 'cpu',\n}\n\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n    'tree_method': 'hist'\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  # Increase this value\n    'task_type': 'CPU',\n}\n\ndef preprocess_feature(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n    \n    return df\n\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.LeakyReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2+10),\n            nn.LeakyReLU(),\n            nn.Linear(encoding_dim*2+10, encoding_dim),\n            nn.LeakyReLU()\n        )\n\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, encoding_dim+15),\n            nn.LeakyReLU(),\n            nn.Linear(encoding_dim+15, encoding_dim*3),\n            nn.LeakyReLU(),\n            nn.Linear(encoding_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n\n\n# encoder du lieu\ndef autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n\n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded","metadata":{"execution":{"iopub.status.busy":"2024-12-19T22:32:54.402669Z","iopub.execute_input":"2024-12-19T22:32:54.403399Z","iopub.status.idle":"2024-12-19T22:32:54.433467Z","shell.execute_reply.started":"2024-12-19T22:32:54.403367Z","shell.execute_reply":"2024-12-19T22:32:54.431993Z"},"papermill":{"duration":0.027763,"end_time":"2024-12-19T21:02:00.661981","exception":false,"start_time":"2024-12-19T21:02:00.634218","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"3493910d","cell_type":"code","source":"# functions\n\n# thong ke parquet file\ndef pre_progress_parquet_data(file_name, parquet_dir):\n    file_path = os.path.join(parquet_dir, file_name, 'part-0.parquet')\n    df = pd.read_parquet(file_path)\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), file_name.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: pre_progress_parquet_data(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-12-19T22:32:54.434820Z","iopub.execute_input":"2024-12-19T22:32:54.435221Z","iopub.status.idle":"2024-12-19T22:32:54.454624Z","shell.execute_reply.started":"2024-12-19T22:32:54.435188Z","shell.execute_reply":"2024-12-19T22:32:54.453277Z"},"papermill":{"duration":0.010967,"end_time":"2024-12-19T21:02:00.676192","exception":false,"start_time":"2024-12-19T21:02:00.665225","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"5b8239a7","cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:32:54.457254Z","iopub.execute_input":"2024-12-19T22:32:54.457620Z","iopub.status.idle":"2024-12-19T22:34:34.943279Z","shell.execute_reply.started":"2024-12-19T22:32:54.457589Z","shell.execute_reply":"2024-12-19T22:34:34.941989Z"}},"outputs":[],"execution_count":null},{"id":"4494abb2","cell_type":"code","source":"import pandas as pd\n\n# nếu mà giá trị PCIAT-PCIAT_Total < 5 và một vài PCIAT-PCIAT còn lại bị missing thì sii  = None vì người trả lời trả lời không chính xác \n\nids_to_update = ['18fdbccc', '053d7d31', '39dd3538', '68fa4631', '6a98537b', '6b9a25e6', '75311a3f', '926bd07e', 'fc8e4de4']\ntrain.loc[train['id'].isin(ids_to_update), 'sii'] = None\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:34:34.945195Z","iopub.execute_input":"2024-12-19T22:34:34.945545Z","iopub.status.idle":"2024-12-19T22:34:34.956968Z","shell.execute_reply.started":"2024-12-19T22:34:34.945513Z","shell.execute_reply":"2024-12-19T22:34:34.955466Z"}},"outputs":[],"execution_count":null},{"id":"da54d033","cell_type":"code","source":"\ntrain.dropna(subset=['sii'], inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:34:34.958245Z","iopub.execute_input":"2024-12-19T22:34:34.958647Z","iopub.status.idle":"2024-12-19T22:34:34.991312Z","shell.execute_reply.started":"2024-12-19T22:34:34.958599Z","shell.execute_reply":"2024-12-19T22:34:34.990007Z"}},"outputs":[],"execution_count":null},{"id":"fb09aa34","cell_type":"code","source":"import pandas as pd\nfrom sklearn.impute import KNNImputer\n\ncolumns_to_impute = [\n    'PCIAT-PCIAT_06', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_15', 'PCIAT-PCIAT_07',\n    'PCIAT-PCIAT_16', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_11',\n    'PCIAT-PCIAT_19', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_14',\n    'PCIAT-PCIAT_04', 'PCIAT-PCIAT_20', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_02',\n    'PCIAT-PCIAT_18', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_05'\n]\n\ndf_non_nan_sii = train[train['sii'].notna()]\n\nknn_imputer = KNNImputer(n_neighbors=10)\n\nimputed_data = knn_imputer.fit_transform(df_non_nan_sii[columns_to_impute])\n\nimputed_data = imputed_data.round()\n\ntrain.loc[train['sii'].notna(), columns_to_impute] = imputed_data\n\ntrain['PCIAT-PCIAT_Total_new'] = train[columns_to_impute].sum(axis=1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:34:34.992585Z","iopub.execute_input":"2024-12-19T22:34:34.993119Z","iopub.status.idle":"2024-12-19T22:34:35.095842Z","shell.execute_reply.started":"2024-12-19T22:34:34.993076Z","shell.execute_reply":"2024-12-19T22:34:35.094355Z"}},"outputs":[],"execution_count":null},{"id":"420bf03b","cell_type":"code","source":"\ndef recalculate_sii(row):\n    if pd.isna(row['PCIAT-PCIAT_Total']):\n        return np.nan\n    if row['PCIAT-PCIAT_Total'] <= 30:\n        return 0\n    elif 31 <= row['PCIAT-PCIAT_Total'] <= 49:\n        return 1\n    elif 50 <= row['PCIAT-PCIAT_Total'] <= 79:\n        return 2\n    elif row['PCIAT-PCIAT_Total'] >= 80:\n        return 3\n    return np.nan\n\n# Thêm cột 'new_sii'\ntrain['new_sii'] = train.apply(recalculate_sii, axis=1)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:34:35.096990Z","iopub.execute_input":"2024-12-19T22:34:35.097508Z","iopub.status.idle":"2024-12-19T22:34:35.186547Z","shell.execute_reply.started":"2024-12-19T22:34:35.097461Z","shell.execute_reply":"2024-12-19T22:34:35.185175Z"}},"outputs":[],"execution_count":null},{"id":"fef7bfe6","cell_type":"code","source":"train['sii'] = train['new_sii']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:34:35.188148Z","iopub.execute_input":"2024-12-19T22:34:35.188573Z","iopub.status.idle":"2024-12-19T22:34:35.196377Z","shell.execute_reply.started":"2024-12-19T22:34:35.188535Z","shell.execute_reply":"2024-12-19T22:34:35.193382Z"}},"outputs":[],"execution_count":null},{"id":"175dee05","cell_type":"code","source":"\n#test_ts = train_ts.head(20)\ndf_train = train_ts.drop('id', axis=1) \ndf_test = test_ts.drop('id', axis=1) # bo cot id\n\nscaler = StandardScaler() # scale du lieu\ndf_scaled = scaler.fit_transform(df_train)\ndata_tensor = torch.FloatTensor(df_scaled)\ndf_scaled2 = scaler.fit_transform(df_test)  \ndata_tensor2 = torch.FloatTensor(df_scaled2)\n\ninput_dim = data_tensor.shape[1]\nautoencoder = AutoEncoder(input_dim, 60)\n\n# encoding ...\ncriterion = nn.MSELoss()\noptimizer = optim.Adam(autoencoder.parameters())\n    \nfor epoch in range(100):\n    for i in range(0, len(data_tensor), 32):\n        batch = data_tensor[i : i + 32]\n        optimizer.zero_grad()\n        reconstructed = autoencoder(batch)\n        loss = criterion(reconstructed, batch)\n        loss.backward()\n        optimizer.step()\n            \n    if (epoch + 1) % 10 == 0:\n        print(f'Epoch [{epoch + 1}/{100}], Loss: {loss.item():.4f}]')\n\nwith torch.no_grad():\n    encoded_data = autoencoder.encoder(data_tensor).numpy()\n    encoded_data2= autoencoder.encoder(data_tensor2).numpy()\n\n# chuyen ve dang dataframe\ntrain_ts_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\ntest_ts_encoded = pd.DataFrame(encoded_data2, columns=[f'Enc_{i + 1}' for i in range(encoded_data2.shape[1])])\n\ntime_series_cols = train_ts_encoded.columns.tolist()\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded['id']=test_ts[\"id\"]\n\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')","metadata":{"execution":{"iopub.status.busy":"2024-12-19T22:34:35.197566Z","iopub.execute_input":"2024-12-19T22:34:35.197886Z","iopub.status.idle":"2024-12-19T22:34:46.532716Z","shell.execute_reply.started":"2024-12-19T22:34:35.197860Z","shell.execute_reply":"2024-12-19T22:34:46.531538Z"},"papermill":{"duration":79.604516,"end_time":"2024-12-19T21:03:20.283811","exception":false,"start_time":"2024-12-19T21:02:00.679295","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"9ce40942","cell_type":"code","source":"cat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n          'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)","metadata":{"execution":{"iopub.status.busy":"2024-12-19T22:34:46.536052Z","iopub.execute_input":"2024-12-19T22:34:46.536766Z","iopub.status.idle":"2024-12-19T22:34:46.571200Z","shell.execute_reply.started":"2024-12-19T22:34:46.536732Z","shell.execute_reply":"2024-12-19T22:34:46.569957Z"},"papermill":{"duration":0.048083,"end_time":"2024-12-19T21:03:20.353055","exception":false,"start_time":"2024-12-19T21:03:20.304972","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"51d84857","cell_type":"code","source":"def create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping_train = create_mapping(col, train)\n    mapping_test = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping_train).astype(int)\n    test[col] = test[col].replace(mapping_test).astype(int)\n\nprint(f'Train Shape : {train.shape} || Test Shape : {test.shape}')","metadata":{"execution":{"iopub.status.busy":"2024-12-19T22:34:46.572575Z","iopub.execute_input":"2024-12-19T22:34:46.572872Z","iopub.status.idle":"2024-12-19T22:34:46.621257Z","shell.execute_reply.started":"2024-12-19T22:34:46.572843Z","shell.execute_reply":"2024-12-19T22:34:46.620148Z"},"papermill":{"duration":0.054298,"end_time":"2024-12-19T21:03:20.424443","exception":false,"start_time":"2024-12-19T21:03:20.370145","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"73c7698b","cell_type":"code","source":"def remove_outliers(df):\n    \n    df = df.drop(df[df['Physical-BMI'] <= 0].index)\n    df = df.drop(df[df['Physical-Diastolic_BP'] <= 0].index)\n    df = df.drop(df[df['Physical-Systolic_BP'] <= 0].index)\n    df = df.drop(df[df['Physical-Diastolic_BP'] > 160].index)\n\n    children = df[df['Basic_Demos-Age'] <= 12]\n    df = df.drop(children[children['FGC-FGC_CU'] > 80].index)\n    df = df.drop(children[children['FGC-FGC_GSND'] > 80].index)\n\n    df = df.drop(df[df['BIA-BIA_BMI'] <= 0].index)\n    df = df.drop(df[df['BIA-BIA_BMC'] > 1000].index)\n    df = df.drop(df[df['BIA-BIA_BMR'] > 40000].index)\n    df = df.drop(df[df['BIA-BIA_DEE'] > 60000].index)\n    df = df.drop(df[df['BIA-BIA_ECW'] > 2000].index)\n    df = df.drop(df[df['BIA-BIA_FFM'] > 2000].index)\n    df = df.drop(df[df['BIA-BIA_ICW'] > 2000].index)\n    df = df.drop(df[df['BIA-BIA_LDM'] > 2000].index)\n    df = df.drop(df[df['BIA-BIA_LST'] > 2000].index)\n    df = df.drop(df[df['BIA-BIA_SMM'] > 2000].index)\n    df = df.drop(df[df['BIA-BIA_TBW'] > 2000].index)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-12-19T22:34:46.622349Z","iopub.execute_input":"2024-12-19T22:34:46.622644Z","iopub.status.idle":"2024-12-19T22:34:46.631744Z","shell.execute_reply.started":"2024-12-19T22:34:46.622620Z","shell.execute_reply":"2024-12-19T22:34:46.630564Z"},"papermill":{"duration":0.025727,"end_time":"2024-12-19T21:03:20.470329","exception":false,"start_time":"2024-12-19T21:03:20.444602","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"8c03519d","cell_type":"code","source":"train = remove_outliers(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T22:34:46.632820Z","iopub.execute_input":"2024-12-19T22:34:46.633147Z","iopub.status.idle":"2024-12-19T22:34:46.695285Z","shell.execute_reply.started":"2024-12-19T22:34:46.633120Z","shell.execute_reply":"2024-12-19T22:34:46.693844Z"}},"outputs":[],"execution_count":null},{"id":"27bf4e20","cell_type":"code","source":"# imputer de fill gia tri nan\nimputer = KNNImputer(n_neighbors=6)\n\nnumeric_cols_train = train.select_dtypes(include=['int32', 'int64', 'float64']).columns\nnumeric_cols_test = test.select_dtypes(include=['int32', 'int64', 'float64']).columns\n\nimputed_train_data = imputer.fit_transform(train[numeric_cols_train])\nimputed_test_data = imputer.fit_transform(test[numeric_cols_test])\n\ntrain_imputed = pd.DataFrame(imputed_train_data, columns=numeric_cols_train)\ntest_imputed = pd.DataFrame(imputed_test_data, columns=numeric_cols_test)\n\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\nfor col in train.columns:\n    if col not in numeric_cols_train:\n        train_imputed[col] = train[col]\n\nfor col in test.columns:\n    if col not in numeric_cols_test:\n        test_imputed[col] = test[col]\n\ntrain = train_imputed\ntest = test_imputed\n\ntrain = preprocess_feature(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = preprocess_feature(test)","metadata":{"execution":{"iopub.status.busy":"2024-12-19T22:34:46.696645Z","iopub.execute_input":"2024-12-19T22:34:46.697095Z","iopub.status.idle":"2024-12-19T22:34:49.902816Z","shell.execute_reply.started":"2024-12-19T22:34:46.697065Z","shell.execute_reply":"2024-12-19T22:34:49.901385Z"},"papermill":{"duration":7.877352,"end_time":"2024-12-19T21:03:28.452917","exception":false,"start_time":"2024-12-19T21:03:20.575565","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"743e0e30-43de-41bc-96fe-bf7995b7f90d","cell_type":"code","source":"# from catboost import CatBoostRegressor, Pool\n# from sklearn.metrics import mean_squared_error\n# from sklearn.model_selection import KFold\n# import optuna\n\n# # Dữ liệu\n# X = data_.drop(columns=['sii'])\n# y = data_['sii']\n\n# def objective_CatBoost(trial):\n#     # Tham số cần tối ưu\n#     params = {\n#         'iterations': trial.suggest_int('iterations', 100, 500),  # Số vòng lặp\n#         'learning_rate': trial.suggest_loguniform('learning_rate', 1e-3, 0.1),  # Tốc độ học\n#         'depth': trial.suggest_int('depth', 4, 10),  # Độ sâu cây\n#         'l2_leaf_reg': trial.suggest_loguniform('l2_leaf_reg', 1e-3, 10.0),  # Regularization L2\n#         'bagging_temperature': trial.suggest_uniform('bagging_temperature', 0.0, 1.0),  # Bagging\n#         'random_strength': trial.suggest_uniform('random_strength', 0.0, 2.0),  # Random strength\n#         'verbose': 0,\n#         'loss_function': 'RMSE',  # Hàm mất mát\n#         'random_seed': 42,\n#     }\n\n#     # K-Fold Cross-Validation\n#     kf = KFold(n_splits=5, shuffle=True, random_state=42)\n#     rmse_scores = []\n\n#     for train_idx, val_idx in kf.split(X):\n#         X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n#         y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n#         # Tạo đối tượng CatBoost Pool\n#         train_pool = Pool(X_train, y_train)\n#         val_pool = Pool(X_val, y_val)\n\n#         # Huấn luyện mô hình\n#         model = CatBoostRegressor(**params)\n#         model.fit(train_pool, eval_set=val_pool, early_stopping_rounds=20)\n\n#         # Dự đoán và tính RMSE\n#         y_val_pred = model.predict(X_val)\n#         rmse = mean_squared_error(y_val, y_val_pred, squared=False)  # RMSE\n#         rmse_scores.append(rmse)\n\n#     # Trả về RMSE trung bình từ K-Fold\n#     return np.mean(rmse_scores)\n\n# # Tối ưu tham số với Optuna\n# study_CatBoost = optuna.create_study(direction='minimize')  # Tối ưu hóa để giảm RMSE\n# study_CatBoost.optimize(objective_CatBoost, n_trials=50)     # Thực hiện 50 lần thử nghiệm\n\n# # In ra tham số và kết quả tốt nhất\n# print(\"Best parameters for CatBoost:\", study_CatBoost.best_params)\n# print(\"Best RMSE:\", study_CatBoost.best_value)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"0b5c7316-3708-4fdc-aca0-4f54f101a896","cell_type":"code","source":"# from xgboost import DMatrix, train\n# from sklearn.metrics import mean_squared_error\n# from sklearn.model_selection import KFold\n# import optuna\n# import numpy as np\n\n# # Chia dữ liệu\n# X = data_.drop(columns=['sii'])  # Dữ liệu đầu vào\n# y = data_['sii']                 # Nhãn\n\n# def objective_XGB(trial):\n#     # Tham số cần tối ưu\n#     params = {\n#         'learning_rate': trial.suggest_loguniform('learning_rate', 1e-3, 0.1),\n#         'max_depth': trial.suggest_int('max_depth', 4, 12),\n#         'subsample': trial.suggest_uniform('subsample', 0.6, 1.0),\n#         'colsample_bytree': trial.suggest_uniform('colsample_bytree', 0.6, 1.0),\n#         'reg_alpha': trial.suggest_loguniform('reg_alpha', 1e-3, 10.0),\n#         'reg_lambda': trial.suggest_loguniform('reg_lambda', 1e-3, 10.0),\n#         'gamma': trial.suggest_loguniform('gamma', 1e-3, 1.0),\n#         'objective': 'reg:squarederror',  # Hồi quy\n#         'tree_method': 'hist',            # Tăng tốc độ huấn luyện\n#         'eval_metric': 'rmse',            # Sử dụng RMSE để đánh giá\n#         'seed': 42\n#     }\n\n#     # Sử dụng K-Fold Cross-Validation\n#     kf = KFold(n_splits=5, shuffle=True, random_state=42)\n#     rmse_scores = []\n\n#     for train__idx, val_idx in kf.split(X):\n#         X_train_, X_val = X.iloc[train__idx], X.iloc[val_idx]\n#         y_train_, y_val = y.iloc[train__idx], y.iloc[val_idx]\n\n#         # Tạo DMatrix cho tập huấn luyện và kiểm tra\n#         dtrain_ = DMatrix(X_train_, label=y_train_)\n#         dval = DMatrix(X_val, label=y_val)\n\n#         # Huấn luyện mô hình với Early Stopping\n#         evals = [(dtrain_, 'train_'), (dval, 'eval')]\n#         booster = train(\n#             params,\n#             dtrain_,\n#             num_boost_round=500,\n#             evals=evals,\n#             early_stopping_rounds=20,  # Dừng sớm nếu không cải thiện\n#             verbose_eval=False\n#         )\n\n#         # Dự đoán trên tập kiểm tra\n#         preds = booster.predict(dval)\n#         rmse = mean_squared_error(y_val, preds, squared=False)  # RMSE\n#         rmse_scores.append(rmse)\n\n#     # Trả về RMSE trung bình từ K-Fold\n#     return np.mean(rmse_scores)\n\n# # Tối ưu tham số với Optuna\n# study_XGB = optuna.create_study(direction='minimize')  # Tối ưu hóa để giảm RMSE\n# study_XGB.optimize(objective_XGB, n_trials=50)         # Thực hiện 50 lần thử nghiệm\n\n# # In ra tham số và kết quả tốt nhất\n# print(\"Best parameters for XGBoost:\", study_XGB.best_params)\n# print(\"Best RMSE:\", study_XGB.best_value)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"47d23efa","cell_type":"code","source":"train_featuresCols += time_series_cols\n\ntrain = train[train_featuresCols]\n# drop cac ban ghi co sii nan\ntrain = train.dropna(subset='sii') \n\ntest_featuresCols += time_series_cols\ntest = test[test_featuresCols]\n\n# con cai nao to qua thi nan\nif np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)\n\nif np.any(np.isinf(test)):\n    test = test.replace([np.inf, -np.inf], np.nan)","metadata":{"execution":{"iopub.status.busy":"2024-12-19T22:34:49.940490Z","iopub.execute_input":"2024-12-19T22:34:49.940876Z","iopub.status.idle":"2024-12-19T22:34:49.980560Z","shell.execute_reply.started":"2024-12-19T22:34:49.940845Z","shell.execute_reply":"2024-12-19T22:34:49.979228Z"},"papermill":{"duration":0.033393,"end_time":"2024-12-19T21:03:28.506530","exception":false,"start_time":"2024-12-19T21:03:28.473137","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"1081677d","cell_type":"code","source":"def eval_models(thresholds, y, y_predict_non_rounded):\n    y_predic = np.where(y_predict_non_rounded < thresholds[0], 0,\n                    np.where(y_predict_non_rounded < thresholds[1], 1,\n                             np.where(y_predict_non_rounded < thresholds[2], 2, 3)))\n    return -cohen_kappa_score(y, y_predic, weights='quadratic')\n\ndef train_models(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        y_train_pred_rounded = y_train_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = cohen_kappa_score(y_train, y_train_pred_rounded, weights='quadratic')        \n        val_kappa = cohen_kappa_score(y_val, y_val_pred_rounded, weights='quadratic')\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n\n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(eval_models,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = np.where(oof_non_rounded < (KappaOPtimizer.x)[0], 0,\n                    np.where(oof_non_rounded < (KappaOPtimizer.x)[1], 1,\n                             np.where(oof_non_rounded < (KappaOPtimizer.x)[2], 2, 3)))\n    \n    tKappa = cohen_kappa_score(y, oof_tuned, weights='quadratic')\n\n    print(f\"----> || Optimized SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = np.where(tpm < (KappaOPtimizer.x)[0], 0,\n                    np.where(tpm < (KappaOPtimizer.x)[1], 1,\n                             np.where(tpm < (KappaOPtimizer.x)[2], 2, 3)))\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"execution":{"iopub.status.busy":"2024-12-19T22:34:49.981766Z","iopub.execute_input":"2024-12-19T22:34:49.982294Z","iopub.status.idle":"2024-12-19T22:34:49.998255Z","shell.execute_reply.started":"2024-12-19T22:34:49.982205Z","shell.execute_reply":"2024-12-19T22:34:49.996595Z"},"papermill":{"duration":0.03042,"end_time":"2024-12-19T21:03:28.596749","exception":false,"start_time":"2024-12-19T21:03:28.566329","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"875ae563","cell_type":"code","source":"import optuna\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model, )\n],weights=[4.604924310158238,2.3338705318246005,5.9112518895527915])","metadata":{"execution":{"iopub.status.busy":"2024-12-19T22:34:49.999555Z","iopub.execute_input":"2024-12-19T22:34:49.999978Z","iopub.status.idle":"2024-12-19T22:34:50.344019Z","shell.execute_reply.started":"2024-12-19T22:34:49.999933Z","shell.execute_reply":"2024-12-19T22:34:50.342713Z"},"papermill":{"duration":0.025801,"end_time":"2024-12-19T21:03:28.639044","exception":false,"start_time":"2024-12-19T21:03:28.613243","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"963e2af3","cell_type":"code","source":"# train\nSubmission1 = train_models(voting_model, test)\n\n# Save submission\nSubmission1.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-12-19T22:34:50.345602Z","iopub.execute_input":"2024-12-19T22:34:50.346288Z","iopub.status.idle":"2024-12-19T22:35:37.627099Z","shell.execute_reply.started":"2024-12-19T22:34:50.346250Z","shell.execute_reply":"2024-12-19T22:35:37.625789Z"},"papermill":{"duration":23.910989,"end_time":"2024-12-19T21:03:52.566487","exception":false,"start_time":"2024-12-19T21:03:28.655498","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"0e18d598","cell_type":"code","source":"Submission1","metadata":{"execution":{"iopub.status.busy":"2024-12-19T22:35:37.628279Z","iopub.execute_input":"2024-12-19T22:35:37.628595Z","iopub.status.idle":"2024-12-19T22:35:37.643778Z","shell.execute_reply.started":"2024-12-19T22:35:37.628565Z","shell.execute_reply":"2024-12-19T22:35:37.642652Z"},"papermill":{"duration":0.049821,"end_time":"2024-12-19T21:03:52.649999","exception":false,"start_time":"2024-12-19T21:03:52.600178","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"fefd5516","cell_type":"code","source":"Submission1['sii'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-12-19T22:35:37.644962Z","iopub.execute_input":"2024-12-19T22:35:37.645379Z","iopub.status.idle":"2024-12-19T22:35:37.657352Z","shell.execute_reply.started":"2024-12-19T22:35:37.645335Z","shell.execute_reply":"2024-12-19T22:35:37.656085Z"},"papermill":{"duration":0.045831,"end_time":"2024-12-19T21:03:52.728735","exception":false,"start_time":"2024-12-19T21:03:52.682904","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null}]}