{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import kagglehub\n\n# Download latest version\npath = kagglehub.dataset_download(\"ryati131457/pytorchtabnet\")\n\nprint(\"Path to dataset files:\", path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:04:54.978012Z","iopub.execute_input":"2024-12-12T09:04:54.982700Z","iopub.status.idle":"2024-12-12T09:04:56.000982Z","shell.execute_reply.started":"2024-12-12T09:04:54.982591Z","shell.execute_reply":"2024-12-12T09:04:55.999628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip -q install /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:04:56.003458Z","iopub.execute_input":"2024-12-12T09:04:56.004143Z","iopub.status.idle":"2024-12-12T09:05:41.111525Z","shell.execute_reply.started":"2024-12-12T09:04:56.004089Z","shell.execute_reply":"2024-12-12T09:05:41.109212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!pip install colorama","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:00:33.819497Z","iopub.execute_input":"2024-12-12T09:00:33.819973Z","iopub.status.idle":"2024-12-12T09:00:33.826737Z","shell.execute_reply.started":"2024-12-12T09:00:33.819926Z","shell.execute_reply":"2024-12-12T09:00:33.825491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!pip install catboost","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:00:33.829219Z","iopub.execute_input":"2024-12-12T09:00:33.829566Z","iopub.status.idle":"2024-12-12T09:00:33.844913Z","shell.execute_reply.started":"2024-12-12T09:00:33.829505Z","shell.execute_reply":"2024-12-12T09:00:33.843849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\nimport plotly.subplots as sp\nimport plotly.express as px\nfrom concurrent.futures import ThreadPoolExecutor\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nfrom IPython.display import display\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:05:41.113655Z","iopub.execute_input":"2024-12-12T09:05:41.114098Z","iopub.status.idle":"2024-12-12T09:05:43.390860Z","shell.execute_reply.started":"2024-12-12T09:05:41.114055Z","shell.execute_reply":"2024-12-12T09:05:43.389856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.base import clone, BaseEstimator, RegressorMixin\nfrom sklearn.ensemble import RandomForestClassifier, RandomForestRegressor\nfrom sklearn.ensemble import StackingRegressor\nfrom sklearn.linear_model import Ridge\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, accuracy_score, precision_score, recall_score, f1_score, roc_curve, auc\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor, HistGradientBoostingRegressor, ExtraTreesRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.model_selection import GridSearchCV","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:05:43.397744Z","iopub.execute_input":"2024-12-12T09:05:43.398082Z","iopub.status.idle":"2024-12-12T09:05:45.735443Z","shell.execute_reply.started":"2024-12-12T09:05:43.398049Z","shell.execute_reply":"2024-12-12T09:05:45.733990Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pytorch_tabnet.tab_model import TabNetRegressor\nfrom pytorch_tabnet.callbacks import Callback\n\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nimport pytorch_tabnet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:05:45.737136Z","iopub.execute_input":"2024-12-12T09:05:45.737829Z","iopub.status.idle":"2024-12-12T09:06:04.859478Z","shell.execute_reply.started":"2024-12-12T09:05:45.737788Z","shell.execute_reply":"2024-12-12T09:06:04.858272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:06:04.861232Z","iopub.execute_input":"2024-12-12T09:06:04.862002Z","iopub.status.idle":"2024-12-12T09:06:04.934629Z","shell.execute_reply.started":"2024-12-12T09:06:04.861962Z","shell.execute_reply":"2024-12-12T09:06:04.933266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"conflict_rows = train[(train['PAQ_A-PAQ_A_Total'].notna()) & (train['PAQ_C-PAQ_C_Total'].notna())]\n\n\n\n# 判斷是否存在衝突行\n\nif not conflict_rows.empty:\n\n    train = train.drop(conflict_rows.index)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:06:04.936036Z","iopub.execute_input":"2024-12-12T09:06:04.936405Z","iopub.status.idle":"2024-12-12T09:06:04.956949Z","shell.execute_reply.started":"2024-12-12T09:06:04.936363Z","shell.execute_reply":"2024-12-12T09:06:04.955333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['PAQ_A-PAQ_A_Total'] = train['PAQ_A-PAQ_A_Total'].fillna(train['PAQ_C-PAQ_C_Total'])\n\ntrain['PAQ_A-Season'] = train['PAQ_A-Season'].fillna(train['PAQ_C-Season'])\n\ntest['PAQ_A-PAQ_A_Total'] = test['PAQ_A-PAQ_A_Total'].fillna(test['PAQ_C-PAQ_C_Total'])\n\ntest['PAQ_A-Season'] = test['PAQ_A-Season'].fillna(test['PAQ_C-Season'])\n\n\n\n# 刪除 column2\n\ntrain = train.drop(columns=['PAQ_C-PAQ_C_Total', 'PAQ_C-Season'])\n\ntest = test.drop(columns=['PAQ_C-PAQ_C_Total', 'PAQ_C-Season'])\n\n\n\ntrain = train.rename(columns={'PAQ_A-Season': 'PAQ-Season'})\n\ntrain = train.rename(columns={'PAQ_A-PAQ_A_Total': 'PAQ-PAQ_Total'})\n\ntest = test.rename(columns={'PAQ_A-Season': 'PAQ-Season'})\n\ntest = test.rename(columns={'PAQ_A-PAQ_A_Total': 'PAQ-PAQ_Total'})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:06:04.958465Z","iopub.execute_input":"2024-12-12T09:06:04.958940Z","iopub.status.idle":"2024-12-12T09:06:04.982864Z","shell.execute_reply.started":"2024-12-12T09:06:04.958897Z","shell.execute_reply":"2024-12-12T09:06:04.981740Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n\n    df.drop('step', axis=1, inplace=True)\n\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n\n    ids = os.listdir(dirname)\n\n    \n\n    with ThreadPoolExecutor() as executor:\n\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n\n    \n\n    stats, indexes = zip(*results)\n\n    \n\n    df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n\n    df['id'] = indexes\n\n    \n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:06:04.984601Z","iopub.execute_input":"2024-12-12T09:06:04.985257Z","iopub.status.idle":"2024-12-12T09:06:04.995494Z","shell.execute_reply.started":"2024-12-12T09:06:04.985203Z","shell.execute_reply":"2024-12-12T09:06:04.994219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PCIAT_cols = [f'PCIAT-PCIAT_{i+1:02d}' for i in range(20)]\n\ndef IncorrectRows(row):\n\n    if pd.isna(row['PCIAT-PCIAT_Total']):\n\n        return np.nan\n\n    max_possible = row['PCIAT-PCIAT_Total'] + row[PCIAT_cols].isna().sum() * 5\n\n    if row['PCIAT-PCIAT_Total'] <= 30 and max_possible <= 30:\n\n        return 0\n\n    elif 31 <= row['PCIAT-PCIAT_Total'] <= 49 and max_possible <= 49:\n\n        return 1\n\n    elif 50 <= row['PCIAT-PCIAT_Total'] <= 79 and max_possible <= 79:\n\n        return 2\n\n    elif row['PCIAT-PCIAT_Total'] >= 80 and max_possible >= 80:\n\n        return 3\n\n    return np.nan\n\n\n\ntrain['recal_sii'] = train.apply(IncorrectRows, axis=1)\n\n\n\nmismatch_rows = train[\n\n    (train['recal_sii'] != train['sii']) & train['sii'].notna()\n\n]\n\nmismatch_indexes = mismatch_rows.index\n\ntrain = train.drop(mismatch_indexes)\n\ntrain = train.drop(['recal_sii'], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:06:04.997216Z","iopub.execute_input":"2024-12-12T09:06:04.997749Z","iopub.status.idle":"2024-12-12T09:06:06.516252Z","shell.execute_reply.started":"2024-12-12T09:06:04.997691Z","shell.execute_reply":"2024-12-12T09:06:06.514850Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:06:06.517692Z","iopub.execute_input":"2024-12-12T09:06:06.518037Z","iopub.status.idle":"2024-12-12T09:06:06.613442Z","shell.execute_reply.started":"2024-12-12T09:06:06.518004Z","shell.execute_reply":"2024-12-12T09:06:06.612102Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:06:14.953883Z","iopub.execute_input":"2024-12-12T09:06:14.954369Z","iopub.status.idle":"2024-12-12T09:07:46.136184Z","shell.execute_reply.started":"2024-12-12T09:06:14.954318Z","shell.execute_reply":"2024-12-12T09:07:46.135012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:08:35.196400Z","iopub.execute_input":"2024-12-12T09:08:35.199529Z","iopub.status.idle":"2024-12-12T09:08:35.216431Z","shell.execute_reply.started":"2024-12-12T09:08:35.199456Z","shell.execute_reply":"2024-12-12T09:08:35.215005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.LeakyReLU(0.2),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.LeakyReLU(0.2),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.LeakyReLU(0.2)\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.LeakyReLU(0.2),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.LeakyReLU(0.2),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n\n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:08:39.649123Z","iopub.execute_input":"2024-12-12T09:08:39.649511Z","iopub.status.idle":"2024-12-12T09:08:39.659072Z","shell.execute_reply.started":"2024-12-12T09:08:39.649479Z","shell.execute_reply":"2024-12-12T09:08:39.657788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n\n    data_tensor = torch.FloatTensor(df_scaled)\n\n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n\n    criterion = F.smooth_l1_loss\n    optimizer = optim.Adam(autoencoder.parameters())\n\n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n\n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n\n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n\n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n\n    return df_encoded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:08:48.786713Z","iopub.execute_input":"2024-12-12T09:08:48.787130Z","iopub.status.idle":"2024-12-12T09:08:48.797451Z","shell.execute_reply.started":"2024-12-12T09:08:48.787095Z","shell.execute_reply":"2024-12-12T09:08:48.796257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\n\ntrain_ts_encoded[\"id\"] = train_ts[\"id\"]\ntest_ts_encoded[\"id\"] = test_ts[\"id\"]\n\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:09:01.987001Z","iopub.execute_input":"2024-12-12T09:09:01.987405Z","iopub.status.idle":"2024-12-12T09:09:13.892914Z","shell.execute_reply.started":"2024-12-12T09:09:01.987372Z","shell.execute_reply":"2024-12-12T09:09:13.891737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:09:18.187162Z","iopub.execute_input":"2024-12-12T09:09:18.187996Z","iopub.status.idle":"2024-12-12T09:09:18.194591Z","shell.execute_reply.started":"2024-12-12T09:09:18.187950Z","shell.execute_reply":"2024-12-12T09:09:18.193169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"featuresCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n               'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW',\n                'PAQ-PAQ_Total',\n                'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nSEASON_COL = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ-Season', 'SDS-Season', 'PreInt_EduHx-Season']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:09:38.528760Z","iopub.execute_input":"2024-12-12T09:09:38.529205Z","iopub.status.idle":"2024-12-12T09:09:38.536346Z","shell.execute_reply.started":"2024-12-12T09:09:38.529167Z","shell.execute_reply":"2024-12-12T09:09:38.535238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    #Age\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['Physical-Waist_Age'] = df['Basic_Demos-Age'] * df['Physical-Waist_Circumference']\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Physical-Height_Age'] = df['Basic_Demos-Age'] * df['Physical-Height']\n    df['SDS_InternetHours'] = df['SDS-SDS_Total_T'] * df['PreInt_EduHx-computerinternet_hoursday']\n\n    #SDS\n    df['SDS_BMI'] = df['BIA-BIA_BMI'] * df['SDS-SDS_Total_T']\n    df['CGAS_SDS'] = df['CGAS-CGAS_Score'] * df['SDS-SDS_Total_T']\n    df['CGAS_Endurance_Mins'] = df['CGAS-CGAS_Score'] * df['Fitness_Endurance-Time_Mins']\n    df['SDS_Activity'] = df['BIA-BIA_Activity_Level_num'] * df['SDS-SDS_Total_T']\n\n    df['BMI_Systolic_BP'] = df['BIA-BIA_BMI'] * df['Physical-Systolic_BP']\n    df['Age_Systolic_BP'] = df['Basic_Demos-Age'] * df['Physical-Systolic_BP']\n    df['PreInt_Systolic_BP'] = df['Physical-Systolic_BP'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['PAQ_A_Activity'] = df['BIA-BIA_Activity_Level_num'] * df['PAQ-PAQ_Total']\n    df['Activity_CU_PU'] = df['BIA-BIA_Activity_Level_num'] * df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n\n    #FGC\n    df['FGC_CU_PU'] = df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n    df['FGC_CU_PU_Age'] = df['FGC-FGC_CU'] * df['FGC-FGC_PU'] * df['Basic_Demos-Age']\n    df['FGC_GSND_GSD'] = df['FGC-FGC_GSND'] * df['FGC-FGC_GSD']\n    df['FGC_GSND_GSD_Age'] = df['FGC-FGC_GSND'] * df['FGC-FGC_GSD'] * df['Basic_Demos-Age']\n    df['CGAS_CU_PU'] = df['CGAS-CGAS_Score'] * df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n    df['PreInt_FGC_CU_PU'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n    df['Endurance_CU_PU'] = df['Fitness_Endurance-Time_Mins'] * df['FGC-FGC_CU'] * df['FGC-FGC_PU']\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:10:14.402462Z","iopub.execute_input":"2024-12-12T09:10:14.402935Z","iopub.status.idle":"2024-12-12T09:10:14.414023Z","shell.execute_reply.started":"2024-12-12T09:10:14.402897Z","shell.execute_reply":"2024-12-12T09:10:14.412508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:10:17.648560Z","iopub.execute_input":"2024-12-12T09:10:17.648982Z","iopub.status.idle":"2024-12-12T09:10:17.688224Z","shell.execute_reply.started":"2024-12-12T09:10:17.648948Z","shell.execute_reply":"2024-12-12T09:10:17.687107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if 'sii' in train.columns:\n    train = train.dropna(subset='sii')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:10:25.119720Z","iopub.execute_input":"2024-12-12T09:10:25.120174Z","iopub.status.idle":"2024-12-12T09:10:25.132216Z","shell.execute_reply.started":"2024-12-12T09:10:25.120136Z","shell.execute_reply":"2024-12-12T09:10:25.130948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = feature_engineering(test)\n\ntrain = train.drop('id', axis=1)\ntest  = test .drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:10:28.825015Z","iopub.execute_input":"2024-12-12T09:10:28.825406Z","iopub.status.idle":"2024-12-12T09:10:28.842450Z","shell.execute_reply.started":"2024-12-12T09:10:28.825373Z","shell.execute_reply":"2024-12-12T09:10:28.840960Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"featuresCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ-PAQ_Total',\n                'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii',\n                'Internet_Hours_Age', 'Physical-Waist_Age', 'BMI_Age', 'SDS_InternetHours',\n                'SDS_BMI', 'CGAS_SDS', 'CGAS_Endurance_Mins', 'SDS_Activity',\n                'BMI_Systolic_BP', 'Age_Systolic_BP', 'PreInt_Systolic_BP', 'PAQ_A_Activity',\n                'Activity_CU_PU', 'FGC_CU_PU', 'FGC_CU_PU_Age', 'FGC_GSND_GSD', 'FGC_GSND_GSD_Age',\n                'CGAS_CU_PU', 'PreInt_FGC_CU_PU', 'Endurance_CU_PU',\n]\nfeaturesCols += time_series_cols\nfeaturesCols += SEASON_COL\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:10:54.508365Z","iopub.execute_input":"2024-12-12T09:10:54.509301Z","iopub.status.idle":"2024-12-12T09:10:54.527287Z","shell.execute_reply.started":"2024-12-12T09:10:54.509247Z","shell.execute_reply":"2024-12-12T09:10:54.525860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"featuresCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ-PAQ_Total',\n                'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday',\n                'Internet_Hours_Age', 'Physical-Waist_Age', 'BMI_Age', 'SDS_InternetHours',\n                'SDS_BMI', 'CGAS_SDS', 'CGAS_Endurance_Mins', 'SDS_Activity',\n                'BMI_Systolic_BP', 'Age_Systolic_BP', 'PreInt_Systolic_BP', 'PAQ_A_Activity',\n                'Activity_CU_PU', 'FGC_CU_PU', 'FGC_CU_PU_Age', 'FGC_GSND_GSD', 'FGC_GSND_GSD_Age',\n                'CGAS_CU_PU', 'PreInt_FGC_CU_PU', 'Endurance_CU_PU',\n]\nfeaturesCols += time_series_cols\nfeaturesCols += SEASON_COL\ntest = test[featuresCols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:12:12.247189Z","iopub.execute_input":"2024-12-12T09:12:12.247709Z","iopub.status.idle":"2024-12-12T09:12:12.258630Z","shell.execute_reply.started":"2024-12-12T09:12:12.247667Z","shell.execute_reply":"2024-12-12T09:12:12.257362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def update(df):\n    global SEASON_COL\n    for c in SEASON_COL: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:12:29.945675Z","iopub.execute_input":"2024-12-12T09:12:29.946130Z","iopub.status.idle":"2024-12-12T09:12:29.978566Z","shell.execute_reply.started":"2024-12-12T09:12:29.946091Z","shell.execute_reply":"2024-12-12T09:12:29.977305Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"season_mapping = {'Spring': 0, 'Summer': 1, 'Fall': 2, 'Winter': 3, 'Missing': 4}\n\nfor col in SEASON_COL:\n\n    train[col] = train[col].map(season_mapping)\n\n    test[col] = test[col].map(season_mapping)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:12:38.752482Z","iopub.execute_input":"2024-12-12T09:12:38.752946Z","iopub.status.idle":"2024-12-12T09:12:38.776284Z","shell.execute_reply.started":"2024-12-12T09:12:38.752906Z","shell.execute_reply":"2024-12-12T09:12:38.775036Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:12:43.262154Z","iopub.execute_input":"2024-12-12T09:12:43.262557Z","iopub.status.idle":"2024-12-12T09:12:43.391782Z","shell.execute_reply.started":"2024-12-12T09:12:43.262509Z","shell.execute_reply":"2024-12-12T09:12:43.390680Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#from sklearn.preprocessing import LabelEncoder\n\n#def encode_categories(df, cat_features):\n#    le_dict = {}\n#    for col in cat_features:\n#        le = LabelEncoder()\n#        df[col] = le.fit_transform(df[col])\n#        le_dict[col] = le  # 保存編碼器以便反向轉換\n#    return df, le_dict\n\n#train, le_dict = encode_categories(train, SEASON_COL)\n#test, _ = encode_categories(test, SEASON_COL)  # 測試集使用相同的編碼器\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:02:39.348347Z","iopub.status.idle":"2024-12-12T09:02:39.349448Z","shell.execute_reply.started":"2024-12-12T09:02:39.349161Z","shell.execute_reply":"2024-12-12T09:02:39.349191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Light = LGBMRegressor(**Params, random_state=42, verbose=-1, n_estimators=300)\n#RF_Model = RandomForestRegressor(**RF_Params)\n#CatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n#XGB_Model = XGBRegressor(**XGB_Params)\n#TabNet_Model = TabNetWrapper(**TabNet_Params)\n#ExtraTrees_Model = ExtraTreesRegressor(**ExtraTrees_Params)\n#HistGB_Model = HistGradientBoostingRegressor(**HistGB_Params)\n#Ridge_Model = Ridge(**Ridge_Params)\n#SVR_Model = SVR(**SVR_Params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:02:39.351346Z","iopub.status.idle":"2024-12-12T09:02:39.351968Z","shell.execute_reply.started":"2024-12-12T09:02:39.351630Z","shell.execute_reply":"2024-12-12T09:02:39.351710Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error\nfrom tqdm import tqdm\nn_splits = 5\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    return tp_rounded\n\nimputer = SimpleImputer(strategy='median')\n\nensemble = StackingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(\n    random_state=42,\n    lambda_l1=0.1,  # L1 regularization\n    lambda_l2=0.1,  # L2 regularization\n    max_depth=7     # Limit tree depth\n))])),\n    ('xgb', Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(\n    random_state=42,\n    reg_alpha=0.1,  # L1 regularization\n    reg_lambda=0.1, # L2 regularization\n    max_depth=7,    # Limit tree depth\n    subsample=0.8,  # Subsampling to prevent overfitting\n    colsample_bytree=0.8  # Use a subset of features\n))])),\n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(\n    random_state=42,\n    silent=True,\n    depth=7,           # Limit tree depth\n    l2_leaf_reg=3.0,   # L2 regularization\n    subsample=0.8      # Use a subset of samples\n))])),\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(\n    random_state=42,\n    max_depth=7,            # Limit tree depth\n    min_samples_split=10,   # Minimum samples to split an internal node\n    min_samples_leaf=4      # Minimum samples in a leaf node\n))])),\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(\n    random_state=42,\n    max_depth=7,          # Limit tree depth\n    learning_rate=0.05,   # Slow down learning\n    subsample=0.8         # Use a subset of samples\n))]))\n])\n\nSubmission = TrainML(ensemble, test)\nSubmission = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission\n})\nSubmission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:13:17.875302Z","iopub.execute_input":"2024-12-12T09:13:17.875849Z","iopub.status.idle":"2024-12-12T09:27:31.263079Z","shell.execute_reply.started":"2024-12-12T09:13:17.875795Z","shell.execute_reply":"2024-12-12T09:27:31.261391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-12T09:02:39.366140Z","iopub.status.idle":"2024-12-12T09:02:39.366716Z","shell.execute_reply.started":"2024-12-12T09:02:39.366403Z","shell.execute_reply":"2024-12-12T09:02:39.366431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nimport matplotlib.pyplot as plt\n\ndef TrainML_with_learning_curve(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n    \n    train_qwk_scores = []\n    val_qwk_scores = []\n    train_losses = []\n    val_losses = []\n\n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(SKF.split(X, y)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        # Predictions\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        # Calculate Metrics\n        train_qwk = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_qwk = quadratic_weighted_kappa(y_val, y_val_pred.round(0).astype(int))\n        train_loss = mean_squared_error(y_train, y_train_pred)\n        val_loss = mean_squared_error(y_val, y_val_pred)\n\n        # Append metrics for learning curve\n        train_qwk_scores.append(train_qwk)\n        val_qwk_scores.append(val_qwk)\n        train_losses.append(train_loss)\n        val_losses.append(val_loss)\n\n        print(f\"Fold {fold+1} - Train QWK: {train_qwk:.4f}, Validation QWK: {val_qwk:.4f}, Train Loss: {train_loss:.4f}, Validation Loss: {val_loss:.4f}\")\n\n    print(f\"Mean Train QWK --> {np.mean(train_qwk_scores):.4f}\")\n    print(f\"Mean Validation QWK --> {np.mean(val_qwk_scores):.4f}\")\n\n    return train_qwk_scores, val_qwk_scores, train_losses, val_losses\n\ndef plot_learning_curve(train_scores, val_scores, metric_name):\n    plt.figure(figsize=(10, 6))\n    plt.plot(range(1, len(train_scores) + 1), train_scores, marker='o', label=f\"Training {metric_name}\")\n    plt.plot(range(1, len(val_scores) + 1), val_scores, marker='o', label=f\"Validation {metric_name}\")\n    plt.xlabel(\"Fold Number\")\n    plt.ylabel(metric_name)\n    plt.title(f\"Learning Curve: Training vs Validation {metric_name}\")\n    plt.legend()\n    plt.grid(True)\n    plt.show()\n\ntrain_qwk_scores, val_qwk_scores, train_losses, val_losses = TrainML_with_learning_curve(ensemble, test)\n\n# Plot Learning Curve for QWK\nplot_learning_curve(train_qwk_scores, val_qwk_scores, \"QWK\")\n\n# Plot Learning Curve for Loss\nplot_learning_curve(train_losses, val_losses, \"Loss (MSE)\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}