{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np  \nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom catboost import CatBoostRegressor\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\n\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import KNNImputer\n\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-21T18:24:01.086455Z","iopub.execute_input":"2024-12-21T18:24:01.086740Z","iopub.status.idle":"2024-12-21T18:24:18.026147Z","shell.execute_reply.started":"2024-12-21T18:24:01.086709Z","shell.execute_reply":"2024-12-21T18:24:18.025508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 讀取數據\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n# 查看數據描述\ndisplay(train.describe())\n\nprint(\"Train Dimension: \", train.shape)\nprint(\"Test Dimension: \", test.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T18:24:18.026958Z","iopub.execute_input":"2024-12-21T18:24:18.027775Z","iopub.status.idle":"2024-12-21T18:24:18.251986Z","shell.execute_reply.started":"2024-12-21T18:24:18.027741Z","shell.execute_reply":"2024-12-21T18:24:18.251255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 顯示前幾行數據\ndisplay(train.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T18:24:18.252840Z","iopub.execute_input":"2024-12-21T18:24:18.253163Z","iopub.status.idle":"2024-12-21T18:24:18.310847Z","shell.execute_reply.started":"2024-12-21T18:24:18.253131Z","shell.execute_reply":"2024-12-21T18:24:18.309998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    # 將某些物理量為 0 的值替換為 NaN\n    df['Physical-Weight'] = df['Physical-Weight'].replace(0, np.nan)\n    df['Physical-Height'] = df['Physical-Height'].replace(0, np.nan)\n    df['Physical-BMI'] = df['Physical-BMI'].replace(0, np.nan)\n    df['Basic_Demos-Age'] = df['Basic_Demos-Age'].replace(0, np.nan)\n    df['Physical-Waist_Circumference'] = df['Physical-Waist_Circumference'].replace(0, np.nan)\n    df['Physical-Diastolic_BP'] = df['Physical-Diastolic_BP'].replace(0, np.nan)\n    df['Physical-HeartRate'] = df['Physical-HeartRate'].replace(0, np.nan)\n    df['Physical-Systolic_BP'] = df['Physical-Systolic_BP'].replace(0, np.nan)\n\n    # 創建新特徵 Total_Fitness_Endurance_Time\n    df['Total_Fitness_Endurance_Time'] = np.where(\n        df['Fitness_Endurance-Time_Mins'].isna() & df['Fitness_Endurance-Time_Sec'].isna(),\n        np.nan,\n        df['Fitness_Endurance-Time_Mins'].fillna(0) + df['Fitness_Endurance-Time_Sec'].fillna(0) / 60\n    )\n    df = df.drop([\"Fitness_Endurance-Time_Mins\", \"Fitness_Endurance-Time_Sec\"], axis=1)\n    return df\n\n# 應用特徵工程\ntrain = feature_engineering(train)\ntest = feature_engineering(test)\n\n# 選擇特徵列\nfeaturesCols = [\n    'Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n    'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n    'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n    'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n    'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n    'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n    'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n    'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n    'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n    'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n    'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n    'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n    'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n    'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n    'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n    'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n    'PreInt_EduHx-computerinternet_hoursday','Total_Fitness_Endurance_Time'\n]\n\n# 提取標籤並去除不必要的列\nlabel_column = train['sii']\ntrain = train.drop(['sii', 'id'], axis=1)\ntest = test.drop(['id'], axis=1)\n\n# 合併訓練集和測試集進行統一處理\nall_data = pd.concat([train, test], axis=0)\n\n# 確定數值型特徵\nnumeric_cols = all_data.select_dtypes(include=['float64', 'int64']).columns\n\n# 使用 KNN Imputer 進行缺失值填補\nimputer = KNNImputer(n_neighbors=5)\nimputed_data = imputer.fit_transform(all_data[numeric_cols])\n\n# 將填補後的數據轉換回 DataFrame\nall_data_imputed = pd.DataFrame(imputed_data, \n                                columns=numeric_cols, \n                                index=all_data.index)\n\n# 恢復非數值型特徵\nfor col in all_data.columns:\n    if col not in numeric_cols:\n        all_data_imputed[col] = all_data[col]\n\n# 分離回訓練集和測試集\ntrain_imputed = all_data_imputed.iloc[:len(train)]\ntest_imputed = all_data_imputed.iloc[len(train):]\n\n# 更新訓練集和測試集\ntrain = train_imputed\ntest = test_imputed\n\n# 選擇特徵\ntrain = train[featuresCols]\ntest = test[featuresCols]\n\n# 添加標籤回訓練集\ntrain['sii'] = label_column\n\n# 去除標籤為 NaN 的樣本\ntrain = train.dropna(subset=['sii'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T18:24:18.312899Z","iopub.execute_input":"2024-12-21T18:24:18.313122Z","iopub.status.idle":"2024-12-21T18:24:25.317959Z","shell.execute_reply.started":"2024-12-21T18:24:18.313103Z","shell.execute_reply":"2024-12-21T18:24:25.317274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 定義類別型特徵\ncat_c = [\n    'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n    'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n    'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season'\n]\n\ndef update(df):\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\n# 應用更新函數\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\n# 將類別特徵映射為整數編碼\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mapping_te = create_mapping(col, test)\n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mapping_te).astype(int)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T18:24:25.319057Z","iopub.execute_input":"2024-12-21T18:24:25.319280Z","iopub.status.idle":"2024-12-21T18:24:25.371999Z","shell.execute_reply.started":"2024-12-21T18:24:25.319261Z","shell.execute_reply":"2024-12-21T18:24:25.371394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T18:24:25.372751Z","iopub.execute_input":"2024-12-21T18:24:25.373074Z","iopub.status.idle":"2024-12-21T18:24:25.377851Z","shell.execute_reply.started":"2024-12-21T18:24:25.373042Z","shell.execute_reply":"2024-12-21T18:24:25.377042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nclass LogisticRegressionRegressor(BaseEstimator, RegressorMixin):\n    def __init__(self, C=1.0, multi_class='multinomial', solver='lbfgs', max_iter=1000, random_state=42):\n        self.C = C\n        self.multi_class = multi_class\n        self.solver = solver\n        self.max_iter = max_iter\n        self.random_state = random_state\n        self.model = LogisticRegression(\n            C=self.C,\n            multi_class=self.multi_class,\n            solver=self.solver,\n            max_iter=self.max_iter,\n            random_state=self.random_state\n        )\n    \n    def fit(self, X, y):\n        self.model.fit(X, y)\n        return self\n    \n    def predict(self, X):\n        # 預測每個類別的概率，並計算期望值作為連續輸出\n        proba = self.model.predict_proba(X)\n        classes = self.model.classes_\n        expected_value = np.dot(proba, classes)\n        return expected_value\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T18:24:25.378605Z","iopub.execute_input":"2024-12-21T18:24:25.378831Z","iopub.status.idle":"2024-12-21T18:24:25.391708Z","shell.execute_reply.started":"2024-12-21T18:24:25.378810Z","shell.execute_reply":"2024-12-21T18:24:25.390879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 設定隨機種子\nSEED = 42\nn_splits = 5\n\n# 定義模型參數\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'l2_leaf_reg': 10,\n}\n\nLGB_Params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,\n    'lambda_l2': 0.01,\n}\n\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,\n    'reg_lambda': 5,\n}\n\n# 創建模型實例\nLight = LGBMRegressor(**LGB_Params, random_state=SEED, n_estimators=200, device='cpu', verbose=-1)\nXGB_Model = XGBRegressor(**XGB_Params, random_state=SEED, tree_method=\"gpu_hist\", verbosity=0)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params, random_seed=SEED, task_type='GPU', verbose=0)\n\n# 創建 Logistic Regression 回歸器\nLogReg_Regressor = LogisticRegressionRegressor(C=0.01, multi_class='multinomial', solver='lbfgs', max_iter=1000, random_state=SEED)\n\n# 定義 Voting Regressor，加入 Logistic Regression\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ('logistic_regression', LogReg_Regressor)\n], weights=[4.0, 4.0, 4.0, 1.0])  # 調整權重\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T18:24:25.392512Z","iopub.execute_input":"2024-12-21T18:24:25.392786Z","iopub.status.idle":"2024-12-21T18:24:25.411025Z","shell.execute_reply.started":"2024-12-21T18:24:25.392751Z","shell.execute_reply":"2024-12-21T18:24:25.410210Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 將數據分成特徵和標籤\nX = train.drop('sii', axis=1)\ny = train['sii'].astype(int)\n\n# 定義 Stratified K-Fold\nskf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n# 儲存交叉驗證的預測結果\noof_preds = np.zeros(X.shape[0])\n\n# 儲存模型在每一折的閾值\nthresholds_list = []\n\nfor fold, (train_idx, valid_idx) in enumerate(skf.split(X, y)):\n    print(f\"Fold {fold + 1}/{n_splits}\")\n    X_train_fold, X_valid_fold = X.iloc[train_idx], X.iloc[valid_idx]\n    y_train_fold, y_valid_fold = y.iloc[train_idx], y.iloc[valid_idx]\n    \n    # 訓練模型\n    voting_model.fit(X_train_fold, y_train_fold)\n    \n    # 預測驗證集\n    oof_valid_preds = voting_model.predict(X_valid_fold)\n    \n    # 優化閾值\n    initial_thresholds = [0.5, 1.5, 2.5]\n    bounds = [(0, 1), (1, 2), (2, 3)]\n    result = minimize(evaluate_predictions, initial_thresholds, args=(y_valid_fold, oof_valid_preds),\n                      method='Nelder-Mead', options={'maxiter': 10000})\n    optimized_thresholds = result.x\n    thresholds_list.append(optimized_thresholds)\n    \n    # 將優化後的閾值應用於驗證集預測\n    oof_preds[valid_idx] = threshold_Rounder(oof_valid_preds, optimized_thresholds)\n    \n    print(f\"Optimized thresholds: {optimized_thresholds}\")\n    print(f\"QWK: {quadratic_weighted_kappa(y_valid_fold, oof_preds[valid_idx])}\\n\")\n\n# 計算整體 QWK\noverall_qwk = quadratic_weighted_kappa(y, oof_preds)\nprint(f\"Overall QWK: {overall_qwk}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T18:24:25.411844Z","iopub.execute_input":"2024-12-21T18:24:25.412117Z","iopub.status.idle":"2024-12-21T18:24:44.141054Z","shell.execute_reply.started":"2024-12-21T18:24:25.412091Z","shell.execute_reply":"2024-12-21T18:24:44.140278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 計算平均閾值\naverage_thresholds = np.mean(thresholds_list, axis=0)\nprint(f\"Average Optimized Thresholds: {average_thresholds}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T18:24:44.141882Z","iopub.execute_input":"2024-12-21T18:24:44.142101Z","iopub.status.idle":"2024-12-21T18:24:44.147023Z","shell.execute_reply.started":"2024-12-21T18:24:44.142082Z","shell.execute_reply":"2024-12-21T18:24:44.146295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 訓練最終模型\nvoting_model.fit(X, y)\n\n# 預測測試集\ntest_preds = voting_model.predict(test)\n\n# 應用平均閾值進行分類\nfinal_predictions = threshold_Rounder(test_preds, average_thresholds)\n\n# 生成提交文件\nsubmission = sample.copy()\nsubmission['sii'] = final_predictions.astype(int)\n\n# 保存提交文件\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission saved to 'submission.csv'\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T18:24:44.147809Z","iopub.execute_input":"2024-12-21T18:24:44.148094Z","iopub.status.idle":"2024-12-21T18:24:47.819558Z","shell.execute_reply.started":"2024-12-21T18:24:44.148071Z","shell.execute_reply":"2024-12-21T18:24:47.818584Z"}},"outputs":[],"execution_count":null}]}