{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30823,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np  \nimport pandas as pd\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.metrics import cohen_kappa_score, classification_report, accuracy_score\nfrom sklearn.model_selection import StratifiedKFold, RandomizedSearchCV\nfrom scipy.optimize import minimize\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import KNNImputer\nfrom catboost import CatBoostRegressor\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\n\nfrom sklearn.ensemble import VotingRegressor\nimport warnings\n\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-23T14:23:40.886506Z","iopub.execute_input":"2024-12-23T14:23:40.886787Z","iopub.status.idle":"2024-12-23T14:23:40.892393Z","shell.execute_reply.started":"2024-12-23T14:23:40.886765Z","shell.execute_reply":"2024-12-23T14:23:40.891313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 讀取數據\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n# 查看數據描述\ndisplay(train.describe())\n\nprint(\"Train Dimension: \", train.shape)\nprint(\"Test Dimension: \", test.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T14:23:40.893417Z","iopub.execute_input":"2024-12-23T14:23:40.893675Z","iopub.status.idle":"2024-12-23T14:23:41.102588Z","shell.execute_reply.started":"2024-12-23T14:23:40.893644Z","shell.execute_reply":"2024-12-23T14:23:41.101604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 顯示前幾行數據\ndisplay(train.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T14:23:41.104171Z","iopub.execute_input":"2024-12-23T14:23:41.104460Z","iopub.status.idle":"2024-12-23T14:23:41.166723Z","shell.execute_reply.started":"2024-12-23T14:23:41.104436Z","shell.execute_reply":"2024-12-23T14:23:41.166021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    # 將某些物理量為 0 的值替換為 NaN\n    df['Physical-Weight'] = df['Physical-Weight'].replace(0, np.nan)\n    df['Physical-Height'] = df['Physical-Height'].replace(0, np.nan)\n    df['Physical-BMI'] = df['Physical-BMI'].replace(0, np.nan)\n    df['Basic_Demos-Age'] = df['Basic_Demos-Age'].replace(0, np.nan)\n    df['Physical-Waist_Circumference'] = df['Physical-Waist_Circumference'].replace(0, np.nan)\n    df['Physical-Diastolic_BP'] = df['Physical-Diastolic_BP'].replace(0, np.nan)\n    df['Physical-HeartRate'] = df['Physical-HeartRate'].replace(0, np.nan)\n    df['Physical-Systolic_BP'] = df['Physical-Systolic_BP'].replace(0, np.nan)\n\n    # 創建新特徵 Total_Fitness_Endurance_Time\n    df['Total_Fitness_Endurance_Time'] = np.where(\n        df['Fitness_Endurance-Time_Mins'].isna() & df['Fitness_Endurance-Time_Sec'].isna(),\n        np.nan,\n        df['Fitness_Endurance-Time_Mins'].fillna(0) + df['Fitness_Endurance-Time_Sec'].fillna(0) / 60\n    )\n    df = df.drop([\"Fitness_Endurance-Time_Mins\", \"Fitness_Endurance-Time_Sec\"], axis=1)\n    return df\n\n# 應用特徵工程\ntrain = feature_engineering(train)\ntest = feature_engineering(test)\n\n# 選擇特徵列\nfeaturesCols = [\n    'Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n    'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n    'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n    'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n    'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n    'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n    'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n    'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n    'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n    'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n    'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n    'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n    'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n    'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n    'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n    'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n    'PreInt_EduHx-computerinternet_hoursday','Total_Fitness_Endurance_Time'\n]\n\n# 提取標籤並去除不必要的列\nlabel_column = train['sii']\ntrain = train.drop(['sii', 'id'], axis=1)\ntest = test.drop(['id'], axis=1)\n\n# 合併訓練集和測試集進行統一處理\nall_data = pd.concat([train, test], axis=0)\n\n# 確定數值型特徵\nnumeric_cols = all_data.select_dtypes(include=['float64', 'int64']).columns\n\n# 使用 KNN Imputer 進行缺失值填補\nimputer = KNNImputer(n_neighbors=5)\nimputed_data = imputer.fit_transform(all_data[numeric_cols])\n\n# 將填補後的數據轉換回 DataFrame\nall_data_imputed = pd.DataFrame(imputed_data, \n                                columns=numeric_cols, \n                                index=all_data.index)\n\n# 恢復非數值型特徵\nfor col in all_data.columns:\n    if col not in numeric_cols:\n        all_data_imputed[col] = all_data[col]\n\n# 分離回訓練集和測試集\ntrain_imputed = all_data_imputed.iloc[:len(train)]\ntest_imputed = all_data_imputed.iloc[len(train):]\n\n# 更新訓練集和測試集\ntrain = train_imputed\ntest = test_imputed\n\n# 選擇特徵\ntrain = train[featuresCols]\ntest = test[featuresCols]\n\n# 添加標籤回訓練集\ntrain['sii'] = label_column\n\n# 去除標籤為 NaN 的樣本\ntrain = train.dropna(subset=['sii'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T14:23:41.167931Z","iopub.execute_input":"2024-12-23T14:23:41.168147Z","iopub.status.idle":"2024-12-23T14:23:48.368355Z","shell.execute_reply.started":"2024-12-23T14:23:41.168128Z","shell.execute_reply":"2024-12-23T14:23:48.367648Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 定義類別型特徵\ncat_c = [\n    'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n    'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n    'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season'\n]\n\ndef update(df):\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\n# 應用更新函數\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\n# 將類別特徵映射為整數編碼\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mapping_te = create_mapping(col, test)\n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mapping_te).astype(int)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T14:23:48.369123Z","iopub.execute_input":"2024-12-23T14:23:48.369410Z","iopub.status.idle":"2024-12-23T14:23:48.423857Z","shell.execute_reply.started":"2024-12-23T14:23:48.369376Z","shell.execute_reply":"2024-12-23T14:23:48.422974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T14:23:48.424577Z","iopub.execute_input":"2024-12-23T14:23:48.424818Z","iopub.status.idle":"2024-12-23T14:23:48.429770Z","shell.execute_reply.started":"2024-12-23T14:23:48.424798Z","shell.execute_reply":"2024-12-23T14:23:48.428823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 設定隨機種子\nSEED = 42\nn_splits = 5\n\n# 定義超參數搜索空間\nCatBoost_Params = {\n    'learning_rate': [0.01, 0.05, 0.1, 0.2],\n    'depth': [4, 6, 8, 10],\n    'iterations': [100, 200, 300],\n    'l2_leaf_reg': [1, 3, 5, 10],\n}\n\nLGB_Params = {\n    'learning_rate': [0.01, 0.046, 0.05, 0.1],\n    'max_depth': [8, 10, 12, 14],\n    'num_leaves': [31, 63, 127, 255],\n    'min_data_in_leaf': [20, 30, 40, 50],\n    'feature_fraction': [0.7, 0.8, 0.9, 1.0],\n    'bagging_fraction': [0.7, 0.8, 0.9, 1.0],\n    'bagging_freq': [1, 2, 3, 4],\n    'lambda_l1': [0, 1, 5, 10],\n    'lambda_l2': [0, 0.01, 0.1, 1],\n}\n\nXGB_Params = {\n    'learning_rate': [0.01, 0.05, 0.1, 0.2],\n    'max_depth': [3, 5, 7, 9],\n    'n_estimators': [100, 200, 300],\n    'subsample': [0.6, 0.8, 1.0],\n    'colsample_bytree': [0.6, 0.8, 1.0],\n    'reg_alpha': [0, 0.1, 1, 5],\n    'reg_lambda': [1, 5, 10],\n}\n\n# 創建模型實例\nLight = LGBMRegressor(random_state=SEED, device='cpu', verbose=-1)\nXGB_Model = XGBRegressor(random_state=SEED, tree_method='hist', verbosity=0)\nCatBoost_Model = CatBoostRegressor(random_seed=SEED, task_type='CPU', verbose=0)\n\n# 定義超參數搜尋器\nLight_Search = RandomizedSearchCV(\n    estimator=Light,\n    param_distributions=LGB_Params,\n    n_iter=50,\n    scoring='neg_mean_squared_error',\n    cv=3,\n    random_state=SEED,\n    n_jobs=-1,\n    verbose=1\n)\n\nXGB_Search = RandomizedSearchCV(\n    estimator=XGB_Model,\n    param_distributions=XGB_Params,\n    n_iter=50,\n    scoring='neg_mean_squared_error',\n    cv=3,\n    random_state=SEED,\n    n_jobs=-1,\n    verbose=1\n)\n\nCatBoost_Search = RandomizedSearchCV(\n    estimator=CatBoost_Model,\n    param_distributions=CatBoost_Params,\n    n_iter=50,\n    scoring='neg_mean_squared_error',\n    cv=3,\n    random_state=SEED,\n    n_jobs=-1,\n    verbose=1\n)\n\n# 執行超參數搜尋\nprint(\"Starting LightGBM hyperparameter search...\")\nLight_Search.fit(train.drop('sii', axis=1), train['sii'])\n\nprint(\"Starting XGBoost hyperparameter search...\")\nXGB_Search.fit(train.drop('sii', axis=1), train['sii'])\n\nprint(\"Starting CatBoost hyperparameter search...\")\nCatBoost_Search.fit(train.drop('sii', axis=1), train['sii'])\n\n# 獲取最佳模型\nbest_Light = Light_Search.best_estimator_\nbest_XGB = XGB_Search.best_estimator_\nbest_CatBoost = CatBoost_Search.best_estimator_\n\nprint(\"Best LightGBM Parameters:\", Light_Search.best_params_)\nprint(\"Best XGBoost Parameters:\", XGB_Search.best_params_)\nprint(\"Best CatBoost Parameters:\", CatBoost_Search.best_params_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T14:23:48.430742Z","iopub.execute_input":"2024-12-23T14:23:48.431039Z","iopub.status.idle":"2024-12-23T14:33:28.739009Z","shell.execute_reply.started":"2024-12-23T14:23:48.431005Z","shell.execute_reply":"2024-12-23T14:33:28.738001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 定義 Voting Regressor，使用最佳模型\nbest_voting_model = VotingRegressor(estimators=[\n    ('lightgbm', best_Light),\n    ('xgboost', best_XGB),\n    ('catboost', best_CatBoost)\n], weights=[4.0, 4.0, 4.0])  # 調整權重根據需要\n\n# 查看 Voting Regressor 的參數\nprint(best_voting_model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T14:41:11.100245Z","iopub.execute_input":"2024-12-23T14:41:11.100480Z","iopub.status.idle":"2024-12-23T14:41:11.110700Z","shell.execute_reply.started":"2024-12-23T14:41:11.100460Z","shell.execute_reply":"2024-12-23T14:41:11.109842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 將數據分成特徵和標籤\nX = train.drop('sii', axis=1)\ny = train['sii'].astype(int)\n\n# 定義 Stratified K-Fold\nskf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n# 儲存交叉驗證的預測結果\noof_preds = np.zeros(X.shape[0])\n\n# 儲存模型在每一折的閾值\nthresholds_list = []\n\nfor fold, (train_idx, valid_idx) in enumerate(skf.split(X, y)):\n    print(f\"Fold {fold + 1}/{n_splits}\")\n    X_train_fold, X_valid_fold = X.iloc[train_idx], X.iloc[valid_idx]\n    y_train_fold, y_valid_fold = y.iloc[train_idx], y.iloc[valid_idx]\n    \n    # 訓練最佳 Voting Regressor\n    best_voting_model.fit(X_train_fold, y_train_fold)\n    \n    # 預測驗證集\n    oof_valid_preds = best_voting_model.predict(X_valid_fold)\n    \n    # 優化閾值\n    initial_thresholds = [0.5, 1.5, 2.5]\n    result = minimize(evaluate_predictions, initial_thresholds, args=(y_valid_fold, oof_valid_preds),\n                      method='Nelder-Mead', options={'maxiter': 10000})\n    optimized_thresholds = result.x\n    thresholds_list.append(optimized_thresholds)\n    \n    # 將優化後的閾值應用於驗證集預測\n    oof_preds[valid_idx] = threshold_Rounder(oof_valid_preds, optimized_thresholds)\n    \n    print(f\"Optimized thresholds: {optimized_thresholds}\")\n    print(f\"QWK: {quadratic_weighted_kappa(y_valid_fold, oof_preds[valid_idx])}\\n\")\n\n# 計算整體 QWK\noverall_qwk = quadratic_weighted_kappa(y, oof_preds)\nprint(f\"Overall QWK: {overall_qwk}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T14:41:14.003933Z","iopub.execute_input":"2024-12-23T14:41:14.004276Z","iopub.status.idle":"2024-12-23T14:41:40.354261Z","shell.execute_reply.started":"2024-12-23T14:41:14.004248Z","shell.execute_reply":"2024-12-23T14:41:40.353276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 計算平均閾值\naverage_thresholds = np.mean(thresholds_list, axis=0)\nprint(f\"Average Optimized Thresholds: {average_thresholds}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T14:41:40.380610Z","iopub.execute_input":"2024-12-23T14:41:40.380938Z","iopub.status.idle":"2024-12-23T14:41:40.394063Z","shell.execute_reply.started":"2024-12-23T14:41:40.380909Z","shell.execute_reply":"2024-12-23T14:41:40.393208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 訓練最終模型\nbest_voting_model.fit(X, y)\n\n# 預測測試集\ntest_preds = best_voting_model.predict(test)\n\n# 應用平均閾值進行分類\nfinal_predictions = threshold_Rounder(test_preds, average_thresholds)\n\n# 生成提交文件\nsubmission = sample.copy()\nsubmission['sii'] = final_predictions.astype(int)\n\n# 保存提交文件\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission saved to 'submission.csv'\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T14:41:52.062859Z","iopub.execute_input":"2024-12-23T14:41:52.063116Z","iopub.status.idle":"2024-12-23T14:41:57.263530Z","shell.execute_reply.started":"2024-12-23T14:41:52.063094Z","shell.execute_reply":"2024-12-23T14:41:57.262366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay, classification_report\n\n# 假設 y 是真實標籤，final_predictions 是模型的最終預測\n# 這裡用隨機數據模擬，如果有實際數據，可以用真實數據替代\nnp.random.seed(42)\ny = np.random.randint(0, 4, 100)  # 假設有4個類別的真實標籤\nfinal_predictions = np.random.randint(0, 4, 100)  # 模擬模型的預測結果\n\n# 計算混淆矩陣\ncm = confusion_matrix(y, final_predictions, labels=[0, 1, 2, 3])\n\n# 繪製混淆矩陣\nplt.figure(figsize=(8, 6))\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=[0, 1, 2, 3])\ndisp.plot(cmap=plt.cm.Blues, ax=plt.gca(), colorbar=True)\nplt.title(\"Confusion Matrix Visualization\")\nplt.show()\n\n# 計算分類指數並打印\nreport = classification_report(y, final_predictions, target_names=[\"Class 0\", \"Class 1\", \"Class 2\", \"Class 3\"])\nprint(\"Classification Report:\\n\", report)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T14:41:57.264753Z","iopub.execute_input":"2024-12-23T14:41:57.264989Z","iopub.status.idle":"2024-12-23T14:41:57.567810Z","shell.execute_reply.started":"2024-12-23T14:41:57.264967Z","shell.execute_reply":"2024-12-23T14:41:57.567173Z"}},"outputs":[],"execution_count":null}]}