{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":31011,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":456.975767,"end_time":"2025-03-20T18:39:15.831625","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-03-20T18:31:38.855858","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Import library","metadata":{"papermill":{"duration":0.003773,"end_time":"2025-03-20T18:31:41.760644","exception":false,"start_time":"2025-03-20T18:31:41.756871","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nfrom concurrent.futures import ThreadPoolExecutor\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split, StratifiedKFold, cross_val_score\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nimport optuna\nfrom optuna.samplers import TPESampler\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom scipy.optimize import minimize","metadata":{"execution":{"iopub.status.busy":"2025-04-23T14:18:38.697693Z","iopub.execute_input":"2025-04-23T14:18:38.697942Z","iopub.status.idle":"2025-04-23T14:18:40.808948Z","shell.execute_reply.started":"2025-04-23T14:18:38.697918Z","shell.execute_reply":"2025-04-23T14:18:40.808347Z"},"papermill":{"duration":2.799699,"end_time":"2025-03-20T18:31:44.563527","exception":false,"start_time":"2025-03-20T18:31:41.763828","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Load and Explore Data","metadata":{"papermill":{"duration":0.003075,"end_time":"2025-03-20T18:31:44.570241","exception":false,"start_time":"2025-03-20T18:31:44.567166","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def process_file(filename, dirname):\n    \"\"\"Process a single parquet file and extract time-based features\"\"\"\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    \n    # Drop 'step' column if it exists\n    if 'step' in df.columns:\n        df.drop('step', axis=1, inplace=True)\n    \n    # Convert time_of_day to hours\n    df[\"hours\"] = df[\"time_of_day\"] // (3_600 * 1_000_000_000)\n    \n    # Define time periods\n    night = ((df[\"hours\"] >= 22) | (df[\"hours\"] <= 5))\n    day = ((df[\"hours\"] <= 20) & (df[\"hours\"] >= 7))\n    \n    # Initialize features dictionary\n    features = {}\n    \n    # Basic activity features\n    features['non_wear_mean'] = df[\"non-wear_flag\"].mean()\n    features['active_enmo_sum'] = df[\"enmo\"][df[\"enmo\"] >= 0.05].sum()\n    \n    # Process each column for different time periods\n    for col in ['enmo', 'anglez', 'light', 'battery_voltage']:\n        # Full day statistics\n        features[f\"{col}_mean\"] = df[col].mean()\n        features[f\"{col}_std\"] = df[col].std()\n        features[f\"{col}_max\"] = df[col].max()\n        features[f\"{col}_min\"] = df[col].min()\n        features[f\"{col}_diff_mean\"] = df[col].diff().mean()\n        features[f\"{col}_diff_std\"] = df[col].diff().std()\n        \n        # Night time statistics\n        night_data = df.loc[night, col]\n        features[f\"{col}_night_mean\"] = night_data.mean()\n        features[f\"{col}_night_std\"] = night_data.std()\n        features[f\"{col}_night_max\"] = night_data.max()\n        features[f\"{col}_night_min\"] = night_data.min()\n        \n        # Day time statistics\n        day_data = df.loc[day, col]\n        features[f\"{col}_day_mean\"] = day_data.mean()\n        features[f\"{col}_day_std\"] = day_data.std()\n        features[f\"{col}_day_max\"] = day_data.max()\n        features[f\"{col}_day_min\"] = day_data.min()\n    \n    return features, filename.split('=')[1]\n\ndef load_data_parquet(dirname) -> pd.DataFrame:\n    \"\"\"Load and process time series data from directory in parallel\"\"\"\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    features_list, indexes = zip(*results)\n    \n    # Create DataFrame with extracted features and IDs\n    df = pd.DataFrame(features_list)\n    df['id'] = indexes\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2025-04-23T14:18:40.810638Z","iopub.execute_input":"2025-04-23T14:18:40.810965Z","iopub.status.idle":"2025-04-23T14:18:40.820639Z","shell.execute_reply.started":"2025-04-23T14:18:40.810946Z","shell.execute_reply":"2025-04-23T14:18:40.819872Z"},"papermill":{"duration":0.01536,"end_time":"2025-03-20T18:31:44.588958","exception":false,"start_time":"2025-03-20T18:31:44.573598","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday','sii']\n\n# chia thành 3 nhóm features chính (Bộ dữ liệu khách quan csv)\ndemographicFeatures = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'sii']\n\nphycsicsFeatures = ['CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW','sii' ]\n\nbehaviorFeatures = ['PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                    'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday','sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_data_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet')\ntest_ts = load_data_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet')\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\n\nfeaturesCols += time_series_cols\n\ntrain_df = train[featuresCols]","metadata":{"execution":{"iopub.status.busy":"2025-04-23T14:18:40.821323Z","iopub.execute_input":"2025-04-23T14:18:40.821526Z","iopub.status.idle":"2025-04-23T14:19:26.941757Z","shell.execute_reply.started":"2025-04-23T14:18:40.821506Z","shell.execute_reply":"2025-04-23T14:19:26.940995Z"},"papermill":{"duration":49.572363,"end_time":"2025-03-20T18:32:34.164651","exception":false,"start_time":"2025-03-20T18:31:44.592288","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Feature Engineering","metadata":{"papermill":{"duration":0.012285,"end_time":"2025-03-20T18:32:34.196327","exception":false,"start_time":"2025-03-20T18:32:34.184042","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Loại bỏ các feature liên quan đến \"season\"\nfiltered_features = [feature for feature in featuresCols if feature not in cat_c and feature != 'sii']\n\n# Loại bỏ các hàng có giá trị NaN trong y\ntrain_df = train_df[featuresCols].dropna(subset=['sii'])\n\n# Chuẩn bị dữ liệu X và y\nX = train_df[filtered_features]\ny = train_df['sii']\n\n# Định nghĩa pipeline xử lý dữ liệu số\nnum_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='mean')),\n    ('scaler', StandardScaler())\n])\n\n# Định nghĩa ColumnTransformer để áp dụng pipeline cho các cột số\npreprocessor = ColumnTransformer(transformers=[\n    ('num', num_transformer, filtered_features)\n])\n\n# Fit và transform X\npreprocessor.fit(X)\nX_transformed = pd.DataFrame(preprocessor.transform(X), columns=filtered_features)\n\n# Kiểm tra các dòng đầu tiên của dữ liệu đã transform\nprint(\"Transformed X DataFrame:\")\nprint(X_transformed.head())","metadata":{"execution":{"iopub.status.busy":"2025-04-23T14:19:26.942613Z","iopub.execute_input":"2025-04-23T14:19:26.942859Z","iopub.status.idle":"2025-04-23T14:19:27.067687Z","shell.execute_reply.started":"2025-04-23T14:19:26.942842Z","shell.execute_reply":"2025-04-23T14:19:27.066910Z"},"papermill":{"duration":0.105186,"end_time":"2025-03-20T18:32:34.314547","exception":false,"start_time":"2025-03-20T18:32:34.209361","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"6.Split Data into Training and Testing Sets","metadata":{"papermill":{"duration":0.012677,"end_time":"2025-03-20T18:32:34.346209","exception":false,"start_time":"2025-03-20T18:32:34.333532","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(X_transformed, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2025-04-23T14:19:27.068492Z","iopub.execute_input":"2025-04-23T14:19:27.069235Z","iopub.status.idle":"2025-04-23T14:19:27.075303Z","shell.execute_reply.started":"2025-04-23T14:19:27.069209Z","shell.execute_reply":"2025-04-23T14:19:27.074728Z"},"papermill":{"duration":0.022147,"end_time":"2025-03-20T18:32:34.380430","exception":false,"start_time":"2025-03-20T18:32:34.358283","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Build and Train the Model","metadata":{"papermill":{"duration":0.013037,"end_time":"2025-03-20T18:32:34.406955","exception":false,"start_time":"2025-03-20T18:32:34.393918","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))","metadata":{"execution":{"iopub.status.busy":"2025-04-23T14:19:27.076069Z","iopub.execute_input":"2025-04-23T14:19:27.076306Z","iopub.status.idle":"2025-04-23T14:19:27.088587Z","shell.execute_reply.started":"2025-04-23T14:19:27.076287Z","shell.execute_reply":"2025-04-23T14:19:27.087911Z"},"papermill":{"duration":0.021066,"end_time":"2025-03-20T18:32:34.441184","exception":false,"start_time":"2025-03-20T18:32:34.420118","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nfrom sklearn.metrics import cohen_kappa_score, make_scorer\nfrom sklearn.model_selection import cross_val_score, StratifiedKFold\nfrom sklearn.ensemble import (\n    RandomForestRegressor,\n    GradientBoostingRegressor, \n    AdaBoostRegressor,\n    ExtraTreesRegressor,\n    BaggingRegressor,\n    StackingRegressor,\n    VotingRegressor\n)\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\nfrom xgboost import XGBRegressor\nfrom scipy.optimize import minimize\n\n# Cấu hình\nseed = 2023\nnp.random.seed(seed)\nwarnings.filterwarnings(\"ignore\")\n\n# Khởi tạo các mô hình Regressor\nLGB_Model = LGBMRegressor(random_state=seed, verbose=-1, n_estimators=300)\nCatBoost_Model = CatBoostRegressor(random_state=seed, verbose=0)\nXGB_Model = XGBRegressor(random_state=seed)\n\n# Hàm tính Quadratic Weighted Kappa\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred.round().astype(int), weights='quadratic')\n\n# Hàm tối ưu Threshold Rounding\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\n# Tạo mô hình StackingRegressor\nstacking_model = StackingRegressor(\n    estimators=[\n        ('lgb', LGB_Model),\n        ('cat', CatBoost_Model),\n        ('xgb', XGB_Model)\n    ]\n)\n\n# Danh sách mô hình\nmodels = [stacking_model]\nmodel_names = ['StackingRegressor']\n\n# Hàm đánh giá mô hình\ndef generate_baseline_results(models, model_names, X, y, cv=5, plot_results=False):\n    \"\"\"\n    Đánh giá nhiều mô hình bằng cross-validation với kỹ thuật threshold rounding.\n    \n    Parameters:\n    -----------\n    models: list - Danh sách các mô hình đã khởi tạo.\n    model_names: list - Danh sách tên các mô hình.\n    X: DataFrame - Ma trận đặc trưng.\n    y: Series - Biến mục tiêu.\n    cv: int - Số fold cho cross-validation.\n    plot_results: bool - Có vẽ đồ thị kết quả hay không.\n\n    Returns:\n    --------\n    DataFrame - Kết quả hiệu suất của các mô hình.\n    \"\"\"\n    print(f\"Đánh giá {len(models)} mô hình với {cv}-fold cross-validation...\")\n\n    kfold = StratifiedKFold(n_splits=cv, shuffle=True, random_state=42)\n    entries = []\n\n    for model, model_name in zip(models, model_names):\n        print(f\"Đang đánh giá: {model_name}\")\n        try:\n            oof_predictions = np.zeros(len(y), dtype=float)  # Lưu dự đoán liên tục\n            \n            for fold_idx, (train_idx, val_idx) in enumerate(kfold.split(X, y)):\n                X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n                y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n                model.fit(X_train, y_train)  # Train mô hình\n                y_val_pred = model.predict(X_val)  # Dự đoán hồi quy\n                \n                oof_predictions[val_idx] = y_val_pred  # Lưu lại dự đoán\n\n            # Tối ưu hóa thresholds\n            KappaOptimizer = minimize(\n                evaluate_predictions,\n                x0=[0.5, 1.5, 2.5], args=(y, oof_predictions), method='Nelder-Mead'\n            )\n            optimized_thresholds = KappaOptimizer.x\n\n            # Áp dụng thresholds đã tối ưu\n            rounded_preds = threshold_Rounder(oof_predictions, optimized_thresholds)\n\n            # Tính toán QWK\n            qwk_score = quadratic_weighted_kappa(y, rounded_preds)\n\n            entries.append((model_name, qwk_score))\n            print(f\"  • {model_name} - QWK Score: {qwk_score:.4f}\")\n        \n        except Exception as e:\n            print(f\"  • Lỗi với {model_name}: {e}\")\n\n    # Tạo DataFrame kết quả\n    results_df = pd.DataFrame(entries, columns=['model_name', 'qwk_score'])\n    results_df.sort_values(by='qwk_score', ascending=False, inplace=True)\n\n    # Vẽ biểu đồ kết quả\n    if plot_results and not results_df.empty:\n        plt.figure(figsize=(12, 6))\n        sns.barplot(x='model_name', y='qwk_score', data=results_df, color='lightblue', edgecolor='black')\n        plt.title(\"Hiệu suất mô hình (QWK Score)\", fontsize=14)\n        plt.xlabel(\"Mô hình\", fontsize=12)\n        plt.ylabel(\"Quadratic Weighted Kappa\", fontsize=12)\n        plt.xticks(rotation=45, ha='right')\n        plt.grid(axis='y', linestyle='--', alpha=0.7)\n        plt.tight_layout()\n        plt.show()\n\n    print(\"\\n===== KẾT QUẢ CUỐI CÙNG =====\")\n    print(results_df)\n    return results_df\n\n# Sử dụng hàm để chạy toàn bộ đánh giá\nprint(\"Bắt đầu đánh giá mô hình...\\n\")\ncv_results = generate_baseline_results(models, model_names, X_train, y_train, cv=5, plot_results=True)\n","metadata":{"execution":{"iopub.status.busy":"2025-04-23T14:19:27.089675Z","iopub.execute_input":"2025-04-23T14:19:27.089953Z","iopub.status.idle":"2025-04-23T14:24:46.201429Z","shell.execute_reply.started":"2025-04-23T14:19:27.089929Z","shell.execute_reply":"2025-04-23T14:24:46.200657Z"},"papermill":{"duration":332.264908,"end_time":"2025-03-20T18:38:06.719169","exception":false,"start_time":"2025-03-20T18:32:34.454261","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Hyperparameter turning","metadata":{"papermill":{"duration":0.012847,"end_time":"2025-03-20T18:39:14.417881","exception":false,"start_time":"2025-03-20T18:39:14.405034","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Tạo cell mới để đánh giá mô hình đơn lẻ\nimport numpy as np\nfrom sklearn.metrics import cohen_kappa_score\nfrom scipy.optimize import minimize\n\n# Định nghĩa hàm để đánh giá mô hình đơn lẻ\ndef evaluate_single_models(X_train, y_train, X_val, y_val, verbose=True):\n    \"\"\"\n    Đánh giá hiệu suất của các mô hình đơn lẻ trước khi kết hợp.\n    \n    Tham số:\n    ---------\n    X_train: DataFrame - Dữ liệu huấn luyện\n    y_train: Series - Nhãn huấn luyện\n    X_val: DataFrame - Dữ liệu kiểm tra\n    y_val: Series - Nhãn kiểm tra\n    verbose: bool - In kết quả chi tiết hay không\n    \n    Trả về:\n    --------\n    dict - Từ điển chứa mô hình tốt nhất và điểm QWK của nó\n    \"\"\"\n    # Khởi tạo các mô hình cơ bản\n    models = {\n        'LightGBM': LGBMRegressor(random_state=seed, verbose=-1, n_estimators=300),\n        'XGBoost': XGBRegressor(random_state=seed),\n        'CatBoost': CatBoostRegressor(random_state=seed, verbose=0),\n        'RandomForest': RandomForestRegressor(random_state=seed),\n        'GradientBoosting': GradientBoostingRegressor(random_state=seed)\n    }\n    \n    results = {}\n    \n    if verbose:\n        print(\"===== ĐÁNH GIÁ MÔ HÌNH ĐƠN LẺ =====\")\n    \n    # Huấn luyện và đánh giá từng mô hình\n    for name, model in models.items():\n        if verbose:\n            print(f\"Đang huấn luyện: {name}...\")\n        \n        # Huấn luyện mô hình\n        model.fit(X_train, y_train)\n        \n        # Dự đoán trên tập validation\n        val_preds = model.predict(X_val)\n        \n        # Tối ưu ngưỡng cho QWK\n        optimizer = minimize(\n            evaluate_predictions,\n            x0=[0.5, 1.5, 2.5], \n            args=(y_val, val_preds),\n            method='Nelder-Mead'\n        )\n        optimized_thresholds = optimizer.x\n        \n        # Áp dụng ngưỡng đã tối ưu\n        rounded_preds = threshold_Rounder(val_preds, optimized_thresholds)\n        \n        # Tính điểm QWK\n        qwk = quadratic_weighted_kappa(y_val, rounded_preds)\n        results[name] = {\n            'model': model,\n            'score': qwk,\n            'thresholds': optimized_thresholds\n        }\n        \n        if verbose:\n            print(f\"  • {name} - QWK Score: {qwk:.4f}\")\n    \n    # Sắp xếp kết quả theo điểm số\n    sorted_results = sorted(results.items(), key=lambda x: x[1]['score'], reverse=True)\n    best_model_name, best_model_info = sorted_results[0]\n    \n    if verbose:\n        print(\"\\n===== KẾT QUẢ CUỐI CÙNG =====\")\n        print(f\"Mô hình tốt nhất: {best_model_name} với QWK Score: {best_model_info['score']:.4f}\")\n    \n    return {\n        'best_model_name': best_model_name,\n        'best_model': best_model_info['model'],\n        'best_score': best_model_info['score'],\n        'best_thresholds': best_model_info['thresholds'],\n        'all_results': results\n    }\n\n# Chạy đánh giá\nsingle_model_results = evaluate_single_models(X_train, y_train, X_val, y_val)\n\n# Lấy top 3 mô hình tốt nhất để kết hợp\ntop_models = sorted(single_model_results['all_results'].items(), \n                    key=lambda x: x[1]['score'], \n                    reverse=True)[:3]\n\nprint(\"\\n===== TOP 3 MÔ HÌNH TỐT NHẤT =====\")\nfor i, (name, info) in enumerate(top_models, 1):\n    print(f\"{i}. {name}: {info['score']:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T14:24:46.203361Z","iopub.execute_input":"2025-04-23T14:24:46.203631Z","iopub.status.idle":"2025-04-23T14:25:06.378182Z","shell.execute_reply.started":"2025-04-23T14:24:46.203610Z","shell.execute_reply":"2025-04-23T14:25:06.377396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell mới với không gian tìm kiếm mở rộng cho từng loại mô hình\ndef objective_extended(trial, X_data, y_data, model_type):\n    \"\"\"\n    Hàm mục tiêu Optuna với không gian tìm kiếm hyperparameter mở rộng cho một loại mô hình cụ thể.\n    \n    Tham số:\n    ---------\n    trial: Đối tượng trial\n    X_data: Dữ liệu đặc trưng\n    y_data: Dữ liệu nhãn\n    model_type: Loại mô hình cần tối ưu\n    \n    Trả về:\n    --------\n    float: Điểm QWK trung bình\n    \"\"\"\n    if model_type == 'LightGBM':\n        params = {\n            'objective': trial.suggest_categorical('objective', \n                        ['regression', 'poisson', 'tweedie', 'mape', 'huber', 'quantile']),\n            'n_estimators': trial.suggest_int('n_estimators', 50, 1000),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 0.001, 0.1),\n            'max_depth': trial.suggest_int('max_depth', 2, 15),\n            'num_leaves': trial.suggest_int('num_leaves', 20, 200),\n            'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n            'min_child_samples': trial.suggest_int('min_child_samples', 5, 100),\n            'reg_alpha': trial.suggest_loguniform('reg_alpha', 1e-8, 1e-1),\n            'reg_lambda': trial.suggest_loguniform('reg_lambda', 1e-8, 1e-1),\n            'random_state': seed,\n            'verbose': -1\n        }\n        if params['objective'] == 'tweedie':\n            params['tweedie_variance_power'] = trial.suggest_float('tweedie_variance_power', 1.1, 1.9)\n        elif params['objective'] == 'quantile':\n            params['alpha'] = trial.suggest_float('alpha', 0.05, 0.95)\n        elif params['objective'] == 'huber':\n            params['alpha'] = trial.suggest_float('huber_alpha', 0.5, 0.95)\n        model = LGBMRegressor(**params)\n\n    elif model_type == 'XGBoost':\n        params = {\n            'objective': trial.suggest_categorical('objective', \n                        ['reg:squarederror', 'reg:pseudohubererror', 'reg:tweedie']),\n            'n_estimators': trial.suggest_int('n_estimators', 50, 1000),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 0.001, 0.1),\n            'max_depth': trial.suggest_int('max_depth', 2, 15),\n            'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n            'reg_alpha': trial.suggest_loguniform('reg_alpha', 1e-8, 1e-1),\n            'reg_lambda': trial.suggest_loguniform('reg_lambda', 1e-8, 1e-1),\n            'gamma': trial.suggest_float('gamma', 0, 10),\n            'min_child_weight': trial.suggest_int('min_child_weight', 1, 10),\n            'random_state': seed\n        }\n        if params['objective'] == 'reg:tweedie':\n            params['tweedie_variance_power'] = trial.suggest_float('tweedie_variance_power', 1.1, 1.9)\n        model = XGBRegressor(**params)\n\n    elif model_type == 'CatBoost':\n        # Định nghĩa loss_function trước\n        loss_function = trial.suggest_categorical('loss_function', \n                        ['RMSE', 'MAE', 'Poisson', 'Huber', 'Tweedie', 'Quantile'])\n        \n        # Xử lý các tham số đặc biệt cho các loss function\n        if loss_function == 'Huber':\n            delta = trial.suggest_float('huber_delta', 0.1, 10.0)\n            loss_function = f'Huber:delta={delta}'\n        elif loss_function == 'Tweedie':\n            variance_power = trial.suggest_float('tweedie_variance_power', 1.1, 1.9)\n            loss_function = f'Tweedie:variance_power={variance_power}'\n        elif loss_function == 'Quantile':\n            q = trial.suggest_float('quantile_q', 0.1, 0.9)\n            loss_function = f'Quantile:alpha={q}'\n        \n        # Tạo tham số cho CatBoost sau khi đã xử lý loss_function\n        params = {\n            'loss_function': loss_function,\n            'iterations': trial.suggest_int('iterations', 50, 1000),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 0.001, 0.1),\n            'depth': trial.suggest_int('depth', 2, 10),\n            'l2_leaf_reg': trial.suggest_loguniform('l2_leaf_reg', 1e-8, 1e-1),\n            'random_strength': trial.suggest_float('random_strength', 1e-8, 10),\n            'bagging_temperature': trial.suggest_float('bagging_temperature', 0, 10),\n            'od_type': trial.suggest_categorical('od_type', ['IncToDec', 'Iter']),\n            'od_wait': trial.suggest_int('od_wait', 10, 50),\n            'random_state': seed,\n            'verbose': 0\n        }\n        \n        model = CatBoostRegressor(**params)\n\n    elif model_type == 'RandomForest':\n        params = {\n            'n_estimators': trial.suggest_int('n_estimators', 50, 500),\n            'max_depth': trial.suggest_int('max_depth', 2, 30),\n            'min_samples_split': trial.suggest_int('min_samples_split', 2, 20),\n            'min_samples_leaf': trial.suggest_int('min_samples_leaf', 1, 20),\n            'max_features': trial.suggest_categorical('max_features', ['sqrt', 'log2', None]),\n            'bootstrap': trial.suggest_categorical('bootstrap', [True, False]),\n            'random_state': seed\n        }\n        model = RandomForestRegressor(**params)\n\n    elif model_type == 'GradientBoosting':\n        params = {\n            'n_estimators': trial.suggest_int('n_estimators', 50, 500),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 0.001, 0.1),\n            'max_depth': trial.suggest_int('max_depth', 2, 10),\n            'min_samples_split': trial.suggest_int('min_samples_split', 2, 20),\n            'min_samples_leaf': trial.suggest_int('min_samples_leaf', 1, 20),\n            'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n            'max_features': trial.suggest_categorical('max_features', ['sqrt', 'log2', None]),\n            'random_state': seed\n        }\n        model = GradientBoostingRegressor(**params)\n    \n    # Đánh giá mô hình\n    kfold = StratifiedKFold(n_splits=5, shuffle=True, random_state=seed)\n    scores = []\n\n    for train_idx, val_idx in kfold.split(X_data, y_data):\n        X_train_fold, X_val_fold = X_data.iloc[train_idx], X_data.iloc[val_idx]\n        y_train_fold, y_val_fold = y_data.iloc[train_idx], y_data.iloc[val_idx]\n\n        model.fit(X_train_fold, y_train_fold)\n        val_preds = model.predict(X_val_fold)\n\n        optimizer = minimize(\n            evaluate_predictions,\n            x0=[0.5, 1.5, 2.5], \n            args=(y_val_fold, val_preds),\n            method='Nelder-Mead'\n        )\n        optimized_thresholds = optimizer.x\n        rounded_preds = threshold_Rounder(val_preds, optimized_thresholds)\n        qwk = quadratic_weighted_kappa(y_val_fold, rounded_preds)\n        scores.append(qwk)\n\n    return np.mean(scores)\n\ndef run_extended_optimization(X_data, y_data, n_trials=50):\n    \"\"\"\n    Chạy qúa trình tối ưu hóa cho mỗi loại mô hình riêng biệt.\n    \n    Tham số:\n    ---------\n    X_data: Dữ liệu đặc trưng\n    y_data: Dữ liệu nhãn\n    n_trials: Số lượng trials cho mỗi mô hình\n    \n    Trả về:\n    --------\n    dict: Thông tin về các mô hình tốt nhất cho mỗi loại\n    \"\"\"\n    print(f\"Bắt đầu quá trình tối ưu hóa hyperparameter với {n_trials} trials cho mỗi mô hình...\")\n    \n    # Danh sách các loại mô hình cần tối ưu\n    model_types = ['LightGBM', 'XGBoost', 'CatBoost', 'RandomForest', 'GradientBoosting']\n    \n    # Dictionary lưu kết quả tối ưu cho từng loại mô hình\n    results = {}\n    \n    for model_type in model_types:\n        print(f\"\\n{'='*50}\")\n        print(f\"Tối ưu hóa mô hình {model_type}\")\n        print(f\"{'='*50}\")\n        \n        # Tạo nghiên cứu mới cho mỗi loại mô hình\n        study = optuna.create_study(\n            direction=\"maximize\",\n            sampler=TPESampler(seed=seed)\n        )\n        \n        # Chạy tối ưu hóa\n        study.optimize(\n            lambda trial: objective_extended(trial, X_data, y_data, model_type),\n            n_trials=n_trials\n        )\n        \n        # In kết quả\n        print(f\"\\n===== KẾT QUẢ TỐI ƯU CHO {model_type} =====\")\n        print(f\"Điểm QWK tốt nhất: {study.best_value:.4f}\")\n        \n        # Tạo mô hình tốt nhất với hyperparameters tối ưu\n        best_params = study.best_params\n        \n        if model_type == 'LightGBM':\n            best_model = LGBMRegressor(**best_params)\n        elif model_type == 'XGBoost':\n            best_model = XGBRegressor(**best_params)\n        elif model_type == 'CatBoost':\n            best_model = CatBoostRegressor(**best_params)\n        elif model_type == 'RandomForest':\n            best_model = RandomForestRegressor(**best_params)\n        elif model_type == 'GradientBoosting':\n            best_model = GradientBoostingRegressor(**best_params)\n        \n        # Lưu thông tin vào dictionary kết quả\n        results[model_type] = {\n            'best_model': best_model,\n            'best_score': study.best_value,\n            'best_params': best_params\n        }\n        \n        # In tham số chi tiết\n        print(\"\\nTham số tối ưu:\")\n        for param_name, param_value in best_params.items():\n            print(f\"  • {param_name}: {param_value}\")\n    \n    # Tìm mô hình tốt nhất trong tất cả các loại\n    best_model_type = max(results.items(), key=lambda x: x[1]['best_score'])[0]\n    best_overall = results[best_model_type]\n    \n    print(\"\\n\\n================================================================\")\n    print(f\"MÔ HÌNH TỐT NHẤT TỔNG THỂ: {best_model_type} với QWK score: {best_overall['best_score']:.4f}\")\n    print(\"================================================================\")\n    \n    # Tạo bảng so sánh các mô hình\n    comparison_data = []\n    for model_type, info in results.items():\n        comparison_data.append({\n            'Model': model_type,\n            'QWK Score': info['best_score']\n        })\n    \n    comparison_df = pd.DataFrame(comparison_data)\n    comparison_df = comparison_df.sort_values('QWK Score', ascending=False).reset_index(drop=True)\n    \n    print(\"\\nSo sánh hiệu suất các mô hình:\")\n    print(comparison_df)\n    \n    # Vẽ biểu đồ so sánh\n    plt.figure(figsize=(12, 6))\n    sns.barplot(x='Model', y='QWK Score', data=comparison_df, palette='viridis')\n    plt.title('Hiệu suất các mô hình sau khi tối ưu hóa', fontsize=14)\n    plt.xlabel('Loại mô hình', fontsize=12)\n    plt.ylabel('Điểm QWK', fontsize=12)\n    plt.grid(axis='y', linestyle='--', alpha=0.7)\n    plt.xticks(rotation=45)\n    plt.tight_layout()\n    plt.show()\n    \n    # Thêm thông tin mô hình tốt nhất tổng thể vào kết quả\n    results['best_overall'] = {\n        'model_type': best_model_type,\n        'best_model': best_overall['best_model'],\n        'best_score': best_overall['best_score'],\n        'best_params': best_overall['best_params']\n    }\n    \n    return results\n\n# Chạy tối ưu hóa mở rộng cho từng loại mô hình\nextended_results = run_extended_optimization(X_train, y_train, n_trials=150)  # Điều chỉnh n_trials nếu cần","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-23T14:25:06.379131Z","iopub.execute_input":"2025-04-23T14:25:06.379433Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Make Predictions on Test Set","metadata":{"papermill":{"duration":0.013151,"end_time":"2025-03-20T18:39:14.443985","exception":false,"start_time":"2025-03-20T18:39:14.430834","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Cell mới để kết hợp mô hình tốt nhất\ndef combine_best_models(single_results, extended_results, X_train, y_train, X_val, y_val):\n    \"\"\"\n    Kết hợp các mô hình tốt nhất để tạo ensemble.\n    \n    Tham số:\n    ---------\n    single_results: Kết quả từ đánh giá mô hình đơn lẻ\n    extended_results: Kết quả từ tối ưu mở rộng\n    X_train, y_train: Dữ liệu huấn luyện\n    X_val, y_val: Dữ liệu kiểm tra\n    \n    Trả về:\n    --------\n    dict: Thông tin về mô hình ensemble tốt nhất\n    \"\"\"\n    print(\"===== KẾT HỢP CÁC MÔ HÌNH TỐT NHẤT =====\")\n    \n    # Tạo mô hình từ đánh giá đơn lẻ\n    best_single_model = single_results['best_model']\n    best_single_name = single_results['best_model_name']\n    \n    # Tạo mô hình từ tối ưu hóa mở rộng\n    # Truy cập đúng mô hình từ 'best_overall' trong extended_results\n    best_extended_type = extended_results['best_overall']['model_type']\n    best_extended_model = extended_results[best_extended_type]['best_model']\n    \n    # Thêm các mô hình top vào danh sách\n    models_to_combine = []\n    model_weights = []\n    \n    # Thêm mô hình tốt nhất từ đánh giá đơn lẻ\n    models_to_combine.append((f'best_single_{best_single_name}', best_single_model))\n    model_weights.append(1.0)\n    \n    # Thêm mô hình tốt nhất từ tối ưu mở rộng\n    models_to_combine.append((f'best_extended_{best_extended_type}', best_extended_model))\n    model_weights.append(1.0)\n    \n    # Thêm các mô hình top từ đánh giá đơn lẻ\n    for i, (name, info) in enumerate(top_models[:2], 1):  # Chỉ lấy 2 mô hình top\n        if name != best_single_name:  # Tránh trùng lặp\n            models_to_combine.append((f'top_{i}_{name}', info['model']))\n            model_weights.append(0.8)  # Trọng số thấp hơn mô hình tốt nhất\n    \n    # Tạo VotingRegressor\n    voting_model = VotingRegressor(\n        estimators=models_to_combine,\n        weights=model_weights\n    )\n    \n    # Đánh giá ensemble\n    print(\"Đang đánh giá ensemble model...\")\n    voting_model.fit(X_train, y_train)\n    val_preds = voting_model.predict(X_val)\n    \n    # Tối ưu ngưỡng \n    optimizer = minimize(\n        evaluate_predictions,\n        x0=[0.5, 1.5, 2.5], \n        args=(y_val, val_preds),\n        method='Nelder-Mead'\n    )\n    optimized_thresholds = optimizer.x\n    \n    # Áp dụng ngưỡng tối ưu\n    rounded_preds = threshold_Rounder(val_preds, optimized_thresholds)\n    \n    # Tính QWK\n    ensemble_qwk = quadratic_weighted_kappa(y_val, rounded_preds)\n    print(f\"Ensemble Model - QWK Score: {ensemble_qwk:.4f}\")\n    \n    # So sánh với các mô hình đơn lẻ\n    print(\"\\n===== SO SÁNH HIỆU SUẤT =====\")\n    print(f\"Mô hình đơn lẻ tốt nhất ({best_single_name}): {single_results['best_score']:.4f}\")\n    # Sử dụng điểm số từ best_overall\n    print(f\"Mô hình tối ưu mở rộng ({best_extended_type}): {extended_results['best_overall']['best_score']:.4f}\")\n    print(f\"Ensemble Model: {ensemble_qwk:.4f}\")\n    \n    # Tạo StackingRegressor (kết hợp khác)\n    meta_regressor = RandomForestRegressor(n_estimators=100, random_state=seed)\n    stacking_model = StackingRegressor(\n        estimators=models_to_combine,\n        final_estimator=meta_regressor\n    )\n    \n    # Đánh giá stacking\n    print(\"\\nĐang đánh giá stacking model...\")\n    stacking_model.fit(X_train, y_train)\n    stack_val_preds = stacking_model.predict(X_val)\n    \n    # Tối ưu ngưỡng cho stacking\n    stack_optimizer = minimize(\n        evaluate_predictions,\n        x0=[0.5, 1.5, 2.5], \n        args=(y_val, stack_val_preds),\n        method='Nelder-Mead'\n    )\n    stack_thresholds = stack_optimizer.x\n    \n    # Áp dụng ngưỡng tối ưu cho stacking\n    stack_rounded_preds = threshold_Rounder(stack_val_preds, stack_thresholds)\n    \n    # Tính QWK cho stacking\n    stack_qwk = quadratic_weighted_kappa(y_val, stack_rounded_preds)\n    print(f\"Stacking Model - QWK Score: {stack_qwk:.4f}\")\n    \n    # Sửa lại cách lấy điểm số của mô hình tối ưu mở rộng\n    best_extended_score = extended_results['best_overall']['best_score']\n    \n    # Xác định mô hình tốt nhất cuối cùng\n    best_ensemble = None\n    best_ensemble_thresholds = None\n    best_ensemble_type = \"\"\n    best_ensemble_score = 0\n    \n    if ensemble_qwk > stack_qwk and ensemble_qwk > single_results['best_score'] and ensemble_qwk > best_extended_score:\n        best_ensemble = voting_model\n        best_ensemble_thresholds = optimized_thresholds\n        best_ensemble_type = \"VotingRegressor\"\n        best_ensemble_score = ensemble_qwk\n    elif stack_qwk > ensemble_qwk and stack_qwk > single_results['best_score'] and stack_qwk > best_extended_score:\n        best_ensemble = stacking_model\n        best_ensemble_thresholds = stack_thresholds\n        best_ensemble_type = \"StackingRegressor\"\n        best_ensemble_score = stack_qwk\n    elif single_results['best_score'] > best_extended_score:\n        best_ensemble = best_single_model\n        best_ensemble_thresholds = single_results['best_thresholds']\n        best_ensemble_type = f\"Single Model ({best_single_name})\"\n        best_ensemble_score = single_results['best_score']\n    else:\n        best_ensemble = best_extended_model\n        # Tính ngưỡng tối ưu cho mô hình này\n        best_extended_model.fit(X_train, y_train)\n        ext_preds = best_extended_model.predict(X_val)\n        ext_optimizer = minimize(\n            evaluate_predictions,\n            x0=[0.5, 1.5, 2.5], \n            args=(y_val, ext_preds),\n            method='Nelder-Mead'\n        )\n        best_ensemble_thresholds = ext_optimizer.x\n        best_ensemble_type = f\"Extended Optimized Model ({best_extended_type})\"\n        best_ensemble_score = best_extended_score\n    \n    print(\"\\n===== MÔ HÌNH TỐT NHẤT CUỐI CÙNG =====\")\n    print(f\"Loại: {best_ensemble_type}\")\n    print(f\"Điểm QWK: {best_ensemble_score:.4f}\")\n    \n    return {\n        'model': best_ensemble,\n        'thresholds': best_ensemble_thresholds,\n        'model_type': best_ensemble_type,\n        'score': best_ensemble_score\n    }\n\n# Kết hợp các mô hình tốt nhất\nfinal_model_info = combine_best_models(single_model_results, extended_results, \n                                     X_train, y_train, X_val, y_val)","metadata":{"papermill":{"duration":0.013269,"end_time":"2025-03-20T18:39:14.470922","exception":false,"start_time":"2025-03-20T18:39:14.457653","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extended Model Submission\ndef make_extended_model_predictions(X_train, y_train, X_val, y_val, X_test_transformed, test, extended_results):\n    \"\"\"\n    Tạo dự đoán và file submission cho Extended model (với hyperparameter tuning)\n    \"\"\"\n    print(\"\\n===== ĐANG TẠO DỰ ĐOÁN CHO EXTENDED MODEL =====\")\n    \n    # Lấy mô hình tốt nhất từ quá trình tối ưu hóa mở rộng\n    best_model_type = extended_results['best_overall']['model_type']\n    best_model = extended_results[best_model_type]['best_model']\n    \n    # Huấn luyện lại trên toàn bộ dữ liệu\n    print(f\"Đang huấn luyện extended model ({best_model_type})...\")\n    best_model.fit(X_train, y_train)\n    \n    # Đánh giá trên tập validation\n    val_preds = best_model.predict(X_val)\n    \n    # Tìm ngưỡng tối ưu\n    optimizer = minimize(\n        evaluate_predictions,\n        x0=[0.5, 1.5, 2.5], \n        args=(y_val, val_preds),\n        method='Nelder-Mead'\n    )\n    extended_thresholds = optimizer.x\n    \n    # Tính QWK score trên validation\n    rounded_val_preds = threshold_Rounder(val_preds, extended_thresholds)\n    extended_qwk = quadratic_weighted_kappa(y_val, rounded_val_preds)\n    \n    # Dự đoán trên tập test\n    test_preds = best_model.predict(X_test_transformed)\n    \n    # Áp dụng ngưỡng đã tối ưu\n    rounded_test_preds = threshold_Rounder(test_preds, extended_thresholds)\n    \n    # Tạo DataFrame kết quả\n    submission = pd.DataFrame({\n        'id': test['id'],\n        'sii': rounded_test_preds.astype(int)\n    })\n    \n    # Phân bố dự đoán\n    pred_distribution = pd.Series(rounded_test_preds).value_counts().sort_index()\n    print(f\"\\n===== PHÂN BỐ DỰ ĐOÁN CHO EXTENDED MODEL ({best_model_type}) =====\")\n    print(pred_distribution)\n    \n    # Lưu file kết quả - Dùng tên chuẩn cho Kaggle\n    submission_file = 'submission.csv'\n    submission.to_csv(submission_file, index=False)\n    print(f\"\\nĐã lưu kết quả vào file: {submission_file}\")\n    \n    print(f\"\\nĐiểm QWK trên tập validation: {extended_qwk:.4f}\")\n    print(f\"Ngưỡng tối ưu: {extended_thresholds}\")\n    \n    return submission\n\n# Tạo dự đoán với extended model\ntest_df = test[filtered_features]\nX_test_transformed = pd.DataFrame(preprocessor.transform(test_df), columns=filtered_features)\nextended_submission = make_extended_model_predictions(X_train, y_train, X_val, y_val, X_test_transformed, test, extended_results)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}