{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib.gridspec as gridspec\nimport pandas as pd\nimport os\nimport warnings\nfrom sklearn import preprocessing\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler, OneHotEncoder\nfrom sklearn.metrics import mean_squared_error, cohen_kappa_score, accuracy_score, r2_score\nfrom sklearn.model_selection import train_test_split, StratifiedGroupKFold, cross_val_score, KFold, StratifiedKFold\nfrom scipy.optimize import minimize\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.decomposition import PCA\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.ensemble import RandomForestRegressor, ExtraTreesRegressor, StackingRegressor\nfrom sklearn.linear_model import LinearRegression\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm.notebook import tqdm\nfrom statsmodels.stats.outliers_influence import variance_inflation_factor\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor ","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:19:11.991180Z","iopub.execute_input":"2024-10-15T08:19:11.991730Z","iopub.status.idle":"2024-10-15T08:19:12.002536Z","shell.execute_reply.started":"2024-10-15T08:19:11.991683Z","shell.execute_reply":"2024-10-15T08:19:12.001265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Loading","metadata":{}},{"cell_type":"code","source":"#Load Data \ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\ndata_dictionary = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-15T07:58:37.929397Z","iopub.execute_input":"2024-10-15T07:58:37.929851Z","iopub.status.idle":"2024-10-15T07:58:37.994657Z","shell.execute_reply.started":"2024-10-15T07:58:37.929809Z","shell.execute_reply":"2024-10-15T07:58:37.993694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_file(filename, dirname):\n    data = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    data.drop('step', axis=1, inplace=True)\n    return data.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname):\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    stats, indexes = zip(*results)\n    data = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    data['id'] = indexes\n    return data","metadata":{"execution":{"iopub.status.busy":"2024-10-15T07:59:29.667041Z","iopub.execute_input":"2024-10-15T07:59:29.668069Z","iopub.status.idle":"2024-10-15T07:59:29.677160Z","shell.execute_reply.started":"2024-10-15T07:59:29.668015Z","shell.execute_reply":"2024-10-15T07:59:29.676007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_parquet = load_time_series('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet')\ntest_parquet = load_time_series('/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet')","metadata":{"execution":{"iopub.status.busy":"2024-10-15T07:59:32.619578Z","iopub.execute_input":"2024-10-15T07:59:32.620272Z","iopub.status.idle":"2024-10-15T08:01:11.215476Z","shell.execute_reply.started":"2024-10-15T07:59:32.620229Z","shell.execute_reply":"2024-10-15T08:01:11.214351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preview data","metadata":{}},{"cell_type":"code","source":"if 'sii' in train.columns:\n  print(\"Cột 'sii' có trong tập train.\")\nelse:\n  print(\"Cột 'sii' không có trong tập train.\")\n\nif 'sii' in test.columns:\n  print(\"Cột 'sii' có trong tập test.\")\nelse:\n  print(\"Cột 'sii' không có trong tập test.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:06:30.590448Z","iopub.execute_input":"2024-10-15T08:06:30.590949Z","iopub.status.idle":"2024-10-15T08:06:30.599179Z","shell.execute_reply.started":"2024-10-15T08:06:30.590904Z","shell.execute_reply":"2024-10-15T08:06:30.597554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'sii' in train_parquet.columns:\n  print(\"Cột 'sii' có trong tập train_parquet.\")\nelse:\n  print(\"Cột 'sii' không có trong tập train_parquet.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:06:32.597650Z","iopub.execute_input":"2024-10-15T08:06:32.598193Z","iopub.status.idle":"2024-10-15T08:06:32.604767Z","shell.execute_reply.started":"2024-10-15T08:06:32.598127Z","shell.execute_reply":"2024-10-15T08:06:32.603799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"warnings.filterwarnings('ignore')\nsns.set(style=\"whitegrid\")\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:06:45.999585Z","iopub.execute_input":"2024-10-15T08:06:46.000001Z","iopub.status.idle":"2024-10-15T08:06:46.011130Z","shell.execute_reply.started":"2024-10-15T08:06:45.999963Z","shell.execute_reply":"2024-10-15T08:06:46.009953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train.head(10))\nprint(f\"Train shape : {train.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:07:32.230757Z","iopub.execute_input":"2024-10-15T08:07:32.231231Z","iopub.status.idle":"2024-10-15T08:07:32.290641Z","shell.execute_reply.started":"2024-10-15T08:07:32.231189Z","shell.execute_reply":"2024-10-15T08:07:32.289262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.info())","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:11:00.763622Z","iopub.execute_input":"2024-10-15T08:11:00.764088Z","iopub.status.idle":"2024-10-15T08:11:00.805954Z","shell.execute_reply.started":"2024-10-15T08:11:00.764043Z","shell.execute_reply":"2024-10-15T08:11:00.804761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.describe())","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:11:10.800899Z","iopub.execute_input":"2024-10-15T08:11:10.801350Z","iopub.status.idle":"2024-10-15T08:11:10.978092Z","shell.execute_reply.started":"2024-10-15T08:11:10.801307Z","shell.execute_reply":"2024-10-15T08:11:10.976897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(test)\nprint(f\"Test shape : {test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:07:53.537577Z","iopub.execute_input":"2024-10-15T08:07:53.538025Z","iopub.status.idle":"2024-10-15T08:07:53.588973Z","shell.execute_reply.started":"2024-10-15T08:07:53.537984Z","shell.execute_reply":"2024-10-15T08:07:53.587848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cols = set(train.columns)\ntest_cols = set(test.columns)\ncols_not_in_test = list(train_cols - test_cols)\ndata_dictionary[data_dictionary['Field'].isin(cols_not_in_test)]","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:09:49.244611Z","iopub.execute_input":"2024-10-15T08:09:49.245490Z","iopub.status.idle":"2024-10-15T08:09:49.281085Z","shell.execute_reply.started":"2024-10-15T08:09:49.245422Z","shell.execute_reply":"2024-10-15T08:09:49.279836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pciat_cols = [col for col in train.columns if col.startswith('PCIAT')]\ntrain[pciat_cols].head(1000)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:10:12.081364Z","iopub.execute_input":"2024-10-15T08:10:12.081896Z","iopub.status.idle":"2024-10-15T08:10:12.141727Z","shell.execute_reply.started":"2024-10-15T08:10:12.081852Z","shell.execute_reply":"2024-10-15T08:10:12.140398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_data = train[pciat_cols].isna().mean() * 100\nprint(missing_data)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:10:21.171901Z","iopub.execute_input":"2024-10-15T08:10:21.172355Z","iopub.status.idle":"2024-10-15T08:10:21.184211Z","shell.execute_reply.started":"2024-10-15T08:10:21.172310Z","shell.execute_reply":"2024-10-15T08:10:21.182984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_pciat = train[train[pciat_cols].isna().any(axis=1)]\nprint(missing_pciat[pciat_cols].head(10)) \nnum_missing = missing_pciat.shape[0]\nprint(f'Số lượng đối tượng thiếu ít nhất một giá trị PCIAT: {num_missing}')","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:10:43.551788Z","iopub.execute_input":"2024-10-15T08:10:43.552263Z","iopub.status.idle":"2024-10-15T08:10:43.587406Z","shell.execute_reply.started":"2024-10-15T08:10:43.552219Z","shell.execute_reply":"2024-10-15T08:10:43.586228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = train.columns.tolist()\nprint(features)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:13:24.819985Z","iopub.execute_input":"2024-10-15T08:13:24.820978Z","iopub.status.idle":"2024-10-15T08:13:24.827079Z","shell.execute_reply.started":"2024-10-15T08:13:24.820927Z","shell.execute_reply":"2024-10-15T08:13:24.825850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train[\"sii\"].head(20))","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:13:34.415807Z","iopub.execute_input":"2024-10-15T08:13:34.416268Z","iopub.status.idle":"2024-10-15T08:13:34.424687Z","shell.execute_reply.started":"2024-10-15T08:13:34.416223Z","shell.execute_reply":"2024-10-15T08:13:34.423330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pciat_min_max = train.groupby('sii')['PCIAT-PCIAT_Total'].agg(['min', 'max'])\npciat_min_max = pciat_min_max.rename(\n    columns={'min': 'Minimum PCIAT total Score', 'max': 'Maximum total PCIAT Score'}\n)\npciat_min_max","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:14:12.805445Z","iopub.execute_input":"2024-10-15T08:14:12.806408Z","iopub.status.idle":"2024-10-15T08:14:12.833009Z","shell.execute_reply.started":"2024-10-15T08:14:12.806358Z","shell.execute_reply":"2024-10-15T08:14:12.831933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dictionary[data_dictionary['Field'] == 'PCIAT-PCIAT_Total']['Value Labels'].iloc[0]","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:14:26.544973Z","iopub.execute_input":"2024-10-15T08:14:26.546218Z","iopub.status.idle":"2024-10-15T08:14:26.559394Z","shell.execute_reply.started":"2024-10-15T08:14:26.546151Z","shell.execute_reply":"2024-10-15T08:14:26.557998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_parquet.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:14:48.646562Z","iopub.execute_input":"2024-10-15T08:14:48.647659Z","iopub.status.idle":"2024-10-15T08:14:48.682850Z","shell.execute_reply.started":"2024-10-15T08:14:48.647593Z","shell.execute_reply":"2024-10-15T08:14:48.681501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_parquet.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:14:50.974166Z","iopub.execute_input":"2024-10-15T08:14:50.974633Z","iopub.status.idle":"2024-10-15T08:14:51.004683Z","shell.execute_reply.started":"2024-10-15T08:14:50.974586Z","shell.execute_reply":"2024-10-15T08:14:51.003384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"train = pd.merge(train, train_parquet, how=\"left\", on='id')\ntest = pd.merge(test, test_parquet, how=\"left\", on='id')","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:15:18.267452Z","iopub.execute_input":"2024-10-15T08:15:18.267927Z","iopub.status.idle":"2024-10-15T08:15:18.302051Z","shell.execute_reply.started":"2024-10-15T08:15:18.267881Z","shell.execute_reply":"2024-10-15T08:15:18.300887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:15:37.449645Z","iopub.execute_input":"2024-10-15T08:15:37.450101Z","iopub.status.idle":"2024-10-15T08:15:37.483081Z","shell.execute_reply.started":"2024-10-15T08:15:37.450058Z","shell.execute_reply":"2024-10-15T08:15:37.481827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.columns)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:15:46.984950Z","iopub.execute_input":"2024-10-15T08:15:46.985400Z","iopub.status.idle":"2024-10-15T08:15:46.992073Z","shell.execute_reply.started":"2024-10-15T08:15:46.985356Z","shell.execute_reply":"2024-10-15T08:15:46.990810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'sii' in train.columns:\n    print(\"Cột 'sii' có trong train_df.\")\nelse:\n    print(\"Cột 'sii' không có trong train_df.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:15:58.174334Z","iopub.execute_input":"2024-10-15T08:15:58.174881Z","iopub.status.idle":"2024-10-15T08:15:58.181457Z","shell.execute_reply.started":"2024-10-15T08:15:58.174828Z","shell.execute_reply":"2024-10-15T08:15:58.180192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def handling_missing(df):\n    categorical_features = df.select_dtypes(include=['object']).columns.tolist()\n    numerical_features = df.select_dtypes(include=['float64', 'int64']).columns.tolist()\n    df[categorical_features] = df[categorical_features].fillna('missing')\n    df[numerical_features] = df[numerical_features].fillna(df[numerical_features].median())\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:16:06.525062Z","iopub.execute_input":"2024-10-15T08:16:06.525537Z","iopub.status.idle":"2024-10-15T08:16:06.532825Z","shell.execute_reply.started":"2024-10-15T08:16:06.525492Z","shell.execute_reply":"2024-10-15T08:16:06.531468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = handling_missing(train)\ntest = handling_missing(test)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:16:33.388679Z","iopub.execute_input":"2024-10-15T08:16:33.389325Z","iopub.status.idle":"2024-10-15T08:16:33.622875Z","shell.execute_reply.started":"2024-10-15T08:16:33.389260Z","shell.execute_reply":"2024-10-15T08:16:33.621737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le = preprocessing.LabelEncoder()\nscaler = StandardScaler()","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:16:42.224608Z","iopub.execute_input":"2024-10-15T08:16:42.225067Z","iopub.status.idle":"2024-10-15T08:16:42.230454Z","shell.execute_reply.started":"2024-10-15T08:16:42.225024Z","shell.execute_reply":"2024-10-15T08:16:42.229209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_categorical_features = train.select_dtypes(include=['object']).columns.tolist()\ntrain_numerical_features = train.select_dtypes(include=['float64', 'int64']).drop(columns=['sii']).columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:16:51.321304Z","iopub.execute_input":"2024-10-15T08:16:51.321791Z","iopub.status.idle":"2024-10-15T08:16:51.344460Z","shell.execute_reply.started":"2024-10-15T08:16:51.321747Z","shell.execute_reply":"2024-10-15T08:16:51.343340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in train_categorical_features:\n    train[col] = le.fit_transform(train[col]).astype(int)\ntrain[train_numerical_features] = scaler.fit_transform(train[train_numerical_features])","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:16:57.650406Z","iopub.execute_input":"2024-10-15T08:16:57.650868Z","iopub.status.idle":"2024-10-15T08:16:57.733527Z","shell.execute_reply.started":"2024-10-15T08:16:57.650827Z","shell.execute_reply":"2024-10-15T08:16:57.732313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_categorical_features = test.select_dtypes(include=['object']).columns.tolist()\ntest_numerical_features = test.select_dtypes(include=['float64', 'int64']).columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:17:05.882303Z","iopub.execute_input":"2024-10-15T08:17:05.882844Z","iopub.status.idle":"2024-10-15T08:17:05.896626Z","shell.execute_reply.started":"2024-10-15T08:17:05.882782Z","shell.execute_reply":"2024-10-15T08:17:05.895376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in test_categorical_features:\n    test[col] = le.fit_transform(test[col]).astype(int)\ntest[test_numerical_features] = scaler.fit_transform(test[test_numerical_features])","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:17:14.003706Z","iopub.execute_input":"2024-10-15T08:17:14.004534Z","iopub.status.idle":"2024-10-15T08:17:14.047666Z","shell.execute_reply.started":"2024-10-15T08:17:14.004488Z","shell.execute_reply":"2024-10-15T08:17:14.046389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_features = test.drop(columns=['id']).columns\ncorrelation_matrix = train[base_features].corr()\n# Use parentheses to call the corrwith method with 'PCIAT-PCIAT_Total' as an argument\ntarget_corr = correlation_matrix.corrwith(train['PCIAT-PCIAT_Total']).drop(labels=['PCIAT-PCIAT_Total'], errors='ignore')  \nthreshold = 0.01\nbase_features = target_corr[abs(target_corr) > threshold].index.tolist()","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:17:18.438600Z","iopub.execute_input":"2024-10-15T08:17:18.439059Z","iopub.status.idle":"2024-10-15T08:17:18.935518Z","shell.execute_reply.started":"2024-10-15T08:17:18.439015Z","shell.execute_reply":"2024-10-15T08:17:18.934495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(25):\n  feature1 = f\"feature_{i}_1\"\n  feature2 = f\"feature_{i}_2\"\n  train[feature1] = np.random.rand(len(train))\n  test[feature1] = np.random.rand(len(test))\n  train[feature2] = np.random.rand(len(train))\n  test[feature2] = np.random.rand(len(test))\n  train[f\"feature_{i}_interaction\"] = train[feature1] * train[feature2]\n  test[f\"feature_{i}_interaction\"] = test[feature1] * test[feature2]","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:17:23.797806Z","iopub.execute_input":"2024-10-15T08:17:23.798258Z","iopub.status.idle":"2024-10-15T08:17:23.899481Z","shell.execute_reply.started":"2024-10-15T08:17:23.798213Z","shell.execute_reply":"2024-10-15T08:17:23.898240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_features = base_features + [col for col in train.columns if col.startswith('feature_')]\n\ntrain_new = train[all_features + ['sii']]\ntest_new = test[all_features]\n\nX = train_new.drop('sii', axis=1)\ny = train_new['sii']\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:17:29.691490Z","iopub.execute_input":"2024-10-15T08:17:29.691963Z","iopub.status.idle":"2024-10-15T08:17:29.722741Z","shell.execute_reply.started":"2024-10-15T08:17:29.691921Z","shell.execute_reply":"2024-10-15T08:17:29.721543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Model","metadata":{}},{"cell_type":"code","source":"SEED = 42\nLGBM_Params = {\n    'n_estimators': 1000,\n    'learning_rate': 0.01,\n    'max_depth': 7,\n    'num_leaves': 31,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'random_state': 42\n}\nXGB_Params = {\n    'n_estimators': 1000,\n    'learning_rate': 0.01,\n    'max_depth': 7,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'random_state': 42\n}\nCatBoost_Params = {\n    'iterations': 1000,\n    'learning_rate': 0.01,\n    'depth': 7,\n    'random_seed': 42\n}\nETR_Params = {\n    'n_estimators': 1000,\n    'max_depth': None,\n    'min_samples_split': 2,\n    'random_state': 42\n}\n\nbase_models = [\n    ('lgb', LGBMRegressor(**LGBM_Params)),\n    ('xgb', XGBRegressor(**XGB_Params)),\n    ('cat', CatBoostRegressor(**CatBoost_Params)),\n    ('etr', ExtraTreesRegressor(**ETR_Params)),\n]","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:17:57.078326Z","iopub.execute_input":"2024-10-15T08:17:57.079036Z","iopub.status.idle":"2024-10-15T08:17:57.098077Z","shell.execute_reply.started":"2024-10-15T08:17:57.078987Z","shell.execute_reply":"2024-10-15T08:17:57.096706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kf = StratifiedKFold(n_splits=5, shuffle=True, random_state=SEED)\npredictions = np.zeros(X.shape[0])\ntest_predictions = np.zeros(test_new.shape[0])\nqwk_scores = []\n\n# Hàm tối ưu hóa QWK bằng cách điều chỉnh ngưỡng\ndef evaluate_predictions(thresholds, y_true, y_pred):\n    thresholds = np.sort(thresholds)  # Đảm bảo ngưỡng theo thứ tự tăng dần\n    y_pred_classes = np.digitize(y_pred, thresholds)\n    return -cohen_kappa_score(y_true, y_pred_classes, weights='quadratic')\n\nfor fold, (train_idx, val_idx) in enumerate(kf.split(X, y)):\n    X_train, y_train = X.iloc[train_idx], y.iloc[train_idx]\n    X_val, y_val = X.iloc[val_idx], y.iloc[val_idx]\n\n    estimator = StackingRegressor(\n        estimators=base_models,\n        final_estimator=LinearRegression(),\n        cv=5\n    )\n\n    # Huấn luyện mô hình\n    estimator.fit(X_train, y_train)\n    predictions[val_idx] = estimator.predict(X_val)\n\n    # Tối ưu hóa ngưỡng cho QWK score\n    initial_thresholds = [0.5, 1.5, 2.5]  # Ngưỡng khởi tạo ban đầu\n    KappaOptimizer = minimize(evaluate_predictions, x0=initial_thresholds, \n                              args=(y_val, predictions[val_idx]), method='Nelder-Mead')\n    best_thresholds = KappaOptimizer.x\n\n    # Chuyển đổi dự đoán thành nhãn lớp với ngưỡng tối ưu\n    val_pred_classes = np.digitize(predictions[val_idx], np.sort(best_thresholds))\n    qwk_score = cohen_kappa_score(y_val, val_pred_classes, weights='quadratic')\n    qwk_scores.append(qwk_score)\n    \n    # Dự đoán trên tập kiểm tra và lưu lại\n    test_preds = estimator.predict(test_new)\n    test_predictions += np.digitize(test_preds, np.sort(best_thresholds)) / kf.n_splits\n\n    print(f\"Fold {fold + 1}: QWK = {qwk_score}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-15T08:20:27.876057Z","iopub.execute_input":"2024-10-15T08:20:27.876586Z","iopub.status.idle":"2024-10-15T09:02:02.537937Z","shell.execute_reply.started":"2024-10-15T08:20:27.876539Z","shell.execute_reply":"2024-10-15T09:02:02.536671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Mean QWK: {np.mean(qwk_scores)}\")\nprint(f\"Std QWK: {np.std(qwk_scores)}\")\nprint(f\"Min QWK: {np.min(qwk_scores)}\")\nprint(f\"Max QWK: {np.max(qwk_scores)}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-15T09:02:02.540042Z","iopub.execute_input":"2024-10-15T09:02:02.540439Z","iopub.status.idle":"2024-10-15T09:02:02.547674Z","shell.execute_reply.started":"2024-10-15T09:02:02.540381Z","shell.execute_reply":"2024-10-15T09:02:02.546508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_pred_classes = np.digitize(test_predictions, np.sort(best_thresholds))\nsubmit_df = pd.DataFrame({'id': test['id'], 'sii': test_pred_classes.astype(int)})\nsubmit_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T09:02:02.548919Z","iopub.execute_input":"2024-10-15T09:02:02.549242Z","iopub.status.idle":"2024-10-15T09:02:02.577381Z","shell.execute_reply.started":"2024-10-15T09:02:02.549207Z","shell.execute_reply":"2024-10-15T09:02:02.576199Z"},"trusted":true},"execution_count":null,"outputs":[]}]}