{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import libraries\nimport optuna\nimport pandas as pd\nimport numpy as np\nimport warnings\nfrom sklearn import preprocessing\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.metrics import mean_squared_error\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom scipy.optimize import minimize\nfrom statsmodels.stats.outliers_influence import variance_inflation_factor\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import StackingRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.linear_model import ElasticNet\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.ensemble import ExtraTreesRegressor\nfrom sklearn.ensemble import HistGradientBoostingRegressor\nfrom sklearn.svm import SVR\nfrom sklearn.neighbors import KNeighborsRegressor\n# from sklearn.linear_model import LinearRegression","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-15T02:48:05.027857Z","iopub.execute_input":"2024-10-15T02:48:05.028550Z","iopub.status.idle":"2024-10-15T02:48:10.398861Z","shell.execute_reply.started":"2024-10-15T02:48:05.028507Z","shell.execute_reply":"2024-10-15T02:48:10.397990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', 1000)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T02:48:25.593090Z","iopub.execute_input":"2024-10-15T02:48:25.593478Z","iopub.status.idle":"2024-10-15T02:48:25.598466Z","shell.execute_reply.started":"2024-10-15T02:48:25.593440Z","shell.execute_reply":"2024-10-15T02:48:25.597369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load data\ntrain_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-15T02:48:28.099971Z","iopub.execute_input":"2024-10-15T02:48:28.100601Z","iopub.status.idle":"2024-10-15T02:48:28.149424Z","shell.execute_reply.started":"2024-10-15T02:48:28.100561Z","shell.execute_reply":"2024-10-15T02:48:28.148258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shape of the data:\nprint(\"train_df :\", train_df.shape)\nprint(\"test_df :\", test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:46:29.794465Z","iopub.execute_input":"2024-10-15T01:46:29.794848Z","iopub.status.idle":"2024-10-15T01:46:29.800731Z","shell.execute_reply.started":"2024-10-15T01:46:29.794811Z","shell.execute_reply":"2024-10-15T01:46:29.799651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'sii' in train_df.columns:\n  print(\"Cột 'sii' có trong tập train.\")\nelse:\n  print(\"Cột 'sii' không có trong tập train.\")\n\nif 'sii' in test_df.columns:\n  print(\"Cột 'sii' có trong tập test.\")\nelse:\n  print(\"Cột 'sii' không có trong tập test.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-15T02:48:34.708035Z","iopub.execute_input":"2024-10-15T02:48:34.709056Z","iopub.status.idle":"2024-10-15T02:48:34.717490Z","shell.execute_reply.started":"2024-10-15T02:48:34.709012Z","shell.execute_reply":"2024-10-15T02:48:34.716329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Custom functions\ndef process_file(filename, dirname):\n    data = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    data.drop('step', axis=1, inplace=True)\n    return data.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname):\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    stats, indexes = zip(*results)\n    data = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    data['id'] = indexes\n    return data","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:46:29.802166Z","iopub.execute_input":"2024-10-15T01:46:29.802611Z","iopub.status.idle":"2024-10-15T01:46:29.813124Z","shell.execute_reply.started":"2024-10-15T01:46:29.802564Z","shell.execute_reply":"2024-10-15T01:46:29.812125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load time series data\ntrain_parquet = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_parquet = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:46:29.814569Z","iopub.execute_input":"2024-10-15T01:46:29.815038Z","iopub.status.idle":"2024-10-15T01:48:04.509805Z","shell.execute_reply.started":"2024-10-15T01:46:29.814986Z","shell.execute_reply":"2024-10-15T01:48:04.508732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'sii' in train_parquet.columns:\n  print(\"Cột 'sii' có trong tập train_parquet.\")\nelse:\n  print(\"Cột 'sii' không có trong tập train_parquet.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:53:55.480989Z","iopub.execute_input":"2024-10-15T01:53:55.481437Z","iopub.status.idle":"2024-10-15T01:53:55.487594Z","shell.execute_reply.started":"2024-10-15T01:53:55.481395Z","shell.execute_reply":"2024-10-15T01:53:55.486220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Merge and preprocess data\ntrain_df = pd.merge(train_df, train_parquet, how=\"left\", on='id')\ntest_df = pd.merge(test_df, test_parquet, how=\"left\", on='id')","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:48:04.511007Z","iopub.execute_input":"2024-10-15T01:48:04.511334Z","iopub.status.idle":"2024-10-15T01:48:04.540436Z","shell.execute_reply.started":"2024-10-15T01:48:04.511300Z","shell.execute_reply":"2024-10-15T01:48:04.539414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 'sii' in train_df.columns:\n  print(\"Cột 'sii' có trong tập train_df.\")\nelse:\n  print(\"Cột 'sii' không có trong tập train_df.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:53:59.603513Z","iopub.execute_input":"2024-10-15T01:53:59.603940Z","iopub.status.idle":"2024-10-15T01:53:59.609871Z","shell.execute_reply.started":"2024-10-15T01:53:59.603902Z","shell.execute_reply":"2024-10-15T01:53:59.608531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_features = test_df.drop(columns=['id']).columns\ntest_id = test_df['id']\ntrain_df = train_df.dropna(subset='sii')","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:48:04.541863Z","iopub.execute_input":"2024-10-15T01:48:04.542284Z","iopub.status.idle":"2024-10-15T01:48:04.554617Z","shell.execute_reply.started":"2024-10-15T01:48:04.542236Z","shell.execute_reply":"2024-10-15T01:48:04.553564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"# check missing values\ntrain_missing_values = train_df[base_features].isnull().sum().sort_values(ascending=False)\ntest_missing_values = test_df.isnull().sum().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:48:04.555934Z","iopub.execute_input":"2024-10-15T01:48:04.556296Z","iopub.status.idle":"2024-10-15T01:48:04.571196Z","shell.execute_reply.started":"2024-10-15T01:48:04.556259Z","shell.execute_reply":"2024-10-15T01:48:04.569997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_missing_values)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:48:04.575035Z","iopub.execute_input":"2024-10-15T01:48:04.575428Z","iopub.status.idle":"2024-10-15T01:48:04.582534Z","shell.execute_reply.started":"2024-10-15T01:48:04.575387Z","shell.execute_reply":"2024-10-15T01:48:04.581455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Remove columns with too many misssing values\n# train_missing_ratio = train_df[base_features].isnull().mean()\n\n# # Remove columns with more than 50% missing values\n# threshold = 0.5\n# base_features = train_missing_ratio[train_missing_ratio < threshold].index","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:48:04.583984Z","iopub.execute_input":"2024-10-15T01:48:04.584366Z","iopub.status.idle":"2024-10-15T01:48:04.596857Z","shell.execute_reply.started":"2024-10-15T01:48:04.584313Z","shell.execute_reply":"2024-10-15T01:48:04.595512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Handling missing values\ndef handling_missing(df):\n    categorical_features = df.select_dtypes(include=['object']).columns.tolist()\n    numerical_features = df.select_dtypes(include=['float64', 'int64']).columns.tolist()\n    df[categorical_features] = df[categorical_features].fillna('missing')\n    df[numerical_features] = df[numerical_features].fillna(df[numerical_features].median()) # mean?\n    return df\n\ntrain_df = handling_missing(train_df)\ntest_df = handling_missing(test_df)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:48:04.598215Z","iopub.execute_input":"2024-10-15T01:48:04.598641Z","iopub.status.idle":"2024-10-15T01:48:04.791685Z","shell.execute_reply.started":"2024-10-15T01:48:04.598600Z","shell.execute_reply":"2024-10-15T01:48:04.790595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encoding data\nle = preprocessing.LabelEncoder()\nscaler = StandardScaler()\n\n# train data\ntrain_categorical_features = train_df.select_dtypes(include=['object']).columns.tolist()\ntrain_numerical_features = train_df.select_dtypes(include=['float64', 'int64']).drop(columns=['sii']).columns.tolist()\n# categorical data\nfor col in train_categorical_features:\n    train_df[col] = le.fit_transform(train_df[col]).astype(int)\n# numerical data\ntrain_df[train_numerical_features] = scaler.fit_transform(train_df[train_numerical_features])\n\n# test data\ntest_categorical_features = test_df.select_dtypes(include=['object']).columns.tolist()\ntest_numerical_features = test_df.select_dtypes(include=['float64', 'int64']).columns.tolist()\n# categorical data\nfor col in test_categorical_features:\n    test_df[col] = le.fit_transform(test_df[col]).astype(int)\n# numerical data\ntest_df[test_numerical_features] = scaler.fit_transform(test_df[test_numerical_features])","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:48:04.793039Z","iopub.execute_input":"2024-10-15T01:48:04.793386Z","iopub.status.idle":"2024-10-15T01:48:04.890719Z","shell.execute_reply.started":"2024-10-15T01:48:04.793331Z","shell.execute_reply":"2024-10-15T01:48:04.889558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# select the features with the strongest correlation with 'PCIAT-PCIAT_Total'\nbase_features_corr = base_features.tolist()  \nbase_features_corr.append('PCIAT-PCIAT_Total')\ncorrelation_matrix = train_df[base_features_corr].corr()\ntarget_corr = correlation_matrix['PCIAT-PCIAT_Total'].drop(labels=['PCIAT-PCIAT_Total'], errors='ignore')\n\nthreshold = 0.01\nbase_features = target_corr[abs(target_corr) > threshold].index.tolist()","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:48:04.892796Z","iopub.execute_input":"2024-10-15T01:48:04.893536Z","iopub.status.idle":"2024-10-15T01:48:05.086155Z","shell.execute_reply.started":"2024-10-15T01:48:04.893475Z","shell.execute_reply":"2024-10-15T01:48:05.085136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.concat([train_df[base_features], train_df['sii']], axis=1)\ntest_data = test_df[base_features]","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:48:05.087795Z","iopub.execute_input":"2024-10-15T01:48:05.088168Z","iopub.status.idle":"2024-10-15T01:48:05.107321Z","shell.execute_reply.started":"2024-10-15T01:48:05.088130Z","shell.execute_reply":"2024-10-15T01:48:05.106194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# combine train data and test data\nall_df = pd.concat([train_data, test_data], sort=False).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:48:05.108716Z","iopub.execute_input":"2024-10-15T01:48:05.109100Z","iopub.status.idle":"2024-10-15T01:48:05.127311Z","shell.execute_reply.started":"2024-10-15T01:48:05.109061Z","shell.execute_reply":"2024-10-15T01:48:05.126031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create new features\nall_df['BMI_Age_sum'] = all_df['Physical-BMI'] + all_df['Basic_Demos-Age']\nall_df['BMI_Age_mult'] = all_df['Physical-BMI'] * all_df['Basic_Demos-Age']\nall_df['Physical_Mean'] = all_df[['Physical-Height', 'Physical-Weight', 'Physical-BMI']].mean(axis=1)\nall_df['Physical_Std'] = all_df[['Physical-Height', 'Physical-Weight', 'Physical-BMI']].std(axis=1)\nall_df['BMI_Rank'] = all_df['Physical-BMI'].rank()\n# all_df['HeartRate_Rank'] = all_df['Physical-HeartRate'].rank(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:48:05.128888Z","iopub.execute_input":"2024-10-15T01:48:05.129727Z","iopub.status.idle":"2024-10-15T01:48:05.145206Z","shell.execute_reply.started":"2024-10-15T01:48:05.129674Z","shell.execute_reply":"2024-10-15T01:48:05.144112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# categorical_features = all_df.select_dtypes(include=['object']).columns.tolist()\n# numerical_features = all_df.select_dtypes(include=['float64', 'int64']).drop(columns=['sii']).columns.tolist()\n# all_df[categorical_features] = all_df[categorical_features].fillna('missing')\n# all_df[numerical_features] = all_df[numerical_features].fillna(all_df[numerical_features].median())","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:48:05.146548Z","iopub.execute_input":"2024-10-15T01:48:05.146912Z","iopub.status.idle":"2024-10-15T01:48:05.158045Z","shell.execute_reply.started":"2024-10-15T01:48:05.146875Z","shell.execute_reply":"2024-10-15T01:48:05.156920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checkt VIF\nX_vif = all_df[base_features]\n\n# Function to calculate VIF\ndef calc_vif(X):\n    vif_data = pd.DataFrame()\n    vif_data['feature'] = X.columns\n    vif_data['VIF'] = [variance_inflation_factor(X.values, i) for i in range(X.shape[1])]\n    return vif_data\n\n# Initial VIF calculation\nvif_df = calc_vif(X_vif)\nvif_df","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:48:05.159509Z","iopub.execute_input":"2024-10-15T01:48:05.159875Z","iopub.status.idle":"2024-10-15T01:48:12.792761Z","shell.execute_reply.started":"2024-10-15T01:48:05.159839Z","shell.execute_reply":"2024-10-15T01:48:12.791583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Iteratively remove variables with VIF exceeding the threshold\nthreshold = 10  # Set threshold for removal, e.g., VIF > 10\nwhile vif_df['VIF'].max() > threshold:\n    # Identify and remove the variable with the highest VIF\n    highest_vif = vif_df[vif_df['VIF'] == vif_df['VIF'].max()]['feature'].iloc[0]\n    print(f'Removing {highest_vif} with VIF: {vif_df[\"VIF\"].max()}')\n    X_vif = X_vif.drop(columns=[highest_vif])\n    \n    # Recalculate VIF\n    vif_df = calc_vif(X_vif)\n\nvif_df","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:48:12.794194Z","iopub.execute_input":"2024-10-15T01:48:12.795207Z","iopub.status.idle":"2024-10-15T01:51:00.936772Z","shell.execute_reply.started":"2024-10-15T01:48:12.795157Z","shell.execute_reply":"2024-10-15T01:51:00.934988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_feature = vif_df['feature']\nall_df = pd.concat([all_df[selected_feature], all_df['sii']], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:51:00.943056Z","iopub.execute_input":"2024-10-15T01:51:00.945017Z","iopub.status.idle":"2024-10-15T01:51:00.962029Z","shell.execute_reply.started":"2024-10-15T01:51:00.944927Z","shell.execute_reply":"2024-10-15T01:51:00.960213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_le = all_df[~all_df['sii'].isnull()]\ntest_df_le = all_df[all_df['sii'].isnull()].drop(columns=['sii'])","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:51:00.964324Z","iopub.execute_input":"2024-10-15T01:51:00.965319Z","iopub.status.idle":"2024-10-15T01:51:00.978428Z","shell.execute_reply.started":"2024-10-15T01:51:00.965260Z","shell.execute_reply":"2024-10-15T01:51:00.976371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# select the features with the strongest correlation with 'sii'\n# correlation_matrix = train_df_le.corr()\n# target_corr = correlation_matrix['sii'].drop('sii')\n\n# threshold = 0.01\n# high_corr_features = target_corr[abs(target_corr) > threshold].index.tolist()","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:51:00.980906Z","iopub.execute_input":"2024-10-15T01:51:00.982216Z","iopub.status.idle":"2024-10-15T01:51:00.988703Z","shell.execute_reply.started":"2024-10-15T01:51:00.982149Z","shell.execute_reply":"2024-10-15T01:51:00.987240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df_le =  pd.concat([train_df_le[high_corr_features], train_df_le['sii']], axis=1)\n# test_df_le = test_df_le[high_corr_features]","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:51:00.991544Z","iopub.execute_input":"2024-10-15T01:51:00.992896Z","iopub.status.idle":"2024-10-15T01:51:00.999240Z","shell.execute_reply.started":"2024-10-15T01:51:00.992830Z","shell.execute_reply":"2024-10-15T01:51:00.997865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"X = train_df_le.drop(columns=['sii'])\ny = train_df_le['sii']","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:51:01.002192Z","iopub.execute_input":"2024-10-15T01:51:01.003516Z","iopub.status.idle":"2024-10-15T01:51:01.013150Z","shell.execute_reply.started":"2024-10-15T01:51:01.003449Z","shell.execute_reply":"2024-10-15T01:51:01.011521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SEED = 42\n# X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:51:01.015859Z","iopub.execute_input":"2024-10-15T01:51:01.017198Z","iopub.status.idle":"2024-10-15T01:51:01.022516Z","shell.execute_reply.started":"2024-10-15T01:51:01.017137Z","shell.execute_reply":"2024-10-15T01:51:01.021202Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LGBM\n# def lgb_objective(trial):\n#     param = {\n#         'objective': 'regression',\n#         'metric': 'rmse',\n#         'boosting_type': 'gbdt',\n#         'max_depth': trial.suggest_int('max_depth', 3, 15),\n#         'num_leaves': trial.suggest_int('num_leaves', 20, 300),\n#         'learning_rate': trial.suggest_loguniform('learning_rate', 1e-4, 1e-1),\n#         'feature_fraction': trial.suggest_uniform('feature_fraction', 0.4, 1.0),\n#         'bagging_fraction': trial.suggest_uniform('bagging_fraction', 0.4, 1.0),\n#         'lambda_l1': trial.suggest_loguniform('lambda_l1', 1e-8, 10.0),\n#         'lambda_l2': trial.suggest_loguniform('lambda_l2', 1e-8, 10.0),\n#         'random_state': SEED\n#     }\n#     model = LGBMRegressor(**param)\n#     model.fit(X_train, y_train)\n#     preds = model.predict(X_val)\n#     rmse = mean_squared_error(y_val, preds, squared=False)\n#     return rmse\n\n# study = optuna.create_study(direction='minimize')\n# study.optimize(lgb_objective, n_trials=50)\n# LGBM_Params = study.best_params","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:51:01.024986Z","iopub.execute_input":"2024-10-15T01:51:01.026136Z","iopub.status.idle":"2024-10-15T01:51:01.033009Z","shell.execute_reply.started":"2024-10-15T01:51:01.026077Z","shell.execute_reply":"2024-10-15T01:51:01.031855Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# XGB\n# def xgb_objective(trial):\n#     param = {\n#         'objective': 'reg:squarederror',\n#         'max_depth': trial.suggest_int('max_depth', 3, 15),\n#         'learning_rate': trial.suggest_loguniform('learning_rate', 1e-4, 1e-1),\n#         'subsample': trial.suggest_uniform('subsample', 0.5, 1.0),\n#         'colsample_bytree': trial.suggest_uniform('colsample_bytree', 0.5, 1.0),\n#         'alpha': trial.suggest_loguniform('alpha', 1e-8, 10.0),\n#         'lambda': trial.suggest_loguniform('lambda', 1e-8, 10.0),\n#         'random_state': SEED\n#     }\n#     model = XGBRegressor(**param)\n#     model.fit(X_train, y_train)\n#     preds = model.predict(X_val)\n#     rmse = mean_squared_error(y_val, preds, squared=False)\n#     return rmse\n\n# study = optuna.create_study(direction='minimize')\n# study.optimize(xgb_objective, n_trials=50)\n# XGB_Params = study.best_params","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:51:01.040189Z","iopub.execute_input":"2024-10-15T01:51:01.040981Z","iopub.status.idle":"2024-10-15T01:51:01.046278Z","shell.execute_reply.started":"2024-10-15T01:51:01.040938Z","shell.execute_reply":"2024-10-15T01:51:01.045029Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CatBoost\n# def cat_objective(trial):\n#     param = {\n#         'depth': trial.suggest_int('depth', 4, 10),\n#         'learning_rate': trial.suggest_loguniform('learning_rate', 1e-4, 1e-1),\n#         'l2_leaf_reg': trial.suggest_loguniform('l2_leaf_reg', 1e-5, 10.0),\n#         'bagging_temperature': trial.suggest_uniform('bagging_temperature', 0, 1),\n#         'border_count': trial.suggest_int('border_count', 32, 255),\n#         'random_state': SEED\n#     }\n#     model = CatBoostRegressor(**param, silent=True)\n#     model.fit(X_train, y_train)\n#     preds = model.predict(X_val)\n#     rmse = mean_squared_error(y_val, preds, squared=False)\n#     return rmse\n\n# study = optuna.create_study(direction='minimize')\n# study.optimize(cat_objective, n_trials=20)\n# CatBoost_Params = study.best_params","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:51:01.047699Z","iopub.execute_input":"2024-10-15T01:51:01.048132Z","iopub.status.idle":"2024-10-15T01:51:01.061775Z","shell.execute_reply.started":"2024-10-15T01:51:01.048089Z","shell.execute_reply":"2024-10-15T01:51:01.060468Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model parameters for LightGBM\nSEED = 42\nLGBM_Params = {\n    'learning_rate': 0.0399, # 0.0399\n    'max_depth': 11, # 12 -> 11\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10, # 10\n    'lambda_l2': 0.3, # 0.3\n    'random_state': SEED\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.03, # 0.05\n    'max_depth': 7, # 6 -> 7\n    'n_estimators': 210, # 200\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1, \n    'reg_lambda': 3, # 3\n    'random_state': SEED\n}\n\n# CatBoost\nCatBoost_Params = {\n    'learning_rate': 0.03, # 0.05\n    'depth': 7, # 6 -> 7\n    'iterations': 210, # 200\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10\n}\n\n# Gradient Boost\nGBR_Params = {\n    'n_estimators': 210,\n    'learning_rate': 0.03,\n    'max_depth': 5,\n    'min_samples_split': 2,\n    'min_samples_leaf': 1,\n    'subsample': 0.8,\n    'random_state': SEED\n}\n\n# Extra Trees\nETR_Params = {\n    'n_estimators': 150,\n    'max_depth': 10,\n    'min_samples_split': 5,\n    'min_samples_leaf': 2,\n    'max_features': 'sqrt',\n    'bootstrap': False,\n    'random_state': SEED\n}\n\n# HistGradientBoosting\nHGBR_Params = {\n    'learning_rate': 0.03,\n    'max_iter': 100,\n    'max_depth': 7,\n    'max_leaf_nodes': 31,\n    'min_samples_leaf': 20,\n    'l2_regularization': 0.1,\n    'random_state': SEED\n}\n\n# # Support Vector\n# SVR_Params = {\n#     'C': 1.0,\n#     'epsilon': 0.1,\n#     'kernel': 'rbf',\n#     'gamma': 'scale'\n# }\n\n# # KNeighbors\n# KNN_Params = {\n#     'n_neighbors': 5,\n#     'weights': 'distance',\n#     'algorithm': 'auto',\n#     'leaf_size': 30,\n#     'p': 2  \n# }","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:51:01.063457Z","iopub.execute_input":"2024-10-15T01:51:01.063830Z","iopub.status.idle":"2024-10-15T01:51:01.074943Z","shell.execute_reply.started":"2024-10-15T01:51:01.063790Z","shell.execute_reply":"2024-10-15T01:51:01.073560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the base models\nbase_models = [\n    ('lgb', LGBMRegressor(**LGBM_Params)),\n    ('xgb', XGBRegressor(**XGB_Params)),\n    ('cat', CatBoostRegressor(**CatBoost_Params)),\n    #('gbr', GradientBoostingRegressor(**GBR_Params))\n    ('etr', ExtraTreesRegressor(**ETR_Params)),\n    #('hgbr', HistGradientBoostingRegressor(**HGBR_Params)),\n    #('svr', SVR(**SVR_Params)),\n    #('knn', KNeighborsRegressor(**KNN_Params))\n    \n]\nmeta_model = RandomForestRegressor(n_estimators=225, # 225\n                                   max_depth=10, # 3 -> 5\n                                   min_samples_split=5,\n                                   min_samples_leaf=13, # 4\n                                   max_features='sqrt', # sqrt \n                                   max_samples=0.8, # 0.8 \n                                   random_state=SEED) # RandomForest\nstacking_model = StackingRegressor(estimators=base_models, final_estimator=meta_model)\n\n# Cross-validation and model training\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=SEED)\npredictions = np.zeros(X.shape[0]) \ntest_predictions = np.zeros(test_df_le.shape[0])\nqwk_scores = []\n\n# Function to optimize the QWK score by adjusting thresholds\ndef evaluate_predictions(thresholds, y_true, y_pred):\n    thresholds = np.sort(thresholds)  # Ensure thresholds are in ascending order\n    y_pred_classes = np.digitize(y_pred, thresholds)\n    return -cohen_kappa_score(y_true, y_pred_classes, weights='quadratic')\n\nfor train_idx, valid_idx in kf.split(X, y):\n    X_train, X_valid = X.iloc[train_idx], X.iloc[valid_idx]\n    y_train, y_valid = y.iloc[train_idx], y.iloc[valid_idx]\n    \n    # Fit the stacking model\n    stacking_model.fit(X_train, y_train)\n\n    # Predict for validation set\n    preds = stacking_model.predict(X_valid)\n    predictions[valid_idx] += preds\n\n    # Optimize thresholds for QWK score using Nelder-Mead\n    initial_thresholds = [0.5, 1.5, 2.5]  # Initial guess for thresholds\n    KappaOptimizer = minimize(evaluate_predictions, x0=initial_thresholds, \n                              args=(y_valid, preds), method='Nelder-Mead')\n    best_thresholds = KappaOptimizer.x\n\n    # Apply optimized thresholds to validation predictions\n    # ensemble_preds = np.digitize(predictions[valid_idx], np.sort(best_thresholds))\n    ensemble_preds = np.digitize(preds, np.sort(best_thresholds))\n\n    # Calculate QWK score for the optimized predictions\n    qwk_score = cohen_kappa_score(y.iloc[valid_idx], ensemble_preds, weights='quadratic')\n    qwk_scores.append(qwk_score)\n\n    # Predict for test set and store using optimized thresholds\n    test_preds = stacking_model.predict(test_df_le)\n    test_predictions += np.digitize(test_preds, np.sort(best_thresholds)) / kf.n_splits","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:51:01.076438Z","iopub.execute_input":"2024-10-15T01:51:01.076854Z","iopub.status.idle":"2024-10-15T01:53:48.802740Z","shell.execute_reply.started":"2024-10-15T01:51:01.076815Z","shell.execute_reply":"2024-10-15T01:53:48.801694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show the average QWK score across all folds\n# print(f\"Average QWK Score: {np.mean(qwk_scores)}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:53:48.804134Z","iopub.execute_input":"2024-10-15T01:53:48.804524Z","iopub.status.idle":"2024-10-15T01:53:48.809319Z","shell.execute_reply.started":"2024-10-15T01:53:48.804484Z","shell.execute_reply":"2024-10-15T01:53:48.808074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Average: ', np.mean(qwk_scores))\nprint('Std: ', np.std(qwk_scores))\nprint('Min: ', np.min(qwk_scores))\nprint('Max: ', np.max(qwk_scores))","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:53:48.810932Z","iopub.execute_input":"2024-10-15T01:53:48.811299Z","iopub.status.idle":"2024-10-15T01:53:48.823748Z","shell.execute_reply.started":"2024-10-15T01:53:48.811262Z","shell.execute_reply":"2024-10-15T01:53:48.822536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"qwk_scores","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:53:48.825120Z","iopub.execute_input":"2024-10-15T01:53:48.825543Z","iopub.status.idle":"2024-10-15T01:53:48.838912Z","shell.execute_reply.started":"2024-10-15T01:53:48.825501Z","shell.execute_reply":"2024-10-15T01:53:48.837835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Final predictions","metadata":{}},{"cell_type":"code","source":"# Prepare the test predictions for submission\ntest_pred_classes = np.digitize(test_predictions, np.sort(best_thresholds))\nsubmit_df = pd.DataFrame({'id': test_id, 'sii': test_pred_classes.astype(int)})\nsubmit_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T01:53:48.840143Z","iopub.execute_input":"2024-10-15T01:53:48.840545Z","iopub.status.idle":"2024-10-15T01:53:48.856249Z","shell.execute_reply.started":"2024-10-15T01:53:48.840506Z","shell.execute_reply":"2024-10-15T01:53:48.855067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}