{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom sklearn.metrics import cohen_kappa_score, make_scorer, confusion_matrix\nfrom sklearn.model_selection import StratifiedKFold, KFold\nfrom scipy.optimize import minimize\nfrom scipy import stats\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom sklearn.feature_selection import RFECV\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nimport optuna\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-25T17:03:39.780000Z","iopub.execute_input":"2024-12-25T17:03:39.780330Z","iopub.status.idle":"2024-12-25T17:03:45.101944Z","shell.execute_reply.started":"2024-12-25T17:03:39.780288Z","shell.execute_reply":"2024-12-25T17:03:45.100890Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42\nn_splits = 20\noptimize_params = False\noptimize_features = False\nn_trials = 25 # n_trials for optuna \nmin_features_to_select = 100 # min features to select with RFECV\nbase_thresholds = [30, 50, 80]\n\n# Load datasets\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-12-25T17:03:45.103064Z","iopub.execute_input":"2024-12-25T17:03:45.103526Z","iopub.status.idle":"2024-12-25T17:03:45.188845Z","shell.execute_reply.started":"2024-12-25T17:03:45.103495Z","shell.execute_reply":"2024-12-25T17:03:45.187879Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def time_features(df):\n    # Convert time_of_day to hours\n    df[\"hours\"] = df[\"time_of_day\"] // (3_600 * 1_000_000_000)\n    # Define conditions for night, day, and no mask (full data)\n    night = ((df[\"hours\"] >= 22) | (df[\"hours\"] <= 5))\n    day = ((df[\"hours\"] <= 20) & (df[\"hours\"] >= 7))\n    no_mask = np.ones(len(df), dtype=bool)\n    # Define weekend and last week conditions\n    weekend = (df[\"weekday\"] >= 6)\n    last_week = df[\"relative_date_PCIAT\"] >= (df[\"relative_date_PCIAT\"].max() - 7)\n    # Create additional weekend features\n    df[\"enmo_weekend\"] = df[\"enmo\"].where(weekend)\n    df[\"anglez_weekend\"] = df[\"anglez\"].where(weekend)\n\n    # List of columns of interest and masks\n    keys = [\"enmo\", \"anglez\", \"light\", \"enmo_weekend\", \"anglez_weekend\"]\n    #masks = [no_mask, night, day, last_week]\n    masks = [no_mask, night, day]\n\n    # Helper function for feature extraction\n    def extract_stats(data):\n        return [\n            data.mean(),\n            data.std(),\n            data.max(),\n            data.min(),\n            data.kurtosis(),\n            data.skew(),\n            data.diff().mean(),\n            data.diff().std(),\n            data.diff().quantile(0.9),\n            data.diff().quantile(0.1)\n        ]\n\n    features = [] # Bỏ 4 giá trị ban đầu\n\n    # Iterate over keys and masks to generate the statistics\n    for key in keys:\n        for i, mask in enumerate(masks):\n            filtered_data = df.loc[mask, key]\n            \n            # Tính toán 4 giá trị ban đầu cho mask == no_mask\n            if i == 0 and key == \"enmo\":\n              features.extend([\n                df[\"non-wear_flag\"].mean(),\n                df[\"battery_voltage\"].mean(),\n                df[\"battery_voltage\"].diff().mean(),\n                df[\"relative_date_PCIAT\"].tail(1).values[0]\n              ])\n            \n            features.extend(extract_stats(filtered_data))\n\n    return features\n\n# Code for parallelized computation of time series data from: Sheikh Muhammad Abdullah \n# https://www.kaggle.com/code/abdmental01/cmi-best-single-model\ndef process_file(filename, dirname):\n    # Process file and extract time features\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return time_features(df), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    # Load time series from directory in parallel\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-12-25T17:03:45.191303Z","iopub.execute_input":"2024-12-25T17:03:45.191675Z","iopub.status.idle":"2024-12-25T17:03:45.206086Z","shell.execute_reply.started":"2024-12-25T17:03:45.191642Z","shell.execute_reply":"2024-12-25T17:03:45.204981Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\ntrain = train[train[\"sii\"].notna()] # Keep rows where target is available\ntrain.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:03:45.207673Z","iopub.execute_input":"2024-12-25T17:03:45.208401Z","iopub.status.idle":"2024-12-25T17:08:06.065098Z","shell.execute_reply.started":"2024-12-25T17:03:45.208354Z","shell.execute_reply":"2024-12-25T17:08:06.064008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot distribution of total scores which determine the sii\n# Note the excess zeros\nsns.set_theme(style=\"whitegrid\")\nplt.hist(train['PCIAT-PCIAT_Total'], bins=40, color=\"darkorange\")\nplt.title('Score Distribution')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:08:06.066386Z","iopub.execute_input":"2024-12-25T17:08:06.066724Z","iopub.status.idle":"2024-12-25T17:08:06.477494Z","shell.execute_reply.started":"2024-12-25T17:08:06.066693Z","shell.execute_reply":"2024-12-25T17:08:06.476184Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Features to exclude, because they're not in test\nexclude = ['PCIAT-Season', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03',\n           'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07',\n           'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11',\n           'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15',\n           'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19',\n           'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total', 'sii']\n\ny = \"PCIAT-PCIAT_Total\" # Score, target for the model\ntarget = \"sii\" # Index, target of the competition\nfeatures = [f for f in train.columns if f not in exclude]\n\n# Categorical features\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n          'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\n# Mapping of categorical columns from: Sheikh Muhammad Abdullah \n# https://www.kaggle.com/code/abdmental01/cmi-best-single-model\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping_train = create_mapping(col, train)\n    mapping_test = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping_train).astype(int)\n    test[col] = test[col].replace(mapping_test).astype(int)\n\n# Simple mean imputation\nfor f in features:\n    f_mean = np.mean(train[f])\n    train.loc[:, f] = train[f].fillna(f_mean)\n    test.loc[:, f] = test[f].fillna(f_mean)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:08:06.479336Z","iopub.execute_input":"2024-12-25T17:08:06.479928Z","iopub.status.idle":"2024-12-25T17:08:06.822359Z","shell.execute_reply.started":"2024-12-25T17:08:06.479876Z","shell.execute_reply":"2024-12-25T17:08:06.821300Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.columns.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:08:06.823541Z","iopub.execute_input":"2024-12-25T17:08:06.823878Z","iopub.status.idle":"2024-12-25T17:08:06.829999Z","shell.execute_reply.started":"2024-12-25T17:08:06.823849Z","shell.execute_reply":"2024-12-25T17:08:06.828615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:08:06.831392Z","iopub.execute_input":"2024-12-25T17:08:06.831880Z","iopub.status.idle":"2024-12-25T17:08:06.873125Z","shell.execute_reply.started":"2024-12-25T17:08:06.831821Z","shell.execute_reply":"2024-12-25T17:08:06.872013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Code for finding optimal thresholds copied from: Michael Semenoff\n# https://www.kaggle.com/code/michaelsemenoff/cmi-actigraphy-feature-engineering-selection\ndef round_with_thresholds(raw_preds, thresholds):\n    return np.where(raw_preds < thresholds[0], int(0),\n                    np.where(raw_preds < thresholds[1], int(1),\n                             np.where(raw_preds < thresholds[2], int(2), int(3))))\n\ndef optimize_thresholds(y_true, raw_preds, start_vals=[0.5, 1.5, 2.5]):\n    def fun(thresholds, y_true, raw_preds):\n        rounded_preds = round_with_thresholds(raw_preds, thresholds)\n        return -cohen_kappa_score(y_true, rounded_preds, weights='quadratic')\n\n    res = minimize(fun, x0=start_vals, args=(y_true, raw_preds), method='Nelder-Mead')\n    assert res.success\n    return res.x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:08:06.877541Z","iopub.execute_input":"2024-12-25T17:08:06.877954Z","iopub.status.idle":"2024-12-25T17:08:06.885085Z","shell.execute_reply.started":"2024-12-25T17:08:06.877921Z","shell.execute_reply":"2024-12-25T17:08:06.883850Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def cross_validate(model_, data, features, score_col, index_col, cv, verbose=False):\n    \"\"\"\n    Perform cross-validation with a given model and compute the out-of-fold \n    predictions and Cohen's Kappa score for each fold.\n\n    Returns:\n    float: Mean Kappa score across all folds.\n    array: Out-of-fold score predictions for the entire dataset.\n    \"\"\"\n    kappa_scores = [] \n    oof_score_predictions = np.zeros(len(data))  \n\n    score_to_index_thresholds = base_thresholds  \n\n    for fold_idx, (train_idx, val_idx) in enumerate(cv.split(data, data[index_col])):\n        # Split data into training and validation sets\n        X_train, X_val = data[features].iloc[train_idx], data[features].iloc[val_idx]\n        y_train_score = data[score_col].iloc[train_idx]  \n        y_val_score = data[score_col].iloc[val_idx]      \n        y_val_index = data[index_col].iloc[val_idx]     \n        \n        # Train model and predict scores for validation set\n        model_.fit(X_train, y_train_score)\n        y_pred_val_score = model_.predict(X_val)\n        \n        oof_score_predictions[val_idx] = y_pred_val_score  # Store OOF predictions\n\n        # Convert predicted scores to index using thresholds\n        y_pred_val_index = round_with_thresholds(y_pred_val_score, score_to_index_thresholds)\n\n        # Calculate Kappa score for the fold\n        kappa_score = cohen_kappa_score(y_val_index, y_pred_val_index, weights='quadratic')\n        kappa_scores.append(kappa_score)\n        \n        if verbose:\n            print(f\"Fold {fold_idx}: Kappa Score = {kappa_score}\")\n    \n    if verbose:\n        print(f\"Mean CV Kappa Score: {np.mean(kappa_scores)}\")\n    \n    return np.mean(kappa_scores), oof_score_predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:08:06.886276Z","iopub.execute_input":"2024-12-25T17:08:06.886661Z","iopub.status.idle":"2024-12-25T17:08:06.905993Z","shell.execute_reply.started":"2024-12-25T17:08:06.886619Z","shell.execute_reply":"2024-12-25T17:08:06.904888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def objective(trial, model_type, X, features, score_col, index_col, cv):\n    # Parameter space to explore if model is xgboost\n    if model_type == 'xgboost':\n        params = {\n            'objective': trial.suggest_categorical('objective', ['reg:squarederror', 'reg:tweedie', 'reg:pseudohubererror']),\n            'random_state': SEED,\n            'n_estimators': trial.suggest_int('n_estimators', 300, 800),\n            'max_depth': trial.suggest_int('max_depth', 3, 10),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 0.02, 0.1),\n            'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n            'gamma': trial.suggest_float('gamma', 0.0, 5.0),\n            'reg_alpha': trial.suggest_loguniform('reg_alpha', 1e-5, 1e-1),\n            'reg_lambda': trial.suggest_loguniform('reg_lambda', 1e-5, 1e-1)\n        }\n        if params['objective'] == 'reg:tweedie':\n            params['tweedie_variance_power'] = trial.suggest_float('tweedie_variance_power', 1, 2)\n        model = XGBRegressor(**params, use_label_encoder=False)\n    \n    # Parameter space to explore if model is lightgbm\n    elif model_type == 'lightgbm':\n        params = {\n            'objective': trial.suggest_categorical('objective', ['poisson', 'tweedie', 'quantile', 'regression']),\n            'random_state': SEED,\n            'verbosity': -1,\n            'n_estimators': trial.suggest_int('n_estimators', 100, 1000),\n            'max_depth': trial.suggest_int('max_depth', 3, 10),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.3),\n            'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n        }\n        if params['objective'] == 'tweedie':\n            params['tweedie_variance_power'] = trial.suggest_float('tweedie_variance_power', 1, 2)\n        model = LGBMRegressor(**params)\n    \n    # Parameter space to explore if model is catboost\n    elif model_type == 'catboost':\n        params = {\n            'loss_function': trial.suggest_categorical('objective', ['Tweedie:variance_power=1.5', 'Quantile', 'Poisson', 'RMSE']),\n            'random_state': SEED,\n            'iterations': trial.suggest_int('iterations', 200, 700),\n            'depth': trial.suggest_int('depth', 4, 10),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.1),\n            'l2_leaf_reg': trial.suggest_loguniform('l2_leaf_reg', 1e-5, 1e-1),\n            'subsample': trial.suggest_float('subsample', 0.5, 1.0)\n        }\n        model = CatBoostRegressor(**params, verbose=0)\n    \n    else:\n        raise ValueError(f\"Unsupported model_type: {model_type}\")\n\n    score, _ = cross_validate(model, X, features, score_col, index_col, cv)\n\n    return score\n\ndef run_optimization(X, features, score_col, index_col, model_type, n_trials=30, cv=None):\n    study = optuna.create_study(direction=\"maximize\")\n    study.optimize(lambda trial: objective(trial, model_type, X, features, score_col, index_col, cv), \n                   n_trials=n_trials)\n    \n    print(f\"Best params for {model_type}: {study.best_params}\")\n    print(f\"Best score: {study.best_value}\")\n    return study.best_params\ndef custom_kappa_scorer(y_true, y_pred):\n    y_pred = round_with_thresholds(y_pred, base_thresholds)\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef perform_rfecv(model, X, y, cv, step=1, scoring='neg_mean_squared_error', min_features_to_select=30, n_jobs=-1, verbose=0):\n    \"\"\"\n    Perform Recursive Feature Elimination with Cross-Validation (RFECV) to select optimal features.\n    \n    Returns:\n    list: Names of the optimal features selected by RFECV.\n    \"\"\"\n    rfecv = RFECV(\n        estimator=model,\n        step=step,\n        cv=cv,\n        scoring=scoring,\n        n_jobs=n_jobs,\n        min_features_to_select=min_features_to_select,\n        verbose=verbose\n    )\n    # Fit the RFECV model and perform feature elimination\n    rfecv.fit(X, y)\n    \n    # Extract optimal features based on ranking\n    optimal_features = [feature for feature, rank in zip(X.columns, rfecv.ranking_) if rank == 1]\n    \n    # Output results\n    print(f\"Optimal number of features: {rfecv.n_features_}\")\n    print(f\"Selected features: {optimal_features}\")\n    \n    return optimal_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:08:06.907402Z","iopub.execute_input":"2024-12-25T17:08:06.907741Z","iopub.status.idle":"2024-12-25T17:08:06.931714Z","shell.execute_reply.started":"2024-12-25T17:08:06.907709Z","shell.execute_reply":"2024-12-25T17:08:06.930287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_features = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI', \n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference', 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_GSND', \n                'FGC-FGC_GSD', 'FGC-FGC_PU', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_TL', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', \n                'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', \n                'BIA-BIA_TBW', 'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-Season', \n                'PreInt_EduHx-computerinternet_hoursday', 'stat_1', 'stat_3', 'stat_4', 'stat_18', 'stat_19', 'stat_20', 'stat_30', 'stat_39', 'stat_41', 'stat_42',\n                'stat_44', 'stat_57', 'stat_64', 'stat_72', 'stat_75', 'stat_76', 'stat_84', 'stat_98', 'stat_102', 'stat_104', 'stat_106', 'stat_116', 'stat_118', \n                'stat_121', 'stat_134', 'stat_136', 'stat_141', 'stat_146', 'stat_149', 'stat_174', 'stat_176', 'stat_184', 'stat_187', 'stat_189', 'stat_192', \n                'stat_196', 'stat_200']\nxgb_features = ['Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-CGAS_Score', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', \n                'Physical-Waist_Circumference', 'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage', \n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-FGC_CU', 'FGC-FGC_GSD', 'FGC-FGC_PU', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_TL', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', \n                'BIA-BIA_Fat', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', \n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-computerinternet_hoursday', 'stat_1', 'stat_3', \n                'stat_4', 'stat_18', 'stat_19', 'stat_20', 'stat_24', 'stat_25', 'stat_28', 'stat_30', 'stat_31', 'stat_33', 'stat_34', 'stat_35', 'stat_38',\n                'stat_39', 'stat_40', 'stat_41', 'stat_44', 'stat_55', 'stat_57', 'stat_58', 'stat_59', 'stat_64', 'stat_65', 'stat_67', 'stat_68', 'stat_69',\n                'stat_70', 'stat_71', 'stat_72', 'stat_73', 'stat_74', 'stat_75', 'stat_76', 'stat_80', 'stat_83', 'stat_96', 'stat_98', 'stat_99', 'stat_100',\n                'stat_102', 'stat_103', 'stat_104', 'stat_105', 'stat_106', 'stat_110', 'stat_111', 'stat_112', 'stat_116', 'stat_118', 'stat_119', 'stat_121',\n                'stat_123', 'stat_134', 'stat_141', 'stat_142', 'stat_144', 'stat_146', 'stat_149', 'stat_151', 'stat_152', 'stat_154', 'stat_156', 'stat_159',\n                'stat_161', 'stat_164', 'stat_174', 'stat_176', 'stat_179', 'stat_180', 'stat_183', 'stat_186', 'stat_188', 'stat_189', 'stat_190', 'stat_191',\n                'stat_192', 'stat_193', 'stat_196', 'stat_197', 'stat_198', 'stat_201']\ncat_features = ['Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', \n                'Physical-Waist_Circumference', 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Season', \n                'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_TL', \n                'BIA-Season', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-Season', \n                'PAQ_A-PAQ_A_Total', 'PAQ_C-Season', 'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', \n                'PreInt_EduHx-computerinternet_hoursday', 'stat_1', 'stat_2', 'stat_3', 'stat_19', 'stat_20', 'stat_21', 'stat_30', 'stat_31', 'stat_33', \n                'stat_34', 'stat_36', 'stat_39', 'stat_40', 'stat_41', 'stat_42', 'stat_43', 'stat_44', 'stat_47', 'stat_54', 'stat_56', 'stat_57', 'stat_60',\n                'stat_62', 'stat_65', 'stat_67', 'stat_68', 'stat_70', 'stat_72', 'stat_73', 'stat_74', 'stat_75', 'stat_76', 'stat_77', 'stat_78', 'stat_80',\n                'stat_84', 'stat_96', 'stat_98', 'stat_100', 'stat_102', 'stat_103', 'stat_104', 'stat_105', 'stat_106', 'stat_109', 'stat_110', 'stat_115',\n                'stat_116', 'stat_121', 'stat_123', 'stat_127', 'stat_134', 'stat_139', 'stat_140', 'stat_142', 'stat_144', 'stat_146', 'stat_149', 'stat_150',\n                'stat_151', 'stat_154', 'stat_159', 'stat_161', 'stat_164', 'stat_167', 'stat_174', 'stat_176', 'stat_181', 'stat_184', 'stat_185', 'stat_186',\n                'stat_189', 'stat_190', 'stat_191', 'stat_192', 'stat_193', 'stat_197', 'stat_200']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:08:06.933818Z","iopub.execute_input":"2024-12-25T17:08:06.934267Z","iopub.status.idle":"2024-12-25T17:08:06.960729Z","shell.execute_reply.started":"2024-12-25T17:08:06.934219Z","shell.execute_reply":"2024-12-25T17:08:06.959432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fea = train.columns.tolist()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_features = features\nxgb_features = features\ncat_features = features","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Parameters for LGBM, XGB and CatBoost\nlgb_params = {\n    'objective': 'regression', \n    'n_estimators': 165, \n    'max_depth': 12, \n    'learning_rate': 0.046, \n    'subsample': 0.6734915155561385, \n    'colsample_bytree': 0.5263197975884975\n}\n\nxgb_params = {\n    'objective': 'reg:squarederror', \n    'n_estimators': 478, \n    'max_depth': 6, \n    'learning_rate': 0.03416918255604553, \n    'subsample': 0.8, \n    'colsample_bytree': 0.8, \n    'gamma': 1.3368387307987768, \n    'reg_alpha': 1, \n    'reg_lambda': 5\n}\n\ncat_params = {\n    'objective': 'RMSE', \n    'iterations': 200, \n    'depth': 6, \n    'learning_rate': 0.05, \n    'l2_leaf_reg': 10, \n    'subsample': 0.5002380405809761\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:08:06.962224Z","iopub.execute_input":"2024-12-25T17:08:06.962684Z","iopub.status.idle":"2024-12-25T17:08:06.984605Z","shell.execute_reply.started":"2024-12-25T17:08:06.962637Z","shell.execute_reply":"2024-12-25T17:08:06.983534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kappa_scorer = make_scorer(custom_kappa_scorer, greater_is_better=True)\nkf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\nif optimize_features:\n    # LightGBM Feature Selection\n    lgb_model = LGBMRegressor(**lgb_params, random_state=SEED, verbosity=0)\n    lgb_features = perform_rfecv(\n        model=lgb_model,\n        X=train[lgb_features],\n        y=train['PCIAT-PCIAT_Total'],\n        cv=kf,\n        scoring=kappa_scorer,\n        min_features_to_select=min_features_to_select,\n        verbose=1\n    )\n\n    # XGBoost Feature Selection\n    xgb_model = XGBRegressor(**xgb_params, random_state=SEED, verbose=0, tree_method = \"hist\")\n    xgb_features = perform_rfecv(\n        model=xgb_model,\n        X=train[xgb_features],\n        y=train['PCIAT-PCIAT_Total'],\n        cv=kf,\n        scoring=kappa_scorer,\n        min_features_to_select=min_features_to_select,\n        verbose=1\n    )\n\n    # CatBoost Feature Selection\n    cat_model = CatBoostRegressor(**cat_params, random_state=SEED, verbose=0)\n    cat_features = perform_rfecv(\n        model=cat_model,\n        X=train[cat_features],\n        y=train['PCIAT-PCIAT_Total'],\n        cv=kf,\n        scoring=kappa_scorer,\n        min_features_to_select=min_features_to_select,\n        verbose=1\n    )\n\nif optimize_params:\n    # LightGBM Optimization\n    lgb_params = run_optimization(train, lgb_features, 'PCIAT-PCIAT_Total', 'sii', 'lightgbm', n_trials=n_trials, cv=kf)\n\n    # XGBoost Optimization\n    xgb_params = run_optimization(train, xgb_features, 'PCIAT-PCIAT_Total', 'sii', 'xgboost', n_trials=n_trials, cv=kf)\n\n    # CatBoost Optimization\n    cat_params = run_optimization(train, cat_features, 'PCIAT-PCIAT_Total', 'sii', 'catboost', n_trials=n_trials, cv=kf)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:08:06.985997Z","iopub.execute_input":"2024-12-25T17:08:06.986384Z","iopub.status.idle":"2024-12-25T17:08:07.007464Z","shell.execute_reply.started":"2024-12-25T17:08:06.986349Z","shell.execute_reply":"2024-12-25T17:08:07.006182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define models\nlgb_model = LGBMRegressor(**lgb_params, random_state=SEED, verbosity=-1)\nxgb_model = XGBRegressor(**xgb_params, random_state=SEED, verbosity=0)\ncat_model = CatBoostRegressor(**cat_params, random_state=SEED, verbose=0)\n\n# Cross-validate LGBM model\nscore_lgb, oof_lgb = cross_validate(lgb_model, train, lgb_features, 'PCIAT-PCIAT_Total', 'sii', kf, verbose=True)\nlgb_model.fit(train[lgb_features], train['PCIAT-PCIAT_Total'])\ntest_lgb = lgb_model.predict(test[lgb_features])\n\n# Cross-validate XGBoost model\nscore_xgb, oof_xgb = cross_validate(xgb_model, train, xgb_features, 'PCIAT-PCIAT_Total', 'sii', kf, verbose=True)\nxgb_model.fit(train[xgb_features], train['PCIAT-PCIAT_Total'])\ntest_xgb = xgb_model.predict(test[xgb_features])\n\n# Cross-validate CatBoost model\nscore_cat, oof_cat = cross_validate(cat_model, train, cat_features, 'PCIAT-PCIAT_Total', 'sii', kf, verbose=True)\ncat_model.fit(train[cat_features], train['PCIAT-PCIAT_Total'])\ntest_cat = cat_model.predict(test[cat_features])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:08:07.008822Z","iopub.execute_input":"2024-12-25T17:08:07.009242Z","iopub.status.idle":"2024-12-25T17:11:37.486816Z","shell.execute_reply.started":"2024-12-25T17:08:07.009198Z","shell.execute_reply":"2024-12-25T17:11:37.485250Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot predicted scores against true scores with the base thresholds for converting score to sii\nsns.set_theme(style=\"white\")\nfig, axes = plt.subplots(1, 3, figsize=(14, 6))\n\nscatter1 = axes[0].scatter(train['PCIAT-PCIAT_Total'], oof_lgb, c=train[\"sii\"], cmap=\"autumn\", alpha=0.5)\naxes[0].set_xlabel(\"True Score\")\naxes[0].set_ylabel(\"OOF Predictions - LGBM\")\n\nthresholds = [30, 50, 80]\nfor threshold in thresholds:\n    axes[0].axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    axes[0].axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n\nscatter2 = axes[1].scatter(train['PCIAT-PCIAT_Total'], oof_xgb, c=train[\"sii\"], cmap=\"autumn\", alpha=0.5)\naxes[1].set_xlabel(\"True Score\")\naxes[1].set_ylabel(\"OOF Predictions - XGB\")\n\nfor threshold in thresholds:\n    axes[1].axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    axes[1].axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    \nscatter3 = axes[2].scatter(train['PCIAT-PCIAT_Total'], oof_cat, c=train[\"sii\"], cmap=\"autumn\", alpha=0.5)\naxes[2].set_xlabel(\"True Score\")\naxes[2].set_ylabel(\"OOF Predictions - Cat\")\n\nfor threshold in thresholds:\n    axes[2].axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    axes[2].axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:11:37.488550Z","iopub.execute_input":"2024-12-25T17:11:37.488958Z","iopub.status.idle":"2024-12-25T17:11:38.726086Z","shell.execute_reply.started":"2024-12-25T17:11:37.488920Z","shell.execute_reply":"2024-12-25T17:11:38.724801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Optimize thresholds for each model's OOF predictions\nlgb_thresholds = optimize_thresholds(train[\"sii\"], oof_lgb, start_vals=base_thresholds)\nprint(f\"LGBM optimized thresholds: {lgb_thresholds}\")\n\nxgb_thresholds = optimize_thresholds(train[\"sii\"], oof_xgb, start_vals=base_thresholds)\nprint(f\"XGBoost optimized thresholds: {xgb_thresholds}\")\n\ncat_thresholds = optimize_thresholds(train[\"sii\"], oof_cat, start_vals=base_thresholds)\nprint(f\"CatBoost optimized thresholds: {cat_thresholds}\")\n\n# Apply the optimized thresholds to OOF predictions\noof_lgb = round_with_thresholds(oof_lgb, lgb_thresholds)\noof_xgb = round_with_thresholds(oof_xgb, xgb_thresholds)\noof_cat = round_with_thresholds(oof_cat, cat_thresholds)\nvoted_oof = stats.mode(np.array([oof_lgb, oof_xgb, oof_cat]), axis=0).mode.flatten().astype(int)\n\n# Calculate Kappa score for voted OOF predictions\nkappa_score = cohen_kappa_score(train[\"sii\"], voted_oof, weights='quadratic')\nprint(f\"Voted ensemble Kappa score: {kappa_score}\")\n# Plot confusion matrix\nconf_matrix = confusion_matrix(train[\"sii\"], voted_oof)\nsns.set_theme(style=\"whitegrid\")\nplt.figure(figsize=(8, 6))\nsns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"autumn\", cbar=False, linewidths=0.5, linecolor='black')\nplt.title('Confusion Matrix', fontsize=16)\nplt.xlabel('Predicted', fontsize=12)\nplt.ylabel('True', fontsize=12)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:11:38.727549Z","iopub.execute_input":"2024-12-25T17:11:38.727940Z","iopub.status.idle":"2024-12-25T17:11:40.521770Z","shell.execute_reply.started":"2024-12-25T17:11:38.727908Z","shell.execute_reply":"2024-12-25T17:11:40.520620Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom scipy import stats\n\ntest_lgb = round_with_thresholds(test_lgb, lgb_thresholds)\ntest_xgb = round_with_thresholds(test_xgb, xgb_thresholds)\ntest_cat = round_with_thresholds(test_cat, cat_thresholds)\n\n# Kết hợp các dự đoán từ ba mô hình\npredictions = np.array([test_lgb, test_xgb, test_cat])\n\n# Tính mode cho mỗi chỉ mục (cột)\nmode_preds, mode_count = stats.mode(predictions, axis=0)\n\n# Kiểm tra số lần xuất hiện của mode:\n# - Nếu mode_count > 1 (có giá trị xuất hiện nhiều nhất), sử dụng mode.\n# - Nếu không (các giá trị khác nhau), tính trung bình (mean).\nvoted_test = np.where(mode_count > 1, mode_preds.flatten(), np.mean(predictions, axis=0))\n\nvoted_test = np.round(voted_test).astype(int)\n\nsubmission = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv\")\n\nsubmission['sii'] = voted_test\n\nsubmission.to_csv(\"submission.csv\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-25T17:11:40.523054Z","iopub.execute_input":"2024-12-25T17:11:40.523368Z","iopub.status.idle":"2024-12-25T17:11:40.545540Z","shell.execute_reply.started":"2024-12-25T17:11:40.523339Z","shell.execute_reply":"2024-12-25T17:11:40.544306Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}