{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score, make_scorer, confusion_matrix\nfrom sklearn.model_selection import StratifiedKFold, KFold\nfrom scipy.optimize import minimize\nfrom scipy import stats\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom sklearn.feature_selection import RFECV\nfrom sklearn.linear_model import ElasticNetCV, LassoCV, Lasso\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nimport optuna\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-30T07:59:11.400946Z","iopub.execute_input":"2024-10-30T07:59:11.401397Z","iopub.status.idle":"2024-10-30T07:59:16.682839Z","shell.execute_reply.started":"2024-10-30T07:59:11.401353Z","shell.execute_reply":"2024-10-30T07:59:16.681590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SEED = 42\nn_splits = 10\noptimize_params = False\nn_trials = 25 # n_trials for optuna \nbase_thresholds = [30, 50, 80]\ny_model = \"PCIAT-PCIAT_Total\" # Score, target for the model\ny_comp = \"sii\" # Index, target of the competition","metadata":{"execution":{"iopub.status.busy":"2024-10-30T08:06:11.626061Z","iopub.execute_input":"2024-10-30T08:06:11.626498Z","iopub.status.idle":"2024-10-30T08:06:11.632854Z","shell.execute_reply.started":"2024-10-30T08:06:11.626446Z","shell.execute_reply":"2024-10-30T08:06:11.631449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load datasets\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-30T07:59:16.694683Z","iopub.execute_input":"2024-10-30T07:59:16.695164Z","iopub.status.idle":"2024-10-30T07:59:16.786531Z","shell.execute_reply.started":"2024-10-30T07:59:16.695108Z","shell.execute_reply":"2024-10-30T07:59:16.785324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def time_features(df):\n    # Convert time_of_day to hours\n    df[\"hours\"] = df[\"time_of_day\"] // (3_600 * 1_000_000_000)\n    # Basic features \n    features = [\n        df[\"non-wear_flag\"].mean(),\n        df[\"enmo\"][df[\"enmo\"] >= 0.05].sum(),\n    ]\n    \n    # Define conditions for night, day, and no mask (full data)\n    night = ((df[\"hours\"] >= 22) | (df[\"hours\"] <= 5))\n    day = ((df[\"hours\"] <= 20) & (df[\"hours\"] >= 7))\n    no_mask = np.ones(len(df), dtype=bool)\n    \n    # List of columns of interest and masks\n    keys = [\"enmo\", \"anglez\", \"light\", \"battery_voltage\"]\n    masks = [no_mask, night, day]\n    \n    # Helper function for feature extraction\n    def extract_stats(data):\n        return [\n            data.mean(), \n            data.std(), \n            data.max(), \n            data.min(), \n            data.diff().mean(), \n            data.diff().std()\n        ]\n    \n    # Iterate over keys and masks to generate the statistics\n    for key in keys:\n        for mask in masks:\n            filtered_data = df.loc[mask, key]\n            features.extend(extract_stats(filtered_data))\n\n    return features\n\n# Code for parallelized computation of time series data from: Sheikh Muhammad Abdullah \n# https://www.kaggle.com/code/abdmental01/cmi-best-single-model\ndef process_file(filename, dirname):\n    # Process file and extract time features\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return time_features(df), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    # Load time series from directory in parallel\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-10-30T08:02:01.317861Z","iopub.execute_input":"2024-10-30T08:02:01.319004Z","iopub.status.idle":"2024-10-30T08:02:01.342933Z","shell.execute_reply.started":"2024-10-30T08:02:01.318932Z","shell.execute_reply":"2024-10-30T08:02:01.341183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\ntrain = train[train[y_comp].notna()] # Keep rows where target is available\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-30T08:02:02.381697Z","iopub.execute_input":"2024-10-30T08:02:02.382169Z","iopub.status.idle":"2024-10-30T08:03:40.530122Z","shell.execute_reply.started":"2024-10-30T08:02:02.382124Z","shell.execute_reply":"2024-10-30T08:03:40.528523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot distribution of total scores which determine the sii\n# Note the excess zeros -> consider other objective functions\nsns.set_theme(style=\"whitegrid\")\nplt.hist(train[y_model], bins=40, color=\"darkorange\")\nplt.title('Score Distribution')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-30T08:05:04.979558Z","iopub.execute_input":"2024-10-30T08:05:04.981600Z","iopub.status.idle":"2024-10-30T08:05:05.496846Z","shell.execute_reply.started":"2024-10-30T08:05:04.981514Z","shell.execute_reply":"2024-10-30T08:05:05.495475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Features to exclude, because they're not in test\nexclude = ['PCIAT-Season', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03',\n           'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07',\n           'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11',\n           'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15',\n           'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19',\n           'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total', 'sii']\n\nfeatures = [f for f in train.columns if f not in exclude]\n\n# Categorical features\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n          'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\nfor col in cat_c:\n    a_map = {}\n    all_unique = set(train[col].unique()) | set(test[col].unique())\n    for i, value in enumerate(all_unique):\n        a_map[value] = i\n\n    train[col] = train[col].map(a_map)\n    test[col] = test[col].map(a_map)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T08:05:17.284043Z","iopub.execute_input":"2024-10-30T08:05:17.284529Z","iopub.status.idle":"2024-10-30T08:05:17.317078Z","shell.execute_reply.started":"2024-10-30T08:05:17.284482Z","shell.execute_reply":"2024-10-30T08:05:17.315457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Impute_With_Model:\n    \n    def __init__(self, na_frac=0.5, min_samples=0):\n        self.model_dict = {}\n        self.mean_dict = {}\n        self.features = None\n        self.na_frac = na_frac\n        self.min_samples = min_samples\n        \n    def find_features(self, data, feature, tmp_features):\n        missing_rows = data[feature].isna()\n        na_fraction = data[missing_rows][tmp_features].isna().mean(axis=0)\n        valid_features = np.array(tmp_features)[na_fraction <= self.na_frac]\n        return valid_features\n\n    def fit_models(self, model, data, features):\n        self.features = features\n        n_data = data.shape[0]\n        for feature in features:\n            self.mean_dict[feature] = np.mean(data[feature])\n        for feature in tqdm(features):\n            if data[feature].isna().sum() > 0:\n                model_clone = clone(model)\n                X = data[data[feature].notna()].copy()\n                tmp_features = [f for f in features if f != feature]\n                tmp_features = self.find_features(data, feature, tmp_features)\n                if len(tmp_features) >= 1 and X.shape[0] > self.min_samples:\n                    for f in tmp_features:\n                        X[f] = X[f].fillna(self.mean_dict[f])\n                    model_clone.fit(X[tmp_features], X[feature])\n                    self.model_dict[feature] = (model_clone, tmp_features.copy())\n                else:\n                    self.model_dict[feature] = (\"mean\", np.mean(data[feature]))\n            \n    def impute(self, data):\n        imputed_data = data.copy()\n        for feature, model in self.model_dict.items():\n            missing_rows = imputed_data[feature].isna()\n            if missing_rows.any():\n                if model[0] == \"mean\":\n                    imputed_data[feature].fillna(model[1], inplace=True)\n                else:\n                    tmp_features = [f for f in self.features if f != feature]\n                    X_missing = data.loc[missing_rows, tmp_features].copy()\n                    for f in tmp_features:\n                        X_missing[f] = X_missing[f].fillna(self.mean_dict[f])\n                    imputed_data.loc[missing_rows, feature] = model[0].predict(X_missing[model[1]])\n        return imputed_data","metadata":{"execution":{"iopub.status.busy":"2024-10-30T08:05:20.112700Z","iopub.execute_input":"2024-10-30T08:05:20.113178Z","iopub.status.idle":"2024-10-30T08:05:20.132248Z","shell.execute_reply.started":"2024-10-30T08:05:20.113135Z","shell.execute_reply":"2024-10-30T08:05:20.130371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Code for finding optimal thresholds copied from: Michael Semenoff\n# https://www.kaggle.com/code/michaelsemenoff/cmi-actigraphy-feature-engineering-selection\ndef round_with_thresholds(raw_preds, thresholds):\n    return np.where(raw_preds < thresholds[0], int(0),\n                    np.where(raw_preds < thresholds[1], int(1),\n                             np.where(raw_preds < thresholds[2], int(2), int(3))))\n\ndef optimize_thresholds(y_true, raw_preds, start_vals=[0.5, 1.5, 2.5]):\n    def fun(thresholds, y_true, raw_preds):\n        rounded_preds = round_with_thresholds(raw_preds, thresholds)\n        return -cohen_kappa_score(y_true, rounded_preds, weights='quadratic')\n\n    res = minimize(fun, x0=start_vals, args=(y_true, raw_preds), method='Nelder-Mead')\n    assert res.success\n    return res.x","metadata":{"execution":{"iopub.status.busy":"2024-10-30T08:05:22.208011Z","iopub.execute_input":"2024-10-30T08:05:22.208471Z","iopub.status.idle":"2024-10-30T08:05:22.219157Z","shell.execute_reply.started":"2024-10-30T08:05:22.208426Z","shell.execute_reply":"2024-10-30T08:05:22.217779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def calculate_weights(series):\n    # Create bins for the target variable and assign weights based on frequency\n    bins = pd.cut(series, bins=10, labels=False)\n    weights = bins.value_counts().reset_index()\n    weights.columns = ['target_bins', 'count']\n    weights['count'] = 1 / weights['count']\n    weight_map = weights.set_index('target_bins')['count'].to_dict()\n    weights = bins.map(weight_map)\n    return weights / weights.mean() ","metadata":{"execution":{"iopub.status.busy":"2024-10-30T08:05:24.071066Z","iopub.execute_input":"2024-10-30T08:05:24.071539Z","iopub.status.idle":"2024-10-30T08:05:24.079741Z","shell.execute_reply.started":"2024-10-30T08:05:24.071495Z","shell.execute_reply":"2024-10-30T08:05:24.078327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cross_validate(model_, data, features, score_col, index_col, cv, sample_weights=False, verbose=False):\n    \"\"\"\n    Perform cross-validation with a given model and compute the out-of-fold \n    predictions and Cohen's Kappa score for each fold.\n\n    Returns:\n    float: Mean Kappa score across all folds.\n    array: Out-of-fold score predictions for the entire dataset.\n    \"\"\"\n    kappa_scores = [] \n    oof_score_predictions = np.zeros(len(data))  \n\n    score_to_index_thresholds = base_thresholds  \n\n    for fold_idx, (train_idx, val_idx) in enumerate(cv.split(data, data[index_col])):\n        X_train, X_val = data[features].iloc[train_idx], data[features].iloc[val_idx]\n        y_train_score = data[score_col].iloc[train_idx]  \n        y_val_score = data[score_col].iloc[val_idx]      \n        y_val_index = data[index_col].iloc[val_idx]     \n        \n        # Train model with sample weights if provided\n        if sample_weights:\n            weights = calculate_weights(y_train_score)\n            model_.fit(X_train, y_train_score, sample_weight=weights)\n        else:\n            model_.fit(X_train, y_train_score)\n        \n        y_pred_val_score = model_.predict(X_val)\n        \n        oof_score_predictions[val_idx] = y_pred_val_score \n\n        y_pred_val_index = round_with_thresholds(y_pred_val_score, score_to_index_thresholds)\n\n        kappa_score = cohen_kappa_score(y_val_index, y_pred_val_index, weights='quadratic')\n        kappa_scores.append(kappa_score)\n        \n        if verbose:\n            print(f\"Fold {fold_idx}: Kappa Score = {kappa_score}\")\n    \n    if verbose:\n        print(f\"## Mean CV Kappa Score: {np.mean(kappa_scores)} ##\")\n    \n    return np.mean(kappa_scores), oof_score_predictions","metadata":{"execution":{"iopub.status.busy":"2024-10-30T08:05:24.569681Z","iopub.execute_input":"2024-10-30T08:05:24.570148Z","iopub.status.idle":"2024-10-30T08:05:24.586763Z","shell.execute_reply.started":"2024-10-30T08:05:24.570075Z","shell.execute_reply":"2024-10-30T08:05:24.585163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def objective(trial, model_type, X, features, score_col, index_col, cv, sample_weights=False):\n    # Parameter space to explore if model is xgboost\n    if model_type == 'xgboost':\n        params = {\n            'objective': trial.suggest_categorical('objective', ['reg:squarederror','reg:tweedie', 'reg:pseudohubererror']),\n            'random_state': SEED,\n            'n_estimators': trial.suggest_int('n_estimators', 300, 600),\n            'max_depth': trial.suggest_int('max_depth', 3, 8),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 0.02, 0.1),\n            'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n            'gamma': trial.suggest_float('gamma', 0.0, 5.0),\n            'reg_alpha': trial.suggest_loguniform('reg_alpha', 1e-5, 1e-1),\n            'reg_lambda': trial.suggest_loguniform('reg_lambda', 1e-5, 1e-1)\n        }\n        if params['objective'] == 'reg:tweedie':\n            params['tweedie_variance_power'] = trial.suggest_float('tweedie_variance_power', 1, 2)\n        model = XGBRegressor(**params, use_label_encoder=False)\n    \n    # Parameter space to explore if model is lightgbm\n    elif model_type == 'lightgbm':\n        params = {\n            'objective': trial.suggest_categorical('objective', ['poisson', 'tweedie', 'regression']),\n            'random_state': SEED,\n            'verbosity': -1,\n            'n_estimators': trial.suggest_int('n_estimators', 100, 600),\n            'max_depth': trial.suggest_int('max_depth', 3, 10),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.3),\n            'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n        }\n        if params['objective'] == 'tweedie':\n            params['tweedie_variance_power'] = trial.suggest_float('tweedie_variance_power', 1, 2)\n        model = LGBMRegressor(**params)\n    \n    # Parameter space to explore if model is catboost\n    elif model_type == 'catboost':\n        params = {\n            'loss_function': trial.suggest_categorical('objective', ['Tweedie:variance_power=1.5', 'Poisson', 'RMSE']),\n            'random_state': SEED,\n            'iterations': trial.suggest_int('iterations', 200, 600),\n            'depth': trial.suggest_int('depth', 4, 10),\n            'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.1),\n            'l2_leaf_reg': trial.suggest_loguniform('l2_leaf_reg', 1e-5, 1e-1),\n            'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n            'bagging_temperature': trial.suggest_float('bagging_temperature', 0.0, 1.0),\n            'random_strength': trial.suggest_float('random_strength', 1e-3, 10.0),\n            'colsample_bylevel': trial.suggest_float('colsample_bylevel', 0.5, 1.0),\n            'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 1, 100),\n        }\n        model = CatBoostRegressor(**params, verbose=0)\n    \n    else:\n        raise ValueError(f\"Unsupported model_type: {model_type}\")\n\n    score, _ = cross_validate(model, X, features, score_col, index_col, cv, sample_weights=True, verbose=False)\n\n    return score\n\ndef run_optimization(X, features, score_col, index_col, model_type, n_trials=30, cv=None, sample_weights=False):\n    study = optuna.create_study(direction=\"maximize\")\n    study.optimize(lambda trial: objective(trial, model_type, X, features, score_col, index_col, cv, sample_weights), \n                   n_trials=n_trials)\n    \n    print(f\"Best params for {model_type}: {study.best_params}\")\n    print(f\"Best score: {study.best_value}\")\n    return study.best_params","metadata":{"execution":{"iopub.status.busy":"2024-10-30T08:05:25.414428Z","iopub.execute_input":"2024-10-30T08:05:25.414865Z","iopub.status.idle":"2024-10-30T08:05:25.436744Z","shell.execute_reply.started":"2024-10-30T08:05:25.414824Z","shell.execute_reply":"2024-10-30T08:05:25.435587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Replace if subsets for features have been selected\nlgb_features = features\nxgb_features = features\ncat_features = features","metadata":{"execution":{"iopub.status.busy":"2024-10-30T08:05:28.272642Z","iopub.execute_input":"2024-10-30T08:05:28.273740Z","iopub.status.idle":"2024-10-30T08:05:28.280440Z","shell.execute_reply.started":"2024-10-30T08:05:28.273680Z","shell.execute_reply":"2024-10-30T08:05:28.278587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parameters for LGBM, XGB and CatBoost\nlgb_params = {\n    'objective': 'tweedie', \n    'n_estimators': 242, \n    'max_depth': 4, \n    'learning_rate': 0.029229916231368648, \n    'subsample': 0.9435713052516868, \n    'colsample_bytree': 0.6372563562692964, \n    'tweedie_variance_power': 1.7598875942201002\n}\n\nxgb_params = {\n    'objective': 'reg:tweedie', \n    'n_estimators': 554, \n    'max_depth': 3, \n    'learning_rate': 0.020148793517835852, \n    'subsample': 0.7245109070247534, \n    'colsample_bytree': 0.7516980111036932, \n    'gamma': 1.4405479996512962, \n    'reg_alpha': 0.00015467164351805926, \n    'reg_lambda': 0.011510449488765364, \n    'tweedie_variance_power': 1.2525085567413385\n}\n\ncat_params = {\n    'objective': 'RMSE', \n    'iterations': 476, \n    'depth': 6, \n    'learning_rate': 0.01508021072978329, \n    'l2_leaf_reg': 0.009219274204258077, \n    'subsample': 0.909899776448952, \n    'bagging_temperature': 0.4068004305795976, \n    'random_strength': 0.13085860045085365, \n    'colsample_bylevel': 0.5000595287404359,\n    'min_data_in_leaf': 27\n}","metadata":{"execution":{"iopub.status.busy":"2024-10-30T08:05:28.832372Z","iopub.execute_input":"2024-10-30T08:05:28.832848Z","iopub.status.idle":"2024-10-30T08:05:28.844519Z","shell.execute_reply.started":"2024-10-30T08:05:28.832806Z","shell.execute_reply":"2024-10-30T08:05:28.842891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Lasso(alpha=0.3, random_state=SEED)\nimputer = Impute_With_Model(na_frac=0.3)\nimputer.fit_models(model, train, features)\ntrain = imputer.impute(train)\ntest = imputer.impute(test)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T08:05:51.560209Z","iopub.execute_input":"2024-10-30T08:05:51.560802Z","iopub.status.idle":"2024-10-30T08:06:11.624041Z","shell.execute_reply.started":"2024-10-30T08:05:51.560745Z","shell.execute_reply":"2024-10-30T08:06:11.622866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\nif optimize_params:\n    # LightGBM Optimization\n    lgb_params = run_optimization(train, lgb_features, y_model, y_comp, 'lightgbm', n_trials=n_trials, cv=kf, sample_weights=True)\n\n    # XGBoost Optimization\n    xgb_params = run_optimization(train, xgb_features, y_model, y_comp, 'xgboost', n_trials=n_trials, cv=kf, sample_weights=True)\n\n    # CatBoost Optimization\n    cat_params = run_optimization(train, cat_features, y_model, y_comp, 'catboost', n_trials=n_trials, cv=kf, sample_weights=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T08:06:17.253729Z","iopub.execute_input":"2024-10-30T08:06:17.254205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define models\nlgb_model = LGBMRegressor(**lgb_params, random_state=SEED, verbosity=-1)\nxgb_model = XGBRegressor(**xgb_params, random_state=SEED, verbosity=0)\ncat_model = CatBoostRegressor(**cat_params, random_state=SEED, verbose=0)\n\nweights = calculate_weights(train[y_model])\n\n# Cross-validate LGBM model\nscore_lgb, oof_lgb = cross_validate(lgb_model, train, lgb_features, y_model, y_comp, kf, verbose=True, sample_weights=True)\nlgb_model.fit(train[lgb_features], train[y_model], sample_weight=weights)\ntest_lgb = lgb_model.predict(test[lgb_features])\n\n# Cross-validate XGBoost model\nscore_xgb, oof_xgb = cross_validate(xgb_model, train, xgb_features, y_model, y_comp, kf, verbose=True, sample_weights=True)\nxgb_model.fit(train[xgb_features], train[y_model], sample_weight=weights)\ntest_xgb = xgb_model.predict(test[xgb_features])\n\n# Cross-validate CatBoost model\nscore_cat, oof_cat = cross_validate(cat_model, train, cat_features, y_model, y_comp, kf, verbose=True, sample_weights=True)\ncat_model.fit(train[cat_features], train[y_model], sample_weight=weights)\ntest_cat = cat_model.predict(test[cat_features])\n\nprint(f'Overall Mean Kappa: {np.mean([score_lgb, score_xgb, score_cat])}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot predicted scores against true scores with the base thresholds for converting score to sii\nsns.set_theme(style=\"white\")\nfig, axes = plt.subplots(1, 3, figsize=(14, 6))\n\nscatter1 = axes[0].scatter(train[y_model], oof_lgb, c=train[y_comp], cmap=\"autumn\", alpha=0.5)\naxes[0].set_xlabel(\"True Score\")\naxes[0].set_ylabel(\"OOF Predictions - LGBM\")\naxes[0].set_ylim(0,np.max(train[y_model]))\naxes[0].set_xlim(0,np.max(train[y_model]))\naxes[0].set_aspect('equal', adjustable='box')\n\nthresholds = [30, 50, 80]\nfor threshold in thresholds:\n    axes[0].axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    axes[0].axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n\nscatter2 = axes[1].scatter(train[y_model], oof_xgb, c=train[y_comp], cmap=\"autumn\", alpha=0.5)\naxes[1].set_xlabel(\"True Score\")\naxes[1].set_ylabel(\"OOF Predictions - XGB\")\naxes[1].set_ylim(0,np.max(train[y_model]))\naxes[1].set_xlim(0,np.max(train[y_model]))\naxes[1].set_aspect('equal', adjustable='box')\n\nfor threshold in thresholds:\n    axes[1].axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    axes[1].axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    \nscatter3 = axes[2].scatter(train[y_model], oof_cat, c=train[y_comp], cmap=\"autumn\", alpha=0.5)\naxes[2].set_xlabel(\"True Score\")\naxes[2].set_ylabel(\"OOF Predictions - Cat\")\naxes[2].set_ylim(0,np.max(train[y_model]))\naxes[2].set_xlim(0,np.max(train[y_model]))\naxes[2].set_aspect('equal', adjustable='box')\n\nfor threshold in thresholds:\n    axes[2].axhline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n    axes[2].axvline(threshold, color=\"blue\", linestyle=\"--\", lw=1)\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_importances = pd.DataFrame({\n    'Feature': features,\n    'Importance': lgb_model.feature_importances_\n}).sort_values(by='Importance', ascending=False)\n\nxgb_importances = pd.DataFrame({\n    'Feature': features,\n    'Importance': xgb_model.feature_importances_\n}).sort_values(by='Importance', ascending=False)\n\ncat_importances = pd.DataFrame({\n    'Feature': features,\n    'Importance': cat_model.feature_importances_\n}).sort_values(by='Importance', ascending=False)\n\n# Set the number of features to display\nn_top_features = 30\n\nfig, axes = plt.subplots(1, 3, figsize=(18, 8))\nsns.set_theme(style=\"whitegrid\")\n\nsns.barplot(ax=axes[0], data=lgb_importances.head(n_top_features),\n            x='Importance', y='Feature', palette=\"autumn\")\naxes[0].set_title('LightGBM Top Feature Importances')\n\nsns.barplot(ax=axes[1], data=xgb_importances.head(n_top_features),\n            x='Importance', y='Feature', palette=\"autumn\")\naxes[1].set_title('XGBoost Top Feature Importances')\n\nsns.barplot(ax=axes[2], data=cat_importances.head(n_top_features),\n            x='Importance', y='Feature', palette=\"autumn\")\naxes[2].set_title('CatBoost Top Feature Importances')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_preds = pd.DataFrame({\n    'lgb': oof_lgb,\n    'xgb': oof_xgb,\n    'cat': oof_cat\n})\ncorr_df = model_preds.corr()\nsns.heatmap(corr_df, annot=True, cmap=\"autumn\", cbar=False, linewidths=0.5, linecolor='black')\nplt.title(\"Correlation Between Models\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-12T09:12:39.765456Z","iopub.execute_input":"2024-10-12T09:12:39.765903Z","iopub.status.idle":"2024-10-12T09:12:39.948953Z","shell.execute_reply.started":"2024-10-12T09:12:39.765858Z","shell.execute_reply":"2024-10-12T09:12:39.947829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Optimize thresholds for each model's OOF predictions\nlgb_thresholds = optimize_thresholds(train[y_comp], oof_lgb, start_vals=base_thresholds)\nprint(f\"LGBM optimized thresholds: {lgb_thresholds}\")\n\nxgb_thresholds = optimize_thresholds(train[y_comp], oof_xgb, start_vals=base_thresholds)\nprint(f\"XGBoost optimized thresholds: {xgb_thresholds}\")\n\ncat_thresholds = optimize_thresholds(train[y_comp], oof_cat, start_vals=base_thresholds)\nprint(f\"CatBoost optimized thresholds: {cat_thresholds}\")\n\n# Apply the optimized thresholds to OOF predictions\noof_lgb = round_with_thresholds(oof_lgb, lgb_thresholds)\noof_xgb = round_with_thresholds(oof_xgb, xgb_thresholds)\noof_cat = round_with_thresholds(oof_cat, cat_thresholds)\nvoted_oof = stats.mode(np.array([oof_lgb, oof_xgb, oof_cat]), axis=0).mode.flatten().astype(int)\n\n# Calculate Kappa score for voted OOF predictions\nkappa_score = cohen_kappa_score(train[y_comp], voted_oof, weights='quadratic')\nprint(f\"Voted ensemble Kappa score: {kappa_score}\")\n# Plot confusion matrix\nconf_matrix = confusion_matrix(train[y_comp], voted_oof)\nsns.set_theme(style=\"whitegrid\")\nplt.figure(figsize=(8, 6))\nsns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"autumn\", cbar=False, linewidths=0.5, linecolor='black')\nplt.title('Confusion Matrix', fontsize=16)\nplt.xlabel('Predicted', fontsize=12)\nplt.ylabel('True', fontsize=12)\nplt.show()\n\n# Apply the optimized thresholds to test predictions\ntest_lgb = round_with_thresholds(test_lgb, lgb_thresholds)\ntest_xgb = round_with_thresholds(test_xgb, xgb_thresholds)\ntest_cat = round_with_thresholds(test_cat, cat_thresholds)\nvoted_test = stats.mode(np.array([test_lgb, test_xgb, test_cat]), axis=0).mode.flatten().astype(int)\n\nsubmission = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv\")\nsubmission[y_comp] = voted_test\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-12T09:12:39.950559Z","iopub.execute_input":"2024-10-12T09:12:39.951313Z","iopub.status.idle":"2024-10-12T09:12:41.429277Z","shell.execute_reply.started":"2024-10-12T09:12:39.951262Z","shell.execute_reply":"2024-10-12T09:12:41.428146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}