{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":false,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport optuna\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import train_test_split\nimport lightgbm as lgb\nfrom sklearn.base import clone\nfrom sklearn.model_selection import KFold\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.ensemble import VotingClassifier\nfrom catboost import CatBoostClassifier\n\nimport warnings\n# Suppress all warnings\nwarnings.filterwarnings(\"ignore\")\n\n# Alternatively, suppress specific types of warnings like FutureWarning or UserWarning\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shape of the data:\nprint(\"train_data :\", train_data.shape)\nprint(\"test_data :\", test_data.shape)\nprint(\"sample_submission_data :\", sample_data.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nfrom IPython.display import clear_output\nfrom concurrent.futures import ThreadPoolExecutor\n\ndef process_file(filename, dirname):\n    data = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    data.drop('step', axis=1, inplace=True)\n    return data.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    data = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    data['id'] = indexes\n    \n    return data\n        \ntrain_parquet = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_parquet = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntime_series_cols = train_parquet.columns.tolist()\ntime_series_cols.remove(\"id\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.merge(train_data, train_parquet, how=\"left\", on='id')\ntest_data = pd.merge(test_data, test_parquet, how=\"left\", on='id')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_data.drop('id',axis=1)\ntest_data = test_data.drop('id',axis=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shape of the data:\nprint(\"train_data :\", train_data.shape)\nprint(\"test_data :\", test_data.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Example columns in your train and test data\ntrain_columns = train_data.columns.tolist()  # Train columns including target\ntest_columns = test_data.columns.tolist()    # Test columns\n\n# Identify common feature columns between train and test (excluding target)\ncommon_columns = [col for col in train_columns if col in test_columns]\n\n# Include the target column explicitly in the final train set\ncommon_columns.append('sii')\n\n# Now, reduce the training data to only the common feature columns + target\ntrain_data = train_data[common_columns]\n\n# Print the resulting columns in the training data\nprint(\"Train data columns:\", len(train_data.columns))\nprint(\"Test data columns:\", len(test_data.columns))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shape of the data:\nprint(\"train_data :\", train_data.shape)\nprint(\"test_data :\", test_data.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.isnull().sum().sort_values(ascending=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calculate missing values\nmissing_values = train_data.isnull().mean() * 100\n\n# Plot\nmissing_values.plot(kind='bar', figsize=(25, 5), color='skyblue')\nplt.title('Percentage of Missing Values by Feature')\nplt.ylabel('Percentage')\nplt.xlabel('Features')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T21:21:19.151469Z","iopub.execute_input":"2024-09-28T21:21:19.151963Z","iopub.status.idle":"2024-09-28T21:21:20.969464Z","shell.execute_reply.started":"2024-09-28T21:21:19.151909Z","shell.execute_reply":"2024-09-28T21:21:20.968158Z"}}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nprint(train_data['sii'].value_counts())\nsns.countplot(x='sii', data=train_data)\nplt.xticks(rotation=60)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_data.dropna(subset='sii')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Distribution of the data:\ntrain_data.drop(['id'],axis=1).hist(figsize=(25,25),color = 'skyblue', edgecolor='black')\nplt.show()","metadata":{}},{"cell_type":"code","source":"test_data.head()\ntest_data.isnull().sum().sort_values(ascending=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calculate missing values\nmissing_values = test_data.isnull().mean() * 100\n\n# Plot\nmissing_values.plot(kind='bar', figsize=(25, 5), color='skyblue')\nplt.title('Percentage of Missing Values by Feature')\nplt.ylabel('Percentage')\nplt.xlabel('Features')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-28T21:21:44.147643Z","iopub.execute_input":"2024-09-28T21:21:44.148110Z","iopub.status.idle":"2024-09-28T21:21:45.743914Z","shell.execute_reply.started":"2024-09-28T21:21:44.148064Z","shell.execute_reply":"2024-09-28T21:21:45.742516Z"}}},{"cell_type":"code","source":"sample_data.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cols = list(train_data.select_dtypes(exclude=['object']).columns.difference(['sii']))\ncat_cols = list(train_data.select_dtypes(include=['object']).columns)\n\nnum_cols_test = list(test_data.select_dtypes(exclude=['object']).columns)\ncat_cols_test = list(test_data.select_dtypes(include=['object']).columns)\n\n#num_cols_test = [col for col in num_cols_test if col not in ['id']]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in cat_cols:\n    train_data[col] = train_data[col].fillna('missing')\n    train_data[col] = train_data[col].astype('category')\n    \n    test_data[col] = test_data[col].fillna('missing')\n    test_data[col] = test_data[col].astype('category')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in num_cols:\n    train_data[col] = train_data[col].fillna(train_data[col].median())\n    test_data[col] = test_data[col].fillna(test_data[col].median())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(cat_cols_test),len(cat_cols)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  object datatype columns encoding:\nfrom sklearn.preprocessing import LabelEncoder\nlabelencoder = LabelEncoder()\nfor col_name in cat_cols:\n    train_data[col_name]=labelencoder.fit_transform(train_data[col_name]).astype(int)\n        \nfor col_name in cat_cols_test:\n    test_data[col_name]=labelencoder.transform(test_data[col_name]).astype(int)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\ntrain_data[num_cols] = scaler.fit_transform(train_data[num_cols])\ntest_data[num_cols] = scaler.transform(test_data[num_cols])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX = train_data.drop(['sii'], axis=1)\ny = train_data['sii']\ntest = test_data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.optimize import minimize\n\nn_splits = 5\ndef Train_model(model_class, test_data):\n    \n    X = train_data.drop(['sii'], axis=1)\n    y = train_data['sii']\n    test = test_data\n\n    SKF = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n    \n    models = []\n    train_pred = []\n    test_pred = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_pred.append(train_kappa)\n        test_pred.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n        models.append(model)\n\n    print(f\"Mean Train QWK : {np.mean(train_pred):.4f}\")\n    print(f\"Mean Validation QWK : {np.mean(test_pred):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions, x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), method='Nelder-Mead') \n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    pred_mean = test_preds.mean(axis=1)\n    pred = threshold_Rounder(pred_mean, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample_data['id'],\n        'sii': pred\n    })\n\n    return submission","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install colorama","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from colorama import Fore, Style, init","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgb_params = {'lambda_l1': 1.8498777274031641, 'lambda_l2': 2.747994241767871, 'num_leaves': 170, 'feature_fraction': 0.6557732374612563, 'bagging_fraction': 0.5511820667548543, 'bagging_freq': 3, 'min_child_samples': 94}\nparams_lgb = {'learning_rate': 0.21512512485545232, 'n_estimators': 937, 'max_depth': 4, 'num_leaves': 67, 'min_data_in_leaf': 77, 'feature_fraction': 0.991968769291252, 'bagging_fraction': 0.8725111301447539, 'bagging_freq': 6, 'lambda_l1': 9.617136134814306, 'lambda_l2': 4.004561047976997}\nparams_xgb = {'learning_rate': 0.16916939570807282, 'n_estimators': 156, 'max_depth': 3, 'min_child_weight': 1, 'gamma': 0.8103297718197263, 'subsample': 0.7876102368548592, 'colsample_bytree': 0.983724689836341, 'lambda': 0.7265688251146536, 'alpha': 3.5063675187665546}\nparams_cat = {'iterations': 422, 'depth': 5, 'learning_rate': 0.21336431777253273, 'l2_leaf_reg': 5.048661970814173, 'random_strength': 7.363242343297449, 'bagging_temperature': 0.43809545008703044, 'border_count': 89}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\nfrom catboost import CatBoostClassifier\nfrom sklearn.ensemble import VotingClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score\n\n# Custom function to calculate Quadratic Weighted Kappa\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n# Split the dataset\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Define individual classifiers\nclf1 = XGBClassifier(**params_xgb, use_label_encoder=False)\nclf2 = LGBMClassifier(**params_lgb, verbosity=-1)\nclf3 = CatBoostClassifier(**params_cat,verbose=0)\n\n# Create the voting classifier (soft voting)\nvoting_clf = VotingClassifier(\n    estimators=[\n        ('xgb', clf1),\n        ('lgb', clf2),\n        ('cat', clf3)\n    ],\n    voting='soft'  # Use 'hard' for majority voting or 'soft' for probability-based voting\n)\n\n# Train the voting classifier\nvoting_clf.fit(X_train, y_train)\n\n# Make predictions\ny_pred = voting_clf.predict(X_test)\n\n# Calculate Quadratic Weighted Kappa\nkappa = quadratic_weighted_kappa(y_test, y_pred)\nprint(f'Voting Classifier Quadratic Weighted Kappa: {kappa:.4f}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the ensemble model\nSubmission = Train_model(voting_clf, test)\n\n# Save submission\nSubmission.to_csv('submission.csv', index=False)\nprint(Submission['sii'].value_counts())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"y = np.where(y == 4, 3, y)\n\n# Custom metric for Quadratic Weighted Kappa\ndef quadratic_weighted_kappa(y_true, y_pred):\n    y_pred_class = np.argmax(y_pred, axis=1)\n    return cohen_kappa_score(y_true, y_pred_class, weights='quadratic')\n\n# Objective function for Optuna\ndef objective(trial):\n    # Define hyperparameters\n    param = {\n        'objective': 'multiclass',\n        'num_class': 4,  # Adjust based on number of classes\n        'metric': 'None',  # Custom metric, so no built-in metric used\n        'boosting_type': 'gbdt',\n        'learning_rate': trial.suggest_float('learning_rate', 1e-3, 0.3),\n        'n_estimators': trial.suggest_int('n_estimators', 50, 1000),\n        'max_depth': trial.suggest_int('max_depth', 3, 10),\n        'num_leaves': trial.suggest_int('num_leaves', 20, 200),\n        'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 20, 200),\n        'feature_fraction': trial.suggest_float('feature_fraction', 0.6, 1.0),\n        'bagging_fraction': trial.suggest_float('bagging_fraction', 0.6, 1.0),\n        'bagging_freq': trial.suggest_int('bagging_freq', 1, 7),\n        'lambda_l1': trial.suggest_float('lambda_l1', 1e-8, 10.0),\n        'lambda_l2': trial.suggest_float('lambda_l2', 1e-8, 10.0),\n        'verbosity' : -1\n    }\n\n    # Split the dataset\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n    # Stratified k-fold cross-validation\n    skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n    kappa_scores = []\n\n    for train_idx, valid_idx in skf.split(X_train, y_train):\n        X_train_cv, X_valid_cv = X_train[train_idx], X_train[valid_idx]\n        y_train_cv, y_valid_cv = y_train[train_idx], y_train[valid_idx]\n\n        # Convert data into LightGBM Dataset\n        dtrain = lgb.Dataset(X_train_cv, label=y_train_cv)\n        dvalid = lgb.Dataset(X_valid_cv, label=y_valid_cv, reference=dtrain)\n\n        model = lgb.train(param,dtrain,valid_sets=[dvalid])\n\n        # Get predictions and calculate quadratic weighted kappa\n        y_pred = model.predict(X_valid_cv, num_iteration=model.best_iteration)\n        kappa = quadratic_weighted_kappa(y_valid_cv, y_pred)\n        kappa_scores.append(kappa)\n\n    # Return the average kappa score from cross-validation\n    return np.mean(kappa_scores)\n\n# Optuna optimization\nstudy = optuna.create_study(direction='maximize')  # Maximize Kappa score\nstudy.optimize(objective, n_trials=50)\n\nprint(f'Best trial: {study.best_trial.params}')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T18:13:43.112745Z","iopub.execute_input":"2024-10-11T18:13:43.113479Z","iopub.status.idle":"2024-10-11T18:33:14.073304Z","shell.execute_reply.started":"2024-10-11T18:13:43.113420Z","shell.execute_reply":"2024-10-11T18:33:14.071008Z"}}},{"cell_type":"markdown","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    y_pred_class = np.argmax(y_pred, axis=1)  # Convert probabilities to predicted classes\n    return cohen_kappa_score(y_true, y_pred_class, weights='quadratic')\ndef objective(trial):\n    # Define the hyperparameters to tune\n    param = {\n        'objective': 'multi:softprob',  # Multiclass classification\n        'num_class': 4,  # Adjust based on number of classes in your problem\n        'eval_metric': 'mlogloss',  # Default metric for multiclass classification\n        'learning_rate': trial.suggest_float('learning_rate', 1e-3, 0.3),\n        'n_estimators': trial.suggest_int('n_estimators', 50, 1000),\n        'max_depth': trial.suggest_int('max_depth', 3, 10),\n        'min_child_weight': trial.suggest_int('min_child_weight', 1, 10),\n        'gamma': trial.suggest_float('gamma', 1e-8, 1.0),\n        'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n        'lambda': trial.suggest_float('lambda', 1e-8, 10.0),\n        'alpha': trial.suggest_float('alpha', 1e-8, 10.0),\n    }\n\n    # Split the data into training and test sets\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n    # Stratified K-Fold Cross Validation (to maintain class distribution)\n    skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n    kappa_scores = []\n\n    for train_idx, valid_idx in skf.split(X_train, y_train):\n        X_train_cv, X_valid_cv = X_train[train_idx], X_train[valid_idx]\n        y_train_cv, y_valid_cv = y_train[train_idx], y_train[valid_idx]\n\n        # Define the XGBoost model\n        model = xgb.XGBClassifier(\n            **param,\n            use_label_encoder=False\n        )\n\n        # Train the model\n        model.fit(X_train_cv, y_train_cv, eval_set=[(X_valid_cv, y_valid_cv)])\n\n        # Get predictions and calculate the quadratic weighted kappa score\n        y_pred = model.predict_proba(X_valid_cv)  # Predict probabilities for the validation set\n        kappa = quadratic_weighted_kappa(y_valid_cv, y_pred)\n        kappa_scores.append(kappa)\n\n    # Return the average Kappa score from cross-validation\n    return np.mean(kappa_scores)\n# Create a study to maximize the Kappa score\nstudy = optuna.create_study(direction='maximize')\nstudy.optimize(objective, n_trials=50)\n\n# Print the best hyperparameters\nprint(f'Best trial: {study.best_trial.params}')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T18:34:30.939781Z","iopub.execute_input":"2024-10-11T18:34:30.940237Z","iopub.status.idle":"2024-10-11T18:54:31.188653Z","shell.execute_reply.started":"2024-10-11T18:34:30.940189Z","shell.execute_reply":"2024-10-11T18:54:31.186244Z"}}},{"cell_type":"code","source":"params_xgb = {'learning_rate': 0.16916939570807282, 'n_estimators': 156, 'max_depth': 3, 'min_child_weight': 1, 'gamma': 0.8103297718197263, 'subsample': 0.7876102368548592, 'colsample_bytree': 0.983724689836341, 'lambda': 0.7265688251146536, 'alpha': 3.5063675187665546}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    y_pred_class = np.argmax(y_pred, axis=1)  # Convert probabilities to class predictions\n    return cohen_kappa_score(y_true, y_pred_class, weights='quadratic')\ndef objective(trial):\n    # Define the hyperparameters for tuning\n    param = {\n        'iterations': trial.suggest_int('iterations', 50, 1000),\n        'depth': trial.suggest_int('depth', 3, 10),\n        'learning_rate': trial.suggest_float('learning_rate', 1e-3, 0.3),\n        'l2_leaf_reg': trial.suggest_float('l2_leaf_reg', 1e-8, 10.0),\n        'random_strength': trial.suggest_float('random_strength', 1e-8, 10.0),\n        'bagging_temperature': trial.suggest_float('bagging_temperature', 0.0, 1.0),\n        'border_count': trial.suggest_int('border_count', 1, 255),\n        'verbose': 0\n    }\n\n    # Split the dataset\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n    # Stratified K-Fold Cross Validation\n    skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n    kappa_scores = []\n\n    for train_idx, valid_idx in skf.split(X_train, y_train):\n        X_train_cv, X_valid_cv = X_train[train_idx], X_train[valid_idx]\n        y_train_cv, y_valid_cv = y_train[train_idx], y_train[valid_idx]\n\n        # Define the CatBoostClassifier model\n        model = CatBoostClassifier(\n            **param,\n            loss_function='MultiClass',  # For multiclass classification\n            eval_metric='MultiClass'\n        )\n\n        # Train the model\n        model.fit(X_train_cv, y_train_cv, eval_set=(X_valid_cv, y_valid_cv), early_stopping_rounds=100)\n\n        # Get predictions and calculate the quadratic weighted kappa score\n        y_pred = model.predict_proba(X_valid_cv)\n        kappa = quadratic_weighted_kappa(y_valid_cv, y_pred)\n        kappa_scores.append(kappa)\n\n    # Return the average kappa score from cross-validation\n    return np.mean(kappa_scores)\n# Create a study to maximize the Kappa score\nstudy = optuna.create_study(direction='maximize')\nstudy.optimize(objective, n_trials=50)\n\n# Print the best hyperparameters\nprint(f'Best trial: {study.best_trial.params}')","metadata":{}}]}