{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport glob\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport lightgbm as lgbm\nimport xgboost as xgb\nfrom sklearn.model_selection import KFold, StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\n\nimport matplotlib.pyplot as plt \nimport seaborn as sns\n\n\nimport warnings\n\nwarnings.filterwarnings('ignore')\npd.set_option('display.max_columns', None)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-25T22:00:59.756893Z","iopub.execute_input":"2024-09-25T22:00:59.757372Z","iopub.status.idle":"2024-09-25T22:00:59.764471Z","shell.execute_reply.started":"2024-09-25T22:00:59.757328Z","shell.execute_reply":"2024-09-25T22:00:59.763307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# settings","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\n","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:00:59.781362Z","iopub.execute_input":"2024-09-25T22:00:59.781801Z","iopub.status.idle":"2024-09-25T22:00:59.842483Z","shell.execute_reply.started":"2024-09-25T22:00:59.781757Z","shell.execute_reply":"2024-09-25T22:00:59.841385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns = train.columns.str.lower().str.replace(r'\\W+', '_', regex=True)\ntest.columns = test.columns.str.lower().str.replace(r'\\W+', '_', regex=True)","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:00:59.844603Z","iopub.execute_input":"2024-09-25T22:00:59.845106Z","iopub.status.idle":"2024-09-25T22:00:59.853901Z","shell.execute_reply.started":"2024-09-25T22:00:59.845052Z","shell.execute_reply":"2024-09-25T22:00:59.852659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train = train[~train['sii'].isnull()]","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:00:59.887089Z","iopub.execute_input":"2024-09-25T22:00:59.887549Z","iopub.status.idle":"2024-09-25T22:00:59.896073Z","shell.execute_reply.started":"2024-09-25T22:00:59.887506Z","shell.execute_reply":"2024-09-25T22:00:59.894820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_columns = set(train.columns) - set(test.columns)\ntarget_columns","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:00:59.912980Z","iopub.execute_input":"2024-09-25T22:00:59.913423Z","iopub.status.idle":"2024-09-25T22:00:59.922148Z","shell.execute_reply.started":"2024-09-25T22:00:59.913378Z","shell.execute_reply":"2024-09-25T22:00:59.920936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.loc[:, list(target_columns)] = train.loc[:, list(target_columns)].fillna(0)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:00:59.938388Z","iopub.execute_input":"2024-09-25T22:00:59.938825Z","iopub.status.idle":"2024-09-25T22:00:59.953069Z","shell.execute_reply.started":"2024-09-25T22:00:59.938781Z","shell.execute_reply":"2024-09-25T22:00:59.951925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['total'] = train[list(target_columns-{'pciat_pciat_total',\n 'pciat_season',\n 'sii'})].sum(axis=1)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:00:59.961366Z","iopub.execute_input":"2024-09-25T22:00:59.961824Z","iopub.status.idle":"2024-09-25T22:00:59.971232Z","shell.execute_reply.started":"2024-09-25T22:00:59.961778Z","shell.execute_reply":"2024-09-25T22:00:59.969968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.boxplot(x=train['sii'], y=train['total'])\n","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:00:59.987012Z","iopub.execute_input":"2024-09-25T22:00:59.987983Z","iopub.status.idle":"2024-09-25T22:01:00.278833Z","shell.execute_reply.started":"2024-09-25T22:00:59.987935Z","shell.execute_reply":"2024-09-25T22:01:00.277229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"0->[0;34]\n\n1->[35;47]\n\n2->[48;75]\n\n3->[76;100]\n","metadata":{}},{"cell_type":"code","source":"bins = [0, 30, 50, 80, 100]\nlabels = [0, 1, 2, 3]\n\ntrain['binned_total'] = pd.cut(train['total'], bins=bins, labels=labels, include_lowest=True).astype(int)\ny_binned = train['binned_total'] ","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:00.281282Z","iopub.execute_input":"2024-09-25T22:01:00.281791Z","iopub.status.idle":"2024-09-25T22:01:00.294797Z","shell.execute_reply.started":"2024-09-25T22:01:00.281727Z","shell.execute_reply":"2024-09-25T22:01:00.293367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(train['total'])","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:00.296373Z","iopub.execute_input":"2024-09-25T22:01:00.297445Z","iopub.status.idle":"2024-09-25T22:01:00.621365Z","shell.execute_reply.started":"2024-09-25T22:01:00.297385Z","shell.execute_reply":"2024-09-25T22:01:00.620138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(np.log1p(train['total']))","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:00.624132Z","iopub.execute_input":"2024-09-25T22:01:00.624548Z","iopub.status.idle":"2024-09-25T22:01:00.980571Z","shell.execute_reply.started":"2024-09-25T22:01:00.624503Z","shell.execute_reply":"2024-09-25T22:01:00.979485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"TODO: try logtransform target ","metadata":{}},{"cell_type":"code","source":"sns.histplot(train['binned_total'])","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:00.982212Z","iopub.execute_input":"2024-09-25T22:01:00.982718Z","iopub.status.idle":"2024-09-25T22:01:01.347812Z","shell.execute_reply.started":"2024-09-25T22:01:00.982660Z","shell.execute_reply":"2024-09-25T22:01:01.346679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(train['sii'].fillna(0))","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:01.349364Z","iopub.execute_input":"2024-09-25T22:01:01.349830Z","iopub.status.idle":"2024-09-25T22:01:01.712944Z","shell.execute_reply.started":"2024-09-25T22:01:01.349786Z","shell.execute_reply":"2024-09-25T22:01:01.711670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_docs(series_id, source='train'):\n    base_dir = f'/kaggle/input/child-mind-institute-problematic-internet-use/series_{source}.parquet'\n    folder_path = os.path.join(base_dir, f\"id={series_id}\")\n    num_docs = len(glob.glob(os.path.join(folder_path, \"part-*.parquet\")))\n    return num_docs\n\n\ntrain['num_docs'] = train['id'].apply(count_docs)\ntest['num_docs'] = test['id'].apply(count_docs, 'test')","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:01.714567Z","iopub.execute_input":"2024-09-25T22:01:01.715016Z","iopub.status.idle":"2024-09-25T22:01:02.483845Z","shell.execute_reply.started":"2024-09-25T22:01:01.714973Z","shell.execute_reply":"2024-09-25T22:01:02.482726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fill_missing_values(train, test, groupby_cols):\n    \"\"\"\n    Fills missing values in both train and test datasets.\n    \n    - Numeric columns: Filled with group-based means (based on groupby_cols).\n    - Categorical columns: Filled with the mode for each column.\n    \n    Parameters:\n    - train: Training DataFrame.\n    - test: Test DataFrame.\n    - groupby_cols: List of columns to group by for filling numeric missing values.\n    \n    Returns:\n    - train: Updated training DataFrame with missing values filled.\n    - test: Updated test DataFrame with missing values filled.\n    \"\"\"\n    # Fill missing age with median\n    median_age = train['basic_demos_age'].median()\n    train['basic_demos_age'].fillna(median_age, inplace=True)\n    test['basic_demos_age'].fillna(median_age, inplace=True)\n\n    # Fill missing sex with mode\n    mode_sex = train['basic_demos_sex'].mode()[0]\n    train['basic_demos_sex'].fillna(mode_sex, inplace=True)\n    test['basic_demos_sex'].fillna(mode_sex, inplace=True)\n    # Fill missing values in numeric columns with group-based means\n    for df in [train, test]:\n        # Separate numeric columns\n        numeric_cols = df.select_dtypes(include='number')\n        \n        # Calculate group-based means\n        group_means = numeric_cols.groupby(groupby_cols).transform('mean')\n        \n        # Fill missing numeric values\n        df.update(numeric_cols.fillna(group_means))\n    \n    # Fill missing values in categorical columns with the mode\n    # Calculate mode for the train dataset\n    mode_agg = train.select_dtypes(include='object').agg(lambda x: x.mode()[0])\n    \n    for df in [train, test]:\n        # Separate categorical columns\n        cat_cols = df.select_dtypes(include='object')\n        \n        # Fill missing categorical values with mode\n        df.update(cat_cols.fillna(mode_agg))\n    return train, test","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:02.485431Z","iopub.execute_input":"2024-09-25T22:01:02.485914Z","iopub.status.idle":"2024-09-25T22:01:02.498410Z","shell.execute_reply.started":"2024-09-25T22:01:02.485861Z","shell.execute_reply":"2024-09-25T22:01:02.497238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set a threshold for the number of unique values to consider for categorical columns\nnunique_threshold = 12  # Adjust the threshold as per your data (e.g., 20 unique values)\n\n# Select columns with 'object' type or columns with less than `nunique_threshold` unique values\ncategorical_columns = [col for col in train.columns if train[col].nunique() < nunique_threshold]","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:02.499883Z","iopub.execute_input":"2024-09-25T22:01:02.500305Z","iopub.status.idle":"2024-09-25T22:01:02.532497Z","shell.execute_reply.started":"2024-09-25T22:01:02.500262Z","shell.execute_reply":"2024-09-25T22:01:02.531310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_columns = set(categorical_columns) - target_columns - {'binned_total'}","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:02.536854Z","iopub.execute_input":"2024-09-25T22:01:02.537286Z","iopub.status.idle":"2024-09-25T22:01:02.542864Z","shell.execute_reply.started":"2024-09-25T22:01:02.537242Z","shell.execute_reply":"2024-09-25T22:01:02.541728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[list(categorical_columns)] = train[list(categorical_columns)].fillna('missing')","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:02.544572Z","iopub.execute_input":"2024-09-25T22:01:02.545069Z","iopub.status.idle":"2024-09-25T22:01:02.571115Z","shell.execute_reply.started":"2024-09-25T22:01:02.545008Z","shell.execute_reply":"2024-09-25T22:01:02.569950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numeric_columns = set(test.columns) - categorical_columns - {'id'}","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:02.572597Z","iopub.execute_input":"2024-09-25T22:01:02.572971Z","iopub.status.idle":"2024-09-25T22:01:02.577984Z","shell.execute_reply.started":"2024-09-25T22:01:02.572929Z","shell.execute_reply":"2024-09-25T22:01:02.576897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_columns","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:02.579551Z","iopub.execute_input":"2024-09-25T22:01:02.579936Z","iopub.status.idle":"2024-09-25T22:01:02.592516Z","shell.execute_reply.started":"2024-09-25T22:01:02.579895Z","shell.execute_reply":"2024-09-25T22:01:02.591421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train[list(numeric_columns)+list(categorical_columns)+['total']]\ntest = test[list(numeric_columns)+list(categorical_columns)]","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:02.596723Z","iopub.execute_input":"2024-09-25T22:01:02.597271Z","iopub.status.idle":"2024-09-25T22:01:02.609450Z","shell.execute_reply.started":"2024-09-25T22:01:02.597223Z","shell.execute_reply":"2024-09-25T22:01:02.608265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train[list(categorical_columns)] = train[list(categorical_columns)].astype(str)\n# test[list(categorical_columns)] = test[list(categorical_columns)].astype(str)\n# train[list(numeric_columns)] = train[list(numeric_columns)].astype(float)\n# test[list(numeric_columns)] = test[list(numeric_columns)].astype(float)","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:02.611012Z","iopub.execute_input":"2024-09-25T22:01:02.611436Z","iopub.status.idle":"2024-09-25T22:01:02.617239Z","shell.execute_reply.started":"2024-09-25T22:01:02.611394Z","shell.execute_reply":"2024-09-25T22:01:02.615827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lbs_to_kg = 0.453592\ninches_to_cm = 2.54\n\ntrain['physical_weight'] = train['physical_weight'] * lbs_to_kg\ntrain['physical_height'] = train['physical_height'] * inches_to_cm\n\n# Recalculate BMI: BMI = weight (kg) / (height (m)^2)\ntrain['physical_bmi'] = np.where(\n    train['physical_weight'].notna() & train['physical_height'].notna(),\n    train['physical_height'] / ((train['physical_height'] / 100) ** 2),\n    np.nan  # If either is NaN, set BMI to NaN\n)\n\ntest['physical_weight'] = test['physical_weight'] * lbs_to_kg\ntest['physical_height'] = test['physical_height'] * inches_to_cm\n\n# Recalculate BMI: BMI = weight (kg) / (height (m)^2)\ntest['physical_bmi'] = np.where(\n    test['physical_weight'].notna() & test['physical_height'].notna(),\n    test['physical_height'] / ((test['physical_height'] / 100) ** 2),\n    np.nan  # If either is NaN, set BMI to NaN\n)","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:02.618921Z","iopub.execute_input":"2024-09-25T22:01:02.619349Z","iopub.status.idle":"2024-09-25T22:01:02.636578Z","shell.execute_reply.started":"2024-09-25T22:01:02.619305Z","shell.execute_reply":"2024-09-25T22:01:02.635293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns\n","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:02.637872Z","iopub.execute_input":"2024-09-25T22:01:02.638301Z","iopub.status.idle":"2024-09-25T22:01:02.651859Z","shell.execute_reply.started":"2024-09-25T22:01:02.638259Z","shell.execute_reply":"2024-09-25T22:01:02.650633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_filled, test_filled = fill_missing_values(train, test, groupby_cols=['basic_demos_sex','basic_demos_age', ])\ntrain_filled, test_filled = fill_missing_values(train_filled, test_filled, groupby_cols=['basic_demos_sex'])\ntrain_filled, test_filled = fill_missing_values(train_filled, test_filled, groupby_cols=['basic_demos_age'])","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:02.653320Z","iopub.execute_input":"2024-09-25T22:01:02.653678Z","iopub.status.idle":"2024-09-25T22:01:03.125059Z","shell.execute_reply.started":"2024-09-25T22:01:02.653636Z","shell.execute_reply":"2024-09-25T22:01:03.123733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_filled[list(categorical_columns)] = train_filled[list(categorical_columns)].astype(str)\ntest_filled[list(categorical_columns)] = test_filled[list(categorical_columns)].astype(str)","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:03.126589Z","iopub.execute_input":"2024-09-25T22:01:03.126967Z","iopub.status.idle":"2024-09-25T22:01:03.161101Z","shell.execute_reply.started":"2024-09-25T22:01:03.126927Z","shell.execute_reply":"2024-09-25T22:01:03.159906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\ndef label_encode(train, test, categorical_cols):\n    \"\"\"\n    Converts the specified categorical columns in both train and test datasets into label-encoded columns.\n    \n    Parameters:\n    - train: Training DataFrame.\n    - test: Test DataFrame.\n    - categorical_cols: List of categorical column names to be label-encoded.\n    \n    Returns:\n    - train: Updated train DataFrame with label-encoded columns.\n    - test: Updated test DataFrame with label-encoded columns.\n    \"\"\"\n    # Loop through each categorical column\n    for col in categorical_cols:\n        # Initialize LabelEncoder\n        le = LabelEncoder()\n        \n        # Fit the label encoder on the combined data from train and test to ensure consistent encoding\n        combined = pd.concat([train[col], test[col]], axis=0)\n        le.fit(combined)\n        \n        # Transform the train and test sets\n        train[col] = le.transform(train[col])\n        test[col] = le.transform(test[col])\n    \n    return train, test\n\n# Example usage:\ntrain_encoded, test_encoded = label_encode(train_filled, test_filled, list(categorical_columns))","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:03.164886Z","iopub.execute_input":"2024-09-25T22:01:03.165306Z","iopub.status.idle":"2024-09-25T22:01:03.218543Z","shell.execute_reply.started":"2024-09-25T22:01:03.165263Z","shell.execute_reply":"2024-09-25T22:01:03.217579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X, y = train_encoded[list(numeric_columns)+list(categorical_columns)], np.log1p(train_encoded['total'])\nX_test = test_encoded[list(numeric_columns)+list(categorical_columns)]\n","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:03.220267Z","iopub.execute_input":"2024-09-25T22:01:03.220775Z","iopub.status.idle":"2024-09-25T22:01:03.236223Z","shell.execute_reply.started":"2024-09-25T22:01:03.220731Z","shell.execute_reply":"2024-09-25T22:01:03.235050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model optimization","metadata":{}},{"cell_type":"code","source":"\n# Function to bin continuous values into discrete labels\ndef bin_to_labels(values, bins, labels):\n    values = np.clip(values, 1e-8, 100)\n    return np.digitize(values, bins) - 1  # Adjust to 0-indexing\n\ndef custom_cross_val(estimator, X, y, n_splits=5, task=\"classification\", bins=bins, labels=labels):\n    kf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n    oof_preds = []\n    oof_true = []\n    fold_scores = []\n    \n    for train_idx, val_idx in kf.split(X, y_binned):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n        \n        # Use LightGBM's early stopping callback if LightGBM model\n        if isinstance(estimator, lgbm.LGBMModel):\n            # Early stopping as a callback for LightGBM\n            estimator.fit(X_train, y_train, \n                          eval_set=[(X_val, y_val)],\n                          callbacks=[lgbm.early_stopping(stopping_rounds=100)])\n        else:\n            # Early stopping as a parameter for other models\n            estimator.fit(X_train, y_train, \n                          eval_set=[(X_val, y_val)],\n                          early_stopping_rounds=100, \n                          verbose=False)\n\n        # Make predictions\n        preds = estimator.predict(X_val)\n        oof_preds.extend(preds)\n        oof_true.extend(y_val)\n        \n        # Compute score\n        if task == \"classification\":\n            score = cohen_kappa_score(y_val, preds, weights=\"quadratic\")\n        elif task == \"regression\":\n            # Bin continuous values to labels for regression\n            y_val = np.expm1(y_val)\n            preds = np.expm1(preds)\n            y_val_binned = bin_to_labels(y_val, bins, labels)\n            preds_binned = bin_to_labels(preds, bins, labels)\n#             preds = np.round(np.clip(preds, 0, 3))\n            # Calculate QWK on binned labels\n            score = cohen_kappa_score(y_val_binned, preds_binned, weights=\"quadratic\")\n        \n        fold_scores.append(score)\n    \n    return oof_preds, oof_true, sum(fold_scores) / len(fold_scores)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:03.238470Z","iopub.execute_input":"2024-09-25T22:01:03.238944Z","iopub.status.idle":"2024-09-25T22:01:03.254546Z","shell.execute_reply.started":"2024-09-25T22:01:03.238890Z","shell.execute_reply":"2024-09-25T22:01:03.253245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import optuna\n\ndef lgbm_regressor_objective(trial):\n        params = {\n#         \"device_type\": trial.suggest_categorical(\"device_type\", ['gpu']),\n        \"n_estimators\": trial.suggest_categorical(\"n_estimators\", [3000]),\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 1e-3, 0.5),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 20, 3000, step=20),\n        \"max_depth\": trial.suggest_int(\"max_depth\", 3, 12),\n        \"min_data_in_leaf\": trial.suggest_int(\"min_data_in_leaf\", 20, 1000, step=20),\n        \"lambda_l1\": trial.suggest_float(\"lambda_l1\", 1e-8, 10),\n        \"lambda_l2\": trial.suggest_float(\"lambda_l2\", 1e-8, 10),\n        \"min_gain_to_split\": trial.suggest_float(\"min_gain_to_split\", 0, 15),\n        \"bagging_fraction\": trial.suggest_float(\"bagging_fraction\", 0.2, 0.95, step=0.1),\n        \"bagging_freq\": trial.suggest_categorical(\"bagging_freq\", [1]),\n        \"feature_fraction\": trial.suggest_float(\"feature_fraction\", 0.2, 0.95, step=0.1),\n        'force_col_wise':True,\n        \"metric\":'rmse',\n        'verbose': -1,\n        \"n_jobs\":-1, \"random_state\":36\n    }\n        estimator = lgbm.LGBMRegressor(**params)\n        _, _, qwk = custom_cross_val(estimator, X, y, 5, task='regression')\n        return qwk \n    ","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:03.255990Z","iopub.execute_input":"2024-09-25T22:01:03.256401Z","iopub.status.idle":"2024-09-25T22:01:03.271562Z","shell.execute_reply.started":"2024-09-25T22:01:03.256354Z","shell.execute_reply":"2024-09-25T22:01:03.270368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if False:\n    # Run Optuna optimization\n    study = optuna.create_study(direction='maximize')\n    study.optimize(lgbm_regressor_objective, n_trials=100, show_progress_bar=True)\n    # Get the best parameters\n    lgbm_best_params = study.best_params\n    print(f\"Best parameters: {lgbm_best_params}\")","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:03.273135Z","iopub.execute_input":"2024-09-25T22:01:03.273621Z","iopub.status.idle":"2024-09-25T22:01:03.281638Z","shell.execute_reply.started":"2024-09-25T22:01:03.273567Z","shell.execute_reply":"2024-09-25T22:01:03.280536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def xgb_regressor_objective(trial):\n    params = {\n        \"n_estimators\": trial.suggest_categorical(\"n_estimators\", [3000]),\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 1e-3, 0.5),\n        \"max_depth\": trial.suggest_int(\"max_depth\", 3, 12),\n        \"min_child_weight\": trial.suggest_int(\"min_child_weight\", 1, 10),\n        \"gamma\": trial.suggest_float(\"gamma\", 0, 5),\n        \"subsample\": trial.suggest_float(\"subsample\", 0.2, 1.0),\n        \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.2, 1.0),\n        \"lambda\": trial.suggest_float(\"lambda\", 1e-8, 10),\n        \"alpha\": trial.suggest_float(\"alpha\", 1e-8, 10),\n#         \"tree_method\": 'gpu_hist',  # use 'hist' if not using GPU\n        \"eval_metric\": 'rmse',\n        \"random_state\": 36,\n        \"n_jobs\": -1\n    }\n    \n    estimator = xgb.XGBRegressor(**params)\n    _, _, qwk = custom_cross_val(estimator, X, y, 5, task='regression')\n    return qwk","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:03.283057Z","iopub.execute_input":"2024-09-25T22:01:03.283452Z","iopub.status.idle":"2024-09-25T22:01:03.295929Z","shell.execute_reply.started":"2024-09-25T22:01:03.283413Z","shell.execute_reply":"2024-09-25T22:01:03.294583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if False:\n    # Run Optuna optimization\n    study = optuna.create_study(direction='maximize')\n    study.optimize(xgb_regressor_objective, n_trials=100, show_progress_bar=True)\n    # Get the best parameters\n    lgbm_best_params = study.best_params\n    print(f\"Best parameters: {lgbm_best_params}\")","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:03.297473Z","iopub.execute_input":"2024-09-25T22:01:03.297929Z","iopub.status.idle":"2024-09-25T22:01:03.313352Z","shell.execute_reply.started":"2024-09-25T22:01:03.297886Z","shell.execute_reply":"2024-09-25T22:01:03.312037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lgbm_classifier_objective(trial):\n    # Define the hyperparameters to be optimized\n    params = {\n        \"n_estimators\": trial.suggest_categorical(\"n_estimators\", [3000]),\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 1e-3, 0.5),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 20, 3000, step=20),\n        \"max_depth\": trial.suggest_int(\"max_depth\", 3, 12),\n        \"min_data_in_leaf\": trial.suggest_int(\"min_data_in_leaf\", 20, 1000, step=20),\n        \"lambda_l1\": trial.suggest_float(\"lambda_l1\", 1e-8, 10),\n        \"lambda_l2\": trial.suggest_float(\"lambda_l2\", 1e-8, 10),\n        \"min_gain_to_split\": trial.suggest_float(\"min_gain_to_split\", 0, 15),\n        \"bagging_fraction\": trial.suggest_float(\"bagging_fraction\", 0.2, 0.95, step=0.1),\n        \"bagging_freq\": trial.suggest_categorical(\"bagging_freq\", [1]),\n        \"feature_fraction\": trial.suggest_float(\"feature_fraction\", 0.2, 0.95, step=0.1),\n        'force_col_wise': True,\n        \"metric\": 'multi_logloss',  # or 'binary_logloss' for binary classification\n        'objective': 'multiclass',  # Use 'binary' for binary classification\n        'num_class': len(np.unique(y_binned)),  # Only needed for multiclass classification\n        'verbose': -1,\n        \"n_jobs\": -1, \"random_state\":36\n    }\n\n    # Instantiate the LGBMClassifier (change to LGBMClassifier)\n    estimator = lgbm.LGBMClassifier(**params)\n\n    # Perform custom cross-validation (defined earlier in your code)\n    _, _, score = custom_cross_val(estimator, X, y_binned, 3, task='classification')\n\n    # Return the evaluation metric (e.g., accuracy, Cohen's Kappa, etc.)\n    return score","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:03.315012Z","iopub.execute_input":"2024-09-25T22:01:03.315483Z","iopub.status.idle":"2024-09-25T22:01:03.328870Z","shell.execute_reply.started":"2024-09-25T22:01:03.315440Z","shell.execute_reply":"2024-09-25T22:01:03.327505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if False:\n    # Run Optuna optimization\n    study = optuna.create_study(direction='maximize')\n    study.optimize(lgbm_classifier_objective, n_trials=100, show_progress_bar=True)\n    # Get the best parameters\n    lgbm_classifier_best_params = study.best_params\n    print(f\"Best parameters: {lgbm_best_params}\")","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:03.336196Z","iopub.execute_input":"2024-09-25T22:01:03.336629Z","iopub.status.idle":"2024-09-25T22:01:03.344138Z","shell.execute_reply.started":"2024-09-25T22:01:03.336576Z","shell.execute_reply":"2024-09-25T22:01:03.343062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_r_params = {'n_estimators': 3000, 'learning_rate': 0.36847545742800797,\n                 'num_leaves': 2740, 'max_depth': 11, 'min_data_in_leaf': 340,\n                 'lambda_l1': 5.985211200209212, 'lambda_l2': 2.904878446193629,\n                 'min_gain_to_split': 5.266288335646141, 'bagging_fraction': 0.6000000000000001,\n                 'bagging_freq': 1, 'feature_fraction': 0.30000000000000004,\n                 'force_col_wise':True,\n                \"metric\":'rmse',\n                'verbose': -1,\n                \"n_jobs\":-1}\n\nxgb_r_params = {\n       'n_estimators': 3000, 'learning_rate': 0.4338349119614098,\n    'max_depth': 3, 'min_child_weight': 4, 'gamma': 2.159683082987327,\n    'subsample': 0.8969840780287761, 'colsample_bytree': 0.37113620202920083,\n    'lambda': 9.10147194027502, 'alpha': 6.868315300372076,\n        \"eval_metric\": 'rmse',\n        \"random_state\": 36,\n        \"n_jobs\": -1\n    }\n\nlgbm_c_params = {'n_estimators': 3000,\n                 'learning_rate': 0.023573100567686217,\n                 'num_leaves': 2560,\n                 'max_depth': 5,\n                 'min_data_in_leaf': 60,\n                 'lambda_l1': 7.819855826570875,\n                 'lambda_l2': 3.9158611299413826,\n                 'min_gain_to_split': 9.390121036346134,\n                 'bagging_fraction': 0.5,\n                 'bagging_freq': 1,\n                 'feature_fraction': 0.6000000000000001,\n                'force_col_wise': True,\n                \"metric\": 'multi_logloss',  # or 'binary_logloss' for binary classification\n                'objective': 'multiclass',  # Use 'binary' for binary classification\n                'num_class': len(np.unique(y_binned)),  # Only needed for multiclass classification\n                'verbose': -1,\n                \"n_jobs\": -1}","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:03.345485Z","iopub.execute_input":"2024-09-25T22:01:03.345854Z","iopub.status.idle":"2024-09-25T22:01:03.360377Z","shell.execute_reply.started":"2024-09-25T22:01:03.345812Z","shell.execute_reply":"2024-09-25T22:01:03.359062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgmb_regressor = lgbm.LGBMRegressor(**lgbm_r_params)\nxgb_regressor = xgb.XGBRegressor(**xgb_r_params)","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:03.364592Z","iopub.execute_input":"2024-09-25T22:01:03.365075Z","iopub.status.idle":"2024-09-25T22:01:03.374844Z","shell.execute_reply.started":"2024-09-25T22:01:03.365031Z","shell.execute_reply":"2024-09-25T22:01:03.373471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_final_predictions(estimator, X, y, X_test, n_splits=3, task=\"classification\", bins=None, labels=None):\n    kf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n    oof_preds = []\n    oof_true = []\n    fold_scores = []\n    \n    test_preds = np.zeros((X_test.shape[0],))  # Store the averaged test predictions\n    \n    for train_idx, val_idx in kf.split(X, y_binned):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n        \n        # Train the model\n        if isinstance(estimator, lgbm.LGBMModel):\n            estimator.fit(X_train, y_train, \n                          eval_set=[(X_val, y_val)],\n                          callbacks=[lgbm.early_stopping(stopping_rounds=100)])\n        else:\n            estimator.fit(X_train, y_train, \n                          eval_set=[(X_val, y_val)],\n                          early_stopping_rounds=100, \n                          verbose=False)\n\n        # Make out-of-fold predictions on validation data\n        preds = estimator.predict(X_val)\n        oof_preds.extend(preds)\n        oof_true.extend(y_val)\n        \n        # Aggregate test predictions (average predictions from each fold)\n        test_preds += estimator.predict(X_test) / n_splits\n        \n        # Compute the score for this fold\n        if task == \"classification\":\n            score = cohen_kappa_score(y_val, preds, weights=\"quadratic\")\n        elif task == \"regression\":\n            # Bin continuous values to labels for regression\n            y_val = np.expm1(y_val)\n            preds = np.expm1(preds)\n            y_val_binned = bin_to_labels(y_val, bins, labels)\n            preds_binned = bin_to_labels(preds, bins, labels)\n#             preds = np.round(np.clip(preds, 0, 3))\n            # Calculate QWK on binned labels\n            score = cohen_kappa_score(y_val_binned, preds_binned, weights=\"quadratic\")\n        \n        fold_scores.append(score)\n    \n    # Print the overall QWK score across folds\n    overall_qwk = sum(fold_scores) / len(fold_scores)\n    print(f\"Overall QWK across folds: {overall_qwk}\")\n\n    # Return the out-of-fold predictions, test predictions, and the overall QWK score\n    return oof_preds, test_preds, overall_qwk","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:03.376457Z","iopub.execute_input":"2024-09-25T22:01:03.376933Z","iopub.status.idle":"2024-09-25T22:01:03.393621Z","shell.execute_reply.started":"2024-09-25T22:01:03.376876Z","shell.execute_reply":"2024-09-25T22:01:03.392118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_preds_r, lgbm_test_preds_r, overall_qwk_r = make_final_predictions(lgmb_regressor, X, y, X_test, task='regression', bins=bins, labels=labels)","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:03.395097Z","iopub.execute_input":"2024-09-25T22:01:03.395514Z","iopub.status.idle":"2024-09-25T22:01:03.868446Z","shell.execute_reply.started":"2024-09-25T22:01:03.395453Z","shell.execute_reply":"2024-09-25T22:01:03.867238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.expm1(lgbm_test_preds_r), bin_to_labels(np.expm1(lgbm_test_preds_r), bins, labels)","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:03.870111Z","iopub.execute_input":"2024-09-25T22:01:03.870604Z","iopub.status.idle":"2024-09-25T22:01:03.879731Z","shell.execute_reply.started":"2024-09-25T22:01:03.870549Z","shell.execute_reply":"2024-09-25T22:01:03.878572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_preds_r, xgb_test_preds_r, overall_qwk_r = make_final_predictions(xgb_regressor, X, y, X_test, task='regression', bins=bins, labels=labels)","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:03.881207Z","iopub.execute_input":"2024-09-25T22:01:03.881677Z","iopub.status.idle":"2024-09-25T22:01:04.996431Z","shell.execute_reply.started":"2024-09-25T22:01:03.881624Z","shell.execute_reply":"2024-09-25T22:01:04.995122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.expm1(xgb_test_preds_r), bin_to_labels(np.expm1(xgb_test_preds_r), bins, labels)","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:04.997842Z","iopub.execute_input":"2024-09-25T22:01:04.998206Z","iopub.status.idle":"2024-09-25T22:01:05.010767Z","shell.execute_reply.started":"2024-09-25T22:01:04.998167Z","shell.execute_reply":"2024-09-25T22:01:05.009608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_hat = (lgbm_test_preds_r + xgb_test_preds_r)/2\ny_hat","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:05.012197Z","iopub.execute_input":"2024-09-25T22:01:05.012658Z","iopub.status.idle":"2024-09-25T22:01:05.023698Z","shell.execute_reply.started":"2024-09-25T22:01:05.012606Z","shell.execute_reply":"2024-09-25T22:01:05.022611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_hat = bin_to_labels(np.expm1(y_hat), bins, labels)\ny_hat","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:05.025172Z","iopub.execute_input":"2024-09-25T22:01:05.025648Z","iopub.status.idle":"2024-09-25T22:01:05.034390Z","shell.execute_reply.started":"2024-09-25T22:01:05.025602Z","shell.execute_reply":"2024-09-25T22:01:05.033198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:05.035862Z","iopub.execute_input":"2024-09-25T22:01:05.036295Z","iopub.status.idle":"2024-09-25T22:01:05.045679Z","shell.execute_reply.started":"2024-09-25T22:01:05.036252Z","shell.execute_reply":"2024-09-25T22:01:05.044496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df['sii'] = y_hat\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:05.047336Z","iopub.execute_input":"2024-09-25T22:01:05.047816Z","iopub.status.idle":"2024-09-25T22:01:05.061816Z","shell.execute_reply.started":"2024-09-25T22:01:05.047755Z","shell.execute_reply":"2024-09-25T22:01:05.060414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-09-25T22:01:05.063639Z","iopub.execute_input":"2024-09-25T22:01:05.064139Z","iopub.status.idle":"2024-09-25T22:01:05.072912Z","shell.execute_reply.started":"2024-09-25T22:01:05.064083Z","shell.execute_reply":"2024-09-25T22:01:05.071833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}