{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Log: \n- no imputation & drop sii na: QWK: 0.46115 (LB Score: 0.295)\n- imputing & use imputed sii: Acc: 0.6300505050505051, QWK: 0.5226428491787309 (LB Score: 0.328)\n- added threshold tuning: Acc: 0.8037, Optimized QWK: 0.759, QWK:0.5913 (LB Score: 0.420) \n- removed outliers, Optimized QWK SCORE :: 0.759, Accuracy: 0.8038 \n- recalculating sii: QWK: 0.5928, Optimized QWK: 0.773, Accuracy: 0.8048 (LB Score: 0.440) \n- merged cols and recalculate bmi: QWK ---> 0.5874, QWK SCORE: 0.767, Accuracy: 0.7965 \n- removed a bunch of FGC cols: QWK: 0.5779, QWK SCORE: 0.760, Accuracy: 0.8063\n- missing values indicator: QWK ---> 0.4986, Optimized QWK: 0.777, Accuracy: 0.8066\n- used OPTUNA QWK: 0.5584, Optimized QWK SCORE: 0.617, Accuracy: 0.7217\n\nTo do: \n- change imputation methods\n- \n\nWhat reduced QWK: \n- winsorizing BMI ","metadata":{}},{"cell_type":"markdown","source":"# Importing libraries and dataset ","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport xgboost as xgb\nimport optuna\nimport lightgbm as lgb \nfrom sklearn.metrics import make_scorer\nfrom scipy.stats.mstats import winsorize\nfrom sklearn.model_selection import train_test_split, GridSearchCV, StratifiedKFold\nfrom sklearn.metrics import accuracy_score, roc_auc_score, make_scorer\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import cross_val_score\nfrom optuna.samplers import TPESampler\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score\nfrom tqdm import tqdm\nfrom scipy.optimize import minimize\nfrom sklearn.utils.class_weight import compute_sample_weight","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:06.640469Z","iopub.execute_input":"2024-10-07T17:06:06.640883Z","iopub.status.idle":"2024-10-07T17:06:10.060564Z","shell.execute_reply.started":"2024-10-07T17:06:06.640843Z","shell.execute_reply":"2024-10-07T17:06:10.059285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:10.065862Z","iopub.execute_input":"2024-10-07T17:06:10.066238Z","iopub.status.idle":"2024-10-07T17:06:10.154048Z","shell.execute_reply.started":"2024-10-07T17:06:10.066193Z","shell.execute_reply":"2024-10-07T17:06:10.152910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data cleaning and preprocessing ","metadata":{}},{"cell_type":"code","source":"def remove_outliers(df):\n    #removing outlier from test \n    df.loc[df['CGAS-CGAS_Score'] > 100, 'CGAS-CGAS_Score'] = np.nan\n    df.loc[df['Physical-Weight'] < 10, 'Physical-Weight'] = np.nan\n    df.loc[df['Physical-Diastolic_BP'] == 0, 'Physical-Diastolic_BP'] =np.nan\n    df.loc[df['Physical-Systolic_BP'] == 0, 'Physical-Systolic_BP'] =np.nan\n    return df\n\ntrain = remove_outliers(train)\ntest = remove_outliers(test)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:10.155493Z","iopub.execute_input":"2024-10-07T17:06:10.155956Z","iopub.status.idle":"2024-10-07T17:06:10.178511Z","shell.execute_reply.started":"2024-10-07T17:06:10.155905Z","shell.execute_reply":"2024-10-07T17:06:10.177240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#recalculate PCIAT\nPCIAT_cols = [ \"PCIAT-PCIAT_01\",\n    \"PCIAT-PCIAT_02\",\n    \"PCIAT-PCIAT_03\",\n    \"PCIAT-PCIAT_04\",\n    \"PCIAT-PCIAT_05\",\n    \"PCIAT-PCIAT_06\",\n    \"PCIAT-PCIAT_07\",\n    \"PCIAT-PCIAT_08\",\n    \"PCIAT-PCIAT_09\",\n    \"PCIAT-PCIAT_10\",\n    \"PCIAT-PCIAT_11\",\n    \"PCIAT-PCIAT_12\",\n    \"PCIAT-PCIAT_13\",\n    \"PCIAT-PCIAT_14\",\n    \"PCIAT-PCIAT_15\",\n    \"PCIAT-PCIAT_16\",    \n    \"PCIAT-PCIAT_17\",\n    \"PCIAT-PCIAT_18\",\n    \"PCIAT-PCIAT_19\",\n    \"PCIAT-PCIAT_20\"]\n\ndef recalculate_sii(row):\n    if pd.isna(row['PCIAT-PCIAT_Total']):\n        return np.nan\n    max_possible = row['PCIAT-PCIAT_Total'] + row[PCIAT_cols].isna().sum() * 5\n    if row['PCIAT-PCIAT_Total'] <= 30 and max_possible <= 30:\n        return 0\n    elif 31 <= row['PCIAT-PCIAT_Total'] <= 49 and max_possible <= 49:\n        return 1\n    elif 50 <= row['PCIAT-PCIAT_Total'] <= 79 and max_possible <= 79:\n        return 2\n    elif row['PCIAT-PCIAT_Total'] >= 80 and max_possible >= 80:\n        return 3\n    return np.nan\n\ntrain['sii'] = train.apply(recalculate_sii, axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:10.182648Z","iopub.execute_input":"2024-10-07T17:06:10.183121Z","iopub.status.idle":"2024-10-07T17:06:11.745115Z","shell.execute_reply.started":"2024-10-07T17:06:10.183073Z","shell.execute_reply":"2024-10-07T17:06:11.744017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Missing values","metadata":{}},{"cell_type":"code","source":"def create_missing_indicators(col_names, df):\n    \"\"\"\n    Creates an indicator variable for each column in col_names, indicating whether the value is missing.\n\n    For each column specified in col_names, a new column will be added to the DataFrame. \n    If a row has a missing value in the corresponding column, the indicator column will be encoded as 1. \n    Otherwise, it will be encoded as 0.\n    \n    Parameters:\n    - col_names: List of column names to create missingness indicators for.\n    - df: The DataFrame where the indicators will be created.\n    \n    Returns:\n    - df: DataFrame with new indicator columns for missing values.\n    \"\"\"\n    for col in col_names:\n        # Create a new column with '_Missing' suffix indicating if a value is missing (1 if missing, 0 otherwise)\n        df[f'{col}_Missing'] = df[col].isna().astype(int)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:11.746411Z","iopub.execute_input":"2024-10-07T17:06:11.746764Z","iopub.status.idle":"2024-10-07T17:06:11.753591Z","shell.execute_reply.started":"2024-10-07T17:06:11.746729Z","shell.execute_reply":"2024-10-07T17:06:11.752257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train\ncreate_missing_indicators(['CGAS-CGAS_Score', 'SDS-SDS_Total_Raw', 'PreInt_EduHx-computerinternet_hoursday', 'Physical-BMI'], train)\n#test\ncreate_missing_indicators(['CGAS-CGAS_Score', 'SDS-SDS_Total_Raw', 'PreInt_EduHx-computerinternet_hoursday', 'Physical-BMI'], test)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:11.755142Z","iopub.execute_input":"2024-10-07T17:06:11.755593Z","iopub.status.idle":"2024-10-07T17:06:11.812321Z","shell.execute_reply.started":"2024-10-07T17:06:11.755551Z","shell.execute_reply":"2024-10-07T17:06:11.811000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to impute all the missing columns \ndef fill_missing_values(train, test, groupby_cols):\n    \"\"\"\n    Fills missing values in both train and test datasets.\n    \n    - Numeric columns: Filled with group-based means (based on groupby_cols).\n    - Categorical columns: Filled with the mode for each column.\n    - Missingness indicators are created only once.\n    \n    Parameters:\n    - train: Training DataFrame.\n    - test: Test DataFrame.\n    - groupby_cols: List of columns to group by for filling numeric missing values.\n    - create_missing_indicators: Boolean to determine whether to create missingness indicators.\n    \n    Returns:\n    - train: Updated training DataFrame with missing values filled.\n    - test: Updated test DataFrame with missing values filled.\n    \"\"\"\n    # Fill missing age with median and create indicator variable\n    median_age = train['Basic_Demos-Age'].median()\n    train['Basic_Demos-Age'].fillna(median_age, inplace=True)\n    test['Basic_Demos-Age'].fillna(median_age, inplace=True)\n\n    # Fill missing sex with mode and create indicator variable\n    mode_sex = train['Basic_Demos-Sex'].mode()[0]\n    train['Basic_Demos-Sex'].fillna(mode_sex, inplace=True)\n    test['Basic_Demos-Sex'].fillna(mode_sex, inplace=True)\n    \n    for df in [train, test]:\n        # Separate numeric columns\n        numeric_cols = df.select_dtypes(include='number')\n\n        # Calculate group-based means\n        group_means = numeric_cols.groupby(groupby_cols).transform('mean')\n\n        # Fill missing numeric values\n        df.update(numeric_cols.fillna(group_means))\n\n        # Fill missing values in categorical columns with the mode\n        # Calculate mode for the train dataset\n        mode_agg = train.select_dtypes(include='object').agg(lambda x: x.mode()[0])\n\n    for df in [train, test]:\n        # Separate categorical columns\n        cat_cols = df.select_dtypes(include='object')\n\n       # Fill missing categorical values with mode\n        df.update(cat_cols.fillna(mode_agg))\n        \n    return train, test\n    \ntrain, test = fill_missing_values(train, test, groupby_cols=['Basic_Demos-Sex','Basic_Demos-Age'])\ntrain, test = fill_missing_values(train, test, groupby_cols=['Basic_Demos-Sex'])\ntrain, test = fill_missing_values(train, test, groupby_cols=['Basic_Demos-Age'])","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:11.813696Z","iopub.execute_input":"2024-10-07T17:06:11.814072Z","iopub.status.idle":"2024-10-07T17:06:12.332320Z","shell.execute_reply.started":"2024-10-07T17:06:11.814034Z","shell.execute_reply":"2024-10-07T17:06:12.331138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Changing some features","metadata":{}},{"cell_type":"code","source":"def preprocess(df):\n    # Merging the PAQ questionnaire\n    df['PAQ_Merged'] = df['PAQ_C-PAQ_C_Total'].combine_first(df['PAQ_A-PAQ_A_Total'])\n    \n    # Recalculating BMI \n    lbs_to_kg = 0.453592\n    inches_to_cm = 2.54\n\n    df['Physical-Weight'] = df['Physical-Weight'] * lbs_to_kg\n    df['Physical-Height'] = df['Physical-Height'] * inches_to_cm\n    df['Physical-Waist_Circumference'] = df['Physical-Waist_Circumference'] * inches_to_cm\n\n    df['Physical-BMI'] = np.where(\n        df['Physical-Weight'].notna() & df['Physical-Height'].notna(),\n        df['Physical-Weight'] / ((df['Physical-Height'] / 100) ** 2),\n        np.nan  # If either is NaN, set BMI to NaN\n    )\n    \n    # Changing endurance timing \n    df['Fitness_Endurance-Total'] = df['Fitness_Endurance-Time_Mins']*60 + df['Fitness_Endurance-Time_Sec']\n    \n    return df\n\n\npreprocess(train)\npreprocess(test)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:12.333544Z","iopub.execute_input":"2024-10-07T17:06:12.333882Z","iopub.status.idle":"2024-10-07T17:06:12.384265Z","shell.execute_reply.started":"2024-10-07T17:06:12.333847Z","shell.execute_reply":"2024-10-07T17:06:12.382979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Specifying features and target variable ","metadata":{}},{"cell_type":"code","source":"# specifiying target columns that needs to be removed \ntarget_cols = [\"PCIAT-Season\", \"PCIAT-PCIAT_01\", \"PCIAT-PCIAT_02\", \"PCIAT-PCIAT_03\", \"PCIAT-PCIAT_04\", \"PCIAT-PCIAT_05\", \"PCIAT-PCIAT_06\", \"PCIAT-PCIAT_07\", \"PCIAT-PCIAT_08\", \"PCIAT-PCIAT_09\", \"PCIAT-PCIAT_10\", \"PCIAT-PCIAT_11\", \"PCIAT-PCIAT_12\", \"PCIAT-PCIAT_13\", \"PCIAT-PCIAT_14\", \"PCIAT-PCIAT_15\", \"PCIAT-PCIAT_16\", \"PCIAT-PCIAT_17\", \"PCIAT-PCIAT_18\", \"PCIAT-PCIAT_19\", \"PCIAT-PCIAT_20\", \"PCIAT-PCIAT_Total\", \"sii\"]","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:12.385606Z","iopub.execute_input":"2024-10-07T17:06:12.385970Z","iopub.status.idle":"2024-10-07T17:06:12.391710Z","shell.execute_reply.started":"2024-10-07T17:06:12.385934Z","shell.execute_reply":"2024-10-07T17:06:12.390578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Removing irrelevant columns ","metadata":{}},{"cell_type":"code","source":"# specifiying target columns that needs to be removed \nirrelevant_cols = [\n'id',\n'PAQ_C-PAQ_C_Total',\n'PAQ_A-PAQ_A_Total',    \n'Fitness_Endurance-Time_Mins',\n'Fitness_Endurance-Time_Sec',\n'FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL_Zone','FGC-FGC_GSND', 'FGC-FGC_GSD',\n#'Physical-Waist_Circumference'  \n]\ncategorical = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season','Fitness_Endurance-Season','FGC-Season',\n'BIA-Season','PAQ_A-Season','PAQ_C-Season','SDS-Season','PreInt_EduHx-Season']\n\ndef remove_columns(df):\n    columns_to_remove = categorical + irrelevant_cols\n    df.drop(columns = columns_to_remove, inplace = True)\n\nremove_columns(train)\nremove_columns(test)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:12.393407Z","iopub.execute_input":"2024-10-07T17:06:12.393781Z","iopub.status.idle":"2024-10-07T17:06:12.409526Z","shell.execute_reply.started":"2024-10-07T17:06:12.393744Z","shell.execute_reply":"2024-10-07T17:06:12.408156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering ","metadata":{}},{"cell_type":"code","source":"BIA_columns =  ['BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW']","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:12.411609Z","iopub.execute_input":"2024-10-07T17:06:12.411984Z","iopub.status.idle":"2024-10-07T17:06:12.422459Z","shell.execute_reply.started":"2024-10-07T17:06:12.411946Z","shell.execute_reply":"2024-10-07T17:06:12.421317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def PCA_BIA_train(train_df):\n#     X_bia_train = train_df[BIA_columns]\n\n#     # Standardize the data\n#     scaler = StandardScaler()\n#     X_bia_train_scaled = scaler.fit_transform(X_bia_train)\n\n#     # Apply PCA\n#     n_components = 3  # Choose number of components to retain\n#     pca = PCA(n_components=n_components, random_state=211)\n#     X_bia_train_pca = pca.fit_transform(X_bia_train_scaled)\n\n#     # Explained variance ratio\n#     explained_variance = pca.explained_variance_ratio_\n#     print(f'Explained variance by PCA components: {explained_variance}')\n\n#     # Create a DataFrame for PCA components\n#     pca_columns = [f'BIA_PCA_{i+1}' for i in range(n_components)]\n#     X_bia_train_pca_df = pd.DataFrame(X_bia_train_pca, columns=pca_columns, index=train_df.index)\n\n#     # Combine PCA components with the original train dataframe (keeping the original features)\n#     train_with_pca = pd.concat([train_df, X_bia_train_pca_df], axis=1)\n\n#     return pca, scaler, train_with_pca\n\n# def PCA_BIA_test(test_df, pca, scaler):\n#     X_bia_test = test_df[BIA_columns]\n\n#     # Standardize the test data using the same scaler fitted on the training data\n#     X_bia_test_scaled = scaler.transform(X_bia_test)\n\n#     # Apply the trained PCA transformation to test data\n#     X_bia_test_pca = pca.transform(X_bia_test_scaled)\n\n#     # Create a DataFrame for PCA components\n#     pca_columns = [f'BIA_PCA_{i+1}' for i in range(pca.n_components)]\n#     X_bia_test_pca_df = pd.DataFrame(X_bia_test_pca, columns=pca_columns, index=test_df.index)\n\n#     # Combine PCA components with the original test dataframe (keeping the original features)\n#     test_with_pca = pd.concat([test_df, X_bia_test_pca_df], axis=1)\n\n#     return test_with_pca\n\n# # Fitting PCA on training data\n# pca_model, scaler_model, train_with_pca = PCA_BIA_train(train)\n\n# # Applying PCA to test data using the fitted model\n# test_with_pca = PCA_BIA_test(test, pca_model, scaler_model)\n\n# train = train_with_pca.drop(columns = BIA_columns)\n# test = test_with_pca.drop(columns = BIA_columns)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:12.423822Z","iopub.execute_input":"2024-10-07T17:06:12.424240Z","iopub.status.idle":"2024-10-07T17:06:12.434826Z","shell.execute_reply.started":"2024-10-07T17:06:12.424190Z","shell.execute_reply":"2024-10-07T17:06:12.433686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating X and Y ","metadata":{}},{"cell_type":"code","source":"# Feature vector \nX = train.drop(columns = target_cols)\n\n# target variable \ny = train['sii']\ny = pd.cut(y, bins=[-np.inf, 0.5, 1.5, 2.5, np.inf], labels=[0, 1, 2, 3]) \ny = y.astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:12.440905Z","iopub.execute_input":"2024-10-07T17:06:12.441340Z","iopub.status.idle":"2024-10-07T17:06:12.458245Z","shell.execute_reply.started":"2024-10-07T17:06:12.441300Z","shell.execute_reply":"2024-10-07T17:06:12.456986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = y.mean()\nb = y.var(ddof=0)\n\ny_min = y.min()\ny_max = y.max()\n\ndef qwk_obj(preds, dtrain):\n    labels = dtrain.get_label()\n    preds = preds.clip(y_min, y_max)\n    f = 1/2 * np.sum((preds - labels)**2)\n    g = 1/2 * np.sum((preds - a)**2 + b)\n    df = preds - labels\n    dg = preds - a\n    grad = (df/g - f*dg/g**2)*len(labels)\n    hess = np.ones(len(labels))\n    return grad, hess\n\ndef quadratic_weighted_kappa_metric(preds, data):\n    y_true = data.get_label()\n    y_pred = preds.clip(y_min, y_max).round()\n    qwk = cohen_kappa_score(y_true, y_pred, weights=\"quadratic\")\n    return 'QWK', qwk, True\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\nqwk_scorer = make_scorer(quadratic_weighted_kappa, greater_is_better=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:12.459658Z","iopub.execute_input":"2024-10-07T17:06:12.460059Z","iopub.status.idle":"2024-10-07T17:06:12.470560Z","shell.execute_reply.started":"2024-10-07T17:06:12.459999Z","shell.execute_reply":"2024-10-07T17:06:12.469412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {\n    \"objective\": qwk_obj,\n    \"metric\": \"None\",\n    \"verbosity\": -1,\n    \"learning_rate\": 0.01,\n    \"num_leaves\": 16,\n    \"feature_fraction\": 0.6\n}\n\n# The initial score of the model is the key parameter.\n# I found that the mean value of the target is a bad choice for this data.\ninit_score = 2.0","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:12.472137Z","iopub.execute_input":"2024-10-07T17:06:12.472620Z","iopub.status.idle":"2024-10-07T17:06:12.486547Z","shell.execute_reply.started":"2024-10-07T17:06:12.472569Z","shell.execute_reply":"2024-10-07T17:06:12.485436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=0)\nfolds = [(idx_train, idx_valid) for idx_train, idx_valid in skf.split(X, y)]\n\nmodels = lgb.cv(\n    params=params,\n    train_set=lgb.Dataset(X, y, init_score=[init_score]*len(X)),\n    num_boost_round=10000,\n    folds=folds,\n    feval=quadratic_weighted_kappa_metric,\n    callbacks=[\n        lgb.early_stopping(stopping_rounds=100, verbose=True),\n        lgb.log_evaluation(100)\n    ],\n    return_cvbooster=True\n    )[\"cvbooster\"].boosters\n\npreds_oof = np.zeros(len(train))\nfor model, (idx_train, idx_valid) in zip(models, folds):\n    preds_oof[idx_valid] = model.predict(X.iloc[idx_valid]) + init_score","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:12.487819Z","iopub.execute_input":"2024-10-07T17:06:12.488238Z","iopub.status.idle":"2024-10-07T17:06:33.272161Z","shell.execute_reply.started":"2024-10-07T17:06:12.488150Z","shell.execute_reply":"2024-10-07T17:06:33.270834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Optuna Hyperparamter tuning ","metadata":{}},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:33.273876Z","iopub.execute_input":"2024-10-07T17:06:33.274256Z","iopub.status.idle":"2024-10-07T17:06:33.281267Z","shell.execute_reply.started":"2024-10-07T17:06:33.274218Z","shell.execute_reply":"2024-10-07T17:06:33.279745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def objective(trial):\n    skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=0)\n    folds = [(idx_train, idx_valid) for idx_train, idx_valid in skf.split(X, y)]\n\n    param = {\n        'objective': qwk_obj,  # Your custom objective function\n        'metric': 'None',  # Custom metric\n        'random_state': 42,\n        'learning_rate': 0.02,\n        'reg_alpha': trial.suggest_float('reg_alpha', 1e-3, 10.0),\n        'reg_lambda': trial.suggest_float('reg_lambda', 1e-3, 10.0),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n        'subsample': trial.suggest_float('subsample', 0.4, 1.0),\n        'max_depth': trial.suggest_int('max_depth', 5, 65),\n        'num_leaves': trial.suggest_int('num_leaves', 10, 300),\n        'min_child_samples': trial.suggest_int('min_child_samples', 1, 300),\n        'cat_smooth': trial.suggest_int('min_data_per_groups', 1, 100),\n        \"verbosity\": -1,\n    }\n    \n    oof_non_rounded = np.zeros(len(X))  # Out-of-fold non-rounded predictions\n    for fold, (train_idx, valid_idx) in enumerate(folds):\n        X_train, y_train = X.iloc[train_idx], y.iloc[train_idx]\n        X_valid, y_valid = X.iloc[valid_idx], y.iloc[valid_idx]\n\n        train_data = lgb.Dataset(X_train, label=y_train)\n        valid_data = lgb.Dataset(X_valid, label=y_valid)\n\n        model = lgb.train(\n            params=param,\n            train_set=train_data,\n            num_boost_round=600,\n            valid_sets=[train_data, valid_data],\n            feval=quadratic_weighted_kappa_metric,\n            callbacks=[\n                lgb.early_stopping(stopping_rounds=100, verbose=True),\n                lgb.log_evaluation(100)\n            ]\n        )\n\n        # Predict and store non-rounded predictions\n        oof_non_rounded[valid_idx] = model.predict(X_valid, num_iteration=model.best_iteration)\n\n    # Optimize thresholds for the predictions\n    thresholds_initial = [0.5, 1.5, 2.5]  # Example initial thresholds\n    KappaOptimizer = minimize(evaluate_predictions,\n                              x0=thresholds_initial, \n                              args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    \n    assert KappaOptimizer.success, \"Optimization did not converge.\"\n\n    # Round the predictions using optimized thresholds\n    oof_rounded = threshold_Rounder(oof_non_rounded, KappaOptimizer.x)\n    \n    # Return out-of-fold score (Kappa)\n    score = quadratic_weighted_kappa(y, oof_rounded)\n\n    # Suppress optuna logging\n    optuna.logging.set_verbosity(optuna.logging.ERROR)\n    \n    return score","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:33.282910Z","iopub.execute_input":"2024-10-07T17:06:33.283347Z","iopub.status.idle":"2024-10-07T17:06:33.300318Z","shell.execute_reply.started":"2024-10-07T17:06:33.283305Z","shell.execute_reply":"2024-10-07T17:06:33.299211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nstudy = optuna.create_study(direction='maximize')\nstudy.optimize(objective, n_trials=200)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T17:06:33.301541Z","iopub.execute_input":"2024-10-07T17:06:33.301879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number of finished trials:', len(study.trials))\nprint('Best trial:', study.best_trial.params)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.visualization.plot_optimization_history(study)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.visualization.plot_slice(study)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params=study.best_params   \nparams['random_state'] = 42\nparams['metric'] = 'None'\nparams['objective'] = qwk_obj\nparams[\"verbosity\"] = -1,\nparams['cat_smooth'] = params.pop('min_data_per_groups')\nparams['learning_rate']= 0.02,\nparams","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Checking results after ","metadata":{}},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=0)\nfolds = [(idx_train, idx_valid) for idx_train, idx_valid in skf.split(X, y)]\n\nparams={'reg_alpha': 5.292045270173479,\n 'reg_lambda': 7.389785522475791,\n 'colsample_bytree': 0.5788687048723766,\n 'subsample': 0.8312228809999682,\n 'learning_rate': 0.01693434250885808,\n 'max_depth': 36,\n 'num_leaves': 344,\n 'min_child_samples': 92,\n 'random_state': 42,\n 'metric': 'None',\n 'objective': qwk_obj,\n 'verbosity': -1,\n 'cat_smooth': 8}\n\nmodels = lgb.cv(\n    params=params,\n    train_set=lgb.Dataset(X, y, init_score=[init_score]*len(X)),\n    num_boost_round=10000,\n    folds=folds,\n    feval=quadratic_weighted_kappa_metric,\n    callbacks=[\n        lgb.early_stopping(stopping_rounds=100, verbose=True),\n        lgb.log_evaluation(100)\n    ],\n    return_cvbooster=True\n    )[\"cvbooster\"].boosters\n\npreds_oof = np.zeros(len(train))\nfor model, (idx_train, idx_valid) in zip(models, folds):\n    preds_oof[idx_valid] = model.predict(X.iloc[idx_valid]) + init_score","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# without threshold optimization\npreds_oof = preds_oof.clip(y_min, y_max).round()\nqwk = cohen_kappa_score(y, preds_oof, weights=\"quadratic\")\nprint(\"QWK:\", qwk)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extended_cohen_kappa_score(labels, preds):\n    f = np.sum((preds - labels)**2)\n    g = np.sum((preds - a) ** 2 + b)\n    return 1 - f / g \n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_confusion_matrix(y_true, y_pred, fold):\n    cm = confusion_matrix(y_true, y_pred)\n    plt.figure(figsize=(8, 6))\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', \n                xticklabels=np.unique(y_true), \n                yticklabels=np.unique(y_true))\n    plt.title(f'Confusion Matrix')\n    plt.xlabel('Predicted')\n    plt.ylabel('True')\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, classification_report\n# accuracy\nplot_confusion_matrix(y, preds_oof, 0)\nprint(f\"Accuracy: {accuracy_score(y, preds_oof)}\")\nprint(f\" QWK: {extended_cohen_kappa_score(y, preds_oof)}\")\nprint(classification_report(y, preds_oof))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model = lgb.train(\n    params=params,\n    train_set=lgb.Dataset(X, label=y),\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Credits to user: Zarnain Syed for the Optimizer code","metadata":{}},{"cell_type":"code","source":"# def plot_feature_importance(X, feature_importances):\n#     feature_names = X.columns\n\n#     avg_feature_importance = np.mean(feature_importances, axis=0)\n#     importance_df = pd.DataFrame({'Feature': feature_names, 'Importance': avg_feature_importance})\n    \n#     # Sort features by importance\n#     importance_df = importance_df.sort_values(by='Importance', ascending=False)\n\n#     # Plot feature importance\n#     plt.figure(figsize=(10, 6))\n#     plt.barh(importance_df['Feature'], importance_df['Importance'], color='skyblue')\n#     plt.xlabel('Importance Score')\n#     plt.title('Feature Importance')\n#     plt.gca().invert_yaxis()  # Invert axis to show most important at the top\n#     plt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.optimize import minimize\nfrom sklearn.metrics import cohen_kappa_score\nfrom lightgbm import LGBMClassifier  # or LGBMRegressor\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\n\ndef TrainML(best_model, X, y, test_data):\n    n_splits = 5\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n\n    train_S = []\n    test_S = []\n\n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        # Predict using the best_model directly for the validation set\n        y_val_pred = best_model.predict(X_val)\n        oof_non_rounded[test_idx] = y_val_pred\n        \n        y_val_pred_rounded = np.round(y_val_pred).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n        \n        # Train kappa calculation\n        train_kappa = quadratic_weighted_kappa(y_train, np.round(best_model.predict(X_train)).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n        \n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        # Predict on test set\n        test_preds[:, fold] = best_model.predict(test_data)\n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    # Optimizing the thresholds for the predictions\n    KappaOptimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], \n                              args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    \n    assert KappaOptimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOptimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {tKappa:.3f}\")\n\n    # Apply optimized thresholds to the test set predictions\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOptimizer.x)\n    plot_confusion_matrix(y, oof_tuned, 0)\n    print(f\"Accuracy: {accuracy_score(y, oof_tuned)}\")\n    \n    # Prepare submission\n    test_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    submission = pd.DataFrame({\n        'id': test_data['id'],\n        'sii': tpTuned\n    })\n\n    return submission, best_model, X","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission, models, X = TrainML(final_model, X, y, test)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install optuna-integration[lightgbm]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from optuna.integration import lightgbm as lgb\nlgb.plot_importance(model, max_num_features=10, figsize=(10,10))\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}