{"metadata":{"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport lightgbm as lgb\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nimport polars as pl\nimport os\nimport lightgbm as lgb\nfrom sklearn.model_selection import StratifiedKFold, GridSearchCV\nfrom sklearn.metrics import make_scorer, cohen_kappa_score\nfrom sklearn.impute import KNNImputer\nfrom sklearn.impute import SimpleImputer\nimport numpy as np\nimport pickle\nfrom itertools import product\n\nimport glob\nfrom tqdm import tqdm\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_DIR = \"/kaggle/input/child-mind-institute-problematic-internet-use/\"\n# INPUT_DIR = f\"{os.getcwd()}/data/\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the data using Polars\ntrain_df = pl.read_csv(INPUT_DIR + \"train.csv\")\ntrain_df_og = train_df.clone()  \ntest_df = pl.read_csv(INPUT_DIR + \"test.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"REMAP_TARGET3 = True\nUSE_ACTIGRAPHY = True\nOVERSAMPLE_SMOTE = True","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here is a summary of the feature engineering steps in the provided code:\n\nData Loading and Filtering:\n\nLoad the train and test datasets while filtering out columns that match a specific regex pattern ('^(?!PCIAT.*$)').\nHandling Missing Values:\n\nFor numeric columns, use the KNNImputer with 5 neighbors to impute missing values.\nFor categorical columns, use SimpleImputer with the 'most frequent' strategy to impute missing values.\nOne-Hot Encoding:\n\nConvert categorical columns into numerical format using one-hot encoding via pd.get_dummies. This ensures that the categorical variables are transformed into binary columns representing each category.\nConsistent Feature Columns:\n\nTo ensure consistent feature columns across both train and test datasets, concatenate the train and test data before performing one-hot encoding. This step guarantees that both datasets have the same set of columns.\nAfter one-hot encoding, the processed data is reindexed to match the full set of columns (all_columns) to ensure no discrepancies between the training and test data.\nTarget Imputation:\n\nThe target variable ('sii') is imputed using the KNNImputer. This involves encoding the features of the training data, concatenating them with the target column, and applying KNN imputation to fill missing target values.\nCross-Validation:\n\nImplement 5-fold stratified cross-validation (StratifiedKFold) to evaluate the model's performance on different subsets of the data.\nLightGBM Model Training:\n\nA fixed set of hyperparameters is used to train a LightGBM classifier on each fold of the training data.\nModel Selection:\n\nThe model with the highest average Cohen's Kappa Score across the validation sets is saved as the best model.\nTest Data Preprocessing and Prediction:\n\nPreprocess the test data using the same imputation and one-hot encoding strategies as applied to the training data, ensuring consistent columns.\nUse the best-trained model to predict the target variable ('sii') on the test data.\nSubmission File Creation:\n\nCreate a CSV submission file containing the predicted target values ('sii') for the test data, along with their corresponding IDs.","metadata":{}},{"cell_type":"markdown","source":"Current Best Kappa Score created with {'num_leaves': 31, 'learning_rate': 0.05, 'lambda_l1': 0.5, 'lambda_l2': 0.1, 'max_depth': 10, 'verbose': -1}: 0.4200","metadata":{}},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"#### Add Actigraphy","metadata":{}},{"cell_type":"code","source":"# pd.read_parquet(rf\"{INPUT_DIR}/series_train.parquet\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport polars as pl\nfrom scipy.stats import entropy\nfrom tqdm import tqdm\n\ndef process_actigraphy_for_id(id_value, dataset_type):\n    # Construct the file path for the specific id\n    base_directory = os.path.join(INPUT_DIR, f'series_{dataset_type}.parquet')\n# #     print(f\"Processing ID {id_value} from {base_directory}\")\n    \n    # Load the actigraphy data for the specific id from the Parquet file using Polars\n    try:\n        actigraphy_data = pl.scan_parquet(base_directory).filter(pl.col('id') == id_value).collect()\n        if actigraphy_data.is_empty():\n#             print(f\"No data found for ID {id_value}\")\n            return pl.DataFrame()  # Return empty DataFrame if no data for the id\n        \n    except Exception as e:\n#         print(f\"Error loading data for ID {id_value}: {e}\")\n        return pl.DataFrame()  # Return empty DataFrame in case of error\n    \n    # Check if the required columns are available\n    required_columns = ['enmo', 'light', 'anglez']\n    if not all(col in actigraphy_data.columns for col in required_columns):\n#         print(f\"Missing required columns for ID {id_value}\")\n        return pl.DataFrame()\n\n    # Ensure we have valid data for feature calculations\n    valid_data = actigraphy_data.select(required_columns).drop_nulls()\n    if valid_data.is_empty():\n#         print(f\"No valid data for required columns for ID {id_value}\")\n        return pl.DataFrame()\n\n    # Feature extraction for the specific id\n    features = {'id': id_value}\n\n    # enmo features\n    enmo = valid_data['enmo']\n    features.update({\n        'enmo_mean': enmo.mean(),\n        'enmo_median': enmo.median(),\n        'enmo_var': enmo.var(),\n        'enmo_min': enmo.min(),\n        'enmo_max': enmo.max(),\n        'enmo_25th': np.percentile(enmo, 25),\n        'enmo_75th': np.percentile(enmo, 75),\n        'enmo_skew': enmo.skew(),\n        'enmo_kurtosis': enmo.kurtosis(),\n        'enmo_entropy': entropy(np.histogram(enmo, bins=10, density=True)[0])\n    })\n\n    # light features\n    light = valid_data['light']\n    features.update({\n        'light_mean': light.mean(),\n        'light_median': light.median(),\n        'light_var': light.var(),\n        'light_min': light.min(),\n        'light_max': light.max(),\n        'light_25th': np.percentile(light, 25),\n        'light_75th': np.percentile(light, 75),\n        'light_skew': light.skew(),\n        'light_kurtosis': light.kurtosis(),\n        'light_entropy': entropy(np.histogram(light, bins=10, density=True)[0])\n    })\n\n    # anglez features\n    anglez = valid_data['anglez']\n    features.update({\n        'anglez_mean': anglez.mean(),\n        'anglez_median': anglez.median(),\n        'anglez_var': anglez.var(),\n        'anglez_min': anglez.min(),\n        'anglez_max': anglez.max()\n    })\n\n    # Additional features\n    features['enmo_IQR'] = features['enmo_75th'] - features['enmo_25th']\n    features['light_IQR'] = features['light_75th'] - features['light_25th']\n    features['enmo_peak_to_trough_ratio'] = features['enmo_max'] / (features['enmo_min'] + 1e-5)\n    features['light_peak_to_trough_ratio'] = features['light_max'] / (features['light_min'] + 1e-5)\n    \n    # Motion and activity features\n    features['no_motion_periods'] = (enmo == 0).sum()\n    features['no_motion_percentage'] = features['no_motion_periods'] / len(enmo)\n\n    mvpa_threshold = 0.1\n    features['mvpa_periods'] = (enmo > mvpa_threshold).sum()\n    features['mvpa_percentage'] = features['mvpa_periods'] / len(enmo)\n\n    # Day and night activity\n    actigraphy_data = actigraphy_data.with_columns(\n            ((pl.col('time_of_day') >= 8 * 3600 * 1e9) & (pl.col('time_of_day') < 21 * 3600 * 1e9)).alias('is_day')\n        )\n    day_data = actigraphy_data.filter(pl.col('is_day'))['enmo']\n    night_data = actigraphy_data.filter(~pl.col('is_day'))['enmo']\n    if not day_data.is_empty() and not night_data.is_empty():\n        features['day_night_ratio'] = day_data.mean() / night_data.mean()\n    else:\n        features['day_night_ratio'] = np.nan\n\n    # Weekday vs. weekend ratio\n    weekday_data = actigraphy_data.filter(pl.col('weekday').is_in([1, 2, 3, 4, 5]))['enmo']\n    weekend_data = actigraphy_data.filter(pl.col('weekday').is_in([6, 7]))['enmo']\n    if not weekday_data.is_empty() and not weekend_data.is_empty():\n        features['weekday_weekend_ratio'] = weekday_data.mean() / weekend_data.mean()\n    else:\n        features['weekday_weekend_ratio'] = np.nan\n\n    # Convert to Polars DataFrame and return\n    return pl.DataFrame([features])\n\ndef process_all_actigraphy_data(dataset_type, id_list):\n    all_actigraphy_data = []\n\n    for id_value in tqdm(id_list, desc=f\"Processing {dataset_type} IDs\"):\n        actigraphy_chunk = process_actigraphy_for_id(id_value, dataset_type)\n        if not actigraphy_chunk.is_empty():\n            all_actigraphy_data.append(actigraphy_chunk)\n\n    # Concatenate all processed data\n    return pl.concat(all_actigraphy_data) if all_actigraphy_data else pl.DataFrame()\n\n# Example usage for processing the train and test actigraphy data:\nif USE_ACTIGRAPHY:\n    train_ids = train_df['id'].unique()  # List of unique train IDs\n    test_ids = test_df['id'].unique()    # List of unique test IDs\n\n    train_actigraphy = process_all_actigraphy_data('train', train_ids)\n    test_actigraphy = process_all_actigraphy_data('test', test_ids)\n\n    # Merge actigraphy data back into train and test\n    train_df = train_df.join(train_actigraphy, on='id', how='left')\n    test_df = test_df.join(test_actigraphy, on='id', how='left')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### recalculate sii","metadata":{}},{"cell_type":"code","source":"#For now, we can conclude that the SII score is sometimes incorrect. Below I recalculate the SII based on PCIAT_Total and the maximum possible score if missing values were answered (5 points), ensuring that the recalculated SII meets the intended thresholds even with some missing answers.\n# idea from https://www.kaggle.com/code/antoninadolgorukova/cmi-piu-features-eda\n# Assume train_df is a Polars DataFrame\ntrain_df = train_df.to_pandas()\ntest_df = test_df.to_pandas()\n\ndef recalculate_sii(row):\n    PCIAT_cols = [f'PCIAT-PCIAT_{i+1:02d}' for i in range(20)]\n    if pd.isna(row['PCIAT-PCIAT_Total']):\n        return np.nan\n    max_possible = row['PCIAT-PCIAT_Total'] + row[PCIAT_cols].isna().sum() * 5\n    if row['PCIAT-PCIAT_Total'] <= 30 and max_possible <= 30:\n        return 0\n    elif 31 <= row['PCIAT-PCIAT_Total'] <= 49 and max_possible <= 49:\n        return 1\n    elif 50 <= row['PCIAT-PCIAT_Total'] <= 79 and max_possible <= 79:\n        return 2\n    elif row['PCIAT-PCIAT_Total'] >= 80 and max_possible >= 80:\n        return 3\n    return np.nan\n\ntrain_df['sii'] = train_df.apply(recalculate_sii, axis=1)\n\nif REMAP_TARGET3:\n    train_df = train_df.replace(3, 2)\n    \ntrain_df = train_df.filter(regex='^(?!PCIAT.*$)')\ntest_df = test_df.filter(regex='^(?!PCIAT.*$)')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(train_df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### extensive feature engineering","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.preprocessing import PolynomialFeatures\nfrom sklearn.cluster import KMeans\nfrom scipy.spatial.distance import euclidean\nfrom statsmodels.stats.outliers_influence import variance_inflation_factor\n\n# Store transformations and columns to be dropped\ntransformations = {}\nkept_columns = None\n\n# Function to calculate VIF\ndef calculate_vif(X):\n    vif_data = pd.DataFrame()\n    vif_data['Variable'] = X.columns\n    vif_data['VIF'] = [min(variance_inflation_factor(X.values, i), 20) if np.isfinite(variance_inflation_factor(X.values, i)) else 20 for i in range(X.shape[1])]  # Clip VIF values at 20 and fill NaNs with 20\n    return vif_data\n\n# Step 1: Polynomial Features (Degree 2)\ndef create_polynomial_features(X, degree=2, fit=True):\n    if fit:\n        transformations['poly'] = PolynomialFeatures(degree=degree, interaction_only=False, include_bias=False)\n        X_poly = transformations['poly'].fit_transform(X)\n    else:\n        X_poly = transformations['poly'].transform(X)\n    \n    poly_feature_names = transformations['poly'].get_feature_names_out(X.columns)\n    return pd.DataFrame(X_poly, columns=poly_feature_names)\n\n# Step 2: Interaction Features (Pairwise Multiplication)\ndef create_interaction_features(X):\n    interaction_features = pd.DataFrame(index=X.index)\n    for i, col1 in enumerate(X.columns):\n        for col2 in X.columns[i + 1:]:\n            interaction_features[f'{col1}_x_{col2}'] = X[col1] * X[col2]\n    return interaction_features\n\n# Step 3: Logarithmic and Exponential Transformations\ndef create_log_exp_features(X):\n    log_exp_features = pd.DataFrame(index=X.index)\n    for col in X.columns:\n        if (X[col] > 0).all():  # Logarithms require positive values\n            log_exp_features[f'log_{col}'] = np.log1p(X[col])\n        log_exp_features[f'exp_{col}'] = np.exp(X[col])\n    return log_exp_features\n\n# Step 4: Binning/Discretization\ndef create_binning_features(X, bins=5, fit=True):\n    binning_features = pd.DataFrame(index=X.index)\n    for col in X.columns:\n        if fit:\n            bin_edges = np.histogram_bin_edges(X[col], bins=bins)\n            transformations[f'binned_{col}'] = bin_edges\n        else:\n            bin_edges = transformations[f'binned_{col}']\n        binning_features[f'binned_{col}'] = np.digitize(X[col], bins=bin_edges, right=True)\n    return binning_features\n\n# Step 5: Feature Ratios\ndef create_feature_ratios(X):\n    ratio_features = pd.DataFrame(index=X.index)\n    for i, col1 in enumerate(X.columns):\n        for col2 in X.columns[i + 1:]:\n            ratio_features[f'{col1}_ratio_{col2}'] = X[col1] / (X[col2] + 1e-5)\n    return ratio_features\n\n# Step 6: Sine and Cosine Transforms (for Cyclic Features like Time)\ndef create_sine_cosine_features(X, period=24):\n    sine_cosine_features = pd.DataFrame(index=X.index)\n    for col in X.columns:\n        sine_cosine_features[f'sin_{col}'] = np.sin(2 * np.pi * X[col] / period)\n        sine_cosine_features[f'cos_{col}'] = np.cos(2 * np.pi * X[col] / period)\n    return sine_cosine_features\n\n# Step 7: Clustering-Based Features (KMeans)\ndef create_clustering_features(X, n_clusters=5, fit=True):\n    if fit:\n        transformations['kmeans'] = KMeans(n_clusters=n_clusters, random_state=42)\n        cluster_labels = transformations['kmeans'].fit_predict(X)\n    else:\n        cluster_labels = transformations['kmeans'].predict(X)\n    \n    return pd.DataFrame({'cluster': cluster_labels}, index=X.index)\n\n# Step 8: Distance Features\ndef create_distance_features(X, reference_point=None):\n    if reference_point is None:\n        reference_point = X.mean().values  # Use centroid of data if no reference is provided\n    distance_features = pd.DataFrame(index=X.index)\n    for col in X.columns:\n        distance_features[f'distance_from_{col}'] = X.apply(lambda row: euclidean(row.values, reference_point), axis=1)\n    return distance_features\n\n# Extensive Feature Engineering (with column tracking for test)\ndef extensive_feature_engineering(X, y=None, vif_threshold=5, fit=True):\n    global kept_columns  # To store which columns are kept\n\n    # Polynomial Features\n    X_poly = create_polynomial_features(X, fit=fit)\n    \n    # Interaction Features\n    X_interaction = create_interaction_features(X)\n    \n    # Logarithmic and Exponential Features\n    X_log_exp = create_log_exp_features(X)\n    \n    # Binned Features\n    X_binned = create_binning_features(X, fit=fit)\n    \n    # Ratio Features\n    X_ratios = create_feature_ratios(X)\n    \n    # Sine and Cosine Transformed Features\n    X_sine_cosine = create_sine_cosine_features(X)\n    \n    # Clustering Features\n    X_clustering = create_clustering_features(X, fit=fit)\n    \n    # Distance Features\n    X_distance = create_distance_features(X)\n\n    # Concatenate all the new features\n    X_new = pd.concat([X, X_poly, X_interaction, X_log_exp, X_binned, X_ratios, \n                       X_sine_cosine, X_clustering, X_distance], axis=1)\n    \n    # For training: Drop features with low correlation to target 'y'\n    if y is not None and fit:\n        correlations = X_new.corrwith(y)  # Compute correlation of each feature with 'y'\n        kept_columns = correlations.abs() >= 0.25  # Keep track of columns to keep\n        X_new = X_new.loc[:, kept_columns]\n    \n    # For testing: Apply the same column selection as in the training set\n    elif not fit and kept_columns is not None:\n        X_new = X_new.loc[:, kept_columns]\n\n    return X_new\n\n# Apply transformations to train and test\nnumeric_cols = train_df.drop(columns=['id', 'sii']).select_dtypes(include=[np.number]).columns\nX_train = train_df[numeric_cols]  # Use only numeric columns initially for transformations\n\n# Extensive feature engineering on train data (fit=True for train)\nX_train_new = extensive_feature_engineering(X_train.fillna(-1), train_df['sii'], fit=True)\n\n# Assign all the newly created columns back to train_df\ntrain_df = pd.concat([train_df, X_train_new], axis=1)\n\n# Extensive feature engineering on test data (fit=False to reuse transformations from train)\nX_test = test_df[numeric_cols]\nX_test_new = extensive_feature_engineering(X_test.fillna(-1), fit=False).copy()\n\n# Assign all the newly created columns back to test_df\ntest_df = pd.concat([test_df, X_test_new], axis=1).copy()\n\ntrain_df = train_df.loc[:, ~train_df.columns.duplicated()]\ntest_df = test_df.loc[:, ~test_df.columns.duplicated()]\nprint(len(train_df.columns))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Preprocessing and KNN imputation functions","metadata":{}},{"cell_type":"code","source":"import itertools\nimport os\nimport numpy as np\nimport lightgbm as lgb\nimport xgboost as xgb\nfrom sklearn.preprocessing import normalize\nfrom sklearn.ensemble import RandomForestRegressor, VotingRegressor\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import make_scorer, cohen_kappa_score\nfrom sklearn.impute import KNNImputer, SimpleImputer\nfrom sklearn.feature_selection import RFECV\nfrom scipy.optimize import minimize\nimport pickle\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler\n\n# Directories\nRESULTS_DIR = 'model_results'\nresults_file = f'{RESULTS_DIR}/hyperparameter_results.csv'\ncheckpoint_file = f'{RESULTS_DIR}/checkpoint.pkl'\n\n# Create required directories and files\nos.makedirs(RESULTS_DIR, exist_ok=True)\nif not os.path.exists(results_file):\n    pd.DataFrame(columns=['model', 'params', 'fold', 'train_qwk', 'val_qwk', 'kappa_tuned', 'num_features', 'selected_features']).to_csv(results_file, index=False)\n\n# Load checkpoint data (to resume from previous progress)\nif os.path.exists(checkpoint_file):\n    with open(checkpoint_file, 'rb') as f:\n        checkpoint_data = pickle.load(f)\nelse:\n    checkpoint_data = None\n\n# Preprocessing function\ndef preprocess_data(df, numeric_cols=None, categorical_cols=None, imputer_numeric=None, imputer_categorical=None, all_columns=None, train=True):\n    if train:\n        imputer_numeric = KNNImputer(n_neighbors=5)  # KNNImputer for numeric columns\n        imputer_categorical = SimpleImputer(strategy='most_frequent')  # SimpleImputer for categorical columns\n    \n    # Process numeric columns individually\n    for col in numeric_cols:\n        # Replace inf and -inf with NaN\n        df[col] = df[col].replace([np.inf, -np.inf], np.nan)\n\n        # Clip extremely large values to avoid issues\n        df[col] = df[col].clip(upper=np.finfo(np.float64).max)\n\n    # Apply numeric imputer\n    df[numeric_cols] = imputer_numeric.fit_transform(df[numeric_cols]) if train else imputer_numeric.transform(df[numeric_cols])\n\n    # Apply categorical imputer\n    df[categorical_cols] = imputer_categorical.fit_transform(df[categorical_cols]) if train else imputer_categorical.transform(df[categorical_cols])\n    \n    # Z-score normalization for numeric columns\n    scaler = StandardScaler()\n    try:\n        df[numeric_cols] = scaler.fit_transform(df[numeric_cols]) if train else scaler.transform(df[numeric_cols])\n    except:\n        df[numeric_cols] = scaler.fit_transform(df[numeric_cols])\n        \n    # One-hot encoding of categorical variables (after imputation)\n    df = pd.get_dummies(df, columns=categorical_cols, drop_first=True)  # drop_first=True to avoid multicollinearity\n\n    # Ensure column names are unique\n    df = df.loc[:, ~df.columns.duplicated()]\n\n    # Reindex to ensure consistent columns across all datasets\n    df = df.reindex(columns=all_columns, fill_value=0)\n    \n    return df, imputer_numeric, imputer_categorical, scaler\n\nfrom sklearn.impute import KNNImputer\nfrom sklearn.preprocessing import normalize\n\n# Preprocessing the target variable 'sii' using KNN Imputation and normalizing numeric columns\ndef impute_target_knn(X_train, y_train):\n    print(\"Starting KNN imputation for the target variable...\")\n\n    # Identify numeric columns\n    numeric_cols = X_train.select_dtypes(include=['int64', 'float64']).columns\n\n    # Replace inf and -inf with NaN column by column and fill NaN\n    for col in numeric_cols:\n        X_train[col] = X_train[col].replace([np.inf, -np.inf], np.nan)  # Replace inf/-inf with NaN\n        X_train[col] = X_train[col].fillna(-1)  # Fill NaN values with a constant (-1 or another value you prefer)\n\n    # # Normalize numeric columns (L2 normalization by default)\n    # X_train[numeric_cols] = normalize(X_train[numeric_cols])\n\n    # One-hot encode categorical columns\n    X_train_encoded = pd.get_dummies(X_train)\n\n    # Concatenate the encoded features with the target variable\n    data_with_target = np.concatenate([X_train_encoded, y_train.values.reshape(-1, 1)], axis=1)\n\n    # Apply KNN imputation\n    imputer = KNNImputer(n_neighbors=5)\n    imputed_data = imputer.fit_transform(data_with_target)\n\n    # Impute the target variable 'sii'\n    y_train_imputed = pd.Series(imputed_data[:, -1]).apply(lambda x: int(round(x, 0)))\n\n    print(\"KNN imputation and normalization completed.\")\n    return y_train_imputed","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from imblearn.over_sampling import SMOTE\n\ny = train_df['sii']  # Target variable\ncategorical_cols = train_df.drop(columns=['id', 'sii']).select_dtypes(include=['object']).columns\nnumeric_cols = train_df.drop(columns=['id', 'sii']).select_dtypes(include=['int64', 'float64']).columns\n\n# Concatenate train and test data to get the full set of columns\nprint(\"Concatenating train and test data for preprocessing...\")\nfull_data = pd.concat([train_df.drop(columns=['sii']), test_df], axis=0)\nfull_data_dummies = pd.get_dummies(full_data, columns=categorical_cols)\nall_columns = full_data_dummies.columns\nprint(\"Data concatenation and one-hot encoding complete.\")\n\n# Impute target variable 'sii' using KNN\ny_imputed = impute_target_knn(train_df.loc[:, train_df.columns != 'sii'], y)\n\n# Preprocess training data\nX_train, imputer_numeric, imputer_categorical, scaler = preprocess_data(\n    train_df.drop(columns=['id', 'sii']), numeric_cols, categorical_cols, all_columns=all_columns, train=True\n)\n\nif OVERSAMPLE_SMOTE:\n    # Apply SMOTE to handle class imbalance\n    print(\"Applying SMOTE for oversampling...\")\n    smote = SMOTE(random_state=1)\n    X_train, y_imputed = smote.fit_resample(X_train, y_imputed)\n    print(\"SMOTE oversampling complete.\")\n\n# Initialize cross-validation\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_imputed.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import optuna\nimport pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import cohen_kappa_score, make_scorer\nfrom scipy.optimize import minimize\n\n# Define a function to apply rounding with optimized thresholds\ndef round_with_thresholds(raw_preds, thresholds):\n    return np.where(raw_preds < thresholds[0], 0,\n                    np.where(raw_preds < thresholds[1], 1,\n                             np.where(raw_preds < thresholds[2], 2, 3)))\n\n# Define the optimization function for tuning thresholds using Optuna\ndef optimize_thresholds(y_true, raw_preds):\n    def objective(trial):\n        thresholds = [\n            trial.suggest_float('threshold_0', 0.0, 1.0),\n            trial.suggest_float('threshold_1', 1.0, 2.0),\n            trial.suggest_float('threshold_2', 2.0, 3.0)\n        ]\n        rounded_preds = round_with_thresholds(raw_preds, thresholds)\n        return -cohen_kappa_score(y_true, rounded_preds, weights='quadratic')\n\n    study = optuna.create_study(direction='minimize')\n    study.optimize(objective, n_trials=50)\n    return [study.best_params['threshold_0'], study.best_params['threshold_1'], study.best_params['threshold_2']]\n\n# Define data preprocessing to handle inf, NaN, and extreme values\ndef preprocess_data(X):\n    # Replace inf values with NaN\n    X = X.replace([np.inf, -np.inf], np.nan)\n    # Fill NaN values with the median of each column\n    X = X.fillna(X.median())\n    # Clip excessively large or small values to a reasonable range\n    X = X.clip(-1e5, 1e5)\n    return X\n\n# Define the objective function for Optuna to optimize LightGBM hyperparameters\ndef objective(trial):\n    params = {\n        'objective': 'regression',\n        'metric': 'rmse',\n        'boosting_type': 'gbdt',\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3),\n        'n_estimators': trial.suggest_int('n_estimators', 100, 1000),\n        'max_depth': trial.suggest_int('max_depth', 3, 12),\n        'num_leaves': trial.suggest_int('num_leaves', 20, 150),\n        'min_child_samples': trial.suggest_int('min_child_samples', 5, 100),\n        'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0)\n    }\n\n    # Initialize LightGBM model with the suggested parameters\n    model = lgb.LGBMRegressor(**params)\n\n    # Preprocess the training data\n    X_train_clean = preprocess_data(X_train.copy())\n    y_train_clean = y_train.copy()\n\n    # Split data for cross-validation\n    X_tr, X_val, y_tr, y_val = train_test_split(X_train_clean, y_train_clean, test_size=0.2, random_state=42)\n\n    # Fit the model\n    model.fit(X_tr, y_tr)\n\n    # Predict on the validation set\n    y_val_pred = model.predict(X_val)\n\n    # Optimize thresholds for the validation set\n    optimized_thresholds = optimize_thresholds(y_val, y_val_pred)\n    y_val_pred_rounded = round_with_thresholds(y_val_pred, optimized_thresholds)\n\n    # Calculate Kappa score\n    kappa = cohen_kappa_score(y_val, y_val_pred_rounded, weights='quadratic')\n\n    # Return negative kappa score for minimization\n    return -kappa\n\n# Split the data into training and test sets\nX_train, X_test, y_train, y_test = train_test_split(X_train, y_imputed, test_size=0.2, random_state=42)\n\n# Create an Optuna study and optimize the hyperparameters\nstudy = optuna.create_study(direction=\"minimize\")\nstudy.optimize(objective, n_trials=100)\n\n# Get the best parameters\nbest_params = study.best_params\nprint(\"Best parameters:\", best_params)\n\n# Re-train the model with the best parameters\nbest_model = lgb.LGBMRegressor(**best_params)\n\n# Preprocess the training data again before fitting\nX_train_clean = preprocess_data(X_train.copy())\nX_test_clean = preprocess_data(X_test.copy())\n\n# Fit the model on the entire training set\nbest_model.fit(X_train_clean, y_train)\n\n# Predict on the test set\ny_pred = best_model.predict(X_test_clean)\n\n# Optimize the thresholds for rounding\ny_optimized_thresholds = optimize_thresholds(y_test, y_pred)\ny_pred_rounded = round_with_thresholds(y_pred, y_optimized_thresholds)\n\n# Evaluate the model on the test data using Kappa score\nkappa = cohen_kappa_score(y_test, y_pred_rounded, weights='quadratic')\nprint(f\"Test Kappa Score: {kappa}\")\n\n# Create a submission DataFrame\nsubmission = pd.DataFrame({'Id': range(len(y_pred_rounded)), 'Target': y_pred_rounded})\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission file created: submission.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}