{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.impute import KNNImputer\nfrom sklearn.linear_model import LinearRegression\nfrom scipy.stats import norm\n\n# Load the data\ntrain = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\n\n# Initialize KNN Imputer for numeric features\nknn_imputer = KNNImputer(n_neighbors=5)\n\n# Helper function to convert weight (lbs) to kg and height (in) to cm\ndef convert_units(weight_lbs, height_in):\n    weight_kg = weight_lbs * 0.453592 if pd.notnull(weight_lbs) else np.nan\n    height_cm = height_in * 2.54 if pd.notnull(height_in) else np.nan\n    return weight_kg, height_cm\n\n# Helper function to calculate BMI\ndef calculate_bmi(weight_kg, height_cm):\n    if pd.notnull(weight_kg) and pd.notnull(height_cm) and height_cm > 0:\n        return weight_kg / ((height_cm / 100) ** 2)\n    return np.nan\n\n# Derived Features: BMR Calculation (Harris-Benedict Formula)\ndef calculate_bmr(weight_kg, height_cm, age, sex):\n    if pd.notnull(weight_kg) and pd.notnull(height_cm) and pd.notnull(age) and pd.notnull(sex):\n        if sex == 0:  # Male\n            return 10 * weight_kg + 6.25 * height_cm - 5 * age + 5\n        elif sex == 1:  # Female\n            return 10 * weight_kg + 6.25 * height_cm - 5 * age - 161\n    return np.nan\n\n# Convert Weight and Height to proper units\ntrain[['Physical-Weight', 'Physical-Height']] = train.apply(\n    lambda row: pd.Series(convert_units(row['Physical-Weight'], row['Physical-Height'])), axis=1\n)\n\n# Fill BMI using Weight and Height\ntrain['Physical-BMI'] = train.apply(lambda row: calculate_bmi(row['Physical-Weight'], row['Physical-Height']), axis=1)\n\n# Impute missing Weight and Height using group medians (Age, Sex), fallback to overall median\nweight_median = train['Physical-Weight'].median()\nheight_median = train['Physical-Height'].median()\ntrain['Physical-Weight'] = train['Physical-Weight'].fillna(\n    train.groupby(['Basic_Demos-Age', 'Basic_Demos-Sex'])['Physical-Weight'].transform('median')\n).fillna(weight_median)\ntrain['Physical-Height'] = train['Physical-Height'].fillna(\n    train.groupby(['Basic_Demos-Age', 'Basic_Demos-Sex'])['Physical-Height'].transform('median')\n).fillna(height_median)\n\n# Calculate BMR\ntrain['BIA-BIA_BMR'] = train.apply(lambda row: calculate_bmr(row['Physical-Weight'], row['Physical-Height'], row['Basic_Demos-Age'], row['Basic_Demos-Sex']), axis=1)\n\n# KNN Imputation for remaining numerical columns\nnumerical_columns = train.select_dtypes(include=[np.number]).columns\ntrain[numerical_columns] = knn_imputer.fit_transform(train[numerical_columns])\n\n# Fill missing values using correlation-based regression where needed\ndef fill_missing_regression(df, target, predictors):\n    missing_idx = df[target].isnull()\n    if missing_idx.any():\n        model = LinearRegression()\n        df_notnull = df.dropna(subset=[target])\n        model.fit(df_notnull[predictors], df_notnull[target])\n        df.loc[missing_idx, target] = model.predict(df.loc[missing_idx, predictors])\n    return df\n\n# Use regression to fill 'PreInt_EduHx-computerinternet_hoursday' if still missing\ntrain = fill_missing_regression(train, 'PreInt_EduHx-computerinternet_hoursday', ['Basic_Demos-Age', 'Basic_Demos-Sex', 'Physical-BMI'])\n\n# Safeguard for division by zero or extremely small values\ndef safe_divide(numerator, denominator):\n    return numerator / denominator if pd.notnull(denominator) and denominator > 1e-6 else np.nan\n\n# Add derived features and explicitly handle remaining nulls\ntrain['BMI_Age'] = train['Physical-BMI'] * train['Basic_Demos-Age']\ntrain['BMI_Internet_Hours'] = train.apply(\n    lambda row: safe_divide(row['Physical-BMI'] * row['PreInt_EduHx-computerinternet_hoursday'], 1), axis=1)\ntrain['Muscle_to_Fat'] = train.apply(lambda row: safe_divide(row['BIA-BIA_SMM'], row['BIA-BIA_Fat']), axis=1)\ntrain['Hydration_Status'] = train.apply(lambda row: safe_divide(row['BIA-BIA_TBW'], row['Physical-Weight']), axis=1)\n\n# Use distribution-based imputation for Muscle_to_Fat and Hydration_Status\nmuscle_mean, muscle_std = train['Muscle_to_Fat'].mean(), train['Muscle_to_Fat'].std()\ntrain['Muscle_to_Fat'] = train['Muscle_to_Fat'].fillna(np.random.normal(muscle_mean, muscle_std))\n\nhydration_mean, hydration_std = train['Hydration_Status'].mean(), train['Hydration_Status'].std()\ntrain['Hydration_Status'] = train['Hydration_Status'].fillna(np.random.normal(hydration_mean, hydration_std))\n\n# Fill categorical missing values with 'Unknown'\ncategorical_columns = train.select_dtypes(include=['object']).columns\ntrain[categorical_columns] = train[categorical_columns].fillna('Unknown')\n\n# Final check for remaining NaNs\nremaining_nans = train.isnull().sum()\nprint(\"Remaining NaN values in the dataset:\")\nprint(remaining_nans[remaining_nans > 0])\n\n# Save the processed file\ntrain.to_csv(\"trainV1.csv\", index=False)\nprint(\"File 'trainV1.csv' has been created with missing values filled, units converted, and derived features added!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T17:15:01.981988Z","iopub.execute_input":"2024-12-19T17:15:01.982579Z","iopub.status.idle":"2024-12-19T17:15:10.642315Z","shell.execute_reply.started":"2024-12-19T17:15:01.982524Z","shell.execute_reply":"2024-12-19T17:15:10.641389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import KNNImputer\nfrom scipy.fftpack import fft\nfrom scipy.stats import skew, kurtosis\n\nclass LSTMAutoEncoder(nn.Module):\n    def __init__(self, input_dim, hidden_dim=32):\n        super(LSTMAutoEncoder, self).__init__()\n        self.encoder = nn.LSTM(input_dim, hidden_dim, batch_first=True)\n        self.decoder = nn.LSTM(hidden_dim, input_dim, batch_first=True)\n\n    def forward(self, x):\n        _, (hidden, _) = self.encoder(x)\n        # Reshape hidden from (1, batch_size, hidden_dim) to (batch_size, 1, hidden_dim)\n        hidden = hidden.permute(1, 0, 2)\n        reconstructed, _ = self.decoder(hidden)\n        return reconstructed\n\ndef process_file(filename, dirname):\n    \"\"\"Enhanced feature extraction per file with sliding window, FFT, and advanced stats\"\"\"\n    df = pd.read_parquet(os.path.join(dirname, filename))\n    df.drop('step', axis=1, inplace=True, errors='ignore')\n\n    features = []\n    for col in ['X', 'Y', 'Z', 'enmo', 'anglez', 'light', 'battery_voltage']:\n        if col in df.columns:\n            # Statistical Features\n            features += [df[col].mean(), df[col].std(), df[col].min(), df[col].max(), df[col].median()]\n            features += [skew(df[col].fillna(0)), kurtosis(df[col].fillna(0))]\n\n            # Sliding Window Features\n            rolling = df[col].rolling(window=100, min_periods=1)\n            features += [rolling.mean().iloc[-1], rolling.std().iloc[-1]]\n\n            # Frequency Domain Features\n            # Convert to numpy array before applying FFT\n            data = df[col].fillna(0).values\n            freq = np.abs(fft(data))[:100]  # Top 100 frequencies\n            features += freq.tolist()\n\n    return features, filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    \"\"\"Efficient parallel loading and feature extraction for time-series data.\"\"\"\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n\n    stats, indexes = zip(*results)\n    df = pd.DataFrame(stats)\n    df.columns = [f'stat_{i}' for i in range(df.shape[1])]\n    df['id'] = indexes\n    return df\n\ndef ensure_all_features(df, feature_list):\n    \"\"\"\n    Ensure all required features in the feature_list exist in the DataFrame.\n    If a feature is missing, it will be added with a default value (e.g., NaN).\n    \"\"\"\n    for feature in feature_list:\n        if feature not in df.columns:\n            print(f\"Adding missing feature: {feature}\")\n            df[feature] = np.nan  # Default value can be set based on feature type\n    return df\n    \ndef perform_autoencoder(df, hidden_dim=60, epochs=100, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    data_tensor = torch.FloatTensor(df_scaled).unsqueeze(1)  # Shape: (batch_size, 1, input_dim)\n\n    input_dim = data_tensor.shape[2]\n    autoencoder = LSTMAutoEncoder(input_dim, hidden_dim)\n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters(), lr=1e-3)\n\n    for epoch in range(epochs):\n        total_loss = 0\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            total_loss += loss.item()\n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {total_loss / len(data_tensor):.4f}')\n\n    with torch.no_grad():\n        _, (hidden, _) = autoencoder.encoder(data_tensor)\n        # hidden shape: (1, 996, hidden_dim)\n        # Squeeze the first dimension to get (996, hidden_dim)\n        encoded_data = hidden.squeeze(0).numpy()\n\n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i}' for i in range(encoded_data.shape[1])])\n    return df_encoded\n\ndef ensure_all_features(df, feature_list):\n    \"\"\"\n    Ensure all required features in the feature_list exist in the DataFrame.\n    If a feature is missing, it will be added with a default value (e.g., NaN).\n    \"\"\"\n    for feature in feature_list:\n        if feature not in df.columns:\n            print(f\"Adding missing feature: {feature}\")\n            df[feature] = np.nan  # Default value can be set based on feature type\n    return df\n\n\ndef feature_engineering(df):\n    \"\"\"Enhanced feature engineering with additional ratios and transformations.\"\"\"\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1, errors='ignore')\n\n    # Existing Interaction Features\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n\n    # New Features\n    df['BMI_HeartRate'] = df['Physical-BMI'] * df['Physical-HeartRate']\n    df['Weight_to_Height'] = df['Physical-Weight'] / (df['Physical-Height'] + 1e-3)\n    df['BMI_Squared'] = df['Physical-BMI'] ** 2\n    df['Fat_Free_Ratio'] = df['BIA-BIA_FFM'] / (df['Physical-Weight'] + 1e-3)\n    df['Hydration_to_BMI'] = df['BIA-BIA_TBW'] / (df['Physical-BMI'] + 1e-3)\n    df['HeartRate_Age'] = df['Physical-HeartRate'] / (df['Basic_Demos-Age'] + 1e-3)\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T17:15:10.643915Z","iopub.execute_input":"2024-12-19T17:15:10.644281Z","iopub.status.idle":"2024-12-19T17:15:14.308747Z","shell.execute_reply.started":"2024-12-19T17:15:10.644242Z","shell.execute_reply":"2024-12-19T17:15:14.307822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.impute import KNNImputer\n\n# Load your data\ntrain = pd.read_csv('/kaggle/working/trainV1.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train, hidden_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, hidden_dim=60, epochs=100, batch_size=32)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\n\n# Add back 'id' columns to the encoded data\ntrain_ts_encoded[\"id\"] = train_ts[\"id\"]\ntest_ts_encoded[\"id\"] = test_ts[\"id\"]\n\n# Merge the encoded time series features with train and test\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\n\n# Impute missing numeric data in train\nimputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\n\n# Round and ensure 'sii' is int\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\n\n# Reattach any non-numeric columns from the original train\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n\ntrain = train_imputed\n\n# Apply feature engineering\ntrain = feature_engineering(train)\ntest = feature_engineering(test)\n\n# Drop rows in train that have too many NAs\ntrain = train.dropna(thresh=10, axis=0)\n\n# These are the features we want to retain\nfeaturesCols = [\n    'id', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-CGAS_Score', 'Physical-BMI',\n    'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n    'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n    'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n    'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone',\n    'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone',\n    'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n    'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n    'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n    'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n    'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total',\n    'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T',\n    'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age', 'Internet_Hours_Age', 'BMI_Internet_Hours',\n    'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n    'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW',\n    'BMI_HeartRate', 'Weight_to_Height', 'BMI_Squared', 'Fat_Free_Ratio', 'Hydration_to_BMI', 'HeartRate_Age'\n]\n\nfeaturesCols += time_series_cols\n\n# Keep only the desired columns in train\ntrain = train[featuresCols]\n\n# Ensure 'sii' is present and drop rows without it\ntrain = train.dropna(subset=['sii'])\n\n# For test, we do not have 'sii', so exclude it\nfeaturesCols_test = [col for col in featuresCols if col != 'sii']\ntest = test[featuresCols_test]\n\n# Save the final train dataset, including 'id' and 'sii'\ntrain.to_csv('trainV2.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T17:15:14.309998Z","iopub.execute_input":"2024-12-19T17:15:14.310544Z","iopub.status.idle":"2024-12-19T17:19:39.553890Z","shell.execute_reply.started":"2024-12-19T17:15:14.310503Z","shell.execute_reply":"2024-12-19T17:19:39.553242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics import cohen_kappa_score\n\n# Handling infinities in the DataFrame\ntrain = train.replace([np.inf, -np.inf], np.nan)\n\n# Ensure 'train' is a pandas DataFrame\n# train = pd.DataFrame(...)  # Uncomment and modify if necessary\n\n# Function to calculate quadratic weighted kappa\ndef quadratic_weighted_kappa(y_true: np.ndarray, y_pred: np.ndarray) -> float:\n    \"\"\"\n    Calculate the quadratic weighted kappa score.\n    \n    Parameters:\n    y_true (np.ndarray): True labels.\n    y_pred (np.ndarray): Predicted labels.\n    \n    Returns:\n    float: Quadratic weighted kappa score.\n    \"\"\"\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n# Function to round predictions based on thresholds\ndef threshold_Rounder(oof_non_rounded: np.ndarray, thresholds: np.ndarray) -> np.ndarray:\n    \"\"\"\n    Round continuous predictions to discrete classes based on thresholds.\n    \n    Parameters:\n    oof_non_rounded (np.ndarray): Continuous predictions.\n    thresholds (np.ndarray): Array of threshold values.\n    \n    Returns:\n    np.ndarray: Discrete class predictions.\n    \"\"\"\n    thresholds = np.sort(thresholds)\n    return np.searchsorted(thresholds, oof_non_rounded)\n\n# Function to evaluate predictions and return negative kappa score\ndef evaluate_predictions(thresholds: np.ndarray, y_true: np.ndarray, oof_non_rounded: np.ndarray) -> float:\n    \"\"\"\n    Evaluate predictions using threshold rounding and return negative quadratic weighted kappa.\n    \n    Parameters:\n    thresholds (np.ndarray): Array of threshold values.\n    y_true (np.ndarray): True labels.\n    oof_non_rounded (np.ndarray): Continuous predictions.\n    \n    Returns:\n    float: Negative quadratic weighted kappa score.\n    \"\"\"\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T17:19:39.555227Z","iopub.execute_input":"2024-12-19T17:19:39.555573Z","iopub.status.idle":"2024-12-19T17:19:39.566833Z","shell.execute_reply.started":"2024-12-19T17:19:39.555543Z","shell.execute_reply":"2024-12-19T17:19:39.566030Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n# If 'id' is present, remove it before training\n    if 'id' in X.columns:\n        X = X.drop('id', axis=1)\n    if 'id' in test_data.columns:\n        test_data = test_data.drop('id', axis=1)\n        \n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T17:19:39.569899Z","iopub.execute_input":"2024-12-19T17:19:39.570263Z","iopub.status.idle":"2024-12-19T17:19:39.581778Z","shell.execute_reply.started":"2024-12-19T17:19:39.570213Z","shell.execute_reply":"2024-12-19T17:19:39.580855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42\nn_splits = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T17:19:39.582822Z","iopub.execute_input":"2024-12-19T17:19:39.583094Z","iopub.status.idle":"2024-12-19T17:19:39.597574Z","shell.execute_reply.started":"2024-12-19T17:19:39.583050Z","shell.execute_reply":"2024-12-19T17:19:39.596814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n    'device': 'gpu'\n\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n    'tree_method': 'gpu_hist',\n\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  # Increase this value\n    'task_type': 'GPU'\n\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T17:19:39.598539Z","iopub.execute_input":"2024-12-19T17:19:39.598863Z","iopub.status.idle":"2024-12-19T17:19:39.610848Z","shell.execute_reply.started":"2024-12-19T17:19:39.598830Z","shell.execute_reply":"2024-12-19T17:19:39.610112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\nfrom sklearn.model_selection import KFold\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom pytorch_tabnet.tab_model import TabNetRegressor\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T17:21:35.984303Z","iopub.execute_input":"2024-12-19T17:21:35.984703Z","iopub.status.idle":"2024-12-19T17:21:37.258198Z","shell.execute_reply.started":"2024-12-19T17:21:35.984666Z","shell.execute_reply":"2024-12-19T17:21:37.257501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom pytorch_tabnet.callbacks import Callback\nfrom pytorch_tabnet.tab_model import TabNetRegressor\nfrom copy import deepcopy\nimport os\nimport torch\nfrom pytorch_tabnet.callbacks import Callback\n\nclass TabNetPretrainedModelCheckpoint(Callback):\n    def __init__(self, filepath, monitor='val_loss', mode='min', \n                 save_best_only=True, verbose=1):\n        super().__init__()  # Initialize parent class\n        self.filepath = filepath\n        self.monitor = monitor\n        self.mode = mode\n        self.save_best_only = save_best_only\n        self.verbose = verbose\n        self.best = float('inf') if mode == 'min' else -float('inf')\n        \n    def on_train_begin(self, logs=None):\n        self.model = self.trainer  # Use trainer itself as model\n        \n    def on_epoch_end(self, epoch, logs=None):\n        logs = logs or {}\n        current = logs.get(self.monitor)\n        if current is None:\n            return\n        \n        if (self.mode == 'min' and current < self.best) or \\\n           (self.mode == 'max' and current > self.best):\n            if self.verbose:\n                print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n            self.best = current\n            if self.save_best_only:\n                self.model.save_model(self.filepath)\n\nclass TabNetWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, **kwargs):\n        self.model = TabNetRegressor(**kwargs)\n        self.kwargs = kwargs\n        self.imputer = SimpleImputer(strategy='median')\n        self.best_model_path = 'best_tabnet_model.pt'\n        \n    def fit(self, X, y):\n        # Handle missing values\n        X_imputed = self.imputer.fit_transform(X)\n        \n        if hasattr(y, 'values'):\n            y = y.values\n            \n        X_train, X_valid, y_train, y_valid = train_test_split(\n            X_imputed, y, test_size=0.2, random_state=42\n        )\n        \n        self.model.fit(\n            X_train=X_train,\n            y_train=y_train.reshape(-1, 1),\n            eval_set=[(X_valid, y_valid.reshape(-1, 1))],\n            eval_name=['valid'],\n            eval_metric=['mse'],\n            max_epochs=500,\n            patience=50,\n            batch_size=1024,\n            virtual_batch_size=128,\n            num_workers=0,\n            drop_last=False,\n            callbacks=[\n                TabNetPretrainedModelCheckpoint(\n                    filepath=self.best_model_path,\n                    monitor='valid_mse',\n                    mode='min',\n                    save_best_only=True,\n                    verbose=True\n                )\n            ]\n        )\n        \n        if os.path.exists(self.best_model_path):\n            self.model.load_model(self.best_model_path)\n            os.remove(self.best_model_path)\n        \n        return self\n    \n    def predict(self, X):\n        X_imputed = self.imputer.transform(X)\n        return self.model.predict(X_imputed).flatten()\n    \n    def __deepcopy__(self, memo):\n        cls = self.__class__\n        result = cls.__new__(cls)\n        memo[id(self)] = result\n        for k, v in self.__dict__.items():\n            setattr(result, k, deepcopy(v, memo))\n        return result\n\n# Make sure TabNet_Params, SEED, Params, XGB_Params, CatBoost_Params, and TrainML are defined above\nTabNet_Params = {\n    'n_d': 64,\n    'n_a': 64,\n    'n_steps': 5,\n    'gamma': 1.5,\n    'n_independent': 2,\n    'n_shared': 2,\n    'lambda_sparse': 1e-4,\n    'optimizer_fn': torch.optim.Adam,\n    'optimizer_params': dict(lr=2e-2, weight_decay=1e-5),\n    'mask_type': 'entmax',\n    'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n    'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n    'verbose': 1,\n    'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n}\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T17:21:39.891767Z","iopub.execute_input":"2024-12-19T17:21:39.892549Z","iopub.status.idle":"2024-12-19T17:21:39.906503Z","shell.execute_reply.started":"2024-12-19T17:21:39.892510Z","shell.execute_reply":"2024-12-19T17:21:39.905529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params) # New","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T17:21:46.570114Z","iopub.execute_input":"2024-12-19T17:21:46.570485Z","iopub.status.idle":"2024-12-19T17:21:46.593230Z","shell.execute_reply.started":"2024-12-19T17:21:46.570438Z","shell.execute_reply":"2024-12-19T17:21:46.592420Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"voting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ('tabnet', TabNet_Model)\n])\n\nSubmission1 = TrainML(voting_model, test)\n\nSubmission1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T17:21:49.278439Z","iopub.execute_input":"2024-12-19T17:21:49.279322Z","iopub.status.idle":"2024-12-19T17:24:27.803101Z","shell.execute_reply.started":"2024-12-19T17:21:49.279289Z","shell.execute_reply":"2024-12-19T17:24:27.802132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport numpy as np\nfrom tqdm import tqdm\nfrom concurrent.futures import ThreadPoolExecutor\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.ensemble import VotingRegressor\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.base import clone\nfrom scipy.optimize import minimize\n\n# File loading functions\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\n\ndef load_time_series(dirname):\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    stats, indexes = zip(*results)\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\n\n# Load datasets\ntrain = pd.read_csv('/kaggle/working/trainV1.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\n# Load and merge new features from trainv2.csv\ntrain_v2 = pd.read_csv('/kaggle/working/trainV2.csv')\n\nnew_features = [\n    'BMI_Age', 'Internet_Hours_Age', 'BMI_Internet_Hours',\n    'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n    'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW',\n    'BMI_HeartRate', 'Weight_to_Height', 'BMI_Squared', 'Fat_Free_Ratio', 'Hydration_to_BMI', 'HeartRate_Age'\n]\n\n# Determine which of the new features are actually in train_v2\navailable_new_features = [f for f in new_features if f in train_v2.columns]\nmissing_new_features = [f for f in new_features if f not in train_v2.columns]\n\n# Merge available features\ntrain = pd.merge(train, train_v2[['id'] + available_new_features], how=\"left\", on=\"id\")\n\n# Ensure test is still a DataFrame\ntest = pd.DataFrame(test)  \n\n# Fill missing features with 0 in train if they are not merged\nfor feature in missing_new_features:\n    if feature not in train.columns:\n        train[feature] = 0\n\n# Fill missing features with 0 in test\nfor feature in new_features:\n    if feature not in test.columns:\n        test[feature] = 0\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols = [\n    'Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n    'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n    'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n    'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n    'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n    'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n    'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n    'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n    'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n    'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n    'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n    'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n    'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n    'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n    'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n    'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n    'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n    'PreInt_EduHx-computerinternet_hoursday', 'sii'\n]\n\nfeaturesCols += time_series_cols\nfeaturesCols += new_features\n\n# Before selecting columns, ensure all new_features exist in train\nfor feature in new_features:\n    if feature not in train.columns:\n        train[feature] = 0\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset=['sii'])\n\ncat_c = [\n    'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season',\n    'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season',\n    'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season'\n]\n\ndef update(df):\n    for c in cat_c:\n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category').cat.codes\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ntrain.replace([np.inf, -np.inf], np.nan, inplace=True)\ntrain.fillna(0, inplace=True)\n\ntest.replace([np.inf, -np.inf], np.nan, inplace=True)\ntest.fillna(0, inplace=True)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    y_true = np.round(y_true).astype(int)\n    y_pred = np.round(y_pred).astype(int)\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef optimize_thresholds(y_true, preds):\n    def score(thresholds):\n        rounded_preds = np.digitize(preds, thresholds)\n        return -quadratic_weighted_kappa(y_true, rounded_preds)\n\n    initial_thresholds = [0.5, 1.5, 2.5]\n    result = minimize(score, initial_thresholds, method='Nelder-Mead')\n    return result.x\n\ndef TrainML(model_class, test_data, n_splits=5, SEED=42):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = KFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    oof_preds = np.zeros(len(y))\n    test_preds = np.zeros(len(test_data))\n\n    for fold, (train_idx, val_idx) in enumerate(SKF.split(X, y)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        oof_preds[val_idx] = model.predict(X_val)\n        test_preds += model.predict(test_data) / n_splits\n\n        print(f\"Fold {fold + 1} QWK: {quadratic_weighted_kappa(y_val, oof_preds[val_idx]):.4f}\")\n\n    thresholds = optimize_thresholds(y, oof_preds)\n    print(f\"Optimized QWK: {quadratic_weighted_kappa(y, np.digitize(oof_preds, thresholds)):.4f}\")\n\n    final_preds = np.digitize(test_preds, thresholds)\n\n    submission = pd.DataFrame({'id': sample['id'], 'sii': final_preds})\n    return submission\n\nParams = {\n    'learning_rate': 0.046, 'max_depth': 12, 'num_leaves': 478, 'min_data_in_leaf': 13,\n    'feature_fraction': 0.893, 'bagging_fraction': 0.784, 'bagging_freq': 4,\n    'lambda_l1': 10, 'lambda_l2': 0.01\n}\n\nXGB_Params = {\n    'learning_rate': 0.05, 'max_depth': 6, 'n_estimators': 200,\n    'subsample': 0.8, 'colsample_bytree': 0.8, 'reg_alpha': 1,\n    'reg_lambda': 5, 'random_state': 42\n}\n\nCatBoost_Params = {\n    'learning_rate': 0.05, 'depth': 6, 'iterations': 200,\n    'random_seed': 42, 'verbose': 0, 'l2_leaf_reg': 10\n}\n\nLight = LGBMRegressor(**Params, random_state=42, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n\nSubmission2 = TrainML(voting_model, test)\n\n# Uncomment to save\n# Submission2.to_csv('submission.csv', index=False)\nSubmission2\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T17:24:33.053150Z","iopub.execute_input":"2024-12-19T17:24:33.053522Z","iopub.status.idle":"2024-12-19T17:26:20.257861Z","shell.execute_reply.started":"2024-12-19T17:24:33.053492Z","shell.execute_reply":"2024-12-19T17:26:20.256896Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport numpy as np\nfrom tqdm import tqdm\nfrom concurrent.futures import ThreadPoolExecutor\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.base import clone\nfrom scipy.optimize import minimize\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\n\n# Make sure to define or import these somewhere in your code:\n# - load_time_series function\n# - TabNetWrapper and TabNet_Params if they are still needed\n# - clear_output if required for progress bar updates (you can comment it out if not)\n# If `Fore`, `Style` from colorama are used, ensure they are imported or remove them.\n\n# Assuming load_time_series is defined as in your original code\ndef load_time_series(dirname):\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    stats, indexes = zip(*results)\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\n# Loading data\ntrain = pd.read_csv('/kaggle/working/trainV1.csv')\ntrain_v2 = pd.read_csv('/kaggle/working/trainV2.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n# Define new features\nnew_features = [\n    'BMI_Age', 'Internet_Hours_Age', 'BMI_Internet_Hours',\n    'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n    'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW',\n    'BMI_HeartRate', 'Weight_to_Height', 'BMI_Squared', 'Fat_Free_Ratio', 'Hydration_to_BMI', 'HeartRate_Age'\n]\n\nfeaturesCols = [\n    'Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n    'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n    'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n    'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n    'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n    'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n    'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n    'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n    'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n    'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n    'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n    'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n    'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n    'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n    'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n    'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n    'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n    'PreInt_EduHx-computerinternet_hoursday', 'sii'\n]\n\ncat_c = [\n    'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season',\n    'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season',\n    'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season'\n]\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\n# Merge train with time series\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\n\n# Determine which of the new features are actually available\navailable_new_features = [f for f in new_features if f in train_v2.columns]\nmissing_new_features = [f for f in new_features if f not in train_v2.columns]\n\n# Merge train with available features\ntrain = pd.merge(train, train_v2[['id'] + available_new_features], how=\"left\", on='id')\n\n# For missing new features, create them in train with NaN or zero\nfor mf in missing_new_features:\n    if mf not in train.columns:\n        train[mf] = np.nan\n\n# Merge test with time series\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\n# Ensure test has all new features\nfor nf in new_features:\n    if nf not in test.columns:\n        test[nf] = np.nan\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\n# Add time series and new features to featuresCols\nfeaturesCols += time_series_cols\nfeaturesCols += new_features\n\n# Ensure all columns are present in train, fill missing if needed\nfor col in featuresCols:\n    if col not in train.columns:\n        train[col] = np.nan\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset=['sii'])\n\ntrain['sii'] = train['sii'].round(0).astype(int)\n\ndef update(df):\n    for c in cat_c:\n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\n# Replace infinities with NaN\ntrain = train.replace([np.inf, -np.inf], np.nan)\ntest = test.replace([np.inf, -np.inf], np.nan)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SEED = 42\n    n_splits = 5\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    # If you want a progress bar, make sure to comment out or remove `clear_output(wait=True)` \n    # if `clear_output` isn't defined.\n    for fold, (train_idx, val_idx) in enumerate(SKF.split(X, y)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_val_pred = model.predict(X_val)\n        oof_non_rounded[val_idx] = y_val_pred\n        test_preds[:, fold] = model.predict(test_data)\n\n        train_kappa = quadratic_weighted_kappa(y_train, model.predict(X_train).round().astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred.round().astype(int))\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n\n    print(f\"Mean Train QWK --> {train_kappa:.4f}\")\n    print(f\"Mean Validation QWK ---> {val_kappa:.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n    print(f\"Optimized QWK: {tKappa:.3f}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    return tp_rounded\n\nimputer = SimpleImputer(strategy='median')\n\n# Ensure TabNetWrapper and TabNet_Params are defined or imported if you still need them.\n# Otherwise, remove the tabnet estimator.\n# For demonstration, let's assume you remove TabNet from the ensemble if it's not defined.\nensemble = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=42))])),\n    ('xgb', Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=42))])),\n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=42, silent=True))])),\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=42))])),\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=42))]))\n    # Add ('tabnet', ...) only if TabNetWrapper and TabNet_Params are defined\n])\n\nSubmission3 = TrainML(ensemble, test)\nSubmission3 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission3\n})\n\n# You can save the submission if needed:\n# Submission3.to_csv('submission.csv', index=False)\nSubmission3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T17:26:44.347287Z","iopub.execute_input":"2024-12-19T17:26:44.347732Z","iopub.status.idle":"2024-12-19T17:31:00.888223Z","shell.execute_reply.started":"2024-12-19T17:26:44.347697Z","shell.execute_reply":"2024-12-19T17:31:00.887189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub1 = Submission1\nsub2 = Submission2\nsub3 = Submission3\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii']\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\nfinal_submission.to_csv('submission.csv', index=False)\n\nprint(\"Majority voting completed and saved to 'Final_Submission.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T17:31:43.130149Z","iopub.execute_input":"2024-12-19T17:31:43.130564Z","iopub.status.idle":"2024-12-19T17:31:43.150844Z","shell.execute_reply.started":"2024-12-19T17:31:43.130529Z","shell.execute_reply":"2024-12-19T17:31:43.149725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T17:31:50.501222Z","iopub.execute_input":"2024-12-19T17:31:50.501996Z","iopub.status.idle":"2024-12-19T17:31:50.510739Z","shell.execute_reply.started":"2024-12-19T17:31:50.501959Z","shell.execute_reply":"2024-12-19T17:31:50.509789Z"}},"outputs":[],"execution_count":null}]}