{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom typing import Tuple, List, Dict\nfrom sklearn.impute import KNNImputer, SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.metrics import cohen_kappa_score\nimport optuna\nimport lightgbm as lgb\nimport xgboost as xgb\nimport catboost as cb\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import cohen_kappa_score\n\n# Data Manipulation and Analysis\nimport pandas as pd\nimport numpy as np\nimport math\n\n# Plotting and Visualization\nimport matplotlib.pyplot as plt\nfrom matplotlib.lines import Line2D\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.subplots as sp\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go\n\n# Machine Learning and Preprocessing\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import StratifiedKFold, KFold\nfrom sklearn.metrics import (\n    cohen_kappa_score, \n    classification_report, \n    confusion_matrix, \n    accuracy_score, \n    mean_squared_error\n)\nfrom sklearn.ensemble import VotingClassifier\n\n# Machine Learning Libraries\nfrom xgboost import XGBClassifier\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostClassifier\nfrom catboost import CatBoostRegressor\nfrom lightgbm import LGBMClassifier\nimport lightgbm as lgb\n\n# Optimization\nimport optuna\n\n# File and OS Utilities\nimport os\nfrom glob import glob\n\n# Suppress Warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nimport optuna\nimport lightgbm as lgb\nimport xgboost as xgb\nimport catboost as cb\nfrom collections import Counter\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import make_scorer\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nimport optuna\nimport numpy as np\nimport pandas as pd\n\nimport numpy as np\nimport pandas as pd\nfrom collections import Counter\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nimport lightgbm as lgb\nimport xgboost as xgb\nimport catboost as cb\nimport optuna\nfrom sklearn.ensemble import VotingClassifier\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:51:07.417650Z","iopub.execute_input":"2024-12-15T16:51:07.418277Z","iopub.status.idle":"2024-12-15T16:51:09.939000Z","shell.execute_reply.started":"2024-12-15T16:51:07.418227Z","shell.execute_reply":"2024-12-15T16:51:09.938313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TabularDataLoader:\n\n    def __init__(self, train_path: str, test_path: str, data_dict_path: str, target_column: str):\n        \"\"\"\n        Initialize the DataLoader with paths to training, test datasets, and data dictionary.\n\n        :param train_path: Path to the training dataset CSV file\n        :param test_path: Path to the test dataset CSV file\n        :param data_dict_path: Path to the data dictionary CSV file\n        :param target_column: The name of the target column\n        \"\"\"\n        self.train_path = train_path\n        self.test_path = test_path\n        self.data_dict_path = data_dict_path\n        self.target_column = target_column\n\n    def load_data(self) -> Tuple[pd.DataFrame, pd.DataFrame, List[str], Dict[str, List[str]]]:\n        \"\"\"\n        Load the training and test datasets, find common features, and categorize features.\n\n        :return: A tuple containing:\n            - Training dataset with common features\n            - Test dataset with common features\n            - List of common features\n            - Dictionary categorizing features into 'numerical', 'ordinal', and 'nominal'\n        \"\"\"\n        # Load datasets\n        train_df = pd.read_csv(self.train_path)\n        test_df = pd.read_csv(self.test_path)\n        data_dict = pd.read_csv(self.data_dict_path)\n\n        # Find common features\n        common_features = list(set(train_df.columns).intersection(set(test_df.columns)))\n\n        # Filter datasets to keep only common features\n        train_features = common_features + [self.target_column]\n        train_df = train_df[train_features]\n        test_df = test_df[common_features]\n\n        # Categorize features\n        feature_types = self._categorize_features(data_dict, common_features)\n\n        return train_df, test_df, common_features, feature_types\n\n    def _categorize_features(self, data_dict: pd.DataFrame, common_features: List[str]) -> Dict[str, List[str]]:\n        \"\"\"\n        Categorize features into numerical, ordinal, and nominal based on the data dictionary.\n\n        :param data_dict: Data dictionary as a pandas DataFrame\n        :param common_features: List of common features between train and test datasets\n        :return: Dictionary with keys 'numerical', 'ordinal', 'nominal' and corresponding feature lists\n        \"\"\"\n        # Initialize category lists\n        continuous = []\n        discrete = []\n        ordinal = []\n        nominal = []\n        id = []\n\n        # Iterate over data dictionary and categorize features\n        for _, row in data_dict.iterrows():\n            column = row['Field']\n            dtype = row['Type']\n\n            if column in common_features:\n                if dtype == 'float':\n                    continuous.append(column)\n                elif dtype == 'int':\n                    discrete.append(column)\n                elif dtype == 'categorical int':\n                    ordinal.append(column)\n                elif dtype == 'str':\n                    if column == 'id':\n                        id.append(column)\n                    else:\n                        nominal.append(column)\n\n        return {\n            'continuous': continuous,\n            'discrete': discrete,\n            'ordinal': ordinal,\n            'nominal': nominal,\n            'id': id\n        }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:51:09.940520Z","iopub.execute_input":"2024-12-15T16:51:09.940992Z","iopub.status.idle":"2024-12-15T16:51:09.951293Z","shell.execute_reply.started":"2024-12-15T16:51:09.940963Z","shell.execute_reply":"2024-12-15T16:51:09.950476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_path = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\ntest_path = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'\ndata_dict_path = '/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv'\nts_train_path = \"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\"\nts_test_path = \"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:51:09.952322Z","iopub.execute_input":"2024-12-15T16:51:09.952651Z","iopub.status.idle":"2024-12-15T16:51:09.962346Z","shell.execute_reply.started":"2024-12-15T16:51:09.952614Z","shell.execute_reply":"2024-12-15T16:51:09.961446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#loader = TabularDataLoader(train_path, test_path, data_dict_path, 'PCIAT-PCIAT_Total')\nloader = TabularDataLoader(train_path, test_path, data_dict_path, 'sii')\ntrain_data, test_data, common_features, feature_types = loader.load_data()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:51:09.964429Z","iopub.execute_input":"2024-12-15T16:51:09.964856Z","iopub.status.idle":"2024-12-15T16:51:10.018645Z","shell.execute_reply.started":"2024-12-15T16:51:09.964818Z","shell.execute_reply":"2024-12-15T16:51:10.017979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:51:10.019500Z","iopub.execute_input":"2024-12-15T16:51:10.019745Z","iopub.status.idle":"2024-12-15T16:51:10.025886Z","shell.execute_reply.started":"2024-12-15T16:51:10.019721Z","shell.execute_reply":"2024-12-15T16:51:10.024988Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = train_data[\ntrain_data['sii'].notnull()\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:51:10.026785Z","iopub.execute_input":"2024-12-15T16:51:10.027129Z","iopub.status.idle":"2024-12-15T16:51:10.033702Z","shell.execute_reply.started":"2024-12-15T16:51:10.027089Z","shell.execute_reply":"2024-12-15T16:51:10.032955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:51:10.034741Z","iopub.execute_input":"2024-12-15T16:51:10.034986Z","iopub.status.idle":"2024-12-15T16:51:10.041674Z","shell.execute_reply.started":"2024-12-15T16:51:10.034962Z","shell.execute_reply":"2024-12-15T16:51:10.040883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:51:10.162760Z","iopub.execute_input":"2024-12-15T16:51:10.162982Z","iopub.status.idle":"2024-12-15T16:51:10.167853Z","shell.execute_reply.started":"2024-12-15T16:51:10.162960Z","shell.execute_reply":"2024-12-15T16:51:10.167013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TabularDataPreprocessor:\n\n    def preprocess_data(self, train_df: pd.DataFrame, test_df: pd.DataFrame, feature_types: Dict[str, List[str]]) -> Tuple[pd.DataFrame, pd.DataFrame]:\n        \"\"\"Preprocess the training and test datasets by handling missing values, encoding categorical data,\n        creating interaction features, and scaling continuous features.\n\n        :param train_df: Training dataset\n        :param test_df: Test dataset\n        :param feature_types: Dictionary categorizing features into 'numerical', 'ordinal', and 'nominal'\n        :return: Tuple containing preprocessed training and test datasets\n        \"\"\"\n        # Handle missing values for continuous features using KNNImputer\n        if feature_types['continuous']:\n            continuous_imputer = KNNImputer(n_neighbors=5)\n            train_df[feature_types['continuous']] = continuous_imputer.fit_transform(train_df[feature_types['continuous']])\n            test_df[feature_types['continuous']] = continuous_imputer.transform(test_df[feature_types['continuous']])\n        \n        # Handle missing values for discrete features using median imputation (more suitable for integers)\n        if feature_types['discrete']:\n            discrete_imputer = SimpleImputer(strategy='median')  # Median to preserve integer nature\n            train_df[feature_types['discrete']] = discrete_imputer.fit_transform(train_df[feature_types['discrete']])\n            test_df[feature_types['discrete']] = discrete_imputer.transform(test_df[feature_types['discrete']])\n\n        # Handle missing values for ordinal features by filling with -1\n        for col in feature_types['ordinal']:\n            train_df[col] = train_df[col].fillna(-1)\n            test_df[col] = test_df[col].fillna(-1)\n    \n        # Handle missing values for nominal features by filling with \"missing\"\n        for col in feature_types['nominal']:\n            train_df[col] = train_df[col].fillna(\"missing\")\n            test_df[col] = test_df[col].fillna(\"missing\")\n    \n        # Apply Custom Label Encoding to nominal features\n        custom_encoding = {'Fall': 3, 'Spring': 1, 'Summer': 2, 'Winter': 0, 'missing': -1}\n        for col in feature_types['nominal']:\n            train_df[col] = train_df[col].map(custom_encoding).astype(int)\n            test_df[col] = test_df[col].map(custom_encoding).astype(int)\n\n        # Create Interaction Features\n        self._create_interaction_features(train_df, test_df)\n\n        # Scale Continuous Features\n        if feature_types['continuous']:\n            scaler = StandardScaler()\n            train_df[feature_types['continuous']] = scaler.fit_transform(train_df[feature_types['continuous']])\n            test_df[feature_types['continuous']] = scaler.transform(test_df[feature_types['continuous']])\n\n        return train_df, test_df\n\n    def _create_interaction_features(self, train_df: pd.DataFrame, test_df: pd.DataFrame):\n        \"\"\"\n        Create interaction features by combining relevant features.\n        \"\"\"\n        interaction_features = [\n            ('Physical-BMI', 'Physical-Waist_Circumference', 'BMI_Waist_Ratio'),\n            ('Physical-HeartRate', 'Fitness_Endurance-Time_Sec', 'HeartRate_Endurance_Product'),\n            ('BIA-BIA_Fat', 'Physical-BMI', 'Fat_BMI_Ratio')\n        ]\n\n        for feature1, feature2, new_feature in interaction_features:\n            if feature1 in train_df.columns and feature2 in train_df.columns:\n                train_df[new_feature] = train_df[feature1] / (train_df[feature2] + 1e-6)\n                test_df[new_feature] = test_df[feature1] / (test_df[feature2] + 1e-6)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:51:10.534482Z","iopub.execute_input":"2024-12-15T16:51:10.534722Z","iopub.status.idle":"2024-12-15T16:51:10.545040Z","shell.execute_reply.started":"2024-12-15T16:51:10.534700Z","shell.execute_reply":"2024-12-15T16:51:10.544253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 2: Initialize the TabularDataPreprocessor and preprocess data\ntabular_data_preprocessor = TabularDataPreprocessor()\n\n# Preprocess the training and test datasets\ntrain_df, test_df = tabular_data_preprocessor.preprocess_data(\n    train_df=train_data,\n    test_df=test_data,\n    feature_types=feature_types\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:51:10.942928Z","iopub.execute_input":"2024-12-15T16:51:10.943562Z","iopub.status.idle":"2024-12-15T16:51:12.190949Z","shell.execute_reply.started":"2024-12-15T16:51:10.943528Z","shell.execute_reply":"2024-12-15T16:51:12.189648Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:51:12.192508Z","iopub.execute_input":"2024-12-15T16:51:12.192862Z","iopub.status.idle":"2024-12-15T16:51:12.200060Z","shell.execute_reply.started":"2024-12-15T16:51:12.192819Z","shell.execute_reply":"2024-12-15T16:51:12.199239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:51:12.201259Z","iopub.execute_input":"2024-12-15T16:51:12.201614Z","iopub.status.idle":"2024-12-15T16:51:12.208859Z","shell.execute_reply.started":"2024-12-15T16:51:12.201577Z","shell.execute_reply":"2024-12-15T16:51:12.207982Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Tabnet","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.linear_model import LogisticRegression\nfrom xgboost import XGBClassifier\nfrom catboost import CatBoostClassifier\nfrom sklearn.metrics import cohen_kappa_score\nfrom collections import Counter\n\n# Quadratic Weighted Kappa (QWK)\ndef quadratic_weighted_kappa(y_true, y_pred):\n    \"\"\"\n    Calculate the quadratic weighted kappa score.\n    \"\"\"\n    return cohen_kappa_score(y_true, y_pred, weights=\"quadratic\")\n\n\ndef train_models_separately_with_cv(\n    X_original: pd.DataFrame,\n    y: pd.Series,\n    n_splits: int = 5,\n    random_state: int = 42,\n    weights: dict = None,\n):\n    \"\"\"\n    Train Logistic Regression, XGBoost, and CatBoost on original features using cross-validation,\n    and return the trained models and cross-validation scores.\n\n    Args:\n        X_original (pd.DataFrame): Original feature matrix.\n        y (pd.Series): Target labels.\n        n_splits (int): Number of cross-validation folds.\n        random_state (int): Random seed for reproducibility.\n        weights (dict): Weights for each model in averaging.\n\n    Returns:\n        dict: Trained models and cross-validation scores.\n    \"\"\"\n    if weights is None:\n        weights = {\n            \"logreg\": 1,\n            \"xgb_original\": 1,\n            \"catboost\": 1,          \n        }\n\n    # Normalize weights\n    total_weight = sum(weights.values())\n    normalized_weights = {key: value / total_weight for key, value in weights.items()}\n\n    # Reset indices to ensure compatibility with StratifiedKFold indices\n    X_original = X_original.reset_index(drop=True)\n    y = y.reset_index(drop=True)\n\n    skf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=random_state)\n\n    # Initialize models\n    models = {\n        \"logreg\": LogisticRegression(\n            C=0.1, class_weight=\"balanced\", solver=\"liblinear\", max_iter=500, random_state=random_state\n        ),\n        \"xgb_original\": XGBClassifier(\n            learning_rate=0.01,\n            max_depth=6,\n            n_estimators=100,\n            objective=\"multi:softprob\",\n            tree_method=\"hist\",\n            device=\"cuda\",\n            random_state=random_state,\n        ),\n        \"catboost\": CatBoostClassifier(\n            depth=10,\n            iterations=100,\n            learning_rate=0.1,\n            class_weights=[0.4291, 0.9370, 1.8095, 20.1176],\n            task_type=\"CPU\",\n            random_state=random_state,\n            verbose=0,\n        ),\n    }\n\n    cv_scores = {}\n\n    for model_name, model in models.items():\n        print(f\"Training {model_name} with cross-validation...\")\n\n        # Initialize out-of-fold predictions for this model\n        oof_preds = np.zeros((len(y), len(np.unique(y))))\n\n        for fold, (train_idx, valid_idx) in enumerate(skf.split(X_original, y), 1):\n            print(f\"  Fold {fold} for {model_name}...\")\n\n            # Convert indices to NumPy arrays for compatibility\n            X_train, X_valid = X_original.iloc[train_idx], X_original.iloc[valid_idx]\n            y_train, y_valid = y.iloc[train_idx], y.iloc[valid_idx]\n\n            # Fit the model on the training set\n            model.fit(X_train, y_train)\n\n            # Predict probabilities for the validation set\n            oof_preds[valid_idx] = model.predict_proba(X_valid)\n\n            # Compute QWK for this fold\n            fold_qwk = quadratic_weighted_kappa(y_valid, np.argmax(oof_preds[valid_idx], axis=1))\n            print(f\"    Fold {fold} QWK: {fold_qwk:.4f}\")\n\n        # Compute overall QWK for the model\n        overall_qwk = quadratic_weighted_kappa(y, np.argmax(oof_preds, axis=1))\n        print(f\"Overall QWK for {model_name}: {overall_qwk:.4f}\")\n        cv_scores[model_name] = overall_qwk\n\n        # Train the model on the entire dataset\n        print(f\"Training {model_name} on the entire dataset...\")\n        model.fit(X_original, y)\n\n    return {\n        \"models\": models,\n        \"cv_scores\": cv_scores,\n    }\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:59:49.684760Z","iopub.execute_input":"2024-12-15T16:59:49.685052Z","iopub.status.idle":"2024-12-15T16:59:49.696943Z","shell.execute_reply.started":"2024-12-15T16:59:49.685026Z","shell.execute_reply":"2024-12-15T16:59:49.696116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Example Usage\nX_train = train_df.drop(columns=['sii', 'id'])\ny_train = train_df['sii']\n\nresults = train_models_separately_with_cv(\n    X_original=X_train,\n    y=y_train\n)\n\n\n# Access cross-validation scores\ncv_scores = results[\"cv_scores\"]\nprint(\"Cross-Validation QWK Scores:\", cv_scores)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:57:51.711909Z","iopub.execute_input":"2024-12-15T16:57:51.712555Z","iopub.status.idle":"2024-12-15T16:59:49.683225Z","shell.execute_reply.started":"2024-12-15T16:57:51.712520Z","shell.execute_reply":"2024-12-15T16:59:49.682245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom scipy.stats import skew, kurtosis\nfrom scipy.fft import fft\n\n# Helper Function: Statistical Features\ndef extract_statistics(data, columns):\n    \"\"\"\n    Extract statistical features (mean, std, min, max, range, skewness, kurtosis) for given columns.\n    \"\"\"\n    stats = {}\n    for col in columns:\n        stats[f\"{col}_mean\"] = data[col].mean()\n        stats[f\"{col}_std\"] = data[col].std()\n        stats[f\"{col}_min\"] = data[col].min()\n        stats[f\"{col}_max\"] = data[col].max()\n        stats[f\"{col}_range\"] = data[col].max() - data[col].min()\n        stats[f\"{col}_skew\"] = skew(data[col], nan_policy='omit')\n        stats[f\"{col}_kurtosis\"] = kurtosis(data[col], nan_policy='omit')\n    return stats\n\n# Helper Function: Energy Features\ndef extract_energy(data, columns):\n    \"\"\"\n    Extract energy features for given columns.\n    \"\"\"\n    energy = {}\n    for col in columns:\n        energy[f\"{col}_energy\"] = np.sum(np.square(data[col]))\n    return energy\n\n# Helper Function: Frequency-Domain Features using FFT\ndef extract_fft(data, column):\n    \"\"\"\n    Extract frequency-domain features using Fast Fourier Transform (FFT).\n    \"\"\"\n    fft_vals = np.abs(fft(data[column]))\n    fft_freqs = np.fft.fftfreq(len(data[column]))\n    dominant_freq = fft_freqs[np.argmax(fft_vals[:len(fft_vals) // 2])]\n    return {\n        f\"{column}_dominant_freq\": dominant_freq,\n        f\"{column}_spectral_energy\": np.sum(fft_vals ** 2)\n    }\n\n# Helper Function: Signal Magnitude Area (SMA)\ndef extract_sma(data, columns):\n    \"\"\"\n    Calculate the Signal Magnitude Area (SMA) for given columns.\n    \"\"\"\n    return {\"SMA\": np.mean(np.abs(data[columns]).sum(axis=1))}\n\n# Helper Function: Non-Wear Time\ndef extract_non_wear(data, column='non-wear_flag'):\n    \"\"\"\n    Calculate the percentage of non-wear time.\n    \"\"\"\n    return {\"non_wear_percentage\": data[column].mean() * 100}\n\n# Helper Function: Percentage of Low Activity\ndef extract_low_activity(data, accel_columns, threshold=0.05):\n    \"\"\"\n    Calculate the percentage of low activity periods.\n    \"\"\"\n    low_activity = (data[accel_columns] <= threshold).all(axis=1).mean() * 100\n    return {\"low_activity_percentage\": low_activity}\n\n# Helper Function: Relative Date Features\ndef extract_relative_date_features(data, column='relative_date_PCIAT'):\n    \"\"\"\n    Extract statistics on relative date since the PCIAT test.\n    \"\"\"\n    return {\n        f\"{column}_mean\": data[column].mean(),\n        f\"{column}_min\": data[column].min(),\n        f\"{column}_max\": data[column].max(),\n        f\"{column}_range\": data[column].max() - data[column].min()\n    }\n\n# Helper Function: Time-of-Day Patterns\ndef extract_time_of_day_patterns(data, value_columns):\n    \"\"\"\n    Calculate average activity per time window (morning, afternoon, evening, night).\n    \"\"\"\n    # Convert time_of_day from nanoseconds to hours\n    data['time_in_hours'] = pd.to_timedelta(data['time_of_day'], unit='ns').dt.total_seconds() // 3600\n    bins = {'night': (0, 6), 'morning': (6, 12), 'afternoon': (12, 18), 'evening': (18, 24)}\n    features = {}\n\n    for window, (start, end) in bins.items():\n        window_data = data[(data['time_in_hours'] >= start) & (data['time_in_hours'] < end)]\n        for col in value_columns:\n            features[f\"{col}_{window}_mean\"] = window_data[col].mean() if not window_data.empty else 0\n    return features\n\n# Helper Function: Weekday vs Weekend Activity\ndef extract_weekday_activity(data, value_columns, column='weekday'):\n    \"\"\"\n    Calculate aggregated activity metrics for weekdays and weekends.\n    \"\"\"\n    features = {}\n    weekday_data = data[data[column] < 6]  # Monday to Friday\n    weekend_data = data[data[column] >= 6]  # Saturday and Sunday\n    for col in value_columns:\n        features[f\"{col}_weekday_mean\"] = weekday_data[col].mean() if not weekday_data.empty else 0\n        features[f\"{col}_weekend_mean\"] = weekend_data[col].mean() if not weekend_data.empty else 0\n    return features\n\n# Helper Function: Battery and Light Features\ndef extract_battery_light_features(data):\n    \"\"\"\n    Calculate mean, std, min, max, and variance for light and battery voltage.\n    \"\"\"\n    return {\n        \"battery_voltage_mean\": data['battery_voltage'].mean(),\n        \"battery_voltage_std\": data['battery_voltage'].std(),\n        \"light_mean\": data['light'].mean(),\n        \"light_std\": data['light'].std(),\n        \"light_min\": data['light'].min(),\n        \"light_max\": data['light'].max(),\n        \"light_variance\": data['light'].var()\n    }\n\n# Main Function: Extract All Features for a Single User\ndef extract_features_for_user(user_df):\n    \"\"\"\n    Extract all features (statistical, energy, frequency, SMA, non-wear time, low activity,\n    relative date features, time-of-day patterns, weekday activity, battery, and light features).\n\n    Parameters\n    ----------\n    user_df : pd.DataFrame\n        DataFrame for a specific user ID containing time series data.\n\n    Returns\n    -------\n    pd.DataFrame\n        DataFrame with one row containing all extracted features for the user.\n    \"\"\"\n    features = {}\n\n    # User ID\n    user_id = user_df['id'].iloc[0]\n    features['id'] = user_id\n\n    accel_columns = ['X', 'Y', 'Z', 'enmo']\n    value_columns = accel_columns + ['light', 'battery_voltage']\n\n    features.update(extract_statistics(user_df, accel_columns))\n    features.update(extract_energy(user_df, accel_columns))\n    features.update(extract_fft(user_df, 'enmo'))\n    features.update(extract_sma(user_df, accel_columns))\n    features.update(extract_non_wear(user_df))\n    #features.update(extract_low_activity(user_df, accel_columns))\n    features.update(extract_relative_date_features(user_df, 'relative_date_PCIAT'))\n    features.update(extract_time_of_day_patterns(user_df, value_columns))\n    features.update(extract_weekday_activity(user_df, value_columns, 'weekday'))\n    features.update(extract_battery_light_features(user_df))\n\n    return pd.DataFrame([features])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:52:44.863441Z","iopub.execute_input":"2024-12-15T16:52:44.863769Z","iopub.status.idle":"2024-12-15T16:52:44.881735Z","shell.execute_reply.started":"2024-12-15T16:52:44.863740Z","shell.execute_reply":"2024-12-15T16:52:44.880868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TimeSeriesDataPreprocessor:\n\n    def aggregate_time_series(self, data_path: str) -> pd.DataFrame:\n        \"\"\"\n        Aggregate time series data for all participants into a single row per participant.\n\n        Args:\n            data_path (str): Path to the folder containing time-series parquet files.\n\n        Returns:\n            pd.DataFrame: Aggregated data with one row per participant.\n        \"\"\"\n        participant_folders = glob(os.path.join(data_path, \"id=*\"))\n        aggregated_data = []\n\n        for participant_folder in participant_folders:\n            # Extract participant ID from folder name\n            participant_id = os.path.basename(participant_folder).split(\"=\")[1]\n\n            # Path to the parquet file\n            file_path = os.path.join(participant_folder, \"part-0.parquet\")\n\n            participant_data = self._process_participant_time_series(file_path, participant_id)\n\n            # Append to the main list if data is not empty\n            if not participant_data.empty:\n                aggregated_data.append(participant_data)\n\n        return pd.concat(aggregated_data, ignore_index=True)\n\n    def _process_participant_time_series(self, file_path: str, participant_id: str) -> pd.DataFrame:\n        \"\"\"\n        Process and aggregate features from a single participant's time-series data.\n\n        Args:\n            file_path (str): Path to the participant's parquet file.\n            participant_id (str): Participant ID.\n\n        Returns:\n            pd.DataFrame: A single-row DataFrame with aggregated features for the participant.\n        \"\"\"\n        try:\n            df = pd.read_parquet(file_path)\n            df['id'] = participant_id\n            stats_df = extract_features_for_user(df)\n\n            return stats_df\n    \n        except Exception as e:\n            print(f\"Error processing file {file_path} for participant {participant_id}: {e}\")\n            return pd.DataFrame()  # Return an empty DataFrame in case of error\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:52:44.882836Z","iopub.execute_input":"2024-12-15T16:52:44.883172Z","iopub.status.idle":"2024-12-15T16:52:44.894105Z","shell.execute_reply.started":"2024-12-15T16:52:44.883130Z","shell.execute_reply":"2024-12-15T16:52:44.893289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df = pd.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=0745c390/part-0.parquet')\n# df['id'] = '0745c390'\n# new_df = extract_features_for_user(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:52:44.895912Z","iopub.execute_input":"2024-12-15T16:52:44.896211Z","iopub.status.idle":"2024-12-15T16:52:44.903914Z","shell.execute_reply.started":"2024-12-15T16:52:44.896184Z","shell.execute_reply":"2024-12-15T16:52:44.903188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ts_preprocessor = TimeSeriesDataPreprocessor()\ntrain_ts_data = ts_preprocessor.aggregate_time_series(ts_train_path)\ntest_ts_data = ts_preprocessor.aggregate_time_series(ts_test_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:52:44.904896Z","iopub.execute_input":"2024-12-15T16:52:44.905220Z","iopub.status.idle":"2024-12-15T16:56:22.534916Z","shell.execute_reply.started":"2024-12-15T16:52:44.905182Z","shell.execute_reply":"2024-12-15T16:56:22.534223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts_data.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:22.535922Z","iopub.execute_input":"2024-12-15T16:56:22.536221Z","iopub.status.idle":"2024-12-15T16:56:22.542828Z","shell.execute_reply.started":"2024-12-15T16:56:22.536193Z","shell.execute_reply":"2024-12-15T16:56:22.541978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts_data.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:22.543872Z","iopub.execute_input":"2024-12-15T16:56:22.544158Z","iopub.status.idle":"2024-12-15T16:56:22.552521Z","shell.execute_reply.started":"2024-12-15T16:56:22.544134Z","shell.execute_reply":"2024-12-15T16:56:22.551637Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ts_data.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:22.553457Z","iopub.execute_input":"2024-12-15T16:56:22.553764Z","iopub.status.idle":"2024-12-15T16:56:22.561704Z","shell.execute_reply.started":"2024-12-15T16:56:22.553724Z","shell.execute_reply":"2024-12-15T16:56:22.560985Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Autoencoder","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\nimport numpy as np\n\n# Define the Autoencoder architecture\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim=38, encoding_dim=16):\n        super(AutoEncoder, self).__init__()\n        # Encoder: Compress to latent space\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim * 3), # Reduce from 38 to 32\n            nn.ReLU(),\n            nn.Linear(encoding_dim * 3, encoding_dim * 2), # Reduce from 32 to 20\n            nn.ReLU(),\n            nn.Linear(encoding_dim * 2, encoding_dim), # Reduce to 16 (latent space)\n            nn.ReLU()\n        )\n        # Decoder: Reconstruct back to original dimension - Mirror the encoder to reconstruct input\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, encoding_dim * 2), # Expand from 16 to 20\n            nn.ReLU(),\n            nn.Linear(encoding_dim * 2, encoding_dim * 3), # Expand from 20 to 32\n            nn.ReLU(),\n            nn.Linear(encoding_dim * 3, input_dim), # Reconstruct to 38\n            nn.Sigmoid()  # Output in range [0, 1]\n        )\n\n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded, encoded\n\n# Function to scale the data using MinMaxScaler\ndef scale_data(df):\n    \"\"\"\n    Scale data using MinMaxScaler to range [0, 1].\n    \"\"\"\n    scaler = MinMaxScaler()\n    scaled_data = scaler.fit_transform(df.values)\n    return scaled_data, scaler\n\n# Function to train the Autoencoder\ndef train_autoencoder(data, input_dim, encoding_dim=16, epochs=50, batch_size=64, lr=0.001):\n    \"\"\"\n    Train the Autoencoder model.\n    \"\"\"\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n    # Convert data to PyTorch tensors\n    data_tensor = torch.tensor(data, dtype=torch.float32)\n    dataset = TensorDataset(data_tensor)\n    dataloader = DataLoader(dataset, batch_size=batch_size, shuffle=True)\n\n    # Initialize model, loss, and optimizer\n    model = AutoEncoder(input_dim=input_dim, encoding_dim=encoding_dim).to(device)\n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(model.parameters(), lr=lr)\n\n    # Training loop\n    model.train()\n    for epoch in range(epochs):\n        epoch_loss = 0\n        for batch in dataloader:\n            batch_data = batch[0].to(device)\n            optimizer.zero_grad()\n            reconstructed, _ = model(batch_data)\n            loss = criterion(reconstructed, batch_data)\n            loss.backward()\n            optimizer.step()\n            epoch_loss += loss.item()\n        print(f\"Epoch {epoch+1}/{epochs}, Loss: {epoch_loss / len(dataloader):.6f}\")\n\n    print(\"Autoencoder training complete.\")\n    return model\n\n# Function to extract features using the trained Autoencoder\ndef extract_features(autoencoder, data):\n    \"\"\"\n    Extract features from the latent space of the Autoencoder.\n    \"\"\"\n    autoencoder.eval()\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n    # Convert to PyTorch tensor\n    data_tensor = torch.tensor(data, dtype=torch.float32).to(device)\n\n    # Extract features\n    with torch.no_grad():\n        features = autoencoder.encoder(data_tensor)\n\n    return features.cpu().numpy()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:22.562960Z","iopub.execute_input":"2024-12-15T16:56:22.563273Z","iopub.status.idle":"2024-12-15T16:56:24.048497Z","shell.execute_reply.started":"2024-12-15T16:56:22.563236Z","shell.execute_reply":"2024-12-15T16:56:24.047789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ids = train_ts_data['id']  # Save the 'id' column\ntrain_features = train_ts_data.drop(columns=['id'])\nprint(\"Features Shape:\", train_features.shape)\n\n# Scale data\ntrain_scaled_data, scaler = scale_data(train_features)\nprint(\"Data Scaled to [0, 1] range.\")\n\n# Define parameters\ninput_dim = train_scaled_data.shape[1]  # Number of input features (e.g., 38)\nencoding_dim = 16  # Latent space dimension\n\nautoencoder = train_autoencoder(\n    data=train_scaled_data,\n    input_dim=input_dim,\n    encoding_dim=encoding_dim,\n    epochs=50,\n    batch_size=32,\n    lr=0.001\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:24.051266Z","iopub.execute_input":"2024-12-15T16:56:24.051900Z","iopub.status.idle":"2024-12-15T16:56:28.666952Z","shell.execute_reply.started":"2024-12-15T16:56:24.051870Z","shell.execute_reply":"2024-12-15T16:56:28.666220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract features from the trained Autoencoder\nextracted_features = extract_features(autoencoder, train_scaled_data)\nprint(\"Extracted Features Shape:\", extracted_features.shape)\n\n# Combine the 'id' column with extracted features\nextracted_features_df = pd.DataFrame(extracted_features, columns=[f\"encoder_feature_{i}\" for i in range(1, encoding_dim + 1)])\nextracted_features_df.insert(0, 'id', ids)  # Add the 'id' column back\nextracted_features_df\n# Save the extracted features to a file\n# extracted_features_df.to_csv(\"extracted_features_with_id.csv\", index=False)\n# print(\"Extracted features saved to 'extracted_features_with_id.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:28.667905Z","iopub.execute_input":"2024-12-15T16:56:28.668473Z","iopub.status.idle":"2024-12-15T16:56:28.702888Z","shell.execute_reply.started":"2024-12-15T16:56:28.668443Z","shell.execute_reply":"2024-12-15T16:56:28.702047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ids = test_ts_data['id']  # Save the 'id' column\ntest_features = test_ts_data.drop(columns=['id'])\nprint(\"Features Shape:\", test_features.shape)\n\n# Scale data\ntest_scaled_data = scaler.transform(test_features)\nprint(\"Data Scaled to [0, 1] range.\")\n\n\n\n# Extract features from the trained Autoencoder\ntest_extracted_features = extract_features(autoencoder, test_scaled_data)\nprint(\"Extracted Features Shape:\", test_extracted_features.shape)\n\ntest_extracted_features_df = pd.DataFrame(test_extracted_features, columns=[f\"encoder_feature_{i}\" for i in range(1, encoding_dim + 1)])\ntest_extracted_features_df.insert(0, 'id', ids)  # Add the 'id' column back\ntest_extracted_features_df\n\n# Save the extracted features to a file\n#extracted_features_df.to_csv(\"extracted_features_with_id.csv\", index=False)\n#print(\"Extracted features saved to 'extracted_features_with_id.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:28.704025Z","iopub.execute_input":"2024-12-15T16:56:28.704334Z","iopub.status.idle":"2024-12-15T16:56:28.728530Z","shell.execute_reply.started":"2024-12-15T16:56:28.704307Z","shell.execute_reply":"2024-12-15T16:56:28.727775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"autoencoder_features =[f\"encoder_feature_{i}\" for i in range(1, encoding_dim + 1)]\nautoencoder_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:28.729519Z","iopub.execute_input":"2024-12-15T16:56:28.729815Z","iopub.status.idle":"2024-12-15T16:56:28.735917Z","shell.execute_reply.started":"2024-12-15T16:56:28.729784Z","shell.execute_reply":"2024-12-15T16:56:28.734960Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.merge(train_df, extracted_features_df, how=\"left\", on=\"id\")\ntest_df = pd.merge(test_df, test_extracted_features_df, how=\"left\", on=\"id\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:28.736923Z","iopub.execute_input":"2024-12-15T16:56:28.737266Z","iopub.status.idle":"2024-12-15T16:56:28.764406Z","shell.execute_reply.started":"2024-12-15T16:56:28.737228Z","shell.execute_reply":"2024-12-15T16:56:28.763663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:28.765393Z","iopub.execute_input":"2024-12-15T16:56:28.765673Z","iopub.status.idle":"2024-12-15T16:56:28.771034Z","shell.execute_reply.started":"2024-12-15T16:56:28.765644Z","shell.execute_reply":"2024-12-15T16:56:28.770007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Apply KNNImputer only on Autoencoder feature columns\nimputer = KNNImputer(n_neighbors=5)  # Use 5 nearest neighbors (default)\nimputed_features = imputer.fit_transform(train_df[autoencoder_features])\n\n# Replace the original Autoencoder feature columns with the imputed values\ntrain_df[autoencoder_features] = imputed_features\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:28.772192Z","iopub.execute_input":"2024-12-15T16:56:28.772547Z","iopub.status.idle":"2024-12-15T16:56:29.098695Z","shell.execute_reply.started":"2024-12-15T16:56:28.772517Z","shell.execute_reply":"2024-12-15T16:56:29.097941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_imputed_features = imputer.transform(test_df[autoencoder_features])\ntest_df[autoencoder_features] = test_imputed_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.099645Z","iopub.execute_input":"2024-12-15T16:56:29.099914Z","iopub.status.idle":"2024-12-15T16:56:29.115846Z","shell.execute_reply.started":"2024-12-15T16:56:29.099888Z","shell.execute_reply":"2024-12-15T16:56:29.115117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define features and target\n# X = train_df.drop(columns=['PCIAT-PCIAT_Total', 'id'])\n# y = train_df['PCIAT-PCIAT_Total']\n\nX = train_df.drop(columns=['sii', 'id'])\ny = train_df['sii']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.116937Z","iopub.execute_input":"2024-12-15T16:56:29.117225Z","iopub.status.idle":"2024-12-15T16:56:29.122993Z","shell.execute_reply.started":"2024-12-15T16:56:29.117200Z","shell.execute_reply":"2024-12-15T16:56:29.122284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n# Define K-Fold cross-validator\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\n# Function to evaluate a model with K-Fold CV using QWK\ndef evaluate_model_with_kfold(model, X, y, thresholds):\n    qwk_scores = []\n\n    for train_idx, val_idx in kf.split(X):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n        # Train the model on the training fold\n        model.fit(X_train, y_train)\n\n        # Predict on the validation fold\n        preds = model.predict(X_val)\n\n        # Threshold predictions into ordinal classes\n        y_pred_class = np.digitize(preds, thresholds)\n        y_val_class = np.digitize(y_val, thresholds)\n\n        # Calculate QWK for the fold\n        qwk = cohen_kappa_score(y_val_class, y_pred_class, weights='quadratic')\n        qwk_scores.append(qwk)\n\n    # Return the mean QWK score across all folds\n    return np.mean(qwk_scores)\n\n# Modify Optuna objective functions for K-Fold CV\ndef objective_lightgbm_kfold(trial):\n    param = {\n        'objective': 'regression',\n        'metric': 'rmse',\n        'boosting_type': 'gbdt',\n        'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.1),\n        'num_leaves': trial.suggest_int('num_leaves', 31, 255),\n        'max_depth': trial.suggest_int('max_depth', 3, 12),\n        'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 10, 100),\n        'lambda_l1': trial.suggest_loguniform('lambda_l1', 1e-8, 10.0),\n        'lambda_l2': trial.suggest_loguniform('lambda_l2', 1e-8, 10.0),\n        'feature_fraction': trial.suggest_uniform('feature_fraction', 0.6, 1.0)\n    }\n\n    model = lgb.LGBMRegressor(**param)\n\n    # Evaluate using K-Fold Cross-Validation\n    thresholds = [30, 50, 80]  # Define boundaries for ordinal classes\n    return evaluate_model_with_kfold(model, X, y, thresholds)\n\ndef objective_xgboost_kfold(trial):\n    param = {\n        'objective': 'reg:squarederror',\n        'eval_metric': 'rmse',\n        'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.1),\n        'max_depth': trial.suggest_int('max_depth', 3, 12),\n        'subsample': trial.suggest_uniform('subsample', 0.6, 1.0),\n        'colsample_bytree': trial.suggest_uniform('colsample_bytree', 0.6, 1.0),\n        'lambda': trial.suggest_loguniform('lambda', 1e-8, 10.0),\n        'alpha': trial.suggest_loguniform('alpha', 1e-8, 10.0)\n    }\n\n    model = xgb.XGBRegressor(**param, n_estimators=1000)\n\n    # Evaluate using K-Fold Cross-Validation\n    thresholds = [30, 50, 80]\n    return evaluate_model_with_kfold(model, X, y, thresholds)\n\ndef objective_catboost_kfold(trial):\n    param = {\n        'depth': trial.suggest_int('depth', 3, 12),\n        'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.1),\n        'l2_leaf_reg': trial.suggest_loguniform('l2_leaf_reg', 1e-8, 10.0),\n        'iterations': 1000,\n        'random_seed': 42,\n        'loss_function': 'RMSE',\n        'verbose': False\n    }\n\n    model = cb.CatBoostRegressor(**param)\n\n    # Evaluate using K-Fold Cross-Validation\n    thresholds = [30, 50, 80]\n    return evaluate_model_with_kfold(model, X, y, thresholds)\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.124044Z","iopub.execute_input":"2024-12-15T16:56:29.124370Z","iopub.status.idle":"2024-12-15T16:56:29.136689Z","shell.execute_reply.started":"2024-12-15T16:56:29.124345Z","shell.execute_reply":"2024-12-15T16:56:29.135861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Run Optuna for each model with K-Fold CV\n# print(\"Optimizing LightGBM with K-Fold CV...\")\n# study_lgb = optuna.create_study(direction='maximize')\n# study_lgb.optimize(objective_lightgbm_kfold, n_trials=20)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.137738Z","iopub.execute_input":"2024-12-15T16:56:29.138600Z","iopub.status.idle":"2024-12-15T16:56:29.148175Z","shell.execute_reply.started":"2024-12-15T16:56:29.138565Z","shell.execute_reply":"2024-12-15T16:56:29.147323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# print(\"Optimizing XGBoost with K-Fold CV...\")\n# study_xgb = optuna.create_study(direction='maximize')\n# study_xgb.optimize(objective_xgboost_kfold, n_trials=20)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.149291Z","iopub.execute_input":"2024-12-15T16:56:29.149574Z","iopub.status.idle":"2024-12-15T16:56:29.155200Z","shell.execute_reply.started":"2024-12-15T16:56:29.149549Z","shell.execute_reply":"2024-12-15T16:56:29.154313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(\"Optimizing CatBoost with K-Fold CV...\")\n# study_cb = optuna.create_study(direction='maximize')\n# study_cb.optimize(objective_catboost_kfold, n_trials=20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.156245Z","iopub.execute_input":"2024-12-15T16:56:29.156527Z","iopub.status.idle":"2024-12-15T16:56:29.162604Z","shell.execute_reply.started":"2024-12-15T16:56:29.156487Z","shell.execute_reply":"2024-12-15T16:56:29.161837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Retrieve best parameters\n# best_params_lgb = study_lgb.best_params\n# best_params_lgb\n# best_params_lgb = {'learning_rate': 0.05007432754045902,\n#  'num_leaves': 31,\n#  'max_depth': 12,\n#  'min_data_in_leaf': 10,\n#  'lambda_l1': 0.016082705420687387,\n#  'lambda_l2': 3.112765171905954e-06,\n#  'feature_fraction': 0.8222451869717737}\n\n#----------------------------------------------\nbest_params_lgb = {'learning_rate': 0.08726088090619541,\n 'num_leaves': 77,\n 'max_depth': 11,\n 'min_data_in_leaf': 52,\n 'lambda_l1': 0.0001122874254953619,\n 'lambda_l2': 0.007802495133278185,\n 'feature_fraction': 0.6650651635137848}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.163498Z","iopub.execute_input":"2024-12-15T16:56:29.163836Z","iopub.status.idle":"2024-12-15T16:56:29.171393Z","shell.execute_reply.started":"2024-12-15T16:56:29.163810Z","shell.execute_reply":"2024-12-15T16:56:29.170536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# best_params_xgb = study_xgb.best_params\n# best_params_xgb\n# best_params_xgb = {'learning_rate': 0.0184111445909722,\n#  'max_depth': 7,\n#  'subsample': 0.8973974031687456,\n#  'colsample_bytree': 0.8945188938106481,\n#  'lambda': 5.082055664747022,\n#  'alpha': 0.007858947844464396}\n\n\n# ----------------------------------------------\nbest_params_xgb = {'learning_rate': 0.03522156033823975,\n 'max_depth': 4,\n 'subsample': 0.8944661619761882,\n 'colsample_bytree': 0.6794186946864124,\n 'lambda': 7.202563429369434,\n 'alpha': 6.9622134111427425e-06}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.172739Z","iopub.execute_input":"2024-12-15T16:56:29.172986Z","iopub.status.idle":"2024-12-15T16:56:29.182126Z","shell.execute_reply.started":"2024-12-15T16:56:29.172963Z","shell.execute_reply":"2024-12-15T16:56:29.181204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#best_params_cb = study_cb.best_params\n# best_params_cb = {'depth': 5,\n#  'learning_rate': 0.041526224339960234,\n#  'l2_leaf_reg': 2.235248251230377e-07}\n#---------------------------\nbest_params_cb = {'depth': 6,\n 'learning_rate': 0.022558746700484376,\n 'l2_leaf_reg': 0.004127600472567024}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.183030Z","iopub.execute_input":"2024-12-15T16:56:29.183322Z","iopub.status.idle":"2024-12-15T16:56:29.190801Z","shell.execute_reply.started":"2024-12-15T16:56:29.183298Z","shell.execute_reply":"2024-12-15T16:56:29.189941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#study_cb.best_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.191786Z","iopub.execute_input":"2024-12-15T16:56:29.192134Z","iopub.status.idle":"2024-12-15T16:56:29.199647Z","shell.execute_reply.started":"2024-12-15T16:56:29.192096Z","shell.execute_reply":"2024-12-15T16:56:29.198814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train final models with best parameters\nfinal_lgb = lgb.LGBMRegressor(**best_params_lgb)\nfinal_xgb = xgb.XGBRegressor(**best_params_xgb)\nfinal_cb = cb.CatBoostRegressor(**best_params_cb)\n\n# final_lgb.fit(X, y)\n# final_xgb.fit(X, y)\n# final_cb.fit(X, y)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.200753Z","iopub.execute_input":"2024-12-15T16:56:29.201383Z","iopub.status.idle":"2024-12-15T16:56:29.206900Z","shell.execute_reply.started":"2024-12-15T16:56:29.201331Z","shell.execute_reply":"2024-12-15T16:56:29.206216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create Voting Regressor\nvoting_regressor = VotingRegressor(\n    estimators=[\n        ('lightgbm', final_lgb),\n        ('xgboost', final_xgb),\n        ('catboost', final_cb)\n    ]\n)\n\n# Train Voting Regressor\n#voting_regressor.fit(X, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.211340Z","iopub.execute_input":"2024-12-15T16:56:29.211643Z","iopub.status.idle":"2024-12-15T16:56:29.216002Z","shell.execute_reply.started":"2024-12-15T16:56:29.211609Z","shell.execute_reply":"2024-12-15T16:56:29.215125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Extract the unique identifier and features from the test set\n# test_ids = test_df['id']\n# X_test = test_df.drop(columns=['id'])  # Drop 'id' to get only the features\n\n# # Predict using the trained Voting Regressor\n# test_predictions = voting_regressor.predict(X_test)\n\n# # Define the thresholds to map continuous predictions to ordinal classes\n# thresholds = [30, 50, 80]\n# test_classes = np.digitize(test_predictions, thresholds)\n\n# # Create the submission dataframe\n# submission = pd.DataFrame({\n#     'id': test_ids,\n#     'sii': test_classes\n# })\n\n# # Save the submission file\n# submission.to_csv('submission.csv', index=False)\n\n# print(\"Submission file created: submission.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.217050Z","iopub.execute_input":"2024-12-15T16:56:29.217352Z","iopub.status.idle":"2024-12-15T16:56:29.224125Z","shell.execute_reply.started":"2024-12-15T16:56:29.217318Z","shell.execute_reply":"2024-12-15T16:56:29.223302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate class weights\nclass_counts = Counter(y)\ntotal_samples = len(y)\nclass_weights = {cls: total_samples / (len(class_counts) * count) for cls, count in class_counts.items()}\nprint(f\"Class weights: {class_weights}\")\n\n# Define Stratified K-Fold Cross-Validation\nkf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\n\n# Quadratic Weighted Kappa (QWK)\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n# Function to evaluate a model with Stratified K-Fold CV using QWK\ndef evaluate_model_with_kfold(model, X, y):\n    qwk_scores = []\n\n    for train_idx, val_idx in kf.split(X, y):\n        X_train_fold, X_val_fold = X.iloc[train_idx], X.iloc[val_idx]\n        y_train_fold, y_val_fold = y.iloc[train_idx], y.iloc[val_idx]\n\n        # Train the model on the training fold\n        model.fit(X_train_fold, y_train_fold, sample_weight=y_train_fold.map(class_weights))\n        # Predict on the validation fold\n        preds = model.predict(X_val_fold)\n\n        # Calculate QWK for the fold\n        qwk = quadratic_weighted_kappa(y_val_fold, preds)\n        qwk_scores.append(qwk)\n\n    # Return the mean QWK score across all folds\n    return np.mean(qwk_scores)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.225099Z","iopub.execute_input":"2024-12-15T16:56:29.225385Z","iopub.status.idle":"2024-12-15T16:56:29.237151Z","shell.execute_reply.started":"2024-12-15T16:56:29.225354Z","shell.execute_reply":"2024-12-15T16:56:29.236327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Optuna objective functions for each model\ndef objective_lightgbm(trial):\n    param = {\n        'objective': 'multiclass',\n        'metric': 'multi_logloss',\n        'num_class': len(class_counts),  # Number of classes\n        'boosting_type': 'gbdt',\n        'device_type': 'gpu',  # Enable GPU for LightGBM\n        'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.1),\n        'num_leaves': trial.suggest_int('num_leaves', 31, 255),\n        'max_depth': trial.suggest_int('max_depth', 3, 12),\n        'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 10, 100),\n        'lambda_l1': trial.suggest_loguniform('lambda_l1', 1e-8, 10.0),\n        'lambda_l2': trial.suggest_loguniform('lambda_l2', 1e-8, 10.0),\n        'feature_fraction': trial.suggest_uniform('feature_fraction', 0.6, 1.0),\n        'class_weight': 'balanced'\n    }\n\n    model = lgb.LGBMClassifier(**param)\n    return evaluate_model_with_kfold(model, X, y)\n\ndef objective_xgboost(trial):\n    param = {\n        'objective': 'multi:softprob',\n        'eval_metric': 'mlogloss',\n        'num_class': len(class_counts),  # Number of classes\n        'tree_method': \"hist\",\n        'device': \"cuda\",\n        'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.1),\n        'max_depth': trial.suggest_int('max_depth', 3, 12),\n        'subsample': trial.suggest_uniform('subsample', 0.6, 1.0),\n        'colsample_bytree': trial.suggest_uniform('colsample_bytree', 0.6, 1.0),\n        'lambda': trial.suggest_loguniform('lambda', 1e-8, 10.0),\n        'alpha': trial.suggest_loguniform('alpha', 1e-8, 10.0),\n        'scale_pos_weight': None  # Class imbalance is handled via sample weights\n    }\n\n    model = xgb.XGBClassifier(**param, use_label_encoder=False)\n    return evaluate_model_with_kfold(model, X, y)\n\ndef objective_catboost(trial):\n    param = {\n        'depth': trial.suggest_int('depth', 3, 12),\n        'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.1),\n        'l2_leaf_reg': trial.suggest_loguniform('l2_leaf_reg', 1e-8, 10.0),\n        'iterations': 1000,\n        'random_seed': 42,\n        'loss_function': 'MultiClass',\n        'class_weights': list(class_weights.values()),  # CatBoost accepts class weights directly\n        'task_type': 'GPU',  # Enable GPU for CatBoost\n        'verbose': False\n    }\n\n    model = cb.CatBoostClassifier(**param)\n    return evaluate_model_with_kfold(model, X, y)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.238531Z","iopub.execute_input":"2024-12-15T16:56:29.238840Z","iopub.status.idle":"2024-12-15T16:56:29.248639Z","shell.execute_reply.started":"2024-12-15T16:56:29.238806Z","shell.execute_reply":"2024-12-15T16:56:29.247781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(\"Optimizing CatBoost...\")\n# study_cb = optuna.create_study(direction='maximize')\n# study_cb.optimize(objective_catboost, n_trials=20)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.249653Z","iopub.execute_input":"2024-12-15T16:56:29.249967Z","iopub.status.idle":"2024-12-15T16:56:29.259239Z","shell.execute_reply.started":"2024-12-15T16:56:29.249941Z","shell.execute_reply":"2024-12-15T16:56:29.258469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# best_params_cb = study_cb.best_params\n# best_params_cb\n# best_params_cb = {'depth': 5,\n#  'learning_rate': 0.07643512432878817,\n#  'l2_leaf_reg': 1.3485070228390585}\n# ----------------------------------\nbest_params_cb = {'depth': 6,\n 'learning_rate': 0.014837619343465627,\n 'l2_leaf_reg': 2.996476398082938}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.260212Z","iopub.execute_input":"2024-12-15T16:56:29.260544Z","iopub.status.idle":"2024-12-15T16:56:29.266395Z","shell.execute_reply.started":"2024-12-15T16:56:29.260503Z","shell.execute_reply":"2024-12-15T16:56:29.265600Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(\"Optimizing XGBoost...\")\n# study_xgb = optuna.create_study(direction='maximize')\n# study_xgb.optimize(objective_xgboost, n_trials=20)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.267340Z","iopub.execute_input":"2024-12-15T16:56:29.267587Z","iopub.status.idle":"2024-12-15T16:56:29.274813Z","shell.execute_reply.started":"2024-12-15T16:56:29.267564Z","shell.execute_reply":"2024-12-15T16:56:29.274051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# best_params_xgb = study_xgb.best_params\n# best_params_xgb\n# best_params_xgb = {'learning_rate': 0.03283355773388092,\n#  'max_depth': 11,\n#  'subsample': 0.7228933869136653,\n#  'colsample_bytree': 0.6145566224459017,\n#  'lambda': 5.5289888333367765,\n#  'alpha': 2.04494920609885e-08}\n# -------------------------------------\nbest_params_xgb = {'learning_rate': 0.037146617915898926,\n 'max_depth': 9,\n 'subsample': 0.8516489160596048,\n 'colsample_bytree': 0.6735124445558762,\n 'lambda': 8.378281015658663,\n 'alpha': 0.031021565562078746}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.275779Z","iopub.execute_input":"2024-12-15T16:56:29.276370Z","iopub.status.idle":"2024-12-15T16:56:29.282693Z","shell.execute_reply.started":"2024-12-15T16:56:29.276331Z","shell.execute_reply":"2024-12-15T16:56:29.281874Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(\"Optimizing LightGBM...\")\n# study_lgb = optuna.create_study(direction='maximize')\n# study_lgb.optimize(objective_lightgbm, n_trials=20)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.283625Z","iopub.execute_input":"2024-12-15T16:56:29.283864Z","iopub.status.idle":"2024-12-15T16:56:29.291025Z","shell.execute_reply.started":"2024-12-15T16:56:29.283841Z","shell.execute_reply":"2024-12-15T16:56:29.290299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# best_params_lgb = study_lgb.best_params\n# best_params_lgb\n# best_params_lgb = {'learning_rate': 0.0540578615280753,\n#  'num_leaves': 84,\n#  'max_depth': 11,\n#  'min_data_in_leaf': 14,\n#  'lambda_l1': 0.031137725698488855,\n#  'lambda_l2': 0.005236805648248619,\n#  'feature_fraction': 0.6917113698255432}\n#-------------------------------------\n\nbest_params_lgb = {'learning_rate': 0.04152850494102635,\n 'num_leaves': 230,\n 'max_depth': 10,\n 'min_data_in_leaf': 24,\n 'lambda_l1': 0.03963647374273808,\n 'lambda_l2': 2.3678169246617127e-06,\n 'feature_fraction': 0.9502539241083896}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.292132Z","iopub.execute_input":"2024-12-15T16:56:29.292428Z","iopub.status.idle":"2024-12-15T16:56:29.297948Z","shell.execute_reply.started":"2024-12-15T16:56:29.292403Z","shell.execute_reply":"2024-12-15T16:56:29.297131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train final models with best parameters\nfinal_lgb = lgb.LGBMClassifier(**best_params_lgb, device_type='gpu')\nfinal_xgb = xgb.XGBClassifier(**best_params_xgb, tree_method='gpu_hist', use_label_encoder=False)\nfinal_cb = cb.CatBoostClassifier(**best_params_cb, task_type='GPU')\n\nfinal_lgb.fit(X, y, sample_weight=y.map(class_weights))\nfinal_xgb.fit(X, y, sample_weight=y.map(class_weights))\nfinal_cb.fit(X, y)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:29.299255Z","iopub.execute_input":"2024-12-15T16:56:29.299644Z","iopub.status.idle":"2024-12-15T16:56:47.916621Z","shell.execute_reply.started":"2024-12-15T16:56:29.299599Z","shell.execute_reply":"2024-12-15T16:56:47.915798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract the unique identifier and features from the test set\ntest_ids = test_df['id']\nX_test = test_df.drop(columns=['id'])  # Drop 'id' to get only the features\nautoencoder_features_final  = autoencoder_features + ['id']\nX_test_tab = test_df.drop(columns=autoencoder_features_final)\n\n# Predict probabilities on test set\nprobs_lgb = final_lgb.predict_proba(X_test)\nprobs_xgb = final_xgb.predict_proba(X_test)\nprobs_cb = final_cb.predict_proba(X_test)\n\ntab_cb = results[\"models\"]['catboost']\nprobs_tab_cb = tab_cb.predict_proba(X_test_tab)\n\n# Average the probabilities\navg_probs = (0.2 * probs_lgb + 0.3 * probs_xgb + 0.2* probs_cb + 0.3* probs_tab_cb) / 4\n\n# Assign the class with the highest probability\nfinal_predictions = np.argmax(avg_probs, axis=1)\n\n\n\n\n#-----------------------------------------------------------\n# Create the submission file\nsubmission = pd.DataFrame({\n    'id': test_data['id'],\n    'sii': final_predictions\n})\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission file created: submission.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T17:51:47.413301Z","iopub.execute_input":"2024-12-15T17:51:47.414180Z","iopub.status.idle":"2024-12-15T17:51:47.656724Z","shell.execute_reply.started":"2024-12-15T17:51:47.414125Z","shell.execute_reply":"2024-12-15T17:51:47.655388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-15T16:56:48.000375Z","iopub.execute_input":"2024-12-15T16:56:48.000673Z","iopub.status.idle":"2024-12-15T16:56:48.016604Z","shell.execute_reply.started":"2024-12-15T16:56:48.000639Z","shell.execute_reply":"2024-12-15T16:56:48.013786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}