{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\n\nfrom sklearn.preprocessing import MinMaxScaler, StandardScaler, RobustScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nSEED = 42\nn_splits = 5","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-08T07:18:07.227874Z","iopub.execute_input":"2024-11-08T07:18:07.228424Z","iopub.status.idle":"2024-11-08T07:18:16.866705Z","shell.execute_reply.started":"2024-11-08T07:18:07.228361Z","shell.execute_reply":"2024-11-08T07:18:16.865446Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T07:18:16.868532Z","iopub.execute_input":"2024-11-08T07:18:16.870149Z","iopub.status.idle":"2024-11-08T07:18:16.880603Z","shell.execute_reply.started":"2024-11-08T07:18:16.870058Z","shell.execute_reply":"2024-11-08T07:18:16.879308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Sparse Autoencoder Model\nclass SparseAutoencoder(nn.Module):\n    def __init__(self, input_dim, sparsity_weight=1e-5):\n        super(SparseAutoencoder, self).__init__()\n        self.sparsity_weight = sparsity_weight\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, 64),\n            nn.ReLU(),\n            nn.Linear(64, 32),\n            nn.ReLU(),\n            nn.Linear(32, 16),\n            nn.ReLU()\n        )\n        \n        self.decoder = nn.Sequential(\n            nn.Linear(16, 32),\n            nn.ReLU(),\n            nn.Linear(32, 64),\n            nn.ReLU(),\n            nn.Linear(64, input_dim),\n            nn.Sigmoid()  # Outputs in the range [0, 1]\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return encoded, decoded\n\n# Preparing Data\n# Option to use different scalers: MinMaxScaler, StandardScaler, RobustScaler\ndef prepare_data(data, scaler_type='MinMaxScaler'):\n    if scaler_type == 'StandardScaler':\n        scaler = StandardScaler()\n    elif scaler_type == 'RobustScaler':\n        scaler = RobustScaler()\n    else:\n        scaler = MinMaxScaler()\n    \n    data_scaled = scaler.fit_transform(data)\n    return torch.tensor(data_scaled, dtype=torch.float32), scaler\n\n# Apply PCA for Dimensionality Reduction\n# This can help focus the autoencoder on the most relevant features\ndef apply_pca(data, n_components=0.95):\n    pca = PCA(n_components=n_components)\n    data_pca = pca.fit_transform(data)\n    return data_pca, pca\n\n# Early Stopping Functionality\ndef early_stopping(patience):\n    class EarlyStopping:\n        def __init__(self, patience=patience):\n            self.patience = patience\n            self.counter = 0\n            self.best_loss = float('inf')\n            self.early_stop = False\n        \n        def __call__(self, loss):\n            if loss < self.best_loss:\n                self.best_loss = loss\n                self.counter = 0\n            else:\n                self.counter += 1\n                if self.counter >= self.patience:\n                    self.early_stop = True\n    return EarlyStopping()\n\n# Training the Sparse Autoencoder with DataFrame Output\ndef perform_autoencoder(data, epochs=100, batch_size=32, learning_rate=0.001, patience=10, scaler_type='MinMaxScaler', use_pca=False, sparsity_weight=1e-5):\n    # Preprocess Data\n    if use_pca:\n        data, pca = apply_pca(data)\n\n    data_tensor, scaler = prepare_data(data, scaler_type=scaler_type)\n    train_data, val_data = train_test_split(data_tensor, test_size=0.2, random_state=42)\n\n    train_loader = DataLoader(TensorDataset(train_data), batch_size=batch_size, shuffle=True)\n    val_loader = DataLoader(TensorDataset(val_data), batch_size=batch_size, shuffle=False)\n\n    model = SparseAutoencoder(input_dim=data.shape[1], sparsity_weight=sparsity_weight)\n    device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n    model.to(device)\n\n    criterion = nn.SmoothL1Loss()  # Changed to Smooth L1 Loss\n    optimizer = optim.Adam(model.parameters(), lr=learning_rate)\n    stopper = early_stopping(patience=patience)\n\n    for epoch in range(epochs):\n        model.train()\n        train_loss = 0.0\n        for batch in train_loader:\n            batch = batch[0].to(device)\n            optimizer.zero_grad()\n            encoded, outputs = model(batch)\n            \n            # Reconstruction loss\n            loss = criterion(outputs, batch)\n            \n            # Sparsity penalty (L1 regularization on encoded activations)\n            l1_penalty = torch.mean(torch.abs(encoded))\n            loss += sparsity_weight * l1_penalty\n            \n            loss.backward()\n            optimizer.step()\n            train_loss += loss.item() * batch.size(0)\n\n        train_loss /= len(train_loader.dataset)\n\n        # Validation\n        model.eval()\n        val_loss = 0.0\n        with torch.no_grad():\n            for batch in val_loader:\n                batch = batch[0].to(device)\n                _, outputs = model(batch)\n                loss = criterion(outputs, batch)\n                val_loss += loss.item() * batch.size(0)\n\n        val_loss /= len(val_loader.dataset)\n        print(f\"Epoch {epoch+1}, Train Loss: {train_loss:.4f}, Validation Loss: {val_loss:.4f}\")\n\n        # Early stopping\n        stopper(val_loss)\n        if stopper.early_stop:\n            print(f\"Early stopping at epoch {epoch + 1}\")\n            break\n\n    # Convert tensor back to DataFrame for consistency\n    _, data_decoded = model(data_tensor.to(device))\n    data_decoded = data_decoded.cpu().detach().numpy()\n    df_encoded = pd.DataFrame(data_decoded, columns=[f'feature_{i}' for i in range(data_decoded.shape[1])])\n    return df_encoded\n\n# Usage example\n# Assuming 'data' is your input dataset as a NumPy array or pandas DataFrame.\n# df_encoded = train_sparse_autoencoder(data, epochs=100, batch_size=32, learning_rate=0.001, patience=10, scaler_type='StandardScaler', use_pca=True, sparsity_weight=1e-5)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-08T07:18:16.882560Z","iopub.execute_input":"2024-11-08T07:18:16.883155Z","iopub.status.idle":"2024-11-08T07:18:16.914557Z","shell.execute_reply.started":"2024-11-08T07:18:16.883105Z","shell.execute_reply":"2024-11-08T07:18:16.913115Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define normal values for relevant variables based on age and sex\nnormal_values = pd.DataFrame({\n    'Basic_Demos-Age': list(range(5, 23)) * 2,\n    'Basic_Demos-Sex': [0] * 18 + [1] * 18,\n    'Normal_BMI': [15 + i * 0.5 for i in range(18)] * 2,\n    'Normal_BMR': [1100 + i * 50 for i in range(18)] * 2,\n    'Normal_HeartRate': [80 for _ in range(36)],\n    'Normal_Systolic_BP': [100 + i for i in range(18)] * 2,\n    'Normal_Diastolic_BP': [65 + i for i in range(18)] * 2,\n})\n\n# Define start month for each season\nseason_start_month = {\n    'Spring': 3, 'Summer': 6, 'Fall': 9, 'Winter': 12\n}\n\ndef feature_engineering(df):\n    # Convert season columns to dummies and add start month\n    season_cols = [col for col in df.columns if 'Season' in col]\n    for col in season_cols:\n        df[col + '_StartMonth'] = df[col].map(season_start_month)  # Map season start month\n\n        # Create dummy variables for seasons\n        for season in season_start_month.keys():\n            df[col + '_' + season] = (df[col] == season).astype(float)\n        \n        # Calculate seasonal month difference with enrollment season\n        if 'Basic_Demos-Enroll_Season' in season_cols and col != 'Basic_Demos-Enroll_Season':\n            df[col + '_MonthDifference'] = df.apply(\n                lambda row: (\n                    12 - abs(row['Basic_Demos-Enroll_Season_StartMonth'] - row[col + '_StartMonth'])\n                    if row['Basic_Demos-Enroll_Season_StartMonth'] > row[col + '_StartMonth'] else\n                    abs(row['Basic_Demos-Enroll_Season_StartMonth'] - row[col + '_StartMonth'])\n                ) if pd.notna(row['Basic_Demos-Enroll_Season_StartMonth']) and pd.notna(row[col + '_StartMonth'])\n                else np.nan, axis=1\n            )\n    \n    # Drop the original season columns after processing\n    df = df.drop(season_cols, axis=1)\n\n    # Merge with normal values based on age and sex\n    df = df.merge(normal_values, on=['Basic_Demos-Age', 'Basic_Demos-Sex'], how='left')\n\n    # Calculate inflation factors\n    df['BMI_Inflation'] = df['Physical-BMI'] / df['Normal_BMI']\n    df['BMR_Inflation'] = df['BIA-BIA_BMR'] / df['Normal_BMR']\n    df['HeartRate_Inflation'] = df['Physical-HeartRate'] / df['Normal_HeartRate']\n    df['Systolic_BP_Inflation'] = df['Physical-Systolic_BP'] / df['Normal_Systolic_BP']\n    df['Diastolic_BP_Inflation'] = df['Physical-Diastolic_BP'] / df['Normal_Diastolic_BP']\n\n    # Existing feature engineering interactions\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n\n    # Additional interaction variables\n    df['Age_Weight'] = df['Basic_Demos-Age'] * df['Physical-Weight']\n    df['Sex_BMI'] = df['Basic_Demos-Sex'] * df['Physical-BMI']\n    df['Sex_HeartRate'] = df['Basic_Demos-Sex'] * df['Physical-HeartRate']\n    df['Age_WaistCirc'] = df['Basic_Demos-Age'] * df['Physical-Waist_Circumference']\n    df['BMI_FitnessMaxStage'] = df['Physical-BMI'] * df['Fitness_Endurance-Max_Stage']\n    df['Weight_GripStrengthDominant'] = df['Physical-Weight'] * df['FGC-FGC_GSD']\n    df['Weight_GripStrengthNonDominant'] = df['Physical-Weight'] * df['FGC-FGC_GSND']\n    df['HeartRate_FitnessTime'] = df['Physical-HeartRate'] * (df['Fitness_Endurance-Time_Mins'] + df['Fitness_Endurance-Time_Sec'])\n    df['Age_PushUp'] = df['Basic_Demos-Age'] * df['FGC-FGC_PU']\n    df['FFMI_Age'] = df['BIA-BIA_FFMI'] * df['Basic_Demos-Age']\n    df['InternetUse_SleepDisturbance'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['SDS-SDS_Total_Raw']\n    df['CGAS_BMI'] = df['CGAS-CGAS_Score'] * df['Physical-BMI']\n    df['CGAS_FitnessMaxStage'] = df['CGAS-CGAS_Score'] * df['Fitness_Endurance-Max_Stage']\n    \n    # KPIs and Interaction Variables\n    df['Sleep_Disturbance_Index'] = (df['SDS-SDS_Total_Raw'] * df['SDS-SDS_Total_T']) / 100\n    df['Strength_Flexibility_Score'] = (df['FGC-FGC_GSD'] + df['FGC-FGC_GSND'] + df['FGC-FGC_SRL'] + df['FGC-FGC_SRR']) / 4\n    df['Hydration_BMI'] = df['BIA-BIA_TBW'] / df['Physical-BMI']\n    df['Metabolic_Risk'] = (df['BIA-BIA_Fat'] + df['Physical-BMI'] + df['Physical-Systolic_BP'] + df['Physical-Diastolic_BP']) / 4\n    df['Age_BMI_BMR'] = (df['Basic_Demos-Age'] * df['Physical-BMI'] * df['BIA-BIA_BMR']) / 100\n    df['Internet_Mental_Resilience'] = df['PreInt_EduHx-computerinternet_hoursday'] * (1 - (df['CGAS-CGAS_Score'] / 100)) * (df['SDS-SDS_Total_T'] / 100)\n    df['Age_FitnessImpact'] = df['Basic_Demos-Age'] * (df['Fitness_Endurance-Max_Stage'] + df['FGC-FGC_CU'] + df['FGC-FGC_PU']) / 3\n    df['Sedentary_Lifestyle_Flag'] = ((df['PreInt_EduHx-computerinternet_hoursday'] >= 3) & ((df['PAQ_A-PAQ_A_Total'] + df['PAQ_C-PAQ_C_Total']) < 1)).astype(int)\n\n    # Inflation factor interactions\n    df['Metabolic_Age'] = df['BMR_Inflation'] * df['Basic_Demos-Age']\n    df['BMI_Age_Inflation'] = df['BMI_Inflation'] * df['Basic_Demos-Age']\n    df['HeartRate_BMI_Inflation'] = df['HeartRate_Inflation'] * df['BMI_Inflation']\n    df['BP_Inflation_Impact'] = (df['Systolic_BP_Inflation'] + df['Diastolic_BP_Inflation']) / 2\n\n    # Additional new features\n    df['Internet_Addiction_MentalImpact'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['CGAS-CGAS_Score'] * (1 - (df['SDS-SDS_Total_T'] / 100))\n    df['Sedentary_Balance'] = ((df['PAQ_A-PAQ_A_Total'] + df['PAQ_C-PAQ_C_Total']) * df['BIA-BIA_DEE']) / (df['PreInt_EduHx-computerinternet_hoursday'] + 1)\n    df['Social_Isolation_Score'] = df[['SDS-SDS_Total_Raw', 'CGAS-CGAS_Score', 'Physical-HeartRate']].mean(axis=1)\n    df['Internet_Addiction_Score'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['CGAS-CGAS_Score']\n    df['Behavioural_Change_Score'] = (df['Physical-BMI'] + df['CGAS-CGAS_Score'] + df['SDS-SDS_Total_Raw']) / 3\n    df['Cellular_Water_Index'] = df['BIA-BIA_ICW'] / df['BIA-BIA_ECW']\n    df['Intracellular_Water_Index'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    df['Extracellular_Water_Index'] = df['BIA-BIA_ECW'] / df['BIA-BIA_TBW']\n    df['Body_Hydration_Percent'] = df['BIA-BIA_TBW'] / df['BIA-BIA_FFM']\n    df['Skeletal_Muscle_Mass_Percent'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FFM']\n    df['Height_to_Waist'] = df['Physical-Height'] / df['Physical-Waist_Circumference']\n    df['Body_Frame_Age_Sex'] = df['BIA-BIA_Frame_num'] * df['Basic_Demos-Age'] * df['Basic_Demos-Sex']\n    df['Mental_Health_Index'] = df['Behavioural_Change_Score'] * df['Sleep_Disturbance_Index']\n\n    return df\n\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train, epochs=100, batch_size=32, learning_rate=0.001, patience=10, use_pca=False, scaler_type='MinMaxScaler', sparsity_weight=1e-5)\ntest_ts_encoded = perform_autoencoder(df_test, epochs=100, batch_size=32, learning_rate=0.001, patience=10, use_pca=False, scaler_type='MinMaxScaler', sparsity_weight=1e-5)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded['id']=test_ts[\"id\"]\n\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T07:18:16.916897Z","iopub.execute_input":"2024-11-08T07:18:16.917317Z","iopub.status.idle":"2024-11-08T07:20:14.974564Z","shell.execute_reply.started":"2024-11-08T07:18:16.917276Z","shell.execute_reply":"2024-11-08T07:20:14.973109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T07:20:14.976428Z","iopub.execute_input":"2024-11-08T07:20:14.977481Z","iopub.status.idle":"2024-11-08T07:20:15.213283Z","shell.execute_reply.started":"2024-11-08T07:20:14.977411Z","shell.execute_reply":"2024-11-08T07:20:15.211846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n        \ntrain = train_imputed\n\ntrain = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T07:20:15.215016Z","iopub.execute_input":"2024-11-08T07:20:15.215568Z","iopub.status.idle":"2024-11-08T07:20:26.258854Z","shell.execute_reply.started":"2024-11-08T07:20:15.215514Z","shell.execute_reply":"2024-11-08T07:20:26.257628Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.drop('id', axis=1)\ntrain","metadata":{"execution":{"iopub.status.busy":"2024-11-08T07:20:26.263678Z","iopub.execute_input":"2024-11-08T07:20:26.264187Z","iopub.status.idle":"2024-11-08T07:20:26.658547Z","shell.execute_reply.started":"2024-11-08T07:20:26.264136Z","shell.execute_reply":"2024-11-08T07:20:26.656928Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.drop('id', axis=1)\ntest","metadata":{"execution":{"iopub.status.busy":"2024-11-08T07:20:26.660204Z","iopub.execute_input":"2024-11-08T07:20:26.660773Z","iopub.status.idle":"2024-11-08T07:20:27.186008Z","shell.execute_reply.started":"2024-11-08T07:20:26.660708Z","shell.execute_reply":"2024-11-08T07:20:27.184730Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Base features\nfeaturesCols = [\n    'Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-CGAS_Score', 'Physical-BMI', 'Physical-Height', \n    'Physical-Weight', 'Physical-Waist_Circumference', 'Physical-Diastolic_BP', 'Physical-HeartRate', \n    'Physical-Systolic_BP', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', \n    'Fitness_Endurance-Time_Sec', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', \n    'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', \n    'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-BIA_Activity_Level_num', \n    'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', \n    'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', \n    'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total', \n    'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-computerinternet_hoursday', 'sii'\n]\n\n# Feature engineered interactions and inflation factor-based features\nengineered_features = [\n    'BMI_Age', 'Internet_Hours_Age', 'BMI_Internet_Hours', 'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', \n    'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight', 'SMM_Height', 'Muscle_to_Fat', \n    'Hydration_Status', 'ICW_TBW', 'Age_Weight', 'Sex_BMI', 'Sex_HeartRate', 'Age_WaistCirc', \n    'BMI_FitnessMaxStage', 'Weight_GripStrengthDominant', 'Weight_GripStrengthNonDominant', \n    'HeartRate_FitnessTime', 'Age_PushUp', 'FFMI_Age', 'InternetUse_SleepDisturbance', 'CGAS_BMI', \n    'CGAS_FitnessMaxStage', 'Sleep_Disturbance_Index', 'Internet_Addiction_MentalImpact', \n    'Sedentary_Balance', 'Social_Isolation_Score', 'Strength_Flexibility_Score', 'Hydration_BMI', \n    'Metabolic_Risk', 'Age_BMI_BMR', 'Internet_Mental_Resilience', 'Age_FitnessImpact', \n    'Sedentary_Lifestyle_Flag', 'Internet_Addiction_Score', 'Behavioural_Change_Score', \n    'Mental_Health_Index', 'Cellular_Water_Index', \n    'Intracellular_Water_Index', 'Extracellular_Water_Index', 'Body_Hydration_Percent', \n    'Skeletal_Muscle_Mass_Percent', 'Height_to_Waist', 'Body_Frame_Age_Sex', 'Metabolic_Age', \n    'BMI_Age_Inflation', 'HeartRate_BMI_Inflation', 'BP_Inflation_Impact'\n]\n\nseason_features = ['Basic_Demos-Enroll_Season_StartMonth',\n 'Basic_Demos-Enroll_Season_Spring',\n 'Basic_Demos-Enroll_Season_Summer',\n 'Basic_Demos-Enroll_Season_Fall',\n 'Basic_Demos-Enroll_Season_Winter',\n 'CGAS-Season_StartMonth',\n 'CGAS-Season_Spring',\n 'CGAS-Season_Summer',\n 'CGAS-Season_Fall',\n 'CGAS-Season_Winter',\n 'CGAS-Season_MonthDifference',\n 'Physical-Season_StartMonth',\n 'Physical-Season_Spring',\n 'Physical-Season_Summer',\n 'Physical-Season_Fall',\n 'Physical-Season_Winter',\n 'Physical-Season_MonthDifference',\n 'Fitness_Endurance-Season_StartMonth',\n 'Fitness_Endurance-Season_Spring',\n 'Fitness_Endurance-Season_Summer',\n 'Fitness_Endurance-Season_Fall',\n 'Fitness_Endurance-Season_Winter',\n 'Fitness_Endurance-Season_MonthDifference',\n 'FGC-Season_StartMonth',\n 'FGC-Season_Spring',\n 'FGC-Season_Summer',\n 'FGC-Season_Fall',\n 'FGC-Season_Winter',\n 'FGC-Season_MonthDifference',\n 'BIA-Season_StartMonth',\n 'BIA-Season_Spring',\n 'BIA-Season_Summer',\n 'BIA-Season_Fall',\n 'BIA-Season_Winter',\n 'BIA-Season_MonthDifference',\n 'PAQ_A-Season_StartMonth',\n 'PAQ_A-Season_Spring',\n 'PAQ_A-Season_Summer',\n 'PAQ_A-Season_Fall',\n 'PAQ_A-Season_Winter',\n 'PAQ_A-Season_MonthDifference',\n 'PAQ_C-Season_StartMonth',\n 'PAQ_C-Season_Spring',\n 'PAQ_C-Season_Summer',\n 'PAQ_C-Season_Fall',\n 'PAQ_C-Season_Winter',\n 'PAQ_C-Season_MonthDifference',\n 'SDS-Season_StartMonth',\n 'SDS-Season_Spring',\n 'SDS-Season_Summer',\n 'SDS-Season_Fall',\n 'SDS-Season_Winter',\n 'SDS-Season_MonthDifference',\n 'PreInt_EduHx-Season_StartMonth',\n 'PreInt_EduHx-Season_Spring',\n 'PreInt_EduHx-Season_Summer',\n 'PreInt_EduHx-Season_Fall',\n 'PreInt_EduHx-Season_Winter',\n 'PreInt_EduHx-Season_MonthDifference']\n\n# Combine with any time series columns\nfeaturesCols += season_features + engineered_features + time_series_cols \n\n# Apply to train and test data\ntrain = train[featuresCols]\ntrain = train.dropna(subset=['sii'])\n\n# Apply the same features to the test set without the 'sii' column\nfeaturesCols.remove('sii')  # Remove 'sii' from feature columns for the test set\ntest = test[featuresCols]","metadata":{"execution":{"iopub.status.busy":"2024-11-08T07:20:27.187944Z","iopub.execute_input":"2024-11-08T07:20:27.188467Z","iopub.status.idle":"2024-11-08T07:20:27.218594Z","shell.execute_reply.started":"2024-11-08T07:20:27.188414Z","shell.execute_reply":"2024-11-08T07:20:27.217198Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2024-11-08T07:20:27.220672Z","iopub.execute_input":"2024-11-08T07:20:27.221174Z","iopub.status.idle":"2024-11-08T07:20:27.563092Z","shell.execute_reply.started":"2024-11-08T07:20:27.221124Z","shell.execute_reply":"2024-11-08T07:20:27.561793Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2024-11-08T07:20:27.564901Z","iopub.execute_input":"2024-11-08T07:20:27.565447Z","iopub.status.idle":"2024-11-08T07:20:28.050783Z","shell.execute_reply.started":"2024-11-08T07:20:27.565389Z","shell.execute_reply":"2024-11-08T07:20:28.049561Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T07:20:28.052462Z","iopub.execute_input":"2024-11-08T07:20:28.052839Z","iopub.status.idle":"2024-11-08T07:20:28.070108Z","shell.execute_reply.started":"2024-11-08T07:20:28.052800Z","shell.execute_reply":"2024-11-08T07:20:28.068381Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_bckp = train\ntest_bckp = test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T07:20:28.071946Z","iopub.execute_input":"2024-11-08T07:20:28.072415Z","iopub.status.idle":"2024-11-08T07:20:28.078675Z","shell.execute_reply.started":"2024-11-08T07:20:28.072369Z","shell.execute_reply":"2024-11-08T07:20:28.077060Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import cohen_kappa_score, make_scorer\nfrom xgboost import XGBRegressor\nfrom sklearn.impute import SimpleImputer\n\ndef apply_thresholds(predictions, thresholds=[0.5, 1.5, 2.5]):\n    labels = np.zeros_like(predictions)\n    labels[predictions >= thresholds[0]] = 1\n    labels[predictions >= thresholds[1]] = 2\n    labels[predictions >= thresholds[2]] = 3\n    return labels\n\n# Define the custom kappa scorer using the thresholded predictions\ndef custom_kappa_scorer(y_true, y_pred):\n    y_pred_labels = apply_thresholds(y_pred)  # Apply thresholds to predictions\n    return cohen_kappa_score(y_true, y_pred_labels, weights=\"quadratic\")\n\n# Create a custom scorer for TPOT\nkappa_scorer = make_scorer(custom_kappa_scorer, greater_is_better=True)\n\nX = train_bckp.drop(['sii'], axis=1)\ny = train_bckp['sii']\ntraining_target = y\n\nimputer = SimpleImputer(strategy=\"median\")\nimputer.fit(X)\ntraining_features = imputer.transform(X)\ntesting_features = imputer.transform(test_bckp)\n\n# Average CV score on the training set was: 0.4926664874087134\nexported_pipeline = XGBRegressor(learning_rate=0.1, max_depth=7, min_child_weight=14, n_estimators=100, n_jobs=1, objective=\"reg:squarederror\", subsample=0.5, verbosity=0)\n# Fix random state in exported estimator\nif hasattr(exported_pipeline, 'random_state'):\n    setattr(exported_pipeline, 'random_state', 42)\n\nexported_pipeline.fit(training_features, training_target)\ny_pred = exported_pipeline.predict(testing_features)\ny_pred_train = exported_pipeline.predict(training_features)\nprint(\"Train Quadratic Weighted Kappa Score:\", custom_kappa_scorer(y, y_pred_train))\nSubmission0 = apply_thresholds(y_pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T07:20:28.080576Z","iopub.execute_input":"2024-11-08T07:20:28.081194Z","iopub.status.idle":"2024-11-08T07:20:31.161013Z","shell.execute_reply.started":"2024-11-08T07:20:28.081130Z","shell.execute_reply":"2024-11-08T07:20:31.159351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission0 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission0\n})\n\nSubmission0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T07:20:31.162704Z","iopub.execute_input":"2024-11-08T07:20:31.163153Z","iopub.status.idle":"2024-11-08T07:20:31.179767Z","shell.execute_reply.started":"2024-11-08T07:20:31.163070Z","shell.execute_reply":"2024-11-08T07:20:31.178392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Submission0.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T07:20:31.181648Z","iopub.execute_input":"2024-11-08T07:20:31.182264Z","iopub.status.idle":"2024-11-08T07:20:31.191322Z","shell.execute_reply.started":"2024-11-08T07:20:31.182215Z","shell.execute_reply":"2024-11-08T07:20:31.190005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"execution":{"iopub.status.busy":"2024-11-08T07:20:31.193019Z","iopub.execute_input":"2024-11-08T07:20:31.193535Z","iopub.status.idle":"2024-11-08T07:20:31.215899Z","shell.execute_reply.started":"2024-11-08T07:20:31.193488Z","shell.execute_reply":"2024-11-08T07:20:31.214535Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n    # 'device': 'gpu'\n    'device': 'cpu'\n}\n\n\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n    # 'tree_method': 'gpu_hist',\n    'tree_method': 'hist'\n}\n\n\nCatBoost_Params = {\n    'random_seed': SEED,\n    'iterations': 804,\n    'learning_rate': 0.007849710402582562, \n    'l2_leaf_reg': 7.31183636902306, \n    'subsample': 0.5630297785016092, \n    'random_strength': 1.7097065892440113, \n    'bagging_temperature': 0.026593521316435192,\n    'border_count': 12\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])","metadata":{"execution":{"iopub.status.busy":"2024-11-08T07:20:31.217844Z","iopub.execute_input":"2024-11-08T07:20:31.219205Z","iopub.status.idle":"2024-11-08T07:20:31.237193Z","shell.execute_reply.started":"2024-11-08T07:20:31.219136Z","shell.execute_reply":"2024-11-08T07:20:31.235330Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission1 = TrainML(voting_model, test)\n\n# Save submission\nSubmission1.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T07:20:31.239813Z","iopub.execute_input":"2024-11-08T07:20:31.240639Z","iopub.status.idle":"2024-11-08T07:22:04.499976Z","shell.execute_reply.started":"2024-11-08T07:20:31.240564Z","shell.execute_reply":"2024-11-08T07:22:04.498414Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission1","metadata":{"execution":{"iopub.status.busy":"2024-11-08T07:22:04.501669Z","iopub.execute_input":"2024-11-08T07:22:04.502168Z","iopub.status.idle":"2024-11-08T07:22:04.519202Z","shell.execute_reply.started":"2024-11-08T07:22:04.502120Z","shell.execute_reply":"2024-11-08T07:22:04.517552Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n\n# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n    # 'device': 'gpu'\n    'device': 'cpu'\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n    # 'tree_method': 'gpu_hist',\n    'tree_method': 'hist',\n}\n\n\nCatBoost_Params = {\n    'random_seed': SEED,\n    'cat_features': cat_c,\n    'iterations': 804,\n    'learning_rate': 0.007849710402582562, \n    'l2_leaf_reg': 7.31183636902306, \n    'subsample': 0.5630297785016092, \n    'random_strength': 1.7097065892440113, \n    'bagging_temperature': 0.026593521316435192,\n    'border_count': 12\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n\n# Train the ensemble model\nSubmission2 = TrainML(voting_model, test)\n\n# Save submission\n#Submission2.to_csv('submission.csv', index=False)\nSubmission2","metadata":{"execution":{"iopub.status.busy":"2024-11-08T07:22:04.522561Z","iopub.execute_input":"2024-11-08T07:22:04.523294Z","iopub.status.idle":"2024-11-08T07:25:19.608975Z","shell.execute_reply.started":"2024-11-08T07:22:04.523229Z","shell.execute_reply":"2024-11-08T07:25:19.607463Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    return tp_rounded\n\nimputer = SimpleImputer(strategy='median')\n\nensemble = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),\n    ('xgb', Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=SEED))])),\n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))]))\n])\n\nSubmission3 = TrainML(ensemble, test)","metadata":{"execution":{"iopub.status.busy":"2024-11-08T07:25:19.612050Z","iopub.execute_input":"2024-11-08T07:25:19.612682Z","iopub.status.idle":"2024-11-08T07:29:43.857153Z","shell.execute_reply.started":"2024-11-08T07:25:19.612621Z","shell.execute_reply":"2024-11-08T07:29:43.855681Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission3 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission3\n})\n\nSubmission3","metadata":{"execution":{"iopub.status.busy":"2024-11-08T07:29:43.859192Z","iopub.execute_input":"2024-11-08T07:29:43.859744Z","iopub.status.idle":"2024-11-08T07:29:43.875034Z","shell.execute_reply.started":"2024-11-08T07:29:43.859686Z","shell.execute_reply":"2024-11-08T07:29:43.873749Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub0 = Submission0\nsub1 = Submission1\nsub2 = Submission2\nsub3 = Submission3\n\nsub0 = sub0.sort_values(by='id').reset_index(drop=True)\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_0': sub0['sii'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii']\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_0', 'sii_3']].apply(majority_vote, axis=1)\n\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\nfinal_submission.to_csv('submission.csv', index=False)\n\nprint(\"Majority voting completed and saved to 'Final_Submission.csv'\")","metadata":{"execution":{"iopub.status.busy":"2024-11-08T07:29:43.876696Z","iopub.execute_input":"2024-11-08T07:29:43.877177Z","iopub.status.idle":"2024-11-08T07:29:43.907505Z","shell.execute_reply.started":"2024-11-08T07:29:43.877127Z","shell.execute_reply":"2024-11-08T07:29:43.906271Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_submission","metadata":{"execution":{"iopub.status.busy":"2024-11-08T07:29:43.909398Z","iopub.execute_input":"2024-11-08T07:29:43.909869Z","iopub.status.idle":"2024-11-08T07:29:43.929783Z","shell.execute_reply.started":"2024-11-08T07:29:43.909824Z","shell.execute_reply":"2024-11-08T07:29:43.928298Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from tpot import TPOTRegressor\n# from sklearn.metrics import cohen_kappa_score, make_scorer\n# from sklearn.model_selection import train_test_split\n# import numpy as np\n\n# def apply_thresholds(predictions, thresholds=[0.5, 1.5, 2.5]):\n#     labels = np.zeros_like(predictions)\n#     labels[predictions >= thresholds[0]] = 1\n#     labels[predictions >= thresholds[1]] = 2\n#     labels[predictions >= thresholds[2]] = 3\n#     return labels\n\n# # Define the custom kappa scorer using the thresholded predictions\n# def custom_kappa_scorer(y_true, y_pred):\n#     y_pred_labels = apply_thresholds(y_pred)  # Apply thresholds to predictions\n#     return cohen_kappa_score(y_true, y_pred_labels, weights=\"quadratic\")\n\n# # Create a custom scorer for TPOT\n# kappa_scorer = make_scorer(custom_kappa_scorer, greater_is_better=True)\n\n# X = train_bckp.drop(['sii'], axis=1)\n# y = train_bckp['sii']\n# # Split the dataset\n# X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# # Initialize TPOTRegressor with the custom kappa scorer\n# tpot = TPOTRegressor(generations=5, population_size=50, verbosity=2, scoring=kappa_scorer, random_state=42)\n\n# # Fit TPOT to the training data\n# tpot.fit(X_train, y_train)\n\n# # Evaluate on the test set\n# y_pred = tpot.predict(X_test)\n# y_pred_labels = apply_thresholds(y_pred)\n# print(\"Test Quadratic Weighted Kappa Score:\", custom_kappa_scorer(y_test, y_pred_labels))\n\n# # Export the best pipeline\n# tpot.export('best_pipeline.py')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-08T07:29:43.931520Z","iopub.execute_input":"2024-11-08T07:29:43.932196Z","iopub.status.idle":"2024-11-08T07:29:43.943464Z","shell.execute_reply.started":"2024-11-08T07:29:43.932131Z","shell.execute_reply":"2024-11-08T07:29:43.941920Z"}},"outputs":[],"execution_count":null}]}