{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import libraries\nimport numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\n\nfrom sklearn.preprocessing import MinMaxScaler, StandardScaler, RobustScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.model_selection import train_test_split\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nSEED = 42\nn_splits = 5\n\nimport optuna\nimport warnings\nimport tensorflow as tf\nimport time\nfrom sklearn import preprocessing\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.metrics import mean_squared_error\nfrom concurrent.futures import ThreadPoolExecutor\nfrom statsmodels.stats.outliers_influence import variance_inflation_factor\nfrom sklearn.ensemble import StackingRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.neural_network import MLPRegressor\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.linear_model import ElasticNet\nfrom sklearn.ensemble import HistGradientBoostingRegressor\nfrom sklearn.manifold import TSNE\nfrom sklearn.linear_model import Ridge\nfrom joblib import Parallel, delayed\nfrom sklearn.cluster import DBSCAN\nimport polars as pl\nimport polars.selectors as cs","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:32.559378Z","iopub.execute_input":"2024-12-13T03:27:32.560404Z","iopub.status.idle":"2024-12-13T03:27:33.777962Z","shell.execute_reply.started":"2024-12-13T03:27:32.560361Z","shell.execute_reply":"2024-12-13T03:27:33.774734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.778810Z","iopub.status.idle":"2024-12-13T03:27:33.779195Z","shell.execute_reply.started":"2024-12-13T03:27:33.779023Z","shell.execute_reply":"2024-12-13T03:27:33.779040Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sparse Autoencoder Model\nclass SparseAutoencoder(nn.Module):\n    def __init__(self, input_dim, sparsity_weight=1e-5):\n        super(SparseAutoencoder, self).__init__()\n        self.sparsity_weight = sparsity_weight\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, 64),\n            nn.ReLU(),\n            nn.Linear(64, 32),\n            nn.ReLU(),\n            nn.Linear(32, 16),\n            nn.ReLU()\n        )\n        \n        self.decoder = nn.Sequential(\n            nn.Linear(16, 32),\n            nn.ReLU(),\n            nn.Linear(32, 64),\n            nn.ReLU(),\n            nn.Linear(64, input_dim),\n            nn.Sigmoid()  # Outputs in the range [0, 1]\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return encoded, decoded\n\n# Preparing Data\n# Option to use different scalers: MinMaxScaler, StandardScaler, RobustScaler\ndef prepare_data(data, scaler_type='MinMaxScaler'):\n    if scaler_type == 'StandardScaler':\n        scaler = StandardScaler()\n    elif scaler_type == 'RobustScaler':\n        scaler = RobustScaler()\n    else:\n        scaler = MinMaxScaler()\n    \n    data_scaled = scaler.fit_transform(data)\n    return torch.tensor(data_scaled, dtype=torch.float32), scaler\n\n# Apply PCA for Dimensionality Reduction\n# This can help focus the autoencoder on the most relevant features\ndef apply_pca(data, n_components=0.95):\n    pca = PCA(n_components=n_components)\n    data_pca = pca.fit_transform(data)\n    return data_pca, pca\n\n# Early Stopping Functionality\ndef early_stopping(patience):\n    class EarlyStopping:\n        def __init__(self, patience=patience):\n            self.patience = patience\n            self.counter = 0\n            self.best_loss = float('inf')\n            self.early_stop = False\n        \n        def __call__(self, loss):\n            if loss < self.best_loss:\n                self.best_loss = loss\n                self.counter = 0\n            else:\n                self.counter += 1\n                if self.counter >= self.patience:\n                    self.early_stop = True\n    return EarlyStopping()\n\n# Training the Sparse Autoencoder with DataFrame Output\ndef perform_autoencoder(data, epochs=100, batch_size=32, learning_rate=0.001, patience=10, scaler_type='MinMaxScaler', use_pca=False, sparsity_weight=1e-5):\n    # Preprocess Data\n    if use_pca:\n        data, pca = apply_pca(data)\n\n    data_tensor, scaler = prepare_data(data, scaler_type=scaler_type)\n    train_data, val_data = train_test_split(data_tensor, test_size=0.2, random_state=42)\n\n    train_loader = DataLoader(TensorDataset(train_data), batch_size=batch_size, shuffle=True)\n    val_loader = DataLoader(TensorDataset(val_data), batch_size=batch_size, shuffle=False)\n\n    model = SparseAutoencoder(input_dim=data.shape[1], sparsity_weight=sparsity_weight)\n    device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n    model.to(device)\n\n    criterion = nn.SmoothL1Loss()  # Changed to Smooth L1 Loss\n    optimizer = optim.Adam(model.parameters(), lr=learning_rate)\n    stopper = early_stopping(patience=patience)\n\n    for epoch in range(epochs):\n        model.train()\n        train_loss = 0.0\n        for batch in train_loader:\n            batch = batch[0].to(device)\n            optimizer.zero_grad()\n            encoded, outputs = model(batch)\n            \n            # Reconstruction loss\n            loss = criterion(outputs, batch)\n            \n            # Sparsity penalty (L1 regularization on encoded activations)\n            l1_penalty = torch.mean(torch.abs(encoded))\n            loss += sparsity_weight * l1_penalty\n            \n            loss.backward()\n            optimizer.step()\n            train_loss += loss.item() * batch.size(0)\n\n        train_loss /= len(train_loader.dataset)\n\n        # Validation\n        model.eval()\n        val_loss = 0.0\n        with torch.no_grad():\n            for batch in val_loader:\n                batch = batch[0].to(device)\n                _, outputs = model(batch)\n                loss = criterion(outputs, batch)\n                val_loss += loss.item() * batch.size(0)\n\n        val_loss /= len(val_loader.dataset)\n        print(f\"Epoch {epoch+1}, Train Loss: {train_loss:.4f}, Validation Loss: {val_loss:.4f}\")\n\n        # Early stopping\n        stopper(val_loss)\n        if stopper.early_stop:\n            print(f\"Early stopping at epoch {epoch + 1}\")\n            break\n\n    # Convert tensor back to DataFrame for consistency\n    _, data_decoded = model(data_tensor.to(device))\n    data_decoded = data_decoded.cpu().detach().numpy()\n    df_encoded = pd.DataFrame(data_decoded, columns=[f'feature_{i}' for i in range(data_decoded.shape[1])])\n    return df_encoded","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.780359Z","iopub.status.idle":"2024-12-13T03:27:33.780736Z","shell.execute_reply.started":"2024-12-13T03:27:33.780561Z","shell.execute_reply":"2024-12-13T03:27:33.780578Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    df['hoursday_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] / df['BMI_Age']\n\n    df['Age_Weight'] = df['Basic_Demos-Age'] * df['Physical-Weight']\n    df['Sex_BMI'] = df['Basic_Demos-Sex'] * df['Physical-BMI']\n    df['Sex_HeartRate'] = df['Basic_Demos-Sex'] * df['Physical-HeartRate']\n    df['Age_WaistCirc'] = df['Basic_Demos-Age'] * df['Physical-Waist_Circumference']\n    df['BMI_FitnessMaxStage'] = df['Physical-BMI'] * df['Fitness_Endurance-Max_Stage']\n    df['Weight_GripStrengthDominant'] = df['Physical-Weight'] * df['FGC-FGC_GSD']\n    df['Weight_GripStrengthNonDominant'] = df['Physical-Weight'] * df['FGC-FGC_GSND']\n    df['HeartRate_FitnessTime'] = df['Physical-HeartRate'] * (df['Fitness_Endurance-Time_Mins'] + df['Fitness_Endurance-Time_Sec'])\n    df['Age_PushUp'] = df['Basic_Demos-Age'] * df['FGC-FGC_PU']\n    df['FFMI_Age'] = df['BIA-BIA_FFMI'] * df['Basic_Demos-Age']\n    df['InternetUse_SleepDisturbance'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['SDS-SDS_Total_Raw']\n    df['CGAS_BMI'] = df['CGAS-CGAS_Score'] * df['Physical-BMI']\n    df['CGAS_FitnessMaxStage'] = df['CGAS-CGAS_Score'] * df['Fitness_Endurance-Max_Stage']\n    \n    return df\n\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train, epochs=100, batch_size=32, learning_rate=0.001, patience=10, use_pca=False, scaler_type='MinMaxScaler', sparsity_weight=1e-5)\ntest_ts_encoded = perform_autoencoder(df_test, epochs=100, batch_size=32, learning_rate=0.001, patience=10, use_pca=False, scaler_type='MinMaxScaler', sparsity_weight=1e-5)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded['id']=test_ts[\"id\"]\n\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.782219Z","iopub.status.idle":"2024-12-13T03:27:33.782611Z","shell.execute_reply.started":"2024-12-13T03:27:33.782437Z","shell.execute_reply":"2024-12-13T03:27:33.782456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.drop('id', axis=1)\ntest_id = test[\"id\"]  # for submit\ntest = test.drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.783664Z","iopub.status.idle":"2024-12-13T03:27:33.784022Z","shell.execute_reply.started":"2024-12-13T03:27:33.783846Z","shell.execute_reply":"2024-12-13T03:27:33.783871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n        \ntrain = train_imputed\n\ntrain = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.785352Z","iopub.status.idle":"2024-12-13T03:27:33.785688Z","shell.execute_reply.started":"2024-12-13T03:27:33.785526Z","shell.execute_reply":"2024-12-13T03:27:33.785542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"featuresCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW','Age_Weight','Sex_BMI','Sex_HeartRate','Age_WaistCirc','BMI_FitnessMaxStage','Weight_GripStrengthDominant','Weight_GripStrengthNonDominant','HeartRate_FitnessTime',\n'Age_PushUp','FFMI_Age','InternetUse_SleepDisturbance','CGAS_BMI','CGAS_FitnessMaxStage']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW','Age_Weight','Sex_BMI','Sex_HeartRate','Age_WaistCirc','BMI_FitnessMaxStage','Weight_GripStrengthDominant','Weight_GripStrengthNonDominant','HeartRate_FitnessTime',\n'Age_PushUp','FFMI_Age','InternetUse_SleepDisturbance','CGAS_BMI','CGAS_FitnessMaxStage']\n\nfeaturesCols += time_series_cols\ntest = test[featuresCols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.787140Z","iopub.status.idle":"2024-12-13T03:27:33.787511Z","shell.execute_reply.started":"2024-12-13T03:27:33.787330Z","shell.execute_reply":"2024-12-13T03:27:33.787354Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.788786Z","iopub.status.idle":"2024-12-13T03:27:33.789153Z","shell.execute_reply.started":"2024-12-13T03:27:33.788982Z","shell.execute_reply":"2024-12-13T03:27:33.789000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.790421Z","iopub.status.idle":"2024-12-13T03:27:33.790777Z","shell.execute_reply.started":"2024-12-13T03:27:33.790604Z","shell.execute_reply":"2024-12-13T03:27:33.790623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)\ntrain_df_le = train\ntest_df_le = test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.792205Z","iopub.status.idle":"2024-12-13T03:27:33.792584Z","shell.execute_reply.started":"2024-12-13T03:27:33.792410Z","shell.execute_reply":"2024-12-13T03:27:33.792429Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.cluster import DBSCAN\nfrom sklearn.manifold import TSNE\nfrom mpl_toolkits.mplot3d import Axes3D\n\n# DBSCAN clustering\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.cluster import DBSCAN\nfrom sklearn.manifold import TSNE\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Save the 'sii' column\ntrain_sii = train_df_le['sii'].copy()\n\n# Find common columns between train and test data\ncommon_columns = train_df_le.columns.intersection(test_df_le.columns)\ntrain_df_le = train_df_le[common_columns]\ntest_df_le = test_df_le[common_columns]\ntrain_df_le['sii'] = train_sii\n\n# Pipeline: Fill missing values with the mean\npreprocessor = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='mean')), \n])\n\ny = train_df_le['sii']\ntrain_df_le = train_df_le.drop(columns=['sii'])\ntrain_df_le = preprocessor.fit_transform(train_df_le)\ntrain_df_le = pd.DataFrame(train_df_le, columns=common_columns)\ntrain_df_le['sii'] = y\n\n# Standardize numerical columns\nnumeric_cols = train_df_le.select_dtypes(include=[np.number]).columns  \nnumeric_cols = numeric_cols.drop('sii')  # Exclude 'sii'\nscaler = StandardScaler()\ntrain_df_le[numeric_cols] = scaler.fit_transform(train_df_le[numeric_cols])\n\n# Preprocess the test data\ntest_df_le = preprocessor.transform(test_df_le)\ntest_df_le = pd.DataFrame(test_df_le, columns=common_columns)\ntest_df_le[numeric_cols] = scaler.transform(test_df_le[numeric_cols])\n\n# Perform DBSCAN clustering\ndbscan = DBSCAN(eps=0.5, min_samples=5)  \ntrain_clusters = dbscan.fit_predict(train_df_le)\n\nprint(f\"Unique clusters: {len(np.unique(train_clusters[train_clusters != -1]))}\")\nprint(f\"Noise points: {np.sum(train_clusters == -1)}\")\n\n# Find the mode of 'sii' for each cluster\ncluster_modes = pd.Series(index=np.unique(train_clusters), dtype=float)\n\nfor cluster_num in np.unique(train_clusters):  \n    cluster_rows = train_df_le[train_clusters == cluster_num]\n    cluster_sii = y[train_clusters == cluster_num]  # 'sii'\n    cluster_modes[cluster_num] = cluster_sii.mode()[0] \n\nprint(\"Cluster 'sii' modes:\")\nprint(cluster_modes)\n\n# Fill missing 'sii' values based on the cluster modes\ntrain_df_le.loc[train_df_le['sii'].isnull(), 'sii'] = [\n    cluster_modes[train_clusters[i]] for i in range(len(train_clusters))\n]\nX = train_df_le\n\n# Perform t-SNE to reduce dimensions to 3D for visualization\ntsne = TSNE(n_components=3, random_state=42, perplexity=30, n_iter=300)\ntrain_df_le_tsne = tsne.fit_transform(train_df_le)\n\n# DBSCAN clustering results (-1 indicates noise)\ntrain_clusters = dbscan.labels_\n\n# Set up the plot size and layout (4 subplots: 3D, 2D, Bar Chart)\nfig, axes = plt.subplots(2, 2, figsize=(18, 12), dpi=80, subplot_kw={'projection': None})\n\n# Define custom color palette for clusters\nnum_clusters = len(np.unique(train_clusters)) - 1  # Exclude noise (-1)\npalette = sns.color_palette(\"Set3\", n_colors=num_clusters)\n\n# Plot 1: Clusters (2D scatter using t-SNE components 1 & 2)\nax1 = axes[0, 0]\nfor cluster_num in np.unique(train_clusters):\n    if cluster_num == -1:  # Noise points\n        ax1.scatter(train_df_le_tsne[train_clusters == cluster_num, 0], \n                    train_df_le_tsne[train_clusters == cluster_num, 1], \n                    color='lightgray', label='Noise', s=30, edgecolor='k', marker='X')\n    else:\n        ax1.scatter(train_df_le_tsne[train_clusters == cluster_num, 0], \n                    train_df_le_tsne[train_clusters == cluster_num, 1], \n                    color=palette[cluster_num % len(palette)], \n                    label=f'Cluster {cluster_num}', s=80, edgecolor='k')\n\nax1.set_title('DBSCAN Clustering (t-SNE 2D)', fontsize=16, weight='bold')\nax1.set_xlabel('t-SNE Component 1', fontsize=14)\nax1.set_ylabel('t-SNE Component 2', fontsize=14)\n\n# Plot 2: Clusters Only (3D scatter)\nax2 = fig.add_subplot(222, projection='3d')\nfor cluster_num in np.unique(train_clusters):\n    if cluster_num != -1:  # Exclude Noise points\n        ax2.scatter(train_df_le_tsne[train_clusters == cluster_num, 0], \n                    train_df_le_tsne[train_clusters == cluster_num, 1], \n                    train_df_le_tsne[train_clusters == cluster_num, 2], \n                    color=palette[cluster_num % len(palette)], \n                    label=f'Cluster {cluster_num}', s=80, edgecolor='k')\n\nax2.set_title('Clusters Without Noise (t-SNE 3D)', fontsize=16, weight='bold')\nax2.set_xlabel('t-SNE Component 1', fontsize=14)\nax2.set_ylabel('t-SNE Component 2', fontsize=14)\nax2.set_zlabel('t-SNE Component 3', fontsize=14)\n\n# Bar chart: Cluster Sizes\nax3 = axes[1, 0]\ncluster_sizes = pd.Series(train_clusters).value_counts().sort_index()\ncluster_sizes = cluster_sizes[cluster_sizes.index != -1]  # Exclude Noise\n\nax3.bar(cluster_sizes.index, cluster_sizes.values, color=palette[:len(cluster_sizes)])\nax3.set_title('Cluster Sizes', fontsize=16, weight='bold')\nax3.set_xlabel('Cluster Number', fontsize=14)\nax3.set_ylabel('Number of Points', fontsize=14)\n\n# Remove the unnecessary fourth plot (bottom-right)\nfig.delaxes(axes[1, 1])\n\n# Add legend\nhandles, labels = ax1.get_legend_handles_labels()\nfig.legend(handles, labels, loc='center left', bbox_to_anchor=(1, 0.5), fontsize=12, title=\"Clusters\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.802503Z","iopub.status.idle":"2024-12-13T03:27:33.803036Z","shell.execute_reply.started":"2024-12-13T03:27:33.802754Z","shell.execute_reply":"2024-12-13T03:27:33.802781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 'sii' \ny = X['sii']\nX = X.drop(columns=['sii'])\nX.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.804659Z","iopub.status.idle":"2024-12-13T03:27:33.805196Z","shell.execute_reply.started":"2024-12-13T03:27:33.804926Z","shell.execute_reply":"2024-12-13T03:27:33.804954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert numpy.ndarray to pandas DataFrame\ntest_df_le = pd.DataFrame(test_df_le)\n\n# Now you can use .info() to inspect the DataFrame\ntest_df_le.info()\n# test_df_le.columns = X.columns\nX_test = test_df_le","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.806497Z","iopub.status.idle":"2024-12-13T03:27:33.807025Z","shell.execute_reply.started":"2024-12-13T03:27:33.806747Z","shell.execute_reply":"2024-12-13T03:27:33.806773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model parameters for LightGBM\nLGBM_Params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  # Increase this value\n    # 'cat_features': cat_c,\n\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.808709Z","iopub.status.idle":"2024-12-13T03:27:33.809230Z","shell.execute_reply.started":"2024-12-13T03:27:33.808963Z","shell.execute_reply":"2024-12-13T03:27:33.808990Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import time\nimport optuna\nimport numpy as np\nfrom scipy.optimize import minimize\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import KFold\nfrom sklearn.ensemble import StackingRegressor\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.preprocessing import StandardScaler\nimport torch\nfrom torch import nn, optim\nimport pandas as pd\n\n# PyTorch Model Definition\nclass PyTorchRegressor(BaseEstimator, RegressorMixin):\n    def __init__(self, input_dim=None, hidden_dim=128, epochs=100, batch_size=32, learning_rate=1e-4, seed=42, device='cpu'):\n        # Hyperparameters initialization\n        self.hidden_dim = hidden_dim\n        self.epochs = epochs\n        self.batch_size = batch_size\n        self.learning_rate = learning_rate\n        self.seed = seed\n        self.device = device\n        \n        # Set random seed for reproducibility\n        torch.manual_seed(self.seed)\n        np.random.seed(self.seed)\n        \n        # Initialize model and input dimension\n        self.model = None\n        self.input_dim = input_dim  # Set input dimension dynamically\n        \n        # Variables to store the scaler and target variable scaling\n        self.scaler = StandardScaler()\n        self.y_mean = None\n        self.y_std = None\n\n    def _build_model(self):\n        # Define the architecture of the model\n        model = nn.Sequential(\n            nn.Linear(self.input_dim, self.hidden_dim),  # Input layer\n            nn.ReLU(),\n            nn.Dropout(p=0.3),\n            nn.Linear(self.hidden_dim, self.hidden_dim),  # Hidden layer\n            nn.ReLU(),\n            nn.Dropout(p=0.3),\n            nn.Linear(self.hidden_dim, 1)  # Output layer (for regression task, 1 output)\n        )\n        return model.to(self.device)\n\n    def fit(self, X, y):\n        # Automatically set input dimension based on the input data shape\n        self.input_dim = X.shape[1]  # Dynamically obtain the number of features\n        self.model = self._build_model()  # Rebuild the model with the correct input dimension\n        \n        self.model.train()  # Set the model to training mode\n        \n        # Normalize input data\n        X_scaled = self.scaler.fit_transform(X)  # Standardize the input features\n        X_tensor = torch.tensor(X_scaled, dtype=torch.float32).to(self.device)\n        \n        # Normalize target variable\n        self.y_mean, self.y_std = y.mean(), y.std()  # Store the mean and std of y\n        y_scaled = (y - self.y_mean) / self.y_std  # Standardize the target variable\n        \n        # If y is a Pandas Series, convert it to a NumPy array\n        if isinstance(y_scaled, pd.Series):\n            y_scaled = y_scaled.values\n        \n        y_tensor = torch.tensor(y_scaled, dtype=torch.float32).view(-1, 1).to(self.device)\n\n        # Use DataLoader for batch processing\n        dataset = torch.utils.data.TensorDataset(X_tensor, y_tensor)\n        dataloader = torch.utils.data.DataLoader(dataset, batch_size=self.batch_size, shuffle=True)\n\n        # Training loop\n        self.optimizer = optim.Adam(self.model.parameters(), lr=self.learning_rate, weight_decay=1e-5)\n        self.criterion = nn.MSELoss()\n\n        for epoch in range(self.epochs):\n            epoch_loss = 0.0\n            for batch_X, batch_y in dataloader:\n                self.optimizer.zero_grad()\n                outputs = self.model(batch_X)  # Forward pass\n                loss = self.criterion(outputs, batch_y)  # Compute loss\n                loss.backward()  # Backward pass\n                self.optimizer.step()  # Update parameters using optimizer\n                epoch_loss += loss.item()\n\n            # Print loss every 10 epochs\n            if epoch % 10 == 0:\n                print(f\"Epoch [{epoch}/{self.epochs}], Loss: {epoch_loss/len(dataloader):.4f}\")\n\n        return self\n\n    def predict(self, X):\n        self.model.eval()  # Set the model to evaluation mode\n        with torch.no_grad():  # Disable gradient computation during inference\n            X_scaled = self.scaler.transform(X)  # Standardize input features\n            X_tensor = torch.tensor(X_scaled, dtype=torch.float32).to(self.device)\n                \n            outputs = self.model(X_tensor).squeeze().cpu().numpy()  # Convert output to a NumPy array\n            \n            # Reverse the scaling of the predictions\n            return outputs * self.y_std + self.y_mean  # Undo scaling to return predictions to the original scale\n\n# Initialize the PyTorch model\ninput_dim = 317  # Set this according to your data\npytorch_model = PyTorchRegressor(input_dim=input_dim, \n                                 hidden_dim=128, \n                                 epochs=100, \n                                 batch_size=32, \n                                 learning_rate=1e-4, \n                                 seed=42, \n                                 device='cuda' if torch.cuda.is_available() else 'cpu')\n\n# Define your other base models\nbase_models = [\n    ('lgb', LGBMRegressor(**LGBM_Params, verbose=-1)),\n    ('xgb', XGBRegressor(**XGB_Params)),\n    ('cat', CatBoostRegressor(**CatBoost_Params)),\n    ('pytorch', pytorch_model),  \n]\n\n# Function to optimize QWK score by adjusting thresholds\ndef evaluate_predictions(thresholds, y_true, y_pred):\n    thresholds = np.sort(thresholds)\n    y_pred_classes = np.digitize(y_pred, thresholds)\n    return -cohen_kappa_score(y_true, y_pred_classes, weights='quadratic')\n\n# Initial threshold values\ninitial_thresholds = [0.5, 1.5, 2.5]\n\n# Objective function for Optuna (Meta-model tuning)\ndef objective(trial):\n    # Suggest a value for alpha (for Ridge regularization)\n    alpha = trial.suggest_float('alpha', 1e-4, 10.0, log=True)\n    \n    # Define the Ridge model with the suggested alpha\n    meta_model = Ridge(alpha=alpha, random_state=SEED)\n\n    # 5-fold CV for parameter tuning\n    scores = []\n    for train_idx, valid_idx in KFold(n_splits=5, shuffle=True, random_state=SEED).split(X, y):\n        X_train, X_valid = X.iloc[train_idx], X.iloc[valid_idx]\n        y_train, y_valid = y[train_idx], y[valid_idx]  # 修正: y.iloc -> y[train_idx] と y[valid_idx] に変更\n\n        # Scale the data\n        scaler = StandardScaler()\n        X_train_scaled = scaler.fit_transform(X_train)\n        X_valid_scaled = scaler.transform(X_valid)\n\n        # Train meta model and get predictions\n        meta_model.fit(X_train_scaled, y_train)\n        preds = meta_model.predict(X_valid_scaled)\n\n        # Optimize thresholds for QWK score\n        KappaOptimizer = minimize(evaluate_predictions, x0=initial_thresholds,\n                                  args=(y_valid, preds), method='Nelder-Mead')\n        best_thresholds = np.sort(KappaOptimizer.x)\n        preds_class = np.digitize(preds, best_thresholds)\n\n        # Calculate QWK score\n        qwk_score = cohen_kappa_score(y_valid, preds_class, weights='quadratic')\n        scores.append(qwk_score)\n\n    # Return the mean QWK score (maximize)\n    return np.mean(scores)\n\n\n# Run Optuna for meta model tuning (QWK maximization)\nstudy = optuna.create_study(direction=\"maximize\")\nstudy.optimize(objective, n_trials=200)\nbest_meta_params = study.best_params\n\n# Updated meta model with best parameters\nmeta_model = Ridge(**best_meta_params, random_state=SEED)\n\n# Stacking Regressor with base models and optimized meta model\nstacking_model = StackingRegressor(estimators=base_models, final_estimator=meta_model)\n\n# Cross-validation setup\nkf = KFold(n_splits=10, shuffle=True, random_state=SEED)\npredictions = np.zeros(X.shape[0])  # Initialize prediction array for train set\ntest_predictions = np.zeros(test_df_le.shape[0])  # Initialize prediction array for test set\nqwk_scores = []  # List to store QWK scores for each fold\nmodel_scores = {name: [] for name, _ in base_models}  # Dictionary to store model scores\n\n# Time-tracking start\nstart_time = time.time()\n\n# K-fold Cross-validation loop\nfor train_idx, valid_idx in kf.split(X, y):\n    # Train/Validation split for this fold\n    X_train, X_valid = X.iloc[train_idx], X.iloc[valid_idx] \n    y_train, y_valid = y[train_idx], y[valid_idx]  # Same for y (numpy array)\n\n    # Scale the data\n    scaler = StandardScaler()\n    X_train_scaled = scaler.fit_transform(X_train)  # Scale X_train\n    X_valid_scaled = scaler.transform(X_valid)  # Scale X_valid using the same scaler\n\n    # Base model predictions\n    base_preds = []  # List to store predictions from base models\n    for name, model in base_models:\n        # Fit base models (without early_stopping_rounds for this example)\n        if isinstance(model, LGBMRegressor):\n            model.fit(X_train_scaled, y_train)  # Train the model\n        elif isinstance(model, XGBRegressor):\n            model.fit(X_train_scaled, y_train)  # Train the model\n        elif isinstance(model, CatBoostRegressor):\n            model.fit(X_train_scaled, y_train)  # Train the model\n        elif isinstance(model, PyTorchRegressor):  # Custom PyTorch model\n            model.fit(X_train_scaled, y_train)  # Custom fit method for PyTorch model\n\n        model_preds = model.predict(X_valid_scaled)  # Predict on the validation set\n\n        # Optimize thresholds for each base model\n        KappaOptimizer = minimize(evaluate_predictions, x0=initial_thresholds,\n                                  args=(y_valid, model_preds), method='Nelder-Mead')\n        best_thresholds = np.sort(KappaOptimizer.x)  # Get optimized thresholds\n        model_preds_class = np.digitize(model_preds, best_thresholds)  # Apply thresholds to predictions\n\n        # QWK score and saving predictions\n        model_qwk_score = cohen_kappa_score(y_valid, model_preds_class, weights='quadratic')  # Calculate QWK score\n        model_scores[name].append(model_qwk_score)  # Store the score for each model\n        base_preds.append(model_preds)  # Store base model predictions\n\n    # Fit stacking model with optimized meta model (Ridge)\n    stacking_model.fit(X_train_scaled, y_train)  # Fit the stacking model\n\n    # Predict for validation set and optimize thresholds\n    preds = stacking_model.predict(X_valid_scaled)  # Get stacking model predictions\n    KappaOptimizer = minimize(evaluate_predictions, x0=initial_thresholds,\n                              args=(y_valid, preds), method='Nelder-Mead')\n    best_thresholds = np.sort(KappaOptimizer.x)  # Get optimized thresholds\n    ensemble_preds = np.digitize(preds, best_thresholds)  # Apply thresholds to predictions\n\n    # Calculate QWK score for the ensemble model\n    qwk_score = cohen_kappa_score(y_valid, ensemble_preds, weights='quadratic')\n    qwk_scores.append(qwk_score)  # Store QWK score for this fold\n\n    # Store predictions\n    predictions[valid_idx] += preds  # Accumulate predictions for train set\n    test_predictions += np.digitize(stacking_model.predict(scaler.transform(test_df_le)), best_thresholds) / kf.n_splits  # Accumulate test predictions\n\n# Time-tracking end\nend_time = time.time()\nprint(f\"Training Time: {end_time - start_time:.2f} seconds\")\n\n# Mean QWK score across all folds\nmean_qwk_score = np.mean(qwk_scores)\nprint(f\"Mean QWK score: {mean_qwk_score:.4f}\")\n\n# Now scale the test data using the same scaler fitted on X_train\ntest_scaled = scaler.transform(test_df_le)  # Scale using the fitted scaler\n\n# Make predictions on test data and apply thresholds\ntest_predictions += np.digitize(stacking_model.predict(test_scaled), best_thresholds) / kf.n_splits\n\n# Display total processing time\nelapsed_time = time.time() - start_time\nprint(f\"Total processing time: {elapsed_time:.2f} seconds\")\n\n# Optionally print or store QWK scores and other metrics\nprint(f\"Mean QWK score: {np.mean(qwk_scores)}\")\n\nprint('Individual model scores:')\nfor name, scores in model_scores.items():\n    print(f'{name}: Average QWK = {np.mean(scores):.4f}, Std = {np.std(scores):.4f}')\n\nprint('Average: ', np.mean(qwk_scores))\nprint('Std: ', np.std(qwk_scores))\nprint('Min: ', np.min(qwk_scores))\nprint('Max: ', np.max(qwk_scores))\n\n# Prepare the test predictions for submission　submit\ntest_pred_classes = np.digitize(test_predictions, np.sort(best_thresholds))\nSubmission1 = pd.DataFrame({'id': test_id, 'sii': test_pred_classes.astype(int)})\n\nSubmission1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.811166Z","iopub.status.idle":"2024-12-13T03:27:33.811570Z","shell.execute_reply.started":"2024-12-13T03:27:33.811389Z","shell.execute_reply":"2024-12-13T03:27:33.811408Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n\n# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n\n# Train the ensemble model\nSubmission2 = TrainML(voting_model, test)\n\n# Save submission\n#Submission2.to_csv('submission.csv', index=False)\nSubmission2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.812910Z","iopub.status.idle":"2024-12-13T03:27:33.813281Z","shell.execute_reply.started":"2024-12-13T03:27:33.813107Z","shell.execute_reply":"2024-12-13T03:27:33.813125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    return tp_rounded\n\nimputer = SimpleImputer(strategy='median')\n\nensemble = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),\n    ('xgb', Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=SEED))])),\n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))]))\n])\n\nSubmission3 = TrainML(ensemble, test)\nSubmission3 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission3\n})\n\nSubmission3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.814301Z","iopub.status.idle":"2024-12-13T03:27:33.814676Z","shell.execute_reply.started":"2024-12-13T03:27:33.814506Z","shell.execute_reply":"2024-12-13T03:27:33.814523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub1 = Submission1\nsub2 = Submission2\nsub3 = Submission3\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii'],\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\nfinal_submission.to_csv('submission.csv', index=False)\n\nprint(\"Majority voting completed and saved to 'Final_Submission.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.815892Z","iopub.status.idle":"2024-12-13T03:27:33.816221Z","shell.execute_reply.started":"2024-12-13T03:27:33.816063Z","shell.execute_reply":"2024-12-13T03:27:33.816079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T03:27:33.817406Z","iopub.status.idle":"2024-12-13T03:27:33.817774Z","shell.execute_reply.started":"2024-12-13T03:27:33.817602Z","shell.execute_reply":"2024-12-13T03:27:33.817620Z"}},"outputs":[],"execution_count":null}]}