{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":504.939801,"end_time":"2024-11-02T09:04:06.431596","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-11-02T08:55:41.491795","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"17ca10c0","cell_type":"code","source":"import numpy as np\n\nimport pandas as pd\n\nimport os\n\nimport numpy as np\n\nimport pandas as pd\n\nimport os\n\nimport re\n\nfrom sklearn.base import clone\n\nfrom sklearn.metrics import cohen_kappa_score\n\nfrom sklearn.model_selection import StratifiedKFold\n\nfrom scipy.optimize import minimize\n\nfrom concurrent.futures import ThreadPoolExecutor\n\nfrom tqdm import tqdm\n\nimport polars as pl\n\nimport polars.selectors as cs\n\nimport matplotlib.pyplot as plt\n\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\n\nimport seaborn as sns\n\n\n\nfrom sklearn.preprocessing import StandardScaler\n\nimport matplotlib.pyplot as plt\n\nfrom keras.models import Model\n\nfrom keras.layers import Input, Dense\n\nfrom keras.optimizers import Adam\n\nimport torch\n\nimport torch.nn as nn\n\nimport torch.optim as optim\n\n\n\nfrom colorama import Fore, Style\n\nfrom IPython.display import clear_output\n\nimport warnings\n\nfrom lightgbm import LGBMRegressor\n\nfrom xgboost import XGBRegressor\n\nfrom catboost import CatBoostRegressor\n\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\n\nfrom sklearn.impute import SimpleImputer, KNNImputer\n\nfrom sklearn.pipeline import Pipeline\n\nwarnings.filterwarnings('ignore')\n\npd.options.display.max_columns = None\n\nSEED = 42\n\nn_splits = 5","metadata":{"execution":{"iopub.status.busy":"2024-11-08T15:59:34.650311Z","iopub.execute_input":"2024-11-08T15:59:34.651198Z","iopub.status.idle":"2024-11-08T15:59:34.661782Z","shell.execute_reply.started":"2024-11-08T15:59:34.651156Z","shell.execute_reply":"2024-11-08T15:59:34.660721Z"},"papermill":{"duration":21.538958,"end_time":"2024-11-02T08:56:05.766772","exception":false,"start_time":"2024-11-02T08:55:44.227814","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"d5c5ab25","cell_type":"markdown","source":"# **Prepare First Model**","metadata":{"papermill":{"duration":0.009942,"end_time":"2024-11-02T08:56:05.787497","exception":false,"start_time":"2024-11-02T08:56:05.777555","status":"completed"},"tags":[]}},{"id":"f4daa8c3","cell_type":"markdown","source":"\n\n## 2. `load_time_series(dirname) -> pd.DataFrame`\n\n\n\nLoads time series data from a directory and processes each file concurrently.\n\n\n\n### Parameters:\n\n- **dirname** (str): The directory containing the parquet files.\n\n\n\n### Returns:\n\n- **pd.DataFrame**: A DataFrame containing statistical summaries and their identifiers.\n","metadata":{"papermill":{"duration":0.009887,"end_time":"2024-11-02T08:56:05.807536","exception":false,"start_time":"2024-11-02T08:56:05.797649","status":"completed"},"tags":[]}},{"id":"35cd7ec3","cell_type":"code","source":"def process_file(filename, dirname):\n\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n\n    df.drop('step', axis=1, inplace=True)\n\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\n\n\ndef load_time_series(dirname) -> pd.DataFrame:\n\n    ids = os.listdir(dirname)\n\n    \n\n    with ThreadPoolExecutor() as executor:\n\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n\n    \n\n    stats, indexes = zip(*results)\n\n    \n\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n\n    df['id'] = indexes\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-11-08T15:59:41.927885Z","iopub.execute_input":"2024-11-08T15:59:41.928584Z","iopub.status.idle":"2024-11-08T15:59:41.938785Z","shell.execute_reply.started":"2024-11-08T15:59:41.928538Z","shell.execute_reply":"2024-11-08T15:59:41.937978Z"},"papermill":{"duration":0.020688,"end_time":"2024-11-02T08:56:05.838339","exception":false,"start_time":"2024-11-02T08:56:05.817651","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"11b95b0f","cell_type":"markdown","source":"## 3. `class AutoEncoder`\n\n\n\nAn AutoEncoder model for dimensionality reduction.\n\n\n\n### Attributes:\n\n- **encoder** (nn.Sequential): The encoder part of the AutoEncoder.\n\n- **decoder** (nn.Sequential): The decoder part of the AutoEncoder.\n\n\n\n### Methods:\n\n- `__init__(input_dim, encoding_dim)`: Initializes the AutoEncoder with specified input and encoding dimensions.\n\n- `forward(x)`: Forward pass through the AutoEncoder.\n","metadata":{"papermill":{"duration":0.009777,"end_time":"2024-11-02T08:56:05.858157","exception":false,"start_time":"2024-11-02T08:56:05.848380","status":"completed"},"tags":[]}},{"id":"26fb31fb","cell_type":"code","source":"\n\nclass AutoEncoder(nn.Module):\n\n    def __init__(self, input_dim, encoding_dim):\n\n        super(AutoEncoder, self).__init__()\n\n        self.encoder = nn.Sequential(\n\n            nn.Linear(input_dim, encoding_dim*3),\n\n            nn.ReLU(),\n\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n\n            nn.ReLU(),\n\n            nn.Linear(encoding_dim*2, encoding_dim),\n\n            nn.ReLU()\n\n        )\n\n        self.decoder = nn.Sequential(\n\n            nn.Linear(encoding_dim, input_dim*2),\n\n            nn.ReLU(),\n\n            nn.Linear(input_dim*2, input_dim*3),\n\n            nn.ReLU(),\n\n            nn.Linear(input_dim*3, input_dim),\n\n            nn.Sigmoid()\n\n        )\n\n        \n\n    def forward(self, x):\n\n        encoded = self.encoder(x)\n\n        decoded = self.decoder(encoded)\n\n        return decoded\n","metadata":{"execution":{"iopub.status.busy":"2024-11-08T15:59:45.009351Z","iopub.execute_input":"2024-11-08T15:59:45.009701Z","iopub.status.idle":"2024-11-08T15:59:45.017795Z","shell.execute_reply.started":"2024-11-08T15:59:45.009670Z","shell.execute_reply":"2024-11-08T15:59:45.016796Z"},"papermill":{"duration":0.02006,"end_time":"2024-11-02T08:56:05.888180","exception":false,"start_time":"2024-11-02T08:56:05.868120","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"ff843161","cell_type":"markdown","source":"## 4. `perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32)`\n\n\n\nTrains an AutoEncoder on the provided DataFrame.\n\n\n\n### Parameters:\n\n- **df** (pd.DataFrame): The input DataFrame to be encoded.\n\n- **encoding_dim** (int): The dimensionality of the encoded representation.\n\n- **epochs** (int): The number of training epochs.\n\n- **batch_size** (int): The size of each training batch.\n\n\n\n### Returns:\n\n- **pd.DataFrame**: A DataFrame containing the encoded features.","metadata":{"papermill":{"duration":0.009945,"end_time":"2024-11-02T08:56:05.908145","exception":false,"start_time":"2024-11-02T08:56:05.898200","status":"completed"},"tags":[]}},{"id":"f71df6e9","cell_type":"code","source":"\n\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n\n    scaler = StandardScaler()\n\n    df_scaled = scaler.fit_transform(df)\n\n    \n\n    data_tensor = torch.FloatTensor(df_scaled)\n\n    \n\n    input_dim = data_tensor.shape[1]\n\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n\n    \n\n    criterion = nn.MSELoss()\n\n    optimizer = optim.Adam(autoencoder.parameters())\n\n    \n\n    for epoch in range(epochs):\n\n        for i in range(0, len(data_tensor), batch_size):\n\n            batch = data_tensor[i : i + batch_size]\n\n            optimizer.zero_grad()\n\n            reconstructed = autoencoder(batch)\n\n            loss = criterion(reconstructed, batch)\n\n            loss.backward()\n\n            optimizer.step()\n\n            \n\n        if (epoch + 1) % 10 == 0:\n\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n\n                 \n\n    with torch.no_grad():\n\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n\n        \n\n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n\n    \n\n    return df_encoded","metadata":{"execution":{"iopub.status.busy":"2024-11-08T15:59:48.787285Z","iopub.execute_input":"2024-11-08T15:59:48.787661Z","iopub.status.idle":"2024-11-08T15:59:48.797200Z","shell.execute_reply.started":"2024-11-08T15:59:48.787626Z","shell.execute_reply":"2024-11-08T15:59:48.796244Z"},"papermill":{"duration":0.021032,"end_time":"2024-11-02T08:56:05.939179","exception":false,"start_time":"2024-11-02T08:56:05.918147","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"e04fed32","cell_type":"markdown","source":"## 5. `feature_engineering(df)`\n\n\n\nPerforms feature engineering on the input DataFrame by creating new features and removing unnecessary columns.\n\n\n\n### Parameters:\n\n- **df** (pd.DataFrame): The input DataFrame containing health and demographic data.\n\n\n\n### Returns:\n\n- **pd.DataFrame**: The modified DataFrame with engineered features.\n\n\n\n### Description:\n\nThis function performs the following operations:\n\n1. **Remove Seasonal Columns**: Drops any columns that contain the substring 'Season'.\n\n2. **Create New Features**:\n\n   - `BMI_Age`: The product of physical BMI and age.","metadata":{"papermill":{"duration":0.009801,"end_time":"2024-11-02T08:56:05.959120","exception":false,"start_time":"2024-11-02T08:56:05.949319","status":"completed"},"tags":[]}},{"id":"bb3800bc","cell_type":"code","source":"def feature_engineering(df):\n\n    season_cols = [col for col in df.columns if 'Season' in col]\n\n    df = df.drop(season_cols, axis=1) \n\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n\n    \n\n    return df\n","metadata":{"execution":{"iopub.status.busy":"2024-11-08T15:59:57.347363Z","iopub.execute_input":"2024-11-08T15:59:57.347961Z","iopub.status.idle":"2024-11-08T15:59:57.357284Z","shell.execute_reply.started":"2024-11-08T15:59:57.347921Z","shell.execute_reply":"2024-11-08T15:59:57.356115Z"},"papermill":{"duration":0.021142,"end_time":"2024-11-02T08:56:05.990643","exception":false,"start_time":"2024-11-02T08:56:05.969501","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"b4c097ac","cell_type":"markdown","source":"## `6. Data Loading and Preprocessing`\n\n\n\n1. **Data Loading**: Loads training, test, and sample submission datasets using `pd.read_csv()`.\n\n2. **Time Series Processing**:\n\n   - Loads and processes time series data from Parquet files using the `load_time_series` function.\n\n   - Drops the 'id' column for further processing.\n\n3. **Autoencoding**: \n\n   - Reduces dimensionality of time series data using an autoencoder with specified encoding dimensions, epochs, and batch size.\n\n   - Adds the 'id' column back to the encoded data for merging.\n\n4. **Merging Data**: Merges the encoded time series features with the original training and test datasets based on 'id'.\n\n5. **Missing Value Imputation**:\n\n   - Uses KNN imputer to fill missing values in numeric columns.\n\n   - Rounds the 'sii' column to integers.\n\n   - Retains non-numeric columns after imputation.\n\n6. **Feature Engineering**: \n\n   - Applies the `feature_engineering` function to both the training and test datasets.\n\n7. **Data Cleanup**: \n\n   - Drops rows in the training dataset that have fewer than 10 non-null values.\n\n   - Drops the 'id' column from both datasets after processing.\n\n\n","metadata":{"papermill":{"duration":0.009799,"end_time":"2024-11-02T08:56:06.010473","exception":false,"start_time":"2024-11-02T08:56:06.000674","status":"completed"},"tags":[]}},{"id":"47d2886d","cell_type":"code","source":"\n\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\n\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\n\n\ndf_train = train_ts.drop('id', axis=1)\n\ndf_test = test_ts.drop('id', axis=1)\n\n\n\ntrain_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\n\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)\n\n\n\ntime_series_cols = train_ts_encoded.columns.tolist()\n\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\n\ntest_ts_encoded['id']=test_ts[\"id\"]\n\n\n\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\n\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\n\n\n\nimputer = KNNImputer(n_neighbors=5)\n\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\n\nimputed_data = imputer.fit_transform(train[numeric_cols])\n\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\n\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\n\nfor col in train.columns:\n\n    if col not in numeric_cols:\n\n        train_imputed[col] = train[col]\n\n        \n\ntrain = train_imputed\n\n\n\ntrain = feature_engineering(train)\n\ntrain = train.dropna(thresh=10, axis=0)\n\ntest = feature_engineering(test)\n\ntrain = train.drop('id', axis=1)\n\ntest  = test .drop('id', axis=1)   \n","metadata":{"execution":{"iopub.status.busy":"2024-11-08T16:00:00.518956Z","iopub.execute_input":"2024-11-08T16:00:00.519310Z","iopub.status.idle":"2024-11-08T16:01:43.449690Z","shell.execute_reply.started":"2024-11-08T16:00:00.519277Z","shell.execute_reply":"2024-11-08T16:01:43.448679Z"},"papermill":{"duration":98.454982,"end_time":"2024-11-02T08:57:44.475474","exception":false,"start_time":"2024-11-02T08:56:06.020492","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"26e561e1","cell_type":"markdown","source":"## `7. Feature Selection and Data Preparation`\n\n\n\nThis section selects relevant features for the training and test datasets, ensuring that all necessary variables are included for further analysis or model training.\n\n\n\n1. **Feature Selection**:\n\n   - Defines a list of feature columns to be included for model training.\n\n   - These features encompass demographic, physical, fitness, and body composition metrics, as well as derived features created during preprocessing.\n\n\n\n2. **Train Dataset Preparation**:\n\n   - Filters the training DataFrame to retain only the specified feature columns.\n\n   - Drops any rows from the training dataset where the target variable 'sii' is missing.\n\n\n\n3. **Test Dataset Preparation**:\n\n   - Repeats the feature selection process for the test dataset, ensuring the same columns are used as in training.\n\n\n","metadata":{"papermill":{"duration":0.03791,"end_time":"2024-11-02T08:57:44.552466","exception":false,"start_time":"2024-11-02T08:57:44.514556","status":"completed"},"tags":[]}},{"id":"f5495d87","cell_type":"code","source":"\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n\n                'CGAS-CGAS_Score', 'Physical-BMI',\n\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n\n                'Fitness_Endurance-Max_Stage',\n\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n\n                'SDS-SDS_Total_T',\n\n                'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW']\n\n\n\nfeaturesCols += time_series_cols\n\n\n\ntrain = train[featuresCols]\n\ntrain = train.dropna(subset='sii')\n\n\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n\n                'CGAS-CGAS_Score', 'Physical-BMI',\n\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n\n                'Fitness_Endurance-Max_Stage',\n\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n\n                'SDS-SDS_Total_T',\n\n                'PreInt_EduHx-computerinternet_hoursday', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW']\n\n\n\nfeaturesCols += time_series_cols\n\ntest = test[featuresCols]","metadata":{"execution":{"iopub.status.busy":"2024-11-08T16:01:43.451418Z","iopub.execute_input":"2024-11-08T16:01:43.452077Z","iopub.status.idle":"2024-11-08T16:01:43.468015Z","shell.execute_reply.started":"2024-11-08T16:01:43.452042Z","shell.execute_reply":"2024-11-08T16:01:43.466975Z"},"papermill":{"duration":0.055853,"end_time":"2024-11-02T08:57:44.646196","exception":false,"start_time":"2024-11-02T08:57:44.590343","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"6d16907b","cell_type":"markdown","source":"## `8. Evaluation Functions`\n\n\n\nThis section defines functions to evaluate model predictions, including handling infinite values, calculating the quadratic weighted kappa, and rounding predictions based on thresholds.\n\n\n\n1. **Handling Infinite Values**:\n\n   - Checks if any infinite values exist in the training dataset and replaces them with NaN to prevent issues during calculations.\n\n\n\n2. **Quadratic Weighted Kappa Calculation**:\n\n   - The `quadratic_weighted_kappa` function computes the kappa score, which measures the level of agreement between true and predicted labels, adjusted for ordinal classifications.\n\n\n\n3. **Threshold Rounding**:\n\n   - The `threshold_Rounder` function thresholds continuous predictions into discrete classes based on specified boundary values.\n\n\n\n4. **Prediction Evaluation**:\n\n   - The `evaluate_predictions` function rounds non-rounded predictions and computes the negative quadratic weighted kappa score to assess prediction accuracy.\n","metadata":{"papermill":{"duration":0.037704,"end_time":"2024-11-02T08:57:44.722166","exception":false,"start_time":"2024-11-02T08:57:44.684462","status":"completed"},"tags":[]}},{"id":"7ed33b9b","cell_type":"code","source":"\n\nif np.any(np.isinf(train)):\n\n    train = train.replace([np.inf, -np.inf], np.nan)\n\n\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n\n    return np.where(oof_non_rounded < thresholds[0], 0,\n\n                    np.where(oof_non_rounded < thresholds[1], 1,\n\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-11-08T16:01:47.907262Z","iopub.execute_input":"2024-11-08T16:01:47.907647Z","iopub.status.idle":"2024-11-08T16:01:47.919972Z","shell.execute_reply.started":"2024-11-08T16:01:47.907611Z","shell.execute_reply":"2024-11-08T16:01:47.919138Z"},"papermill":{"duration":0.051392,"end_time":"2024-11-02T08:57:44.811340","exception":false,"start_time":"2024-11-02T08:57:44.759948","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"aa97bbd5","cell_type":"markdown","source":"## `9. Training and Prediction Function`\n\n\n\nThe `TrainML` function trains a specified machine learning model using cross-validation and generates predictions for a test dataset. It also optimizes the thresholds for the predictions based on a quadratic weighted kappa score.\n\n\n\n1. **Input Parameters**:\n\n   - `model_class`: Specifies the machine learning model class to be trained.\n\n   - `test_data`: The dataset on which predictions will be made.\n\n\n\n2. **Data Preparation**:\n\n   - The function separates the features and target variable from the training dataset.\n\n\n\n3. **Cross-Validation**:\n\n   - Utilizes `StratifiedKFold` for cross-validation to maintain the distribution of classes across folds.\n\n   - For each fold, the model is trained, and predictions are made on both the training and validation sets.\n\n\n\n4. **Kappa Score Calculation**:\n\n   - Calculates the Quadratic Weighted Kappa score for both training and validation predictions, storing the results for later analysis.\n\n\n\n5. **Threshold Optimization**:\n\n   - Optimizes thresholds for the predictions using the `minimize` function from SciPy, targeting an improved kappa score.\n\n\n\n6. **Final Predictions**:\n\n   - Averages predictions across folds and applies optimized thresholds.\n\n   - Returns a submission DataFrame with IDs and the final predictions.\n\n\n","metadata":{"papermill":{"duration":0.038082,"end_time":"2024-11-02T08:57:44.887465","exception":false,"start_time":"2024-11-02T08:57:44.849383","status":"completed"},"tags":[]}},{"id":"3c7dd139","cell_type":"code","source":"def TrainML(model_class, test_data):\n\n    X = train.drop(['sii'], axis=1)\n\n    y = train['sii']\n\n\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n    \n\n    train_S = []\n\n    test_S = []\n\n    \n\n    oof_non_rounded = np.zeros(len(y), dtype=float) \n\n    oof_rounded = np.zeros(len(y), dtype=int) \n\n    test_preds = np.zeros((len(test_data), n_splits))\n\n\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n\n\n        model = clone(model_class)\n\n        model.fit(X_train, y_train)\n\n\n\n        y_train_pred = model.predict(X_train)\n\n        y_val_pred = model.predict(X_val)\n\n\n\n        oof_non_rounded[test_idx] = y_val_pred\n\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n\n\n        train_S.append(train_kappa)\n\n        test_S.append(val_kappa)\n\n        \n\n        test_preds[:, fold] = model.predict(test_data)\n\n        \n\n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n\n        clear_output(wait=True)\n\n\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n\n                              method='Nelder-Mead')\n\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n    \n\n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n\n\n    tpm = test_preds.mean(axis=1)\n\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    \n\n    submission = pd.DataFrame({\n\n        'id': sample['id'],\n\n        'sii': tpTuned\n\n    })\n\n\n\n    return submission","metadata":{"execution":{"iopub.status.busy":"2024-11-08T16:01:51.366328Z","iopub.execute_input":"2024-11-08T16:01:51.366959Z","iopub.status.idle":"2024-11-08T16:01:51.380793Z","shell.execute_reply.started":"2024-11-08T16:01:51.366919Z","shell.execute_reply":"2024-11-08T16:01:51.379811Z"},"papermill":{"duration":0.054353,"end_time":"2024-11-02T08:57:44.979858","exception":false,"start_time":"2024-11-02T08:57:44.925505","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"6fc59c69","cell_type":"markdown","source":"## `10. Model Parameters`\n\n\n\n1. Model parameters for LightGBM\n\n2. XGBoost parameters\n\n3. Catboost Parameters","metadata":{"papermill":{"duration":0.037846,"end_time":"2024-11-02T08:57:45.055477","exception":false,"start_time":"2024-11-02T08:57:45.017631","status":"completed"},"tags":[]}},{"id":"ee779a34","cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-11-08T16:07:42.670607Z","iopub.execute_input":"2024-11-08T16:07:42.671426Z","iopub.status.idle":"2024-11-08T16:07:42.678655Z","shell.execute_reply.started":"2024-11-08T16:07:42.671387Z","shell.execute_reply":"2024-11-08T16:07:42.677704Z"},"papermill":{"duration":0.04691,"end_time":"2024-11-02T08:57:45.140220","exception":false,"start_time":"2024-11-02T08:57:45.093310","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"879a8e71","cell_type":"code","source":"# Model parameters for LightGBM\n\nParams = {\n\n    'learning_rate': 0.046,\n\n    'max_depth': 13,\n\n    'num_leaves': 478,\n\n    'min_data_in_leaf': 13,\n\n    'feature_fraction': 0.893,\n\n    'bagging_fraction': 0.893,\n\n    'bagging_freq': 4,\n\n    'lambda_l1': 9,  # Increased from 6.59\n\n    'lambda_l2': 0.01,  # Increased from 2.68e-06\n\n    'device': 'gpu'\n\n\n\n}\n\n# XGBoost parameters\n\nXGB_Params = {\n\n    'learning_rate': 0.05,\n\n    'max_depth': 8,\n\n    'n_estimators': 200,\n\n    'subsample': 0.8,\n\n    'colsample_bytree': 0.8,\n\n    'reg_alpha': 1,  # Increased from 0.1\n\n    'reg_lambda': 5,  # Increased from 1\n\n    'random_state': SEED,\n\n    'tree_method': 'gpu_hist',\n\n\n\n}\n\nCatBoost_Params = {\n\n    'learning_rate': 0.05,\n\n    'depth': 8,\n\n    'iterations': 200,\n\n    'random_seed': SEED,\n\n    'verbose': 0,\n\n    'l2_leaf_reg': 10,  # Increase this value\n\n    'task_type': 'GPU'\n\n\n\n}\n\n# Create model instances\n\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=200)\n\nXGB_Model = XGBRegressor(**XGB_Params)\n\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Combine models using Voting Regressor\n\nvoting_model = VotingRegressor(estimators=[\n\n    ('lightgbm', Light),\n\n    ('xgboost', XGB_Model),\n\n    ('catboost', CatBoost_Model)\n\n])\n\n# Train the ensemble model\n\nSubmission1 = TrainML(voting_model, test)\n\n\nprint(Submission1['sii'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2024-11-08T16:26:37.919902Z","iopub.execute_input":"2024-11-08T16:26:37.920829Z","iopub.status.idle":"2024-11-08T16:27:06.315830Z","shell.execute_reply.started":"2024-11-08T16:26:37.920773Z","shell.execute_reply":"2024-11-08T16:27:06.315001Z"},"papermill":{"duration":44.070816,"end_time":"2024-11-02T08:58:29.248979","exception":false,"start_time":"2024-11-02T08:57:45.178163","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"id":"3be21c8f","cell_type":"markdown","source":"# **Prepare Second Model**","metadata":{"papermill":{"duration":0.039041,"end_time":"2024-11-02T08:58:29.336682","exception":false,"start_time":"2024-11-02T08:58:29.297641","status":"completed"},"tags":[]}},{"id":"df38472c","cell_type":"markdown","source":"\n\n## `1. Load Data`\n\n\n\n1. **Library Imports**: Load required libraries at the start of the script.\n\n2. **Data Loading**: Read the CSV files for training, testing, and sample submission.\n\n3. **Function `process_file`**: \n\n   - Reads a parquet file.\n\n   - Drops unnecessary columns and returns descriptive statistics and an identifier.\n\n4. **Function `load_time_series`**:\n\n   - Processes multiple parquet files concurrently and aggregates their statistics.\n\n   - Returns a DataFrame of statistics with corresponding IDs.\n\n5. **Data Preparation**:\n\n   - Load and merge the time series data with the main datasets.\n\n   - Drop the 'id' column for the final datasets used in training and testing.\n\n\n","metadata":{"papermill":{"duration":0.038087,"end_time":"2024-11-02T08:58:29.413696","exception":false,"start_time":"2024-11-02T08:58:29.375609","status":"completed"},"tags":[]}},{"id":"eb35da37","cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n\n\ndef process_file(filename, dirname):\n\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n\n    df.drop('step', axis=1, inplace=True)\n\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\n\n\ndef load_time_series(dirname) -> pd.DataFrame:\n\n    ids = os.listdir(dirname)\n\n    \n\n    with ThreadPoolExecutor() as executor:\n\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n\n    \n\n    stats, indexes = zip(*results)\n\n    \n\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n\n    df['id'] = indexes\n\n    return df\n\n        \n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\n\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\n\n\ntime_series_cols = train_ts.columns.tolist()\n\ntime_series_cols.remove(\"id\")\n\n\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\n\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\n\n\ntrain = train.drop('id', axis=1)\n\ntest = test.drop('id', axis=1)  ","metadata":{"execution":{"iopub.execute_input":"2024-11-02T08:58:29.492487Z","iopub.status.busy":"2024-11-02T08:58:29.492076Z","iopub.status.idle":"2024-11-02T08:59:48.828169Z","shell.execute_reply":"2024-11-02T08:59:48.827099Z"},"papermill":{"duration":79.378283,"end_time":"2024-11-02T08:59:48.830305","exception":false,"start_time":"2024-11-02T08:58:29.452022","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"35896e7c","cell_type":"markdown","source":"## `2. Features Selection`\n\n\n\n1. **Feature Column Definition**:\n\n   - A list named `featuresCols` is defined to specify the features to be included in the training dataset. This includes various demographic, physical, and activity-related attributes.\n\n\n\n2. **Time Series Columns Addition**:\n\n   - The previously defined `time_series_cols` are appended to `featuresCols`, incorporating relevant time series features into the training dataset.\n\n\n\n3. **Data Filtering**:\n\n   - The training dataset (`train`) is filtered to retain only the columns specified in `featuresCols`.\n","metadata":{"papermill":{"duration":0.066716,"end_time":"2024-11-02T08:59:48.965291","exception":false,"start_time":"2024-11-02T08:59:48.898575","status":"completed"},"tags":[]}},{"id":"f2a12ec7","cell_type":"code","source":"\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\n\n\nfeaturesCols += time_series_cols\n\n\n\ntrain = train[featuresCols]\n\ntrain = train.dropna(subset='sii')\n\n\n\n\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n\n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n\n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']","metadata":{"execution":{"iopub.execute_input":"2024-11-02T08:59:49.100642Z","iopub.status.busy":"2024-11-02T08:59:49.100254Z","iopub.status.idle":"2024-11-02T08:59:49.113323Z","shell.execute_reply":"2024-11-02T08:59:49.112449Z"},"papermill":{"duration":0.082964,"end_time":"2024-11-02T08:59:49.115333","exception":false,"start_time":"2024-11-02T08:59:49.032369","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"0a455122","cell_type":"markdown","source":"## `3. Data Mapping`\n\n1. **Function `update(df)`**: Processes categorical columns in a DataFrame, filling missing values and converting types.\n\n2. **Function `create_mapping(column, dataset)`**: Generates a mapping of unique values in a column to integers.\n\n3. **Mapping Application**: Applies mappings to both the training and test datasets, replacing categorical values with integers.","metadata":{"papermill":{"duration":0.066429,"end_time":"2024-11-02T08:59:49.248161","exception":false,"start_time":"2024-11-02T08:59:49.181732","status":"completed"},"tags":[]}},{"id":"e0ee868e","cell_type":"code","source":"\n\ndef update(df):\n\n    global cat_c\n\n    for c in cat_c: \n\n        df[c] = df[c].fillna('Missing')\n\n        df[c] = df[c].astype('category')\n\n    return df\n\n        \n\ntrain = update(train)\n\ntest = update(test)\n\n\n\ndef create_mapping(column, dataset):\n\n    unique_values = dataset[column].unique()\n\n    return {value: idx for idx, value in enumerate(unique_values)}\n\n\n\nfor col in cat_c:\n\n    mapping = create_mapping(col, train)\n\n    mappingTe = create_mapping(col, test)\n\n    \n\n    train[col] = train[col].replace(mapping).astype(int)\n\n    test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"execution":{"iopub.execute_input":"2024-11-02T08:59:49.384250Z","iopub.status.busy":"2024-11-02T08:59:49.383364Z","iopub.status.idle":"2024-11-02T08:59:49.447292Z","shell.execute_reply":"2024-11-02T08:59:49.446418Z"},"papermill":{"duration":0.134058,"end_time":"2024-11-02T08:59:49.449255","exception":false,"start_time":"2024-11-02T08:59:49.315197","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"19bd9e81","cell_type":"markdown","source":"## `4. Evaluation Functions`\n\n\n\nThis section defines functions to evaluate model predictions, including handling infinite values, calculating the quadratic weighted kappa, and rounding predictions based on thresholds.\n\n\n\n1. **Handling Infinite Values**:\n\n   - Checks if any infinite values exist in the training dataset and replaces them with NaN to prevent issues during calculations.\n\n\n\n2. **Quadratic Weighted Kappa Calculation**:\n\n   - The `quadratic_weighted_kappa` function computes the kappa score, which measures the level of agreement between true and predicted labels, adjusted for ordinal classifications.\n\n\n\n3. **Threshold Rounding**:\n\n   - The `threshold_Rounder` function thresholds continuous predictions into discrete classes based on specified boundary values.\n\n\n\n4. **Prediction Evaluation**:\n\n   - The `evaluate_predictions` function rounds non-rounded predictions and computes the negative quadratic weighted kappa score to assess prediction accuracy.\n","metadata":{"papermill":{"duration":0.066235,"end_time":"2024-11-02T08:59:49.582935","exception":false,"start_time":"2024-11-02T08:59:49.516700","status":"completed"},"tags":[]}},{"id":"e6b68d22","cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n\n    return np.where(oof_non_rounded < thresholds[0], 0,\n\n                    np.where(oof_non_rounded < thresholds[1], 1,\n\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n","metadata":{"execution":{"iopub.execute_input":"2024-11-02T08:59:49.717751Z","iopub.status.busy":"2024-11-02T08:59:49.717119Z","iopub.status.idle":"2024-11-02T08:59:49.723541Z","shell.execute_reply":"2024-11-02T08:59:49.722664Z"},"papermill":{"duration":0.076013,"end_time":"2024-11-02T08:59:49.725437","exception":false,"start_time":"2024-11-02T08:59:49.649424","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"85a6299f","cell_type":"markdown","source":"## `5. Training and Prediction Function`\n\n\n\nThe `TrainML` function trains a specified machine learning model using cross-validation and generates predictions for a test dataset. It also optimizes the thresholds for the predictions based on a quadratic weighted kappa score.\n\n\n\n2. **Data Preparation**:\n\n   - The function separates the features and target variable from the training dataset.\n\n\n\n3. **Cross-Validation**:\n\n   - Utilizes `StratifiedKFold` for cross-validation to maintain the distribution of classes across folds.\n\n   - For each fold, the model is trained, and predictions are made on both the training and validation sets.\n\n\n\n4. **Kappa Score Calculation**:\n\n   - Calculates the Quadratic Weighted Kappa score for both training and validation predictions, storing the results for later analysis.\n\n\n\n5. **Threshold Optimization**:\n\n   - Optimizes thresholds for the predictions using the `minimize` function from SciPy, targeting an improved kappa score.\n\n\n\n6. **Final Predictions**:\n\n   - Averages predictions across folds and applies optimized thresholds.\n\n   - Returns a submission DataFrame with IDs and the final predictions.\n","metadata":{"papermill":{"duration":0.06656,"end_time":"2024-11-02T08:59:49.859094","exception":false,"start_time":"2024-11-02T08:59:49.792534","status":"completed"},"tags":[]}},{"id":"56ad0e13","cell_type":"code","source":"def TrainML(model_class, test_data):\n\n    X = train.drop(['sii'], axis=1)\n\n    y = train['sii']\n\n\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n    \n\n    train_S = []\n\n    test_S = []\n\n    \n\n    oof_non_rounded = np.zeros(len(y), dtype=float) \n\n    oof_rounded = np.zeros(len(y), dtype=int) \n\n    test_preds = np.zeros((len(test_data), n_splits))\n\n\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n\n\n        model = clone(model_class)\n\n        model.fit(X_train, y_train)\n\n\n\n        y_train_pred = model.predict(X_train)\n\n        y_val_pred = model.predict(X_val)\n\n\n\n        oof_non_rounded[test_idx] = y_val_pred\n\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n\n\n        train_S.append(train_kappa)\n\n        test_S.append(val_kappa)\n\n        \n\n        test_preds[:, fold] = model.predict(test_data)\n\n        \n\n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n\n        clear_output(wait=True)\n\n\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n\n                              method='Nelder-Mead')\n\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n    \n\n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n\n\n    tpm = test_preds.mean(axis=1)\n\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    \n\n    submission = pd.DataFrame({\n\n        'id': sample['id'],\n\n        'sii': tpTuned\n\n    })\n\n\n\n    return submission\n","metadata":{"execution":{"iopub.execute_input":"2024-11-02T08:59:49.995086Z","iopub.status.busy":"2024-11-02T08:59:49.994365Z","iopub.status.idle":"2024-11-02T08:59:50.007791Z","shell.execute_reply":"2024-11-02T08:59:50.006915Z"},"papermill":{"duration":0.083843,"end_time":"2024-11-02T08:59:50.009796","exception":false,"start_time":"2024-11-02T08:59:49.925953","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"a889cfb0","cell_type":"markdown","source":"## `6. Model Parameters`\n\n\n\n1. Model parameters for LightGBM\n\n2. XGBoost parameters\n\n3. Catboost Parameters","metadata":{"papermill":{"duration":0.0662,"end_time":"2024-11-02T08:59:50.142464","exception":false,"start_time":"2024-11-02T08:59:50.076264","status":"completed"},"tags":[]}},{"id":"68088e54","cell_type":"code","source":"\n\n# Model parameters for LightGBM\n\nParams = {\n\n    'learning_rate': 0.046,\n\n    'max_depth': 12,\n\n    'num_leaves': 478,\n\n    'min_data_in_leaf': 13,\n\n    'feature_fraction': 0.893,\n\n    'bagging_fraction': 0.784,\n\n    'bagging_freq': 4,\n\n    'lambda_l1': 10,  # Increased from 6.59\n\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n\n}\n\n\n\n\n\n# XGBoost parameters\n\nXGB_Params = {\n\n    'learning_rate': 0.05,\n\n    'max_depth': 6,\n\n    'n_estimators': 200,\n\n    'subsample': 0.8,\n\n    'colsample_bytree': 0.8,\n\n    'reg_alpha': 1,  # Increased from 0.1\n\n    'reg_lambda': 5,  # Increased from 1\n\n    'random_state': SEED\n\n}\n\n\n\n\n\nCatBoost_Params = {\n\n    'learning_rate': 0.05,\n\n    'depth': 6,\n\n    'iterations': 200,\n\n    'random_seed': SEED,\n\n    'cat_features': cat_c,\n\n    'verbose': 0,\n\n    'l2_leaf_reg': 10  # Increase this value\n\n}\n","metadata":{"execution":{"iopub.execute_input":"2024-11-02T08:59:50.277850Z","iopub.status.busy":"2024-11-02T08:59:50.277019Z","iopub.status.idle":"2024-11-02T08:59:50.283930Z","shell.execute_reply":"2024-11-02T08:59:50.283039Z"},"papermill":{"duration":0.077036,"end_time":"2024-11-02T08:59:50.285904","exception":false,"start_time":"2024-11-02T08:59:50.208868","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"e8b61832","cell_type":"markdown","source":"## `7. Model Instances and Combine Model`\n","metadata":{"papermill":{"duration":0.068204,"end_time":"2024-11-02T08:59:50.421222","exception":false,"start_time":"2024-11-02T08:59:50.353018","status":"completed"},"tags":[]}},{"id":"c5e6d26c","cell_type":"code","source":"\n\n# Create model instances\n\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\n\nXGB_Model = XGBRegressor(**XGB_Params)\n\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n\n\n# Combine models using Voting Regressor\n\nvoting_model = VotingRegressor(estimators=[\n\n    ('lightgbm', Light),\n\n    ('xgboost', XGB_Model),\n\n    ('catboost', CatBoost_Model)\n\n])\n\n\n\n# Train the ensemble model\n\nSubmission2 = TrainML(voting_model, test)\n\n\n\nprint(Submission2['sii'].value_counts())","metadata":{"execution":{"iopub.execute_input":"2024-11-02T08:59:50.556893Z","iopub.status.busy":"2024-11-02T08:59:50.556251Z","iopub.status.idle":"2024-11-02T09:00:41.601120Z","shell.execute_reply":"2024-11-02T09:00:41.599378Z"},"papermill":{"duration":51.116111,"end_time":"2024-11-02T09:00:41.603954","exception":false,"start_time":"2024-11-02T08:59:50.487843","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"a88a4029","cell_type":"markdown","source":"# **Prepare Third Model**","metadata":{"papermill":{"duration":0.067563,"end_time":"2024-11-02T09:00:41.740305","exception":false,"start_time":"2024-11-02T09:00:41.672742","status":"completed"},"tags":[]}},{"id":"b0a949c4","cell_type":"markdown","source":"\n\n## Overview\n\nThis notebook focuses on predicting problematic internet use in children through data processing and machine learning.\n\n\n\n## Steps\n\n\n\n### 1. Load Data\n\nWe start by loading the training and testing datasets, as well as a sample submission file.\n\n\n\n### 2. Define Features\n\nNext, we identify the features (columns) that will be used for modeling, including categorical variables.\n\n\n\n### 3. Load Time Series Data\n\nWe then load additional time series data and merge it with the main datasets to enrich the features.\n\n\n\n### 4. Preprocess Data\n\nWe handle missing values and convert categorical variables into a suitable format for modeling.\n\n\n\n### 5. Train the Model\n\nUsing a machine learning ensemble approach, we train the model with the prepared data while implementing a custom evaluation metric to assess performance.\n\n\n\n### 6. Prepare Submission\n\nFinally, we create a submission file that contains the predicted scores for the test dataset.\n","metadata":{"papermill":{"duration":0.067311,"end_time":"2024-11-02T09:00:41.874993","exception":false,"start_time":"2024-11-02T09:00:41.807682","status":"completed"},"tags":[]}},{"id":"14962909","cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\n\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n\n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n\n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\n\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\n\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\n\n\ntime_series_cols = train_ts.columns.tolist()\n\ntime_series_cols.remove(\"id\")\n\n\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\n\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\n\n\ntrain = train.drop('id', axis=1)\n\ntest = test.drop('id', axis=1)\n\n\n\nfeaturesCols += time_series_cols\n\n\n\ntrain = train[featuresCols]\n\ntrain = train.dropna(subset='sii')\n\n\n\ndef update(df):\n\n    global cat_c\n\n    for c in cat_c: \n\n        df[c] = df[c].fillna('Missing')\n\n        df[c] = df[c].astype('category')\n\n    return df\n\n\n\ntrain = update(train)\n\ntest = update(test)\n\n\n\ndef create_mapping(column, dataset):\n\n    unique_values = dataset[column].unique()\n\n    return {value: idx for idx, value in enumerate(unique_values)}\n\n\n\nfor col in cat_c:\n\n    mapping = create_mapping(col, train)\n\n    mappingTe = create_mapping(col, test)\n\n    \n\n    train[col] = train[col].replace(mapping).astype(int)\n\n    test[col] = test[col].replace(mappingTe).astype(int)\n\n\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n\n    return np.where(oof_non_rounded < thresholds[0], 0,\n\n                    np.where(oof_non_rounded < thresholds[1], 1,\n\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\n\n\ndef TrainML(model_class, test_data):\n\n    X = train.drop(['sii'], axis=1)\n\n    y = train['sii']\n\n\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n    \n\n    train_S = []\n\n    test_S = []\n\n    \n\n    oof_non_rounded = np.zeros(len(y), dtype=float) \n\n    oof_rounded = np.zeros(len(y), dtype=int) \n\n    test_preds = np.zeros((len(test_data), n_splits))\n\n\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n\n\n        model = clone(model_class)\n\n        model.fit(X_train, y_train)\n\n\n\n        y_train_pred = model.predict(X_train)\n\n        y_val_pred = model.predict(X_val)\n\n\n\n        oof_non_rounded[test_idx] = y_val_pred\n\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n\n\n        train_S.append(train_kappa)\n\n        test_S.append(val_kappa)\n\n        \n\n        test_preds[:, fold] = model.predict(test_data)\n\n        \n\n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n\n        clear_output(wait=True)\n\n\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n\n                              method='Nelder-Mead')\n\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n    \n\n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n\n\n    tpm = test_preds.mean(axis=1)\n\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n\n\n    return tp_rounded\n\n\n\nimputer = SimpleImputer(strategy='median')\n\n\n\nensemble = VotingRegressor(estimators=[\n\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),\n\n    ('xgb', Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=SEED))])),\n\n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))])),\n\n    \n\n])\n\n\n\nSubmission3 = TrainML(ensemble, test)\n\n\n\n\n\nSubmission3 = pd.DataFrame({\n\n    'id': sample['id'],\n\n    'sii': Submission3\n\n})\n\n\n\nprint(Submission3['sii'].value_counts())","metadata":{"execution":{"iopub.execute_input":"2024-11-02T09:00:42.012161Z","iopub.status.busy":"2024-11-02T09:00:42.011803Z","iopub.status.idle":"2024-11-02T09:04:02.809448Z","shell.execute_reply":"2024-11-02T09:04:02.808356Z"},"papermill":{"duration":200.869036,"end_time":"2024-11-02T09:04:02.811367","exception":false,"start_time":"2024-11-02T09:00:41.942331","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"662bf06a","cell_type":"markdown","source":"# **Combine Model result and make final submission**","metadata":{"papermill":{"duration":0.06734,"end_time":"2024-11-02T09:04:02.946257","exception":false,"start_time":"2024-11-02T09:04:02.878917","status":"completed"},"tags":[]}},{"id":"05fd177f","cell_type":"markdown","source":"\n\n1. **Sort Submissions**: Each of the three submission DataFrames (`s1`, `s2`, `s3`) is sorted by the `id` column and reset to ensure the indices align.\n\n\n\n2. **Combine Predictions**: A new DataFrame is created that includes:\n\n   - The `id` column from one of the submissions.\n\n   - The predicted scores (`sii`) from each of the three submissions, labeled as `sii_1`, `sii_2`, and `sii_3`.\n\n\n\n3. **Voting Mechanism**: A function is defined to determine the most frequent prediction (mode) across the three submissions for each row. This function is applied to the predictions to generate a final score, `final_sii`.\n\n\n\n4. **Prepare Final Submission**: A new DataFrame is created to include just the `id` and the final predicted scores, renaming `final_sii` to `sii`.\n\n\n\n5. **Export to CSV**: The final submission DataFrame is saved as a CSV file named `submission.csv`.\n\n\n\n6. **Count Predictions**: Finally, the code prints the counts of each unique prediction in the final submission.\n\n\n","metadata":{"papermill":{"duration":0.067186,"end_time":"2024-11-02T09:04:03.081293","exception":false,"start_time":"2024-11-02T09:04:03.014107","status":"completed"},"tags":[]}},{"id":"ea4ccb06","cell_type":"code","source":"s1 = Submission1\n\ns2 = Submission2\n\ns3 = Submission3\n\n\n\n\n\ns1 = s1.sort_values(by='id').reset_index(drop=True)\n\ns2 = s2.sort_values(by='id').reset_index(drop=True)\n\ns3 = s3.sort_values(by='id').reset_index(drop=True)\n\n\n\n\n\n\n\ncombined = pd.DataFrame({\n\n    'id': s1['id'],\n\n    'sii_1': s1['sii'],\n\n    'sii_2': s2['sii'],\n\n    'sii_3': s3['sii']\n\n   \n\n})\n\n\n\ndef vote(row):\n\n    return row.mode()[0]\n\n\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(vote, axis=1)\n\n\n\nFinalSubmission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\n\n\nFinalSubmission.to_csv('submission.csv', index=False)\n\nprint(FinalSubmission['sii'].value_counts())","metadata":{"execution":{"iopub.execute_input":"2024-11-02T09:04:03.217056Z","iopub.status.busy":"2024-11-02T09:04:03.216727Z","iopub.status.idle":"2024-11-02T09:04:03.236611Z","shell.execute_reply":"2024-11-02T09:04:03.235557Z"},"papermill":{"duration":0.09028,"end_time":"2024-11-02T09:04:03.238482","exception":false,"start_time":"2024-11-02T09:04:03.148202","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}