{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7933327,"sourceType":"datasetVersion","datasetId":3914692}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q autogluon --no-index --find-links=file:///kaggle/input/autogluon/v1.0.0","metadata":{"execution":{"iopub.status.busy":"2024-10-23T21:22:29.648674Z","iopub.execute_input":"2024-10-23T21:22:29.649154Z","iopub.status.idle":"2024-10-23T21:26:31.738990Z","shell.execute_reply.started":"2024-10-23T21:22:29.649118Z","shell.execute_reply":"2024-10-23T21:26:31.737813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from autogluon.tabular import TabularPredictor\nimport pandas as pd\nimport numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None","metadata":{"execution":{"iopub.status.busy":"2024-10-23T21:26:31.741491Z","iopub.execute_input":"2024-10-23T21:26:31.741923Z","iopub.status.idle":"2024-10-23T21:26:55.230199Z","shell.execute_reply.started":"2024-10-23T21:26:31.741873Z","shell.execute_reply":"2024-10-23T21:26:55.229425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\n\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n\n\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n                 \n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded\n\ndef feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    \n    return df\n\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train, encoding_dim=60, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=60, epochs=100, batch_size=32)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded['id']=test_ts[\"id\"]\n\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\n\nimputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n        \ntrain = train_imputed\n\ntrain = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)\n\ntrain = train.drop('id', axis=1)\ntest  = test .drop('id', axis=1)   \n\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'BMI_Age','Internet_Hours_Age','BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW']\n\nfeaturesCols += time_series_cols\ntest = test[featuresCols]\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-23T21:26:55.231702Z","iopub.execute_input":"2024-10-23T21:26:55.232438Z","iopub.status.idle":"2024-10-23T21:28:29.866976Z","shell.execute_reply.started":"2024-10-23T21:26:55.232402Z","shell.execute_reply":"2024-10-23T21:28:29.866169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n# null_percent_sorted = null_percent.sort_values()\n\n# # Increasing the plot size and re-plotting\n# plt.figure(figsize=(25, 12))\n# null_percent_sorted.plot(kind='bar', color='lightcoral')\n# plt.title('Percentage of Null Values for Each Feature (Sorted Ascending)')\n# plt.ylabel('Percentage of Null Values')\n# plt.xlabel('Features')\n# plt.xticks(rotation=90)\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-23T21:28:29.868895Z","iopub.execute_input":"2024-10-23T21:28:29.869203Z","iopub.status.idle":"2024-10-23T21:28:29.873239Z","shell.execute_reply.started":"2024-10-23T21:28:29.869171Z","shell.execute_reply":"2024-10-23T21:28:29.872388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom autogluon.tabular import TabularPredictor\nfrom sklearn.metrics import cohen_kappa_score\nfrom scipy.optimize import minimize\nfrom sklearn.impute import KNNImputer\nfrom sklearn.model_selection import StratifiedKFold\n\n# Check for infinity values and replace with NaN\nif np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)\n\ndf = train.copy()\n\n# Separate features and labels\nX = df.drop(columns=['sii'])\ny = df['sii']\n\n# Initialize StratifiedKFold\nskf = StratifiedKFold(n_splits=5, shuffle = False)\n\n# List to store results and best model hyperparameters\nresults_list = []\nbest_models_hyperparams = []\n\n# Initialize preprocessor to use later for the entire dataset\npreprocessor = None\nbest_test_qwk = -np.inf  # To track the best test QWK score\nbest_model_info = None  # To store the best model info\n\nfor fold, (train_index, test_index) in enumerate(skf.split(X, y)):\n    print(f\"\\nFold {fold+1}\")\n    X_train_raw, X_test_raw = X.iloc[train_index], X.iloc[test_index]\n    y_train, y_test = y.iloc[train_index], y.iloc[test_index]\n\n    # Combine features and labels for processing\n    df_train = X_train_raw.copy()\n    df_train['sii'] = y_train.reset_index(drop=True)\n    df_test = X_test_raw.copy()\n    df_test['sii'] = y_test.reset_index(drop=True)\n\n    # Preprocess the data\n    numerical_cols = df_train.select_dtypes(include=['float64', 'int64']).columns.drop('sii')\n    categorical_cols = df_train.select_dtypes(include=['object']).columns\n\n    numerical_pipeline = Pipeline(steps=[('imputer', KNNImputer(n_neighbors=5))])\n    categorical_pipeline = Pipeline(steps=[\n        ('imputer', SimpleImputer(strategy='most_frequent')),\n        ('encoder', OrdinalEncoder())\n    ])\n\n    preprocessor = ColumnTransformer(transformers=[\n        ('num', numerical_pipeline, numerical_cols),\n        ('cat', categorical_pipeline, categorical_cols)\n    ])\n\n    # Fit and transform training data\n    X_train_processed = pd.DataFrame(\n        preprocessor.fit_transform(df_train.drop('sii', axis=1)),\n        columns=numerical_cols.tolist() + categorical_cols.tolist()\n    )\n\n    # Transform test data\n    X_test_processed = pd.DataFrame(\n        preprocessor.transform(df_test.drop('sii', axis=1)),\n        columns=numerical_cols.tolist() + categorical_cols.tolist()\n    )\n\n    X_train_processed['sii'] = y_train.reset_index(drop=True)\n    X_test_processed['sii'] = y_test.reset_index(drop=True)\n\n    # Set up the model\n    label = 'sii'\n    predictor = TabularPredictor(\n        label=label,\n        problem_type='regression',\n        eval_metric='mean_squared_error',\n        sample_weight='balance_weight'\n        \n    ).fit(\n        train_data=X_train_processed,\n        time_limit=8000,\n        hyperparameters={\n            'GBM': {},\n            'CAT': {},\n            'XGB': {},\n            'RF': {},\n            'XT': {}\n        }\n    )\n\n    # Make predictions on train and test sets\n    y_train_pred_cont = predictor.predict(X_train_processed.drop(columns=['sii'])).to_numpy()\n    y_test_pred_cont = predictor.predict(X_test_processed.drop(columns=['sii'])).to_numpy()\n\n    # Optimize thresholds on training data\n    def threshold_rounder(preds, thresholds):\n        return np.where(preds < thresholds[0], 0,\n                        np.where(preds < thresholds[1], 1,\n                                 np.where(preds < thresholds[2], 2, 3)))\n\n    def evaluate_qwk(thresholds, y_true, preds):\n        y_pred = threshold_rounder(preds, thresholds)\n        return -cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n    initial_thresholds = [0.5, 1.5, 2.5]\n    result = minimize(\n        evaluate_qwk,\n        x0=initial_thresholds,\n        args=(y_train.values, y_train_pred_cont),\n        method='Nelder-Mead'\n    )\n    optimal_thresholds = result.x\n\n    y_train_pred_opt = threshold_rounder(y_train_pred_cont, optimal_thresholds)\n    y_test_pred_opt = threshold_rounder(y_test_pred_cont, optimal_thresholds)\n\n    train_qwk = cohen_kappa_score(y_train.values, y_train_pred_opt, weights='quadratic')\n    test_qwk = cohen_kappa_score(y_test.values, y_test_pred_opt, weights='quadratic')\n\n    # Simply rounding the predictions and comparing to actual values\n    y_train_pred_rounded = np.round(y_train_pred_cont).astype(int)\n    y_test_pred_rounded = np.round(y_test_pred_cont).astype(int)\n\n    train_qwk_rounded = cohen_kappa_score(y_train.values, y_train_pred_rounded, weights='quadratic')\n    test_qwk_rounded = cohen_kappa_score(y_test.values, y_test_pred_rounded, weights='quadratic')\n\n    # Print results for this fold\n    print(f\"Train QWK: {train_qwk:.4f}, Test QWK: {test_qwk:.4f}\")\n    print(f\"Train QWK (rounded): {train_qwk_rounded:.4f}, Test QWK (rounded): {test_qwk_rounded:.4f}\")\n\n    # Save results in list\n    results_list.append({\n        'fold': fold + 1,\n        'train_qwk': train_qwk,\n        'test_qwk': test_qwk,\n        'train_qwk_rounded': train_qwk_rounded,\n        'test_qwk_rounded': test_qwk_rounded\n    })\n\n\n# Train final model on the entire dataset using the best hyperparameters\nX_full = train.drop(columns=['sii'])\ny_full = train['sii']\n\n# Preprocess the entire dataset using the pre-fitted preprocessor\nX_full_processed = pd.DataFrame(\n    preprocessor.transform(X_full),\n    columns=numerical_cols.tolist() + categorical_cols.tolist()\n)\n\nX_full_processed['sii'] = y_full.reset_index(drop=True)\n\n# Use the best hyperparameters to train on the full dataset\npredictor_full = TabularPredictor(\n    label=label,\n    problem_type='regression',\n    eval_metric='mean_squared_error',\n    sample_weight='balance_weight'\n).fit(\n    train_data=X_full_processed,\n    time_limit=12000,\n    hyperparameters={\n            'GBM': {},\n            'CAT': {},\n            'XGB': {},\n            'RF': {},\n            'XT': {}\n        }\n\n)\n\n# Display final results DataFrame\nresults_df = pd.DataFrame(results_list)\nprint(results_df)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-23T22:20:09.523041Z","iopub.execute_input":"2024-10-23T22:20:09.523434Z","iopub.status.idle":"2024-10-23T22:22:33.403514Z","shell.execute_reply.started":"2024-10-23T22:20:09.523397Z","shell.execute_reply":"2024-10-23T22:22:33.402627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictor_full = TabularPredictor(\n    label=label,\n    problem_type='regression',\n    eval_metric='mean_squared_error',\n    sample_weight='balance_weight'\n).fit(\n    train_data=X_full_processed,\n    time_limit=12000,\n    hyperparameters={\n            'GBM': {},\n            'CAT': {},\n            'XGB': {},\n            'RF': {},\n            'XT': {}\n        }\n\n)\n\n# Display final results DataFrame\nresults_df = pd.DataFrame(results_list)\nprint(results_df)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-23T22:23:19.241786Z","iopub.execute_input":"2024-10-23T22:23:19.242674Z","iopub.status.idle":"2024-10-23T22:23:49.386187Z","shell.execute_reply.started":"2024-10-23T22:23:19.242634Z","shell.execute_reply":"2024-10-23T22:23:49.385165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Data\nfolds = results_df['fold']\ntrain_qwk = results_df['train_qwk']\ntest_qwk = results_df['test_qwk']\ntrain_qwk_rounded = results_df['train_qwk_rounded']\ntest_qwk_rounded = results_df['test_qwk_rounded']\n\n# Create a figure with two subplots: one for train and one for test\nfig, axs = plt.subplots(2, 1, figsize=(10, 10))\n\n# Train QWK Scores\naxs[0].plot(folds, train_qwk, label='Train QWK', marker='o')\naxs[0].plot(folds, train_qwk_rounded, label='Train QWK (Rounded)', marker='o')\n\n# Add titles and labels for the train plot\naxs[0].set_title('Train QWK Scores Across Folds', fontsize=14)\naxs[0].set_xlabel('Fold', fontsize=12)\naxs[0].set_ylabel('QWK Score', fontsize=12)\naxs[0].legend()\naxs[0].grid(True)\n\n# Test QWK Scores\naxs[1].plot(folds, test_qwk, label='Test QWK', marker='o')\naxs[1].plot(folds, test_qwk_rounded, label='Test QWK (Rounded)', marker='o')\n\n# Add titles and labels for the test plot\naxs[1].set_title('Test QWK Scores Across Folds', fontsize=14)\naxs[1].set_xlabel('Fold', fontsize=12)\naxs[1].set_ylabel('QWK Score', fontsize=12)\naxs[1].legend()\naxs[1].grid(True)\n\n# Adjust layout and show the plot\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-23T22:23:49.387849Z","iopub.execute_input":"2024-10-23T22:23:49.388160Z","iopub.status.idle":"2024-10-23T22:23:49.954911Z","shell.execute_reply.started":"2024-10-23T22:23:49.388127Z","shell.execute_reply":"2024-10-23T22:23:49.954022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_id = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n# test_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n# test = pd.merge(test, test_ts, how=\"left\", on='id')\n# id_col = test['id']\n# test_data = test.drop('id', axis=1)\n\n\n\ntest_preprocessed = pd.DataFrame(preprocessor.transform(test),\n                                 columns=numerical_cols.tolist() + categorical_cols.tolist())\n\n\ny_pred_cont_test = predictor_full.predict(test_preprocessed).to_numpy()\n\n\ny_pred_opt_test = threshold_rounder(y_pred_cont_test, optimal_thresholds)\n\n\nsubmission = pd.DataFrame({\n    'id': test_id['id'],  \n    'sii': y_pred_opt_test\n})\n\n\n# submission.to_csv('submission.csv', index=False)\n\n# # Print the distribution of predicted 'sii' values\n# print(submission['sii'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2024-10-23T22:23:49.956214Z","iopub.execute_input":"2024-10-23T22:23:49.956615Z","iopub.status.idle":"2024-10-23T22:23:50.229091Z","shell.execute_reply.started":"2024-10-23T22:23:49.956551Z","shell.execute_reply":"2024-10-23T22:23:50.228121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-23T22:23:50.231515Z","iopub.execute_input":"2024-10-23T22:23:50.232053Z","iopub.status.idle":"2024-10-23T22:23:50.237202Z","shell.execute_reply.started":"2024-10-23T22:23:50.232015Z","shell.execute_reply":"2024-10-23T22:23:50.236228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-10-23T22:23:50.238372Z","iopub.execute_input":"2024-10-23T22:23:50.238758Z","iopub.status.idle":"2024-10-23T22:23:50.252550Z","shell.execute_reply.started":"2024-10-23T22:23:50.238725Z","shell.execute_reply":"2024-10-23T22:23:50.251700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}