{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:57:33.736993Z","iopub.execute_input":"2024-12-17T17:57:33.73743Z","iopub.status.idle":"2024-12-17T17:57:35.001893Z","shell.execute_reply.started":"2024-12-17T17:57:33.737388Z","shell.execute_reply":"2024-12-17T17:57:35.000594Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Predicting Problematic Internet Use Among Children and Adolescents¶\n1. Overview\nIn this notebook, we predict the level of problematic internet use (PIU) among children and adolescents using their physical activity and fitness data. This project leverages the LightGBM regressor model, trained and evaluated using a robust methodology involving stratified K-fold cross-validation, to ensure reliable performance.\n\nThe task focuses on utilizing accessible fitness indicators to serve as proxies for early detection of PIU. This model aims to contribute to early interventions that promote healthier digital habits among young people.\n\n2. Project Description\nMotivation\nProblematic internet use (PIU) is linked to significant mental health challenges, including depression and anxiety. Traditional assessments of PIU are often inaccessible due to cultural, linguistic, and logistical barriers. This project proposes leveraging easily available physical fitness indicators as a predictive tool for PIU, particularly in resource-constrained settings.\n\nKey Objectives\nPrediction: Use accessible physical activity data to predict PIU levels effectively.\nIntervention: Enable early detection of PIU to prompt timely behavioral and health interventions.\n3. Model Roadmap\nData Preparation\nThe train dataset was split to include a separate subset (test1) for unseen validation during training.\nOnly features common between train and test datasets were used for modeling.\nModel Choice\nThe model used in this project is LightGBM Regressor, configured with optimized hyperparameters.\n\nTraining Strategy\nStratified K-Fold Cross-Validation: The dataset was split into 7 folds to ensure balanced representation of target classes in each fold.\nMetrics: The Quadratic Weighted Kappa (QWK) score was used to measure the model's performance.\nThreshold Optimization\nThe model outputs continuous predictions, which were converted to discrete classes (0 to 3) using optimized thresholds. These thresholds were tuned to maximize the QWK score.\n\n4. Model Implementation\nCore Steps\nFeature Engineering:\nEnsured alignment of features between train and test datasets.\nModel Training:\nThe LightGBM regressor was trained for each fold of the stratified K-Fold split.\nValidation:\nPerformance was evaluated using the QWK score on both training and validation sets.\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd\nfrom sklearn.base import clone\nfrom copy import deepcopy\nimport optuna\nfrom scipy.optimize import minimize\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport re\nfrom colorama import Fore, Style\n\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nimport lightgbm as lgb\nfrom catboost import CatBoostRegressor, CatBoostClassifier\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\n\nSEED = 42\nn_splits = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:57:35.004723Z","iopub.execute_input":"2024-12-17T17:57:35.00526Z","iopub.status.idle":"2024-12-17T17:57:35.014074Z","shell.execute_reply.started":"2024-12-17T17:57:35.005171Z","shell.execute_reply":"2024-12-17T17:57:35.01292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:57:35.015586Z","iopub.execute_input":"2024-12-17T17:57:35.016031Z","iopub.status.idle":"2024-12-17T17:57:35.029955Z","shell.execute_reply.started":"2024-12-17T17:57:35.015981Z","shell.execute_reply":"2024-12-17T17:57:35.028819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:57:35.031359Z","iopub.execute_input":"2024-12-17T17:57:35.031773Z","iopub.status.idle":"2024-12-17T17:57:35.053307Z","shell.execute_reply.started":"2024-12-17T17:57:35.031722Z","shell.execute_reply":"2024-12-17T17:57:35.051987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:57:35.056208Z","iopub.execute_input":"2024-12-17T17:57:35.056645Z","iopub.status.idle":"2024-12-17T18:00:00.175859Z","shell.execute_reply.started":"2024-12-17T17:57:35.056605Z","shell.execute_reply":"2024-12-17T18:00:00.174493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.dtypes)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T18:00:00.177442Z","iopub.execute_input":"2024-12-17T18:00:00.177999Z","iopub.status.idle":"2024-12-17T18:00:00.186294Z","shell.execute_reply.started":"2024-12-17T18:00:00.177949Z","shell.execute_reply":"2024-12-17T18:00:00.18472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Selecting columns with 'object' dtype (categorical columns)\ncategorical_columns = train.select_dtypes(include=['object']).columns.tolist()\n\n# Print the list of categorical columns\nprint(\"Categorical columns:\", categorical_columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T18:00:00.188263Z","iopub.execute_input":"2024-12-17T18:00:00.188788Z","iopub.status.idle":"2024-12-17T18:00:00.210426Z","shell.execute_reply.started":"2024-12-17T18:00:00.188722Z","shell.execute_reply":"2024-12-17T18:00:00.208816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train['CGAS-Season'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T18:00:00.212164Z","iopub.execute_input":"2024-12-17T18:00:00.212657Z","iopub.status.idle":"2024-12-17T18:00:00.226196Z","shell.execute_reply.started":"2024-12-17T18:00:00.212596Z","shell.execute_reply":"2024-12-17T18:00:00.224873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# 1. Drop rows with missing target column 'sii' in training data\ntrain = train.dropna(subset=['sii'])\n\n# 2. Replace missing values in numeric columns with the mean of each column\n# Identify numeric columns in the training set\nnumeric_columns = train.select_dtypes(include=[np.number]).columns\n\n# Exclude the target column 'sii' from numeric columns\nnumeric_columns = [col for col in numeric_columns if col != 'sii']\n\n# Replace missing values in numeric columns with the mean for training data\ntrain[numeric_columns] = train[numeric_columns].fillna(train[numeric_columns].mean())\n\n# 3. Convert categorical columns into numeric\n# Define categorical columns (you can expand or change based on your dataset)\ncategorical_columns = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n                       'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n                       'PAQ_A-Season', 'PAQ_C-Season', 'PCIAT-Season', 'SDS-Season', \n                       'PreInt_EduHx-Season']\n\n# Map the session values to numeric values (Fall: 0, Winter: 1, Summer: 2, Spring: 3)\nseason_mapping = {'Fall': 0, 'Winter': 1, 'Summer': 2, 'Spring': 3}\n\n# Apply the mapping to the categorical columns in the train data\nfor col in categorical_columns:\n    train[col] = train[col].map(season_mapping)\n\n# Ensure there are no missing values after the conversion in train data\ntrain = train.fillna(train.mean())  # In case there are any remaining NaNs\n\n# Now apply the same transformations to the test data:\n\n# 4. Replace missing values in numeric columns with the mean for test data\n# Identify numeric columns in the test set (common columns between train and test)\ntest_numeric_columns = [col for col in numeric_columns if col in test.columns]\n\n# Ensure that test data contains only numeric columns\ntest[test_numeric_columns] = test[test_numeric_columns].apply(pd.to_numeric, errors='coerce')\n\n# Replace missing values in numeric columns with the mean for test data\ntest[test_numeric_columns] = test[test_numeric_columns].fillna(test[test_numeric_columns].mean())\n\n# 5. Convert categorical columns in the test set into numeric using the same mapping\nfor col in categorical_columns:\n    if col in test.columns:  # Check if the column exists in the test set\n        test[col] = test[col].map(season_mapping)\n\n# Ensure there are no missing values after the conversion in test data\ntest = test.fillna(test.mean())  # In case there are any remaining NaNs\n\n# Now train and test data should be ready for modeling\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T18:00:00.227814Z","iopub.execute_input":"2024-12-17T18:00:00.228191Z","iopub.status.idle":"2024-12-17T18:00:00.615785Z","shell.execute_reply.started":"2024-12-17T18:00:00.228155Z","shell.execute_reply":"2024-12-17T18:00:00.614461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Find the number of missing values in each column\nmissing_values = train.isnull().sum()\n\n# Print the result\nprint(missing_values)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T18:00:00.618876Z","iopub.execute_input":"2024-12-17T18:00:00.619455Z","iopub.status.idle":"2024-12-17T18:00:00.644002Z","shell.execute_reply.started":"2024-12-17T18:00:00.619386Z","shell.execute_reply":"2024-12-17T18:00:00.641814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count unique values in the column\nunique_values_count = train['sii'].nunique()\n\n# Count missing values in the column\nmissing_values_count = train['sii'].isnull().sum()\n\nprint(f\"Unique values in 'sii': {unique_values_count}\")\nprint(f\"Missing values in 'sii': {missing_values_count}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T18:00:00.647077Z","iopub.execute_input":"2024-12-17T18:00:00.647695Z","iopub.status.idle":"2024-12-17T18:00:00.658558Z","shell.execute_reply.started":"2024-12-17T18:00:00.647621Z","shell.execute_reply":"2024-12-17T18:00:00.655942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T18:00:00.660904Z","iopub.execute_input":"2024-12-17T18:00:00.66142Z","iopub.status.idle":"2024-12-17T18:00:00.757852Z","shell.execute_reply.started":"2024-12-17T18:00:00.66136Z","shell.execute_reply":"2024-12-17T18:00:00.756633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T18:00:00.759346Z","iopub.execute_input":"2024-12-17T18:00:00.759708Z","iopub.status.idle":"2024-12-17T18:00:00.868517Z","shell.execute_reply.started":"2024-12-17T18:00:00.759675Z","shell.execute_reply":"2024-12-17T18:00:00.867283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T18:00:00.873383Z","iopub.execute_input":"2024-12-17T18:00:00.874088Z","iopub.status.idle":"2024-12-17T18:00:01.009851Z","shell.execute_reply.started":"2024-12-17T18:00:00.874014Z","shell.execute_reply":"2024-12-17T18:00:01.008319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T18:00:01.011151Z","iopub.execute_input":"2024-12-17T18:00:01.01152Z","iopub.status.idle":"2024-12-17T18:00:01.103266Z","shell.execute_reply.started":"2024-12-17T18:00:01.011486Z","shell.execute_reply":"2024-12-17T18:00:01.101937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the number of columns\nnum_columns = test.shape[1]\nprint(\"Number of columns:\", num_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T18:00:01.104805Z","iopub.execute_input":"2024-12-17T18:00:01.105593Z","iopub.status.idle":"2024-12-17T18:00:01.111314Z","shell.execute_reply.started":"2024-12-17T18:00:01.105553Z","shell.execute_reply":"2024-12-17T18:00:01.109819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.base import clone\nfrom scipy.optimize import minimize\nfrom tqdm import tqdm\nfrom colorama import Fore, Style\nimport lightgbm as lgb\n\n# Parameters\nSEED = 42\nn_splits = 7\n\n# Define functions for Quadratic Weighted Kappa and thresholding\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    # Separate features and target\n    # Align train and test data columns\n    common_columns = test_data.columns  # Get columns in the test data\n    X_full = train[common_columns]      # Use only columns present in test data\n    y_full = train['sii']               # Target column remains unchanged\n\n    # Separate last 200 rows into `test1` and the rest into `X_train` and `y_train`\n    X_train = X_full.iloc[:-200]\n    y_train = y_full.iloc[:-200]\n    X_test1 = X_full.iloc[-200:]\n    y_test1 = y_full.iloc[-200:]\n\n    # Initialize stratified K-fold\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    # Arrays to store out-of-fold and test predictions\n    oof_non_rounded = np.zeros(len(y_train), dtype=float)\n    oof_rounded = np.zeros(len(y_train), dtype=int)\n    test_preds = np.zeros((len(test_data), n_splits))\n\n    train_S = []  # List to store training QWK scores\n    test_S = []   # List to store validation QWK scores\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X_train, y_train), desc=\"Training Folds\", total=n_splits)):\n        X_tr, X_val = X_train.iloc[train_idx], X_train.iloc[test_idx]\n        y_tr, y_val = y_train.iloc[train_idx], y_train.iloc[test_idx]\n\n        # Clone and train the model\n        model = clone(model_class)\n        model.fit(X_tr, y_tr)\n\n        # Make predictions for training and validation\n        y_train_pred = model.predict(X_tr)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        # Calculate QWK for training and validation\n        train_kappa = quadratic_weighted_kappa(y_tr, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n\n        # Make predictions for the test set\n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n\n    # Print mean QWK scores\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    # Optimize thresholds for final predictions\n    KappaOptimizer = minimize(\n        evaluate_predictions, x0=[0.5, 1.5, 2.5],\n        args=(y_train, oof_non_rounded), method='Nelder-Mead'\n    )\n    assert KappaOptimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOptimizer.x)\n    tKappa = quadratic_weighted_kappa(y_train, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    # Final predictions\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOptimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    # Predict and evaluate on `test1`\n    test1_preds = model.predict(X_test1)\n    test1_rounded = threshold_Rounder(test1_preds, KappaOptimizer.x)\n    test1_kappa = quadratic_weighted_kappa(y_test1, test1_rounded)\n\n    print(f\"----> || Test1 QWK SCORE :: {Fore.GREEN}{Style.BRIGHT} {test1_kappa:.3f}{Style.RESET_ALL}\")\n\n    return submission, model, test1_kappa\n\n# Define and preprocess your train and test data\n# Assuming `train` and `test` are Pandas DataFrames already loaded in your environment\ntest1 = train.iloc[-200:].copy()  # Create test1 from the last 200 rows of train\ntrain = train.iloc[:-200].copy()  # Update train to exclude test1 rows\n\n# Model Parameters\nParams7 = {\n    'learning_rate': 0.013951544541065234, \n    'max_depth': 30, \n    'num_leaves': 349, \n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.6617479801088895, \n    'bagging_fraction': 0.8193545617693619, \n    'bagging_freq': 9, \n    'lambda_l1': 37.916331094750447, \n    'lambda_l2': 2.258386103169817,\n    'n_estimators': 952\n}\n\n# Instantiate the model\nLight = lgb.LGBMRegressor(**Params7, random_state=SEED, verbose=-1)\n\n# Train model and generate submission\nSubmission, model, Test1_QWK = TrainML(Light, test)\nprint(f\"Final QWK Score on Test1: {Test1_QWK:.3f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T18:00:01.113743Z","iopub.execute_input":"2024-12-17T18:00:01.114161Z","iopub.status.idle":"2024-12-17T18:00:09.692945Z","shell.execute_reply.started":"2024-12-17T18:00:01.114125Z","shell.execute_reply":"2024-12-17T18:00:09.691731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission.to_csv('submission.csv', index=False)\nprint(Submission['sii'].value_counts())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T18:00:09.694549Z","iopub.execute_input":"2024-12-17T18:00:09.694906Z","iopub.status.idle":"2024-12-17T18:00:09.707623Z","shell.execute_reply.started":"2024-12-17T18:00:09.694871Z","shell.execute_reply":"2024-12-17T18:00:09.706326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}