{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\nfrom sklearn.preprocessing import MinMaxScaler, StandardScaler, RobustScaler\n\nSEED = 42\nn_splits = 5","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-30T03:48:55.075511Z","iopub.execute_input":"2024-10-30T03:48:55.076047Z","iopub.status.idle":"2024-10-30T03:48:55.087673Z","shell.execute_reply.started":"2024-10-30T03:48:55.075985Z","shell.execute_reply":"2024-10-30T03:48:55.086329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Func for Data loading & Feature extraction","metadata":{}},{"cell_type":"code","source":"GLOBAL_TS_LENGTH=[]\nimport pandas as pd\nimport numpy as np\nfrom scipy import stats\n\ndef extract_advanced_features(data):\n    \"\"\"\n    Extract advanced features from actigraphy data for SII prediction.\n    \n    Parameters:\n    data (pd.DataFrame): Input DataFrame with actigraphy measurements\n    \n    Returns:\n    pd.DataFrame: Single row DataFrame with extracted features\n    \"\"\"\n    # Initial data preprocessing\n    data = data.copy()\n    data['timestamp'] = pd.to_datetime(data['relative_date_PCIAT'], unit='D') + pd.to_timedelta(data['time_of_day'])\n    data = data[data['non-wear_flag'] == 0]\n    \n    # Calculate basic metrics\n    data['magnitude'] = np.sqrt(data['X']**2 + data['Y']**2 + data['Z']**2)\n    data['velocity'] = data['magnitude']\n    data['distance'] = data['velocity'] * 5  # 5 seconds per observation\n    data['date'] = data['timestamp'].dt.date\n    hour = pd.to_datetime(data['time_of_day']).dt.hour\n    \n    # Calculate aggregated distances\n    distances = {\n        'daily': data.groupby('date')['distance'].sum(),\n        'monthly': data.groupby(data['timestamp'].dt.to_period('M'))['distance'].sum(),\n        'quarterly': data.groupby('quarter')['distance'].sum()\n    }\n    \n    # Initialize features dictionary\n    features = {}\n    \n    # Time masks for different periods\n    time_masks = {\n        'morning': (hour >= 6) & (hour < 12),\n        'afternoon': (hour >= 12) & (hour < 18),\n        'evening': (hour >= 18) & (hour < 22),\n        'night': (hour >= 22) | (hour < 6)\n    }\n    \n    # 1. Activity Pattern Features\n    for period, mask in time_masks.items():\n        features.update({\n            f'{period}_activity_mean': data.loc[mask, 'magnitude'].mean(),\n            f'{period}_activity_std': data.loc[mask, 'magnitude'].std(),\n            f'{period}_enmo_mean': data.loc[mask, 'enmo'].mean()\n        })\n    \n    # 2. Sleep Quality Features\n    sleep_hours = time_masks['night']\n    magnitude_threshold = data['magnitude'].mean() + data['magnitude'].std()\n    \n    features.update({\n        'sleep_movement_mean': data.loc[sleep_hours, 'magnitude'].mean(),\n        'sleep_movement_std': data.loc[sleep_hours, 'magnitude'].std(),\n        'sleep_disruption_count': len(data.loc[sleep_hours & (data['magnitude'] > \n            data['magnitude'].mean() + 2 * data['magnitude'].std())]),\n        'light_exposure_during_sleep': data.loc[sleep_hours, 'light'].mean(),\n        'sleep_position_changes': len(data.loc[sleep_hours & \n            (abs(data['anglez'].diff()) > 45)]),\n        'good_sleep_cycle': int(data.loc[sleep_hours, 'light'].mean() < 50)\n    })\n    \n    # 3. Activity Intensity Features\n    features.update({\n        'sedentary_time_ratio': (data['magnitude'] < magnitude_threshold * 0.5).mean(),\n        'moderate_activity_ratio': ((data['magnitude'] >= magnitude_threshold * 0.5) & \n            (data['magnitude'] < magnitude_threshold * 1.5)).mean(),\n        'vigorous_activity_ratio': (data['magnitude'] >= magnitude_threshold * 1.5).mean(),\n        'activity_peaks_per_day': len(data[data['magnitude'] > \n            data['magnitude'].quantile(0.95)]) / len(data.groupby('relative_date_PCIAT'))\n    })\n    \n    # 4. Circadian Rhythm Features\n    hourly_activity = data.groupby(hour)['magnitude'].mean()\n    features.update({\n        'circadian_regularity': hourly_activity.std() / hourly_activity.mean(),\n        'peak_activity_hour': hourly_activity.idxmax(),\n        'trough_activity_hour': hourly_activity.idxmin(),\n        'activity_range': hourly_activity.max() - hourly_activity.min()\n    })\n    \n    # 5-11. Additional Feature Groups\n    weekend_mask = data['weekday'].isin([6, 7])\n    \n    features.update({\n        # Movement Patterns\n        'movement_entropy': stats.entropy(pd.qcut(data['magnitude'], q=10, duplicates='drop').value_counts()),\n        'direction_changes': len(data[abs(data['anglez'].diff()) > 30]) / len(data),\n        'sustained_activity_periods': len(data[data['magnitude'].rolling(12).mean() > \n            magnitude_threshold]) / len(data),\n        \n        # Weekend vs Weekday\n        'weekend_activity_ratio': data.loc[weekend_mask, 'magnitude'].mean() / \n            data.loc[~weekend_mask, 'magnitude'].mean(),\n        'weekend_sleep_difference': data.loc[weekend_mask & sleep_hours, 'magnitude'].mean() - \n            data.loc[~weekend_mask & sleep_hours, 'magnitude'].mean(),\n        \n        # Non-wear Time\n        'wear_time_ratio': (data['non-wear_flag'] == 0).mean(),\n        'wear_consistency': len(data['non-wear_flag'].value_counts()),\n        'longest_wear_streak': data['non-wear_flag'].eq(0).astype(int).groupby(\n            data['non-wear_flag'].ne(0).cumsum()).sum().max(),\n        \n        # Device Usage\n        'screen_time_proxy': (data['light'] > data['light'].quantile(0.75)).mean(),\n        'dark_environment_ratio': (data['light'] < data['light'].quantile(0.25)).mean(),\n        'light_variation': data['light'].std() / data['light'].mean() if data['light'].mean() != 0 else 0,\n        \n        # Battery Usage\n        'battery_drain_rate': -np.polyfit(range(len(data)), data['battery_voltage'], 1)[0],\n        'battery_variability': data['battery_voltage'].std(),\n        'low_battery_time': (data['battery_voltage'] < data['battery_voltage'].quantile(0.1)).mean(),\n        \n        # Time-based\n        'days_monitored': data['relative_date_PCIAT'].nunique(),\n        'total_active_hours': len(data[data['magnitude'] > magnitude_threshold * 0.5]) * 5 / 3600,\n        'activity_regularity': data.groupby('weekday')['magnitude'].mean().std()\n    })\n    \n    # Variability Features for multiple columns\n    for col in ['X', 'Y', 'Z', 'enmo', 'anglez']:\n        features.update({\n            f'{col}_skewness': data[col].skew(),\n            f'{col}_kurtosis': data[col].kurtosis(),\n            f'{col}_trend': np.polyfit(range(len(data)), data[col], 1)[0]\n        })\n    \n    return pd.DataFrame([features])\n\ndef process_file(filename, dirname):\n    df= pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    data=extract_advanced_features(df)\n    array_1=data.values[0]\n    array_2=df.describe().values.reshape(-1), filename.split('=')[1]\n    # Combine the two arrays\n    combined_array = np.concatenate((array_1, array_2[0]))\n    combined_tuple=(array_1,array_2[1])\n    return combined_tuple\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-10-30T03:48:55.090341Z","iopub.execute_input":"2024-10-30T03:48:55.090857Z","iopub.status.idle":"2024-10-30T03:48:55.129889Z","shell.execute_reply.started":"2024-10-30T03:48:55.090804Z","shell.execute_reply":"2024-10-30T03:48:55.128607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-10-30T03:48:55.131354Z","iopub.execute_input":"2024-10-30T03:48:55.131754Z","iopub.status.idle":"2024-10-30T03:58:46.764863Z","shell.execute_reply.started":"2024-10-30T03:48:55.131715Z","shell.execute_reply":"2024-10-30T03:58:46.763068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature extraction and Training: Model 2","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n\n# Model parameters for LightGBM\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n\n# Train the ensemble model\nSubmission2 = TrainML(voting_model, test)\n\n# Save submission\n#Submission2.to_csv('submission.csv', index=False)\nSubmission2","metadata":{"execution":{"iopub.status.busy":"2024-10-30T03:58:46.768176Z","iopub.execute_input":"2024-10-30T03:58:46.768600Z","iopub.status.idle":"2024-10-30T04:01:30.098535Z","shell.execute_reply.started":"2024-10-30T03:58:46.768555Z","shell.execute_reply":"2024-10-30T04:01:30.097324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature extraction and Training: Model 3","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    return tp_rounded\n\nimputer = SimpleImputer(strategy='median')\n\nensemble = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),\n    ('xgb', Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=SEED))])),\n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))]))\n])\n\nSubmission3 = TrainML(ensemble, test)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:01:30.100454Z","iopub.execute_input":"2024-10-30T04:01:30.100972Z","iopub.status.idle":"2024-10-30T04:05:33.854615Z","shell.execute_reply.started":"2024-10-30T04:01:30.100895Z","shell.execute_reply":"2024-10-30T04:05:33.853374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Submission3 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission3\n})\n\nSubmission3","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:33.856070Z","iopub.execute_input":"2024-10-30T04:05:33.856428Z","iopub.status.idle":"2024-10-30T04:05:33.870425Z","shell.execute_reply.started":"2024-10-30T04:05:33.856391Z","shell.execute_reply":"2024-10-30T04:05:33.869230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature extraction and Training: Model 1","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\n\nfrom sklearn.preprocessing import MinMaxScaler, StandardScaler, RobustScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nSEED = 42\nn_splits = 5","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:33.871954Z","iopub.execute_input":"2024-10-30T04:05:33.872392Z","iopub.status.idle":"2024-10-30T04:05:33.888257Z","shell.execute_reply.started":"2024-10-30T04:05:33.872350Z","shell.execute_reply":"2024-10-30T04:05:33.887019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Sparse Autoencoder Model\nclass SparseAutoencoder(nn.Module):\n    def __init__(self, input_dim, sparsity_weight=1e-5):\n        super(SparseAutoencoder, self).__init__()\n        self.sparsity_weight = sparsity_weight\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, 64),\n            nn.ReLU(),\n            nn.Linear(64, 32),\n            nn.ReLU(),\n            nn.Linear(32, 16),\n            nn.ReLU()\n        )\n        \n        self.decoder = nn.Sequential(\n            nn.Linear(16, 32),\n            nn.ReLU(),\n            nn.Linear(32, 64),\n            nn.ReLU(),\n            nn.Linear(64, input_dim),\n            nn.Sigmoid()  # Outputs in the range [0, 1]\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return encoded, decoded\n\n# Preparing Data\n# Option to use different scalers: MinMaxScaler, StandardScaler, RobustScaler\ndef prepare_data(data, scaler_type='MinMaxScaler'):\n    if scaler_type == 'StandardScaler':\n        scaler = StandardScaler()\n    elif scaler_type == 'RobustScaler':\n        scaler = RobustScaler()\n    else:\n        scaler = MinMaxScaler()\n    \n    data_scaled = scaler.fit_transform(data)\n    return torch.tensor(data_scaled, dtype=torch.float32), scaler\n\n# Apply PCA for Dimensionality Reduction\n# This can help focus the autoencoder on the most relevant features\ndef apply_pca(data, n_components=0.95):\n    pca = PCA(n_components=n_components)\n    data_pca = pca.fit_transform(data)\n    return data_pca, pca\n\n# Early Stopping Functionality\ndef early_stopping(patience):\n    class EarlyStopping:\n        def __init__(self, patience=patience):\n            self.patience = patience\n            self.counter = 0\n            self.best_loss = float('inf')\n            self.early_stop = False\n        \n        def __call__(self, loss):\n            if loss < self.best_loss:\n                self.best_loss = loss\n                self.counter = 0\n            else:\n                self.counter += 1\n                if self.counter >= self.patience:\n                    self.early_stop = True\n    return EarlyStopping()\n\n# Training the Sparse Autoencoder with DataFrame Output\ndef perform_autoencoder(data, epochs=100, batch_size=32, learning_rate=0.001, patience=10, scaler_type='MinMaxScaler', use_pca=False, sparsity_weight=1e-5):\n    # Preprocess Data\n    if use_pca:\n        data, pca = apply_pca(data)\n\n    data_tensor, scaler = prepare_data(data, scaler_type=scaler_type)\n    train_data, val_data = train_test_split(data_tensor, test_size=0.2, random_state=42)\n\n    train_loader = DataLoader(TensorDataset(train_data), batch_size=batch_size, shuffle=True)\n    val_loader = DataLoader(TensorDataset(val_data), batch_size=batch_size, shuffle=False)\n\n    model = SparseAutoencoder(input_dim=data.shape[1], sparsity_weight=sparsity_weight)\n    device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n    model.to(device)\n\n    criterion = nn.SmoothL1Loss()  # Changed to Smooth L1 Loss\n    optimizer = optim.Adam(model.parameters(), lr=learning_rate)\n    stopper = early_stopping(patience=patience)\n\n    for epoch in range(epochs):\n        model.train()\n        train_loss = 0.0\n        for batch in train_loader:\n            batch = batch[0].to(device)\n            optimizer.zero_grad()\n            encoded, outputs = model(batch)\n            \n            # Reconstruction loss\n            loss = criterion(outputs, batch)\n            \n            # Sparsity penalty (L1 regularization on encoded activations)\n            l1_penalty = torch.mean(torch.abs(encoded))\n            loss += sparsity_weight * l1_penalty\n            \n            loss.backward()\n            optimizer.step()\n            train_loss += loss.item() * batch.size(0)\n\n        train_loss /= len(train_loader.dataset)\n\n        # Validation\n        model.eval()\n        val_loss = 0.0\n        with torch.no_grad():\n            for batch in val_loader:\n                batch = batch[0].to(device)\n                _, outputs = model(batch)\n                loss = criterion(outputs, batch)\n                val_loss += loss.item() * batch.size(0)\n\n        val_loss /= len(val_loader.dataset)\n        print(f\"Epoch {epoch+1}, Train Loss: {train_loss:.4f}, Validation Loss: {val_loss:.4f}\")\n\n        # Early stopping\n        stopper(val_loss)\n        if stopper.early_stop:\n            print(f\"Early stopping at epoch {epoch + 1}\")\n            break\n\n    # Convert tensor back to DataFrame for consistency\n    _, data_decoded = model(data_tensor.to(device))\n    data_decoded = data_decoded.cpu().detach().numpy()\n    df_encoded = pd.DataFrame(data_decoded, columns=[f'feature_{i}' for i in range(data_decoded.shape[1])])\n    return df_encoded\n\n# Usage example\n# Assuming 'data' is your input dataset as a NumPy array or pandas DataFrame.\n# df_encoded = train_sparse_autoencoder(data, epochs=100, batch_size=32, learning_rate=0.001, patience=10, scaler_type='StandardScaler', use_pca=True, sparsity_weight=1e-5)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:33.890224Z","iopub.execute_input":"2024-10-30T04:05:33.890712Z","iopub.status.idle":"2024-10-30T04:05:33.916803Z","shell.execute_reply.started":"2024-10-30T04:05:33.890657Z","shell.execute_reply":"2024-10-30T04:05:33.915504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n\n    df['Age_Weight'] = df['Basic_Demos-Age'] * df['Physical-Weight']\n    df['Sex_BMI'] = df['Basic_Demos-Sex'] * df['Physical-BMI']\n    df['Sex_HeartRate'] = df['Basic_Demos-Sex'] * df['Physical-HeartRate']\n    df['Age_WaistCirc'] = df['Basic_Demos-Age'] * df['Physical-Waist_Circumference']\n    df['BMI_FitnessMaxStage'] = df['Physical-BMI'] * df['Fitness_Endurance-Max_Stage']\n    df['Weight_GripStrengthDominant'] = df['Physical-Weight'] * df['FGC-FGC_GSD']\n    df['Weight_GripStrengthNonDominant'] = df['Physical-Weight'] * df['FGC-FGC_GSND']\n    df['HeartRate_FitnessTime'] = df['Physical-HeartRate'] * (df['Fitness_Endurance-Time_Mins'] + df['Fitness_Endurance-Time_Sec'])\n    df['Age_PushUp'] = df['Basic_Demos-Age'] * df['FGC-FGC_PU']\n    df['FFMI_Age'] = df['BIA-BIA_FFMI'] * df['Basic_Demos-Age']\n    df['InternetUse_SleepDisturbance'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['SDS-SDS_Total_Raw']\n    df['CGAS_BMI'] = df['CGAS-CGAS_Score'] * df['Physical-BMI']\n    df['CGAS_FitnessMaxStage'] = df['CGAS-CGAS_Score'] * df['Fitness_Endurance-Max_Stage']\n    \n    return df\n\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n# train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:33.921877Z","iopub.execute_input":"2024-10-30T04:05:33.922738Z","iopub.status.idle":"2024-10-30T04:05:34.194943Z","shell.execute_reply.started":"2024-10-30T04:05:33.922685Z","shell.execute_reply":"2024-10-30T04:05:34.193828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ts","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:34.196381Z","iopub.execute_input":"2024-10-30T04:05:34.196730Z","iopub.status.idle":"2024-10-30T04:05:34.311428Z","shell.execute_reply.started":"2024-10-30T04:05:34.196691Z","shell.execute_reply":"2024-10-30T04:05:34.310327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ts_encoded = train_ts\ntest_ts_encoded = test_ts\n\ntime_series_cols = train_ts_encoded.columns.tolist()\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded['id']=test_ts[\"id\"]\n\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:34.312929Z","iopub.execute_input":"2024-10-30T04:05:34.313424Z","iopub.status.idle":"2024-10-30T04:05:34.333111Z","shell.execute_reply.started":"2024-10-30T04:05:34.313367Z","shell.execute_reply":"2024-10-30T04:05:34.331738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:34.334775Z","iopub.execute_input":"2024-10-30T04:05:34.335291Z","iopub.status.idle":"2024-10-30T04:05:34.343468Z","shell.execute_reply.started":"2024-10-30T04:05:34.335233Z","shell.execute_reply":"2024-10-30T04:05:34.342118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:34.345377Z","iopub.execute_input":"2024-10-30T04:05:34.345851Z","iopub.status.idle":"2024-10-30T04:05:34.359117Z","shell.execute_reply.started":"2024-10-30T04:05:34.345796Z","shell.execute_reply":"2024-10-30T04:05:34.357971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imputer = KNNImputer(n_neighbors=7)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n\ntrain = train_imputed\n\ntrain = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:34.360764Z","iopub.execute_input":"2024-10-30T04:05:34.361637Z","iopub.status.idle":"2024-10-30T04:05:50.252968Z","shell.execute_reply.started":"2024-10-30T04:05:34.361582Z","shell.execute_reply":"2024-10-30T04:05:50.251790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=train\n\n# Sample DataFrame (replace this with your actual DataFrame)\n# df = pd.read_csv('your_data.csv')  # Load your data\n\n# 1. Age Categories\nage_bins = [0, 18, 25, 35, 45, 55, 65, 100]\nage_labels = ['<18', '18-25', '26-35', '36-45', '46-55', '56-65', '65+']\n# df['Age_Category'] = pd.cut(df['Basic_Demos-Age'], bins=age_bins, labels=age_labels)\n\n# # 2. BMI Categories\n# def categorize_bmi(bmi):\n#     if bmi < 18.5:\n#         return 'Underweight'\n#     elif 18.5 <= bmi < 24.9:\n#         return 'Normal'\n#     elif 25 <= bmi < 29.9:\n#         return 'Overweight'\n#     else:\n#         return 'Obese'\n\n# df['BMI_Category'] = df['Physical-BMI'].apply(categorize_bmi)\n\n# 3. Physical Health Index\ndf['Physical_Health_Index'] = (df['Physical-BMI'] + df['Physical-Waist_Circumference'] +\n                                df['Physical-Diastolic_BP'] + df['Physical-Systolic_BP']) / 4\n\n# 4. Fitness Endurance Score\ndf['Fitness_Endurance_Score'] = df['Fitness_Endurance-Max_Stage'] + (df['Fitness_Endurance-Time_Mins'] * 60 + df['Fitness_Endurance-Time_Sec'])\n\n# 5. Normalize Height and Weight\ndf['Height_Norm'] = (df['Physical-Height'] - df['Physical-Height'].mean()) / df['Physical-Height'].std()\ndf['Weight_Norm'] = (df['Physical-Weight'] - df['Physical-Weight'].mean()) / df['Physical-Weight'].std()\n\n# 6. Body Composition Ratios\ndf['Fat_to_Lean_Mass_Ratio'] = df['BIA-BIA_Fat'] / df['BIA-BIA_FFM']\n\n# 7. Create Engagement Metric\ndf['Computer_Engagement'] = df['PreInt_EduHx-computerinternet_hoursday'] * 7  # Weekly hours\n\n# 8. Cumulative Scores\ndf['Cumulative_PAQ'] = df['PAQ_A-PAQ_A_Total'] + df['PAQ_C-PAQ_C_Total']\n\n# 9. Analyze Zone Percentage (example for FGC)\ndf['FGC_Percentage'] = df[['FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', \n                             'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', \n                             'FGC-FGC_GSD_Zone']].sum(axis=1) / len(df.columns)\n\n# 10. Flag Missing Values\ndf['Missing_Health_Data'] = df[['Physical-BMI', 'Physical-Height', 'Physical-Weight']].isnull().any(axis=1).astype(int)\n# df['Age_Category'] = df['Age_Category'].astype(str)\ntrain=df\n\n# test_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ndf=test\n# test_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\nimport pandas as pd\nimport numpy as np\n\n# Sample DataFrame (replace this with your actual DataFrame)\n# df = pd.read_csv('your_data.csv')  # Load your data\n\n# 1. Age Categories\nage_bins = [0, 18, 25, 35, 45, 55, 65, 100]\nage_labels = ['<18', '18-25', '26-35', '36-45', '46-55', '56-65', '65+']\n# df['Age_Category'] = pd.cut(df['Basic_Demos-Age'], bins=age_bins, labels=age_labels)\n\n# # 2. BMI Categories\n# def categorize_bmi(bmi):\n#     if bmi < 18.5:\n#         return 'Underweight'\n#     elif 18.5 <= bmi < 24.9:\n#         return 'Normal'\n#     elif 25 <= bmi < 29.9:\n#         return 'Overweight'\n#     else:\n#         return 'Obese'\n\n# df['BMI_Category'] = df['Physical-BMI'].apply(categorize_bmi)\n\n# 3. Physical Health Index\ndf['Physical_Health_Index'] = (df['Physical-BMI'] + df['Physical-Waist_Circumference'] +\n                                df['Physical-Diastolic_BP'] + df['Physical-Systolic_BP']) / 4\n\n# 4. Fitness Endurance Score\ndf['Fitness_Endurance_Score'] = df['Fitness_Endurance-Max_Stage'] + (df['Fitness_Endurance-Time_Mins'] * 60 + df['Fitness_Endurance-Time_Sec'])\n\n# 5. Normalize Height and Weight\ndf['Height_Norm'] = (df['Physical-Height'] - df['Physical-Height'].mean()) / df['Physical-Height'].std()\ndf['Weight_Norm'] = (df['Physical-Weight'] - df['Physical-Weight'].mean()) / df['Physical-Weight'].std()\n\n# 6. Body Composition Ratios\ndf['Fat_to_Lean_Mass_Ratio'] = df['BIA-BIA_Fat'] / df['BIA-BIA_FFM']\n\n# 7. Create Engagement Metric\ndf['Computer_Engagement'] = df['PreInt_EduHx-computerinternet_hoursday'] * 7  # Weekly hours\n\n# 8. Cumulative Scores\ndf['Cumulative_PAQ'] = df['PAQ_A-PAQ_A_Total'] + df['PAQ_C-PAQ_C_Total']\n\n# 9. Analyze Zone Percentage (example for FGC)\ndf['FGC_Percentage'] = df[['FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', \n                             'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', \n                             'FGC-FGC_GSD_Zone']].sum(axis=1) / len(df.columns)\n\n# 10. Flag Missing Values\ndf['Missing_Health_Data'] = df[['Physical-BMI', 'Physical-Height', 'Physical-Weight']].isnull().any(axis=1).astype(int)\n# df['Age_Category'] = df['Age_Category'].astype(str)\ntest=df\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\n# train = pd.merge(train, train_ts, how=\"left\", on='id')\n# test = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday','Missing_Health_Data','FGC_Percentage','Cumulative_PAQ',\n 'Fitness_Endurance_Score',\n 'Height_Norm',\n 'Weight_Norm',\n 'Computer_Engagement',\n 'Fat_to_Lean_Mass_Ratio',\n 'Physical_Health_Index', 'sii']\n\nfeaturesCols += time_series_cols","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:50.254813Z","iopub.execute_input":"2024-10-30T04:05:50.255245Z","iopub.status.idle":"2024-10-30T04:05:50.301668Z","shell.execute_reply.started":"2024-10-30T04:05:50.255205Z","shell.execute_reply":"2024-10-30T04:05:50.300522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(featuresCols)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:50.303288Z","iopub.execute_input":"2024-10-30T04:05:50.303769Z","iopub.status.idle":"2024-10-30T04:05:50.312306Z","shell.execute_reply.started":"2024-10-30T04:05:50.303714Z","shell.execute_reply":"2024-10-30T04:05:50.310940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"time_series_cols","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:50.314275Z","iopub.execute_input":"2024-10-30T04:05:50.315459Z","iopub.status.idle":"2024-10-30T04:05:50.324443Z","shell.execute_reply.started":"2024-10-30T04:05:50.315401Z","shell.execute_reply":"2024-10-30T04:05:50.323251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:50.325890Z","iopub.execute_input":"2024-10-30T04:05:50.326790Z","iopub.status.idle":"2024-10-30T04:05:50.338416Z","shell.execute_reply.started":"2024-10-30T04:05:50.326734Z","shell.execute_reply":"2024-10-30T04:05:50.337200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col=list(test.columns)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:50.340104Z","iopub.execute_input":"2024-10-30T04:05:50.340533Z","iopub.status.idle":"2024-10-30T04:05:50.348965Z","shell.execute_reply.started":"2024-10-30T04:05:50.340491Z","shell.execute_reply":"2024-10-30T04:05:50.347623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col.append('sii')","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:50.350542Z","iopub.execute_input":"2024-10-30T04:05:50.350974Z","iopub.status.idle":"2024-10-30T04:05:50.360883Z","shell.execute_reply.started":"2024-10-30T04:05:50.350926Z","shell.execute_reply":"2024-10-30T04:05:50.359404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:50.362377Z","iopub.execute_input":"2024-10-30T04:05:50.362807Z","iopub.status.idle":"2024-10-30T04:05:50.377719Z","shell.execute_reply.started":"2024-10-30T04:05:50.362765Z","shell.execute_reply":"2024-10-30T04:05:50.376587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train[col]","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:50.379622Z","iopub.execute_input":"2024-10-30T04:05:50.380011Z","iopub.status.idle":"2024-10-30T04:05:50.392863Z","shell.execute_reply.started":"2024-10-30T04:05:50.379963Z","shell.execute_reply":"2024-10-30T04:05:50.391583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:50.394638Z","iopub.execute_input":"2024-10-30T04:05:50.395175Z","iopub.status.idle":"2024-10-30T04:05:50.562118Z","shell.execute_reply.started":"2024-10-30T04:05:50.395118Z","shell.execute_reply":"2024-10-30T04:05:50.561083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain = train.dropna(subset='sii')\n","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:50.566474Z","iopub.execute_input":"2024-10-30T04:05:50.566865Z","iopub.status.idle":"2024-10-30T04:05:50.578281Z","shell.execute_reply.started":"2024-10-30T04:05:50.566824Z","shell.execute_reply":"2024-10-30T04:05:50.577039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:50.579854Z","iopub.execute_input":"2024-10-30T04:05:50.580504Z","iopub.status.idle":"2024-10-30T04:05:50.595662Z","shell.execute_reply.started":"2024-10-30T04:05:50.580448Z","shell.execute_reply":"2024-10-30T04:05:50.594425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:50.597185Z","iopub.execute_input":"2024-10-30T04:05:50.597548Z","iopub.status.idle":"2024-10-30T04:05:50.615964Z","shell.execute_reply.started":"2024-10-30T04:05:50.597510Z","shell.execute_reply":"2024-10-30T04:05:50.614554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  # Increased from 6.59\n    'lambda_l2': 0.01  # Increased from 2.68e-06\n}\n\n\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n    'tree_method': 'exact'\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10  # Increase this value\n}\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:50.617597Z","iopub.execute_input":"2024-10-30T04:05:50.618129Z","iopub.status.idle":"2024-10-30T04:05:50.633199Z","shell.execute_reply.started":"2024-10-30T04:05:50.618073Z","shell.execute_reply":"2024-10-30T04:05:50.631935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Submission1 = TrainML(voting_model, test)\n\n# Save submission\nSubmission1.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:05:50.634834Z","iopub.execute_input":"2024-10-30T04:05:50.635364Z","iopub.status.idle":"2024-10-30T04:06:55.836998Z","shell.execute_reply.started":"2024-10-30T04:05:50.635308Z","shell.execute_reply":"2024-10-30T04:06:55.835970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Final submission: majority vote from submission 1, 2 and 3","metadata":{}},{"cell_type":"code","source":"sub1 = Submission1\nsub2 = Submission2\nsub3 = Submission3\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii']\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\nfinal_submission.to_csv('submission.csv', index=False)\n\nprint(\"Majority voting completed and saved to 'Final_Submission.csv'\")","metadata":{"execution":{"iopub.status.busy":"2024-10-30T04:06:55.838434Z","iopub.execute_input":"2024-10-30T04:06:55.838796Z","iopub.status.idle":"2024-10-30T04:06:55.862496Z","shell.execute_reply.started":"2024-10-30T04:06:55.838758Z","shell.execute_reply":"2024-10-30T04:06:55.861328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}