{"metadata":{"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"papermill":{"default_parameters":{},"duration":1390.651064,"end_time":"2024-10-26T13:26:42.016500","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-10-26T13:03:31.365436","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\nfrom sklearn.preprocessing import MinMaxScaler, StandardScaler, RobustScaler\n\nSEED = 42\nn_splits = 5","metadata":{"execution":{"iopub.status.busy":"2024-11-03T12:03:19.773936Z","iopub.execute_input":"2024-11-03T12:03:19.774278Z","iopub.status.idle":"2024-11-03T12:03:37.776559Z","shell.execute_reply.started":"2024-11-03T12:03:19.774245Z","shell.execute_reply":"2024-11-03T12:03:37.775578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GLOBAL_TS_LENGTH=[]\nimport pandas as pd\nimport numpy as np\nfrom scipy import stats\n\ndef extract_advanced_features(data):\n    \n    \n    data = data.copy()\n    data['timestamp'] = pd.to_datetime(data['relative_date_PCIAT'], unit='D') + pd.to_timedelta(data['time_of_day'])\n    data = data[data['non-wear_flag'] == 0]\n    \n    \n    data['magnitude'] = np.sqrt(data['X']**2 + data['Y']**2 + data['Z']**2)\n    data['velocity'] = data['magnitude']\n    data['distance'] = data['velocity'] * 5  \n    data['date'] = data['timestamp'].dt.date\n    hour = pd.to_datetime(data['time_of_day']).dt.hour\n    \n    \n    distances = {\n        'daily': data.groupby('date')['distance'].sum(),\n        'monthly': data.groupby(data['timestamp'].dt.to_period('M'))['distance'].sum(),\n        'quarterly': data.groupby('quarter')['distance'].sum()\n    }\n    \n    \n    features = {}\n    \n    \n    time_masks = {\n        'morning': (hour >= 6) & (hour < 12),\n        'afternoon': (hour >= 12) & (hour < 18),\n        'evening': (hour >= 18) & (hour < 22),\n        'night': (hour >= 22) | (hour < 6)\n    }\n    \n    \n    for period, mask in time_masks.items():\n        features.update({\n            f'{period}_activity_mean': data.loc[mask, 'magnitude'].mean(),\n            f'{period}_activity_std': data.loc[mask, 'magnitude'].std(),\n            f'{period}_enmo_mean': data.loc[mask, 'enmo'].mean()\n        })\n    \n    \n    sleep_hours = time_masks['night']\n    magnitude_threshold = data['magnitude'].mean() + data['magnitude'].std()\n    \n    features.update({\n        'sleep_movement_mean': data.loc[sleep_hours, 'magnitude'].mean(),\n        'sleep_movement_std': data.loc[sleep_hours, 'magnitude'].std(),\n        'sleep_disruption_count': len(data.loc[sleep_hours & (data['magnitude'] > \n            data['magnitude'].mean() + 2 * data['magnitude'].std())]),\n        'light_exposure_during_sleep': data.loc[sleep_hours, 'light'].mean(),\n        'sleep_position_changes': len(data.loc[sleep_hours & \n            (abs(data['anglez'].diff()) > 45)]),\n        'good_sleep_cycle': int(data.loc[sleep_hours, 'light'].mean() < 50)\n    })\n    \n    \n    features.update({\n        'sedentary_time_ratio': (data['magnitude'] < magnitude_threshold * 0.5).mean(),\n        'moderate_activity_ratio': ((data['magnitude'] >= magnitude_threshold * 0.5) & \n            (data['magnitude'] < magnitude_threshold * 1.5)).mean(),\n        'vigorous_activity_ratio': (data['magnitude'] >= magnitude_threshold * 1.5).mean(),\n        'activity_peaks_per_day': len(data[data['magnitude'] > \n            data['magnitude'].quantile(0.95)]) / len(data.groupby('relative_date_PCIAT'))\n    })\n    \n    \n    hourly_activity = data.groupby(hour)['magnitude'].mean()\n    features.update({\n        'circadian_regularity': hourly_activity.std() / hourly_activity.mean(),\n        'peak_activity_hour': hourly_activity.idxmax(),\n        'trough_activity_hour': hourly_activity.idxmin(),\n        'activity_range': hourly_activity.max() - hourly_activity.min()\n    })\n    \n    \n    weekend_mask = data['weekday'].isin([6, 7])\n    \n    features.update({\n        \n        'movement_entropy': stats.entropy(pd.qcut(data['magnitude'], q=10, duplicates='drop').value_counts()),\n        'direction_changes': len(data[abs(data['anglez'].diff()) > 30]) / len(data),\n        'sustained_activity_periods': len(data[data['magnitude'].rolling(12).mean() > \n            magnitude_threshold]) / len(data),\n        \n        \n        'weekend_activity_ratio': data.loc[weekend_mask, 'magnitude'].mean() / \n            data.loc[~weekend_mask, 'magnitude'].mean(),\n        'weekend_sleep_difference': data.loc[weekend_mask & sleep_hours, 'magnitude'].mean() - \n            data.loc[~weekend_mask & sleep_hours, 'magnitude'].mean(),\n        \n        \n        'wear_time_ratio': (data['non-wear_flag'] == 0).mean(),\n        'wear_consistency': len(data['non-wear_flag'].value_counts()),\n        'longest_wear_streak': data['non-wear_flag'].eq(0).astype(int).groupby(\n            data['non-wear_flag'].ne(0).cumsum()).sum().max(),\n        \n    \n        'screen_time_proxy': (data['light'] > data['light'].quantile(0.75)).mean(),\n        'dark_environment_ratio': (data['light'] < data['light'].quantile(0.25)).mean(),\n        'light_variation': data['light'].std() / data['light'].mean() if data['light'].mean() != 0 else 0,\n        \n     \n        'battery_drain_rate': -np.polyfit(range(len(data)), data['battery_voltage'], 1)[0],\n        'battery_variability': data['battery_voltage'].std(),\n        'low_battery_time': (data['battery_voltage'] < data['battery_voltage'].quantile(0.1)).mean(),\n        \n   \n        'days_monitored': data['relative_date_PCIAT'].nunique(),\n        'total_active_hours': len(data[data['magnitude'] > magnitude_threshold * 0.5]) * 5 / 3600,\n        'activity_regularity': data.groupby('weekday')['magnitude'].mean().std()\n    })\n    \n  \n    for col in ['X', 'Y', 'Z', 'enmo', 'anglez']:\n        features.update({\n            f'{col}_skewness': data[col].skew(),\n            f'{col}_kurtosis': data[col].kurtosis(),\n            f'{col}_trend': np.polyfit(range(len(data)), data[col], 1)[0]\n        })\n    \n    return pd.DataFrame([features])\n\ndef process_file(filename, dirname):\n    df= pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    data=extract_advanced_features(df)\n    array_1=data.values[0]\n    array_2=df.describe().values.reshape(-1), filename.split('=')[1]\n   \n    combined_array = np.concatenate((array_1, array_2[0]))\n    combined_tuple=(array_1,array_2[1])\n    return combined_tuple\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n","metadata":{"papermill":{"duration":0.060736,"end_time":"2024-10-26T13:03:59.605112","exception":false,"start_time":"2024-10-26T13:03:59.544376","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:03:37.778869Z","iopub.execute_input":"2024-11-03T12:03:37.779667Z","iopub.status.idle":"2024-11-03T12:03:37.816058Z","shell.execute_reply.started":"2024-11-03T12:03:37.779621Z","shell.execute_reply":"2024-11-03T12:03:37.815075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\n","metadata":{"papermill":{"duration":834.539775,"end_time":"2024-10-26T13:17:54.157698","exception":false,"start_time":"2024-10-26T13:03:59.617923","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:03:37.817190Z","iopub.execute_input":"2024-11-03T12:03:37.817587Z","iopub.status.idle":"2024-11-03T12:14:50.211741Z","shell.execute_reply.started":"2024-11-03T12:03:37.817536Z","shell.execute_reply":"2024-11-03T12:14:50.210456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        clear_output(wait=True)\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n\n\nParams = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  \n    'lambda_l2': 0.01  \n}\n\n\n\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1, \n    'reg_lambda': 5,  \n    'random_state': SEED\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'cat_features': cat_c,\n    'verbose': 0,\n    'l2_leaf_reg': 10  \n}\n\n\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n\n# Train the ensemble model\nSubmission2 = TrainML(voting_model, test)\n\nSubmission2\n","metadata":{"papermill":{"duration":174.271896,"end_time":"2024-10-26T13:20:48.506227","exception":false,"start_time":"2024-10-26T13:17:54.234331","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:14:50.214492Z","iopub.execute_input":"2024-11-03T12:14:50.214845Z","iopub.status.idle":"2024-11-03T12:17:01.802385Z","shell.execute_reply.started":"2024-11-03T12:14:50.214800Z","shell.execute_reply":"2024-11-03T12:17:01.801454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        \n        clear_output(wait=True)\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    return tp_rounded\n\nimputer = SimpleImputer(strategy='median')\n\nensemble = VotingRegressor(estimators=[\n    ('lgb', Pipeline(steps=[('imputer', imputer), ('regressor', LGBMRegressor(random_state=SEED))])),\n    ('xgb', Pipeline(steps=[('imputer', imputer), ('regressor', XGBRegressor(random_state=SEED))])),\n    ('cat', Pipeline(steps=[('imputer', imputer), ('regressor', CatBoostRegressor(random_state=SEED, silent=True))])),\n    ('rf', Pipeline(steps=[('imputer', imputer), ('regressor', RandomForestRegressor(random_state=SEED))])),\n    ('gb', Pipeline(steps=[('imputer', imputer), ('regressor', GradientBoostingRegressor(random_state=SEED))]))\n])\n\nSubmission3 = TrainML(ensemble, test)\n","metadata":{"papermill":{"duration":260.243223,"end_time":"2024-10-26T13:25:08.812396","exception":false,"start_time":"2024-10-26T13:20:48.569173","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:17:01.803680Z","iopub.execute_input":"2024-11-03T12:17:01.804038Z","iopub.status.idle":"2024-11-03T12:20:23.588182Z","shell.execute_reply.started":"2024-11-03T12:17:01.804004Z","shell.execute_reply":"2024-11-03T12:20:23.587328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Submission3 = pd.DataFrame({\n    'id': sample['id'],\n    'sii': Submission3\n})\n\nSubmission3\n","metadata":{"papermill":{"duration":0.084803,"end_time":"2024-10-26T13:25:08.962251","exception":false,"start_time":"2024-10-26T13:25:08.877448","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:20:23.589756Z","iopub.execute_input":"2024-11-03T12:20:23.590159Z","iopub.status.idle":"2024-11-03T12:20:23.601860Z","shell.execute_reply.started":"2024-11-03T12:20:23.590101Z","shell.execute_reply":"2024-11-03T12:20:23.600964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\n\nfrom sklearn.preprocessing import MinMaxScaler, StandardScaler, RobustScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nSEED = 42\nn_splits = 5\n","metadata":{"papermill":{"duration":0.08199,"end_time":"2024-10-26T13:25:09.398509","exception":false,"start_time":"2024-10-26T13:25:09.316519","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:20:23.603098Z","iopub.execute_input":"2024-11-03T12:20:23.603425Z","iopub.status.idle":"2024-11-03T12:20:23.618050Z","shell.execute_reply.started":"2024-11-03T12:20:23.603386Z","shell.execute_reply":"2024-11-03T12:20:23.617264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nclass SparseAutoencoder(nn.Module):\n    def __init__(self, input_dim, sparsity_weight=1e-5):\n        super(SparseAutoencoder, self).__init__()\n        self.sparsity_weight = sparsity_weight\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, 64),\n            nn.ReLU(),\n            nn.Linear(64, 32),\n            nn.ReLU(),\n            nn.Linear(32, 16),\n            nn.ReLU()\n        )\n        \n        self.decoder = nn.Sequential(\n            nn.Linear(16, 32),\n            nn.ReLU(),\n            nn.Linear(32, 64),\n            nn.ReLU(),\n            nn.Linear(64, input_dim),\n            nn.Sigmoid()  \n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return encoded, decoded\n\n\ndef prepare_data(data, scaler_type='MinMaxScaler'):\n    if scaler_type == 'StandardScaler':\n        scaler = StandardScaler()\n    elif scaler_type == 'RobustScaler':\n        scaler = RobustScaler()\n    else:\n        scaler = MinMaxScaler()\n    \n    data_scaled = scaler.fit_transform(data)\n    return torch.tensor(data_scaled, dtype=torch.float32), scaler\n\n\ndef apply_pca(data, n_components=0.95):\n    pca = PCA(n_components=n_components)\n    data_pca = pca.fit_transform(data)\n    return data_pca, pca\n\n\ndef early_stopping(patience):\n    class EarlyStopping:\n        def __init__(self, patience=patience):\n            self.patience = patience\n            self.counter = 0\n            self.best_loss = float('inf')\n            self.early_stop = False\n        \n        def __call__(self, loss):\n            if loss < self.best_loss:\n                self.best_loss = loss\n                self.counter = 0\n            else:\n                self.counter += 1\n                if self.counter >= self.patience:\n                    self.early_stop = True\n    return EarlyStopping()\n\n\ndef perform_autoencoder(data, epochs=100, batch_size=32, learning_rate=0.001, patience=10, scaler_type='MinMaxScaler', use_pca=False, sparsity_weight=1e-5):\n    \n    if use_pca:\n        data, pca = apply_pca(data)\n\n    data_tensor, scaler = prepare_data(data, scaler_type=scaler_type)\n    train_data, val_data = train_test_split(data_tensor, test_size=0.2, random_state=42)\n\n    train_loader = DataLoader(TensorDataset(train_data), batch_size=batch_size, shuffle=True)\n    val_loader = DataLoader(TensorDataset(val_data), batch_size=batch_size, shuffle=False)\n\n    model = SparseAutoencoder(input_dim=data.shape[1], sparsity_weight=sparsity_weight)\n    device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n    model.to(device)\n\n    criterion = nn.SmoothL1Loss()  \n    optimizer = optim.Adam(model.parameters(), lr=learning_rate)\n    stopper = early_stopping(patience=patience)\n\n    for epoch in range(epochs):\n        model.train()\n        train_loss = 0.0\n        for batch in train_loader:\n            batch = batch[0].to(device)\n            optimizer.zero_grad()\n            encoded, outputs = model(batch)\n            \n            \n            loss = criterion(outputs, batch)\n            \n            \n            l1_penalty = torch.mean(torch.abs(encoded))\n            loss += sparsity_weight * l1_penalty\n            \n            loss.backward()\n            optimizer.step()\n            train_loss += loss.item() * batch.size(0)\n\n        train_loss /= len(train_loader.dataset)\n\n        \n        model.eval()\n        val_loss = 0.0\n        with torch.no_grad():\n            for batch in val_loader:\n                batch = batch[0].to(device)\n                _, outputs = model(batch)\n                loss = criterion(outputs, batch)\n                val_loss += loss.item() * batch.size(0)\n\n        val_loss /= len(val_loader.dataset)\n       \n\n        \n        stopper(val_loss)\n        if stopper.early_stop:\n            print(f\"Early stopping at epoch {epoch + 1}\")\n            break\n\n    \n    _, data_decoded = model(data_tensor.to(device))\n    data_decoded = data_decoded.cpu().detach().numpy()\n    df_encoded = pd.DataFrame(data_decoded, columns=[f'feature_{i}' for i in range(data_decoded.shape[1])])\n    return df_encoded\n","metadata":{"papermill":{"duration":0.098455,"end_time":"2024-10-26T13:25:09.560827","exception":false,"start_time":"2024-10-26T13:25:09.462372","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:20:23.619353Z","iopub.execute_input":"2024-11-03T12:20:23.619664Z","iopub.status.idle":"2024-11-03T12:20:23.643589Z","shell.execute_reply.started":"2024-11-03T12:20:23.619630Z","shell.execute_reply":"2024-11-03T12:20:23.642585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineering(df):\n    season_cols = [col for col in df.columns if 'Season' in col]\n    df = df.drop(season_cols, axis=1) \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n\n    df['Age_Weight'] = df['Basic_Demos-Age'] * df['Physical-Weight']\n    df['Sex_BMI'] = df['Basic_Demos-Sex'] * df['Physical-BMI']\n    df['Sex_HeartRate'] = df['Basic_Demos-Sex'] * df['Physical-HeartRate']\n    df['Age_WaistCirc'] = df['Basic_Demos-Age'] * df['Physical-Waist_Circumference']\n    df['BMI_FitnessMaxStage'] = df['Physical-BMI'] * df['Fitness_Endurance-Max_Stage']\n    df['Weight_GripStrengthDominant'] = df['Physical-Weight'] * df['FGC-FGC_GSD']\n    df['Weight_GripStrengthNonDominant'] = df['Physical-Weight'] * df['FGC-FGC_GSND']\n    df['HeartRate_FitnessTime'] = df['Physical-HeartRate'] * (df['Fitness_Endurance-Time_Mins'] + df['Fitness_Endurance-Time_Sec'])\n    df['Age_PushUp'] = df['Basic_Demos-Age'] * df['FGC-FGC_PU']\n    df['FFMI_Age'] = df['BIA-BIA_FFMI'] * df['Basic_Demos-Age']\n    df['InternetUse_SleepDisturbance'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['SDS-SDS_Total_Raw']\n    df['CGAS_BMI'] = df['CGAS-CGAS_Score'] * df['Physical-BMI']\n    df['CGAS_FitnessMaxStage'] = df['CGAS-CGAS_Score'] * df['Fitness_Endurance-Max_Stage']\n    \n    return df\n\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n","metadata":{"papermill":{"duration":0.361068,"end_time":"2024-10-26T13:25:09.986124","exception":false,"start_time":"2024-10-26T13:25:09.625056","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:20:23.645039Z","iopub.execute_input":"2024-11-03T12:20:23.645579Z","iopub.status.idle":"2024-11-03T12:20:23.867617Z","shell.execute_reply.started":"2024-11-03T12:20:23.645534Z","shell.execute_reply":"2024-11-03T12:20:23.866728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ts_encoded = train_ts\ntest_ts_encoded = test_ts\n\ntime_series_cols = train_ts_encoded.columns.tolist()\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded['id']=test_ts[\"id\"]\n\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\n","metadata":{"papermill":{"duration":0.091537,"end_time":"2024-10-26T13:25:10.424902","exception":false,"start_time":"2024-10-26T13:25:10.333365","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:20:23.870963Z","iopub.execute_input":"2024-11-03T12:20:23.871371Z","iopub.status.idle":"2024-11-03T12:20:23.886840Z","shell.execute_reply.started":"2024-11-03T12:20:23.871335Z","shell.execute_reply":"2024-11-03T12:20:23.886000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imputer = KNNImputer(n_neighbors=7)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n\ntrain = train_imputed\n\ntrain = feature_engineering(train)\ntrain = train.dropna(thresh=10, axis=0)\ntest = feature_engineering(test)\n","metadata":{"papermill":{"duration":17.438246,"end_time":"2024-10-26T13:25:28.220269","exception":false,"start_time":"2024-10-26T13:25:10.782023","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:20:23.889813Z","iopub.execute_input":"2024-11-03T12:20:23.890358Z","iopub.status.idle":"2024-11-03T12:20:37.535080Z","shell.execute_reply.started":"2024-11-03T12:20:23.890324Z","shell.execute_reply":"2024-11-03T12:20:37.534305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=train\n\n\nage_bins = [0, 18, 25, 35, 45, 55, 65, 100]\nage_labels = ['<18', '18-25', '26-35', '36-45', '46-55', '56-65', '65+']\n\ndf['Physical_Health_Index'] = (df['Physical-BMI'] + df['Physical-Waist_Circumference'] +\n                                df['Physical-Diastolic_BP'] + df['Physical-Systolic_BP']) / 4\n\n\ndf['Fitness_Endurance_Score'] = df['Fitness_Endurance-Max_Stage'] + (df['Fitness_Endurance-Time_Mins'] * 60 + df['Fitness_Endurance-Time_Sec'])\n\n\ndf['Height_Norm'] = (df['Physical-Height'] - df['Physical-Height'].mean()) / df['Physical-Height'].std()\ndf['Weight_Norm'] = (df['Physical-Weight'] - df['Physical-Weight'].mean()) / df['Physical-Weight'].std()\n\n\ndf['Fat_to_Lean_Mass_Ratio'] = df['BIA-BIA_Fat'] / df['BIA-BIA_FFM']\n\n\ndf['Computer_Engagement'] = df['PreInt_EduHx-computerinternet_hoursday'] * 7  # Weekly hours\n\n\ndf['Cumulative_PAQ'] = df['PAQ_A-PAQ_A_Total'] + df['PAQ_C-PAQ_C_Total']\n\n\ndf['FGC_Percentage'] = df[['FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', \n                             'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', \n                             'FGC-FGC_GSD_Zone']].sum(axis=1) / len(df.columns)\n\n\ndf['Missing_Health_Data'] = df[['Physical-BMI', 'Physical-Height', 'Physical-Weight']].isnull().any(axis=1).astype(int)\n\ntrain=df\n\n\ndf=test\n\nimport pandas as pd\nimport numpy as np\n\n\nage_bins = [0, 18, 25, 35, 45, 55, 65, 100]\nage_labels = ['<18', '18-25', '26-35', '36-45', '46-55', '56-65', '65+']\n\ndf['Physical_Health_Index'] = (df['Physical-BMI'] + df['Physical-Waist_Circumference'] +\n                                df['Physical-Diastolic_BP'] + df['Physical-Systolic_BP']) / 4\n\n\ndf['Fitness_Endurance_Score'] = df['Fitness_Endurance-Max_Stage'] + (df['Fitness_Endurance-Time_Mins'] * 60 + df['Fitness_Endurance-Time_Sec'])\n\n\ndf['Height_Norm'] = (df['Physical-Height'] - df['Physical-Height'].mean()) / df['Physical-Height'].std()\ndf['Weight_Norm'] = (df['Physical-Weight'] - df['Physical-Weight'].mean()) / df['Physical-Weight'].std()\n\n\ndf['Fat_to_Lean_Mass_Ratio'] = df['BIA-BIA_Fat'] / df['BIA-BIA_FFM']\n\n\ndf['Computer_Engagement'] = df['PreInt_EduHx-computerinternet_hoursday'] * 7  # Weekly hours\n\n\ndf['Cumulative_PAQ'] = df['PAQ_A-PAQ_A_Total'] + df['PAQ_C-PAQ_C_Total']\n\n\ndf['FGC_Percentage'] = df[['FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', \n                             'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', \n                             'FGC-FGC_GSD_Zone']].sum(axis=1) / len(df.columns)\n\n\ndf['Missing_Health_Data'] = df[['Physical-BMI', 'Physical-Height', 'Physical-Weight']].isnull().any(axis=1).astype(int)\n\ntest=df\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday','Missing_Health_Data','FGC_Percentage','Cumulative_PAQ',\n 'Fitness_Endurance_Score',\n 'Height_Norm',\n 'Weight_Norm',\n 'Computer_Engagement',\n 'Fat_to_Lean_Mass_Ratio',\n 'Physical_Health_Index', 'sii']\n\nfeaturesCols += time_series_cols\n","metadata":{"papermill":{"duration":0.12686,"end_time":"2024-10-26T13:25:28.414243","exception":false,"start_time":"2024-10-26T13:25:28.287383","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:20:37.536305Z","iopub.execute_input":"2024-11-03T12:20:37.536634Z","iopub.status.idle":"2024-11-03T12:20:37.578169Z","shell.execute_reply.started":"2024-11-03T12:20:37.536601Z","shell.execute_reply":"2024-11-03T12:20:37.577252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col=list(test.columns)\n","metadata":{"papermill":{"duration":0.085879,"end_time":"2024-10-26T13:25:29.020580","exception":false,"start_time":"2024-10-26T13:25:28.934701","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:20:37.579545Z","iopub.execute_input":"2024-11-03T12:20:37.579946Z","iopub.status.idle":"2024-11-03T12:20:37.584740Z","shell.execute_reply.started":"2024-11-03T12:20:37.579898Z","shell.execute_reply":"2024-11-03T12:20:37.583577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col.append('sii')\n","metadata":{"papermill":{"duration":0.079524,"end_time":"2024-10-26T13:25:29.168354","exception":false,"start_time":"2024-10-26T13:25:29.088830","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:20:37.585898Z","iopub.execute_input":"2024-11-03T12:20:37.586432Z","iopub.status.idle":"2024-11-03T12:20:37.594027Z","shell.execute_reply.started":"2024-11-03T12:20:37.586386Z","shell.execute_reply":"2024-11-03T12:20:37.593275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train[col]\n","metadata":{"papermill":{"duration":0.081472,"end_time":"2024-10-26T13:25:29.502461","exception":false,"start_time":"2024-10-26T13:25:29.420989","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:20:37.595309Z","iopub.execute_input":"2024-11-03T12:20:37.595597Z","iopub.status.idle":"2024-11-03T12:20:37.606059Z","shell.execute_reply.started":"2024-11-03T12:20:37.595566Z","shell.execute_reply":"2024-11-03T12:20:37.605241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain = train.dropna(subset='sii')\n","metadata":{"papermill":{"duration":0.091303,"end_time":"2024-10-26T13:25:30.021309","exception":false,"start_time":"2024-10-26T13:25:29.930006","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:20:37.607266Z","iopub.execute_input":"2024-11-03T12:20:37.607555Z","iopub.status.idle":"2024-11-03T12:20:37.619339Z","shell.execute_reply.started":"2024-11-03T12:20:37.607524Z","shell.execute_reply":"2024-11-03T12:20:37.618429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if np.any(np.isinf(train)):\n    train = train.replace([np.inf, -np.inf], np.nan)\n","metadata":{"papermill":{"duration":0.092515,"end_time":"2024-10-26T13:25:30.185567","exception":false,"start_time":"2024-10-26T13:25:30.093052","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:20:37.620571Z","iopub.execute_input":"2024-11-03T12:20:37.621600Z","iopub.status.idle":"2024-11-03T12:20:37.632177Z","shell.execute_reply.started":"2024-11-03T12:20:37.621555Z","shell.execute_reply":"2024-11-03T12:20:37.631321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        clear_output(wait=True)\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission\n","metadata":{"papermill":{"duration":0.096557,"end_time":"2024-10-26T13:25:30.354310","exception":false,"start_time":"2024-10-26T13:25:30.257753","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:20:37.633560Z","iopub.execute_input":"2024-11-03T12:20:37.633855Z","iopub.status.idle":"2024-11-03T12:20:37.648342Z","shell.execute_reply.started":"2024-11-03T12:20:37.633823Z","shell.execute_reply":"2024-11-03T12:20:37.647480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10,  \n    'lambda_l2': 0.01  \n}\n\n\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  \n    'reg_lambda': 5,  \n    'random_state': SEED,\n    'tree_method': 'exact'\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10  \n}\n\n\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model)\n])\n","metadata":{"papermill":{"duration":0.087234,"end_time":"2024-10-26T13:25:30.513166","exception":false,"start_time":"2024-10-26T13:25:30.425932","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:20:37.649566Z","iopub.execute_input":"2024-11-03T12:20:37.649861Z","iopub.status.idle":"2024-11-03T12:20:37.661160Z","shell.execute_reply.started":"2024-11-03T12:20:37.649828Z","shell.execute_reply":"2024-11-03T12:20:37.660254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Submission1 = TrainML(voting_model, test)\nSubmission1.to_csv('submission.csv', index=False)\n","metadata":{"papermill":{"duration":68.063415,"end_time":"2024-10-26T13:26:38.648805","exception":false,"start_time":"2024-10-26T13:25:30.585390","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:20:37.662270Z","iopub.execute_input":"2024-11-03T12:20:37.662567Z","iopub.status.idle":"2024-11-03T12:21:30.580956Z","shell.execute_reply.started":"2024-11-03T12:20:37.662536Z","shell.execute_reply":"2024-11-03T12:21:30.579998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub1 = Submission1\nsub2 = Submission2\nsub3 = Submission3\n\nsub1 = sub1.sort_values(by='id').reset_index(drop=True)\nsub2 = sub2.sort_values(by='id').reset_index(drop=True)\nsub3 = sub3.sort_values(by='id').reset_index(drop=True)\n\ncombined = pd.DataFrame({\n    'id': sub1['id'],\n    'sii_1': sub1['sii'],\n    'sii_2': sub2['sii'],\n    'sii_3': sub3['sii']\n})\n\ndef majority_vote(row):\n    return row.mode()[0]\n\ncombined['final_sii'] = combined[['sii_1', 'sii_2', 'sii_3']].apply(majority_vote, axis=1)\n\nfinal_submission = combined[['id', 'final_sii']].rename(columns={'final_sii': 'sii'})\n\nfinal_submission.to_csv('submission.csv', index=False)\n","metadata":{"papermill":{"duration":0.127591,"end_time":"2024-10-26T13:26:39.294655","exception":false,"start_time":"2024-10-26T13:26:39.167064","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2024-11-03T12:21:30.582257Z","iopub.execute_input":"2024-11-03T12:21:30.582588Z","iopub.status.idle":"2024-11-03T12:21:30.601011Z","shell.execute_reply.started":"2024-11-03T12:21:30.582555Z","shell.execute_reply":"2024-11-03T12:21:30.600243Z"},"trusted":true},"execution_count":null,"outputs":[]}]}