{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd\nfrom sklearn.base import clone\nfrom copy import deepcopy\nimport optuna\nfrom scipy.optimize import minimize\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport re\nfrom colorama import Fore, Style\n\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nfrom lightgbm import LGBMRegressor\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\n\nwarnings.filterwarnings('ignore')\n\nSEED = 42\nn_splits = 5","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2024-12-17T15:00:53.665898Z","iopub.execute_input":"2024-12-17T15:00:53.666282Z","iopub.status.idle":"2024-12-17T15:00:53.672939Z","shell.execute_reply.started":"2024-12-17T15:00:53.666240Z","shell.execute_reply":"2024-12-17T15:00:53.671978Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"Stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df\n\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\nimputer = KNNImputer(n_neighbors=5)\nnumeric_cols = train.select_dtypes(include=['float64', 'int64']).columns\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\n\ntrain = train_imputed\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n          'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\n\"\"\"This Mapping Works Fine For me I also Check Each Values in Train and test Using Logic. There no Data Lekage.\"\"\"\n\nfor col in cat_c:\n    mapping_train = create_mapping(col, train)\n    mapping_test = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping_train).astype(int)\n    test[col] = test[col].replace(mapping_test).astype(int)\n\nprint(f'Train Shape : {train.shape} || Test Shape : {test.shape}')","metadata":{"execution":{"iopub.status.busy":"2024-12-17T15:00:53.674773Z","iopub.execute_input":"2024-12-17T15:00:53.675267Z","iopub.status.idle":"2024-12-17T15:02:11.062211Z","shell.execute_reply.started":"2024-12-17T15:00:53.675237Z","shell.execute_reply":"2024-12-17T15:02:11.061175Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-12-17T15:02:11.063652Z","iopub.execute_input":"2024-12-17T15:02:11.064266Z","iopub.status.idle":"2024-12-17T15:02:11.162215Z","shell.execute_reply.started":"2024-12-17T15:02:11.064220Z","shell.execute_reply":"2024-12-17T15:02:11.161278Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-12-17T15:02:11.164494Z","iopub.execute_input":"2024-12-17T15:02:11.164909Z","iopub.status.idle":"2024-12-17T15:02:11.261543Z","shell.execute_reply.started":"2024-12-17T15:02:11.164866Z","shell.execute_reply":"2024-12-17T15:02:11.260518Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:02:11.262695Z","iopub.execute_input":"2024-12-17T15:02:11.263005Z","iopub.status.idle":"2024-12-17T15:02:11.268371Z","shell.execute_reply.started":"2024-12-17T15:02:11.262976Z","shell.execute_reply":"2024-12-17T15:02:11.267486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def objective(trial):\n    # Define the hyperparameters to tune\n    params = {\n        'objective': 'reg:squarederror',  # Use 'reg:squarederror' for regression tasks\n        'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.1),\n        'max_depth': trial.suggest_int('max_depth', 3, 12),\n        'num_leaves': trial.suggest_int('num_leaves', 30, 1000),\n        'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 1, 50),\n        'feature_fraction': trial.suggest_uniform('feature_fraction', 0.6, 1.0),\n        'bagging_fraction': trial.suggest_uniform('bagging_fraction', 0.6, 1.0),\n        'bagging_freq': trial.suggest_int('bagging_freq', 1, 5),\n        'lambda_l1': trial.suggest_loguniform('lambda_l1', 1e-5, 10),\n        'lambda_l2': trial.suggest_loguniform('lambda_l2', 1e-5, 10),\n        'random_state': SEED,\n        'n_estimators': 1000  # Adjust the number of estimators or allow it to vary\n    }\n\n    # Prepare the data (this part is assumed to be defined outside this function)\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n    \n    n_splits = 5  # Define the number of splits for Stratified K-Folds (adjust as needed)\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n    oof_non_rounded = np.zeros(len(y), dtype=float)\n    val_kappas = []  # To store validation QWK for each fold\n    \n    model_class = XGBRegressor(**params)\n\n    # Cross-validation loop\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)  # Clone model to ensure fresh instance each fold\n        model.fit(X_train, y_train)\n\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n\n        # Calculate validation QWK (quadratic weighted kappa) for the fold\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n        val_kappas.append(val_kappa)\n\n    # Calculate the average validation QWK across all folds (mean of the list)\n    mean_val_kappa = np.mean(val_kappas)\n\n    print(f\"Mean Validation QWK: {mean_val_kappa:.4f}\")\n    \n    # Return the negative validation QWK for optimization (since Optuna minimizes the objective)\n    return -mean_val_kappa","metadata":{"execution":{"iopub.status.busy":"2024-12-17T15:02:11.269861Z","iopub.execute_input":"2024-12-17T15:02:11.270155Z","iopub.status.idle":"2024-12-17T15:02:11.283352Z","shell.execute_reply.started":"2024-12-17T15:02:11.270128Z","shell.execute_reply":"2024-12-17T15:02:11.282560Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Optuna Study setup (only do one time)\nparam = {'learning_rate': 0.015344666145156564,\n 'max_depth': 5,\n 'num_leaves': 565,\n 'min_data_in_leaf': 19,\n 'feature_fraction': 0.8634552281905575,\n 'bagging_fraction': 0.6560133574841832,\n 'bagging_freq': 2,\n 'lambda_l1': 1.7007561360254027,\n 'lambda_l2': 0.00031751742017441855}\n\nif param is None:\n    study = optuna.create_study(direction='minimize') \n    study.optimize(objective, n_trials=50)  # Number of trials to perform\n    print(\"Number of finished trials: \", len(study.trials))\n    print(\"Best trial:\")\n    trial = study.best_trial\n    print(\"  Value: {}\".format(trial.value))\n    print(\"  Params: \")\n    param = {key: value for key, value in trial.params.items()}\nparam","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:02:11.284436Z","iopub.execute_input":"2024-12-17T15:02:11.284794Z","iopub.status.idle":"2024-12-17T15:02:11.300866Z","shell.execute_reply.started":"2024-12-17T15:02:11.284721Z","shell.execute_reply":"2024-12-17T15:02:11.299865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\ndef TrainML(model_class, test_data):\n    \n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead') # Nelder-Mead | # Powell\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission,model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:02:11.302011Z","iopub.execute_input":"2024-12-17T15:02:11.302335Z","iopub.status.idle":"2024-12-17T15:02:11.319021Z","shell.execute_reply.started":"2024-12-17T15:02:11.302298Z","shell.execute_reply":"2024-12-17T15:02:11.318119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Params = {\n    'learning_rate': 0.046,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 10, \n    'lambda_l2': 0.01, \n    'device': 'gpu'\n\n}\n\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,\n    'reg_lambda': 5, \n    'random_state': SEED,\n    'tree_method': 'gpu_hist',\n\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,\n    'task_type': 'GPU'\n\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:02:11.321554Z","iopub.execute_input":"2024-12-17T15:02:11.321884Z","iopub.status.idle":"2024-12-17T15:02:11.336665Z","shell.execute_reply.started":"2024-12-17T15:02:11.321849Z","shell.execute_reply":"2024-12-17T15:02:11.335850Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb = XGBRegressor(**XGB_Params)\nlight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:02:11.337867Z","iopub.execute_input":"2024-12-17T15:02:11.338183Z","iopub.status.idle":"2024-12-17T15:02:11.348131Z","shell.execute_reply.started":"2024-12-17T15:02:11.338149Z","shell.execute_reply":"2024-12-17T15:02:11.347298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nmodel = VotingRegressor(estimators=[\n    ('lightgbm', light),\n    ('xgboost', xgb)\n])\n\nSubmission,model = TrainML(model,test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:02:11.349146Z","iopub.execute_input":"2024-12-17T15:02:11.349452Z","iopub.status.idle":"2024-12-17T15:02:28.960001Z","shell.execute_reply.started":"2024-12-17T15:02:11.349427Z","shell.execute_reply":"2024-12-17T15:02:28.959051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\nSubmission.to_csv('submission.csv', index=False)\nprint(Submission['sii'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T15:02:28.961124Z","iopub.execute_input":"2024-12-17T15:02:28.961421Z","iopub.status.idle":"2024-12-17T15:02:28.969062Z","shell.execute_reply.started":"2024-12-17T15:02:28.961392Z","shell.execute_reply":"2024-12-17T15:02:28.968048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}