{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30776,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd\nfrom sklearn.base import clone\nfrom copy import deepcopy\nimport optuna\nfrom scipy.optimize import minimize\nimport os\n\nimport re\nfrom colorama import Fore, Style\n\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nimport optuna\n\nimport lightgbm as lgb\nfrom catboost import CatBoostRegressor, CatBoostClassifier\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\n\nSEED = 42\nn_splits = 5","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:08:04.145279Z","iopub.execute_input":"2024-10-03T17:08:04.145597Z","iopub.status.idle":"2024-10-03T17:08:11.488516Z","shell.execute_reply.started":"2024-10-03T17:08:04.145562Z","shell.execute_reply":"2024-10-03T17:08:11.487570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n\n    stats, indexes = zip(*results)\n\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n\n    return df\n\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id',axis=1)\ntest = test.drop('id',axis=1)\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n       'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n       'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n       'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n       'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n       'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n       'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n       'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n       'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n       'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n       'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n       'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n       'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n       'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n       'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n       'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n       'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n       'PreInt_EduHx-computerinternet_hoursday','sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season','Fitness_Endurance-Season','FGC-Season',\n 'BIA-Season','PAQ_A-Season','PAQ_C-Season','SDS-Season','PreInt_EduHx-Season']\n\ndef update(df):\n\n    global cat_c\n    for c in cat_c : \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n\n    return df\n\ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n\n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:08:11.490700Z","iopub.execute_input":"2024-10-03T17:08:11.491382Z","iopub.status.idle":"2024-10-03T17:09:37.006950Z","shell.execute_reply.started":"2024-10-03T17:08:11.491337Z","shell.execute_reply":"2024-10-03T17:09:37.005920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"updated_col = ['Basic_Demos-Age', 'PreInt_EduHx-computerinternet_hoursday',  'SDS-SDS_Total_Raw',\n              'Basic_Demos-Sex', 'SDS-SDS_Total_T', 'FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_SRL', \n               'Physical-HeartRate', 'Physical-Height', 'stat_48', 'BIA-BIA_FFMI', 'Physical-Systolic_BP',\n               'Physical-BMI', 'CGAS-CGAS_Score', 'Physical-Weight', \n               'BIA-BIA_DEE', 'Physical-Diastolic_BP', 'stat_28', 'BIA-BIA_LDM']","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:46:16.506069Z","iopub.execute_input":"2024-10-03T17:46:16.506732Z","iopub.status.idle":"2024-10-03T17:46:16.511791Z","shell.execute_reply.started":"2024-10-03T17:46:16.506692Z","shell.execute_reply":"2024-10-03T17:46:16.510864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n        return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n\n#     X = train.drop(['sii'], axis=1)\n    X = train[updated_col]\n    y = train['sii']\n    \n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n    train_S = []\n    test_S = []\n\n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n\n        test_preds[:, fold] = model.predict(test_data)\n\n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead') # Nelder-Mead | # Powell\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:54:57.712327Z","iopub.execute_input":"2024-10-03T17:54:57.713036Z","iopub.status.idle":"2024-10-03T17:54:57.728767Z","shell.execute_reply.started":"2024-10-03T17:54:57.712996Z","shell.execute_reply":"2024-10-03T17:54:57.727764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X = train.drop(['sii'], axis=1)\n# y = train['sii']","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:20:17.052805Z","iopub.execute_input":"2024-10-03T17:20:17.053644Z","iopub.status.idle":"2024-10-03T17:20:17.060879Z","shell.execute_reply.started":"2024-10-03T17:20:17.053597Z","shell.execute_reply":"2024-10-03T17:20:17.059830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Select parameters to find best value**","metadata":{}},{"cell_type":"code","source":"# def objective(trial,data=X,target=y):\n    \n#     train_x, test_x, train_y, test_y = train_test_split(data, target, test_size=0.2,random_state=42)\n#     param = {\n#         \n#     \n#         'objective': 'multiclass',\n#         'num_class': 4,\n#           'device' : 'gpu',\n#           'min_data_in_leaf':  trial.suggest_int('min_data_in_leaf', 10,100),\n#           'bagging_freq':  trial.suggest_int('bagging_freq', 1,10),\n#           'feature_fraction': trial.suggest_loguniform('feature_fraction', 1e-7,1),\n#           'n_estimators': trial.suggest_int('n_estimators', 1000,30000),\n#           'bagging_fraction': trial.suggest_float('bagging_fraction', 0,1),\n#           'lambda_l1': trial.suggest_loguniform('lambda_l1', 1e-7,10.0),\n#           'lambda_l2': trial.suggest_loguniform('lambda_l1', 1e-7,1),\n         \n        \n#         'reg_alpha': trial.suggest_loguniform('reg_alpha', 1e-3, 10.0),\n#         'reg_lambda': trial.suggest_loguniform('reg_lambda', 1e-3, 10.0),\n#         'colsample_bytree': trial.suggest_categorical('colsample_bytree', [0.3,0.4,0.5,0.6,0.7,0.8,0.9, 1.0]),\n#         'subsample': trial.suggest_categorical('subsample', [0.4,0.5,0.6,0.7,0.8,1.0]),\n#         'learning_rate': trial.suggest_categorical('learning_rate', [0.006,0.008,0.01,0.014,0.017,0.02]),\n#         'max_depth': trial.suggest_int('max_depth', 1 , 100),\n#         'num_leaves' : trial.suggest_int('num_leaves', 1, 1000),\n#         'min_child_samples': trial.suggest_int('min_child_samples', 1, 300),\n#         'cat_smooth' : trial.suggest_int('min_data_per_groups', 1, 100)\n#     }\n#     model = lgb.LGBMRegressor(**param, verbose=-1)  \n    \n#     model.fit(train_x,train_y,eval_set=[(test_x,test_y)])\n    \n#     preds = model.predict(test_x)\n#     vd_preds = np.argmax(preds,axis=1)\n\n#     test_y1 = test_y.round()\n#      print(\"Pred\", np.unique(preds), np.unique(test_y))\n    \n#     loss = quadratic_weighted_kappa(test_y, preds.round(1).astype(int))\n    \n#     return loss","metadata":{"execution":{"iopub.status.busy":"2024-09-28T12:09:22.517741Z","iopub.execute_input":"2024-09-28T12:09:22.518418Z","iopub.status.idle":"2024-09-28T12:09:22.528141Z","shell.execute_reply.started":"2024-09-28T12:09:22.518366Z","shell.execute_reply":"2024-09-28T12:09:22.527034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# study = optuna.create_study(direction='maximize')\n# study.optimize(objective, n_trials=50)\n# print('Number of finished trials:', len(study.trials))\n# print('Best trial:', study.best_trial.params)","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:09:37.048778Z","iopub.execute_input":"2024-10-03T17:09:37.049060Z","iopub.status.idle":"2024-10-03T17:09:37.056586Z","shell.execute_reply.started":"2024-10-03T17:09:37.049030Z","shell.execute_reply":"2024-10-03T17:09:37.055841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import shap","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:21:26.930136Z","iopub.execute_input":"2024-10-03T17:21:26.930776Z","iopub.status.idle":"2024-10-03T17:21:32.194563Z","shell.execute_reply.started":"2024-10-03T17:21:26.930736Z","shell.execute_reply":"2024-10-03T17:21:32.193741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_x, test_x, train_y, test_y = train_test_split(X, y, test_size=0.2,random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:20:21.221141Z","iopub.execute_input":"2024-10-03T17:20:21.221550Z","iopub.status.idle":"2024-10-03T17:20:21.232231Z","shell.execute_reply.started":"2024-10-03T17:20:21.221511Z","shell.execute_reply":"2024-10-03T17:20:21.231467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:20:09.317352Z","iopub.execute_input":"2024-10-03T17:20:09.317790Z","iopub.status.idle":"2024-10-03T17:20:09.348317Z","shell.execute_reply.started":"2024-10-03T17:20:09.317751Z","shell.execute_reply":"2024-10-03T17:20:09.347006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shap.summary_plot(shap_values, test_x, plot_type=\"bar\")\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"updated_test = test[updated_col]","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:51:15.424310Z","iopub.execute_input":"2024-10-03T17:51:15.424751Z","iopub.status.idle":"2024-10-03T17:51:15.430973Z","shell.execute_reply.started":"2024-10-03T17:51:15.424710Z","shell.execute_reply":"2024-10-03T17:51:15.429706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Params = {'learning_rate': 0.04603534510792164, 'max_depth': 12, 'num_leaves': 478, 'min_data_in_leaf': 13,\n#               'feature_fraction': 0.8935304204489449, 'bagging_fraction': 0.7840117449237969, 'bagging_freq': 4,\n#               'lambda_l1': 6.596560434072009, 'lambda_l2': 2.680080551210706e-06} \nParams = {'n_estimators': 10625, 'reg_alpha': 4.276247198601698, 'reg_lambda': 0.036903362030235545, 'colsample_bytree': 0.8,  #Best Parameters\n          'subsample': 0.8, 'learning_rate': 0.01, 'max_depth': 38, 'num_leaves': 450,\n          'min_child_samples': 177, 'min_data_per_groups': 96}\n\nLight = lgb.LGBMRegressor(**Params,random_state=SEED, verbose=-1)\nSubmission = TrainML(Light,updated_test)","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:55:03.780896Z","iopub.execute_input":"2024-10-03T17:55:03.781271Z","iopub.status.idle":"2024-10-03T17:55:29.045526Z","shell.execute_reply.started":"2024-10-03T17:55:03.781235Z","shell.execute_reply":"2024-10-03T17:55:29.044532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Submission.to_csv('submission.csv', index=False)\nprint(Submission['sii'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:56:50.721522Z","iopub.execute_input":"2024-10-03T17:56:50.722618Z","iopub.status.idle":"2024-10-03T17:56:50.733340Z","shell.execute_reply.started":"2024-10-03T17:56:50.722574Z","shell.execute_reply":"2024-10-03T17:56:50.732340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Light.fit(train_x,train_y,eval_set=[(test_x,test_y)])","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:24:24.726160Z","iopub.execute_input":"2024-10-03T17:24:24.726819Z","iopub.status.idle":"2024-10-03T17:24:43.786483Z","shell.execute_reply.started":"2024-10-03T17:24:24.726775Z","shell.execute_reply":"2024-10-03T17:24:43.785497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shap_values = shap.TreeExplainer(Light).shap_values(test_x)\n# shap_interaction_values = shap.TreeExplainer(Light).shap_interaction_values(test_x)","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:27:48.346206Z","iopub.execute_input":"2024-10-03T17:27:48.346624Z","iopub.status.idle":"2024-10-03T17:29:41.915924Z","shell.execute_reply.started":"2024-10-03T17:27:48.346586Z","shell.execute_reply":"2024-10-03T17:29:41.914865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shap.summary_plot(shap_values, test_x, plot_type=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:31:26.487394Z","iopub.execute_input":"2024-10-03T17:31:26.488307Z","iopub.status.idle":"2024-10-03T17:31:26.990211Z","shell.execute_reply.started":"2024-10-03T17:31:26.488268Z","shell.execute_reply":"2024-10-03T17:31:26.989159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:44:46.492693Z","iopub.execute_input":"2024-10-03T17:44:46.493557Z","iopub.status.idle":"2024-10-03T17:44:46.499407Z","shell.execute_reply.started":"2024-10-03T17:44:46.493514Z","shell.execute_reply":"2024-10-03T17:44:46.498202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:44:48.405697Z","iopub.execute_input":"2024-10-03T17:44:48.406080Z","iopub.status.idle":"2024-10-03T17:44:48.412010Z","shell.execute_reply.started":"2024-10-03T17:44:48.406043Z","shell.execute_reply":"2024-10-03T17:44:48.410968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# updated_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-03T17:44:59.260795Z","iopub.execute_input":"2024-10-03T17:44:59.261447Z","iopub.status.idle":"2024-10-03T17:44:59.294217Z","shell.execute_reply.started":"2024-10-03T17:44:59.261392Z","shell.execute_reply":"2024-10-03T17:44:59.293310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Notebook copied from VYACHESLAV BOLOTIN\n# Try Optuna to find best parameters. \n\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Shaply Value","metadata":{}}]}