{"cells":[{"cell_type":"code","execution_count":null,"id":"c610ce89","metadata":{"execution":{"iopub.execute_input":"2024-11-14T13:53:36.185492Z","iopub.status.busy":"2024-11-14T13:53:36.185081Z","iopub.status.idle":"2024-11-14T13:53:39.091997Z","shell.execute_reply":"2024-11-14T13:53:39.090648Z"},"papermill":{"duration":2.917103,"end_time":"2024-11-14T13:53:39.095212","exception":false,"start_time":"2024-11-14T13:53:36.178109","status":"completed"},"tags":[]},"outputs":[],"source":"import pandas as pdimport numpy as npimport osimport pyarrow.parquet as pqimport matplotlib.pyplot as pltfrom sklearn.impute import SimpleImputerfrom sklearn.preprocessing import StandardScalerfrom sklearn.decomposition import PCAfrom sklearn.cluster import KMeansfrom sklearn.ensemble import RandomForestClassifierfrom sklearn.model_selection import train_test_split, cross_val_scorefrom sklearn.metrics import accuracy_score, classification_report# Load datatrain_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')test_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')# Check the first few rowsprint(train_data.head())print(test_data.head())"},{"cell_type":"code","execution_count":null,"id":"5c8c76c0","metadata":{"execution":{"iopub.execute_input":"2024-11-14T13:53:39.107078Z","iopub.status.busy":"2024-11-14T13:53:39.106485Z","iopub.status.idle":"2024-11-14T13:53:39.120696Z","shell.execute_reply":"2024-11-14T13:53:39.119320Z"},"papermill":{"duration":0.023263,"end_time":"2024-11-14T13:53:39.123554","exception":false,"start_time":"2024-11-14T13:53:39.100291","status":"completed"},"tags":[]},"outputs":[],"source":"# Check for missing valuesmissing_values_train = train_data.isnull().sum() / len(train_data)# print(missing_values_train)print(missing_values_train)"},{"cell_type":"markdown","id":"c2b74de6","metadata":{"papermill":{"duration":0.004831,"end_time":"2024-11-14T13:53:39.133689","exception":false,"start_time":"2024-11-14T13:53:39.128858","status":"completed"},"tags":[]},"source":"#### Ref > [Ensemble Model](https://www.kaggle.com/code/abdullah0a/ensamble-models) "},{"cell_type":"code","execution_count":null,"id":"7d9e6ed7","metadata":{"execution":{"iopub.execute_input":"2024-11-14T13:53:39.145740Z","iopub.status.busy":"2024-11-14T13:53:39.144730Z","iopub.status.idle":"2024-11-14T13:53:59.058830Z","shell.execute_reply":"2024-11-14T13:53:59.057379Z"},"papermill":{"duration":19.922892,"end_time":"2024-11-14T13:53:59.061435","exception":false,"start_time":"2024-11-14T13:53:39.138543","status":"completed"},"tags":[]},"outputs":[],"source":"import numpy as npimport pandas as pdimport osimport numpy as npimport pandas as pdimport osimport refrom sklearn.base import clonefrom sklearn.metrics import cohen_kappa_scorefrom sklearn.model_selection import StratifiedKFoldfrom scipy.optimize import minimizefrom concurrent.futures import ThreadPoolExecutorfrom tqdm import tqdmimport polars as plimport polars.selectors as csimport matplotlib.pyplot as pltfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatterimport seaborn as snsfrom sklearn.preprocessing import StandardScalerimport matplotlib.pyplot as pltfrom keras.models import Modelfrom keras.layers import Input, Densefrom keras.optimizers import Adamimport torchimport torch.nn as nnimport torch.optim as optimfrom colorama import Fore, Stylefrom IPython.display import clear_outputimport warningsfrom lightgbm import LGBMRegressorfrom xgboost import XGBRegressorfrom catboost import CatBoostRegressorfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressorfrom sklearn.impute import SimpleImputer, KNNImputerfrom sklearn.pipeline import Pipelinewarnings.filterwarnings('ignore')pd.options.display.max_columns = NoneSEED = 42n_splits = 5"},{"cell_type":"code","execution_count":null,"id":"1a9d7a75","metadata":{"execution":{"iopub.execute_input":"2024-11-14T13:53:59.073213Z","iopub.status.busy":"2024-11-14T13:53:59.072480Z","iopub.status.idle":"2024-11-14T13:55:34.488688Z","shell.execute_reply":"2024-11-14T13:55:34.487509Z"},"papermill":{"duration":95.425042,"end_time":"2024-11-14T13:55:34.491380","exception":false,"start_time":"2024-11-14T13:53:59.066338","status":"completed"},"tags":[]},"outputs":[],"source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')def process_file(filename, dirname):    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))    df.drop('step', axis=1, inplace=True)    return df.describe().values.reshape(-1), filename.split('=')[1]def load_time_series(dirname) -> pd.DataFrame:    ids = os.listdir(dirname)        with ThreadPoolExecutor() as executor:        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))        stats, indexes = zip(*results)        df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])    df['id'] = indexes    return df        train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")test_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")time_series_cols = train_ts.columns.tolist()time_series_cols.remove(\"id\")train = pd.merge(train, train_ts, how=\"left\", on='id')test = pd.merge(test, test_ts, how=\"left\", on='id')train = train.drop('id', axis=1)test = test.drop('id', axis=1)  "},{"cell_type":"code","execution_count":null,"id":"9333ff01","metadata":{"execution":{"iopub.execute_input":"2024-11-14T13:55:34.549024Z","iopub.status.busy":"2024-11-14T13:55:34.548565Z","iopub.status.idle":"2024-11-14T13:55:34.565389Z","shell.execute_reply":"2024-11-14T13:55:34.564187Z"},"papermill":{"duration":0.048103,"end_time":"2024-11-14T13:55:34.567841","exception":false,"start_time":"2024-11-14T13:55:34.519738","status":"completed"},"tags":[]},"outputs":[],"source":"featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',                'PreInt_EduHx-computerinternet_hoursday', 'sii']print(len(featuresCols))featuresCols += time_series_colstrain = train[featuresCols]train = train.dropna(subset='sii')cat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season',           'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season',           'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']"},{"cell_type":"code","execution_count":null,"id":"2426de1b","metadata":{"execution":{"iopub.execute_input":"2024-11-14T13:55:34.624147Z","iopub.status.busy":"2024-11-14T13:55:34.623717Z","iopub.status.idle":"2024-11-14T13:55:34.698407Z","shell.execute_reply":"2024-11-14T13:55:34.697209Z"},"papermill":{"duration":0.105747,"end_time":"2024-11-14T13:55:34.700890","exception":false,"start_time":"2024-11-14T13:55:34.595143","status":"completed"},"tags":[]},"outputs":[],"source":"def update(df):    global cat_c    for c in cat_c:         df[c] = df[c].fillna('Missing')        df[c] = df[c].astype('category')    return df        train = update(train)test = update(test)def create_mapping(column, dataset):    unique_values = dataset[column].unique()    return {value: idx for idx, value in enumerate(unique_values)}for col in cat_c:    mapping = create_mapping(col, train)    mappingTe = create_mapping(col, test)        train[col] = train[col].replace(mapping).astype(int)    test[col] = test[col].replace(mappingTe).astype(int)"},{"cell_type":"code","execution_count":null,"id":"4f801086","metadata":{"execution":{"iopub.execute_input":"2024-11-14T13:55:34.758755Z","iopub.status.busy":"2024-11-14T13:55:34.757897Z","iopub.status.idle":"2024-11-14T13:55:34.764939Z","shell.execute_reply":"2024-11-14T13:55:34.763826Z"},"papermill":{"duration":0.038518,"end_time":"2024-11-14T13:55:34.767447","exception":false,"start_time":"2024-11-14T13:55:34.728929","status":"completed"},"tags":[]},"outputs":[],"source":"from IPython.display import display, HTMLdef display_scrollable_dataframe(df, height=300):    html = df.to_html()    scrollable_html = f\"\"\"    <div style=\"overflow-y: scroll; height: {height}px; border: 1px solid #ccc; padding: 10px;\">        {html}    </div>    \"\"\"    display(HTML(scrollable_html))"},{"cell_type":"code","execution_count":null,"id":"842010f7","metadata":{"execution":{"iopub.execute_input":"2024-11-14T13:55:34.824049Z","iopub.status.busy":"2024-11-14T13:55:34.823597Z","iopub.status.idle":"2024-11-14T13:55:34.854082Z","shell.execute_reply":"2024-11-14T13:55:34.852994Z"},"papermill":{"duration":0.06164,"end_time":"2024-11-14T13:55:34.856485","exception":false,"start_time":"2024-11-14T13:55:34.794845","status":"completed"},"tags":[]},"outputs":[],"source":"double_check = pd.DataFrame(train.isnull().sum())display_scrollable_dataframe(double_check.T, height=100)"},{"cell_type":"code","execution_count":null,"id":"a0800774","metadata":{"execution":{"iopub.execute_input":"2024-11-14T13:55:34.915382Z","iopub.status.busy":"2024-11-14T13:55:34.914969Z","iopub.status.idle":"2024-11-14T13:55:42.448169Z","shell.execute_reply":"2024-11-14T13:55:42.446838Z"},"papermill":{"duration":7.565686,"end_time":"2024-11-14T13:55:42.450914","exception":false,"start_time":"2024-11-14T13:55:34.885228","status":"completed"},"tags":[]},"outputs":[],"source":"imputer = KNNImputer(n_neighbors= 3)train_imputed = pd.DataFrame(imputer.fit_transform(train), columns=train.columns)#print(train_imputed)"},{"cell_type":"code","execution_count":null,"id":"bfb1f445","metadata":{"execution":{"iopub.execute_input":"2024-11-14T13:55:42.509746Z","iopub.status.busy":"2024-11-14T13:55:42.509318Z","iopub.status.idle":"2024-11-14T13:55:42.537401Z","shell.execute_reply":"2024-11-14T13:55:42.536320Z"},"papermill":{"duration":0.059955,"end_time":"2024-11-14T13:55:42.539980","exception":false,"start_time":"2024-11-14T13:55:42.480025","status":"completed"},"tags":[]},"outputs":[],"source":"double_check = pd.DataFrame(train_imputed.isnull().sum())display_scrollable_dataframe(double_check.T, height=100)"},{"cell_type":"code","execution_count":null,"id":"c6296fd6","metadata":{"execution":{"iopub.execute_input":"2024-11-14T13:55:42.601435Z","iopub.status.busy":"2024-11-14T13:55:42.600473Z","iopub.status.idle":"2024-11-14T13:55:42.608973Z","shell.execute_reply":"2024-11-14T13:55:42.607268Z"},"papermill":{"duration":0.04336,"end_time":"2024-11-14T13:55:42.611702","exception":false,"start_time":"2024-11-14T13:55:42.568342","status":"completed"},"tags":[]},"outputs":[],"source":"# Evaluate Acc Scoredef quadratic_weighted_kappa(y_true, y_pred):    return cohen_kappa_score(y_true, y_pred, weights='quadratic')def threshold_Rounder(oof_non_rounded, thresholds):    return np.where(oof_non_rounded < thresholds[0], 0,                    np.where(oof_non_rounded < thresholds[1], 1,                             np.where(oof_non_rounded < thresholds[2], 2, 3)))def evaluate_predictions(thresholds, y_true, oof_non_rounded):    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)    return -quadratic_weighted_kappa(y_true, rounded_p)"},{"cell_type":"code","execution_count":null,"id":"00562ace","metadata":{"execution":{"iopub.execute_input":"2024-11-14T13:55:42.670509Z","iopub.status.busy":"2024-11-14T13:55:42.670068Z","iopub.status.idle":"2024-11-14T13:55:42.684953Z","shell.execute_reply":"2024-11-14T13:55:42.683872Z"},"papermill":{"duration":0.046757,"end_time":"2024-11-14T13:55:42.687253","exception":false,"start_time":"2024-11-14T13:55:42.640496","status":"completed"},"tags":[]},"outputs":[],"source":"def TrainML(model_class, train, test_data):    X = train.drop(['sii'], axis=1)    y = train['sii']    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)        train_S = []    test_S = []        oof_non_rounded = np.zeros(len(y), dtype=float)     oof_rounded = np.zeros(len(y), dtype=int)     test_preds = np.zeros((len(test_data), n_splits))    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]        model = clone(model_class)        model.fit(X_train, y_train)        y_train_pred = model.predict(X_train)        y_val_pred = model.predict(X_val)        oof_non_rounded[test_idx] = y_val_pred        y_val_pred_rounded = y_val_pred.round(0).astype(int)        oof_rounded[test_idx] = y_val_pred_rounded        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)        train_S.append(train_kappa)        test_S.append(val_kappa)                test_preds[:, fold] = model.predict(test_data)                print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")        clear_output(wait=True)    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")    KappaOPtimizer = minimize(evaluate_predictions,                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded),                               method='Nelder-Mead')    assert KappaOPtimizer.success, \"Optimization did not converge.\"        oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)    tKappa = quadratic_weighted_kappa(y, oof_tuned)    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")    tpm = test_preds.mean(axis=1)    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)        submission = pd.DataFrame({        'id': sample['id'],        'sii': tpTuned    })    return submission"},{"cell_type":"code","execution_count":null,"id":"b874f1d1","metadata":{"execution":{"iopub.execute_input":"2024-11-14T13:55:42.761678Z","iopub.status.busy":"2024-11-14T13:55:42.760296Z","iopub.status.idle":"2024-11-14T13:55:42.767101Z","shell.execute_reply":"2024-11-14T13:55:42.765889Z"},"papermill":{"duration":0.049277,"end_time":"2024-11-14T13:55:42.770274","exception":false,"start_time":"2024-11-14T13:55:42.720997","status":"completed"},"tags":[]},"outputs":[],"source":"# XGBoost parametersXGB_Params = {    'learning_rate': 0.05,    'max_depth': 6,    'n_estimators': 200,    'subsample': 0.8,    'colsample_bytree': 0.8,    'reg_alpha': 2,  # Increased from 0.1    'reg_lambda': 10,  # Increased from 1    'random_state': SEED}"},{"cell_type":"code","execution_count":null,"id":"550784df","metadata":{"execution":{"iopub.execute_input":"2024-11-14T13:55:42.832706Z","iopub.status.busy":"2024-11-14T13:55:42.832291Z","iopub.status.idle":"2024-11-14T13:55:54.669916Z","shell.execute_reply":"2024-11-14T13:55:54.668624Z"},"papermill":{"duration":11.869877,"end_time":"2024-11-14T13:55:54.672549","exception":false,"start_time":"2024-11-14T13:55:42.802672","status":"completed"},"tags":[]},"outputs":[],"source":"# Create model instancesXGB_Model = XGBRegressor(**XGB_Params)# Train the ensemble modelsubmission = TrainML(XGB_Model, train_imputed, test)print(submission['sii'].value_counts())submission.to_csv('/kaggle/working/submission.csv', index=False)"}],"metadata":{"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"papermill":{"default_parameters":{},"duration":143.498849,"end_time":"2024-11-14T13:55:56.828817","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-11-14T13:53:33.329968","version":"2.6.0"}},"nbformat":4,"nbformat_minor":4}