{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.model_selection import train_test_split, GroupKFold\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\nimport lightgbm as lgb\nimport xgboost as xgb\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom scipy.optimize import minimize\nfrom tqdm import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-05T18:56:52.620830Z","iopub.execute_input":"2024-10-05T18:56:52.621403Z","iopub.status.idle":"2024-10-05T18:56:52.841166Z","shell.execute_reply.started":"2024-10-05T18:56:52.621352Z","shell.execute_reply":"2024-10-05T18:56:52.839750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:56:54.716348Z","iopub.execute_input":"2024-10-05T18:56:54.716847Z","iopub.status.idle":"2024-10-05T18:56:54.779978Z","shell.execute_reply.started":"2024-10-05T18:56:54.716803Z","shell.execute_reply":"2024-10-05T18:56:54.778786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:56:57.338079Z","iopub.execute_input":"2024-10-05T18:56:57.338492Z","iopub.status.idle":"2024-10-05T18:56:57.348235Z","shell.execute_reply.started":"2024-10-05T18:56:57.338452Z","shell.execute_reply":"2024-10-05T18:56:57.346837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGET_COLS = [\n    \"PCIAT-PCIAT_01\",\n    \"PCIAT-PCIAT_02\",\n    \"PCIAT-PCIAT_03\",\n    \"PCIAT-PCIAT_04\",\n    \"PCIAT-PCIAT_05\",\n    \"PCIAT-PCIAT_06\",\n    \"PCIAT-PCIAT_07\",\n    \"PCIAT-PCIAT_08\",\n    \"PCIAT-PCIAT_09\",\n    \"PCIAT-PCIAT_10\",\n    \"PCIAT-PCIAT_11\",\n    \"PCIAT-PCIAT_12\",\n    \"PCIAT-PCIAT_13\",\n    \"PCIAT-PCIAT_14\",\n    \"PCIAT-PCIAT_15\",\n    \"PCIAT-PCIAT_16\",    \n    \"PCIAT-PCIAT_17\",\n    \"PCIAT-PCIAT_18\",\n    \"PCIAT-PCIAT_19\",\n    \"PCIAT-PCIAT_20\",\n    \"PCIAT-PCIAT_Total\",\n    \"sii\",\n]","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:57:00.466623Z","iopub.execute_input":"2024-10-05T18:57:00.467168Z","iopub.status.idle":"2024-10-05T18:57:00.478383Z","shell.execute_reply.started":"2024-10-05T18:57:00.467116Z","shell.execute_reply":"2024-10-05T18:57:00.476874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for nulls in target columns\nnull_counts = train[TARGET_COLS].isnull().sum()\nnull_counts","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:57:03.310386Z","iopub.execute_input":"2024-10-05T18:57:03.310898Z","iopub.status.idle":"2024-10-05T18:57:03.324884Z","shell.execute_reply.started":"2024-10-05T18:57:03.310839Z","shell.execute_reply":"2024-10-05T18:57:03.323469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.dropna(subset=TARGET_COLS)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:57:07.406407Z","iopub.execute_input":"2024-10-05T18:57:07.406895Z","iopub.status.idle":"2024-10-05T18:57:07.417835Z","shell.execute_reply.started":"2024-10-05T18:57:07.406852Z","shell.execute_reply":"2024-10-05T18:57:07.416267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:57:09.662405Z","iopub.execute_input":"2024-10-05T18:57:09.662870Z","iopub.status.idle":"2024-10-05T18:57:09.671772Z","shell.execute_reply.started":"2024-10-05T18:57:09.662827Z","shell.execute_reply":"2024-10-05T18:57:09.670339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_data(df):\n    # Handle numerical columns\n    num_cols = df.select_dtypes(include=np.number).columns\n    df[num_cols] = df[num_cols].fillna(df[num_cols].median())\n    \n    # Handle categorical columns\n    cat_cols = df.select_dtypes(include='object').columns\n    for col in cat_cols:\n        df[col] = df[col].fillna(df[col].mode()[0])  # Fill missing with the most frequent value\n        df[col] = LabelEncoder().fit_transform(df[col].astype(str))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:57:12.526476Z","iopub.execute_input":"2024-10-05T18:57:12.526946Z","iopub.status.idle":"2024-10-05T18:57:12.535960Z","shell.execute_reply.started":"2024-10-05T18:57:12.526902Z","shell.execute_reply":"2024-10-05T18:57:12.534329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract the target and drop unnecessary columns\ntarget = train['PCIAT-PCIAT_Total']\ntrain = train.drop(columns=TARGET_COLS)\ntrain = train.drop(columns=['id'])\ntrain = train.drop(columns=['PCIAT-Season'])","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:57:16.016179Z","iopub.execute_input":"2024-10-05T18:57:16.016658Z","iopub.status.idle":"2024-10-05T18:57:16.029479Z","shell.execute_reply.started":"2024-10-05T18:57:16.016592Z","shell.execute_reply":"2024-10-05T18:57:16.028216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ids = test['id']\ntest = test.drop(columns=['id'])","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:57:19.161078Z","iopub.execute_input":"2024-10-05T18:57:19.161619Z","iopub.status.idle":"2024-10-05T18:57:19.169839Z","shell.execute_reply.started":"2024-10-05T18:57:19.161573Z","shell.execute_reply":"2024-10-05T18:57:19.168307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Columns with 70% or more NaNs:\ncolumns_to_drop = ['Physical-Waist_Circumference', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total']","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:57:21.692804Z","iopub.execute_input":"2024-10-05T18:57:21.693286Z","iopub.status.idle":"2024-10-05T18:57:21.700059Z","shell.execute_reply.started":"2024-10-05T18:57:21.693241Z","shell.execute_reply":"2024-10-05T18:57:21.698346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.drop(columns=columns_to_drop)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:57:23.801191Z","iopub.execute_input":"2024-10-05T18:57:23.801692Z","iopub.status.idle":"2024-10-05T18:57:23.810517Z","shell.execute_reply.started":"2024-10-05T18:57:23.801614Z","shell.execute_reply":"2024-10-05T18:57:23.809115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = test.drop(columns=columns_to_drop)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:57:26.012744Z","iopub.execute_input":"2024-10-05T18:57:26.013210Z","iopub.status.idle":"2024-10-05T18:57:26.020824Z","shell.execute_reply.started":"2024-10-05T18:57:26.013157Z","shell.execute_reply":"2024-10-05T18:57:26.019319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:57:29.774669Z","iopub.execute_input":"2024-10-05T18:57:29.775164Z","iopub.status.idle":"2024-10-05T18:57:29.786031Z","shell.execute_reply.started":"2024-10-05T18:57:29.775120Z","shell.execute_reply":"2024-10-05T18:57:29.784658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:57:31.966671Z","iopub.execute_input":"2024-10-05T18:57:31.967115Z","iopub.status.idle":"2024-10-05T18:57:31.976700Z","shell.execute_reply.started":"2024-10-05T18:57:31.967075Z","shell.execute_reply":"2024-10-05T18:57:31.975166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selection = ['Physical-Height',\n 'Basic_Demos-Age',\n 'PreInt_EduHx-computerinternet_hoursday',\n 'Physical-Weight',\n 'FGC-FGC_CU',\n 'BIA-BIA_BMI',\n 'SDS-SDS_Total_T',\n #'PAQ_A-Season',\n 'FGC-FGC_PU',\n 'BIA-BIA_Frame_num',\n 'Physical-Systolic_BP',\n 'FGC-FGC_TL',\n 'PAQ_C-Season',\n 'BIA-BIA_FFMI',\n 'FGC-FGC_SRR_Zone',\n 'FGC-FGC_SRL_Zone']","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:39:13.801870Z","iopub.execute_input":"2024-10-05T18:39:13.802426Z","iopub.status.idle":"2024-10-05T18:39:13.809253Z","shell.execute_reply.started":"2024-10-05T18:39:13.802378Z","shell.execute_reply":"2024-10-05T18:39:13.807837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train[selection]\ntest = test[selection]","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:39:17.538895Z","iopub.execute_input":"2024-10-05T18:39:17.539563Z","iopub.status.idle":"2024-10-05T18:39:17.551804Z","shell.execute_reply.started":"2024-10-05T18:39:17.539493Z","shell.execute_reply":"2024-10-05T18:39:17.549142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = preprocess_data(X)\nX_test = preprocess_data(test)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:39:19.631953Z","iopub.execute_input":"2024-10-05T18:39:19.632569Z","iopub.status.idle":"2024-10-05T18:39:19.672137Z","shell.execute_reply.started":"2024-10-05T18:39:19.632526Z","shell.execute_reply":"2024-10-05T18:39:19.670911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert(scores):\n    scores = np.array(scores)*1.3\n    bins = np.zeros_like(scores)\n    bins[scores <= 30] = 0\n    bins[(scores > 30) & (scores < 50)] = 1\n    bins[(scores >= 50) & (scores < 80)] = 2\n    bins[scores >= 80] = 3\n    return bins","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:44:03.094970Z","iopub.execute_input":"2024-10-05T18:44:03.095503Z","iopub.status.idle":"2024-10-05T18:44:03.103907Z","shell.execute_reply.started":"2024-10-05T18:44:03.095450Z","shell.execute_reply":"2024-10-05T18:44:03.102450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    y_true_cat = convert(y_true)\n    y_pred_cat = convert(y_pred)\n    return cohen_kappa_score(y_true_cat, y_pred_cat, weights='quadratic')","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:45:14.855041Z","iopub.execute_input":"2024-10-05T18:45:14.855608Z","iopub.status.idle":"2024-10-05T18:45:14.862805Z","shell.execute_reply.started":"2024-10-05T18:45:14.855558Z","shell.execute_reply":"2024-10-05T18:45:14.861378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_folds(model_class, X, y, test_data,n_splits=5, params=None):\n    skf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n    \n    oof_non_rounded = np.zeros(len(y))\n    test_preds = np.zeros((len(test_data), n_splits))\n    val_kappas = []  # Store QWK for each fold\n    \n    for fold, (train_idx, val_idx) in enumerate(tqdm(skf.split(X, y), total=n_splits, desc=\"Training Folds\")):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n        \n        model = model_class(**params) if params else model_class()\n        model.fit(X_train, y_train)\n        \n        # Predict validation\n        y_val_pred = model.predict(X_val)\n        oof_non_rounded[val_idx] = y_val_pred\n        \n        # Round validation predictions\n#         y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        \n        # Compute QWK for validation data\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred)\n        val_kappas.append(val_kappa)\n        print(f\"Fold {fold+1} - Validation QWK: {val_kappa:.4f}\")\n        \n        # Predict test\n        test_preds[:, fold] = model.predict(test_data)\n    \n    # Average test predictions across folds\n    test_preds_mean = test_preds.mean(axis=1)\n    \n    # Print mean QWK score across all folds\n    mean_kappa = np.mean(val_kappas)\n    print(f\"Mean Validation QWK across folds: {mean_kappa:.4f}\")\n    \n    return test_preds_mean\n","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:45:17.726387Z","iopub.execute_input":"2024-10-05T18:45:17.726905Z","iopub.status.idle":"2024-10-05T18:45:17.740058Z","shell.execute_reply.started":"2024-10-05T18:45:17.726862Z","shell.execute_reply":"2024-10-05T18:45:17.738830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# params = {'learning_rate': 0.09056475084257094, 'max_depth': 5, 'num_leaves': 429, 'min_data_in_leaf': 43, \n#           'feature_fraction': 0.8916815865803562, 'bagging_fraction': 0.8276271977726875, 'bagging_freq': 6, \n#           'lambda_l1': 0.07688998999378223, 'lambda_l2': 0.00026663802273190103}","metadata":{"execution":{"iopub.status.busy":"2024-09-28T16:43:31.550382Z","iopub.execute_input":"2024-09-28T16:43:31.550833Z","iopub.status.idle":"2024-09-28T16:43:31.556923Z","shell.execute_reply.started":"2024-09-28T16:43:31.550790Z","shell.execute_reply":"2024-09-28T16:43:31.555661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params =  {'learning_rate': 0.03266434756847073, 'n_estimators': 223, 'num_leaves': 81, 'max_depth': 4, 'min_child_samples': 45, 'subsample': 0.567109684953732, 'colsample_bytree': 0.6395599971419116, 'reg_alpha': 2.5930738324072244, 'reg_lambda': 0.012468184419224114}","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:45:22.768866Z","iopub.execute_input":"2024-10-05T18:45:22.769401Z","iopub.status.idle":"2024-10-05T18:45:22.781254Z","shell.execute_reply.started":"2024-10-05T18:45:22.769354Z","shell.execute_reply":"2024-10-05T18:45:22.779717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = train_folds(lgb.LGBMRegressor, X_train, target, X_test, n_splits=10, params=params)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:45:24.614430Z","iopub.execute_input":"2024-10-05T18:45:24.616154Z","iopub.status.idle":"2024-10-05T18:45:25.265142Z","shell.execute_reply.started":"2024-10-05T18:45:24.616097Z","shell.execute_reply":"2024-10-05T18:45:25.263605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prepare submission\nsubmission = pd.DataFrame({\n    'id': test_ids.values,\n    'sii': convert(predictions)\n})\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:46:35.246596Z","iopub.execute_input":"2024-10-05T18:46:35.247113Z","iopub.status.idle":"2024-10-05T18:46:35.258204Z","shell.execute_reply.started":"2024-10-05T18:46:35.247069Z","shell.execute_reply":"2024-10-05T18:46:35.256448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-10-05T18:46:37.682294Z","iopub.execute_input":"2024-10-05T18:46:37.682766Z","iopub.status.idle":"2024-10-05T18:46:37.700896Z","shell.execute_reply.started":"2024-10-05T18:46:37.682724Z","shell.execute_reply":"2024-10-05T18:46:37.699536Z"},"trusted":true},"execution_count":null,"outputs":[]}]}