{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport os\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score\nfrom scipy.optimize import minimize\nfrom xgboost import XGBRegressor\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-09-27T08:01:45.596450Z","iopub.execute_input":"2024-09-27T08:01:45.597058Z","iopub.status.idle":"2024-09-27T08:01:46.704573Z","shell.execute_reply.started":"2024-09-27T08:01:45.596998Z","shell.execute_reply":"2024-09-27T08:01:46.703390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:01:46.706276Z","iopub.execute_input":"2024-09-27T08:01:46.707005Z","iopub.status.idle":"2024-09-27T08:01:46.800300Z","shell.execute_reply.started":"2024-09-27T08:01:46.706929Z","shell.execute_reply":"2024-09-27T08:01:46.798901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:01:46.801893Z","iopub.execute_input":"2024-09-27T08:01:46.802302Z","iopub.status.idle":"2024-09-27T08:01:46.837520Z","shell.execute_reply.started":"2024-09-27T08:01:46.802259Z","shell.execute_reply":"2024-09-27T08:01:46.836299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"series_train = pd.read_parquet('/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=0417c91e/part-0.parquet')\nseries_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:01:46.840853Z","iopub.execute_input":"2024-09-27T08:01:46.841333Z","iopub.status.idle":"2024-09-27T08:01:46.933188Z","shell.execute_reply.started":"2024-09-27T08:01:46.841290Z","shell.execute_reply":"2024-09-27T08:01:46.931828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:01:46.934786Z","iopub.execute_input":"2024-09-27T08:01:46.935203Z","iopub.status.idle":"2024-09-27T08:01:47.086082Z","shell.execute_reply.started":"2024-09-27T08:01:46.935148Z","shell.execute_reply":"2024-09-27T08:01:47.084741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.drop('id',axis=1)\ntest = test.drop('id',axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:01:47.087719Z","iopub.execute_input":"2024-09-27T08:01:47.088195Z","iopub.status.idle":"2024-09-27T08:01:47.097547Z","shell.execute_reply.started":"2024-09-27T08:01:47.088149Z","shell.execute_reply":"2024-09-27T08:01:47.096278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n           'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n           'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n           'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n           'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n           'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n           'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n           'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n           'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n           'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n           'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n           'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n           'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n           'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n           'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n           'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n           'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n           'PreInt_EduHx-computerinternet_hoursday','sii']\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:01:47.100118Z","iopub.execute_input":"2024-09-27T08:01:47.100882Z","iopub.status.idle":"2024-09-27T08:01:47.116003Z","shell.execute_reply.started":"2024-09-27T08:01:47.100826Z","shell.execute_reply":"2024-09-27T08:01:47.114841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season','Fitness_Endurance-Season','FGC-Season',\n     'BIA-Season','PAQ_A-Season','PAQ_C-Season','SDS-Season','PreInt_EduHx-Season']","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:01:47.117526Z","iopub.execute_input":"2024-09-27T08:01:47.118101Z","iopub.status.idle":"2024-09-27T08:01:47.126080Z","shell.execute_reply.started":"2024-09-27T08:01:47.118048Z","shell.execute_reply":"2024-09-27T08:01:47.124555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cat = train[categorical]\ntest_cat = test[categorical]","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:01:47.127646Z","iopub.execute_input":"2024-09-27T08:01:47.128037Z","iopub.status.idle":"2024-09-27T08:01:47.140753Z","shell.execute_reply.started":"2024-09-27T08:01:47.127999Z","shell.execute_reply":"2024-09-27T08:01:47.139511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:01:47.142400Z","iopub.execute_input":"2024-09-27T08:01:47.142853Z","iopub.status.idle":"2024-09-27T08:01:47.165862Z","shell.execute_reply.started":"2024-09-27T08:01:47.142811Z","shell.execute_reply":"2024-09-27T08:01:47.164602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numeric_train = train.select_dtypes(include=['float', 'int'])\nnumeric_test = test.select_dtypes(include=['float', 'int'])\n\n# Count null values in each numeric column\nnull_train = numeric_train.isna().sum()\nnull_test = numeric_test.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:01:47.167302Z","iopub.execute_input":"2024-09-27T08:01:47.167615Z","iopub.status.idle":"2024-09-27T08:01:47.178013Z","shell.execute_reply.started":"2024-09-27T08:01:47.167581Z","shell.execute_reply":"2024-09-27T08:01:47.176576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_train\nnull_test\nfor column in numeric_train.columns:\n    mode_train = numeric_train[column].mode()[0] \n    numeric_train[column].fillna(mode_train, inplace=True)\n\nfor column in numeric_test.columns:\n    mode_test = numeric_test[column].mode()[0]\n    numeric_test[column].fillna(mode_test, inplace=True)\n\ntrain[numeric_train.columns] = numeric_train\ntest[numeric_test.columns] = numeric_test","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:01:47.179776Z","iopub.execute_input":"2024-09-27T08:01:47.180210Z","iopub.status.idle":"2024-09-27T08:01:47.249954Z","shell.execute_reply.started":"2024-09-27T08:01:47.180168Z","shell.execute_reply":"2024-09-27T08:01:47.248745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\ndef plot_histograms(df, num_bins=30):\n    sns.set(style=\"whitegrid\")\n\n    num_features = len(df.columns)\n\n    plt.figure(figsize=(15, num_features * 4))\n\n    for i, column in enumerate(df.columns):\n        plt.subplot(num_features, 1, i + 1)\n        sns.histplot(df[column].dropna(), bins=num_bins, kde=False, color='skyblue')\n        plt.title(f'Histogram of {column}', fontsize=14)\n        plt.xlabel(column)\n        plt.ylabel('Count')\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:01:47.255042Z","iopub.execute_input":"2024-09-27T08:01:47.255473Z","iopub.status.idle":"2024-09-27T08:01:47.403717Z","shell.execute_reply.started":"2024-09-27T08:01:47.255433Z","shell.execute_reply":"2024-09-27T08:01:47.402335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_histograms(train)","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:01:47.405079Z","iopub.execute_input":"2024-09-27T08:01:47.405723Z","iopub.status.idle":"2024-09-27T08:02:13.036781Z","shell.execute_reply.started":"2024-09-27T08:01:47.405670Z","shell.execute_reply":"2024-09-27T08:02:13.034653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[categorical] = train_cat.fillna('missing')\ntest[categorical] = test_cat.fillna('missing')\ntrain[categorical] = train_cat.astype('category')\ntest[categorical] = test_cat.astype('category')\n\nfrom sklearn.preprocessing import LabelEncoder\n\ndef label_encode_columns(dataset, categorical_columns):\n    label_encoders = {}\n    for col in categorical_columns:\n        le = LabelEncoder()\n        dataset[col] = le.fit_transform(dataset[col].astype(str))\n        label_encoders[col] = le \n    return dataset, label_encoders\n\n\ntrain_encoded, label_encoders_train = label_encode_columns(train, categorical)\ntest_encoded, label_encoders_test = label_encode_columns(test, categorical)","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:02:13.038701Z","iopub.execute_input":"2024-09-27T08:02:13.039197Z","iopub.status.idle":"2024-09-27T08:02:13.096672Z","shell.execute_reply.started":"2024-09-27T08:02:13.039147Z","shell.execute_reply":"2024-09-27T08:02:13.095469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:02:13.098092Z","iopub.execute_input":"2024-09-27T08:02:13.098448Z","iopub.status.idle":"2024-09-27T08:02:13.123128Z","shell.execute_reply.started":"2024-09-27T08:02:13.098410Z","shell.execute_reply":"2024-09-27T08:02:13.121831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:02:13.124998Z","iopub.execute_input":"2024-09-27T08:02:13.125548Z","iopub.status.idle":"2024-09-27T08:02:13.131907Z","shell.execute_reply.started":"2024-09-27T08:02:13.125492Z","shell.execute_reply":"2024-09-27T08:02:13.130476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:02:13.133304Z","iopub.execute_input":"2024-09-27T08:02:13.133677Z","iopub.status.idle":"2024-09-27T08:02:13.145718Z","shell.execute_reply.started":"2024-09-27T08:02:13.133638Z","shell.execute_reply":"2024-09-27T08:02:13.144239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:02:13.147719Z","iopub.execute_input":"2024-09-27T08:02:13.148269Z","iopub.status.idle":"2024-09-27T08:02:13.158814Z","shell.execute_reply.started":"2024-09-27T08:02:13.148213Z","shell.execute_reply":"2024-09-27T08:02:13.157468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_data(df):\n    cat_columns = df.select_dtypes(include=['object', 'category']).columns\n    for col in cat_columns:\n        df[col] = df[col].astype('category')\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:02:13.160488Z","iopub.execute_input":"2024-09-27T08:02:13.160922Z","iopub.status.idle":"2024-09-27T08:02:13.176813Z","shell.execute_reply.started":"2024-09-27T08:02:13.160881Z","shell.execute_reply":"2024-09-27T08:02:13.175558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def TrainXGB(train, test_data=None, n_splits=5, SEED=42):\n    train = prepare_data(train)\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    if test_data is not None:\n        test_data = prepare_data(test_data)\n\n    params = {\n        'learning_rate': 0.01,\n        'max_depth': 10,\n        'n_estimators': 200,\n        'min_child_weight': 41,\n        'colsample_bytree': 0.6,\n        'reg_alpha': 9.9,\n        'reg_lambda': 4.3,\n        'random_state': SEED,\n        'enable_categorical': True\n    }\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n\n    train_S = []\n    test_S = []\n    accuracies = []\n    models = []\n\n    oof_non_rounded = np.zeros(len(y), dtype=float)\n    oof_rounded = np.zeros(len(y), dtype=int)\n    \n    if test_data is not None:\n        test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = XGBRegressor(**params)\n        model.fit(\n            X_train, y_train,\n            eval_set=[(X_val, y_val)],\n            early_stopping_rounds=50,\n            verbose=0\n        )\n        \n        models.append(model)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n        accuracy = accuracy_score(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        accuracies.append(accuracy)\n\n        if test_data is not None:\n            test_preds[:, fold] = model.predict(test_data)\n\n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}, Accuracy: {accuracy:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n    print(f\"Mean Accuracy ---> {np.mean(accuracies):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n\n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n    final_accuracy = accuracy_score(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {tKappa:.3f}\")\n    print(f\"----> || Final Accuracy :: {final_accuracy:.3f}\")\n\n    if test_data is not None:\n        tpm = test_preds.mean(axis=1)\n        tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n\n        # Create an 'id' column if it doesn't exist\n        if 'id' not in test_data.columns:\n            test_data['id'] = range(len(test_data))\n\n        submission = pd.DataFrame({\n            'id': test_data['id'],\n            'sii': tpTuned\n        })\n        return submission, models, X\n    else:\n        return oof_tuned, models, X\n\n# Usage\nSEED = 42\nParams = {\n    'learning_rate': 0.01,\n    'max_depth': 10,\n    'n_estimators': 100,\n    'min_child_weight': 41,\n    'colsample_bytree': 0.6,\n    'subsample': 0.91,\n    'reg_alpha': 5,\n    'reg_lambda': 4.3,\n    'random_state': SEED,\n    'enable_categorical': True\n}\n\nXGB = XGBRegressor(**Params)\nSubmission, models_train1, X = TrainXGB(train, test)\n\nSubmission.to_csv('submission.csv', index=False)\nprint(Submission['sii'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2024-09-27T08:02:28.155246Z","iopub.execute_input":"2024-09-27T08:02:28.155664Z","iopub.status.idle":"2024-09-27T08:02:32.448552Z","shell.execute_reply.started":"2024-09-27T08:02:28.155626Z","shell.execute_reply":"2024-09-27T08:02:32.447425Z"},"trusted":true},"execution_count":null,"outputs":[]}]}