{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import necessary libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom scipy.optimize import minimize\nfrom xgboost import XGBRegressor\nfrom sklearn.base import clone\nfrom tqdm import tqdm\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Suppress warnings and set pandas display options","metadata":{}},{"cell_type":"code","source":"warnings.filterwarnings('ignore')\npd.options.display.max_columns = None","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Loading","metadata":{}},{"cell_type":"code","source":"def load_data():\n    train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    sample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n    return train, test, sample","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocess data","metadata":{}},{"cell_type":"code","source":"\ndef preprocess_data(train, test):\n    # Drop 'id' column\n    train = train.drop('id', axis=1)\n    test = test.drop('id', axis=1)\n    \n    # Handle categorical variables\n    cat_cols = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n                'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n                'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n    \n    for col in cat_cols:\n        train[col] = train[col].fillna('Missing').astype('category')\n        test[col] = test[col].fillna('Missing').astype('category')\n        \n        # Create mappings for categorical features\n        train[col] = train[col].cat.codes\n        test[col] = test[col].cat.codes\n    \n    return train, test","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Define quadratic weighted kappa metric","metadata":{}},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Define threshold rounding function","metadata":{}},{"cell_type":"code","source":"def threshold_rounder(predictions, thresholds):\n    return np.digitize(predictions, thresholds)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Define evaluation function for predictions","metadata":{}},{"cell_type":"code","source":"def evaluate_predictions(thresholds, y_true, predictions):\n    rounded_predictions = threshold_rounder(predictions, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_predictions)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Define main training function","metadata":{}},{"cell_type":"code","source":"def train_model(X, y, test_data, n_splits=5):\n    skf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n    oof_predictions = np.zeros(len(y))\n    test_predictions = np.zeros((len(test_data), n_splits))\n    \n    for fold, (train_idx, val_idx) in enumerate(tqdm(skf.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n        \n        model = XGBRegressor(\n            learning_rate=0.05,\n            max_depth=6,\n            n_estimators=200,\n            subsample=0.8,\n            colsample_bytree=0.8,\n            reg_alpha=1,\n            reg_lambda=5,\n            random_state=42\n        )\n        model.fit(X_train, y_train)\n        \n        oof_predictions[val_idx] = model.predict(X_val)\n        test_predictions[:, fold] = model.predict(test_data)\n        \n        train_kappa = quadratic_weighted_kappa(y_train, model.predict(X_train).round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, oof_predictions[val_idx].round(0).astype(int))\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n    \n    print(f\"Mean Validation QWK: {quadratic_weighted_kappa(y, oof_predictions.round(0).astype(int)):.4f}\")\n    \n    return oof_predictions, test_predictions.mean(axis=1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def main():\n    # Load and preprocess data\n    train, test, sample = load_data()\n    train, test = preprocess_data(train, test)\n    \n    X = train.drop('sii', axis=1)\n    y = train['sii']\n    \n    # Train model and get predictions\n    oof_predictions, test_predictions = train_model(X, y, test)\n    \n    # Optimize thresholds\n    optimized_thresholds = minimize(evaluate_predictions,\n                                    x0=[0.5, 1.5, 2.5],\n                                    args=(y, oof_predictions),\n                                    method='Nelder-Mead').x\n    \n    # Apply optimized thresholds to predictions\n    final_predictions = threshold_rounder(test_predictions, optimized_thresholds)\n    \n    # Create submission file\n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': final_predictions\n    })\n    \n    submission.to_csv('submission.csv', index=False)\n    print(\"Submission file created.\")\n    print(f\"Optimized QWK SCORE: {Fore.CYAN}{Style.BRIGHT}{quadratic_weighted_kappa(y, threshold_rounder(oof_predictions, optimized_thresholds)):.3f}{Style.RESET_ALL}\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}