{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n'''\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))'''\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-10T00:55:38.973624Z","iopub.execute_input":"2024-12-10T00:55:38.974020Z","iopub.status.idle":"2024-12-10T00:55:38.980287Z","shell.execute_reply.started":"2024-12-10T00:55:38.973992Z","shell.execute_reply":"2024-12-10T00:55:38.979396Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Importing libraries\nimport matplotlib.pyplot as plt\nimport xgboost as xgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score, cohen_kappa_score\n\n# Custom QWK Metric\ndef quadratic_weighted_kappa(preds, dtrain):\n    labels = dtrain.get_label()\n    \n    # Round predictions for classification\n    preds = np.round(preds).astype(int)\n    \n    # Ensure that both preds and labels are within the range of valid classes\n    min_class = min(np.min(labels), np.min(preds))\n    max_class = max(np.max(labels), np.max(preds))\n    \n    # Compute Cohen's Quadratic Weighted Kappa\n    qwk = cohen_kappa_score(labels, preds, weights='quadratic')\n    \n    # Return the QWK value\n    return 'qwk', qwk\n","metadata":{"execution":{"iopub.status.busy":"2024-12-10T00:55:38.981790Z","iopub.execute_input":"2024-12-10T00:55:38.982059Z","iopub.status.idle":"2024-12-10T00:55:38.994938Z","shell.execute_reply.started":"2024-12-10T00:55:38.982022Z","shell.execute_reply":"2024-12-10T00:55:38.994191Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import datasets\ntrain_path = '/kaggle/input/child-mind-institute-problematic-internet-use/train.csv'\ntrain = pd.read_csv(train_path)\n\ntest_path = '/kaggle/input/child-mind-institute-problematic-internet-use/test.csv'\ntest = pd.read_csv(test_path)","metadata":{"execution":{"iopub.status.busy":"2024-12-10T00:55:38.995716Z","iopub.execute_input":"2024-12-10T00:55:38.995954Z","iopub.status.idle":"2024-12-10T00:55:39.043790Z","shell.execute_reply.started":"2024-12-10T00:55:38.995932Z","shell.execute_reply":"2024-12-10T00:55:39.043012Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Data-preprocessing:**\n# 1. Deal with missing values.","metadata":{}},{"cell_type":"code","source":"# Shows the graph after rows missing 'sii' are removed, those rows are useless since they're missing the target value anyways\n\n# Assuming 'df' is your original DataFrame\noriginal_missing_values = train.isnull().sum()\n\n# Remove rows where the 'sii' column has missing values\ndf_filtered = train.dropna(subset=['sii'])\nfiltered_missing_values = df_filtered.isnull().sum()\n\n# Filter only columns with missing values in the original DataFrame\nmissing_columns = original_missing_values[original_missing_values > 0].index\noriginal_missing_values = original_missing_values[missing_columns]\nfiltered_missing_values = filtered_missing_values[missing_columns]\n\n# Plotting\nplt.figure(figsize=(6, 14))\n\n# Original missing values (before dropping)\nplt.barh(missing_columns, original_missing_values, color='skyblue', label='Original')\n\n# Missing values after dropping 'sii' rows\nplt.barh(missing_columns, filtered_missing_values, color='lightcoral', label='After Dropping SII Rows')\n\n# Adding values to the end of each bar\nfor index, value in enumerate(original_missing_values):\n    plt.text(value + 0.1, index, str(value), va='center', color='blue')\n\nfor index, value in enumerate(filtered_missing_values):\n    plt.text(value + 0.1, index, str(value), va='center', color='red')\n\n# Adding labels and legend\nplt.xlabel(\"Missing Values Count\")\nplt.title(\"Comparison of Missing Values Before and After Dropping Rows with Missing 'sii'\")\nplt.legend()\nplt.grid(axis=\"x\", linestyle=\"--\", linewidth=0.5)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-10T00:55:39.045272Z","iopub.execute_input":"2024-12-10T00:55:39.045644Z","iopub.status.idle":"2024-12-10T00:55:40.199492Z","shell.execute_reply.started":"2024-12-10T00:55:39.045607Z","shell.execute_reply":"2024-12-10T00:55:40.198427Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove rows where the 'sii' (target) column has missing values\ntrain = train.dropna(subset=['sii'])\n\n# Co-fill in values for bmi from 'Physical-BMI' and 'BIA-BIA_BMI'\ntrain['Combined-BMI'] = train['BIA-BIA_BMI'].combine_first(train['Physical-BMI'])","metadata":{"execution":{"iopub.status.busy":"2024-12-10T00:55:40.201995Z","iopub.execute_input":"2024-12-10T00:55:40.202288Z","iopub.status.idle":"2024-12-10T00:55:40.211501Z","shell.execute_reply.started":"2024-12-10T00:55:40.202260Z","shell.execute_reply":"2024-12-10T00:55:40.210494Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Defining categorical columns\n\ncategorical_columns = [\n    'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n    'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', 'PAQ_A-Season',\n    'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season'\n]\n\n# Ensure categorical columns are marked as categorical\n\nfor col in categorical_columns:\n    train[col] = train[col].astype('category')\n\n# Define features\n\nfeatures = [\n    'Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex', \n    'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-Height',\n    'Physical-Weight', 'Physical-Waist_Circumference', 'Physical-Diastolic_BP',\n    'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Season',\n    'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins',\n    'Fitness_Endurance-Time_Sec', 'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', \n    'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone',\n    'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone',\n    'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n    'BIA-Season', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMR',\n    'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI',\n    'BIA-BIA_Fat', 'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM',\n    'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total',\n    'PAQ_C-Season', 'PAQ_C-PAQ_C_Total',\n    'SDS-Season', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n    'PreInt_EduHx-computerinternet_hoursday', 'Combined-BMI'\n]\n\n# Prepare training data\nX = train[features]\ny = train['sii']\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=0)\n\n# Create DMatrix with categorical columns enabled\ndtrain = xgb.DMatrix(X_train, label=y_train, enable_categorical=True, feature_names=X_train.columns.to_list())\ndtest = xgb.DMatrix(X_test, label=y_test, enable_categorical=True, feature_names=X_test.columns.to_list())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T00:55:40.213003Z","iopub.execute_input":"2024-12-10T00:55:40.213287Z","iopub.status.idle":"2024-12-10T00:55:40.417628Z","shell.execute_reply.started":"2024-12-10T00:55:40.213262Z","shell.execute_reply":"2024-12-10T00:55:40.416918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Set parameters for XGBoost\n\nparams = {\n    'objective': \"multi:softmax\",  # For multi-class classification\n    'num_class': 4,\n    \"device\": \"cuda\" # Enables GPU acceleration\n}\n\nevals = [\n    (dtrain, 'train'), \n    (dtest, 'eval')\n]\n\n# Perform cross-validation\ncv_results = xgb.cv(\n    params=params,\n    dtrain=dtrain,\n    num_boost_round=500,\n    nfold=5,  # Number of folds\n    early_stopping_rounds=10,\n    custom_metric=quadratic_weighted_kappa,\n    as_pandas=True,\n    seed=0\n)\n\nprint(cv_results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T00:55:40.418382Z","iopub.execute_input":"2024-12-10T00:55:40.418622Z","iopub.status.idle":"2024-12-10T00:55:41.917517Z","shell.execute_reply.started":"2024-12-10T00:55:40.418597Z","shell.execute_reply":"2024-12-10T00:55:41.916556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the model with custom QWK evaluation metric\nmodel = xgb.train(\n    params=params,\n    dtrain=dtrain,\n    evals=evals,\n    custom_metric=quadratic_weighted_kappa,  # Use custom QWK metric\n    num_boost_round=100,\n    early_stopping_rounds=10\n)\n\n# Predictions\ny_pred = model.predict(dtest)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T00:55:41.918634Z","iopub.execute_input":"2024-12-10T00:55:41.918898Z","iopub.status.idle":"2024-12-10T00:55:42.500648Z","shell.execute_reply.started":"2024-12-10T00:55:41.918874Z","shell.execute_reply":"2024-12-10T00:55:42.500030Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Final model\ntest['Combined-BMI'] = test['BIA-BIA_BMI'].combine_first(test['Physical-BMI'])\n# Ensure categorical columns are marked as categorical\nfor col in categorical_columns:\n    test[col] = test[col].astype('category')\n\nfinal_test = test[features]\n\n# Create DMatrix with categorical columns enabled\nd_final = xgb.DMatrix(final_test, enable_categorical=True)\n\n# Predictions\nfinal_preds = model.predict(d_final)\n\noutput = pd.DataFrame({'id': test.id,\n                       'sii': final_preds})\noutput.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T00:55:42.501378Z","iopub.execute_input":"2024-12-10T00:55:42.501616Z","iopub.status.idle":"2024-12-10T00:55:42.528702Z","shell.execute_reply.started":"2024-12-10T00:55:42.501591Z","shell.execute_reply":"2024-12-10T00:55:42.528132Z"}},"outputs":[],"execution_count":null}]}