{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"raw","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport seaborn as sns\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor\nimport matplotlib.pyplot as plt\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n","metadata":{"execution":{"iopub.status.busy":"2024-12-10T00:16:29.660933Z","iopub.execute_input":"2024-12-10T00:16:29.661373Z","iopub.status.idle":"2024-12-10T00:16:29.670224Z","shell.execute_reply.started":"2024-12-10T00:16:29.661334Z","shell.execute_reply":"2024-12-10T00:16:29.668652Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42\nn_splits = 5\n\n# Load datasets\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T00:16:37.800562Z","iopub.execute_input":"2024-12-10T00:16:37.801023Z","iopub.status.idle":"2024-12-10T00:18:57.151102Z","shell.execute_reply.started":"2024-12-10T00:16:37.800984Z","shell.execute_reply":"2024-12-10T00:18:57.150125Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Understanding of Project\nThe aim of this competition is to predict the Severity Impairment Index (sii), which measures the level of problematic internet use among children and adolescents, based on physical activity data and other features.\n\n\nsii is derived from PCIAT-PCIAT_Total, the sum of scores from the Parent-Child Internet Addiction Test (PCIAT: 20 questions, scored 0-5).\n\n\n\nTarget Variable (sii) is defined as:\n\n\n0: None (PCIAT-PCIAT_Total from 0 to 30)\n1: Mild (PCIAT-PCIAT_Total from 31 to 49)\n2: Moderate (PCIAT-PCIAT_Total from 50 to 79)\n3: Severe (PCIAT-PCIAT_Total 80 and more)\n\nThis makes sii an ordinal categorical variable with four levels, where the order of categories is meaningful.\n\nType of Machine Learning Problem we can use with sii as a target:\n\nOrdinal classification (ordinal logistic regression, models with custom ordinal loss functions)\nMulticlass classification (treat sii as a nominal categorical variable without considering the order)\nRegression (ignore the discrete nature of categories and treat sii as a continuous variable, then round prediction)\nCustom (e.g. loss functions that penalize errors based on the distance between categories)\nWe can also use PCIAT-PCIAT_Total as a continuous target variable, and implement regression on PCIAT-PCIAT_Total and then map predictions to sii categories.\n\n\nFinally, another strategy involves predicting responses to each question of the Parent-Child Internet Addiction Test: i.e. pedict individual question scores as separate targets, sum the predicted scores to get the PCIAT-PCIAT_Total and map predictions to the corresponding sii category.","metadata":{}},{"cell_type":"markdown","source":"Feature Selection: The dataset contains features related to physical characteristics (e.g., BMI, Height, Weight), behavioral aspects (e.g., internet usage), and fitness data (e.g., endurance time).\n\nCategorical Feature Encoding: Categorical features are mapped to numerical values using custom mappings for each unique category within the dataset. This ensures compatibility with machine learning algorithms that require numerical input.\n\nTime Series Aggregation: Time series statistics (e.g., mean, standard deviation) from the actigraphy data are computed and merged into the main dataset to create additional features for model training.","metadata":{}},{"cell_type":"markdown","source":"Target Variable (sii) is defined as:\n\n0: None (PCIAT-PCIAT_Total from 0 to 30)\n\n1: Mild (PCIAT-PCIAT_Total from 31 to 49)\n\n2: Moderate (PCIAT-PCIAT_Total from 50 to 79)\n\n3: Severe (PCIAT-PCIAT_Total 80 and more)\n\nThis makes sii an ordinal categorical variable with four levels, where the order of categories is meaningful.","metadata":{}},{"cell_type":"markdown","source":"# 1. Exploratory Analysis","metadata":{}},{"cell_type":"code","source":"print(\"train\",train.shape)\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:26.678572Z","iopub.execute_input":"2024-12-09T23:31:26.678935Z","iopub.status.idle":"2024-12-09T23:31:26.814126Z","shell.execute_reply.started":"2024-12-09T23:31:26.678899Z","shell.execute_reply":"2024-12-09T23:31:26.813039Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Data Description","metadata":{}},{"cell_type":"code","source":"train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:26.815717Z","iopub.execute_input":"2024-12-09T23:31:26.816208Z","iopub.status.idle":"2024-12-09T23:31:27.262231Z","shell.execute_reply.started":"2024-12-09T23:31:26.816150Z","shell.execute_reply":"2024-12-09T23:31:27.261124Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Visualizing the value counts of columns","metadata":{}},{"cell_type":"code","source":"train1 = train.copy()\n# 2. Visualize value_counts for other columns\nfor column in train1.columns:\n    if train1[column].dtype == 'object' or len(train1[column].unique()) < 20:  # Adjust this threshold as needed\n        plt.figure(figsize=(8, 5))\n        sns.countplot(y=train1[column], order=train1[column].value_counts().index)\n        plt.title(f'Value Counts of {column}')\n        plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:27.265078Z","iopub.execute_input":"2024-12-09T23:31:27.265567Z","iopub.status.idle":"2024-12-09T23:31:38.650705Z","shell.execute_reply.started":"2024-12-09T23:31:27.265513Z","shell.execute_reply":"2024-12-09T23:31:38.649564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Calculate value counts\nvc = train['Basic_Demos-Enroll_Season'].value_counts()\n\n# Create a pie chart\nplt.figure(figsize=(10, 5))\nplt.pie(vc.values, labels=vc.index, autopct='%1.1f%%', startangle=90)\nplt.title('Season of Enrollment', fontsize=16)\nplt.axis('equal')  # Equal aspect ratio ensures that pie is drawn as a circle\n# Add a legend\nplt.legend(title=\"Seasons\", loc=\"best\", bbox_to_anchor=(1, 0, 0.5, 1))\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:38.652310Z","iopub.execute_input":"2024-12-09T23:31:38.652771Z","iopub.status.idle":"2024-12-09T23:31:38.985987Z","shell.execute_reply.started":"2024-12-09T23:31:38.652718Z","shell.execute_reply":"2024-12-09T23:31:38.984765Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vc = train['Basic_Demos-Sex'].value_counts()\nplt.pie(vc.values, labels=['boys', 'girls'], autopct='%1.1f%%')\nplt.title('Sex of participant')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:38.987490Z","iopub.execute_input":"2024-12-09T23:31:38.987859Z","iopub.status.idle":"2024-12-09T23:31:39.114481Z","shell.execute_reply.started":"2024-12-09T23:31:38.987817Z","shell.execute_reply":"2024-12-09T23:31:39.112868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator\n\n_, axs = plt.subplots(2, 1, sharex=True, figsize=(10, 10))\nlabels = ['boys', 'girls']\ncolors = ['lightblue', 'coral']\n\nfor sex in range(2):\n    ax = axs[sex]\n    vc = train[train['Basic_Demos-Sex'] == sex]['Basic_Demos-Age'].value_counts().sort_index()\n    \n    ax.bar(vc.index,\n           vc.values,\n           color=colors[sex],\n           label=labels[sex])\n    \n    ax.xaxis.set_major_locator(MaxNLocator(integer=True))\n    ax.set_ylabel('Count')\n    ax.legend()\n    \n    # Add percentage labels on top of each bar\n    total = vc.sum()\n    for i, v in enumerate(vc):\n        ax.text(vc.index[i], v, f'{v/total*100:.1f}%', \n                ha='center', va='bottom')\n\nplt.suptitle('Age Distribution by Sex', fontsize=16)\naxs[1].set_xlabel('Age (years)')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:39.116229Z","iopub.execute_input":"2024-12-09T23:31:39.116833Z","iopub.status.idle":"2024-12-09T23:31:39.907364Z","shell.execute_reply.started":"2024-12-09T23:31:39.116766Z","shell.execute_reply":"2024-12-09T23:31:39.906172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import PercentFormatter\n\n_, axs = plt.subplots(2, 1, sharex=True, sharey=True, figsize=(10, 8))\nlabels = ['boys', 'girls']\ncolors = ['lightblue', 'coral']\ntarget_labels = ['Typical', 'Mild', 'Moderate', 'Severe']  # Assuming these are your target labels\n\nfor sex in range(2):\n    ax = axs[sex]\n    vc = train[train['Basic_Demos-Sex'] == sex]['sii'].value_counts().sort_index()\n    \n    ax.bar(vc.index,\n           vc.values / vc.sum(),\n           color=colors[sex],\n           label=labels[sex])\n    \n    ax.set_xticks(np.arange(4), target_labels)\n    ax.yaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\n    ax.set_ylabel('Percentage')\n    ax.legend()\n    \n    # Add percentage labels on top of each bar\n    for i, v in enumerate(vc):\n        ax.text(vc.index[i], v/vc.sum(), f'{v/vc.sum()*100:.1f}%', \n                ha='center', va='bottom')\n\nplt.suptitle('Target Distribution by Sex', fontsize=16)\naxs[1].set_xlabel('Severity Impairment Index (sii)')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:39.909031Z","iopub.execute_input":"2024-12-09T23:31:39.909504Z","iopub.status.idle":"2024-12-09T23:31:40.384934Z","shell.execute_reply.started":"2024-12-09T23:31:39.909452Z","shell.execute_reply":"2024-12-09T23:31:40.383677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Assuming 'train' is your pandas DataFrame\nsupervised_usable = train.dropna(subset=['sii'])\n\nplt.figure(figsize=(14, 12))\n\ncolumns_to_correlate = [\n    'PCIAT-PCIAT_Total', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'Physical-BMI', \n    'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n    'Physical-Diastolic_BP', 'Physical-Systolic_BP', 'Physical-HeartRate',\n    'PreInt_EduHx-computerinternet_hoursday', 'SDS-SDS_Total_T', 'PAQ_A-PAQ_A_Total',\n    'PAQ_C-PAQ_C_Total', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins','Fitness_Endurance-Time_Sec',\n    'FGC-FGC_CU', 'FGC-FGC_GSND','FGC-FGC_GSD','FGC-FGC_PU','FGC-FGC_SRL','FGC-FGC_SRR','FGC-FGC_TL','BIA-BIA_Activity_Level_num', \n    'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n    'BIA-BIA_FFMI','BIA-BIA_FMI', 'BIA-BIA_Fat','BIA-BIA_Frame_num','BIA-BIA_ICW','BIA-BIA_LDM','BIA-BIA_LST',\n    'BIA-BIA_SMM','BIA-BIA_TBW'\n]\n\ncorr_matrix = supervised_usable[columns_to_correlate].corr()\n\nsii_corr = corr_matrix['PCIAT-PCIAT_Total'].drop('PCIAT-PCIAT_Total')\nfiltered_corr = sii_corr[(sii_corr.abs() > 0.1)]\n\nprint(filtered_corr)\n\nplt.figure(figsize=(10, 8))\nax = filtered_corr.sort_values().plot(kind='barh', color='coral')\nplt.title('Features with |Correlation| > 0.1 with PCIAT-PCIAT_Total')\nplt.xlabel('Correlation coefficient', fontsize=12)\nplt.ylabel('Features', fontsize=12)\n\n# Add correlation values to the end of each bar\nfor i, v in enumerate(filtered_corr.sort_values()):\n    ax.text(v, i, f' {v:.2f}', va='center', fontsize=10)\n\nplt.tight_layout()\nplt.show()\n\n# Heatmap of the correlation matrix\nplt.figure(figsize=(16, 14))\nsns.heatmap(corr_matrix, annot=False, cmap='coolwarm', vmin=-1, vmax=1, center=0)\nplt.title('Correlation Heatmap of All Features', fontsize=16)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:40.386348Z","iopub.execute_input":"2024-12-09T23:31:40.386677Z","iopub.status.idle":"2024-12-09T23:31:42.300575Z","shell.execute_reply.started":"2024-12-09T23:31:40.386644Z","shell.execute_reply":"2024-12-09T23:31:42.299216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:42.305733Z","iopub.execute_input":"2024-12-09T23:31:42.306248Z","iopub.status.idle":"2024-12-09T23:31:42.434356Z","shell.execute_reply.started":"2024-12-09T23:31:42.306199Z","shell.execute_reply":"2024-12-09T23:31:42.433169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def update(df):\n        global cat_c\n        for c in cat_c: \n            df[c] = df[c].fillna('Missing')\n            df[c] = df[c].astype('category')\n        return df\n\ntrain = update(train)\ntest = update(test)\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:42.436338Z","iopub.execute_input":"2024-12-09T23:31:42.436672Z","iopub.status.idle":"2024-12-09T23:31:42.585146Z","shell.execute_reply.started":"2024-12-09T23:31:42.436639Z","shell.execute_reply":"2024-12-09T23:31:42.584124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T00:19:17.973786Z","iopub.execute_input":"2024-12-10T00:19:17.974173Z","iopub.status.idle":"2024-12-10T00:19:18.165007Z","shell.execute_reply.started":"2024-12-10T00:19:17.974139Z","shell.execute_reply":"2024-12-10T00:19:18.163873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['sii'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:42.587042Z","iopub.execute_input":"2024-12-09T23:31:42.587501Z","iopub.status.idle":"2024-12-09T23:31:42.600359Z","shell.execute_reply.started":"2024-12-09T23:31:42.587453Z","shell.execute_reply":"2024-12-09T23:31:42.599221Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Extraction/Dimensionality Reduction based Analysis","metadata":{}},{"cell_type":"markdown","source":"Without Feature Extraction","metadata":{}},{"cell_type":"code","source":"## without pca\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\nfor col in cat_c:\n    # Create mapping from the training dataset\n    mapping = create_mapping(col, train)\n    # Map training data\n    train[col] = train[col].replace(mapping).astype(int)\n    # Map test data, using the same mapping\n    test[col] = test[col].replace(mapping).astype(int)\n    \n    # Optionally, if you want to handle unseen values in the test set:\n    test[col] = test[col].fillna(-1)  # Assign a special value for missing or unseen categories\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\ndef TrainML(model_class, test_data):\n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n       \n        model = clone(model_class)\n        model.fit(X_train, y_train)\n        \n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n    return submission, model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:42.601949Z","iopub.execute_input":"2024-12-09T23:31:42.602471Z","iopub.status.idle":"2024-12-09T23:31:42.664163Z","shell.execute_reply.started":"2024-12-09T23:31:42.602415Z","shell.execute_reply":"2024-12-09T23:31:42.662842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Params = {'learning_rate': 0.02884249148676999, \n          'max_depth': 15, \n          'num_leaves': 470,\n          'min_data_in_leaf': 14,\n           \n          'feature_fraction': 0.7987976913702801, \n          'bagging_fraction': 0.7602261703576205, \n          'bagging_freq': 4, \n           'lambda_l1': 4.735462555910575, \n          'lambda_l2': 4.735028557007343e-06\n         }\n\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=200)\n\n\n# Train the ensemble model\nSubmission, model = TrainML(Light, test)\n\n# Save submission\n# Submission.to_csv('submission.csv', index=False)\nprint(Submission['sii'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:42.665459Z","iopub.execute_input":"2024-12-09T23:31:42.665796Z","iopub.status.idle":"2024-12-09T23:31:58.339249Z","shell.execute_reply.started":"2024-12-09T23:31:42.665761Z","shell.execute_reply":"2024-12-09T23:31:58.337963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_importance_df = pd.DataFrame({\n    'Feature': model.booster_.feature_name(),\n    'Importance': model.booster_.feature_importance(importance_type='gain')\n})\n\nfeature_importance_df = feature_importance_df.sort_values(by='Importance', ascending=False)\n\nplt.figure(figsize=(10, 5))\nsns.barplot(x='Importance', y='Feature', data=feature_importance_df.head(20)) \nplt.title(\"Top 20 Feature Importance\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:58.340658Z","iopub.execute_input":"2024-12-09T23:31:58.341009Z","iopub.status.idle":"2024-12-09T23:31:58.782701Z","shell.execute_reply.started":"2024-12-09T23:31:58.340974Z","shell.execute_reply":"2024-12-09T23:31:58.781240Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Feature Extraction (PCA)","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.base import clone\nfrom scipy.optimize import minimize\n\ndef TrainML_PCA(model_class, test_data):\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n    \n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X_pca, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X_pca[train_idx], X_pca[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n       \n        model = clone(model_class)\n        model.fit(X_train, y_train)\n        \n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n    \n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n    \n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n    print(f\"----> || Optimized QWK SCORE :: {tKappa:.3f}\")\n    \n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n    return submission\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:58.784670Z","iopub.execute_input":"2024-12-09T23:31:58.785192Z","iopub.status.idle":"2024-12-09T23:31:58.800358Z","shell.execute_reply.started":"2024-12-09T23:31:58.785137Z","shell.execute_reply":"2024-12-09T23:31:58.799023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Here, we have applied PCA with 10% error rate or 90% explained variance\n# Preprocess data\nX = train.drop(['sii'], axis=1)\ny = train['sii']\n\n# Handling missing values\nimputer = SimpleImputer(strategy='mean')\nX_imputed = imputer.fit_transform(X)\ntest_imputed = imputer.transform(test)\n\n# Standardizing data\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X_imputed)\ntest_scaled = scaler.transform(test_imputed)\n\n# Applying PCA\npca = PCA(n_components=0.90)\nX_pca = pca.fit_transform(X_scaled)\ntest_pca = pca.transform(test_scaled)\n\n# PCA details\nprint(f\"PCA retained {pca.n_components_} components, explaining {np.sum(pca.explained_variance_ratio_):.2%} of variance.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:58.801748Z","iopub.execute_input":"2024-12-09T23:31:58.802163Z","iopub.status.idle":"2024-12-09T23:31:58.909739Z","shell.execute_reply.started":"2024-12-09T23:31:58.802096Z","shell.execute_reply":"2024-12-09T23:31:58.906130Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Calculate cumulative explained variance ratio\ncumulative_variance_ratio = np.cumsum(pca.explained_variance_ratio_)\n\n# Create the plot\nplt.figure(figsize=(10, 6))\nplt.plot(range(1, len(cumulative_variance_ratio) + 1), cumulative_variance_ratio, 'bo-')\nplt.xlabel('Number of Components')\nplt.ylabel('Cumulative Explained Variance Ratio')\nplt.title('Cumulative Explained Variance Ratio vs Number of PCA Components')\nplt.grid(True)\n\n# Add a horizontal line at 90% explained variance\nplt.axhline(y=0.9, color='r', linestyle='--', label='90% Explained Variance')\n\n# Add the number of components for 90% explained variance\nn_components_90 = next(i for i, var in enumerate(cumulative_variance_ratio) if var >= 0.9) + 1\nplt.annotate(f'{n_components_90}', \n             xy=(n_components_90, 0.9), \n             xytext=(n_components_90 + 1, 0.86))\n\nplt.legend()\nplt.tight_layout()\nplt.savefig(\"pca_plotting.pdf\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:58.911191Z","iopub.execute_input":"2024-12-09T23:31:58.915404Z","iopub.status.idle":"2024-12-09T23:31:59.658628Z","shell.execute_reply.started":"2024-12-09T23:31:58.915339Z","shell.execute_reply":"2024-12-09T23:31:59.657496Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Train LightGBM model with PCA\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=200)\nSubmission = TrainML_PCA(Light, test_pca)\nprint(Submission['sii'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:31:59.660246Z","iopub.execute_input":"2024-12-09T23:31:59.660701Z","iopub.status.idle":"2024-12-09T23:32:07.724162Z","shell.execute_reply.started":"2024-12-09T23:31:59.660647Z","shell.execute_reply":"2024-12-09T23:32:07.722971Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"WITH t-SNE","metadata":{}},{"cell_type":"code","source":"from sklearn.manifold import TSNE\n\ndef TrainML_tSNE(model_class, test_data):\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n    \n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X_tsne, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X_tsne[train_idx], X_tsne[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n       \n        model = clone(model_class)\n        model.fit(X_train, y_train)\n        \n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n    \n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n    \n    KappaOptimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    assert KappaOptimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOptimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n    print(f\"----> || Optimized QWK SCORE :: {tKappa:.3f}\")\n    \n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOptimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:32:07.725659Z","iopub.execute_input":"2024-12-09T23:32:07.726011Z","iopub.status.idle":"2024-12-09T23:32:07.738619Z","shell.execute_reply.started":"2024-12-09T23:32:07.725975Z","shell.execute_reply":"2024-12-09T23:32:07.737432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocess data\nX = train.drop(['sii'], axis=1)\ny = train['sii']\n\n# Handling missing values\nimputer = SimpleImputer(strategy='mean')\nX_imputed = imputer.fit_transform(X)\ntest_imputed = imputer.transform(test)\n\n# Standardizing data\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X_imputed)\ntest_scaled = scaler.transform(test_imputed)\n\n# Applying t-SNE\n# Adjust perplexity based on the dataset size\nperplexity_train = min(30, X_scaled.shape[0] - 1)  # Ensure perplexity < n_samples\nperplexity_test = min(30, test_scaled.shape[0] - 1)\n\n# Apply t-SNE to training data\ntsne_train = TSNE(n_components=3, random_state=SEED, perplexity=perplexity_train, n_iter=1000)\nX_tsne = tsne_train.fit_transform(X_scaled)\n\n# Apply t-SNE to test data\ntsne_test = TSNE(n_components=3, random_state=SEED, perplexity=perplexity_test, n_iter=1000)\ntest_tsne = tsne_test.fit_transform(test_scaled)\n\nprint(f\"t-SNE reduced training data to {X_tsne.shape[1]} components.\")\nprint(f\"t-SNE reduced test data to {test_tsne.shape[1]} components.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:32:07.740444Z","iopub.execute_input":"2024-12-09T23:32:07.740816Z","iopub.status.idle":"2024-12-09T23:33:57.917273Z","shell.execute_reply.started":"2024-12-09T23:32:07.740780Z","shell.execute_reply":"2024-12-09T23:33:57.915753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom mpl_toolkits.mplot3d import Axes3D\nimport seaborn as sns\n\n# 2D Plot\nplt.figure(figsize=(10, 8))\nscatter = plt.scatter(X_tsne[:, 0], X_tsne[:, 1], c=y, cmap='viridis', alpha=0.7)\nplt.colorbar(scatter)\nplt.title('2D t-SNE Visualization of Training Data')\nplt.xlabel('t-SNE Component 1')\nplt.ylabel('t-SNE Component 2')\nplt.show()\n\n# 3D Plot\nfig = plt.figure(figsize=(10, 8))\nax = fig.add_subplot(111, projection='3d')\nscatter = ax.scatter(X_tsne[:, 0], X_tsne[:, 1], X_tsne[:, 2], c=y, cmap='viridis', alpha=0.7)\nfig.colorbar(scatter)\nax.set_title('3D t-SNE Visualization of Training Data')\nax.set_xlabel('t-SNE Component 1')\nax.set_ylabel('t-SNE Component 2')\nax.set_zlabel('t-SNE Component 3')\nplt.savefig(\"3d_plotting.pdf\")\nplt.show()\n\n# Pairplot\ntsne_df = pd.DataFrame(data=X_tsne, columns=['Component 1', 'Component 2', 'Component 3'])\ntsne_df['SII'] = y\nsns.pairplot(tsne_df, hue='SII', palette='viridis', diag_kind='kde')\nplt.suptitle('Pairplot of t-SNE Components', y=1.02)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:33:57.919034Z","iopub.execute_input":"2024-12-09T23:33:57.919596Z","iopub.status.idle":"2024-12-09T23:34:03.677181Z","shell.execute_reply.started":"2024-12-09T23:33:57.919538Z","shell.execute_reply":"2024-12-09T23:34:03.676069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train LightGBM model with t-SNE\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=200)\nSubmission = TrainML_tSNE(Light, test_tsne)\nprint(Submission['sii'].value_counts())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:34:03.678586Z","iopub.execute_input":"2024-12-09T23:34:03.678993Z","iopub.status.idle":"2024-12-09T23:34:04.918021Z","shell.execute_reply.started":"2024-12-09T23:34:03.678947Z","shell.execute_reply":"2024-12-09T23:34:04.916816Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"It is noted that with PCA and t-SNE, there is significant performance drop as compared to the original dataset.","metadata":{}},{"cell_type":"markdown","source":"# 2. Model application","metadata":{}},{"cell_type":"markdown","source":"* Model Types: Various models are used, including:\n    - LightGBM: A gradient-boosting framework known for its speed and efficiency with large datasets.\n    - XGBoost: Another powerful gradient-boosting model used for structured data.\n    - CatBoost: Optimized for categorical features without the need for extensive preprocessing.\n    - Voting Regressor: An ensemble model that combines the predictions of LightGBM, XGBoost, and CatBoost for better accuracy.\n* Cross-Validation: Stratified K-Folds cross-validation is employed to split the data into training and validation sets, ensuring balanced class distribution in each fold.\n* Quadratic Weighted Kappa (QWK): The performance of the models is evaluated using QWK, which measures the agreement between predicted and actual values, taking into account the ordinal nature of the target variable.\n* Threshold Optimization: The minimize function from scipy.optimize is used to fine-tune decision thresholds that map continuous predictions to discrete categories (None, Mild, Moderate, Severe).","metadata":{}},{"cell_type":"markdown","source":"### LGBM model","metadata":{}},{"cell_type":"code","source":"Params = {'learning_rate': 0.02884249148676999, \n          'max_depth': 15, \n          'num_leaves': 470,\n          'min_data_in_leaf': 14,\n           \n          'feature_fraction': 0.7987976913702801, \n          'bagging_fraction': 0.7602261703576205, \n          'bagging_freq': 4, \n           'lambda_l1': 4.735462555910575, \n          'lambda_l2': 4.735028557007343e-06\n         }\n\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=200)\n\n\n# Train the ensemble model\nSubmission,_ = TrainML(Light, test)\n\n# Save submission\n# Submission.to_csv('submission.csv', index=False)\nprint(Submission['sii'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:34:04.919748Z","iopub.execute_input":"2024-12-09T23:34:04.920260Z","iopub.status.idle":"2024-12-09T23:34:20.442725Z","shell.execute_reply.started":"2024-12-09T23:34:04.920193Z","shell.execute_reply":"2024-12-09T23:34:20.441468Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### XGB model","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBRegressor\n\n# Assuming SEED and TrainML function are defined elsewhere in your code\n\n# XGBoost\nxgb_params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED}\n\nxgb_model = XGBRegressor(**xgb_params)\nxgb_submission,_ = TrainML(xgb_model, test)\nxgb_submission.to_csv('xgb_submission.csv', index=False)\nprint(\"XGBoost Submission:\")\nprint(xgb_submission['sii'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:34:20.444272Z","iopub.execute_input":"2024-12-09T23:34:20.444748Z","iopub.status.idle":"2024-12-09T23:35:05.213230Z","shell.execute_reply.started":"2024-12-09T23:34:20.444693Z","shell.execute_reply":"2024-12-09T23:35:05.212122Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### CatBoost Model","metadata":{}},{"cell_type":"markdown","source":"LightGBM Parameters: Hyperparameters such as learning_rate, max_depth, num_leaves, and feature_fraction are tuned to improve the performance of the LightGBM model. These parameters control the complexity of the model and its ability to generalize to new data.\nXGBoost and CatBoost Parameters: Similar tuning is applied for XGBoost and CatBoost, adjusting parameters such as n_estimators, max_depth, learning_rate, subsample, and regularization terms (reg_alpha, reg_lambda). These help in controlling overfitting and ensuring the model's robustness.","metadata":{}},{"cell_type":"code","source":"from catboost import CatBoostRegressor\n\n# CatBoost\ncat_params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 3,  \n}\n\ncat_model = CatBoostRegressor(**cat_params)\ncat_submission, _ = TrainML(cat_model, test)\n# cat_submission.to_csv('catboost_submission.csv', index=False)\nprint(\"\\nCatBoost Submission:\")\nprint(cat_submission['sii'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:35:05.214717Z","iopub.execute_input":"2024-12-09T23:35:05.215172Z","iopub.status.idle":"2024-12-09T23:35:19.892298Z","shell.execute_reply.started":"2024-12-09T23:35:05.215123Z","shell.execute_reply":"2024-12-09T23:35:19.891239Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### RF model","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.pipeline import Pipeline\n\n# Create an imputer\nimputer = SimpleImputer(strategy='mean')\nrf_params = {\n    'n_estimators': 200,\n    'max_depth': 15,\n    'min_samples_split': 2,\n    'min_samples_leaf': 1,\n    'max_features': 'sqrt',\n    'bootstrap': True\n}\n\nrf_pipeline = Pipeline([\n    ('imputer', imputer),\n    ('rf', RandomForestRegressor(**rf_params, random_state=SEED))\n])\n\nrf_submission,_ = TrainML(rf_pipeline, test)\n# rf_submission.to_csv('rf_submission.csv', index=False)\nprint(\"\\nRandom Forest Submission:\")\nprint(rf_submission['sii'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:35:19.893927Z","iopub.execute_input":"2024-12-09T23:35:19.894376Z","iopub.status.idle":"2024-12-09T23:35:29.898223Z","shell.execute_reply.started":"2024-12-09T23:35:19.894329Z","shell.execute_reply":"2024-12-09T23:35:29.897113Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### KNN","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom scipy.optimize import minimize\nfrom tqdm import tqdm\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainKNN(n_neighbors, X, y, test_data, n_splits=5, random_state=42):\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=random_state)\n    \n    # Create a pipeline with imputation\n    knn_pipeline = Pipeline([\n        ('imputer', SimpleImputer(strategy='mean')),\n        ('knn', KNeighborsRegressor(n_neighbors=n_neighbors))\n    ])\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float)\n    test_preds = np.zeros((len(test_data), n_splits))\n    \n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=f\"Training Folds for k={n_neighbors}\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n       \n        knn_pipeline.fit(X_train, y_train)\n        \n        y_train_pred = knn_pipeline.predict(X_train)\n        y_val_pred = knn_pipeline.predict(X_val)\n        \n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        \n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n        \n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = knn_pipeline.predict(test_data)\n    \n    # Threshold optimization\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n    \n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n    \n    return {\n        'train_qwk': np.mean(train_S),\n        'val_qwk': np.mean(test_S),\n        'optimized_qwk': tKappa,\n        'submission': submission\n    }\n\n# Prepare data\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mapping).astype(int)\n\nX = train.drop(['sii'], axis=1)\ny = train['sii']\n\n# Convergence analysis\nk_values = range(1, 25)\nresults = []\n\nfor k in k_values:\n    result = TrainKNN(k, X, y, test)\n    results.append({\n        'k': k,\n        'train_qwk': result['train_qwk'],\n        'val_qwk': result['val_qwk'],\n        'optimized_qwk': result['optimized_qwk'],\n        'submission': result['submission']\n    })\n\n# Plotting\nplt.figure(figsize=(12, 6))\nresults_df = pd.DataFrame(results)\n\nplt.plot(results_df['k'], results_df['train_qwk'], label='Train QWK', marker='o')\nplt.plot(results_df['k'], results_df['val_qwk'], label='Validation QWK', marker='s')\nplt.plot(results_df['k'], results_df['optimized_qwk'], label='Optimized QWK', marker='^')\n\nplt.title('KNN Performance Convergence')\nplt.xlabel('Number of Neighbors (k)')\nplt.ylabel('Quadratic Weighted Kappa')\nplt.legend()\nplt.grid(True)\nplt.tight_layout()\nplt.show()\n\n# Print results for reference\nprint(results_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:43:39.764862Z","iopub.execute_input":"2024-12-09T23:43:39.766217Z","iopub.status.idle":"2024-12-09T23:44:02.643341Z","shell.execute_reply.started":"2024-12-09T23:43:39.766166Z","shell.execute_reply":"2024-12-09T23:44:02.642152Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(results_df['submission'][5].value_counts)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:45:35.043414Z","iopub.execute_input":"2024-12-09T23:45:35.043831Z","iopub.status.idle":"2024-12-09T23:45:35.051393Z","shell.execute_reply.started":"2024-12-09T23:45:35.043790Z","shell.execute_reply":"2024-12-09T23:45:35.050173Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Baysian","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.preprocessing import StandardScaler\nfrom scipy.optimize import minimize\nfrom scipy.spatial.distance import euclidean, mahalanobis\nfrom tqdm import tqdm\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\nclass RobustBayesianDistanceClassifier:\n    def __init__(self, distance_metric='euclidean', regularization=1e-6):\n        self.distance_metric = distance_metric\n        self.regularization = regularization\n        self.X_train = None\n        self.y_train = None\n        self.class_means = None\n        self.class_covs = None\n        self.class_priors = None\n        \n    def fit(self, X, y):\n        self.X_train = X.values\n        self.y_train = y\n        \n        # Compute class priors\n        self.class_priors = {cls: np.mean(self.y_train == cls) for cls in np.unique(self.y_train)}\n        \n        # Compute class means and regularized covariances\n        self.class_means = {}\n        self.class_covs = {}\n        for cls in np.unique(self.y_train):\n            class_data = self.X_train[self.y_train == cls]\n            \n            # Compute mean\n            self.class_means[cls] = np.mean(class_data, axis=0)\n            \n            # Compute covariance with regularization\n            cov = np.cov(class_data.T)\n            cov += np.eye(cov.shape[0]) * self.regularization\n            self.class_covs[cls] = cov\n        \n        return self\n    \n    def predict(self, X):\n        X_test = X.values\n        predictions = []\n        \n        for sample in X_test:\n            distances = {}\n            for cls in self.class_means.keys():\n                if self.distance_metric == 'euclidean':\n                    distance = euclidean(sample, self.class_means[cls])\n                elif self.distance_metric == 'mahalanobis':\n                    # Compute Mahalanobis distance with regularized inverse\n                    cov_inv = np.linalg.inv(self.class_covs[cls])\n                    distance = mahalanobis(sample, self.class_means[cls], cov_inv)\n                \n                # Compute probability based on distance\n                distances[cls] = -distance * self.class_priors[cls]\n            \n            predictions.append(max(distances, key=distances.get))\n        \n        return np.array(predictions)\n\ndef TrainBayesianDistance(distance_metric, X, y, test_data, n_splits=5, random_state=42):\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=random_state)\n    \n    # Create a pipeline with imputation and scaling\n    pipeline = Pipeline([\n        ('imputer', SimpleImputer(strategy='mean')),\n        ('scaler', StandardScaler())\n    ])\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float)\n    test_preds = np.zeros((len(test_data), n_splits))\n    \n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=f\"Training Folds for {distance_metric}\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n       \n        # Impute and scale data\n        X_train_processed = pipeline.fit_transform(X_train)\n        X_val_processed = pipeline.transform(X_val)\n        test_processed = pipeline.transform(test_data)\n        \n        # Train Bayesian classifier\n        model = RobustBayesianDistanceClassifier(distance_metric=distance_metric)\n        model.fit(pd.DataFrame(X_train_processed), y_train)\n        \n        y_train_pred = model.predict(pd.DataFrame(X_train_processed))\n        y_val_pred = model.predict(pd.DataFrame(X_val_processed))\n        \n        # Assign predictions\n        oof_non_rounded[test_idx] = y_val_pred\n        \n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred)\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred)\n        \n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(pd.DataFrame(test_processed))\n    \n    # Threshold optimization\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead')\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n    \n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n    \n    return {\n        'train_qwk': np.mean(train_S),\n        'val_qwk': np.mean(test_S),\n        'optimized_qwk': tKappa,\n        'submission': submission\n    }\n\n# Prepare data\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mapping).astype(int)\n\nX = train.drop(['sii'], axis=1)\ny = train['sii']\n\n# Convergence analysis for both distance metrics\ndistance_metrics = ['euclidean', 'mahalanobis']\nresults = []\n\nfor metric in distance_metrics:\n    result = TrainBayesianDistance(metric, X, y, test)\n    results.append({\n        'metric': metric,\n        'train_qwk': result['train_qwk'],\n        'val_qwk': result['val_qwk'],\n        'optimized_qwk': result['optimized_qwk'],\n        'submission': result['submission']\n    })\n\n# Print results\nresults_df = pd.DataFrame(results)\nprint(results_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:45:47.554761Z","iopub.execute_input":"2024-12-09T23:45:47.555214Z","iopub.status.idle":"2024-12-09T23:46:57.200639Z","shell.execute_reply.started":"2024-12-09T23:45:47.555173Z","shell.execute_reply":"2024-12-09T23:46:57.199515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(results_df['submission'][0]['sii'].value_counts())\nprint(results_df['submission'][1]['sii'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:47:48.303841Z","iopub.execute_input":"2024-12-09T23:47:48.304261Z","iopub.status.idle":"2024-12-09T23:47:48.312601Z","shell.execute_reply.started":"2024-12-09T23:47:48.304221Z","shell.execute_reply":"2024-12-09T23:47:48.311470Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### SVM","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import SVR\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.pipeline import Pipeline\n\n# Create an imputer\nimputer = SimpleImputer(strategy='mean')\n\n# Defining SVM parameters\nsvm_params = {\n    'kernel': 'rbf',  # Radial Basis Function kernel\n    'C': 10.0,         # Regularization parameter\n    'epsilon': 0.5,   # Epsilon in the epsilon-SVR model\n    'gamma': 'scale' # Kernel coefficient\n}\n\n# Create SVM model instance\nsvm_pipeline = Pipeline([\n    ('imputer', imputer),\n    ('rf', SVR(**svm_params))\n])\n\n\n# Train the SVM model\nSubmission_svm, trained_svm = TrainML(svm_pipeline, test)\n\n# Save submission\n# Submission.to_csv('svm_submission.csv', index=False)\nprint(Submission_svm['sii'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:47:55.864425Z","iopub.execute_input":"2024-12-09T23:47:55.864823Z","iopub.status.idle":"2024-12-09T23:47:59.382046Z","shell.execute_reply.started":"2024-12-09T23:47:55.864784Z","shell.execute_reply":"2024-12-09T23:47:59.381041Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Stacking","metadata":{}},{"cell_type":"code","source":"\nfrom sklearn.linear_model import ElasticNet\n# from pyswarm import pso\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neural_network import MLPClassifier,MLPRegressor\nfrom sklearn.linear_model import Ridge\nfrom sklearn.ensemble import StackingClassifier, StackingRegressor\nfrom sklearn.linear_model import LinearRegression\n\n# Create a pipeline for Linear Regression\nlinear_pipeline = Pipeline([\n    ('imputer', imputer),\n    ('lr', LinearRegression())\n])\n\n# Proposed Stacking model\n# Define base models\nParams = {'learning_rate': 0.02884249148676999, \n          'max_depth': 15, \n          'num_leaves': 470,\n          'min_data_in_leaf': 14,\n           \n          'feature_fraction': 0.7987976913702801, \n          'bagging_fraction': 0.7602261703576205, \n          'bagging_freq': 4, \n           'lambda_l1': 4.735462555910575, \n          'lambda_l2': 4.735028557007343e-06\n         }\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=1, n_estimators=200)\n\n\nrf_pipeline = Pipeline([\n    ('imputer', imputer),\n    ('rf', RandomForestRegressor(**rf_params, random_state=SEED))\n])\n\nmlp_pipeline = Pipeline([\n    ('imputer', imputer),\n    ('rf', MLPRegressor(hidden_layer_sizes=(int(140),), alpha=0.01, random_state=SEED) )\n])\n\n# Add feature importance-based feature selection\nfrom sklearn.feature_selection import SelectFromModel\n\nbest_base_models = [\n        ('linear', linear_pipeline),\n        ('xgb', xgb_model),\n        ('rf', rf_pipeline),\n        ('lgb', Light),\n        ('cat', cat_model),\n        # ('ann', mlp_pipeline),\n]\nmeta_learner = Ridge(alpha=0.6,random_state=SEED)\n\nbest_stacking_regressor = StackingRegressor(estimators=best_base_models, final_estimator=meta_learner, \n                                            cv=5, verbose=1)\n\nstacking_submission, _ = TrainML(best_stacking_regressor, test)\nprint(stacking_submission['sii'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:47:59.383736Z","iopub.execute_input":"2024-12-09T23:47:59.384121Z","iopub.status.idle":"2024-12-09T23:55:43.907116Z","shell.execute_reply.started":"2024-12-09T23:47:59.384073Z","shell.execute_reply":"2024-12-09T23:55:43.905673Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Ensemble Learning and Submission Preparation\n\nEnsemble Learning: The model uses a Voting Regressor, which combines the predictions from LightGBM, XGBoost, and CatBoost. This approach is beneficial as it leverages the strengths of multiple models, reducing overfitting and improving overall model performance.\n\nOut-of-Fold (OOF) Predictions: During cross-validation, out-of-fold predictions are generated for the training set, which helps in model evaluation without data leakage.\n\nKappa Optimizer: The Kappa Optimizer ensures that the predicted values are as close to the actual values as possible by adjusting the thresholds used to convert raw model outputs into class labels.\n\nTest Set Predictions: After the model is trained and thresholds are optimized, the test dataset is processed, and predictions are generated using the ensemble model. These predictions are converted into the appropriate format for submission.\n\nSubmission File Creation: The predictions are saved in a CSV file following the required format for submission (e.g., for a Kaggle competition), which includes columns like id and sii (Severity Impairment Index).\n\nFinal Results and Performance Metrics\nTrain and Validation Scores: After training across multiple folds, the mean Quadratic Weighted Kappa (QWK) score is calculated for both the training and validation datasets, providing an indicator of model performance.\n\nOptimized QWK Score: The final optimized QWK score after threshold tuning is displayed, showcasing the model's ability to predict the severity levels effectively.\n\nTest Predictions: The test set predictions are evaluated, and a breakdown of the predicted severity levels (None, Mild, Moderate, Severe) is shown, along with their respective counts.\n","metadata":{}},{"cell_type":"markdown","source":"### Voting (Fusion)","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import VotingRegressor\n# from pytorch_tabnet.tab_model import TabNetRegressor\n\n# Create model instances\nLight = LGBMRegressor(**Params, random_state=SEED, verbose=-1, n_estimators=300)\nXGB_Model = XGBRegressor(**xgb_params)\nCatBoost_Model = CatBoostRegressor(**cat_params)\n# TabNet_Model = TabNetWrapper(**TabNet_Params) # New\n\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ('rf_model', rf_pipeline),\n],weights=[4.0,4.0,5.0,4.0])\n\nSubmission_voting,_ = TrainML(voting_model, test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:55:43.909567Z","iopub.execute_input":"2024-12-09T23:55:43.910037Z","iopub.status.idle":"2024-12-09T23:57:14.596483Z","shell.execute_reply.started":"2024-12-09T23:55:43.909985Z","shell.execute_reply":"2024-12-09T23:57:14.595175Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission_voting.to_csv('submission.csv', index=False)\nprint(Submission_voting['sii'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:57:14.598435Z","iopub.execute_input":"2024-12-09T23:57:14.598777Z","iopub.status.idle":"2024-12-09T23:57:14.610668Z","shell.execute_reply.started":"2024-12-09T23:57:14.598742Z","shell.execute_reply":"2024-12-09T23:57:14.609525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission_voting","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T23:35:30.927076Z","iopub.status.idle":"2024-12-09T23:35:30.927599Z","shell.execute_reply.started":"2024-12-09T23:35:30.927331Z","shell.execute_reply":"2024-12-09T23:35:30.927360Z"}},"outputs":[],"execution_count":null}]}