{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:37.390012Z","iopub.execute_input":"2024-11-28T16:55:37.390414Z","iopub.status.idle":"2024-11-28T16:55:39.149957Z","shell.execute_reply.started":"2024-11-28T16:55:37.390374Z","shell.execute_reply":"2024-11-28T16:55:39.148877Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Read train and test csv files","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\ntrain","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:39.152237Z","iopub.execute_input":"2024-11-28T16:55:39.152688Z","iopub.status.idle":"2024-11-28T16:55:39.302672Z","shell.execute_reply.started":"2024-11-28T16:55:39.152643Z","shell.execute_reply":"2024-11-28T16:55:39.301155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:39.304466Z","iopub.execute_input":"2024-11-28T16:55:39.304883Z","iopub.status.idle":"2024-11-28T16:55:39.343561Z","shell.execute_reply.started":"2024-11-28T16:55:39.304845Z","shell.execute_reply":"2024-11-28T16:55:39.342416Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Retain only test columns in train","metadata":{}},{"cell_type":"code","source":"common_columns = list(test.columns.intersection(train.columns)) + ['sii']\ntrain = train[common_columns]\ntrain","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:39.345989Z","iopub.execute_input":"2024-11-28T16:55:39.346488Z","iopub.status.idle":"2024-11-28T16:55:39.390411Z","shell.execute_reply.started":"2024-11-28T16:55:39.346449Z","shell.execute_reply":"2024-11-28T16:55:39.389040Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Drop rows where sii is na","metadata":{}},{"cell_type":"code","source":"train.dropna(subset=['sii'], inplace=True)\nassert not train['sii'].isna().any()\ntrain","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:39.392110Z","iopub.execute_input":"2024-11-28T16:55:39.392563Z","iopub.status.idle":"2024-11-28T16:55:39.431406Z","shell.execute_reply.started":"2024-11-28T16:55:39.392504Z","shell.execute_reply":"2024-11-28T16:55:39.430109Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Change Datatype of Columns","metadata":{}},{"cell_type":"code","source":"from sklearn.base import BaseEstimator, TransformerMixin\nclass ObjectTypeConverter(BaseEstimator, TransformerMixin):\n    def __init__(self, target_dtype):\n        self.target_dtype = target_dtype\n    \n    # The fit method is not necessary, as there is no fitting required for type conversion\n    def fit(self, X, y=None):\n        return self\n    \n    # Transform method will convert object columns to the target data type\n    def transform(self, X, y=None):\n        X_copy = X.copy()\n        X_copy[X_copy.select_dtypes(include='object').columns] = X_copy.select_dtypes(include='object').astype(self.target_dtype)\n        return X_copy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:39.432813Z","iopub.execute_input":"2024-11-28T16:55:39.433154Z","iopub.status.idle":"2024-11-28T16:55:41.189772Z","shell.execute_reply.started":"2024-11-28T16:55:39.433117Z","shell.execute_reply":"2024-11-28T16:55:41.188819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['sii'] = train['sii'].astype('int')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:41.191495Z","iopub.execute_input":"2024-11-28T16:55:41.192200Z","iopub.status.idle":"2024-11-28T16:55:41.200536Z","shell.execute_reply.started":"2024-11-28T16:55:41.192129Z","shell.execute_reply":"2024-11-28T16:55:41.199277Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Check if there are any columns that have more than a threshold of na values","metadata":{}},{"cell_type":"code","source":"threshold_of_nan_values_allowed_per_column = 0.8\nnan_percentage = train.isna().mean()\ncolumns_above_threshold_nan = nan_percentage[nan_percentage > threshold_of_nan_values_allowed_per_column].index.tolist()\ncolumns_above_threshold_nan","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:41.202148Z","iopub.execute_input":"2024-11-28T16:55:41.202609Z","iopub.status.idle":"2024-11-28T16:55:41.232844Z","shell.execute_reply.started":"2024-11-28T16:55:41.202558Z","shell.execute_reply":"2024-11-28T16:55:41.231492Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Remove Columns that have more than a threshold percentage of na values","metadata":{}},{"cell_type":"code","source":"# Custom transformer to drop specified columns\nclass ColumnDropperTransformer(BaseEstimator, TransformerMixin):\n    def __init__(self, columns_to_drop):\n        self.columns_to_drop = columns_to_drop\n    \n    def fit(self, X, y=None):\n        # No fitting required for dropping columns\n        return self\n    \n    def transform(self, X, y=None):\n        # Drop specified columns\n        X_copy = X.copy()\n        return X_copy.drop(columns=self.columns_to_drop)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:41.234057Z","iopub.execute_input":"2024-11-28T16:55:41.234541Z","iopub.status.idle":"2024-11-28T16:55:41.250299Z","shell.execute_reply.started":"2024-11-28T16:55:41.234490Z","shell.execute_reply":"2024-11-28T16:55:41.248816Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Impute NaN Values (Only for Random Forest)","metadata":{}},{"cell_type":"code","source":"# Custom transformer to impute NaN columns\nclass ColumnImputer(BaseEstimator, TransformerMixin):\n    def __init__(self,\n                 categorical_column_impute_strategy,\n                 numerical_column_impute_strategy,\n                 numerical_constant_value,\n                 categorical_constant_value):\n        self.categorical_column_impute_strategy = categorical_column_impute_strategy\n        self.numerical_column_impute_strategy = numerical_column_impute_strategy\n        self.numerical_constant_value = numerical_constant_value\n        self.categorical_constant_value = categorical_constant_value\n        \n    def fit(self, X, y=None):\n        \n        X_copy = X.copy()\n        numerical_cols = X_copy.select_dtypes(include='number')\n        categorical_cols = X_copy.select_dtypes(include=['object', 'category'])\n\n        if self.numerical_column_impute_strategy == 'median':\n            self.numerical_columns_fillna=numerical_cols.median().to_dict()\n        elif self.numerical_column_impute_strategy == 'constant':\n            self.numerical_columns_fillna={col: self.numerical_constant_value for col in numerical_cols.columns}\n\n        if self.categorical_column_impute_strategy == 'mode':\n            self.categorical_columns_fillna=categorical_cols.mode().to_dict()\n        elif self.categorical_column_impute_strategy == 'constant':\n            self.categorical_columns_fillna={col: self.categorical_constant_value for col in categorical_cols.columns}\n        \n        return self\n    \n    def transform(self, X, y=None):\n        X_copy = X.copy()\n        numerical_data = X_copy.select_dtypes(include='number')\n        categorical_data = X_copy.select_dtypes(include=['object', 'category'])\n        if self.categorical_column_impute_strategy == 'constant':\n            for col in list(categorical_data.columns):\n                categorical_data[col] = categorical_data[col].cat.add_categories(self.categorical_constant_value)\n        numerical_data_imputed = numerical_data.apply(lambda col: col.fillna(self.numerical_columns_fillna[col.name]))\n        categorical_data_imputed = categorical_data.apply(lambda col: col.fillna(self.categorical_columns_fillna[col.name]))\n            \n        X_copy = pd.concat([numerical_data_imputed, categorical_data_imputed], axis=1)\n        return X_copy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:41.255410Z","iopub.execute_input":"2024-11-28T16:55:41.255818Z","iopub.status.idle":"2024-11-28T16:55:41.277348Z","shell.execute_reply.started":"2024-11-28T16:55:41.255783Z","shell.execute_reply":"2024-11-28T16:55:41.275834Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# One Hot encode categorical columns (Only For Random Forest)","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n# Custom transformer to one-hot encode categorical columns\nclass OneHotEncodeCategorical(BaseEstimator, TransformerMixin):\n    def __init__(self):\n        self.encoder = OneHotEncoder(sparse_output=False)\n    def fit(self, X, y=None):\n        categorical_columns=X.select_dtypes(include=['object', 'category']).columns.tolist()\n        \n        self.encoder.fit(X[categorical_columns])\n        return self\n    \n    def transform(self, X, y=None):\n        X_copy = X.copy()\n        categorical_columns=X_copy.select_dtypes(include=['object', 'category']).columns.tolist()\n        indices = list(X_copy.index)\n        X_copy_numerical = X_copy.drop(columns=categorical_columns,axis=1)\n        X_copy_categorical_one_hot = pd.DataFrame(self.encoder.transform(X_copy[categorical_columns]),\n                                                  columns=self.encoder.get_feature_names_out(categorical_columns),\n                                                  index=indices)\n        X_copy = pd.concat([X_copy_numerical,X_copy_categorical_one_hot],axis=1)\n        return X_copy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:41.278485Z","iopub.execute_input":"2024-11-28T16:55:41.278863Z","iopub.status.idle":"2024-11-28T16:55:41.355463Z","shell.execute_reply.started":"2024-11-28T16:55:41.278819Z","shell.execute_reply":"2024-11-28T16:55:41.354235Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Separate target from predictors","metadata":{}},{"cell_type":"code","source":"y = train.sii\nX = train.drop(['sii'], axis=1)\nX","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:41.357242Z","iopub.execute_input":"2024-11-28T16:55:41.357760Z","iopub.status.idle":"2024-11-28T16:55:41.406508Z","shell.execute_reply.started":"2024-11-28T16:55:41.357695Z","shell.execute_reply":"2024-11-28T16:55:41.405100Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Define Evaluation Metric","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import cohen_kappa_score,make_scorer\nkappa_scorer = make_scorer(cohen_kappa_score,weights='quadratic')\n\ndef generate_labels_from_ordinal_probabilities(y):\n    classes = np.array([0,1,2,3])\n    p_0 = 1-y[:,0]\n    p_1 = y[:,0]-y[:,1]\n    p_2 = y[:,1]-y[:,2]\n    p_3 = y[:,2]\n    y = np.vstack((p_0,p_1,p_2,p_3)).T\n    y = classes[np.argmax(y,axis=1).reshape(-1)]\n    return y\n\ndef kappa_ordinal_input(y_true_ordinal,y_pred):\n    y_true_ordinal = np.reshape(y_true_ordinal,(-1,3))\n    y_pred = np.reshape(y_pred,(-1,3))\n    \n    y_pred_labels = generate_labels_from_ordinal_probabilities(y_pred)\n    \n    # y_pred_labels = classes[np.argmax(y_pred < 0.5, axis=1).reshape(-1)]\n    y_true_labels =  generate_labels_from_ordinal_probabilities(y_true_ordinal)\n    return cohen_kappa_score(y_true_labels,y_pred_labels,weights='quadratic')\n    \ndef kappa_ordinal_input_custom_obj(y_true_ordinal,y_pred):\n    y_true_ordinal = np.reshape(y_true_ordinal,(-1,3))\n    y_pred = np.reshape(y_pred,(-1,3))\n    y_pred = expit(y_pred)\n    y_pred_labels = generate_labels_from_ordinal_probabilities(y_pred)\n    \n    # y_pred_labels = classes[np.argmax(y_pred < 0.5, axis=1).reshape(-1)]\n    y_true_labels =  generate_labels_from_ordinal_probabilities(y_true_ordinal)\n    return cohen_kappa_score(y_true_labels,y_pred_labels,weights='quadratic')\n\nkappa_ordinal_input_scorer = make_scorer(kappa_ordinal_input)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:41.408262Z","iopub.execute_input":"2024-11-28T16:55:41.408747Z","iopub.status.idle":"2024-11-28T16:55:41.587905Z","shell.execute_reply.started":"2024-11-28T16:55:41.408698Z","shell.execute_reply":"2024-11-28T16:55:41.586780Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Split into train and validation for calculating ensemble metrics","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Generate X_valid and y_valid for checking metrics of the ensemble model\nX, X_valid, y, y_valid = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:41.589326Z","iopub.execute_input":"2024-11-28T16:55:41.589688Z","iopub.status.idle":"2024-11-28T16:55:41.623167Z","shell.execute_reply.started":"2024-11-28T16:55:41.589640Z","shell.execute_reply":"2024-11-28T16:55:41.622047Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Cleaning","metadata":{}},{"cell_type":"code","source":"# from sklearn.base import BaseEstimator, TransformerMixin\n# class ProcessOutliersDataCleaning(BaseEstimator, TransformerMixin):\n#     def __init__(self):\n#       pass    \n#     def fit(self, X_, y=None):\n#         return self\n    \n#     def transform(self, X_, y=None):\n#         X_copy = X_.copy()\n#         # CGAS-CGAS_Score greater than 100 is not possible. Replace it with np.nan\n#         X_copy.loc[X_copy['CGAS-CGAS_Score'] > 100, 'CGAS-CGAS_Score'] = np.nan\n\n#         # Some physical weight/Height/BMI columns are 0. Set them to NaN and let XGBoost/LGBM handle them natively\n#         X_copy.loc[X_copy['Physical-BMI'] == 0, 'Physical-BMI'] = np.nan\n#         X_copy.loc[X_copy['Physical-Height'] == 0, 'Physical-Height'] = np.nan\n#         X_copy.loc[X_copy['Physical-Weight'] == 0, 'Physical-Weight'] = np.nan\n#         # If systolic BP is less than diastolic BP, the values were probably swapped. So swap them back\n#         mask = X_copy['Physical-Systolic_BP'] < X_copy['Physical-Diastolic_BP']\n#         X_copy.loc[mask, ['Physical-Systolic_BP', 'Physical-Diastolic_BP']] = X_copy.loc[mask, ['Physical-Diastolic_BP', 'Physical-Systolic_BP']].values\n\n#         X_copy.loc[X_copy['Physical-Systolic_BP']>180,'Physical-Systolic_BP']=np.nan\n#         X_copy.loc[X_copy['Physical-Systolic_BP']<60,'Physical-Systolic_BP']=np.nan\n        \n#         X_copy.loc[X_copy['Physical-Diastolic_BP']>140,'Physical-Diastolic_BP']=np.nan\n#         X_copy.loc[X_copy['Physical-Diastolic_BP']<40,'Physical-Diastolic_BP']=np.nan\n        \n        \n#         # Club info from Fitness_Endurance-Time_Mins and Fitness_Endurance-Time_Sec \n#         X_copy['Fitness_Endurance_Total_Time'] = X_copy['Fitness_Endurance-Time_Mins']*60 + X_copy['Fitness_Endurance-Time_Sec']\n#         X_copy= X_copy.drop(['Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec'],axis=1)\n\n        \n#         # FGC_FGC_GSND_zone_median = X.groupby('FGC-FGC_GSND_Zone')['FGC-FGC_GSND'].median() # Get FGC-FGC_GSND, FGC-FGC_GSND_Zone wise\n#         # FGC_FGC_GSND_median = X['FGC-FGC_GSND'].median()\n#         # FGC_FGC_GSND_zone_median_mapping = FGC_FGC_GSND_zone_median.to_dict()  \n#         # FGC_FGC_GSND_zone_median_mapping[np.nan]=FGC_FGC_GSND_median # To handle the case where FGC-FGC_GSND_Zone in NaN\n#         # X_copy.loc[X_copy['FGC-FGC_GSND'] > 75, 'FGC-FGC_GSND'] = X_copy['FGC-FGC_GSND_Zone'].map(FGC_FGC_GSND_zone_median)\n\n#         # FGC_FGC_GSD_zone_median = X.groupby('FGC-FGC_GSD_Zone')['FGC-FGC_GSD'].median() # Get FGC-FGC_GSD, FGC-FGC_GSD_Zone wise\n#         # FGC_FGC_GSD_median = X['FGC-FGC_GSD'].median()\n#         # FGC_FGC_GSD_zone_median_mapping = FGC_FGC_GSD_zone_median.to_dict()  \n#         # FGC_FGC_GSD_zone_median_mapping[np.nan]=FGC_FGC_GSD_median # To handle the case where FGC-FGC_GSD_Zone in NaN\n#         # X_copy.loc[X_copy['FGC-FGC_GSD'] > 75, 'FGC-FGC_GSD'] = X_copy['FGC-FGC_GSD_Zone'].map(FGC_FGC_GSD_zone_median)\n\n#         # BIA_BIA_BMC_median = X['BIA-BIA_BMC'].median()\n#         # X_copy['BIA-BIA_BMC'] = X_copy['BIA-BIA_BMC'].apply(lambda x: BIA_BIA_BMC_median if x > 1500 else x)\n\n#         # BIA_BIA_BMR_median = X['BIA-BIA_BMR'].median()\n#         # X_copy['BIA-BIA_BMR'] = X_copy['BIA-BIA_BMR'].apply(lambda x: BIA_BIA_BMR_median if x > 8000 else x)\n\n#         # BIA_BIA_DEE_median = X['BIA-BIA_DEE'].median()\n#         # X_copy['BIA-BIA_DEE'] = X_copy['BIA-BIA_DEE'].apply(lambda x: BIA_BIA_DEE_median if x > 8000 else x)\n\n#         # BIA_BIA_ECW_median = X['BIA-BIA_ECW'].median()\n#         # X_copy['BIA-BIA_ECW'] = X_copy['BIA-BIA_ECW'].apply(lambda x: BIA_BIA_ECW_median if x > 500 else x)\n\n#         # BIA_BIA_FFM_median = X['BIA-BIA_FFM'].median()\n#         # X_copy['BIA-BIA_FFM'] = X_copy['BIA-BIA_FFM'].apply(lambda x: BIA_BIA_FFM_median if x > 500 else x)\n        \n#         return X_copy\n# # ProcessOutliers().transform(train).loc[[59,3955,2039]]['FGC-FGC_GSND']\n# # train.plot(x='FGC-FGC_GSND', y='FGC-FGC_GSND_Zone', kind='scatter')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:41.625250Z","iopub.execute_input":"2024-11-28T16:55:41.625739Z","iopub.status.idle":"2024-11-28T16:55:41.634580Z","shell.execute_reply.started":"2024-11-28T16:55:41.625686Z","shell.execute_reply":"2024-11-28T16:55:41.633043Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Sample Weights and Class Weights","metadata":{}},{"cell_type":"code","source":"from sklearn.utils.class_weight import compute_sample_weight,compute_class_weight\nsii_ = list(y)\nsample_weights = compute_sample_weight(class_weight='balanced', y=sii_)\nclass_weights = compute_class_weight(class_weight='balanced', classes=np.unique(sii_), y=sii_)\nclass_weights = dict(zip(np.unique(sii_), class_weights))\nsample_weights","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:41.636210Z","iopub.execute_input":"2024-11-28T16:55:41.636710Z","iopub.status.idle":"2024-11-28T16:55:41.669268Z","shell.execute_reply.started":"2024-11-28T16:55:41.636650Z","shell.execute_reply":"2024-11-28T16:55:41.667693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_weights","metadata":{"execution":{"iopub.status.busy":"2024-11-28T16:55:41.670884Z","iopub.execute_input":"2024-11-28T16:55:41.671306Z","iopub.status.idle":"2024-11-28T16:55:41.683770Z","shell.execute_reply.started":"2024-11-28T16:55:41.671270Z","shell.execute_reply":"2024-11-28T16:55:41.682558Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Ordinal Encoding","metadata":{}},{"cell_type":"code","source":"# Define the mapping\nencoding_map = {\n    0: np.array([0, 0, 0]),\n    1: np.array([1, 0, 0]),\n    2: np.array([1, 1, 0]),\n    3: np.array([1, 1, 1])\n}\ny_ordinal = y.map(encoding_map)\ny_valid_ordinal = y_valid.map(encoding_map)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:41.685749Z","iopub.execute_input":"2024-11-28T16:55:41.686145Z","iopub.status.idle":"2024-11-28T16:55:41.705262Z","shell.execute_reply.started":"2024-11-28T16:55:41.686111Z","shell.execute_reply":"2024-11-28T16:55:41.703830Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Custom Objective","metadata":{}},{"cell_type":"code","source":"from collections import Counter\ndef focal_obj_wrapper(gamma = 0,single_output=False):\n    def focal_obj(y_true,z,sample_weight=None):\n        if (single_output == False):\n            y_true = np.reshape(y_true,(-1,3))\n\n        if sample_weight is None:\n            row_counts = Counter(map(tuple, y_true))\n            sample_weight = np.array([len(y_true)/(row_counts[tuple(row)]*4) for row in y_true])\n            sample_weight = sample_weight.reshape(-1,1)\n            \n        grad = -gamma*y_true*(1 - 1/(1 + np.exp(-z)))**gamma*np.exp(-z)*np.log(1/(1 + np.exp(-z)))/((1 - 1/(1 + np.exp(-z)))*(1 + np.exp(-z))**2) + gamma*(1 - y_true)*(1/(1 + np.exp(-z)))**gamma*np.exp(-z)*np.log(1 - 1/(1 + np.exp(-z)))/(1 + np.exp(-z)) + y_true*(1 - 1/(1 + np.exp(-z)))**gamma*np.exp(-z)/(1 + np.exp(-z)) - (1 - y_true)*(1/(1 + np.exp(-z)))**gamma*np.exp(-z)/((1 - 1/(1 + np.exp(-z)))*(1 + np.exp(-z))**2)\n        hess = gamma**2*y_true*(1 - 1/(1 + np.exp(-z)))**gamma*np.exp(-2*z)*np.log(1/(1 + np.exp(-z)))/((1 - 1/(1 + np.exp(-z)))**2*(1 + np.exp(-z))**4) + gamma**2*(1 - y_true)*(1/(1 + np.exp(-z)))**gamma*np.exp(-2*z)*np.log(1 - 1/(1 + np.exp(-z)))/(1 + np.exp(-z))**2 + gamma*y_true*(1 - 1/(1 + np.exp(-z)))**gamma*np.exp(-z)*np.log(1/(1 + np.exp(-z)))/((1 - 1/(1 + np.exp(-z)))*(1 + np.exp(-z))**2) - 2*gamma*y_true*(1 - 1/(1 + np.exp(-z)))**gamma*np.exp(-2*z)*np.log(1/(1 + np.exp(-z)))/((1 - 1/(1 + np.exp(-z)))*(1 + np.exp(-z))**3) - 2*gamma*y_true*(1 - 1/(1 + np.exp(-z)))**gamma*np.exp(-2*z)/((1 - 1/(1 + np.exp(-z)))*(1 + np.exp(-z))**3) - gamma*y_true*(1 - 1/(1 + np.exp(-z)))**gamma*np.exp(-2*z)*np.log(1/(1 + np.exp(-z)))/((1 - 1/(1 + np.exp(-z)))**2*(1 + np.exp(-z))**4) - gamma*(1 - y_true)*(1/(1 + np.exp(-z)))**gamma*np.exp(-z)*np.log(1 - 1/(1 + np.exp(-z)))/(1 + np.exp(-z)) + gamma*(1 - y_true)*(1/(1 + np.exp(-z)))**gamma*np.exp(-2*z)*np.log(1 - 1/(1 + np.exp(-z)))/(1 + np.exp(-z))**2 - 2*gamma*(1 - y_true)*(1/(1 + np.exp(-z)))**gamma*np.exp(-2*z)/((1 - 1/(1 + np.exp(-z)))*(1 + np.exp(-z))**3) - y_true*(1 - 1/(1 + np.exp(-z)))**gamma*np.exp(-z)/(1 + np.exp(-z)) + y_true*(1 - 1/(1 + np.exp(-z)))**gamma*np.exp(-2*z)/(1 + np.exp(-z))**2 + (1 - y_true)*(1/(1 + np.exp(-z)))**gamma*np.exp(-z)/((1 - 1/(1 + np.exp(-z)))*(1 + np.exp(-z))**2) - 2*(1 - y_true)*(1/(1 + np.exp(-z)))**gamma*np.exp(-2*z)/((1 - 1/(1 + np.exp(-z)))*(1 + np.exp(-z))**3) - (1 - y_true)*(1/(1 + np.exp(-z)))**gamma*np.exp(-2*z)/((1 - 1/(1 + np.exp(-z)))**2*(1 + np.exp(-z))**4)\n\n        grad = sample_weight * grad\n        hess = sample_weight * hess\n        \n        grad = -grad\n        hess = -hess\n        return grad.ravel(),hess.ravel()\n    return focal_obj\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T18:14:07.701853Z","iopub.execute_input":"2024-11-28T18:14:07.702330Z","iopub.status.idle":"2024-11-28T18:14:07.725848Z","shell.execute_reply.started":"2024-11-28T18:14:07.702291Z","shell.execute_reply":"2024-11-28T18:14:07.724250Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.model_selection import StratifiedKFold\n# np.set_printoptions(threshold=np.inf)\nclass CustomStratifiedKFold:\n    def __init__(self, n_splits=3, shuffle=False, random_state=None):\n        self.n_splits = n_splits\n        self.shuffle = shuffle\n        self.random_state = random_state\n        self.splits = []\n\n    def split(self, X, y_ordinal,groups=None):\n        # Convert one-hot encoded labels to a single label per sample\n        zero_locations = (y_ordinal == 0)\n        y_labels =np.argmax(y_ordinal==0,axis=1)\n        no_zero = ~zero_locations.any(axis=1)\n        y_labels[no_zero] = 3\n        # Create StratifiedKFold object\n        skf = StratifiedKFold(n_splits=self.n_splits, shuffle=self.shuffle, random_state=self.random_state)\n\n        # Generate train/test indices\n        for train_index, test_index in skf.split(X, y_labels):\n            self.splits.append((train_index, test_index))\n        return self.splits\n\n    def get_splits(self):\n        return self.splits\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:41.822327Z","iopub.execute_input":"2024-11-28T16:55:41.822708Z","iopub.status.idle":"2024-11-28T16:55:41.845847Z","shell.execute_reply.started":"2024-11-28T16:55:41.822671Z","shell.execute_reply":"2024-11-28T16:55:41.844462Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# XGBoost Model CV","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_predict,cross_val_score,cross_validate\nfrom xgboost import XGBClassifier,XGBRegressor,callback\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom scipy.special import expit\ndef xgboost_objective(trial):\n    # Suggest hyperparameters\n    early_stopping_rounds = trial.suggest_int('early_stopping_rounds',32,128)\n    gamma_focal_loss = trial.suggest_float('gamma_focal_loss', 0.1 ,2)\n    param = {\n        'verbosity': 0,\n        'booster': trial.suggest_categorical('booster', ['gbtree', 'dart']),\n        'lambda': trial.suggest_float('lambda', 1e-8, 1.0),\n        'alpha': trial.suggest_float('alpha', 1e-8, 1.0),\n        'learning_rate': trial.suggest_float('learning_rate', 0.001, 0.03),\n        'n_estimators': trial.suggest_int('n_estimators', 256, 512),\n        'max_depth': trial.suggest_int('max_depth', 3, 10),\n        'min_child_weight': trial.suggest_int('min_child_weight', 1, 10),\n        'gamma': trial.suggest_float('gamma', 0, 1),\n        'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n    }\n    early_stopping_cb = callback.EarlyStopping(\n        rounds=early_stopping_rounds,  # Number of rounds without improvement before stopping\n        save_best=True,  # Save the model with the best iteration\n        maximize=True,  # False for metrics like log loss (smaller is better)\n        data_name=\"validation_0\",\n        # metric_name=\"kappa_ordinal_input\",\n        metric_name=\"kappa_ordinal_input_custom_obj\",\n        min_delta=0.001\n    )\n    xgboost_pipeline = Pipeline(steps=[( 'object_type_converter', ObjectTypeConverter(target_dtype=\"category\")\n                                       ),\n                                       ('column_dropper_transformer',ColumnDropperTransformer(columns_above_threshold_nan+['id'])\n                                       ),\n                                       # ('process_outliers', ProcessOutliersDataCleaning()),\n                                       ('xgb_model', XGBRegressor(**param,\n                                                          random_state=0,\n                                                          enable_categorical=True,\n                                                          # objective='reg:logistic', \n                                                          # num_class=4,\n                                                          n_jobs=-1,\n                                                          objective=focal_obj_wrapper(gamma=0),\n                                                          eval_metric=kappa_ordinal_input_custom_obj,\n                                                          disable_default_eval_metric=True,\n                                                          callbacks = [early_stopping_cb]\n                                                         ))\n                                 ])\n    \n    \n    X_xgboost,y_xgboost=X,np.reshape(np.concatenate(y_ordinal.values),(-1,3))\n    preprocessing_pipeline = xgboost_pipeline[:-1]\n    X_valid_xgboost,y_valid_xgboost = preprocessing_pipeline.fit(X=X_valid).transform(X=X_valid),np.reshape(np.concatenate(y_valid_ordinal.values),(-1,3))\n    skf = CustomStratifiedKFold(n_splits=2, shuffle=False)\n\n    predictions = cross_val_predict(xgboost_pipeline, X_xgboost, y_xgboost,\n                                       cv = skf,\n                                       fit_params={\n                                           'xgb_model__eval_set': [(X_valid_xgboost,y_valid_xgboost)],\n                                           'xgb_model__verbose': False,\n                                           'xgb_model__sample_weight': sample_weights\n                                       }\n                                   )\n    ########### REMEMBER TO COMMENT THIS IF USING BUILT-IN OBJECTIVE ####################\n    predictions = expit(predictions)\n    #####################################################################################\n    predictions = generate_labels_from_ordinal_probabilities(predictions)\n    # skf = CustomStratifiedKFold(n_splits=2, shuffle=False)\n    # cv_results = cross_validate(xgboost_pipeline, X_xgboost, y_xgboost, \n    #                            cv=skf,\n    #                             scoring=kappa_ordinal_input_scorer,\n    #                             return_train_score=True,\n    #                             fit_params={\n    #                                 'xgb_model__eval_set': [(X_valid_xgboost,y_valid_xgboost)],\n    #                                 'xgb_model__verbose': False,\n    #                                 'xgb_model__sample_weight': sample_weights\n    #                             }\n    #                             )\n    print(\"Actual:\\n\",y.values)\n    print(\"Predictions:\\n\", predictions)\n    # print(\"CV_results:\\n\", cv_results)\n    cm = confusion_matrix(y.values, predictions)\n    print(\"Confusion Matrix:\\n\",cm)    \n    return cohen_kappa_score(y.values,predictions,weights=\"quadratic\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:41.848158Z","iopub.execute_input":"2024-11-28T16:55:41.848547Z","iopub.status.idle":"2024-11-28T16:55:42.580374Z","shell.execute_reply.started":"2024-11-28T16:55:41.848513Z","shell.execute_reply":"2024-11-28T16:55:42.579109Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# XGBoost Hyperparameter Search","metadata":{}},{"cell_type":"code","source":"import optuna\nstudy = optuna.create_study(direction='maximize')  \nstudy.optimize(xgboost_objective, n_trials=50)\nbest_trial = study.best_trial\nxgb_best_params = best_trial.params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T16:55:42.582006Z","iopub.execute_input":"2024-11-28T16:55:42.582695Z","iopub.status.idle":"2024-11-28T17:00:07.841776Z","shell.execute_reply.started":"2024-11-28T16:55:42.582641Z","shell.execute_reply":"2024-11-28T17:00:07.838992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_best_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:00:07.843049Z","iopub.execute_input":"2024-11-28T17:00:07.843505Z","iopub.status.idle":"2024-11-28T17:00:07.852959Z","shell.execute_reply.started":"2024-11-28T17:00:07.843451Z","shell.execute_reply":"2024-11-28T17:00:07.851632Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Fit and Save XGBoost PIpeline","metadata":{}},{"cell_type":"code","source":"import pickle\n\nearly_stopping_cb = callback.EarlyStopping(\n        rounds=xgb_best_params['early_stopping_rounds'],  # Number of rounds without improvement before stopping\n        save_best=True,  # Save the model with the best iteration\n        maximize=True,  # False for metrics like log loss (smaller is better)\n        data_name=\"validation_0\",\n        metric_name=\"kappa_ordinal_input_custom_obj\",\n        min_delta=0.001\n    )\nxgboost_pipeline = Pipeline(steps=[( 'object_type_converter', ObjectTypeConverter(target_dtype=\"category\")\n                                       ),\n                                       ('column_dropper_transformer',ColumnDropperTransformer(columns_above_threshold_nan+['id'])\n                                       ),\n                                       # ('process_outliers', ProcessOutliersDataCleaning()),\n                                       ('xgb_model', XGBRegressor(**xgb_best_params,\n                                                          random_state=0,\n                                                          enable_categorical=True,\n                                                          # objective='reg:logistic', \n                                                          # num_class=4,\n                                                          n_jobs=-1,\n                                                          objective=focal_obj_wrapper(gamma=xgb_best_params['gamma_focal_loss']),\n                                                          eval_metric=kappa_ordinal_input_custom_obj,\n                                                          disable_default_eval_metric=True,\n                                                          callbacks = [early_stopping_cb]\n                                                         ))\n                                 ])\n    \n    \nX_xgboost,y_xgboost=X,np.reshape(np.concatenate(y_ordinal.values),(-1,3))\npreprocessing_pipeline = xgboost_pipeline[:-1]\nX_valid_xgboost,y_valid_xgboost = preprocessing_pipeline.fit(X=X_valid).transform(X=X_valid),np.reshape(np.concatenate(y_valid_ordinal.values),(-1,3))\nxgboost_pipeline.fit(X_xgboost,y_xgboost,\n                     xgb_model__eval_set= [(X_valid_xgboost,y_valid_xgboost)],\n                     xgb_model__verbose= False,\n                     xgb_model__sample_weight= sample_weights\n                    )\ny_valid_pred_xgboost = xgboost_pipeline.predict(X_valid)\n########### REMEMBER TO COMMENT THIS IF USING BUILT-IN OBJECTIVE ####################\ny_valid_pred_xgboost = expit(y_valid_pred_xgboost)\n#####################################################################################\ny_valid_pred_xgboost = generate_labels_from_ordinal_probabilities(y_valid_pred_xgboost)\nprint(\"XGB Validation Kappa: \", cohen_kappa_score(y_valid,y_valid_pred_xgboost,weights=\"quadratic\"))\n# with open('xgboost_pipeline.pkl','wb') as file:\n#     pickle.dump(xgboost_pipeline, file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:13:19.441377Z","iopub.execute_input":"2024-11-28T17:13:19.441835Z","iopub.status.idle":"2024-11-28T17:13:59.790469Z","shell.execute_reply.started":"2024-11-28T17:13:19.441797Z","shell.execute_reply":"2024-11-28T17:13:59.788982Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Random Forest Model CV","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_predict,cross_val_score,cross_validate,StratifiedKFold\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import cohen_kappa_score\ndef random_forest_objective(trial):\n    # Suggest hyperparameters\n    categorical_column_impute_strategy = trial.suggest_categorical('categorical_column_impute_strategy', ['mode','constant'])\n    numerical_column_impute_strategy =  trial.suggest_categorical('numerical_column_impute_strategy', ['median','constant'])\n    # threshold = trial.suggest_categorical('threshold', [0.3, 0.4, 0.5, 0.6, 0.7])\n    params={\n        'n_estimators' : trial.suggest_int('n_estimators', 100, 1000),\n        'max_depth' : trial.suggest_int('max_depth', 2, 32),\n        'min_samples_split' : trial.suggest_int('min_samples_split', 2, 32),\n        'min_samples_leaf' : trial.suggest_int('min_samples_leaf', 1, 32),\n        'max_features' : trial.suggest_categorical('max_features', ['sqrt', 'log2']),\n        'min_weight_fraction_leaf' : trial.suggest_float('min_weight_fraction_leaf', 0, 0.5)\n        \n    }\n    \n    random_forest_pipeline = Pipeline(steps=[( 'object_type_converter', ObjectTypeConverter(target_dtype=\"category\")\n                                  ),\n                                  ('column_dropper_transformer',ColumnDropperTransformer(columns_above_threshold_nan+['id'])\n                                  ),\n                                  ('column_imputer', ColumnImputer(categorical_column_impute_strategy=categorical_column_impute_strategy,\n                                                                   numerical_column_impute_strategy=numerical_column_impute_strategy,\n                                                                   categorical_constant_value='Z',\n                                                                   numerical_constant_value=-10000)),\n                                  ('one_hot_encode_categorical', OneHotEncodeCategorical()),\n#                                  \n                                  \n                                  ('rf_model', RandomForestRegressor(**params,random_state=42,\n                                                                      # class_weight=class_weights\n                                                                     ))\n                                 ])\n    skf = CustomStratifiedKFold(n_splits=3, shuffle=False)\n    X_rf,y_rf=X,np.reshape(np.concatenate(y_ordinal.values),(-1,3))\n    # downsample_majority_class.fit(X,y).transform(X,y)\n    predictions = cross_val_predict(random_forest_pipeline, X_rf, y_rf,\n                                   cv = skf,\n                                   fit_params={\n                                        \"rf_model__sample_weight\": sample_weights\n                                   }\n                                   )\n\n    predictions = generate_labels_from_ordinal_probabilities(predictions)\n    # cv_results = cross_validate(random_forest_pipeline, X, y, \n    #                             cv=skf,\n    #                             scoring=kappa_scorer,\n    #                             return_train_score=True,\n    #                            )\n    print(\"Actual:\\n\", y.values)\n    print(\"Predictions:\\n\", predictions)\n    # print(\"CV_results:\\n\", cv_results)\n    cm = confusion_matrix(y.values, predictions)\n    print(\"Confusion Matrix:\\n\",cm)\n    return cohen_kappa_score(y.values,predictions,weights=\"quadratic\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:00:31.862068Z","iopub.execute_input":"2024-11-28T17:00:31.862415Z","iopub.status.idle":"2024-11-28T17:00:32.289515Z","shell.execute_reply.started":"2024-11-28T17:00:31.862380Z","shell.execute_reply":"2024-11-28T17:00:32.288156Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Random Forest Hyperparameter Search","metadata":{}},{"cell_type":"code","source":"import optuna\nstudy = optuna.create_study(direction='maximize')  \nstudy.optimize(random_forest_objective, n_trials=50)\nbest_trial = study.best_trial\nrandom_forest_best_params = best_trial.params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:00:32.291358Z","iopub.execute_input":"2024-11-28T17:00:32.291745Z","iopub.status.idle":"2024-11-28T17:00:34.954352Z","shell.execute_reply.started":"2024-11-28T17:00:32.291709Z","shell.execute_reply":"2024-11-28T17:00:34.952092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_forest_best_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:00:34.957524Z","iopub.execute_input":"2024-11-28T17:00:34.957966Z","iopub.status.idle":"2024-11-28T17:00:34.969058Z","shell.execute_reply.started":"2024-11-28T17:00:34.957916Z","shell.execute_reply":"2024-11-28T17:00:34.966430Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_column_impute_strategy=random_forest_best_params['categorical_column_impute_strategy']\nnumerical_column_impute_strategy=random_forest_best_params['numerical_column_impute_strategy']\n# random_forest_threshold = random_forest_best_params['threshold']\ndel random_forest_best_params['categorical_column_impute_strategy']\ndel random_forest_best_params['numerical_column_impute_strategy']\n# del random_forest_best_params['threshold']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:00:34.981306Z","iopub.execute_input":"2024-11-28T17:00:34.982785Z","iopub.status.idle":"2024-11-28T17:00:34.995643Z","shell.execute_reply.started":"2024-11-28T17:00:34.982717Z","shell.execute_reply":"2024-11-28T17:00:34.992373Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Fit and Save Random Forest Pipeline","metadata":{}},{"cell_type":"code","source":"random_forest_pipeline = Pipeline(steps=[( 'object_type_converter', ObjectTypeConverter(target_dtype=\"category\")\n                                         ),\n                                  ('column_dropper_transformer',ColumnDropperTransformer(columns_above_threshold_nan+['id'])\n                                  ),\n                                  ('column_imputer', ColumnImputer(categorical_column_impute_strategy=categorical_column_impute_strategy,\n                                                                   numerical_column_impute_strategy=numerical_column_impute_strategy,\n                                                                   categorical_constant_value='Z',\n                                                                   numerical_constant_value=-10000)),\n                                  ('one_hot_encode_categorical', OneHotEncodeCategorical()),\n#                                  \n                                  \n                                  ('rf_model', RandomForestRegressor(**random_forest_best_params,random_state=42,\n                                                                      # class_weight=class_weights\n                                                                     ))\n                                 ])\nX_rf,y_rf=X,np.reshape(np.concatenate(y_ordinal.values),(-1,3))\nrandom_forest_pipeline.fit(X_rf,y_rf,rf_model__sample_weight=sample_weights)\n\ny_valid_pred_rf = random_forest_pipeline.predict(X_valid)\n\ny_valid_pred_rf = generate_labels_from_ordinal_probabilities(y_valid_pred_rf)\nprint(\"Random Forest Validation Kappa: \", cohen_kappa_score(y_valid,y_valid_pred_rf,weights=\"quadratic\"))\nwith open('random_forest_pipeline.pkl','wb') as file:\n    pickle.dump(random_forest_pipeline, file)\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:00:34.998483Z","iopub.execute_input":"2024-11-28T17:00:34.999157Z","iopub.status.idle":"2024-11-28T17:00:35.997021Z","shell.execute_reply.started":"2024-11-28T17:00:34.999113Z","shell.execute_reply":"2024-11-28T17:00:35.994949Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LGBM Model CV","metadata":{}},{"cell_type":"code","source":"from lightgbm import LGBMRegressor,early_stopping,log_evaluation\nfrom sklearn.multioutput import MultiOutputRegressor\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\n\ndef lgbm_objective(trial):\n    # Suggest hyperparameters\n    early_stopping_rounds = trial.suggest_int(\"early_stopping_rounds\", 32, 128)\n    params = {\n        # \"objective\": \"multiclass\",\n        # \"num_class\": len(set(y)),  # Set the number of classes\n        # \"metric\": \"multi_logloss\",,\n        \"boosting_type\": \"gbdt\",\n        \"n_estimators\":  trial.suggest_int(\"n_estimators\", 128, 512),\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.001, 0.1),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 16, 512),\n        \"max_depth\": trial.suggest_int(\"max_depth\", 4, 16),\n        \"min_data_in_leaf\": trial.suggest_int(\"min_data_in_leaf\", 10, 100),\n        \"feature_fraction\": trial.suggest_float(\"feature_fraction\", 0.5, 1.0),\n        \"bagging_fraction\": trial.suggest_float(\"bagging_fraction\", 0.5, 1.0),\n        \"bagging_freq\": trial.suggest_int(\"bagging_freq\", 1, 7),\n        \"lambda_l1\": trial.suggest_float(\"lambda_l1\", 1e-1, 10.0),\n        \"lambda_l2\": trial.suggest_float(\"lambda_l2\", 1e-1, 10.0)\n    }\n    skf = CustomStratifiedKFold(n_splits=3, shuffle=False)\n    \n    X_lgbm,y_lgbm = X,np.reshape(np.concatenate(y_ordinal.values),(-1,3))\n    lgbm_pipeline = Pipeline(steps=[( 'object_type_converter', ObjectTypeConverter(target_dtype=\"category\")\n                                         ),\n                                  ('column_dropper_transformer',ColumnDropperTransformer(columns_above_threshold_nan+['id'])\n                                  ),\n                                  \n                                  ('lgbm_model', MultiOutputRegressor(LGBMRegressor(**params,\n                                                            random_state=42,\n                                                            # class_weight=class_weights,\n                                                            verbosity=-1,\n                                                            # objective=\"binary\",\n                                                            objective=focal_obj_wrapper(gamma=0,single_output=True)\n                                                            # metric='custom',\n                                                          )))\n                                 ])\n    preprocessing_pipeline = lgbm_pipeline[:-1]\n    X_valid_lgbm,y_valid_lgbm = preprocessing_pipeline.fit(X=X_valid).transform(X=X_valid),np.reshape(np.concatenate(y_valid_ordinal.values),(-1,3))\n    eval_sets = [(X_valid_lgbm, y_valid_lgbm[:, i]) for i in range(y_valid_lgbm.shape[1])]\n    predictions = cross_val_predict(lgbm_pipeline,X_lgbm, y_lgbm,\n                                   cv = skf,\n                                   fit_params={\n                                       'lgbm_model__eval_set': eval_sets,\n                                    'lgbm_model__sample_weight': sample_weights,\n                                    'lgbm_model__callbacks': [early_stopping(early_stopping_rounds,verbose=0),log_evaluation(period=0)],\n                                    }\n                                   )\n    ########### REMEMBER TO COMMENT THIS IF USING BUILT-IN OBJECTIVE ####################\n    predictions = expit(predictions)\n    #####################################################################################\n        # cv_results = cross_validate(lgbm_pipeline, X_lgbm, y_lgbm, \n    #                             cv=skf,\n    #                             scoring=kappa_scorer,\n    #                             return_train_score=True,\n    #                             fit_params={\n    #                                 lgbm_model__sample_weight: sample_weights\n    #                             }\n    #                            )\n\n    predictions = generate_labels_from_ordinal_probabilities(predictions)\n    print(\"Actual:\\n\", y.values)\n    print(\"Predictions:\\n\", predictions)\n    # print(\"CV_results:\\n\", cv_results)\n    cm = confusion_matrix(y.values, predictions)\n    print(\"Confusion Matrix:\\n\",cm)\n    return cohen_kappa_score(y.values,predictions,weights=\"quadratic\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T18:15:57.766420Z","iopub.execute_input":"2024-11-28T18:15:57.766875Z","iopub.status.idle":"2024-11-28T18:15:57.782411Z","shell.execute_reply.started":"2024-11-28T18:15:57.766835Z","shell.execute_reply":"2024-11-28T18:15:57.780807Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LGBM Hyperparameter Search","metadata":{}},{"cell_type":"code","source":"import optuna\nstudy = optuna.create_study(direction='maximize')  \nstudy.optimize(lgbm_objective, n_trials=50)\nbest_trial = study.best_trial\nlgbm_best_params = best_trial.params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T18:16:05.399658Z","iopub.execute_input":"2024-11-28T18:16:05.400071Z","iopub.status.idle":"2024-11-28T18:16:09.949218Z","shell.execute_reply.started":"2024-11-28T18:16:05.400033Z","shell.execute_reply":"2024-11-28T18:16:09.947752Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_best_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T18:14:33.514328Z","iopub.execute_input":"2024-11-28T18:14:33.514754Z","iopub.status.idle":"2024-11-28T18:14:33.523392Z","shell.execute_reply.started":"2024-11-28T18:14:33.514714Z","shell.execute_reply":"2024-11-28T18:14:33.521783Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Fit and Save LGBM  Pipeline","metadata":{}},{"cell_type":"code","source":"X_lgbm,y_lgbm = X,np.reshape(np.concatenate(y_ordinal.values),(-1,3))\nlgbm_pipeline = Pipeline(steps=[( 'object_type_converter', ObjectTypeConverter(target_dtype=\"category\")\n                                         ),\n                                  ('column_dropper_transformer',ColumnDropperTransformer(columns_above_threshold_nan+['id'])\n                                  ),\n                                  \n                                  ('lgbm_model', MultiOutputRegressor(LGBMRegressor(**lgbm_best_params,\n                                                            random_state=42,\n                                                            # class_weight=class_weights,\n                                                            verbose=-1,\n                                                            verbose_eval= False,\n                                                            objective=focal_obj_wrapper(gamma=0,single_output=True),\n                                                            # metric='custom',\n                                                          )))\n                                 ])\npreprocessing_pipeline = lgbm_pipeline[:-1]\nX_valid_lgbm,y_valid_lgbm = preprocessing_pipeline.fit(X=X_valid).transform(X=X_valid),np.reshape(np.concatenate(y_valid_ordinal.values),(-1,3))\neval_sets = [(X_valid_lgbm, y_valid_lgbm[:, i]) for i in range(y_valid_lgbm.shape[1])]\nlgbm_pipeline.fit(X_lgbm,y_lgbm,\n                  lgbm_model__sample_weight=sample_weights,\n                  lgbm_model__eval_set= eval_sets,\n                  # lgbm_model__eval_metric=  kappa_ordinal_input_lgbm,\n                  lgbm_model__callbacks= [early_stopping(lgbm_best_params['early_stopping_rounds'],verbose=0),log_evaluation(period=0)]\n                 )\n\ny_valid_pred_lgbm = lgbm_pipeline.predict(X_valid)\n########### REMEMBER TO COMMENT THIS IF USING BUILT-IN OBJECTIVE ####################\ny_valid_pred_lgbm = expit(y_valid_pred_lgbm)\n#####################################################################################\n\ny_valid_pred_lgbm = generate_labels_from_ordinal_probabilities(y_valid_pred_lgbm)\n\n\nprint(\"LGBM Validation Kappa: \", cohen_kappa_score(y_valid,y_valid_pred_lgbm,weights=\"quadratic\"))\n# with open('lgbm_pipeline.pkl','wb') as file:\n#     pickle.dump(lgbm_pipeline, file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T18:14:37.240370Z","iopub.execute_input":"2024-11-28T18:14:37.240771Z","iopub.status.idle":"2024-11-28T18:14:38.674765Z","shell.execute_reply.started":"2024-11-28T18:14:37.240735Z","shell.execute_reply":"2024-11-28T18:14:38.673378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.base import BaseEstimator, ClassifierMixin\n\nclass CustomVotingClassifier(BaseEstimator, ClassifierMixin):\n    def __init__(self, estimators):\n        self.estimators = estimators\n\n    def fit(self, X1, y1, X2, y2, X3, y3):\n        for (estimator, X, y) in zip(self.estimators, [X1, X2, X3], [y1, y2, y3]):\n            estimator.fit(X, y)\n\n    def predict(self, X):\n        ensemble_predictions = np.concatenate([estimator.predict(X) for estimator in self.estimators]).reshape(len(self.estimators),-1,3)\n        ########### REMEMBER TO COMMENT THIS IF USING BUILT-IN OBJECTIVE ####################\n        ensemble_predictions = expit(ensemble_predictions)\n        #####################################################################################\n        ensemble_predictions_labels = []\n        for predictions in ensemble_predictions:\n            ensemble_predictions_labels.append(generate_labels_from_ordinal_probabilities(predictions))\n        ensemble_predictions_labels = np.reshape(np.concatenate(ensemble_predictions_labels),(len(self.estimators),len(X)))\n        ensemble_predictions_labels = np.transpose(ensemble_predictions_labels)\n        ensemble_vote = []\n        for votes in ensemble_predictions_labels:\n            # No clear majority\n            if (len(np.unique(votes)) == len(votes)):\n                ensemble_vote.append(np.median(votes))\n            # Clear majority\n            else:\n                ensemble_vote.append(np.bincount(votes).argmax())\n                \n        return np.array(ensemble_vote)\n\n\n\n# Initialize your estimators\nestimators = [xgboost_pipeline, random_forest_pipeline, lgbm_pipeline]\n\n# Create custom voting classifier\nvoting_classifier = CustomVotingClassifier(estimators)\n\n# Fit the classifier with different datasets\n# voting_classifier.fit(X_xgboost,y_xgboost, X_rf,y_rf, X_lgbm,y_lgbm)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T18:14:50.573369Z","iopub.execute_input":"2024-11-28T18:14:50.573766Z","iopub.status.idle":"2024-11-28T18:14:50.587769Z","shell.execute_reply.started":"2024-11-28T18:14:50.573730Z","shell.execute_reply":"2024-11-28T18:14:50.586252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on the test data\ny_pred = voting_classifier.predict(X_valid)\nprint(\"Voting Classifier Validation Kappa: \", cohen_kappa_score(y_valid,y_pred,weights=\"quadratic\"))\n# with open('voting_classifier.pkl','wb') as file:\n#     pickle.dump(voting_classifier, file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T18:14:55.046490Z","iopub.execute_input":"2024-11-28T18:14:55.047585Z","iopub.status.idle":"2024-11-28T18:14:55.233967Z","shell.execute_reply.started":"2024-11-28T18:14:55.047536Z","shell.execute_reply":"2024-11-28T18:14:55.232544Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submit","metadata":{}},{"cell_type":"code","source":"test_ids = test.id\ntest_preds = voting_classifier.predict(test)\ntest_submission = pd.DataFrame({\n    'id': test_ids,\n    'sii': test_preds\n})\ntest_submission.to_csv(\"submission.csv\",index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-28T17:00:39.733541Z","iopub.status.idle":"2024-11-28T17:00:39.734349Z","shell.execute_reply.started":"2024-11-28T17:00:39.733834Z","shell.execute_reply":"2024-11-28T17:00:39.734023Z"}},"outputs":[],"execution_count":null}]}