{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":1655918,"sourceType":"datasetVersion","datasetId":980175}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import libraries\nimport optuna\nimport pandas as pd\nimport numpy as np\nimport warnings\nfrom sklearn import preprocessing\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split, StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.metrics import mean_squared_error\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom scipy.optimize import minimize\nfrom statsmodels.stats.outliers_influence import variance_inflation_factor\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import StackingRegressor\nfrom sklearn.ensemble import RandomForestRegressor\nimport tensorflow as tf\nfrom tensorflow.keras.layers import LSTM\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Input, Dense, Dropout, LSTM, Conv1D, Flatten, BatchNormalization\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.regularizers import l2\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.neural_network import MLPRegressor\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.linear_model import ElasticNet\n# from sklearn.ensemble import GradientBoostingRegressor\n# from sklearn.ensemble import ExtraTreesRegressor\nfrom sklearn.ensemble import HistGradientBoostingRegressor\n# from sklearn.svm import SVR\n# from sklearn.neighbors import KNeighborsRegressor\n# from sklearn.linear_model import LinearRegression","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-28T03:59:48.615385Z","iopub.execute_input":"2024-10-28T03:59:48.616089Z","iopub.status.idle":"2024-10-28T03:59:49.647372Z","shell.execute_reply.started":"2024-10-28T03:59:48.616022Z","shell.execute_reply":"2024-10-28T03:59:49.645396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', 1000)\npd.set_option('display.max_rows', 1000)","metadata":{"execution":{"iopub.status.busy":"2024-10-28T03:59:49.649932Z","iopub.execute_input":"2024-10-28T03:59:49.650510Z","iopub.status.idle":"2024-10-28T03:59:49.660704Z","shell.execute_reply.started":"2024-10-28T03:59:49.650446Z","shell.execute_reply":"2024-10-28T03:59:49.659255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load data\ntrain_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-28T03:59:49.662417Z","iopub.execute_input":"2024-10-28T03:59:49.662947Z","iopub.status.idle":"2024-10-28T03:59:49.727548Z","shell.execute_reply.started":"2024-10-28T03:59:49.662885Z","shell.execute_reply":"2024-10-28T03:59:49.726177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Shape of the data:\nprint(\"train_df :\", train_df.shape)\nprint(\"test_df :\", test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-28T03:59:49.731587Z","iopub.execute_input":"2024-10-28T03:59:49.732244Z","iopub.status.idle":"2024-10-28T03:59:49.739955Z","shell.execute_reply.started":"2024-10-28T03:59:49.732156Z","shell.execute_reply":"2024-10-28T03:59:49.738407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Custom functions\ndef process_file(filename, dirname):\n    data = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    data.drop('step', axis=1, inplace=True)\n    return data.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname):\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    stats, indexes = zip(*results)\n    data = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    data['id'] = indexes\n    return data","metadata":{"execution":{"iopub.status.busy":"2024-10-28T03:59:49.741551Z","iopub.execute_input":"2024-10-28T03:59:49.742169Z","iopub.status.idle":"2024-10-28T03:59:49.756594Z","shell.execute_reply.started":"2024-10-28T03:59:49.742117Z","shell.execute_reply":"2024-10-28T03:59:49.755174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load time series data\ntrain_parquet = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_parquet = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-10-28T03:59:49.758450Z","iopub.execute_input":"2024-10-28T03:59:49.758914Z","iopub.status.idle":"2024-10-28T04:02:15.441023Z","shell.execute_reply.started":"2024-10-28T03:59:49.758869Z","shell.execute_reply":"2024-10-28T04:02:15.439824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_features = test_df.drop(columns=['id']).columns\ntest_id = test_df['id']\ntrain_df = train_df.dropna(subset='sii')\ntrain_df = train_df.dropna(thresh=10, axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:15.442839Z","iopub.execute_input":"2024-10-28T04:02:15.443305Z","iopub.status.idle":"2024-10-28T04:02:15.464336Z","shell.execute_reply.started":"2024-10-28T04:02:15.443259Z","shell.execute_reply":"2024-10-28T04:02:15.462932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"# check missing values\ntrain_missing_values = train_df.isnull().sum().sort_values(ascending=False)\ntest_missing_values = test_df.isnull().sum().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:15.466354Z","iopub.execute_input":"2024-10-28T04:02:15.466891Z","iopub.status.idle":"2024-10-28T04:02:15.480409Z","shell.execute_reply.started":"2024-10-28T04:02:15.466835Z","shell.execute_reply":"2024-10-28T04:02:15.478827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Remove columns with too many misssing values\n# train_missing_ratio = train_df[base_features].isnull().mean()\n\n# # Remove columns with more than 70% missing values\n# threshold = 0.3\n# base_features = train_missing_ratio[train_missing_ratio < threshold].index","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:15.482281Z","iopub.execute_input":"2024-10-28T04:02:15.482781Z","iopub.status.idle":"2024-10-28T04:02:15.494713Z","shell.execute_reply.started":"2024-10-28T04:02:15.482724Z","shell.execute_reply":"2024-10-28T04:02:15.493333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df = pd.concat([train_df[base_features], train_df['sii']], axis=1)\n# test_df = test_df[base_features]","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:15.499770Z","iopub.execute_input":"2024-10-28T04:02:15.500284Z","iopub.status.idle":"2024-10-28T04:02:15.507535Z","shell.execute_reply.started":"2024-10-28T04:02:15.500227Z","shell.execute_reply":"2024-10-28T04:02:15.506325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Handling missing values\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nfrom sklearn.ensemble import ExtraTreesRegressor\nfrom sklearn.impute import KNNImputer\n\ndef categorical_handling_missing(df):\n    categorical_features = df.select_dtypes(include=['object']).columns.tolist()\n    df[categorical_features] = df[categorical_features].fillna('missing')\n    return df\n\ndef numerical_handling_missing(df):\n    # numerical_features = df.select_dtypes(include=['float64', 'int64']).columns.tolist()\n\n    # Median\n    col_median = [\n        'Physical-BMI',  'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n        'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Time_Sec', \n        'FGC-FGC_CU', 'FGC-FGC_GSND', 'FGC-FGC_GSD', 'FGC-FGC_PU', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n        'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat',\n        'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n        'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total', \n        # 'stat_0', 'stat_1', 'stat_2', 'stat_3', 'stat_4', 'stat_5', 'stat_6', 'stat_7', 'stat_8',\n        # 'stat_9', 'stat_10', 'stat_11', 'stat_12', 'stat_13', 'stat_14', 'stat_15', 'stat_16', 'stat_17', 'stat_18',\n        # 'stat_19', 'stat_20', 'stat_21', 'stat_23', 'stat_24', 'stat_25', 'stat_26', 'stat_27', 'stat_28', 'stat_30',\n        # 'stat_33', 'stat_35', 'stat_36', 'stat_37', 'stat_38', 'stat_40', 'stat_41', 'stat_42', 'stat_43',\n        # 'stat_47', 'stat_48', 'stat_49', 'stat_50', 'stat_51', 'stat_52', 'stat_54', 'stat_55', \n        # 'stat_59', 'stat_60', 'stat_61', 'stat_62', 'stat_63', 'stat_64', 'stat_66', 'stat_67', 'stat_68',\n        # 'stat_71', 'stat_72', 'stat_73', 'stat_74', 'stat_75', 'stat_76', 'stat_78', 'stat_79', 'stat_80',\n        # 'stat_83', 'stat_84', 'stat_85', 'stat_86', 'stat_87', 'stat_88', 'stat_90', 'stat_91', 'stat_92', 'stat_93', 'stat_95'\n    ]\n    df[col_median] = df[col_median].fillna(df[col_median].median())\n    \n    # KNN\n    col_knn = [\n        'FGC-FGC_SRL', 'FGC-FGC_SRR', 'FGC-FGC_TL', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T',  # 'stat_22'\n        'CGAS-CGAS_Score',  'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'FGC-FGC_CU_Zone',\n        'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone',\n        'FGC-FGC_TL_Zone', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_Frame_num',\n        'PreInt_EduHx-computerinternet_hoursday'\n    ]\n    imputer = KNNImputer(n_neighbors=5)\n    df[col_knn] = imputer.fit_transform(df[col_knn])\n    \n    # if 'sii' in df.columns:\n    #     df['sii'] = imputer.fit_transform(df[['sii']])\n    #     df['sii'] = df['sii'].round().astype(int)\n\n    # Other\n    # col_other = [\n    #     'CGAS-CGAS_Score',  'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'FGC-FGC_CU_Zone',\n    #     'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone',\n    #     'FGC-FGC_TL_Zone', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_Frame_num',\n    #     'PreInt_EduHx-computerinternet_hoursday', \n        # 'stat_29', 'stat_31', 'stat_32', 'stat_34', 'stat_39', 'stat_44', 'stat_45', 'stat_46',\n        # 'stat_53', 'stat_56', 'stat_57', 'stat_58', 'stat_65', 'stat_69', 'stat_70', 'stat_77',\n        # 'stat_81', 'stat_82', 'stat_89', 'stat_94'\n    # ]\n    \n    # df.loc[:, col_other] = df[col_other].fillna(-1)\n    # df[numerical_features] = df[numerical_features].fillna(df[numerical_features].median()) # mean?\n    return df\n    \ntrain_df = categorical_handling_missing(train_df)\ntest_df = categorical_handling_missing(test_df)\n\ntrain_df = numerical_handling_missing(train_df)\ntest_df = numerical_handling_missing(test_df)","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:15.509396Z","iopub.execute_input":"2024-10-28T04:02:15.509805Z","iopub.status.idle":"2024-10-28T04:02:16.853153Z","shell.execute_reply.started":"2024-10-28T04:02:15.509760Z","shell.execute_reply":"2024-10-28T04:02:16.851839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create new features\ndef create_new_features(df):\n    df['BMI_Age_sum'] = df['Physical-BMI'] + df['Basic_Demos-Age']\n    df['BMI_Age_mult'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Physical_Mean'] = df[['Physical-Height', 'Physical-Weight', 'Physical-BMI']].mean(axis=1)\n    df['Physical_Std'] = df[['Physical-Height', 'Physical-Weight', 'Physical-BMI']].std(axis=1)\n    \n    # df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    # df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    # df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    # df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    # df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    # df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    # df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    # df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    # df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    # df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    # df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    # df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    # df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    # df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    # df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    return df\n\ntrain_df = create_new_features(train_df)\ntest_df = create_new_features(test_df)","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:16.854906Z","iopub.execute_input":"2024-10-28T04:02:16.855365Z","iopub.status.idle":"2024-10-28T04:02:16.880082Z","shell.execute_reply.started":"2024-10-28T04:02:16.855320Z","shell.execute_reply":"2024-10-28T04:02:16.878564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_features = test_df.drop(columns=['id']).columns","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:16.881700Z","iopub.execute_input":"2024-10-28T04:02:16.882091Z","iopub.status.idle":"2024-10-28T04:02:16.891880Z","shell.execute_reply.started":"2024-10-28T04:02:16.882050Z","shell.execute_reply":"2024-10-28T04:02:16.890526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df[base_features] = train_df[base_features].replace([np.inf, -np.inf], np.nan)\n# train_df[base_features] = train_df[base_features].fillna(train_df[base_features].max())","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:16.893478Z","iopub.execute_input":"2024-10-28T04:02:16.893981Z","iopub.status.idle":"2024-10-28T04:02:16.906610Z","shell.execute_reply.started":"2024-10-28T04:02:16.893925Z","shell.execute_reply":"2024-10-28T04:02:16.905291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encoding data\nle = preprocessing.LabelEncoder()\nscaler = StandardScaler()\n\n# train data\ntrain_categorical_features = train_df.select_dtypes(include=['object']).columns.tolist()\n# categorical data\nfor col in train_categorical_features:\n    train_df[col] = le.fit_transform(train_df[col]).astype(int)\n# scaling\n# train_df_scaled = train_df[train_df[base_features] != -1]\n# train_df_scaled[base_features] = scaler.fit_transform(train_df_scaled[base_features])\n# train_df.update(train_df_scaled)\ntrain_df[base_features] = scaler.fit_transform(train_df[base_features])\n\n# test data\ntest_categorical_features = test_df.select_dtypes(include=['object']).columns.tolist()\n# categorical data\nfor col in test_categorical_features:\n    test_df[col] = le.fit_transform(test_df[col]).astype(int)\n# scaling\n# test_df_scaled = test_df[test_df[base_features] != -1]\n# test_df_scaled[base_features] = scaler.fit_transform(test_df_scaled[base_features])\n# test_df.update(test_df_scaled)\ntest_df[base_features] = scaler.fit_transform(test_df[base_features])","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:16.911386Z","iopub.execute_input":"2024-10-28T04:02:16.912108Z","iopub.status.idle":"2024-10-28T04:02:16.988615Z","shell.execute_reply.started":"2024-10-28T04:02:16.912060Z","shell.execute_reply":"2024-10-28T04:02:16.987374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# select the features with the strongest correlation with 'PCIAT-PCIAT_Total'\n# base_features_corr = base_features.tolist()  \n# base_features_corr.append('PCIAT-PCIAT_Total')\n# correlation_matrix = train_df[base_features_corr].corr()\n# target_corr = correlation_matrix['PCIAT-PCIAT_Total'].drop(labels=['PCIAT-PCIAT_Total'], errors='ignore')\n\n# threshold = 0.02 # 0.05\n# base_features = target_corr[abs(target_corr) > threshold].index.tolist()","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:16.990388Z","iopub.execute_input":"2024-10-28T04:02:16.990911Z","iopub.status.idle":"2024-10-28T04:02:17.000210Z","shell.execute_reply.started":"2024-10-28T04:02:16.990854Z","shell.execute_reply":"2024-10-28T04:02:16.998897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_data = pd.concat([train_df[base_features], train_df['sii']], axis=1)\n# test_data = test_df[base_features]","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:17.001969Z","iopub.execute_input":"2024-10-28T04:02:17.002491Z","iopub.status.idle":"2024-10-28T04:02:17.012985Z","shell.execute_reply.started":"2024-10-28T04:02:17.002432Z","shell.execute_reply":"2024-10-28T04:02:17.011657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# combine train data and test data\n# all_df = pd.concat([train_data, test_data], sort=False).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:17.014641Z","iopub.execute_input":"2024-10-28T04:02:17.015165Z","iopub.status.idle":"2024-10-28T04:02:17.030478Z","shell.execute_reply.started":"2024-10-28T04:02:17.015106Z","shell.execute_reply":"2024-10-28T04:02:17.029240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # checkt VIF\n# X_vif = all_df[base_features]\n\n# # Function to calculate VIF\n# def calc_vif(X):\n#     vif_data = pd.DataFrame()\n#     vif_data['feature'] = X.columns\n#     vif_data['VIF'] = [variance_inflation_factor(X.values, i) for i in range(X.shape[1])]\n#     return vif_data\n\n# # Initial VIF calculation\n# vif_df = calc_vif(X_vif)\n# vif_df","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:17.032149Z","iopub.execute_input":"2024-10-28T04:02:17.033102Z","iopub.status.idle":"2024-10-28T04:02:17.043672Z","shell.execute_reply.started":"2024-10-28T04:02:17.033023Z","shell.execute_reply":"2024-10-28T04:02:17.042400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Iteratively remove variables with VIF exceeding the threshold\n# threshold = 10  # Set threshold for removal, e.g., VIF > 10\n# while vif_df['VIF'].max() > threshold:\n#     # Identify and remove the variable with the highest VIF\n#     highest_vif = vif_df[vif_df['VIF'] == vif_df['VIF'].max()]['feature'].iloc[0]\n#     print(f'Removing {highest_vif} with VIF: {vif_df[\"VIF\"].max()}')\n#     X_vif = X_vif.drop(columns=[highest_vif])\n    \n#     # Recalculate VIF\n#     vif_df = calc_vif(X_vif)\n\n# vif_df","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:17.045294Z","iopub.execute_input":"2024-10-28T04:02:17.045839Z","iopub.status.idle":"2024-10-28T04:02:17.060748Z","shell.execute_reply.started":"2024-10-28T04:02:17.045777Z","shell.execute_reply":"2024-10-28T04:02:17.059505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# selected_feature = vif_df['feature']\n# all_df = pd.concat([all_df[selected_feature], all_df['sii']], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:17.062685Z","iopub.execute_input":"2024-10-28T04:02:17.063120Z","iopub.status.idle":"2024-10-28T04:02:17.074657Z","shell.execute_reply.started":"2024-10-28T04:02:17.063046Z","shell.execute_reply":"2024-10-28T04:02:17.073027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df_le = all_df[~all_df['sii'].isnull()]\n# test_df_le = all_df[all_df['sii'].isnull()].drop(columns=['sii'])","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:17.076572Z","iopub.execute_input":"2024-10-28T04:02:17.077108Z","iopub.status.idle":"2024-10-28T04:02:17.086869Z","shell.execute_reply.started":"2024-10-28T04:02:17.077051Z","shell.execute_reply":"2024-10-28T04:02:17.085719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# select the features with the strongest correlation with 'sii'\n# correlation_matrix = train_df_le.corr()\n# target_corr = correlation_matrix['sii'].drop('sii')\n\n# threshold = 0.01\n# high_corr_features = target_corr[abs(target_corr) > threshold].index.tolist()","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:17.088529Z","iopub.execute_input":"2024-10-28T04:02:17.089629Z","iopub.status.idle":"2024-10-28T04:02:17.097727Z","shell.execute_reply.started":"2024-10-28T04:02:17.089571Z","shell.execute_reply":"2024-10-28T04:02:17.096493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df_le =  pd.concat([train_df_le[high_corr_features], train_df_le['sii']], axis=1)\n# test_df_le = test_df_le[high_corr_features]","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:17.102078Z","iopub.execute_input":"2024-10-28T04:02:17.102689Z","iopub.status.idle":"2024-10-28T04:02:17.111174Z","shell.execute_reply.started":"2024-10-28T04:02:17.102644Z","shell.execute_reply":"2024-10-28T04:02:17.109885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_le = pd.concat([train_df[base_features], train_df['sii']], axis=1)\ntest_df_le = test_df.drop(columns=['id'])","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:17.112969Z","iopub.execute_input":"2024-10-28T04:02:17.113532Z","iopub.status.idle":"2024-10-28T04:02:17.137908Z","shell.execute_reply.started":"2024-10-28T04:02:17.113475Z","shell.execute_reply":"2024-10-28T04:02:17.136743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"X = train_df_le.drop(columns=['sii'])\ny = train_df_le['sii']","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:17.139436Z","iopub.execute_input":"2024-10-28T04:02:17.139855Z","iopub.status.idle":"2024-10-28T04:02:17.149676Z","shell.execute_reply.started":"2024-10-28T04:02:17.139803Z","shell.execute_reply":"2024-10-28T04:02:17.148283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model parameters for LightGBM\n\nSEED = 42\n# LGBM\nLGBM_Params = {\n    'learning_rate': 0.0399, # 0.0399\n    'max_depth': 14, # 12 -> 11\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.893,\n    'bagging_fraction': 0.784,\n    'bagging_freq': 4,\n    'lambda_l1': 3, # 10\n    'lambda_l2': 0.3, # 0.3\n    'random_state': SEED\n}\n\n# XGBoost parameters\nXGB_Params = {\n    'learning_rate': 0.03, # 0.03\n    'max_depth': 10, # 10\n    'n_estimators': 300, # 300\n    'subsample': 0.8, # 0.8\n    'colsample_bytree': 0.8, # 0.8\n    'gamma': 2,\n    'reg_alpha': 1,  # 1\n    'reg_lambda': 3, # 3\n    'random_state': SEED\n}\n\n# CatBoost\nCatBoost_Params = {\n    'learning_rate': 0.045, # 0.05\n    'depth': 8, # 6 -> 7\n    'iterations': 250, # 200\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10\n}\n\n# RandomForest\nRF_Params = {\n    'n_estimators': 700, \n    'max_depth': 20, \n    'min_samples_split': 10,\n    'min_samples_leaf': 8, \n    'max_features': 0.6,\n    'bootstrap': True,\n    'random_state': SEED\n}\n\n# Gradient Boost\n# GBR_Params = {\n#     'n_estimators': 210,\n#     'learning_rate': 0.05,\n#     'max_depth': 8,\n#     'min_samples_split': 4,\n#     'min_samples_leaf': 5,\n#     'max_features': 'sqrt',\n#     'subsample': 0.8,\n#     'random_state': SEED\n# }\n\n# Extra Trees\n# ETR_Params = {\n#     'n_estimators': 150,\n#     'max_depth': 9,\n#     'min_samples_split': 5,\n#     'min_samples_leaf': 2,\n#     'max_features': 'sqrt',\n#     'bootstrap': False,\n#     'random_state': SEED\n# }\n\n# HistGradientBoosting\nHGBR_Params = {\n    'learning_rate': 0.04,\n    'max_iter': 150,\n    'max_depth': 8,\n    'max_leaf_nodes': 40,\n    'min_samples_leaf': 25,\n    'l2_regularization': 0.3,\n    'random_state': SEED\n}","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:17.151540Z","iopub.execute_input":"2024-10-28T04:02:17.151954Z","iopub.status.idle":"2024-10-28T04:02:17.165162Z","shell.execute_reply.started":"2024-10-28T04:02:17.151910Z","shell.execute_reply":"2024-10-28T04:02:17.163937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class KerasNNRegressor(BaseEstimator, RegressorMixin):\n    def __init__(self, input_dim, epochs=100, batch_size=32, learning_rate=0.0001, dropout_rate=0.09, l2_reg=0.005, seed=SEED):\n        self.input_dim = input_dim\n        self.epochs = epochs\n        self.batch_size = batch_size\n        self.learning_rate = learning_rate\n        self.dropout_rate = dropout_rate\n        self.l2_reg = l2_reg\n        self.seed = seed\n        self.model = None\n\n    def _build_model(self):\n        tf.random.set_seed(self.seed)\n        model = Sequential()\n        \n        # 1st hidden layer with L2 regularization and dropout\n        model.add(Dense(256, input_dim=self.input_dim, activation='relu', kernel_regularizer=tf.keras.regularizers.l2(self.l2_reg)))\n        model.add(Dropout(self.dropout_rate))\n        \n        # 2nd hidden layer with L2 regularization and dropout\n        model.add(Dense(128, activation='relu', kernel_regularizer=tf.keras.regularizers.l2(self.l2_reg)))\n        model.add(Dropout(self.dropout_rate))\n        \n        # 3rd hidden layer with L2 regularization and dropout\n        model.add(Dense(64, activation='relu', kernel_regularizer=tf.keras.regularizers.l2(self.l2_reg)))\n        model.add(Dropout(self.dropout_rate))\n        \n        # 4th hidden layer with L2 regularization and dropout\n        model.add(Dense(32, activation='relu', kernel_regularizer=tf.keras.regularizers.l2(self.l2_reg)))\n        model.add(Dropout(self.dropout_rate))\n        \n        # Output layer for regression\n        model.add(Dense(1, activation='linear'))\n        \n        # Compile the model with Adam optimizer\n        model.compile(optimizer=Adam(learning_rate=self.learning_rate), loss='mean_squared_error')\n        \n        return model\n\n    def fit(self, X, y):\n        # Build the model\n        self.model = self._build_model()\n        \n        # Early stopping callback\n        early_stopping = EarlyStopping(monitor='val_loss', patience=15, restore_best_weights=True)\n        \n        # Fit the model with validation split and early stopping\n        self.model.fit(X, y, \n                       epochs=self.epochs, \n                       batch_size=self.batch_size, \n                       validation_split=0.2,\n                       callbacks=[early_stopping], \n                       verbose=0)\n        \n        return self\n\n    def predict(self, X):\n        return self.model.predict(X).squeeze()\n\n# Define the NN model with your specific input dimension\ninput_dim = X.shape[1]\nnn_model = KerasNNRegressor(input_dim=input_dim)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-10-28T04:02:17.173103Z","iopub.execute_input":"2024-10-28T04:02:17.173574Z","iopub.status.idle":"2024-10-28T04:02:17.191968Z","shell.execute_reply.started":"2024-10-28T04:02:17.173530Z","shell.execute_reply":"2024-10-28T04:02:17.190569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class FCNNRegressor(BaseEstimator, RegressorMixin):\n    def __init__(self, input_shape, epochs=150, batch_size=64, learning_rate=0.0001, dropout_rate=0.2, l2_reg=0.01, seed=SEED):\n        self.input_shape = input_shape\n        self.epochs = epochs\n        self.batch_size = batch_size\n        self.learning_rate = learning_rate\n        self.dropout_rate = dropout_rate\n        self.l2_reg = l2_reg\n        self.seed = seed\n        self.model = self._build_model()\n\n    def _build_model(self):\n        tf.random.set_seed(self.seed)\n        model = Sequential()\n        # Fully connected layers with dropout and batch normalization\n        model.add(Dense(128, activation='relu', input_shape=(self.input_shape[0],), kernel_regularizer=l2(self.l2_reg)))\n        model.add(BatchNormalization())\n        model.add(Dropout(self.dropout_rate))\n        model.add(Dense(64, activation='relu', kernel_regularizer=l2(self.l2_reg)))\n        model.add(BatchNormalization())\n        model.add(Dropout(self.dropout_rate))\n        model.add(Dense(32, activation='relu', kernel_regularizer=l2(self.l2_reg)))\n        model.add(BatchNormalization())\n        model.add(Dropout(self.dropout_rate))\n        model.add(Dense(1, activation='linear'))\n        \n        model.compile(optimizer=Adam(learning_rate=self.learning_rate), loss='mean_squared_error')\n        return model\n\n    def fit(self, X, y):\n        early_stopping = EarlyStopping(monitor='val_loss', patience=15, restore_best_weights=True)\n        self.model.fit(X, y, epochs=self.epochs, batch_size=self.batch_size, validation_split=0.2, callbacks=[early_stopping], verbose=0) \n        return self\n\n    def predict(self, X):\n        return self.model.predict(X).flatten()\n\n# Initialize the model with the updated input shape\ninput_shape = (X.shape[1],)\nfcnn_model = FCNNRegressor(input_shape=input_shape)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-10-28T04:02:17.193931Z","iopub.execute_input":"2024-10-28T04:02:17.194763Z","iopub.status.idle":"2024-10-28T04:02:17.391235Z","shell.execute_reply.started":"2024-10-28T04:02:17.194699Z","shell.execute_reply":"2024-10-28T04:02:17.389911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# MLP Regressor\n# mlp_model = MLPRegressor(hidden_layer_sizes=(100, 50, 25),   \n#                          activation='relu',              \n#                          solver='adam',                  \n#                          alpha=0.0001,                   \n#                          learning_rate='adaptive',       \n#                          learning_rate_init=0.0001,       \n#                          max_iter=500,                   \n#                          random_state=SEED,                \n#                          early_stopping=True,            \n#                          validation_fraction=0.2)        ","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-10-28T04:02:17.392743Z","iopub.execute_input":"2024-10-28T04:02:17.393210Z","iopub.status.idle":"2024-10-28T04:02:17.399184Z","shell.execute_reply.started":"2024-10-28T04:02:17.393149Z","shell.execute_reply":"2024-10-28T04:02:17.397827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # PyTorch Regressor model\n# class PyTorchRegressor(BaseEstimator, RegressorMixin):\n#     def __init__(self, input_dim, hidden_dim=64, epochs=100, batch_size=32, learning_rate=0.001, seed=SEED, device='cpu'):\n#         self.input_dim = input_dim\n#         self.hidden_dim = hidden_dim\n#         self.epochs = epochs\n#         self.batch_size = batch_size\n#         self.learning_rate = learning_rate\n#         self.seed = seed\n#         self.device = device\n#         self.model = self._build_model()\n#         self.criterion = nn.MSELoss()\n#         self.optimizer = optim.Adam(self.model.parameters(), lr=self.learning_rate, weight_decay=1e-5)\n\n#     def _build_model(self):\n#         torch.manual_seed(self.seed)\n#         model = nn.Sequential(\n#             nn.Linear(self.input_dim, self.hidden_dim),\n#             nn.ReLU(),\n#             nn.Dropout(p=0.4),\n#             nn.Linear(self.hidden_dim, self.hidden_dim),\n#             nn.ReLU(),\n#             nn.Dropout(p=0.4),\n#             nn.Linear(self.hidden_dim, 1)\n#         )\n#         return model.to(self.device)\n\n#     def fit(self, X, y):\n#         self.model.train()\n#         X_tensor = torch.tensor(X.values, dtype=torch.float32).to(self.device)\n#         y_tensor = torch.tensor(y.values, dtype=torch.float32).view(-1, 1).to(self.device)\n\n#         for epoch in range(self.epochs):\n#             self.optimizer.zero_grad()\n#             outputs = self.model(X_tensor)\n#             loss = self.criterion(outputs, y_tensor)\n#             loss.backward()\n#             self.optimizer.step()\n\n#             # Optional: Log the loss every 10 epochs\n#             if epoch % 10 == 0:\n#                 print(f\"Epoch [{epoch}/{self.epochs}], Loss: {loss.item():.4f}\")\n\n#         return self\n\n#     def predict(self, X):\n#         self.model.eval()\n#         with torch.no_grad():\n#             X_tensor = torch.tensor(X.values, dtype=torch.float32).to(self.device)\n#             outputs = self.model(X_tensor).squeeze().cpu().numpy()\n#         return outputs\n\n# pytorch_model = PyTorchRegressor(input_dim=input_dim, device='cuda' if torch.cuda.is_available() else 'cpu')\n","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-10-28T04:02:17.401525Z","iopub.execute_input":"2024-10-28T04:02:17.402066Z","iopub.status.idle":"2024-10-28T04:02:17.417806Z","shell.execute_reply.started":"2024-10-28T04:02:17.402005Z","shell.execute_reply":"2024-10-28T04:02:17.416468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PyTorch Regressor model\nclass PyTorchRegressor(BaseEstimator, RegressorMixin):\n    def __init__(self, input_dim, hidden_dim=64, epochs=100, batch_size=32, learning_rate=0.001, seed=SEED, device='cpu'):\n        self.input_dim = input_dim\n        self.hidden_dim = hidden_dim\n        self.epochs = epochs\n        self.batch_size = batch_size\n        self.learning_rate = learning_rate\n        self.seed = seed\n        self.device = device\n        self.model = self._build_model()\n        self.criterion = nn.MSELoss()\n        self.optimizer = optim.Adam(self.model.parameters(), lr=self.learning_rate, weight_decay=1e-5)\n\n    def _build_model(self):\n        torch.manual_seed(self.seed)\n        model = nn.Sequential(\n            nn.Linear(self.input_dim, self.hidden_dim),\n            nn.ReLU(),\n            nn.Dropout(p=0.4),\n            nn.Linear(self.hidden_dim, self.hidden_dim),\n            nn.ReLU(),\n            nn.Dropout(p=0.4),\n            nn.Linear(self.hidden_dim, 1)\n        )\n        return model.to(self.device)\n\n    def fit(self, X, y):\n        self.model.train()\n        X_tensor = torch.tensor(X.values if hasattr(X, \"values\") else X, dtype=torch.float32).to(self.device)\n        y_tensor = torch.tensor(y.values if hasattr(y, \"values\") else y, dtype=torch.float32).view(-1, 1).to(self.device)\n\n        for epoch in range(self.epochs):\n            self.optimizer.zero_grad()\n            outputs = self.model(X_tensor)\n            loss = self.criterion(outputs, y_tensor)\n            loss.backward()\n            self.optimizer.step()\n\n            # Optional\n            if epoch % 10 == 0:\n                print(f\"Epoch [{epoch}/{self.epochs}], Loss: {loss.item():.4f}\")\n\n        return self\n\n    def predict(self, X):\n        self.model.eval()\n        with torch.no_grad():\n            X_tensor = torch.tensor(X.values if hasattr(X, \"values\") else X, dtype=torch.float32).to(self.device)\n            outputs = self.model(X_tensor).squeeze().cpu().numpy()\n        return outputs\n\npytorch_model = PyTorchRegressor(input_dim=input_dim, device='cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:17.419509Z","iopub.execute_input":"2024-10-28T04:02:17.419978Z","iopub.status.idle":"2024-10-28T04:02:17.442964Z","shell.execute_reply.started":"2024-10-28T04:02:17.419930Z","shell.execute_reply":"2024-10-28T04:02:17.441802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.optimize import minimize\nimport time\nimport optuna\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.linear_model import Ridge\n\n# Define the base models\nbase_models = [\n    ('lgb', LGBMRegressor(**LGBM_Params, verbose=-1)),\n    ('xgb', XGBRegressor(**XGB_Params)),\n    ('cat', CatBoostRegressor(**CatBoost_Params)),\n    ('rf', RandomForestRegressor(**RF_Params))\n    #('hgbr', HistGradientBoostingRegressor(**HGBR_Params)),\n    # ('pytorch', pytorch_model),\n]\n\n# Function to optimize QWK score by adjusting thresholds\ndef evaluate_predictions(thresholds, y_true, y_pred):\n    thresholds = np.sort(thresholds)\n    y_pred_classes = np.digitize(y_pred, thresholds)\n    return -cohen_kappa_score(y_true, y_pred_classes, weights='quadratic')\n\ninitial_thresholds = [0.5, 1.5, 2.5]\n\n# Optuna for meta model tuning with QWK maximization\ndef objective(trial):\n#     meta_params = {\n#         'n_estimators': trial.suggest_int('n_estimators', 100, 1000),\n#         'max_depth': trial.suggest_int('max_depth', 3, 20),\n#         'min_samples_split': trial.suggest_int('min_samples_split', 2, 10),\n#         'min_samples_leaf': trial.suggest_int('min_samples_leaf', 1, 13),\n#         'max_features': trial.suggest_categorical('max_features', ['sqrt', 'log2', None]),\n#         'bootstrap': trial.suggest_categorical('bootstrap', [True, False]),\n#         'random_state': SEED\n#     }\n    \n#     meta_model = RandomForestRegressor(**meta_params)\n    \n    # Suggest a value for alpha\n    alpha = trial.suggest_loguniform('alpha', 1e-4, 10.0)\n    # Define the Ridge model with suggested alpha\n    meta_model = Ridge(alpha=alpha, random_state=SEED)\n   \n    # alpha = trial.suggest_loguniform('alpha', 1e-4, 10)\n    # l1_ratio = trial.suggest_float('l1_ratio', 0.0, 1.0)\n    # Define the ElasticNet model with suggested parameters\n    # model = ElasticNet(alpha=alpha, l1_ratio=l1_ratio, random_state=0)\n    \n    # 5-fold CV for parameter tuning\n    scores = []\n    for train_idx, valid_idx in StratifiedKFold(n_splits=5, shuffle=True, random_state=SEED).split(X, y):\n        X_train, X_valid = X.iloc[train_idx], X.iloc[valid_idx]\n        y_train, y_valid = y.iloc[train_idx], y.iloc[valid_idx]\n        \n        # Train meta model and get predictions\n        meta_model.fit(X_train, y_train)\n        preds = meta_model.predict(X_valid)\n        \n        # Optimize thresholds for QWK score\n        KappaOptimizer = minimize(evaluate_predictions, x0=initial_thresholds,\n                                  args=(y_valid, preds), method='Nelder-Mead')\n        best_thresholds = np.sort(KappaOptimizer.x)\n        preds_class = np.digitize(preds, best_thresholds)\n        \n        # Calculate QWK score\n        qwk_score = cohen_kappa_score(y_valid, preds_class, weights='quadratic')\n        scores.append(qwk_score)\n        \n    # Return the mean QWK score (maximize)\n    return np.mean(scores)\n\n# Run Optuna for meta model tuning (QWK maximization)\nstudy = optuna.create_study(direction=\"maximize\")\nstudy.optimize(objective, n_trials=50)\nbest_meta_params = study.best_params\n\n# Updated meta model with best parameters\n# meta_model = RandomForestRegressor(**best_meta_params, random_state=SEED)\nmeta_model = Ridge(**best_meta_params, random_state=SEED)\n# meta_model = ElasticNet(**best_meta_params, random_state=SEED)\n\n# Stacking Regressor with base models and optimized meta model\nstacking_model = StackingRegressor(estimators=base_models, final_estimator=meta_model)\n\n# Cross-validation setup\nkf = StratifiedKFold(n_splits=10, shuffle=True, random_state=SEED)\npredictions = np.zeros(X.shape[0])\ntest_predictions = np.zeros(test_df_le.shape[0])\nqwk_scores = []\nmodel_scores = {name: [] for name, _ in base_models}\n\n# Time-tracking start\nstart_time = time.time()\n\nfor train_idx, valid_idx in kf.split(X, y):\n    X_train, X_valid = X.iloc[train_idx], X.iloc[valid_idx]\n    y_train, y_valid = y.iloc[train_idx], y.iloc[valid_idx]\n    \n    # Base model predictions\n    base_preds = []\n    for name, model in base_models:\n        model.fit(X_train, y_train)\n        model_preds = model.predict(X_valid)\n        \n        # Optimize thresholds for each base model\n        KappaOptimizer = minimize(evaluate_predictions, x0=initial_thresholds,\n                                  args=(y_valid, model_preds), method='Nelder-Mead')\n        best_thresholds = np.sort(KappaOptimizer.x)\n        model_preds_class = np.digitize(model_preds, best_thresholds)\n        \n        # QWK score and saving predictions\n        model_qwk_score = cohen_kappa_score(y_valid, model_preds_class, weights='quadratic')\n        model_scores[name].append(model_qwk_score)\n        base_preds.append(model_preds)\n    \n    # Fit stacking model with optimized RandomForestRegressor\n    stacking_model.fit(X_train, y_train)\n    \n    # Predict for validation set and optimize thresholds\n    preds = stacking_model.predict(X_valid)\n    KappaOptimizer = minimize(evaluate_predictions, x0=initial_thresholds,\n                              args=(y_valid, preds), method='Nelder-Mead')\n    best_thresholds = np.sort(KappaOptimizer.x)\n    ensemble_preds = np.digitize(preds, best_thresholds)\n    \n    # QWK score and test predictions\n    qwk_score = cohen_kappa_score(y_valid, ensemble_preds, weights='quadratic')\n    qwk_scores.append(qwk_score)\n    \n    # Store predictions\n    predictions[valid_idx] += preds\n    test_predictions += np.digitize(stacking_model.predict(test_df_le), best_thresholds) / kf.n_splits\n\n# Display total processing time\nelapsed_time = time.time() - start_time\nprint(f\"Total processing time: {elapsed_time:.2f} seconds\")","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:02:17.444930Z","iopub.execute_input":"2024-10-28T04:02:17.445368Z","iopub.status.idle":"2024-10-28T04:27:19.395808Z","shell.execute_reply.started":"2024-10-28T04:02:17.445324Z","shell.execute_reply":"2024-10-28T04:27:19.393687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import RandomizedSearchCV\n# from sklearn.ensemble import RandomForestRegressor\n# import time\n\n# # Define the base models\n# base_models = [\n#     ('lgb', LGBMRegressor(**LGBM_Params, verbose=-1)),\n#     ('xgb', XGBRegressor(**XGB_Params)),\n#     ('cat', CatBoostRegressor(**CatBoost_Params)),\n#     ('hgbr', HistGradientBoostingRegressor(**HGBR_Params)),\n# ]\n\n# # Define the RandomForestRegressor for meta model with parameter distributions\n# meta_model = RandomForestRegressor(random_state=SEED)\n# meta_param_dist = {\n#     'n_estimators': [220, 225, 250, 300, 350, 400],  \n#     'max_depth': [3, 5, 7, 8],                    \n#     'min_samples_split': [1, 2, 3, 4, 5],                   \n#     'min_samples_leaf': [10, 11, 12, 13],                    \n#     'max_features': ['sqrt'],                     \n#     'bootstrap': [True]                            \n# }\n\n# # Use RandomizedSearchCV to optimize the RandomForest meta model\n# random_search = RandomizedSearchCV(\n#     meta_model, param_distributions=meta_param_dist,\n#     n_iter=10,  # Adjust the number of iterations as needed\n#     cv=3,\n#     scoring='neg_mean_squared_error',\n#     random_state=SEED,\n#     n_jobs=-1\n# )\n\n# # Stacking Regressor with the base models and meta model\n# stacking_model = StackingRegressor(estimators=base_models, final_estimator=random_search)\n\n# # Cross-validation and model training\n# kf = StratifiedKFold(n_splits=5, shuffle=True, random_state=SEED)\n# predictions = np.zeros(X.shape[0])\n# test_predictions = np.zeros(test_df_le.shape[0])\n# qwk_scores = []\n# model_scores = {name: [] for name, _ in base_models}\n\n# # Function to optimize QWK score by adjusting thresholds\n# def evaluate_predictions(thresholds, y_true, y_pred):\n#     thresholds = np.sort(thresholds)\n#     y_pred_classes = np.digitize(y_pred, thresholds)\n#     return -cohen_kappa_score(y_true, y_pred_classes, weights='quadratic')\n\n# # Timing the process\n# start_time = time.time()\n\n# for train_idx, valid_idx in kf.split(X, y):\n#     X_train, X_valid = X.iloc[train_idx], X.iloc[valid_idx]\n#     y_train, y_valid = y.iloc[train_idx], y.iloc[valid_idx]\n    \n#     # Individual base model predictions\n#     for name, model in base_models:\n#         model.fit(X_train, y_train)\n#         model_preds = model.predict(X_valid)\n        \n#         # Optimize thresholds for individual base model using Nelder-Mead\n#         initial_thresholds = [0.5, 1.5, 2.5]\n#         KappaOptimizer = minimize(evaluate_predictions, x0=initial_thresholds,\n#                                   args=(y_valid, model_preds), method='Nelder-Mead')\n#         best_thresholds = KappaOptimizer.x\n#         model_preds_class = np.digitize(model_preds, np.sort(best_thresholds))\n        \n#         model_qwk_score = cohen_kappa_score(y.iloc[valid_idx], model_preds_class, weights='quadratic')\n#         model_scores[name].append(model_qwk_score)\n    \n#     # Fit the stacking model with RandomizedSearchCV optimized RandomForestRegressor\n#     stacking_model.fit(X_train, y_train)\n    \n#     # Predict for validation set\n#     preds = stacking_model.predict(X_valid)\n#     predictions[valid_idx] += preds\n    \n#     # Optimize thresholds for ensemble model using Nelder-Mead\n#     KappaOptimizer = minimize(evaluate_predictions, x0=initial_thresholds,\n#                               args=(y_valid, preds), method='Nelder-Mead')\n#     best_thresholds = KappaOptimizer.x\n    \n#     # Apply optimized thresholds to validation predictions\n#     ensemble_preds = np.digitize(preds, np.sort(best_thresholds))\n    \n#     # Calculate QWK score for ensemble model\n#     qwk_score = cohen_kappa_score(y.iloc[valid_idx], ensemble_preds, weights='quadratic')\n#     qwk_scores.append(qwk_score)\n    \n#     # Predict for test set and store using optimized thresholds\n#     test_preds = stacking_model.predict(test_df_le)\n#     test_predictions += np.digitize(test_preds, np.sort(best_thresholds)) / kf.n_splits\n\n# # Timing output\n# elapsed_time = time.time() - start_time\n# print(f\"Total processing time: {elapsed_time:.2f} seconds\")","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-10-28T04:27:19.400107Z","iopub.execute_input":"2024-10-28T04:27:19.401363Z","iopub.status.idle":"2024-10-28T04:27:19.415852Z","shell.execute_reply.started":"2024-10-28T04:27:19.401296Z","shell.execute_reply":"2024-10-28T04:27:19.413447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from scipy.optimize import minimize\n# import time\n\n# # Timing the process to ensure it doesn't exceed expected limits\n# start_time = time.time()\n\n# kf = StratifiedKFold(n_splits=5, shuffle=True, random_state=SEED)\n# predictions = np.zeros(X.shape[0])\n# test_predictions = np.zeros(test_df_le.shape[0])\n# qwk_scores = []\n\n# # Function to optimize QWK score by adjusting thresholds\n# def evaluate_predictions(thresholds, y_true, y_pred):\n#     thresholds = np.sort(thresholds)\n#     y_pred_classes = np.digitize(y_pred, thresholds)\n#     return -cohen_kappa_score(y_true, y_pred_classes, weights='quadratic')\n\n# for train_idx, valid_idx in kf.split(X, y):\n#     X_train, X_valid = X.iloc[train_idx], X.iloc[valid_idx]\n#     y_train, y_valid = y.iloc[train_idx], y.iloc[valid_idx]\n    \n#     # Train and predict with only Pytorch model\n#     pytorch_model.fit(X_train, y_train)\n#     model_preds = pytorch_model.predict(X_valid)\n    \n#     # Optimize thresholds for Pytorch model using Nelder-Mead\n#     initial_thresholds = [0.5, 1.5, 2.5]\n#     KappaOptimizer = minimize(evaluate_predictions, x0=initial_thresholds, \n#                               args=(y_valid, model_preds), method='Nelder-Mead')\n#     best_thresholds = KappaOptimizer.x\n#     model_preds_class = np.digitize(model_preds, np.sort(best_thresholds))\n    \n#     # Calculate QWK score for Pytorch model\n#     model_qwk_score = cohen_kappa_score(y.iloc[valid_idx], model_preds_class, weights='quadratic')\n#     qwk_scores.append(model_qwk_score)\n    \n#     # Store predictions\n#     predictions[valid_idx] = model_preds_class\n    \n#     # Predict for test set using optimized thresholds\n#     test_preds = pytorch_model.predict(test_df_le)\n#     test_predictions += np.digitize(test_preds, np.sort(best_thresholds)) / kf.n_splits\n\n# # Timing output\n# elapsed_time = time.time() - start_time\n# print(f\"Total processing time: {elapsed_time:.2f} seconds\")\n# print(\"Average QWK Score for Pytorch model only:\", np.mean(qwk_scores))","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-10-28T04:27:19.418458Z","iopub.execute_input":"2024-10-28T04:27:19.418901Z","iopub.status.idle":"2024-10-28T04:27:19.442746Z","shell.execute_reply.started":"2024-10-28T04:27:19.418856Z","shell.execute_reply":"2024-10-28T04:27:19.441104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# RandomForest\n# 'n_estimators': 848, 'max_depth': 19, 'min_samples_split': 2, 'min_samples_leaf': 10, 'max_features': None, 'bootstrap': True}. Best is trial 37","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:27:19.444422Z","iopub.execute_input":"2024-10-28T04:27:19.444848Z","iopub.status.idle":"2024-10-28T04:27:19.464173Z","shell.execute_reply.started":"2024-10-28T04:27:19.444804Z","shell.execute_reply":"2024-10-28T04:27:19.462642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Define the base models\n# base_models = [\n#     ('lgb', LGBMRegressor(**LGBM_Params, verbose=-1)),\n#     ('xgb', XGBRegressor(**XGB_Params)),\n#     ('cat', CatBoostRegressor(**CatBoost_Params)),\n#     # ('nn', nn_model),\n#     # ('fcnn', fcnn_model),\n#     # ('mlp', mlp_model),\n#     # ('pytorch', pytorch_model),\n# ]\n\n# meta_model = RandomForestRegressor(n_estimators=848, # 225\n#                                    max_depth=19, # 3 -> 5\n#                                    min_samples_split=2,\n#                                    min_samples_leaf=10, # 4\n#                                    max_features='None', # sqrt \n#                                    max_samples=0.8, # 0.8 \n#                                    # boostrap='True',\n#                                    random_state=SEED) # RandomForest\n# #'n_estimators': 848, 'max_depth': 19, 'min_samples_split': 2, 'min_samples_leaf': 10, 'max_features': None, 'bootstrap': True\n\n# stacking_model = StackingRegressor(estimators=base_models, final_estimator=meta_model)\n\n# # Function to optimize QWK score by adjusting thresholds\n# def evaluate_predictions(thresholds, y_true, y_pred):\n#     thresholds = np.sort(thresholds)\n#     y_pred_classes = np.digitize(y_pred, thresholds)\n#     return -cohen_kappa_score(y_true, y_pred_classes, weights='quadratic')\n\n# initial_thresholds = [0.5, 1.5, 2.5]\n\n# # Cross-validation and model training\n# kf = StratifiedKFold(n_splits=5, shuffle=True, random_state=SEED)\n# predictions = np.zeros(X.shape[0]) \n# test_predictions = np.zeros(test_df_le.shape[0])\n# qwk_scores = []\n# model_scores = {name: [] for name, _ in base_models}\n\n# # Lists to store predictions for each fold\n# stacking_valid_predictions = np.zeros(X.shape[0])\n# averaging_valid_predictions = np.zeros(X.shape[0])\n\n# for train_idx, valid_idx in kf.split(X, y):\n#     X_train, X_valid = X.iloc[train_idx], X.iloc[valid_idx]\n#     y_train, y_valid = y.iloc[train_idx], y.iloc[valid_idx]\n    \n#     # Individual base model predictions and averaging\n#     base_model_preds = np.zeros((len(valid_idx), len(base_models)))  # Store predictions for each base model\n    \n#     for i, (name, model) in enumerate(base_models):\n#         model.fit(X_train, y_train)\n#         model_preds = model.predict(X_valid)\n        \n#         # initial_thresholds = [0.5, 1.5, 2.5]  # Initial guess for thresholds\n#         KappaOptimizer = minimize(evaluate_predictions, x0=initial_thresholds, \n#                                   args=(y_valid, model_preds), method='Nelder-Mead')\n#         best_thresholds = KappaOptimizer.x\n#         model_preds_class = np.digitize(model_preds, np.sort(best_thresholds))\n        \n#         model_qwk_score = cohen_kappa_score(y.iloc[valid_idx], model_preds_class, weights='quadratic')\n#         model_scores[name].append(model_qwk_score)\n        \n#         # Store base model predictions\n#         base_model_preds[:, i] = model_preds\n    \n#     # Calculate averaging result\n#     averaging_preds = base_model_preds.mean(axis=1)\n    \n#     # Optimize thresholds for averaging result\n#     KappaOptimizer = minimize(evaluate_predictions, x0=initial_thresholds, \n#                               args=(y_valid, averaging_preds), method='Nelder-Mead')\n#     best_thresholds = KappaOptimizer.x\n#     averaging_preds_class = np.digitize(averaging_preds, np.sort(best_thresholds))\n    \n#     averaging_qwk_score = cohen_kappa_score(y.iloc[valid_idx], averaging_preds_class, weights='quadratic')\n\n#     # Stacking model predictions\n#     stacking_model.fit(X_train, y_train)\n#     stacking_preds = stacking_model.predict(X_valid)\n    \n#     # Optimize thresholds for stacking model result\n#     KappaOptimizer = minimize(evaluate_predictions, x0=initial_thresholds, \n#                               args=(y_valid, stacking_preds), method='Nelder-Mead')\n#     best_thresholds = KappaOptimizer.x\n#     stacking_preds_class = np.digitize(stacking_preds, np.sort(best_thresholds))\n    \n#     stacking_qwk_score = cohen_kappa_score(y.iloc[valid_idx], stacking_preds_class, weights='quadratic')\n\n#     # Compare stacking model and averaging results and choose the better one\n#     if stacking_qwk_score > averaging_qwk_score:\n#         predictions[valid_idx] = stacking_preds_class\n#     else:\n#         predictions[valid_idx] = averaging_preds_class\n    \n#     # Handle test data prediction similarly\n#     test_preds_stacking = stacking_model.predict(test_df_le)  # Test set predictions from stacking model\n\n#     # Test set predictions from the base model averaging\n#     base_model_test_preds = np.zeros((test_df_le.shape[0], len(base_models)))  # Shape based on test set size\n\n#     # Generate predictions for the test set using the base models (for averaging)\n#     for i, (name, model) in enumerate(base_models):\n#         base_model_test_preds[:, i] = model.predict(test_df_le)  # Predict for each base model\n\n#     # Average predictions from the base models for the test set\n#     test_preds_averaging = base_model_test_preds.mean(axis=1)\n\n#     # Compare stacking model and averaging results and choose the better one\n#     if stacking_qwk_score > averaging_qwk_score:\n#         test_predictions += np.digitize(test_preds_stacking, np.sort(best_thresholds)) / kf.n_splits\n#     else:\n#         test_predictions += np.digitize(test_preds_averaging, np.sort(best_thresholds)) / kf.n_splits\n","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-10-28T04:27:19.466492Z","iopub.execute_input":"2024-10-28T04:27:19.466985Z","iopub.status.idle":"2024-10-28T04:27:19.486671Z","shell.execute_reply.started":"2024-10-28T04:27:19.466939Z","shell.execute_reply":"2024-10-28T04:27:19.485180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Individual model scores:')\nfor name, scores in model_scores.items():\n    print(f'{name}: Average QWK = {np.mean(scores):.4f}, Std = {np.std(scores):.4f}')","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:27:19.488473Z","iopub.execute_input":"2024-10-28T04:27:19.488881Z","iopub.status.idle":"2024-10-28T04:27:19.509371Z","shell.execute_reply.started":"2024-10-28T04:27:19.488840Z","shell.execute_reply":"2024-10-28T04:27:19.507999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Define the base models\n# base_models = [\n#     ('lgb', LGBMRegressor(**LGBM_Params)),\n#     ('xgb', XGBRegressor(**XGB_Params)),\n#     ('cat', CatBoostRegressor(**CatBoost_Params)),\n#     ('nn', nn_model),\n#     ('fcnn', fcnn_model),\n#     ('mlp', mlp_model),\n#     #('gbr', GradientBoostingRegressor(**GBR_Params)),\n#     #('etr', ExtraTreesRegressor(**ETR_Params)),\n#     #('hgbr', HistGradientBoostingRegressor(**HGBR_Params)),\n#     #('svr', SVR(**SVR_Params)),\n#     #('knn', KNeighborsRegressor(**KNN_Params))\n    \n# ]\n# meta_model = RandomForestRegressor(n_estimators=225, # 225\n#                                    max_depth=5, # 3 -> 5\n#                                    min_samples_split=5,\n#                                    min_samples_leaf=13, # 4\n#                                    max_features='sqrt', # sqrt \n#                                    max_samples=0.8, # 0.8 \n#                                    random_state=SEED) # RandomForest\n# # meta_model = HistGradientBoostingRegressor(**HGBR_Params) # HistGradientBoost\n# stacking_model = StackingRegressor(estimators=base_models, final_estimator=meta_model)\n\n# # Cross-validation and model training\n# kf = StratifiedKFold(n_splits=5, shuffle=True, random_state=SEED)\n# predictions = np.zeros(X.shape[0]) \n# test_predictions = np.zeros(test_df_le.shape[0])\n# qwk_scores = []\n\n# # Function to optimize the QWK score by adjusting thresholds\n# def evaluate_predictions(thresholds, y_true, y_pred):\n#     thresholds = np.sort(thresholds)  # Ensure thresholds are in ascending order\n#     y_pred_classes = np.digitize(y_pred, thresholds)\n#     return -cohen_kappa_score(y_true, y_pred_classes, weights='quadratic')\n\n# for train_idx, valid_idx in kf.split(X, y):\n#     X_train, X_valid = X.iloc[train_idx], X.iloc[valid_idx]\n#     y_train, y_valid = y.iloc[train_idx], y.iloc[valid_idx]\n    \n#     # Fit the stacking model\n#     stacking_model.fit(X_train, y_train)\n\n#     # Predict for validation set\n#     preds = stacking_model.predict(X_valid)\n#     predictions[valid_idx] += preds\n\n#     # Optimize thresholds for QWK score using Nelder-Mead\n#     initial_thresholds = [0.5, 1.5, 2.5]  # Initial guess for thresholds\n#     KappaOptimizer = minimize(evaluate_predictions, x0=initial_thresholds, \n#                               args=(y_valid, preds), method='Nelder-Mead')\n#     best_thresholds = KappaOptimizer.x\n\n#     # Apply optimized thresholds to validation predictions\n#     # ensemble_preds = np.digitize(predictions[valid_idx], np.sort(best_thresholds))\n#     ensemble_preds = np.digitize(preds, np.sort(best_thresholds))\n\n#     # Calculate QWK score for the optimized predictions\n#     qwk_score = cohen_kappa_score(y.iloc[valid_idx], ensemble_preds, weights='quadratic')\n#     qwk_scores.append(qwk_score)\n\n#     # Predict for test set and store using optimized thresholds\n#     test_preds = stacking_model.predict(test_df_le)\n#     test_predictions += np.digitize(test_preds, np.sort(best_thresholds)) / kf.n_splits","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-10-28T04:27:19.514124Z","iopub.execute_input":"2024-10-28T04:27:19.515478Z","iopub.status.idle":"2024-10-28T04:27:19.526280Z","shell.execute_reply.started":"2024-10-28T04:27:19.515413Z","shell.execute_reply":"2024-10-28T04:27:19.524895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Average: ', np.mean(qwk_scores))\nprint('Std: ', np.std(qwk_scores))\nprint('Min: ', np.min(qwk_scores))\nprint('Max: ', np.max(qwk_scores))","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:27:19.528349Z","iopub.execute_input":"2024-10-28T04:27:19.529526Z","iopub.status.idle":"2024-10-28T04:27:19.548760Z","shell.execute_reply.started":"2024-10-28T04:27:19.529464Z","shell.execute_reply":"2024-10-28T04:27:19.547272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"qwk_scores","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:27:19.550752Z","iopub.execute_input":"2024-10-28T04:27:19.551324Z","iopub.status.idle":"2024-10-28T04:27:19.567561Z","shell.execute_reply.started":"2024-10-28T04:27:19.551265Z","shell.execute_reply":"2024-10-28T04:27:19.566260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Final predictions","metadata":{}},{"cell_type":"code","source":"# Prepare the test predictions for submission\ntest_pred_classes = np.digitize(test_predictions, np.sort(best_thresholds))\nsubmit_df = pd.DataFrame({'id': test_id, 'sii': test_pred_classes.astype(int)})\nsubmit_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-28T04:27:19.569321Z","iopub.execute_input":"2024-10-28T04:27:19.569761Z","iopub.status.idle":"2024-10-28T04:27:19.588460Z","shell.execute_reply.started":"2024-10-28T04:27:19.569715Z","shell.execute_reply":"2024-10-28T04:27:19.587137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}