{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":9682055,"sourceType":"datasetVersion","datasetId":5918194}],"dockerImageVersionId":30776,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install tsflex --no-index --find-links=file:///kaggle/input/cmi-time-series-tools \n!pip install seglearn --no-index --find-links=file:///kaggle/input/cmi-time-series-tools  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:48:29.652167Z","iopub.execute_input":"2024-12-17T16:48:29.652406Z","iopub.status.idle":"2024-12-17T16:48:48.838593Z","shell.execute_reply.started":"2024-12-17T16:48:29.652381Z","shell.execute_reply":"2024-12-17T16:48:48.837717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars as pl\nfrom seglearn.feature_functions import base_features, emg_features\nfrom tsflex.features import FeatureCollection, MultipleFeatureDescriptors\nfrom tsflex.features.integrations import seglearn_feature_dict_wrapper\nimport os\nimport random\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer\nfrom imblearn.over_sampling import SMOTE\nfrom sklearn.metrics import make_scorer\nfrom sklearn.model_selection import GridSearchCV, RandomizedSearchCV\nfrom sklearn.preprocessing import StandardScaler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:48:48.840594Z","iopub.execute_input":"2024-12-17T16:48:48.840866Z","iopub.status.idle":"2024-12-17T16:48:55.461422Z","shell.execute_reply.started":"2024-12-17T16:48:48.840840Z","shell.execute_reply":"2024-12-17T16:48:55.460740Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set random seeds for reproducibility\nSEED = 42\nnp.random.seed(SEED)\nrandom.seed(SEED)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:48:55.462477Z","iopub.execute_input":"2024-12-17T16:48:55.463131Z","iopub.status.idle":"2024-12-17T16:48:55.467812Z","shell.execute_reply.started":"2024-12-17T16:48:55.463092Z","shell.execute_reply":"2024-12-17T16:48:55.467032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Suppress warnings\nwarnings.filterwarnings('ignore')\n\n# Pandas option for displaying all columns\npd.options.display.max_columns = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:48:55.470127Z","iopub.execute_input":"2024-12-17T16:48:55.470355Z","iopub.status.idle":"2024-12-17T16:48:55.481442Z","shell.execute_reply.started":"2024-12-17T16:48:55.470332Z","shell.execute_reply":"2024-12-17T16:48:55.480639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Constants\nn_splits = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:48:55.482295Z","iopub.execute_input":"2024-12-17T16:48:55.482525Z","iopub.status.idle":"2024-12-17T16:48:55.496088Z","shell.execute_reply.started":"2024-12-17T16:48:55.482502Z","shell.execute_reply":"2024-12-17T16:48:55.495517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load datasets\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\nnumeric_columns = train.select_dtypes(include = ['float64', 'int64']).columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:48:55.496866Z","iopub.execute_input":"2024-12-17T16:48:55.497146Z","iopub.status.idle":"2024-12-17T16:48:55.591759Z","shell.execute_reply.started":"2024-12-17T16:48:55.497122Z","shell.execute_reply":"2024-12-17T16:48:55.590930Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TSPipeline:\n    @staticmethod\n    def build_tsflex_features(df: pd.DataFrame):\n\n        df = df.to_pandas()\n        len = df.shape[0]\n\n        basic_feats = MultipleFeatureDescriptors(\n            functions=seglearn_feature_dict_wrapper(base_features()),\n            series_names=['X', 'Y', 'Z', 'enmo', 'anglez', 'light', 'battery_voltage'],\n            windows=[len],\n            strides=[len],\n        )\n        \n        emg_feats = emg_features()\n        del emg_feats['simple square integral']\n        \n        emg_feats = MultipleFeatureDescriptors(\n            functions=seglearn_feature_dict_wrapper(emg_feats),\n            series_names=['X', 'Y', 'Z', 'enmo', 'anglez', 'light', 'battery_voltage'],\n            windows=[len],\n            strides=[len],\n        )\n        \n        fc = FeatureCollection([basic_feats, emg_feats])\n        \n        df = fc.calculate(df,\n                          return_df=True, \n                          include_final_window=True, \n                          approve_sparsity=True, \n                          window_idx=\"begin\").astype(np.float32)\n\n        return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:48:55.592820Z","iopub.execute_input":"2024-12-17T16:48:55.593092Z","iopub.status.idle":"2024-12-17T16:48:55.599341Z","shell.execute_reply.started":"2024-12-17T16:48:55.593067Z","shell.execute_reply":"2024-12-17T16:48:55.598491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_file(path) -> np.ndarray: \n    df = pl.read_parquet(path)\n    df = df.pipe(TSPipeline.build_tsflex_features)\n    return df.to_numpy().flatten()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:48:55.600616Z","iopub.execute_input":"2024-12-17T16:48:55.601267Z","iopub.status.idle":"2024-12-17T16:48:55.616940Z","shell.execute_reply.started":"2024-12-17T16:48:55.601227Z","shell.execute_reply":"2024-12-17T16:48:55.616348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_time_series(dir_path: str) -> dict:\n    time_series_store = {}\n    file_ind = os.listdir(dir_path)\n    for file_name in tqdm(file_ind):\n        file_id = file_name.split('=')[1]\n        file_path = os.path.join(dir_path + file_name, 'part-0.parquet')\n        df = read_file(file_path)\n        time_series_store[file_id] = df\n        \n    return time_series_store","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:48:55.617686Z","iopub.execute_input":"2024-12-17T16:48:55.617903Z","iopub.status.idle":"2024-12-17T16:48:55.632303Z","shell.execute_reply.started":"2024-12-17T16:48:55.617880Z","shell.execute_reply":"2024-12-17T16:48:55.631530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load time series data\ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet/\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:48:55.634869Z","iopub.execute_input":"2024-12-17T16:48:55.635339Z","iopub.status.idle":"2024-12-17T16:59:37.661116Z","shell.execute_reply.started":"2024-12-17T16:48:55.635315Z","shell.execute_reply":"2024-12-17T16:59:37.660075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts = pd.DataFrame(train_ts).T\ntrain_ts.columns = [f'f{i}' for i in range(train_ts.shape[1])]\ntrain_ts = train_ts.reset_index()\ntrain_ts.rename(columns={'index': 'id'}, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:37.662435Z","iopub.execute_input":"2024-12-17T16:59:37.662735Z","iopub.status.idle":"2024-12-17T16:59:37.678576Z","shell.execute_reply.started":"2024-12-17T16:59:37.662705Z","shell.execute_reply":"2024-12-17T16:59:37.677694Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ts = pd.DataFrame(test_ts).T\ntest_ts.columns = [f'f{i}' for i in range(test_ts.shape[1])]\ntest_ts = test_ts.reset_index()\ntest_ts.rename(columns={'index': 'id'}, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:37.679593Z","iopub.execute_input":"2024-12-17T16:59:37.679883Z","iopub.status.idle":"2024-12-17T16:59:37.702226Z","shell.execute_reply.started":"2024-12-17T16:59:37.679858Z","shell.execute_reply":"2024-12-17T16:59:37.701642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Merge datasets\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:37.703335Z","iopub.execute_input":"2024-12-17T16:59:37.703974Z","iopub.status.idle":"2024-12-17T16:59:37.754662Z","shell.execute_reply.started":"2024-12-17T16:59:37.703913Z","shell.execute_reply":"2024-12-17T16:59:37.753870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[time_series_cols] = train[time_series_cols].fillna(value=0)  \ntest[time_series_cols] = test[time_series_cols].fillna(value=0)  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:37.755621Z","iopub.execute_input":"2024-12-17T16:59:37.755858Z","iopub.status.idle":"2024-12-17T16:59:37.794769Z","shell.execute_reply.started":"2024-12-17T16:59:37.755834Z","shell.execute_reply":"2024-12-17T16:59:37.794118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define feature columns\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-Season', 'CGAS-CGAS_Score',\n                'Physical-Season', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Season',\n                'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-Season',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone',\n                'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone',\n                'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat',\n                'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW',\n                'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season', 'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season', 'PreInt_EduHx-computerinternet_hoursday', 'sii']\nfeaturesCols += time_series_cols\ntrain = train[featuresCols]\ntrain = train.dropna(subset=['sii'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:37.795598Z","iopub.execute_input":"2024-12-17T16:59:37.795824Z","iopub.status.idle":"2024-12-17T16:59:37.813330Z","shell.execute_reply.started":"2024-12-17T16:59:37.795801Z","shell.execute_reply":"2024-12-17T16:59:37.812550Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', 'FGC-Season',\n         'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c:\n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n\ntrain = update(train)\ntest = update(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:37.814107Z","iopub.execute_input":"2024-12-17T16:59:37.814304Z","iopub.status.idle":"2024-12-17T16:59:37.845083Z","shell.execute_reply.started":"2024-12-17T16:59:37.814282Z","shell.execute_reply":"2024-12-17T16:59:37.844479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(create_mapping(col, test)).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:37.846014Z","iopub.execute_input":"2024-12-17T16:59:37.846246Z","iopub.status.idle":"2024-12-17T16:59:37.890032Z","shell.execute_reply.started":"2024-12-17T16:59:37.846223Z","shell.execute_reply":"2024-12-17T16:59:37.889250Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sii = train['sii']\ntrain = train.drop(['sii'], axis=1)\n\ncol_features_imp = [item for item in featuresCols if item != 'sii']\nimputer = SimpleImputer(strategy='mean')\n\ntrain = imputer.fit_transform(train)                          \ntrain = pd.DataFrame(data = train, columns = col_features_imp)\n\ntest = imputer.transform(test)\ntest = pd.DataFrame(data = test, columns = col_features_imp)\n\ntrain['sii'] = sii.to_list()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:37.890981Z","iopub.execute_input":"2024-12-17T16:59:37.891210Z","iopub.status.idle":"2024-12-17T16:59:37.940436Z","shell.execute_reply.started":"2024-12-17T16:59:37.891187Z","shell.execute_reply":"2024-12-17T16:59:37.939598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = StandardScaler()\n\nnum_columns = [c for c in train.columns if c in numeric_columns and c not in ['sii']] \nnum_columns = num_columns + time_series_cols\n\ntrain_scalar = scaler.fit_transform(train[num_columns])  \ntest_scalar = scaler.transform(test[num_columns])\n\ntrain_scalar = pd.DataFrame(data = train_scalar, columns = num_columns)\ntest_scalar = pd.DataFrame(data = test_scalar, columns = num_columns)\n\nfor col in num_columns:\n    train.loc[:, col] = train_scalar[col].values\n    test.loc[:, col] = test_scalar[col].values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:37.941408Z","iopub.execute_input":"2024-12-17T16:59:37.941659Z","iopub.status.idle":"2024-12-17T16:59:38.011156Z","shell.execute_reply.started":"2024-12-17T16:59:37.941635Z","shell.execute_reply":"2024-12-17T16:59:38.010539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in cat_c:\n    train[i] = train[i].astype(int) \n    test[i] = test[i].astype(int) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:38.011904Z","iopub.execute_input":"2024-12-17T16:59:38.012128Z","iopub.status.idle":"2024-12-17T16:59:38.022654Z","shell.execute_reply.started":"2024-12-17T16:59:38.012106Z","shell.execute_reply":"2024-12-17T16:59:38.021946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_params = {\n    'learning_rate': np.linspace(0.01, 0.1, num=100).tolist(),  \n    'max_depth': [3, 5, 8, 10, 12, 14],\n    'n_estimators': list(range(10, 400, 5)),\n    'num_leaves': list(range(10, 500, 5)),\n    'min_data_in_leaf': list(range(10, 50, 3)),\n    'feature_fraction': [round(x, 3) for x in np.linspace(0.3, 1.0, num=40)],  \n    'bagging_fraction': [round(x, 3) for x in np.linspace(0.3, 1.0, num=40)],  \n    'bagging_freq': list(range(2, 60, 2)),\n    'lambda_l1': np.linspace(0, 5, num=40).tolist(),\n    'lambda_l2': np.linspace(0, 5, num=40).tolist(),\n    'device': ['gpu']\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:38.023739Z","iopub.execute_input":"2024-12-17T16:59:38.024136Z","iopub.status.idle":"2024-12-17T16:59:38.034098Z","shell.execute_reply.started":"2024-12-17T16:59:38.024063Z","shell.execute_reply":"2024-12-17T16:59:38.033500Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_params = {\n    'learning_rate': np.linspace(0.01, 0.1, num=100).tolist(),\n    'max_depth': [3, 5, 8, 10, 12, 14],\n    'n_estimators': list(range(10, 400, 5)),\n    'subsample': np.arange(0.1, 1.01, 0.05).tolist(),\n    'colsample_bytree': np.arange(0.1, 1.01, 0.05).tolist(),\n    'reg_alpha': np.arange(0.1, 10, 0.5).tolist(), \n    'reg_lambda': np.arange(0.1, 10, 0.5).tolist(),\n    'random_state': [42],\n    'tree_method': ['hist', 'approx', 'gpu_hist'],\n    'device': ['cuda']\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:38.035020Z","iopub.execute_input":"2024-12-17T16:59:38.035329Z","iopub.status.idle":"2024-12-17T16:59:38.047351Z","shell.execute_reply.started":"2024-12-17T16:59:38.035287Z","shell.execute_reply":"2024-12-17T16:59:38.046585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_boost_params = {\n    'learning_rate': np.linspace(0.01, 0.1, num=100).tolist(),\n    'depth': [3, 4, 5, 6, 7],\n    'iterations': list(range(10, 250, 10)),\n    'random_seed': [42],\n    'verbose': [0],\n    'l2_leaf_reg': list(range(10, 50, 10)),\n    'task_type': ['GPU']\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:38.048478Z","iopub.execute_input":"2024-12-17T16:59:38.048831Z","iopub.status.idle":"2024-12-17T16:59:38.070649Z","shell.execute_reply.started":"2024-12-17T16:59:38.048796Z","shell.execute_reply":"2024-12-17T16:59:38.069911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rf_params = {\n    'n_estimators': list(range(10, 400, 5)),  \n    'max_depth': [3, 5, 8, 10, 11, 12, 14, 15],  \n    'min_samples_split': list(range(10, 50)),  \n    'min_samples_leaf': list(range(10, 50)),   \n    'max_features': ['auto', 'sqrt', 'log2'],  \n    'bootstrap': [True, False],  \n    'random_state': [42],  \n    'n_jobs': [-1],  \n    'verbose': [0],  \n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:38.071514Z","iopub.execute_input":"2024-12-17T16:59:38.071716Z","iopub.status.idle":"2024-12-17T16:59:38.086272Z","shell.execute_reply.started":"2024-12-17T16:59:38.071694Z","shell.execute_reply":"2024-12-17T16:59:38.085666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gb_params = {\n    'learning_rate': np.linspace(0.01, 0.1, num=100).tolist(),  \n    'n_estimators': list(range(50, 300, 10)), \n    'max_depth': [3, 5, 8, 10, 11, 12, 14, 15],  \n    'min_samples_split': list(range(10, 50)),  \n    'min_samples_leaf': list(range(10, 50)),  \n    'subsample': np.arange(0.1, 1.0, 0.05).tolist(),  \n    'max_features': ['auto', 'sqrt', 'log2'],  \n    'alpha': np.linspace(0.1, 0.9, num=30).tolist(),  \n    'random_state': [42],  \n    'verbose': [0],  \n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:38.087117Z","iopub.execute_input":"2024-12-17T16:59:38.087345Z","iopub.status.idle":"2024-12-17T16:59:38.099814Z","shell.execute_reply.started":"2024-12-17T16:59:38.087322Z","shell.execute_reply":"2024-12-17T16:59:38.099207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:38.100761Z","iopub.execute_input":"2024-12-17T16:59:38.101020Z","iopub.status.idle":"2024-12-17T16:59:38.113715Z","shell.execute_reply.started":"2024-12-17T16:59:38.100996Z","shell.execute_reply":"2024-12-17T16:59:38.112958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def search_optimal_param(X: pd.DataFrame, model, param: dict):\n    \n    skfolds = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n    \n    x = X.copy()\n    y = x['sii']\n    _ = x.pop('sii') \n    \n    for num_fold, (train_index, val_index) in enumerate(skfolds.split(x, y)):\n        print('fold: ', num_fold+1)\n        X_train, X_val = x.iloc[train_index], x.iloc[val_index]\n        y_train, y_val = y.iloc[train_index], y.iloc[val_index]\n        \n        model_fold = clone(model)\n        kappa_scorer = make_scorer(cohen_kappa_score)\n        search = RandomizedSearchCV(model_fold, param, scoring = kappa_scorer)\n        \n        search.fit(X_train, y_train)\n        predictions = search.predict(X_val)  \n        predictions = np.round(predictions)\n        \n        kappa_fold = quadratic_weighted_kappa(y_val, predictions)\n        \n        if num_fold == 0:\n            best_model_score = kappa_fold\n            best_model = search.best_estimator_  \n            best_params = search.best_params_\n            print('kappa: ', kappa_fold)\n            print('best_params:', best_params)\n            \n        elif kappa_fold > best_model_score:\n            best_model_score = kappa_fold\n            best_model = search.best_estimator_  \n            best_params = search.best_params_\n            print('kappa: ', kappa_fold)\n            print('best_params:', best_params)\n            \n    return best_model, best_params, best_model_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:38.114727Z","iopub.execute_input":"2024-12-17T16:59:38.115236Z","iopub.status.idle":"2024-12-17T16:59:38.124283Z","shell.execute_reply.started":"2024-12-17T16:59:38.115200Z","shell.execute_reply":"2024-12-17T16:59:38.123647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# lgb = LGBMRegressor(verbose=-1)\n# lgb_best_model, lgb_best_params, lgb_best_score = search_optimal_param(train, lgb, lgb_params)\n\nlgb_best_params = {'num_leaves': 70,\n                   'n_estimators': 110,\n                   'min_data_in_leaf': 10,\n                   'max_depth': 3,\n                   'learning_rate': 0.07272727272727272,\n                   'lambda_l2': 3.7179487179487176,\n                   'lambda_l1': 0.641025641025641,\n                   'feature_fraction': 0.803,\n                   'device': 'gpu',             \n                   'bagging_freq': 2,\n                   'bagging_fraction': 0.569}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:38.128113Z","iopub.execute_input":"2024-12-17T16:59:38.128354Z","iopub.status.idle":"2024-12-17T16:59:38.142795Z","shell.execute_reply.started":"2024-12-17T16:59:38.128331Z","shell.execute_reply":"2024-12-17T16:59:38.141968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# xgb = XGBRegressor()\n# xgb_best_model, xgb_best_params, xgb_best_score = search_optimal_param(train, xgb, xgb_params)\n\nxgb_best_params = {'tree_method': 'gpu_hist',    \n                   'subsample': 0.3500000000000001,\n                   'reg_lambda': 5.6,\n                   'reg_alpha': 6.6,\n                   'random_state': 42,\n                   'n_estimators': 365,\n                   'max_depth': 12,\n                   'learning_rate': 0.024545454545454547,\n                   'device': 'cuda',  \n                   'colsample_bytree': 0.7000000000000002}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:38.143850Z","iopub.execute_input":"2024-12-17T16:59:38.144162Z","iopub.status.idle":"2024-12-17T16:59:38.162645Z","shell.execute_reply.started":"2024-12-17T16:59:38.144137Z","shell.execute_reply":"2024-12-17T16:59:38.162061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# cat_boost = CatBoostRegressor(cat_features = cat_c)\n# cb_best_model, cb_best_params, cb_best_score = search_optimal_param(train, cat_boost, cat_boost_params)\n\ncb_best_params = {'verbose': 0,\n                  'task_type': 'GPU',    \n                  'random_seed': 42,\n                  'learning_rate': 0.07454545454545455,\n                  'l2_leaf_reg': 20,\n                  'iterations': 180,\n                  'depth': 3}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:38.163586Z","iopub.execute_input":"2024-12-17T16:59:38.164043Z","iopub.status.idle":"2024-12-17T16:59:38.177291Z","shell.execute_reply.started":"2024-12-17T16:59:38.164018Z","shell.execute_reply":"2024-12-17T16:59:38.176637Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# rf = RandomForestRegressor()\n# rf_best_model, rf_best_params, rf_best_score = search_optimal_param(train, rf, rf_params)\n\nrf_best_params = {'verbose': 0, \n                  'random_state': 42, \n                  'n_jobs': -1, \n                  'n_estimators': 15, \n                  'min_samples_split': 31, \n                  'min_samples_leaf': 29, \n                  'max_features': 'auto', \n                  'max_depth': 15, \n                  'bootstrap': True}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:38.178206Z","iopub.execute_input":"2024-12-17T16:59:38.178530Z","iopub.status.idle":"2024-12-17T16:59:38.189524Z","shell.execute_reply.started":"2024-12-17T16:59:38.178494Z","shell.execute_reply":"2024-12-17T16:59:38.188955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# gb = GradientBoostingRegressor()\n# gb_best_model, gb_best_params, gb_best_score = search_optimal_param(train, gb, gb_params)\n\ngb_best_params = {'verbose': 0, \n                  'subsample': 0.5000000000000001, \n                  'random_state': 42, \n                  'n_estimators': 250, \n                  'min_samples_split': 19, \n                  'min_samples_leaf': 48, \n                  'max_features': 'sqrt', \n                  'max_depth': 11, \n                  'learning_rate': 0.07727272727272727, \n                  'alpha': 0.21034482758620693}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T16:59:38.190332Z","iopub.execute_input":"2024-12-17T16:59:38.190534Z","iopub.status.idle":"2024-12-17T16:59:38.208257Z","shell.execute_reply.started":"2024-12-17T16:59:38.190513Z","shell.execute_reply":"2024-12-17T16:59:38.207617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def threshold_Rounder(oof_non_rounded, thresholds):\n    # print('thresholds:', thresholds)\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:07:14.639563Z","iopub.execute_input":"2024-12-17T17:07:14.639912Z","iopub.status.idle":"2024-12-17T17:07:14.645144Z","shell.execute_reply.started":"2024-12-17T17:07:14.639883Z","shell.execute_reply":"2024-12-17T17:07:14.644362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def TrainML(model_class, test_data):\n    \n    X = train.drop(['sii'], axis=1)    \n    y = train['sii']   \n\n    X_res, y_res = X, y\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    train_S = []\n    test_S = []\n    oof_non_rounded = np.zeros(len(y_res), dtype=float)\n    oof_rounded = np.zeros(len(y_res), dtype=int)\n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, val_idx) in enumerate(SKF.split(X_res, y_res)):\n        X_train, X_val = X_res.iloc[train_idx], X_res.iloc[val_idx]\n        y_train, y_val = y_res.iloc[train_idx], y_res.iloc[val_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[val_idx] = y_val_pred\n        y_val_pred_rounded = np.round(y_val_pred).astype(int)\n        oof_rounded[val_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, np.round(y_train_pred).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n\n        test_preds[:, fold] = model.predict(test_data)\n\n        print(f\"Fold {fold + 1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    # Optimize thresholds with Nelder-Mead method\n    KappaOptimizer = minimize(evaluate_predictions, x0=[0.5, 1.5, 2.5], args=(y_res, oof_non_rounded), method='Nelder-Mead')\n    assert KappaOptimizer.success, \"Optimization did not converge.\"\n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOptimizer.x)\n    tKappa = quadratic_weighted_kappa(y_res, oof_tuned)\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOptimizer.x)\n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:07:17.822355Z","iopub.execute_input":"2024-12-17T17:07:17.822691Z","iopub.status.idle":"2024-12-17T17:07:17.832219Z","shell.execute_reply.started":"2024-12-17T17:07:17.822661Z","shell.execute_reply":"2024-12-17T17:07:17.831382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create model instances\nLight = LGBMRegressor(**lgb_best_params, verbose=-1)                              # 0.4449055632128198\n\nXGB_Model = XGBRegressor(**xgb_best_params)                                       # 0.46584022243922796\n\nCatBoost_Model = CatBoostRegressor(**cb_best_params, cat_features = cat_c)        # 0.4500549273522687\n\nRFReg_Model = RandomForestRegressor(**rf_best_params)                             # 0.44319127442746176\n\ngb = GradientBoostingRegressor(**gb_best_params)                                  # 0.4710221522545822","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:07:22.742276Z","iopub.execute_input":"2024-12-17T17:07:22.742615Z","iopub.status.idle":"2024-12-17T17:07:22.747890Z","shell.execute_reply.started":"2024-12-17T17:07:22.742584Z","shell.execute_reply":"2024-12-17T17:07:22.746818Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Combine models using Voting Regressor\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n    ('rfreg', RFReg_Model),\n    ('gb', gb)\n], weights=[25, 10, 25, 10, 3])  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:07:26.103329Z","iopub.execute_input":"2024-12-17T17:07:26.104133Z","iopub.status.idle":"2024-12-17T17:07:26.108451Z","shell.execute_reply.started":"2024-12-17T17:07:26.104098Z","shell.execute_reply":"2024-12-17T17:07:26.107559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the ensemble model\nSubmission = TrainML(voting_model, test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T17:07:29.421174Z","iopub.execute_input":"2024-12-17T17:07:29.421520Z","iopub.status.idle":"2024-12-17T17:07:57.999666Z","shell.execute_reply.started":"2024-12-17T17:07:29.421486Z","shell.execute_reply":"2024-12-17T17:07:57.998801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save submission\nSubmission.to_csv('submission.csv', index=False)\nprint(Submission['sii'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2024-12-17T17:07:58.001038Z","iopub.execute_input":"2024-12-17T17:07:58.001321Z","iopub.status.idle":"2024-12-17T17:07:58.008182Z","shell.execute_reply.started":"2024-12-17T17:07:58.001295Z","shell.execute_reply":"2024-12-17T17:07:58.007259Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}