{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":7453542,"sourceType":"datasetVersion","datasetId":921302}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom scipy.optimize import minimize\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nimport lightgbm as lgb\nfrom xgboost import XGBRegressor\n\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.impute import KNNImputer\n\nfrom scipy.optimize import minimize\nimport optuna\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score\nfrom sklearn.preprocessing import StandardScaler\n\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, TensorDataset\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-12-20T01:50:31.897313Z","iopub.execute_input":"2024-12-20T01:50:31.897653Z","iopub.status.idle":"2024-12-20T01:50:52.092709Z","shell.execute_reply.started":"2024-12-20T01:50:31.897612Z","shell.execute_reply":"2024-12-20T01:50:52.091865Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip -q install /kaggle/input/pytorchtabnet/pytorch_tabnet-4.1.0-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:50:52.094668Z","iopub.execute_input":"2024-12-20T01:50:52.095302Z","iopub.status.idle":"2024-12-20T01:51:32.455833Z","shell.execute_reply.started":"2024-12-20T01:50:52.095272Z","shell.execute_reply":"2024-12-20T01:51:32.454563Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def feature_engineering(df):\n    # Mevcut özellik oluşturma işlemleri\n    \n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    \n   \n    df['FFMI_FMI_Avg'] = (df['FFMI_BFP'] + df['FMI_BFP']) / 2\n    df['BFP_BMR_BFP_DEE_Sum'] = df['BFP_BMR'] + df['BFP_DEE']\n    \n    #Güç Oranı (Dominant / Non-Dominant):\n    df['Grip_Strength_Ratio'] = df['FGC-FGC_GSD'] / df['FGC-FGC_GSND']\n    #Curl Up Performansı Yaşa Göre\n    df['Curl_Up_Age'] = df['FGC-FGC_CU'] * df['Basic_Demos-Age']\n    #Push-Up Performansı ve BMI:\n    df['Push_Up_BMI'] = df['FGC-FGC_PU'] / df['Physical-BMI']\n    #Esneklik Farkı (Sol / Sağ)\n    df['Flexibility_Diff'] = df['FGC-FGC_SRL'] - df['FGC-FGC_SRR']\n    #Fitness Seviyesi ile Kuvvet:\n    df['Grip_Strength_Fitness'] = df['FGC-FGC_GSD'] * (df['FGC-FGC_GSD_Zone'] + 1)\n    #BMI ile trunk lift oranını karşılaştırır.\n    df['Trunk_Lift_BMI'] = df['FGC-FGC_TL'] / df['Physical-BMI']\n\n    \n    df['Overall_Fitness_Zone_Avg'] = df[['FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone', \n                                     'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone', \n                                     'FGC-FGC_TL_Zone']].mean(axis=1)\n    \n    \n    \"\"\"    \n    df['Curl_Up_Fitness'] = df['FGC-FGC_CU'] * df['FGC-FGC_CU_Zone']\n    df['Grip_Strength_ND_Fitness'] = df['FGC-FGC_GSND'] * df['FGC-FGC_GSND_Zone']\n    df['Grip_Strength_D_Fitness'] = df['FGC-FGC_GSD'] * df['FGC-FGC_GSD_Zone']\n    df['Push_Up_Fitness'] = df['FGC-FGC_PU'] * df['FGC-FGC_PU_Zone']\n    df['Sit_Reach_Left_Fitness'] = df['FGC-FGC_SRL'] * df['FGC-FGC_SRL_Zone']\n    df['Sit_Reach_Right_Fitness'] = df['FGC-FGC_SRR'] * df['FGC-FGC_SRR_Zone']\n    df['Trunk_Lift_Fitness'] = df['FGC-FGC_TL'] * df['FGC-FGC_TL_Zone']\n    \"\"\"\n\n    \n    return df\n\n\n\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df\n\n\nclass AutoEncoder(nn.Module):\n    def __init__(self, input_dim, encoding_dim):\n        super(AutoEncoder, self).__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(input_dim, encoding_dim*3),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*3, encoding_dim*2),\n            nn.ReLU(),\n            nn.Linear(encoding_dim*2, encoding_dim),\n            nn.ReLU()\n        )\n        self.decoder = nn.Sequential(\n            nn.Linear(encoding_dim, input_dim*2),\n            nn.ReLU(),\n            nn.Linear(input_dim*2, input_dim*3),\n            nn.ReLU(),\n            nn.Linear(input_dim*3, input_dim),\n            nn.Sigmoid()\n        )\n        \n    def forward(self, x):\n        encoded = self.encoder(x)\n        decoded = self.decoder(encoded)\n        return decoded\n\n\ndef perform_autoencoder(df, encoding_dim=50, epochs=50, batch_size=32):\n    scaler = StandardScaler()\n    df_scaled = scaler.fit_transform(df)\n    \n    data_tensor = torch.FloatTensor(df_scaled)\n    \n    input_dim = data_tensor.shape[1]\n    autoencoder = AutoEncoder(input_dim, encoding_dim)\n    \n    criterion = nn.MSELoss()\n    optimizer = optim.Adam(autoencoder.parameters())\n    \n    for epoch in range(epochs):\n        for i in range(0, len(data_tensor), batch_size):\n            batch = data_tensor[i : i + batch_size]\n            optimizer.zero_grad()\n            reconstructed = autoencoder(batch)\n            loss = criterion(reconstructed, batch)\n            loss.backward()\n            optimizer.step()\n            \n        if (epoch + 1) % 10 == 0:\n            print(f'Epoch [{epoch + 1}/{epochs}], Loss: {loss.item():.4f}]')\n                 \n    with torch.no_grad():\n        encoded_data = autoencoder.encoder(data_tensor).numpy()\n        \n    df_encoded = pd.DataFrame(encoded_data, columns=[f'Enc_{i + 1}' for i in range(encoded_data.shape[1])])\n    \n    return df_encoded","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:32.457489Z","iopub.execute_input":"2024-12-20T01:51:32.457796Z","iopub.status.idle":"2024-12-20T01:51:32.478250Z","shell.execute_reply.started":"2024-12-20T01:51:32.457767Z","shell.execute_reply":"2024-12-20T01:51:32.477500Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_col  = \"sii\"","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:32.479368Z","iopub.execute_input":"2024-12-20T01:51:32.480023Z","iopub.status.idle":"2024-12-20T01:51:32.494588Z","shell.execute_reply.started":"2024-12-20T01:51:32.479984Z","shell.execute_reply":"2024-12-20T01:51:32.493743Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\n\"\"\"train_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\n\ndf_train = train_ts.drop('id', axis=1)\ndf_test = test_ts.drop('id', axis=1)\n\ntrain_ts_encoded = perform_autoencoder(df_train, encoding_dim=50, epochs=100, batch_size=32)\ntest_ts_encoded = perform_autoencoder(df_test, encoding_dim=50, epochs=100, batch_size=32)\n\ntime_series_cols = train_ts_encoded.columns.tolist()\ntrain_ts_encoded[\"id\"]=train_ts[\"id\"]\ntest_ts_encoded['id']=test_ts[\"id\"]\n\ntrain = pd.merge(train, train_ts_encoded, how=\"left\", on='id')\ntest = pd.merge(test, test_ts_encoded, how=\"left\", on='id')\"\"\"","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:32.495627Z","iopub.execute_input":"2024-12-20T01:51:32.495891Z","iopub.status.idle":"2024-12-20T01:51:32.577641Z","shell.execute_reply.started":"2024-12-20T01:51:32.495839Z","shell.execute_reply":"2024-12-20T01:51:32.576896Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\n#featuresCols += time_series_cols\n\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset=[\"sii\"]).reset_index(drop=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:32.578715Z","iopub.execute_input":"2024-12-20T01:51:32.579051Z","iopub.status.idle":"2024-12-20T01:51:32.602101Z","shell.execute_reply.started":"2024-12-20T01:51:32.579025Z","shell.execute_reply":"2024-12-20T01:51:32.601168Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_columns = ['Basic_Demos-Age', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', \n 'FGC-FGC_GSND', 'FGC-FGC_GSD', 'FGC-FGC_SRL', 'FGC-FGC_SRR', 'BIA-BIA_BMC', \n 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', \n 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_ICW', 'BIA-BIA_LDM', \n 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total','FFMI_FMI_Avg', 'BFP_BMR_BFP_DEE_Sum', 'Grip_Strength_Ratio', \n    'Curl_Up_Age', 'Push_Up_BMI', 'Flexibility_Diff', 'Grip_Strength_Fitness', \n    'Trunk_Lift_BMI','Enc_1',\n]\n\ncat_cols = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n 'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', 'PAQ_A-Season', \n 'PAQ_C-Season', 'PCIAT-Season', 'SDS-Season', 'PreInt_EduHx-Season','Basic_Demos-Sex', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone', \n 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL_Zone', \n 'BIA-BIA_Activity_Level_num', 'BIA-BIA_Frame_num', 'PreInt_EduHx-computerinternet_hoursday','Overall_Fitness_Zone_Avg']\n\n","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:32.606007Z","iopub.execute_input":"2024-12-20T01:51:32.606261Z","iopub.status.idle":"2024-12-20T01:51:32.611764Z","shell.execute_reply.started":"2024-12-20T01:51:32.606237Z","shell.execute_reply":"2024-12-20T01:51:32.610966Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def updateCats(df):\n    global cat_cols\n    for c in cat_cols: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        ","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:32.612789Z","iopub.execute_input":"2024-12-20T01:51:32.613138Z","iopub.status.idle":"2024-12-20T01:51:32.626802Z","shell.execute_reply.started":"2024-12-20T01:51:32.613101Z","shell.execute_reply":"2024-12-20T01:51:32.626153Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ntrain = feature_engineering(train)\ntest  = feature_engineering(test)","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:32.627651Z","iopub.execute_input":"2024-12-20T01:51:32.627881Z","iopub.status.idle":"2024-12-20T01:51:32.656741Z","shell.execute_reply.started":"2024-12-20T01:51:32.627833Z","shell.execute_reply":"2024-12-20T01:51:32.655892Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"train = updateCats(train)\ntest  = updateCats(test) \"\"\"\n\n\nnumerical_columns = train.select_dtypes(include=['int64', 'float64']).columns\nnumerical_columns = numerical_columns.drop('sii') \ntrain[numerical_columns] = train[numerical_columns].replace([np.inf, -np.inf], np.nan)\ntest[numerical_columns] = test[numerical_columns].replace([np.inf, -np.inf], np.nan)\n\nscaler = StandardScaler()\n\ntrain[numerical_columns] = scaler.fit_transform(train[numerical_columns])\ntest[numerical_columns] = scaler.transform(test[numerical_columns])\n","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:32.657613Z","iopub.execute_input":"2024-12-20T01:51:32.657874Z","iopub.status.idle":"2024-12-20T01:51:32.711499Z","shell.execute_reply.started":"2024-12-20T01:51:32.657826Z","shell.execute_reply":"2024-12-20T01:51:32.710888Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1: Identify columns with less than 60% missing values\ncolumns_to_keep = train.columns[train.isnull().mean() < 0.8]\n\n# Step 2: Ensure that 'sii' is included even if it has more missing values\nif 'sii' not in columns_to_keep:\n    columns_to_keep = columns_to_keep.append(pd.Index(['sii']))\n\n# Step 3: Filter both train and test datasets based on these columns\ntrain = train[columns_to_keep]  # Keep only the desired columns in the training set\ntest = test[columns_to_keep.drop('sii')]  # Apply the same to test set, excluding 'sii'","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:32.712501Z","iopub.execute_input":"2024-12-20T01:51:32.712776Z","iopub.status.idle":"2024-12-20T01:51:32.730149Z","shell.execute_reply.started":"2024-12-20T01:51:32.712752Z","shell.execute_reply":"2024-12-20T01:51:32.729456Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert columns_to_keep to a list before using set operations\ndropped_columns = set(numerical_columns.tolist() + cat_cols) - set(columns_to_keep.tolist())\n\n# Remove the dropped columns from numerical_columns and cat_cols\nnumerical_columns = [col for col in numerical_columns if col not in dropped_columns]\ncat_cols = [col for col in cat_cols if col not in dropped_columns]\n","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:32.731187Z","iopub.execute_input":"2024-12-20T01:51:32.731580Z","iopub.status.idle":"2024-12-20T01:51:32.742503Z","shell.execute_reply.started":"2024-12-20T01:51:32.731551Z","shell.execute_reply":"2024-12-20T01:51:32.741734Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\"\"\"num_imputer = SimpleImputer(strategy='mean')\ntrain[numerical_columns] = num_imputer.fit_transform(train[numerical_columns])\ntest[numerical_columns] = num_imputer.transform(test[numerical_columns])\"\"\"\n\nknn_imputer = KNNImputer(n_neighbors=5)\n\n# Apply KNN imputation to numerical columns\ntrain[numerical_columns] = knn_imputer.fit_transform(train[numerical_columns])\ntest[numerical_columns] = knn_imputer.fit_transform(test[numerical_columns])\n\n\n\ncat_imputer = SimpleImputer(strategy='most_frequent')\ntrain[cat_cols] = cat_imputer.fit_transform(train[cat_cols])\ntest[cat_cols] = cat_imputer.fit_transform(test[cat_cols])","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:32.743501Z","iopub.execute_input":"2024-12-20T01:51:32.744318Z","iopub.status.idle":"2024-12-20T01:51:35.497815Z","shell.execute_reply.started":"2024-12-20T01:51:32.744281Z","shell.execute_reply":"2024-12-20T01:51:35.497163Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"season_features = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n                   'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n                   'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\nseason_F = [col for col in season_features if col not in dropped_columns]","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.498709Z","iopub.execute_input":"2024-12-20T01:51:35.498977Z","iopub.status.idle":"2024-12-20T01:51:35.503127Z","shell.execute_reply.started":"2024-12-20T01:51:35.498951Z","shell.execute_reply":"2024-12-20T01:51:35.502286Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_cols = [col for col in cat_cols if col not in season_F]","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.504150Z","iopub.execute_input":"2024-12-20T01:51:35.504462Z","iopub.status.idle":"2024-12-20T01:51:35.514831Z","shell.execute_reply.started":"2024-12-20T01:51:35.504429Z","shell.execute_reply":"2024-12-20T01:51:35.514058Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\nfrom sklearn.preprocessing import LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.515935Z","iopub.execute_input":"2024-12-20T01:51:35.516219Z","iopub.status.idle":"2024-12-20T01:51:35.524702Z","shell.execute_reply.started":"2024-12-20T01:51:35.516182Z","shell.execute_reply":"2024-12-20T01:51:35.524004Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"low_unique_cols = [col for col in cat_cols if train[col].nunique() <= 7]","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.525684Z","iopub.execute_input":"2024-12-20T01:51:35.526022Z","iopub.status.idle":"2024-12-20T01:51:35.539231Z","shell.execute_reply.started":"2024-12-20T01:51:35.525987Z","shell.execute_reply":"2024-12-20T01:51:35.538580Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.columns","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.540249Z","iopub.execute_input":"2024-12-20T01:51:35.540787Z","iopub.status.idle":"2024-12-20T01:51:35.553378Z","shell.execute_reply.started":"2024-12-20T01:51:35.540749Z","shell.execute_reply":"2024-12-20T01:51:35.552690Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nencoder = OneHotEncoder(drop='first', sparse=False, handle_unknown='ignore')\n\n# 7 veya daha az benzersiz değeri olan sütunları belirleme\nlow_unique_cols = [col for col in train.columns if train[col].nunique() <= 7 and train[col].dtype == 'category']\n\n# Train verisine encoding uygulama\ntrain_encoded = pd.DataFrame(encoder.fit_transform(train[low_unique_cols]), columns=encoder.get_feature_names_out(low_unique_cols))\ntrain = train.drop(columns=low_unique_cols).reset_index(drop=True)  # Orijinal sütunları kaldırıyoruz\ntrain = pd.concat([train, train_encoded], axis=1)  # Encoded sütunları ekliyoruz\n\n# Test verisine aynı encoder ile encoding uygulama\ntest_encoded = pd.DataFrame(encoder.transform(test[low_unique_cols]), columns=encoder.get_feature_names_out(low_unique_cols))\ntest = test.drop(columns=low_unique_cols).reset_index(drop=True)  # Orijinal sütunları kaldırıyoruz\ntest = pd.concat([test, test_encoded], axis=1)  # Encoded sütunları ekliyoruz\n\n\n# OrdinalEncoder başlatma\nencoder = OrdinalEncoder(\n    dtype=np.int32,\n    handle_unknown='use_encoded_value',\n    unknown_value=-1\n)\n\n# 7'den fazla benzersiz değeri olan kategorik sütunları belirleme\nmore_7_features = [col for col in cat_cols if train[col].nunique() > 7 and train[col].dtype == 'category']\n\n# Train setine Ordinal Encoding uygulama\ntrain[more_7_features] = encoder.fit_transform(train[more_7_features])\ntrain[more_7_features] = train[more_7_features].astype('category')\n\n# Test setine aynı encoder ile Ordinal Encoding uygulama\ntest[more_7_features] = encoder.transform(test[more_7_features])\ntest[more_7_features] = test[more_7_features].astype('category')\n    \n\n#removing season features\n\"\"\"train = train.drop(columns=season_features)\ntest = test.drop(columns=season_features)\"\"\"\n# Sadece season özelliklerine LabelEncoder uygulayın\nfor col in season_F:\n    le = LabelEncoder()\n    train[col] = le.fit_transform(train[col])\n    test[col] = le.transform(test[col])","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.554550Z","iopub.execute_input":"2024-12-20T01:51:35.555030Z","iopub.status.idle":"2024-12-20T01:51:35.609145Z","shell.execute_reply.started":"2024-12-20T01:51:35.554993Z","shell.execute_reply":"2024-12-20T01:51:35.608524Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"object_columns = ['Basic_Demos-Sex', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone', \n                  'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL_Zone', \n                  'BIA-BIA_Activity_Level_num', 'BIA-BIA_Frame_num', 'PreInt_EduHx-computerinternet_hoursday','Overall_Fitness_Zone_Avg']\n\nfor col in object_columns:\n    train[col] = train[col].astype('category')\n    test[col] = test[col].astype('category')","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.609968Z","iopub.execute_input":"2024-12-20T01:51:35.610220Z","iopub.status.idle":"2024-12-20T01:51:35.634277Z","shell.execute_reply.started":"2024-12-20T01:51:35.610186Z","shell.execute_reply":"2024-12-20T01:51:35.633560Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\nimport numpy as np\nimport pandas as pd\n\ndef add_outlier_flag(train_df, test_df, target_col='sii', outlier_threshold=3):\n    # Target sütununu ayırma\n    train_target = train_df[target_col].reset_index(drop=True)  # Target değerini saklama\n    train_df = train_df.drop(columns=[target_col])  # Target değerini veri setinden çıkarma\n    tr  = train_df.copy()\n    ts  = test_df.copy()\n    scaler = StandardScaler()\n\n    train_df[numerical_columns] = scaler.fit_transform(train_df[numerical_columns])\n    test_df[numerical_columns] = scaler.transform(test_df[numerical_columns])\n    # PCA işlemi (train verisi ile fit, hem train hem test için transform yapılır)\n    pca = PCA(n_components=2)\n    train_pca = pca.fit_transform(train_df)\n    test_pca = pca.transform(test_df)\n    \n    # PCA sonuçlarını DataFrame'e dönüştürme\n    train_pca_df = pd.DataFrame(train_pca, columns=['PC1', 'PC2'])\n    test_pca_df = pd.DataFrame(test_pca, columns=['PC1', 'PC2'])\n    \n    # Uç nokta belirleme (Z-score yöntemi)\n    train_z_scores = np.abs((train_pca_df - train_pca_df.mean()) / train_pca_df.std())\n    test_z_scores = np.abs((test_pca_df - train_pca_df.mean()) / train_pca_df.std())  # Test için train ortalama ve std kullanılır\n\n    # Outlier threshold kullanarak uç noktaları belirleme\n    train_outliers = (train_z_scores['PC1'] > outlier_threshold) | (train_z_scores['PC2'] > outlier_threshold)\n    test_outliers = (test_z_scores['PC1'] > outlier_threshold) | (test_z_scores['PC2'] > outlier_threshold)\n    \n    # Orijinal veri setlerine is_outlier sütununu ekleme\n    \n    tr['is_outlier'] = train_outliers.astype(int)\n    ts['is_outlier'] = test_outliers.astype(int)\n    \n    \n    tr[target_col] = train_target  # Target sütunu geri ekleme\n    \n    # Target sütununu en sona getirme\n    tr = tr[[col for col in tr.columns if col != target_col] + [target_col]]\n    \n    return tr, ts\n\n# Train ve test veri setlerine is_outlier özelliği ekleme\n#train, test = add_outlier_flag(train, test)","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.635278Z","iopub.execute_input":"2024-12-20T01:51:35.635529Z","iopub.status.idle":"2024-12-20T01:51:35.649635Z","shell.execute_reply.started":"2024-12-20T01:51:35.635505Z","shell.execute_reply":"2024-12-20T01:51:35.648890Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.653327Z","iopub.execute_input":"2024-12-20T01:51:35.653574Z","iopub.status.idle":"2024-12-20T01:51:35.702793Z","shell.execute_reply.started":"2024-12-20T01:51:35.653550Z","shell.execute_reply":"2024-12-20T01:51:35.702008Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SEED = 42","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.703685Z","iopub.execute_input":"2024-12-20T01:51:35.703952Z","iopub.status.idle":"2024-12-20T01:51:35.707555Z","shell.execute_reply.started":"2024-12-20T01:51:35.703928Z","shell.execute_reply":"2024-12-20T01:51:35.706673Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def quadratic_weighted_kappa(estimator, X, y_true):\n    y_pred = estimator.predict(X).round()\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.708492Z","iopub.execute_input":"2024-12-20T01:51:35.708706Z","iopub.status.idle":"2024-12-20T01:51:35.716294Z","shell.execute_reply.started":"2024-12-20T01:51:35.708685Z","shell.execute_reply":"2024-12-20T01:51:35.715380Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def threshold_rounder(y_pred, thresholds):\n    return np.where(y_pred < thresholds[0], 0,\n                    np.where(y_pred < thresholds[1], 1,\n                             np.where(y_pred < thresholds[2], 2, 3)))","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.717324Z","iopub.execute_input":"2024-12-20T01:51:35.718016Z","iopub.status.idle":"2024-12-20T01:51:35.728593Z","shell.execute_reply.started":"2024-12-20T01:51:35.717978Z","shell.execute_reply":"2024-12-20T01:51:35.727890Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def eval_preds(thresholds, y_true, y_pred):\n    y_pred = threshold_rounder(y_pred, thresholds)\n    score = cohen_kappa_score(y_true, y_pred, weights='quadratic')\n    return -score","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.729469Z","iopub.execute_input":"2024-12-20T01:51:35.729775Z","iopub.status.idle":"2024-12-20T01:51:35.740340Z","shell.execute_reply.started":"2024-12-20T01:51:35.729740Z","shell.execute_reply":"2024-12-20T01:51:35.739593Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import make_scorer\nKAPPA_SCORER = make_scorer(\n    cohen_kappa_score, \n    greater_is_better=True, \n    weights='quadratic',\n)","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.741307Z","iopub.execute_input":"2024-12-20T01:51:35.741625Z","iopub.status.idle":"2024-12-20T01:51:35.752110Z","shell.execute_reply.started":"2024-12-20T01:51:35.741590Z","shell.execute_reply":"2024-12-20T01:51:35.751401Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CustomLGBMRegressor(lgb.LGBMRegressor):\n    '''\n    Custom LightGBM Regressor\n    \n    It optimizes threshold values during fitting.\n    Main goal is preventing overfit on validation data.\n    '''\n    def fit(self, X, y, **kwargs):\n        super().fit(X, y, **kwargs)\n        y_pred = super().predict(X, **kwargs)\n        \n        self.optimizer = minimize(\n            eval_preds, \n            x0=[0.5, 1.5, 2.5], \n            args=(y, y_pred), \n            method='Nelder-Mead',\n        )\n        \n    def predict(self, X, **kwargs):\n        y_pred = super().predict(X, **kwargs)\n        y_pred = threshold_rounder(y_pred, self.optimizer.x)\n        return y_pred","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.753066Z","iopub.execute_input":"2024-12-20T01:51:35.753363Z","iopub.status.idle":"2024-12-20T01:51:35.761156Z","shell.execute_reply.started":"2024-12-20T01:51:35.753317Z","shell.execute_reply":"2024-12-20T01:51:35.760369Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import optuna\nfrom xgboost import XGBRegressor\nfrom sklearn.model_selection import cross_val_score, KFold\nimport numpy as np\n\ndef xgb_objective(trial):\n    # Hyperparameter tuning ranges\n    params = {\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.1, log=True),\n        'max_depth': trial.suggest_int('max_depth', 3, 10),\n        'n_estimators': trial.suggest_int('n_estimators', 50, 500),\n        'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n        'reg_alpha': trial.suggest_float('reg_alpha', 0.0, 10.0),\n        'reg_lambda': trial.suggest_float('reg_lambda', 0.0, 10.0),\n        'random_state': SEED,\n        'tree_method': 'gpu_hist',  # If using GPU\n         'enable_categorical': True\n    }\n\n    # Define the model\n    estimator = XGBRegressor(**params)\n    X = train.drop([\"sii\"], axis=1)\n    y = train[target_col]\n\n    # Check for NaNs in the dataset\n    assert not X.isnull().values.any(), \"X contains NaN values\"\n    assert not y.isnull().values.any(), \"y contains NaN values\"\n\n    # Use KFold for cross-validation in regression tasks\n    cv = KFold(5, shuffle=True, random_state=SEED)\n    val_scores = cross_val_score(estimator=estimator, X=X, y=y, cv=cv, scoring=KAPPA_SCORER)\n\n    # Return the mean validation score\n    return np.mean(val_scores)\n","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.762181Z","iopub.execute_input":"2024-12-20T01:51:35.762433Z","iopub.status.idle":"2024-12-20T01:51:35.774512Z","shell.execute_reply.started":"2024-12-20T01:51:35.762408Z","shell.execute_reply":"2024-12-20T01:51:35.773815Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def lgb_objective(trial):\n    # Hyperparameters to be tuned\n    params = {\n        'objective':         'l2',\n        'verbosity':         -1,\n        'random_state':      SEED,\n        'boosting_type':     'gbdt',\n        'lambda_l1':         trial.suggest_float('lambda_l1', 1e-3, 10.0, log=True),\n        'lambda_l2':         trial.suggest_float('lambda_l2', 1e-3, 10.0, log=True),\n        'learning_rate':     trial.suggest_float('learning_rate', 1e-2, 1e-1, log=True),\n        'max_depth':         trial.suggest_int('max_depth', 4, 8),\n        'num_leaves':        trial.suggest_int('num_leaves', 16, 256),\n        'colsample_bytree':  trial.suggest_float('colsample_bytree', 0.4, 1.0),\n        'colsample_bynode':  trial.suggest_float('colsample_bynode', 0.4, 1.0),\n        'bagging_fraction':  trial.suggest_float('bagging_fraction', 0.4, 1.0),\n        'bagging_freq':      trial.suggest_int('bagging_freq', 1, 7),\n        'min_data_in_leaf':  trial.suggest_int('min_data_in_leaf', 5, 100),\n    }\n\n    # n_iter (num_boost_round) için bir aralık seçiyoruz\n    n_iter = trial.suggest_int('n_iter', 50, 500)\n    \n    # Veri bölme\n    X = train.drop([\"sii\"], axis=1)\n    y = train[target_col]\n\n    # Modeli n_iter parametresi ile oluşturma\n    estimator = CustomLGBMRegressor(**params, n_estimators=n_iter)\n    \n    # Stratified K-Fold ile cross-validation\n    cv = StratifiedKFold(5, shuffle=True, random_state=SEED)\n    val_scores = cross_val_score(\n        estimator=estimator, \n        X=X, y=y, \n        cv=cv, \n        scoring=KAPPA_SCORER,\n    )\n\n    return np.mean(val_scores)\n","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.775366Z","iopub.execute_input":"2024-12-20T01:51:35.775678Z","iopub.status.idle":"2024-12-20T01:51:35.788295Z","shell.execute_reply.started":"2024-12-20T01:51:35.775644Z","shell.execute_reply":"2024-12-20T01:51:35.787421Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"do_tunin_xgb = False","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.789254Z","iopub.execute_input":"2024-12-20T01:51:35.789594Z","iopub.status.idle":"2024-12-20T01:51:35.800721Z","shell.execute_reply.started":"2024-12-20T01:51:35.789570Z","shell.execute_reply":"2024-12-20T01:51:35.800054Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if do_tunin_xgb:  \n    study = optuna.create_study(direction='maximize', study_name='classification')\n    study.optimize(xgb_objective, n_trials=100, show_progress_bar=True)","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.801569Z","iopub.execute_input":"2024-12-20T01:51:35.801801Z","iopub.status.idle":"2024-12-20T01:51:35.812302Z","shell.execute_reply.started":"2024-12-20T01:51:35.801778Z","shell.execute_reply":"2024-12-20T01:51:35.811587Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"do_tunin_lgb = False","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.813178Z","iopub.execute_input":"2024-12-20T01:51:35.813411Z","iopub.status.idle":"2024-12-20T01:51:35.823039Z","shell.execute_reply.started":"2024-12-20T01:51:35.813388Z","shell.execute_reply":"2024-12-20T01:51:35.822255Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if do_tunin_lgb:  \n    study = optuna.create_study(direction='maximize', study_name='Regressor')\n    study.optimize(lgb_objective, n_trials=100, show_progress_bar=True)","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.824050Z","iopub.execute_input":"2024-12-20T01:51:35.824292Z","iopub.status.idle":"2024-12-20T01:51:35.833282Z","shell.execute_reply.started":"2024-12-20T01:51:35.824270Z","shell.execute_reply":"2024-12-20T01:51:35.832655Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.base import BaseEstimator, RegressorMixin\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom pytorch_tabnet.callbacks import Callback\nimport os\nimport torch\nfrom pytorch_tabnet.callbacks import Callback\nfrom pytorch_tabnet.tab_model import TabNetRegressor\n\n\nclass TabNetWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, **kwargs):\n        self.model = TabNetRegressor(**kwargs)\n        self.kwargs = kwargs\n        self.imputer = SimpleImputer(strategy='median')\n        self.best_model_path = 'best_tabnet_model.pt'\n        \n    def fit(self, X, y):\n        # Handle missing values\n        X_imputed = self.imputer.fit_transform(X)\n        \n        if hasattr(y, 'values'):\n            y = y.values\n            \n        # Create internal validation set\n        X_train, X_valid, y_train, y_valid = train_test_split(\n            X_imputed, \n            y, \n            test_size=0.2,\n            random_state=42\n        )\n        \n        # Train TabNet model\n        history = self.model.fit(\n            X_train=X_train,\n            y_train=y_train.reshape(-1, 1),\n            eval_set=[(X_valid, y_valid.reshape(-1, 1))],\n            eval_name=['valid'],\n            eval_metric=['mse'],\n            max_epochs=500,\n            patience=50,\n            batch_size=1024,\n            virtual_batch_size=128,\n            num_workers=0,\n            drop_last=False,\n            callbacks=[\n                TabNetPretrainedModelCheckpoint(\n                    filepath=self.best_model_path,\n                    monitor='valid_mse',\n                    mode='min',\n                    save_best_only=True,\n                    verbose=True\n                )\n            ]\n        )\n        \n        # Load the best model\n        if os.path.exists(self.best_model_path):\n            self.model.load_model(self.best_model_path)\n            os.remove(self.best_model_path)  # Remove temporary file\n        \n        return self\n    \n    def predict(self, X):\n        X_imputed = self.imputer.transform(X)\n        return self.model.predict(X_imputed).flatten()\n    \n    def __deepcopy__(self, memo):\n        # Add deepcopy support for scikit-learn\n        cls = self.__class__\n        result = cls.__new__(cls)\n        memo[id(self)] = result\n        for k, v in self.__dict__.items():\n            setattr(result, k, deepcopy(v, memo))\n        return result\n\n# TabNet hyperparameters\nTabNet_Params = {\n    'n_d': 64,              # Width of the decision prediction layer\n    'n_a': 64,              # Width of the attention embedding for each step\n    'n_steps': 5,           # Number of steps in the architecture\n    'gamma': 1.5,           # Coefficient for feature selection regularization\n    'n_independent': 2,     # Number of independent GLU layer in each GLU block\n    'n_shared': 2,          # Number of shared GLU layer in each GLU block\n    'lambda_sparse': 1e-4,  # Sparsity regularization\n    'optimizer_fn': torch.optim.Adam,\n    'optimizer_params': dict(lr=2e-2, weight_decay=1e-5),\n    'mask_type': 'entmax',\n    'scheduler_params': dict(mode=\"min\", patience=10, min_lr=1e-5, factor=0.5),\n    'scheduler_fn': torch.optim.lr_scheduler.ReduceLROnPlateau,\n    'verbose': 1,\n    'device_name': 'cuda' if torch.cuda.is_available() else 'cpu'\n}\n\nclass TabNetPretrainedModelCheckpoint(Callback):\n    def __init__(self, filepath, monitor='val_loss', mode='min', \n                 save_best_only=True, verbose=1):\n        super().__init__()  # Initialize parent class\n        self.filepath = filepath\n        self.monitor = monitor\n        self.mode = mode\n        self.save_best_only = save_best_only\n        self.verbose = verbose\n        self.best = float('inf') if mode == 'min' else -float('inf')\n        \n    def on_train_begin(self, logs=None):\n        self.model = self.trainer  # Use trainer itself as model\n        \n    def on_epoch_end(self, epoch, logs=None):\n        logs = logs or {}\n        current = logs.get(self.monitor)\n        if current is None:\n            return\n        \n        # Check if current metric is better than best\n        if (self.mode == 'min' and current < self.best) or \\\n           (self.mode == 'max' and current > self.best):\n            if self.verbose:\n                print(f'\\nEpoch {epoch}: {self.monitor} improved from {self.best:.4f} to {current:.4f}')\n            self.best = current\n            if self.save_best_only:\n                self.model.save_model(self.filepath)  # Save the entire model","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.834321Z","iopub.execute_input":"2024-12-20T01:51:35.834580Z","iopub.status.idle":"2024-12-20T01:51:35.890391Z","shell.execute_reply.started":"2024-12-20T01:51:35.834532Z","shell.execute_reply":"2024-12-20T01:51:35.889783Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_splits = 5","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.891253Z","iopub.execute_input":"2024-12-20T01:51:35.891508Z","iopub.status.idle":"2024-12-20T01:51:35.895449Z","shell.execute_reply.started":"2024-12-20T01:51:35.891483Z","shell.execute_reply":"2024-12-20T01:51:35.894508Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    \n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead') # Nelder-Mead | # Powell\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission,model","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.896666Z","iopub.execute_input":"2024-12-20T01:51:35.896947Z","iopub.status.idle":"2024-12-20T01:51:35.908065Z","shell.execute_reply.started":"2024-12-20T01:51:35.896922Z","shell.execute_reply":"2024-12-20T01:51:35.907196Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LParams = {\n  'verbose' : -1,\n    'lambda_l1': 6.370803099031535,\n    'lambda_l2': 0.35820072977715545,\n    'learning_rate': 0.0416445038025349,\n    'max_depth': 10,\n    'num_leaves': 449,\n    'min_data_in_leaf': 38,\n    'feature_fraction': 0.8369460932871426,\n    'bagging_fraction': 0.4745752661386946,\n    'bagging_freq': 1\n}\n\n\nCatBoost_Params = {\n    'learning_rate': 0.05,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': SEED,\n    'verbose': 0,\n    'l2_leaf_reg': 10,  # Increase this value\n    'task_type': 'GPU',\n    'cat_features': cat_cols,\n\n}\n\n\nXGB_Params = {\n    'learning_rate': 0.05,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'reg_alpha': 1,  # Increased from 0.1\n    'reg_lambda': 5,  # Increased from 1\n    'random_state': SEED,\n    'tree_method': 'gpu_hist',\n    'enable_categorical': True,\n\n}","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.908914Z","iopub.execute_input":"2024-12-20T01:51:35.909123Z","iopub.status.idle":"2024-12-20T01:51:35.921671Z","shell.execute_reply.started":"2024-12-20T01:51:35.909102Z","shell.execute_reply":"2024-12-20T01:51:35.920902Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.922625Z","iopub.execute_input":"2024-12-20T01:51:35.923011Z","iopub.status.idle":"2024-12-20T01:51:35.931890Z","shell.execute_reply.started":"2024-12-20T01:51:35.922974Z","shell.execute_reply":"2024-12-20T01:51:35.931138Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Light = CustomLGBMRegressor(**LParams)\nXGB_Model = xgb.XGBRegressor(**XGB_Params)\nTabNet_Model = TabNetWrapper(**TabNet_Params)","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.933113Z","iopub.execute_input":"2024-12-20T01:51:35.933444Z","iopub.status.idle":"2024-12-20T01:51:35.944795Z","shell.execute_reply.started":"2024-12-20T01:51:35.933409Z","shell.execute_reply":"2024-12-20T01:51:35.944082Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = VotingRegressor([\n    ('lgb_0', CustomLGBMRegressor(**LParams, random_state=12)),\n    ('lgb_1', CustomLGBMRegressor(**LParams, random_state=22)),\n    ('lgb_2', CustomLGBMRegressor(**LParams, random_state=32)),\n    ('lgb_3', CustomLGBMRegressor(**LParams, random_state=42)),\n    ('lgb_7', CustomLGBMRegressor(**LParams, random_state=52)),\n    ('lgb_8', CustomLGBMRegressor(**LParams, random_state=62)),\n    ('lgb_9', CustomLGBMRegressor(**LParams, random_state=72)),\n    ('lgb_10', CustomLGBMRegressor(**LParams, random_state=82)),\n    ('lgb_11', CustomLGBMRegressor(**LParams, random_state=92)),\n    ('lgb_12', CustomLGBMRegressor(**LParams, random_state=102))\n])","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.945646Z","iopub.execute_input":"2024-12-20T01:51:35.945907Z","iopub.status.idle":"2024-12-20T01:51:35.957729Z","shell.execute_reply.started":"2024-12-20T01:51:35.945878Z","shell.execute_reply":"2024-12-20T01:51:35.956896Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"voting_model = VotingRegressor(estimators=[\n    ('lightgbm', model),\n    ('xgboost', XGB_Model),\n   \n])","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.958831Z","iopub.execute_input":"2024-12-20T01:51:35.959179Z","iopub.status.idle":"2024-12-20T01:51:35.966670Z","shell.execute_reply.started":"2024-12-20T01:51:35.959143Z","shell.execute_reply":"2024-12-20T01:51:35.966013Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission,model = TrainML(voting_model,test)","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:51:35.967459Z","iopub.execute_input":"2024-12-20T01:51:35.967671Z","iopub.status.idle":"2024-12-20T01:53:02.123181Z","shell.execute_reply.started":"2024-12-20T01:51:35.967650Z","shell.execute_reply":"2024-12-20T01:53:02.122327Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:53:02.124393Z","iopub.execute_input":"2024-12-20T01:53:02.125232Z","iopub.status.idle":"2024-12-20T01:53:02.133669Z","shell.execute_reply.started":"2024-12-20T01:53:02.125190Z","shell.execute_reply":"2024-12-20T01:53:02.132888Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Submission.to_csv('submission.csv', index=False)\nprint(Submission['sii'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2024-12-20T01:53:02.134753Z","iopub.execute_input":"2024-12-20T01:53:02.135021Z","iopub.status.idle":"2024-12-20T01:53:02.152685Z","shell.execute_reply.started":"2024-12-20T01:53:02.134997Z","shell.execute_reply":"2024-12-20T01:53:02.151907Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}