{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <p style=\"background-color:#6A1B9A; font-family:'Dancing Script', cursive; color:#FFFFFF; font-size:150%; text-align:center; border: 3px solid #FFEB3B; border-radius:40px; padding: 10px;\">Child Mind Institute | SIngleLGBM</p>","metadata":{}},{"cell_type":"markdown","source":"drop(['step', 'non-wear_flag', 'battery_voltage', 'time_of_day', 'weekday', 'relative_date_PCIAT']","metadata":{}},{"cell_type":"markdown","source":"featuresCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n       'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n       'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n       'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n       'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n       'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n       'FGC-Season', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU_Zone','FGC-FGC_SRL_Zone',\n       'FGC-FGC_SRR_Zone','FGC-FGC_TL_Zone', 'BIA-Season',\n       'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n       'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n       'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Frame_num',\n       'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n       'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n       'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n       'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n       'PreInt_EduHx-computerinternet_hoursday']\n       \n cat_c = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season','Fitness_Endurance-Season','FGC-Season',\n 'BIA-Season','SDS-Season','PreInt_EduHx-Season']\n \n \n ","metadata":{}},{"cell_type":"markdown","source":"group wise filled the following\n\n['Physical-Systolic_BP', 'Physical-Diastolic_BP', 'Physical-HeartRate', \n                'SDS-SDS_Total_T', 'SDS-SDS_Total_Raw', 'Physical-BMI', \n                'Physical-Height', 'Physical-Weight']","metadata":{}},{"cell_type":"markdown","source":"remove wrist diameter","metadata":{}},{"cell_type":"markdown","source":"Combine PAQ_AC_Total\n\ndrop index=[2065, 3205]\n\nclip BIA-BIA_BMC and BIA-BIA_BMR\n\nadd Blank_Fitness_Endurance\n\nadd Difference_populationweight\n\nbound heartrate","metadata":{}},{"cell_type":"markdown","source":"cv=10","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nimport pandas as pd\nfrom sklearn.base import clone\nfrom copy import deepcopy\nimport optuna\nfrom scipy.optimize import minimize\nimport os\n\nimport re\nfrom colorama import Fore, Style\n\nfrom tqdm import tqdm\nfrom IPython.display import clear_output\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport warnings\nwarnings.filterwarnings('ignore')\npd.options.display.max_columns = None\n\nimport lightgbm as lgb\nfrom catboost import CatBoostRegressor, CatBoostClassifier\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\n\nSEED = 42\nn_splits = 10","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-07T15:54:46.968479Z","iopub.execute_input":"2024-10-07T15:54:46.969464Z","iopub.status.idle":"2024-10-07T15:54:50.871389Z","shell.execute_reply.started":"2024-10-07T15:54:46.969370Z","shell.execute_reply":"2024-10-07T15:54:50.870164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntrain = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop(['step', 'non-wear_flag', 'battery_voltage', 'time_of_day', 'weekday', 'relative_date_PCIAT'], axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    \n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df\n        \ntrain_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")\ntime_series_cols = train_ts.columns.tolist()\ntime_series_cols.remove(\"id\")\n\ntrain = pd.merge(train, train_ts, how=\"left\", on='id')\ntest = pd.merge(test, test_ts, how=\"left\", on='id')\n\ntrain = train.drop('id',axis=1)\ntest = test.drop('id',axis=1)\n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n       'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n       'Physical-Height', 'Physical-Weight',\n       'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n       'Fitness_Endurance-Max_Stage',#'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec','Fitness_Endurance-Season', \n       'FGC-Season', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU_Zone','FGC-FGC_SRL_Zone',\n       'FGC-FGC_SRR_Zone','FGC-FGC_TL_Zone', 'BIA-Season',\n       'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n       'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n       'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Frame_num',\n       'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n       'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n       'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n       'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n       'PreInt_EduHx-computerinternet_hoursday']\n\nfeaturesCols += time_series_cols\n\ntrain = train[featuresCols + ['sii']]\ntest = test[featuresCols]\n\ncat_c = ['Basic_Demos-Enroll_Season','CGAS-Season','Physical-Season','FGC-Season', #'Fitness_Endurance-Season',\n 'BIA-Season','SDS-Season','PreInt_EduHx-Season']\n\ndef update(df):\n    \n    global cat_c\n    for c in cat_c : \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n        \n    return df\n        \ntrain = update(train)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:54:50.873961Z","iopub.execute_input":"2024-10-07T15:54:50.874765Z","iopub.status.idle":"2024-10-07T15:55:58.769486Z","shell.execute_reply.started":"2024-10-07T15:54:50.874707Z","shell.execute_reply":"2024-10-07T15:55:58.768085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <p style=\"background-color:#6A1B9A; font-family:'Dancing Script', cursive; color:#FFFFFF; font-size:120%; text-align:center; border: 3px solid #FFEB3B; border-radius:40px; padding: 10px;\">Basic Preprocess</p>","metadata":{}},{"cell_type":"code","source":"train.drop(index=[2065, 3205], inplace=True)   # Drop the row with index 2065\ntrain.reset_index(drop=True, inplace=True)  # Reset the index to avoid gaps\n","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:55:58.770868Z","iopub.execute_input":"2024-10-07T15:55:58.771279Z","iopub.status.idle":"2024-10-07T15:55:58.780563Z","shell.execute_reply.started":"2024-10-07T15:55:58.771233Z","shell.execute_reply":"2024-10-07T15:55:58.779141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# List of columns you want to fill based on Basic_Demos-Sex and Basic_Demos-Age\ncols_to_fill = ['Physical-Systolic_BP', 'Physical-Diastolic_BP', 'Physical-HeartRate', \n                'Physical-BMI', 'Physical-Height', 'Physical-Weight']\n\n# Function to calculate group-wise means from the training dataset\ndef calculate_group_means(train_df, cols, group_cols):\n    group_means = {}\n    for col in cols:\n        # Calculate mean for each group\n        means = train_df.groupby(group_cols)[col].mean()\n        group_means[col] = means.to_dict()  # Store as a dictionary\n    return group_means\n\n# Function to fill NaNs based on calculated group means\ndef fill_with_group_means(df, group_means, group_cols):\n    for col, means in group_means.items():\n        df[col] = df.apply(\n            lambda row: means.get((row[group_cols[0]], row[group_cols[1]]), row[col]), axis=1\n        )\n    return df\n\n# Calculate group means using the training dataset\ngroup_cols = ['Basic_Demos-Sex', 'Basic_Demos-Age']\ngroup_means = calculate_group_means(train, cols_to_fill, group_cols)\n\n# Fill missing values in both train and test DataFrames\ntrain = fill_with_group_means(train, group_means, group_cols)\ntest = fill_with_group_means(test, group_means, group_cols)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:55:58.783774Z","iopub.execute_input":"2024-10-07T15:55:58.784345Z","iopub.status.idle":"2024-10-07T15:55:59.404190Z","shell.execute_reply.started":"2024-10-07T15:55:58.784204Z","shell.execute_reply":"2024-10-07T15:55:59.402732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Median BMI values for each age and sex group\nmedian_bmi = {\n    'male': {\n        4: 16.1, 5: 16.1, 6: 16.1, 7: 16.1, 8: 17.3, 9: 17.3, 10: 19.2, 11: 19.2, 12: 20.6, 13: 20.6, \n        14: 22.0, 15: 22.0, 16: 24.1, 17: 24.1, 18: 25.0, 19: 25.0, 20: 25.5, 21: 25.5, 22: 25.5, \n        23: 25.5, 24: 25.5, 25: 26.4, 26: 26.4, 27: 26.4, 28: 26.4, 29: 26.4, 30: 26.4\n    },\n    'female': {\n        4: 16.0, 5: 16.0, 6: 16.0, 7: 16.0, 8: 17.0, 9: 17.0, 10: 19.0, 11: 19.0, 12: 20.5, 13: 20.5, \n        14: 22.0, 15: 22.0, 16: 23.5, 17: 23.5, 18: 24.5, 19: 24.5, 20: 24.5, 21: 24.5, 22: 24.5, \n        23: 24.5, 24: 24.5, 25: 25.0, 26: 25.0, 27: 25.0, 28: 25.0, 29: 25.0, 30: 25.0\n    }\n}\n\n# Function to calculate Relative-BMI\ndef calculate_relative_bmi(row):\n    if pd.isna(row['Physical-BMI']):\n        return None  # Return None if Physical-BMI is NaN\n    sex = 'male' if row['Basic_Demos-Sex'] == 0 else 'female'\n    age = row['Basic_Demos-Age']\n    median_bmi_value = median_bmi[sex].get(age, None)\n    if median_bmi_value:\n        return (row['Physical-BMI'] / median_bmi_value) * 100\n    else:\n        return None\n\n# Apply the function to the DataFrames without dropping rows\nfor df in [train, test]:\n    df['Relative-BMI'] = df.apply(calculate_relative_bmi, axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:55:59.405566Z","iopub.execute_input":"2024-10-07T15:55:59.405970Z","iopub.status.idle":"2024-10-07T15:55:59.550333Z","shell.execute_reply.started":"2024-10-07T15:55:59.405927Z","shell.execute_reply":"2024-10-07T15:55:59.549222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a new column 'PAQ_AC_Total' in the train DataFrame\ntrain['PAQ_AC_Total'] = train['PAQ_C-PAQ_C_Total'].combine_first(train['PAQ_A-PAQ_A_Total'])\ntest['PAQ_AC_Total'] = test['PAQ_C-PAQ_C_Total'].combine_first(test['PAQ_A-PAQ_A_Total'])\n\n# Optional: Drop the old columns if no longer needed\ntrain = train.drop(['PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total'], axis=1)\ntest = test.drop(['PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:55:59.551859Z","iopub.execute_input":"2024-10-07T15:55:59.552392Z","iopub.status.idle":"2024-10-07T15:55:59.568158Z","shell.execute_reply.started":"2024-10-07T15:55:59.552328Z","shell.execute_reply":"2024-10-07T15:55:59.566750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['BIA-BIA_BMC'] = train['BIA-BIA_BMC'].clip(upper=10)\ntest['BIA-BIA_BMC'] = test['BIA-BIA_BMC'].clip(upper=10)\n\ntrain['BIA-BIA_BMR'] = train['BIA-BIA_BMR'].clip(upper=4000)\ntest['BIA-BIA_BMR'] = test['BIA-BIA_BMR'].clip(upper=4000)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:55:59.570290Z","iopub.execute_input":"2024-10-07T15:55:59.570748Z","iopub.status.idle":"2024-10-07T15:55:59.585280Z","shell.execute_reply.started":"2024-10-07T15:55:59.570704Z","shell.execute_reply":"2024-10-07T15:55:59.583866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For the training DataFrame\ntrain['Blank_Fitness_Endurance'] = train['Fitness_Endurance-Max_Stage'].isna().astype(int)\n\n# For the testing DataFrame\ntest['Blank_Fitness_Endurance'] = test['Fitness_Endurance-Max_Stage'].isna().astype(int)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:55:59.587089Z","iopub.execute_input":"2024-10-07T15:55:59.587684Z","iopub.status.idle":"2024-10-07T15:55:59.596253Z","shell.execute_reply.started":"2024-10-07T15:55:59.587636Z","shell.execute_reply":"2024-10-07T15:55:59.594941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Combined average weights for males and females based on age (as a nested dictionary)\naverage_weights = {\n    0: {  # Male\n        1: 24.8, 2: 28.4, 3: 31.8, 4: 36.0, 5: 40.5,\n        6: 45.5, 7: 50.5, 8: 56.0, 9: 61.5, 10: 68.0,\n        11: 75.0, 12: 82.0, 13: 90.0, 14: 98.0, 15: 106.0,\n        16: 115.0, 17: 125.0, 18: 135.0, 19: 145.0,\n        '20-29': 155.0, '30-39': 165.0, '40-49': 175.0,\n        '50-59': 185.0, '60-69': 195.0, '70-79': 185.0,\n        '80-89': 175.0, '90-100': 165.0\n    },\n    1: {  # Female\n        1: 23.4, 2: 26.5, 3: 30.0, 4: 34.0, 5: 38.5,\n        6: 43.0, 7: 48.0, 8: 53.0, 9: 58.5, 10: 65.0,\n        11: 72.0, 12: 80.0, 13: 88.0, 14: 96.0, 15: 104.0,\n        16: 112.0, 17: 120.0, 18: 128.0, 19: 135.0,\n        '20-29': 145.0, '30-39': 155.0, '40-49': 165.0,\n        '50-59': 175.0, '60-69': 165.0, '70-79': 155.0,\n        '80-89': 145.0, '90-100': 135.0\n    }\n}\n\n# Function to determine age group for ages 20 and above\ndef get_age_group(age):\n    if age >= 20:\n        if age < 30:\n            return '20-29'\n        elif age < 40:\n            return '30-39'\n        elif age < 50:\n            return '40-49'\n        elif age < 60:\n            return '50-59'\n        elif age < 70:\n            return '60-69'\n        elif age < 80:\n            return '70-79'\n        elif age < 90:\n            return '80-89'\n        else:\n            return '90-100'\n    return age\n\n# Function to create the average_weight_US feature\ndef add_average_weight(df):\n    df['average_weight_US'] = df.apply(\n        lambda row: average_weights[row['Basic_Demos-Sex']].get(get_age_group(row['Basic_Demos-Age']), None), axis=1\n    )\n\n# Add the new feature to both DataFrames\nadd_average_weight(train)\nadd_average_weight(test)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:55:59.597880Z","iopub.execute_input":"2024-10-07T15:55:59.598353Z","iopub.status.idle":"2024-10-07T15:55:59.694188Z","shell.execute_reply.started":"2024-10-07T15:55:59.598307Z","shell.execute_reply":"2024-10-07T15:55:59.693224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['Difference_populationweight'] = train.apply(\n    lambda row: row['Physical-Weight'] - row['average_weight_US'] if pd.notna(row['Physical-Weight']) else None, axis=1\n)\ntest['Difference_populationweight'] = test.apply(\n    lambda row: row['Physical-Weight'] - row['average_weight_US'] if pd.notna(row['Physical-Weight']) else None, axis=1\n)\n\ntrain.drop(columns=['average_weight_US'], inplace=True)\ntest.drop(columns=['average_weight_US'], inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:55:59.697549Z","iopub.execute_input":"2024-10-07T15:55:59.697967Z","iopub.status.idle":"2024-10-07T15:55:59.813265Z","shell.execute_reply.started":"2024-10-07T15:55:59.697924Z","shell.execute_reply":"2024-10-07T15:55:59.812084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['Physical-HeartRate'] = train['Physical-HeartRate'].apply(lambda x: 85 if x < 50 or x > 120 else x)\ntest['Physical-HeartRate'] = test['Physical-HeartRate'].apply(lambda x: 85 if x < 50 or x > 120 else x)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:55:59.814660Z","iopub.execute_input":"2024-10-07T15:55:59.815066Z","iopub.status.idle":"2024-10-07T15:55:59.825424Z","shell.execute_reply.started":"2024-10-07T15:55:59.815023Z","shell.execute_reply":"2024-10-07T15:55:59.824108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:55:59.826982Z","iopub.execute_input":"2024-10-07T15:55:59.827414Z","iopub.status.idle":"2024-10-07T15:55:59.935825Z","shell.execute_reply.started":"2024-10-07T15:55:59.827366Z","shell.execute_reply":"2024-10-07T15:55:59.934466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:55:59.937668Z","iopub.execute_input":"2024-10-07T15:55:59.938077Z","iopub.status.idle":"2024-10-07T15:56:00.043880Z","shell.execute_reply.started":"2024-10-07T15:55:59.938036Z","shell.execute_reply":"2024-10-07T15:56:00.042233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.dropna(subset='sii')","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:56:00.045450Z","iopub.execute_input":"2024-10-07T15:56:00.045996Z","iopub.status.idle":"2024-10-07T15:56:00.057893Z","shell.execute_reply.started":"2024-10-07T15:56:00.045934Z","shell.execute_reply":"2024-10-07T15:56:00.056562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:56:00.059774Z","iopub.execute_input":"2024-10-07T15:56:00.060317Z","iopub.status.idle":"2024-10-07T15:56:00.070337Z","shell.execute_reply.started":"2024-10-07T15:56:00.060260Z","shell.execute_reply":"2024-10-07T15:56:00.069001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:56:00.071841Z","iopub.execute_input":"2024-10-07T15:56:00.072283Z","iopub.status.idle":"2024-10-07T15:56:00.090634Z","shell.execute_reply.started":"2024-10-07T15:56:00.072237Z","shell.execute_reply":"2024-10-07T15:56:00.089365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <p style=\"background-color:#6A1B9A; font-family:'Dancing Script', cursive; color:#FFFFFF; font-size:120%; text-align:center; border: 3px solid #FFEB3B; border-radius:40px; padding: 10px;\">Modeling | Single LGBM</p>","metadata":{}},{"cell_type":"code","source":"%%time\n\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\ndef TrainML(model_class, test_data):\n    \n    X = train.drop(['sii'], axis=1)\n    y = train['sii']\n\n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    train_S = []\n    test_S = []\n    \n    oof_non_rounded = np.zeros(len(y), dtype=float) \n    oof_rounded = np.zeros(len(y), dtype=int) \n    test_preds = np.zeros((len(test_data), n_splits))\n\n    for fold, (train_idx, test_idx) in enumerate(tqdm(SKF.split(X, y), desc=\"Training Folds\", total=n_splits)):\n        X_train, X_val = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[test_idx]\n\n        model = clone(model_class)\n        model.fit(X_train, y_train)\n\n        y_train_pred = model.predict(X_train)\n        y_val_pred = model.predict(X_val)\n\n        oof_non_rounded[test_idx] = y_val_pred\n        y_val_pred_rounded = y_val_pred.round(0).astype(int)\n        oof_rounded[test_idx] = y_val_pred_rounded\n\n        train_kappa = quadratic_weighted_kappa(y_train, y_train_pred.round(0).astype(int))\n        val_kappa = quadratic_weighted_kappa(y_val, y_val_pred_rounded)\n\n        train_S.append(train_kappa)\n        test_S.append(val_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        \n        print(f\"Fold {fold+1} - Train QWK: {train_kappa:.4f}, Validation QWK: {val_kappa:.4f}\")\n        clear_output(wait=True)\n\n    print(f\"Mean Train QWK --> {np.mean(train_S):.4f}\")\n    print(f\"Mean Validation QWK ---> {np.mean(test_S):.4f}\")\n\n    KappaOPtimizer = minimize(evaluate_predictions,\n                              x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), \n                              method='Nelder-Mead') # Nelder-Mead | # Powell\n    assert KappaOPtimizer.success, \"Optimization did not converge.\"\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOPtimizer.x)\n    tKappa = quadratic_weighted_kappa(y, oof_tuned)\n\n    print(f\"----> || Optimized QWK SCORE :: {Fore.CYAN}{Style.BRIGHT} {tKappa:.3f}{Style.RESET_ALL}\")\n\n    tpm = test_preds.mean(axis=1)\n    tpTuned = threshold_Rounder(tpm, KappaOPtimizer.x)\n    \n    submission = pd.DataFrame({\n        'id': sample['id'],\n        'sii': tpTuned\n    })\n\n    return submission","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:56:00.093355Z","iopub.execute_input":"2024-10-07T15:56:00.094051Z","iopub.status.idle":"2024-10-07T15:56:00.118309Z","shell.execute_reply.started":"2024-10-07T15:56:00.093974Z","shell.execute_reply":"2024-10-07T15:56:00.116967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nParams1 = {'learning_rate': 0.040096788285119124, 'max_depth': 7, 'num_leaves': 595, 'min_data_in_leaf': 10, 'feature_fraction': 0.8425318989982216, \n          'bagging_fraction': 0.558213183171439, 'bagging_freq': 7, 'lambda_l1': 9.895769394818767, 'lambda_l2': 0.014174738258541517} #0.3817 group filled, then relative-BMI\n\nParams4 = {'learning_rate': 0.04356076941529689, 'max_depth': 4, 'num_leaves': 332, 'min_data_in_leaf': 37, 'feature_fraction': 0.6436776163844733, \n           'bagging_fraction': 0.5011085267000686, 'bagging_freq': 4, 'lambda_l1': 1.7876121889129078e-07, 'lambda_l2': 0.0033160284648320734} #0.39894 good!\n\nParams1 = {'learning_rate': 0.07211505923648924, 'max_depth': 11, 'num_leaves': 402, 'min_data_in_leaf': 17, 'feature_fraction': 0.7276473832251896, \n          'bagging_fraction': 0.9302918142957443, 'bagging_freq': 2, 'lambda_l1': 5.068637587054631, 'lambda_l2': 4.910771992387766} #0.40902\n\nParams2 = {'learning_rate': 0.056936260368106555, 'max_depth': 9, 'num_leaves': 465, 'min_data_in_leaf': 20, 'feature_fraction': 0.75498677577314, \n          'bagging_fraction': 0.980917970181553, 'bagging_freq': 1, 'lambda_l1': 9.442548891860381, 'lambda_l2': 7.095936839464524e-08} #0.4055 best public s V11\n\nParams = {'learning_rate': 0.03512370538099075, 'max_depth': 5, 'num_leaves': 588, 'min_data_in_leaf': 58, 'feature_fraction': 0.5461045135357695, \n          'bagging_fraction': 0.7291599781596566, 'bagging_freq': 8, 'lambda_l1': 0.1858447193415156, 'lambda_l2': 2.561940607803013e-07} #0.39956 group filled, then relative-BMI 300 trials\n#old params\n\nLight = lgb.LGBMRegressor(**Params,random_state=SEED, verbose=-1,n_estimators=200)\nSubmission = TrainML(Light,test)","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:56:00.119976Z","iopub.execute_input":"2024-10-07T15:56:00.120501Z","iopub.status.idle":"2024-10-07T15:56:08.837121Z","shell.execute_reply.started":"2024-10-07T15:56:00.120456Z","shell.execute_reply":"2024-10-07T15:56:08.835790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <p style=\"background-color:#6A1B9A; font-family:'Dancing Script', cursive; color:#FFFFFF; font-size:120%; text-align:center; border: 3px solid #FFEB3B; border-radius:40px; padding: 10px;\">Submission</p>","metadata":{}},{"cell_type":"code","source":"%%time\n\nSubmission.to_csv('submission.csv', index=False)\nprint(Submission['sii'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2024-10-07T15:56:08.838534Z","iopub.execute_input":"2024-10-07T15:56:08.838934Z","iopub.status.idle":"2024-10-07T15:56:08.850685Z","shell.execute_reply.started":"2024-10-07T15:56:08.838887Z","shell.execute_reply":"2024-10-07T15:56:08.849632Z"},"trusted":true},"execution_count":null,"outputs":[]}]}