{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":9624647,"sourceType":"datasetVersion","datasetId":5874898}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np\nimport pandas as pd\nimport os\nimport random\nfrom sklearn.base import clone\nfrom sklearn.metrics import cohen_kappa_score,  make_scorer\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score\nfrom sklearn.feature_selection import SelectKBest, f_classif\nfrom sklearn.ensemble import VotingRegressor, RandomForestClassifier, VotingClassifier\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import PolynomialFeatures, LabelEncoder, StandardScaler\nfrom scipy.optimize import minimize, minimize_scalar\nfrom scipy import stats\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nfrom lightgbm import LGBMRegressor, LGBMClassifier\nfrom xgboost import XGBRegressor, XGBClassifier\nfrom catboost import CatBoostRegressor, CatBoostClassifier\nfrom imblearn.over_sampling import SMOTE\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nimport optuna\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\nwarnings.filterwarnings(\"ignore\", category=UserWarning, module=\"lightgbm\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:21:48.214098Z","iopub.execute_input":"2024-10-17T22:21:48.214481Z","iopub.status.idle":"2024-10-17T22:21:55.385217Z","shell.execute_reply.started":"2024-10-17T22:21:48.214437Z","shell.execute_reply":"2024-10-17T22:21:55.384221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create output folders\noutput_folder = 'output'\nif not os.path.exists(output_folder):\n    os.makedirs(output_folder)\n\n# Create separate analysis output folders\nanalysis_output_folder = 'analysis_output'\nos.makedirs(analysis_output_folder, exist_ok=True)\n\nphysical_analysis_output_folder = 'analysis_output/physical'\nos.makedirs(physical_analysis_output_folder, exist_ok=True)\n\nfitness_analysis_output_folder = 'analysis_output/fitness'\nos.makedirs(fitness_analysis_output_folder, exist_ok=True)\n\nbia_analysis_output_folder = 'analysis_output/bia'\nos.makedirs(bia_analysis_output_folder, exist_ok=True)\n\nchild_info_analysis_output_folder = 'analysis_output/child_info'\nos.makedirs(child_info_analysis_output_folder, exist_ok=True)\n\nactigraphy_analysis_output_folder = 'analysis_output/actigraphy'\nos.makedirs(actigraphy_analysis_output_folder, exist_ok=True)\n\n\n# Set display all columns in dataframes property\npd.options.display.max_columns = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:21:55.387403Z","iopub.execute_input":"2024-10-17T22:21:55.388394Z","iopub.status.idle":"2024-10-17T22:21:55.395985Z","shell.execute_reply.started":"2024-10-17T22:21:55.388355Z","shell.execute_reply":"2024-10-17T22:21:55.395186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load data functions\ndef process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    return df.describe().values.reshape(-1), filename.split('=')[1]\n\ndef load_time_series(dirname) -> pd.DataFrame:\n    ids = os.listdir(dirname)\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    stats, indexes = zip(*results)\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:21:55.397081Z","iopub.execute_input":"2024-10-17T22:21:55.397951Z","iopub.status.idle":"2024-10-17T22:21:55.412046Z","shell.execute_reply.started":"2024-10-17T22:21:55.397903Z","shell.execute_reply":"2024-10-17T22:21:55.411214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load data\ntrain_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts_data = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts_data = load_time_series(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:21:55.413275Z","iopub.execute_input":"2024-10-17T22:21:55.413663Z","iopub.status.idle":"2024-10-17T22:23:16.525715Z","shell.execute_reply.started":"2024-10-17T22:21:55.413623Z","shell.execute_reply":"2024-10-17T22:23:16.524785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remove id column from time series data\ntime_series_columns = train_ts_data.columns.tolist()\ntime_series_columns.remove(\"id\")\n\n# Merge data\ntrain_data = pd.merge(train_data, train_ts_data, how=\"left\", on='id')\ntest_data = pd.merge(test_data, test_ts_data, how=\"left\", on='id')\ntrain_data = train_data.drop('id', axis=1)\ntest_data = test_data.drop('id', axis=1)\n\ntrain_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:23:16.527917Z","iopub.execute_input":"2024-10-17T22:23:16.528839Z","iopub.status.idle":"2024-10-17T22:23:16.707588Z","shell.execute_reply.started":"2024-10-17T22:23:16.528791Z","shell.execute_reply":"2024-10-17T22:23:16.706651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Skew removal for some BIA columns\nskewed_columns = [\n    'BIA-BIA_BMC', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_Fat',\n    'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', \n    'BIA-BIA_TBW', 'CGAS-CGAS_Score', 'stat_23', 'stat_35', 'stat_38', 'stat_40', 'stat_47',\n    'stat_54', 'stat_66', 'stat_78', 'stat_80', 'stat_88', 'stat_90'\n]\nlambda_params = {}\n\n# Define the box-cox function to remove skew\ndef box_cox_transform(df, column, lambda_param=None):\n    # Create a copy of the dataframe\n    df_copy = df.copy()\n    \n    # Drop NaN values for the specific column\n    df_copy = df_copy.dropna(subset=[column])\n    \n    # Ensure all values are positive\n    min_value = df_copy[column].min()\n    if min_value <= 0:\n        df_copy[column] = df_copy[column] - min_value + 1  # Add 1 to ensure all values are positive\n    \n    # Perform Box-Cox transformation\n    if lambda_param is None:\n        df_copy[f'{column}_boxcox'], lambda_param = stats.boxcox(df_copy[column])\n        print(f\"Transforming column: {column}\")\n        print(f\"Optimal lambda for Box-Cox transformation: {lambda_param}\")\n    else:\n        df_copy[f'{column}_boxcox'] = stats.boxcox(df_copy[column], lmbda=lambda_param)\n        print(f\"Applying transformation to column: {column} with lambda: {lambda_param}\")\n    \n    print(f\"Number of rows before transformation: {len(df)}\")\n    print(f\"Number of rows after removing NaN values: {len(df_copy)}\")\n    \n    return df_copy, lambda_param\n\n# Apply Box-Cox transformation to train data and store lambda values\nfor column in skewed_columns:\n    transformed_train_data, lambda_params[column] = box_cox_transform(train_data, column)\n    # Update only the new transformed column in the original dataframe\n    train_data[f'{column}_boxcox'] = transformed_train_data[f'{column}_boxcox']\n\n# Apply the same transformation to test data using stored lambda values\nfor column in skewed_columns:\n    transformed_test_data, _ = box_cox_transform(test_data, column, lambda_param=lambda_params[column])\n    # Update only the new transformed column in the original dataframe\n    test_data[f'{column}_boxcox'] = transformed_test_data[f'{column}_boxcox']\n\n# Function to handle infinite values\ndef replace_inf_with_max(df):\n    for column in df.columns:\n        if df[column].dtype == 'float64':\n            max_value = df[column][~np.isinf(df[column])].max()\n            df[column] = df[column].replace([np.inf, -np.inf], max_value)\n    return df\n\n# Replace infinite values with the maximum non-infinite value in each column\ntrain_data = replace_inf_with_max(train_data)\ntest_data = replace_inf_with_max(test_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:23:16.708696Z","iopub.execute_input":"2024-10-17T22:23:16.708977Z","iopub.status.idle":"2024-10-17T22:23:17.478568Z","shell.execute_reply.started":"2024-10-17T22:23:16.708946Z","shell.execute_reply":"2024-10-17T22:23:17.477756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature engineering\ndef engineer_features(df, is_train=True):\n    # Create a copy of the dataframe to avoid modifying the original\n    df = df.copy()\n    \n    # Combine all grip strength\n    df['FGC-FGC_GS'] = df['FGC-FGC_GSD_Zone'] + df['FGC-FGC_GSND_Zone']\n\n    # Combine all sit and reach\n    df['FGC-FGC_SR'] = df['FGC-FGC_SRL_Zone'] + df['FGC-FGC_SRR_Zone']\n\n    # Create a fitness score by adding the zone fitness data\n    df['fitness_score'] = df['FGC-FGC_GS'] + df['FGC-FGC_SR'] + df['FGC-FGC_CU_Zone'] + df['FGC-FGC_PU_Zone'] + df['FGC-FGC_TL_Zone']\n\n    # Combine PAQ_A-PAQ_A_Total and PAQ_C-PAQ_C_Total into one column\n    df['PAQ_Total'] = df['PAQ_A-PAQ_A_Total'].combine_first(df['PAQ_C-PAQ_C_Total'])\n\n    # Create interaction features\n    interaction_features = [\n        ('Physical-Weight', 'BIA-BIA_BMI', 'Physical_Weight_X_BIA_BMI'),\n        ('BIA-BIA_Fat', 'Physical-Weight', 'BIA_Fat_X_Physical_Weight'),\n    ]\n\n    for feature1, feature2, new_feature in interaction_features:\n        if is_train or (feature2 != 'PCIAT-PCIAT_Total' and feature1 != 'PCIAT-PCIAT_Total'):\n            df[new_feature] = df[feature1] * df[feature2]\n        else:\n            # For test set, create a placeholder column filled with NaN\n            df[new_feature] = np.nan\n            print(f\"Warning: {new_feature} could not be created for test set due to missing 'PCIAT-PCIAT_Total'\")\n\n    # Combine identical actigraphy stats\n    columns_to_combine = ['stat_0', 'stat_1', 'stat_2', 'stat_3', 'stat_4', 'stat_5', \n                          'stat_6', 'stat_7', 'stat_8', 'stat_9', 'stat_10', 'stat_11']\n    \n    # Check if columns are identical\n    is_identical = df[columns_to_combine].nunique().eq(1).all()\n    \n    if is_identical:\n        # If they are identical, we can just take the first column and rename it\n        df['combined_actigraphy_stat'] = df[columns_to_combine[0]]\n        print(f\"Columns {columns_to_combine} have been combined into combined_actigraphy_stat\")\n    else:\n        # If not exactly identical, take the mean\n        print(\"Warning: The actigraphy stat columns are not identical. Taking the average instead.\")\n        df['combined_actigraphy_stat'] = df[columns_to_combine].mean(axis=1)\n    \n    # Drop the original columns\n    df = df.drop(columns=columns_to_combine)\n\n    return df\n\n# Apply feature engineering to train data\ntrain_data = engineer_features(train_data, is_train=True)\n\n# Apply feature engineering to test data\ntest_data = engineer_features(test_data, is_train=False)\n\nprint(\"Feature engineering completed for both train and test data.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:23:17.479786Z","iopub.execute_input":"2024-10-17T22:23:17.480481Z","iopub.status.idle":"2024-10-17T22:23:17.524143Z","shell.execute_reply.started":"2024-10-17T22:23:17.480430Z","shell.execute_reply":"2024-10-17T22:23:17.523223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Isolate the physical attribute columns and some contextual columns for analysis\nphysical_columns = [\n    'Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n    'CGAS-Season', 'CGAS-CGAS_Score_boxcox', 'Physical-Season', 'Physical-BMI',\n    'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference', 'BIA_Fat_X_Physical_Weight',\n    'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n    'fitness_score', 'BIA-BIA_Frame_num', 'BIA-BIA_BMI', 'PreInt_EduHx-computerinternet_hoursday',\n    'Physical_Weight_X_BIA_BMI', 'sii'\n]\n\n# Isolate the fitness attributes\n# Removed columns: 'FGC-FGC_CU' 'FGC-FGC_PU', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_SRL' \n# 'FGC-FGC_GSND_Zone' 'FGC-FGC_GSND' 'FGC-FGC_GSD' 'FGC-FGC_GSD_Zone' 'Fitness_Endurance-Time_Sec'\nfitness_columns = [\n    'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', \n    'FGC-Season', 'FGC-FGC_CU_Zone', 'FGC-FGC_SR', 'FGC-FGC_PU_Zone', 'FGC-FGC_TL', 'BIA_Fat_X_Physical_Weight',\n    'FGC-FGC_TL_Zone', 'FGC-FGC_GS', 'fitness_score', 'BIA-BIA_BMI', 'Physical-BMI', \n    'BIA-BIA_Frame_num', 'PreInt_EduHx-computerinternet_hoursday',\n    'Physical_Weight_X_BIA_BMI', 'sii'\n]\n\n# Isolate the BIA attributes\nbia_columns = [\n    'BIA-Season', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_Frame_num', 'BIA-BIA_BMC_boxcox', 'BIA_Fat_X_Physical_Weight',\n    'BIA-BIA_BMI', 'BIA-BIA_BMR_boxcox', 'BIA-BIA_DEE_boxcox', 'BIA-BIA_ECW_boxcox', 'BIA-BIA_FFM_boxcox', 'BIA-BIA_FFMI_boxcox',\n    'BIA-BIA_FMI_boxcox', 'BIA-BIA_Fat_boxcox', 'BIA-BIA_ICW_boxcox', 'BIA-BIA_LDM_boxcox', 'BIA-BIA_LST_boxcox', 'BIA-BIA_TBW_boxcox',\n    'fitness_score', 'Physical-BMI', 'PreInt_EduHx-computerinternet_hoursday', 'Physical_Weight_X_BIA_BMI', \n    'sii'\n]\n\n# Isolate the PAQ, PCIAT, and SDS\nchild_info_columns = [\n    'PreInt_EduHx-computerinternet_hoursday', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total',\n    'PAQ_Total', 'PCIAT-Season', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04',\n    'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'BIA_Fat_X_Physical_Weight',\n    'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', \n    'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', \n    'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total', 'SDS-Season', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T',\n    'PreInt_EduHx-Season', 'PreInt_EduHx-computerinternet_hoursday', 'BIA-BIA_BMI', 'fitness_score',\n    'Physical-BMI', 'Physical_Weight_X_BIA_BMI', 'sii'\n]\n\n# Isolate the Actigraphy data\n# removed columns: 'stat_41', 'stat_42', 'stat_39','stat_92_boxcox' 'stat_0', 'stat_1', 'stat_2', 'stat_3', 'stat_4', 'stat_5', 'stat_6', 'stat_7', \n# 'stat_8', 'stat_9', 'stat_10','stat_11'\nactigraphy_columns = [\n    'combined_actigraphy_stat', 'stat_12', 'stat_13', 'stat_14', 'stat_15', 'stat_16', 'stat_17', 'stat_18', 'stat_19', 'stat_20',\n    'stat_21', 'stat_22', 'stat_23_boxcox', 'stat_24', 'stat_25', 'stat_26', 'stat_27', 'stat_28', 'stat_29', 'stat_30',\n    'stat_31', 'stat_32', 'stat_33', 'stat_34', 'stat_35_boxcox', 'stat_36', 'stat_37', 'stat_38_boxcox', 'stat_40_boxcox',\n    'stat_43', 'stat_44', 'stat_45', 'stat_46', 'stat_47_boxcox', 'stat_48', 'stat_49', 'stat_50', 'BIA_Fat_X_Physical_Weight',\n    'stat_51', 'stat_52', 'stat_53', 'stat_54_boxcox', 'stat_55', 'stat_56', 'stat_57', 'stat_58', 'stat_59', 'stat_60',\n    'stat_61', 'stat_62', 'stat_63', 'stat_64', 'stat_65', 'stat_66_boxcox', 'stat_67', 'stat_68', 'stat_69', 'stat_70',\n    'stat_71', 'stat_72', 'stat_73', 'stat_74', 'stat_75', 'stat_76', 'stat_77', 'stat_78_boxcox', 'stat_79', 'stat_80_boxcox',\n    'stat_81', 'stat_82', 'stat_83', 'stat_84', 'stat_85', 'stat_86', 'stat_87', 'stat_88_boxcox', 'stat_89', 'stat_90_boxcox',\n    'stat_91', 'stat_93', 'stat_94', 'stat_95', 'PreInt_EduHx-computerinternet_hoursday', \n    'BIA-BIA_Frame_num', 'SDS-SDS_Total_T', 'BIA-BIA_BMI', 'Physical-BMI', 'sii'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:23:17.525411Z","iopub.execute_input":"2024-10-17T22:23:17.525727Z","iopub.status.idle":"2024-10-17T22:23:17.538492Z","shell.execute_reply.started":"2024-10-17T22:23:17.525694Z","shell.execute_reply":"2024-10-17T22:23:17.537514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to analyze columns\ndef analyze_column(column):\n    total_count = len(train_data)\n    missing_count = train_data[column].isnull().sum()\n    missing_percentage = (missing_count / total_count) * 100\n    unique_values = train_data[column].nunique()\n    \n    if pd.api.types.is_numeric_dtype(train_data[column]):\n        mean_value = train_data[column].mean()\n        median_value = train_data[column].median()\n        std_dev = train_data[column].std()\n        min_value = train_data[column].min()\n        max_value = train_data[column].max()\n        return {\n            \"Column\": column,\n            \"Total Count\": total_count,\n            \"Missing Count\": missing_count,\n            \"Missing Percentage\": f\"{missing_percentage:.2f}%\",\n            \"Unique Values\": unique_values,\n            \"Data Type\": train_data[column].dtype,\n            \"Mean\": mean_value,\n            \"Median\": median_value,\n            \"Standard Deviation\": std_dev,\n            \"Minimum\": min_value,\n            \"Maximum\": max_value\n        }\n    else:\n        top_values = train_data[column].value_counts().head(3).to_dict()\n        return {\n            \"Column\": column,\n            \"Total Count\": total_count,\n            \"Missing Count\": missing_count,\n            \"Missing Percentage\": f\"{missing_percentage:.2f}%\",\n            \"Unique Values\": unique_values,\n            \"Data Type\": train_data[column].dtype,\n            \"Top 3 Values\": top_values\n        }\n\n# Physical column profiles        \nphysical_column_profiles = [analyze_column(col) for col in physical_columns]\nphysical_column_profiles_df = pd.DataFrame(physical_column_profiles)\n\n# Save column profiles to CSV\nphysical_column_profiles_df.to_csv(os.path.join(physical_analysis_output_folder, 'physical_column_profiles.csv'), index=False)\nprint(f\"Column profiles saved to {os.path.join(physical_analysis_output_folder, 'physical_column_profiles.csv')}\")\n\n# Fitness column profiles\nfitness_column_profiles = [analyze_column(col) for col in fitness_columns]\nfitness_column_profiles_df = pd.DataFrame(fitness_column_profiles)\n\n# Save column profiles to CSV\nfitness_column_profiles_df.to_csv(os.path.join(analysis_output_folder, 'fitness_column_profiles.csv'), index=False)\nprint(f\"Column profiles saved to {os.path.join(analysis_output_folder, 'fitness_column_profiles.csv')}\")\n\n# BIA column profiles\nbia_column_profiles = [analyze_column(col) for col in bia_columns]\nbia_column_profiles_df = pd.DataFrame(bia_column_profiles)\n\n# Save column profiles to CSV\nbia_column_profiles_df.to_csv(os.path.join(analysis_output_folder, 'bia_column_profiles.csv'), index=False)\nprint(f\"Column profiles saved to {os.path.join(analysis_output_folder, 'bia_column_profiles.csv')}\")\n\n# Child info column profiles\nchild_info_column_profiles = [analyze_column(col) for col in child_info_columns]\nchild_info_column_profiles_df = pd.DataFrame(child_info_column_profiles)\n\n# Save column profiles to CSV\nchild_info_column_profiles_df.to_csv(os.path.join(analysis_output_folder, 'child_info_column_profiles.csv'), index=False)\nprint(f\"Column profiles saved to {os.path.join(analysis_output_folder, 'child_info_column_profiles.csv')}\")\n\n# Actigraphy info column profiles\nactigraphy_column_profiles = [analyze_column(col) for col in actigraphy_columns]\nactigraphy_column_profiles_df = pd.DataFrame(actigraphy_column_profiles)\n\n# Save column profiles to CSV\nactigraphy_column_profiles_df.to_csv(os.path.join(analysis_output_folder, 'actigraphy_column_profiles.csv'), index=False)\nprint(f\"Column profiles saved to {os.path.join(analysis_output_folder, 'actigraphy_column_profiles.csv')}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:23:17.539719Z","iopub.execute_input":"2024-10-17T22:23:17.539997Z","iopub.status.idle":"2024-10-17T22:23:17.746272Z","shell.execute_reply.started":"2024-10-17T22:23:17.539966Z","shell.execute_reply":"2024-10-17T22:23:17.745240Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize missing data\n# Physical columns\nplt.figure(figsize=(12, 6))\nsns.heatmap(train_data[physical_columns].isnull(), cbar=False, yticklabels=False, cmap='viridis')\nplt.title('Missing Data in Physical Attribute Columns')\nplt.xlabel('Columns')\nplt.ylabel('Rows')\nplt.tight_layout()\nplt.savefig(os.path.join(physical_analysis_output_folder, 'physical_missing_data_heatmap.png'))\nplt.close()\nprint(f\"Missing data heatmap saved to {os.path.join(physical_analysis_output_folder, 'physical_missing_data_heatmap.png')}\")\n\n# Fitness columns\nplt.figure(figsize=(12, 6))\nsns.heatmap(train_data[fitness_columns].isnull(), cbar=False, yticklabels=False, cmap='viridis')\nplt.title('Missing Data in Fitness Attribute Columns')\nplt.xlabel('Columns')\nplt.ylabel('Rows')\nplt.tight_layout()\nplt.savefig(os.path.join(fitness_analysis_output_folder, 'fitness_missing_data_heatmap.png'))\nplt.close()\nprint(f\"Missing data heatmap saved to {os.path.join(fitness_analysis_output_folder, 'fitness_missing_data_heatmap.png')}\")\n\n# BIA columns\nplt.figure(figsize=(12, 6))\nsns.heatmap(train_data[bia_columns].isnull(), cbar=False, yticklabels=False, cmap='viridis')\nplt.title('Missing Data in BIA Attribute Columns')\nplt.xlabel('Columns')\nplt.ylabel('Rows')\nplt.tight_layout()\nplt.savefig(os.path.join(bia_analysis_output_folder, 'bia_missing_data_heatmap.png'))\nplt.close()\nprint(f\"Missing data heatmap saved to {os.path.join(bia_analysis_output_folder, 'bia_missing_data_heatmap.png')}\")\n\n# Child info columns\nplt.figure(figsize=(12, 6))\nsns.heatmap(train_data[child_info_columns].isnull(), cbar=False, yticklabels=False, cmap='viridis')\nplt.title('Missing Data in Child info Attribute Columns')\nplt.xlabel('Columns')\nplt.ylabel('Rows')\nplt.tight_layout()\nplt.savefig(os.path.join(child_info_analysis_output_folder, 'child_info_missing_data_heatmap.png'))\nplt.close()\nprint(f\"Missing data heatmap saved to {os.path.join(child_info_analysis_output_folder, 'child_info_missing_data_heatmap.png')}\")\n\n# Actigraphy info columns\nplt.figure(figsize=(40, 6))\nsns.heatmap(train_data[actigraphy_columns].isnull(), cbar=False, yticklabels=False, cmap='viridis')\nplt.title('Missing Data in Actigraphy info Attribute Columns')\nplt.xlabel('Columns')\nplt.ylabel('Rows')\nplt.tight_layout()\nplt.savefig(os.path.join(actigraphy_analysis_output_folder, 'actigraphy_missing_data_heatmap.png'))\nplt.close()\nprint(f\"Missing data heatmap saved to {os.path.join(actigraphy_analysis_output_folder, 'actigraphy_missing_data_heatmap.png')}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:23:17.747645Z","iopub.execute_input":"2024-10-17T22:23:17.747978Z","iopub.status.idle":"2024-10-17T22:23:20.924058Z","shell.execute_reply.started":"2024-10-17T22:23:17.747941Z","shell.execute_reply":"2024-10-17T22:23:20.923016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Correlation matrix for physical numeric columns\nphysical_numeric_columns = train_data[physical_columns].select_dtypes(include=[np.number]).columns\nphysical_correlation_matrix = train_data[physical_numeric_columns].corr()\n\nplt.figure(figsize=(20, 18))\nsns.heatmap(physical_correlation_matrix, annot=True, cmap='coolwarm', linewidths=0.5)\nplt.title('Correlation Matrix of Numeric Physical Attributes')\nplt.tight_layout()\nplt.savefig(os.path.join(physical_analysis_output_folder, 'physical_correlation_matrix.png'))\nplt.close()\nprint(f\"Physical correlation matrix saved to {os.path.join(physical_analysis_output_folder, 'physical_correlation_matrix.png')}\")\n\n# Correlation matrix for fitness numeric columns\nfitness_numeric_columns = train_data[fitness_columns].select_dtypes(include=[np.number]).columns\nfitness_correlation_matrix = train_data[fitness_numeric_columns].corr()\n\nplt.figure(figsize=(20, 18))\nsns.heatmap(fitness_correlation_matrix, annot=True, cmap='coolwarm', linewidths=0.5)\nplt.title('Correlation Matrix of Numeric Fitness Attributes')\nplt.tight_layout()\nplt.savefig(os.path.join(fitness_analysis_output_folder, 'fitness_correlation_matrix.png'))\nplt.close()\nprint(f\"Fitness correlation matrix saved to {os.path.join(fitness_analysis_output_folder, 'fitness_correlation_matrix.png')}\")\n\n# Correlation matrix for bia numeric columns\nbia_numeric_columns = train_data[bia_columns].select_dtypes(include=[np.number]).columns\nbia_correlation_matrix = train_data[bia_numeric_columns].corr()\n\nplt.figure(figsize=(20, 18))\nsns.heatmap(bia_correlation_matrix, annot=True, cmap='coolwarm', linewidths=0.5)\nplt.title('Correlation Matrix of Numeric BIA Attributes')\nplt.tight_layout()\nplt.savefig(os.path.join(bia_analysis_output_folder, 'bia_correlation_matrix.png'))\nplt.close()\nprint(f\"BIA correlation matrix saved to {os.path.join(bia_analysis_output_folder, 'BIA_correlation_matrix.png')}\")\n\n# Correlation matrix for child info numeric columns\nchild_info_numeric_columns = train_data[child_info_columns].select_dtypes(include=[np.number]).columns\nchild_info_correlation_matrix = train_data[child_info_numeric_columns].corr()\n\nplt.figure(figsize=(24, 22))\nsns.heatmap(child_info_correlation_matrix, annot=True, cmap='coolwarm', linewidths=0.5)\nplt.title('Correlation Matrix of Numeric Child info Attributes')\nplt.tight_layout()\nplt.savefig(os.path.join(child_info_analysis_output_folder, 'child_info_correlation_matrix.png'))\nplt.close()\nprint(f\"BIA correlation matrix saved to {os.path.join(child_info_analysis_output_folder, 'child_info_correlation_matrix.png')}\")\n\n# Correlation matrix for actigraphy numeric columns\nactigraphy_numeric_columns = train_data[actigraphy_columns].select_dtypes(include=[np.number]).columns\nactigraphy_correlation_matrix = train_data[actigraphy_numeric_columns].corr()\n\nplt.figure(figsize=(80, 78))\nsns.heatmap(actigraphy_correlation_matrix, annot=True, cmap='coolwarm', linewidths=0.5)\nplt.title('Correlation Matrix of Numeric Actigraphy Attributes')\nplt.tight_layout()\nplt.savefig(os.path.join(actigraphy_analysis_output_folder, 'actigraphy_correlation_matrix.png'))\nplt.close()\nprint(f\"BIA correlation matrix saved to {os.path.join(actigraphy_analysis_output_folder, 'actigraphy_correlation_matrix.png')}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:23:20.925213Z","iopub.execute_input":"2024-10-17T22:23:20.925524Z","iopub.status.idle":"2024-10-17T22:23:49.957540Z","shell.execute_reply.started":"2024-10-17T22:23:20.925491Z","shell.execute_reply":"2024-10-17T22:23:49.956593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Combine all columns into a single list\nall_columns = []\nall_columns.extend(physical_numeric_columns)\nall_columns.extend(fitness_numeric_columns)\nall_columns.extend(bia_numeric_columns)\nall_columns.extend(child_info_numeric_columns)\nall_columns.extend(actigraphy_numeric_columns)\n\n# Create a tqdm progress bar\nwith tqdm(total=len(all_columns), desc=\"Creating distribution plots\") as pbar:\n    # Distribution plots for physical numeric columns\n    for column in physical_numeric_columns:\n        plt.figure(figsize=(10, 6))\n        sns.histplot(train_data[column].dropna(), kde=True)\n        plt.title(f'Distribution of physical {column}')\n        plt.xlabel(column)\n        plt.ylabel('Count')\n        plt.tight_layout()\n        plt.savefig(os.path.join(physical_analysis_output_folder, f'physical_{column}_distribution.png'))\n        plt.close()\n        pbar.update(1)\n\n    # Distribution plots for fitness numeric columns\n    for column in fitness_numeric_columns:\n        plt.figure(figsize=(10, 6))\n        sns.histplot(train_data[column].dropna(), kde=True)\n        plt.title(f'Distribution of fitness {column}')\n        plt.xlabel(column)\n        plt.ylabel('Count')\n        plt.tight_layout()\n        plt.savefig(os.path.join(fitness_analysis_output_folder, f'fitness_{column}_distribution.png'))\n        plt.close()\n        pbar.update(1)\n\n    # Distribution plots for BIA numeric columns\n    for column in bia_numeric_columns:\n        plt.figure(figsize=(10, 6))\n        sns.histplot(train_data[column].dropna(), kde=True)\n        plt.title(f'Distribution of bia {column}')\n        plt.xlabel(column)\n        plt.ylabel('Count')\n        plt.tight_layout()\n        plt.savefig(os.path.join(bia_analysis_output_folder, f'bia_{column}_distribution.png'))\n        plt.close()\n        pbar.update(1)\n        \n    # Distribution plots for Child info numeric columns\n    for column in child_info_numeric_columns:\n        plt.figure(figsize=(10, 6))\n        sns.histplot(train_data[column].dropna(), kde=True)\n        plt.title(f'Distribution of child info {column}')\n        plt.xlabel(column)\n        plt.ylabel('Count')\n        plt.tight_layout()\n        plt.savefig(os.path.join(child_info_analysis_output_folder, f'child_info_{column}_distribution.png'))\n        plt.close()\n        pbar.update(1)\n        \n    # Distribution plots for actigraphy numeric columns\n    for column in actigraphy_numeric_columns:\n        plt.figure(figsize=(10, 6))\n        sns.histplot(train_data[column].dropna(), kde=True)\n        plt.title(f'Distribution of actigraphy {column}')\n        plt.xlabel(column)\n        plt.ylabel('Count')\n        plt.tight_layout()\n        plt.savefig(os.path.join(actigraphy_analysis_output_folder, f'actigraphy_{column}_distribution.png'))\n        plt.close()\n        pbar.update(1)\n\nprint(\"All analyses completed and saved to the 'analysis_output' folder.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:23:49.959007Z","iopub.execute_input":"2024-10-17T22:23:49.959506Z","iopub.status.idle":"2024-10-17T22:25:17.028833Z","shell.execute_reply.started":"2024-10-17T22:23:49.959456Z","shell.execute_reply":"2024-10-17T22:25:17.027862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Supplement missing data with data from WHO\n# Load who data and group\ndef load_who_bmi_data(file_path):\n    who_data = pd.read_csv(file_path)\n    who_data = who_data.groupby(['age', 'sex']).agg({\n        'L': 'mean', 'mean_bmi': 'mean', 'S': 'mean'\n    }).reset_index()\n    who_data = who_data.set_index(['sex', 'age'])\n    return who_data\n\ndef load_who_height_data(file_path):\n    who_data = pd.read_csv(file_path)\n    who_data = who_data.groupby(['age', 'sex']).agg({\n        'mean_height': 'mean'\n    }).reset_index()\n    who_data = who_data.set_index(['sex', 'age'])\n    return who_data\n\nwho_bmi_data = load_who_bmi_data('/kaggle/input/who-bmi-and-height-data-5-to-19/bmi_for_age_5_to_19.csv')\nwho_height_data = load_who_height_data('/kaggle/input/who-bmi-and-height-data-5-to-19/height_for_age_5_to_19.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:25:17.030554Z","iopub.execute_input":"2024-10-17T22:25:17.030958Z","iopub.status.idle":"2024-10-17T22:25:17.065728Z","shell.execute_reply.started":"2024-10-17T22:25:17.030912Z","shell.execute_reply":"2024-10-17T22:25:17.064956Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Defining functions to Impute with data from WHO\ndef get_who_stats(age, sex, data_type='bmi'):\n    try:\n        if data_type == 'bmi':\n            stats = who_bmi_data.loc[(sex, age), ['mean_bmi', 'S']]\n            return stats['mean_bmi'], stats['S']\n        elif data_type == 'height':\n            stats = who_height_data.loc[(sex, age), 'mean_height']\n            return stats\n    except KeyError:\n        return None, None if data_type == 'bmi' else None\n    \ndef impute_bmi(age, sex):\n    mean_bmi, sd = get_who_stats(age, sex, 'bmi')\n    if mean_bmi is not None and sd is not None:\n        imputed_bmi = np.random.normal(mean_bmi, sd)\n        return round(imputed_bmi, 8)\n    else:\n        return None\n\ndef impute_height(age, sex):\n    mean_height_cm = get_who_stats(age, sex, 'height')\n    if mean_height_cm is not None:\n        mean_height_inches = mean_height_cm / 2.54  # Convert cm to inches\n        return round(mean_height_inches, 2)\n    else:\n        return None\n    \ndef impute_weight(bmi, height_inches):\n    if bmi is not None and height_inches is not None:\n        height_meters = height_inches * 0.0254  # Convert inches to meters\n        weight_kg = bmi * (height_meters ** 2)\n        weight_lbs = weight_kg * 2.20462  # Convert kg to lbs\n        return round(weight_lbs, 2)\n    else:\n        return None\n\ndef impute_waist_circumference(weight, height, age, sex):\n    if weight is not None and height is not None:\n        if sex == '0':\n            waist = (weight * 0.5) + (height * 0.2) - (age * 0.1)\n        else:\n            waist = (weight * 0.4) + (height * 0.3) - (age * 0.1)\n        return round(waist, 2)\n    else:\n        return None\n\ndef impute_body_fat(bmi, age, sex):\n    if bmi is not None:\n        if sex == '0':\n            body_fat = (1.20 * bmi) + (0.23 * age) - 16.2\n        else:\n            body_fat = (1.20 * bmi) + (0.23 * age) - 5.4\n        return round(max(body_fat, 0), 2)  # Ensure body fat is not negative\n    else:\n        return None\n    \ndef impute_sds_total_raw():\n    # Based on the distribution in Image 1\n    mu, sigma = 40, 12  # Estimated from the graph\n    imputed_value = np.random.normal(mu, sigma)\n    return round(max(min(imputed_value, 90), 20))  # Clip between 20 and 90\n\ndef impute_sds_total_t(sds_total_raw):\n    if sds_total_raw is not None:\n        # Rough linear transformation based on the relationship between raw and T scores\n        t_score = 50 + (sds_total_raw - 40) * (10 / 12)\n        return round(max(min(t_score, 100), 40))  # Clip between 40 and 100\n    else:\n        # If raw score is not available, impute based on the distribution in Image 2\n        mu, sigma = 60, 15  # Estimated from the graph\n        imputed_value = np.random.normal(mu, sigma)\n        return round(max(min(imputed_value, 100), 40))  # Clip between 40 and 100\n\ndef impute_sii_function(bmi, waist_circumference, body_fat, sds_total_raw, sds_total_t):\n    # Create a simple scoring system\n    score = 0\n    \n    # BMI-based score\n    if bmi is not None:\n        if bmi < 18.5:\n            score += 1\n        elif bmi >= 25:\n            score += 2\n        elif bmi >= 30:\n            score += 3\n    \n    # Waist circumference-based score (using general guidelines)\n    if waist_circumference is not None:\n        if waist_circumference > 60:  # This is a general threshold, might need adjustment\n            score += 1\n    \n    if sds_total_t is not None:\n        if sds_total_t > 80:\n            score += 1\n        if sds_total_t > 90:\n            score += 1\n    \n    # Map the score to sii values without forcing a 3\n    if score <= 2:\n        return 0\n    elif score <= 5:\n        return 1\n    else:\n        return 2\n\ndef apply_imputation(df, impute_sii=True):\n    def impute_if_missing(row):\n        age = row['Basic_Demos-Age']\n        sex = row['Basic_Demos-Sex']\n        \n        if pd.isna(row['Physical-BMI']) and 5 <= age <= 19:\n            row['Physical-BMI'] = impute_bmi(age, sex)\n        \n        if pd.isna(row['Physical-Height']) and 5 <= age <= 19:\n            row['Physical-Height'] = impute_height(age, sex)\n        \n        if pd.isna(row['Physical-Weight']) and row['Physical-BMI'] is not None and row['Physical-Height'] is not None:\n            row['Physical-Weight'] = impute_weight(row['Physical-BMI'], row['Physical-Height'])\n        \n        if pd.isna(row['Physical-Waist_Circumference']):\n            row['Physical-Waist_Circumference'] = impute_waist_circumference(\n                row['Physical-Weight'], row['Physical-Height'], age, sex\n            )\n        \n        if pd.isna(row['BIA-BIA_Fat']):\n            row['BIA-BIA_Fat'] = impute_body_fat(row['Physical-BMI'], age, sex)\n\n        if pd.isna(row['SDS-SDS_Total_Raw']):\n            row['SDS-SDS_Total_Raw'] = impute_sds_total_raw()\n        \n        if pd.isna(row['SDS-SDS_Total_T']):\n            row['SDS-SDS_Total_T'] = impute_sds_total_t(row['SDS-SDS_Total_Raw'])\n        \n        # Impute sii if it's missing and impute_sii is True\n        if impute_sii and pd.isna(row['sii']):\n            row['sii'] = impute_sii_function(\n                row['Physical-BMI'],\n                row['Physical-Waist_Circumference'],\n                row['BIA-BIA_Fat'],\n                row['SDS-SDS_Total_Raw'],\n                row['SDS-SDS_Total_T']\n            )\n        \n        return row\n    \n    return df.apply(impute_if_missing, axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:25:17.069342Z","iopub.execute_input":"2024-10-17T22:25:17.069656Z","iopub.status.idle":"2024-10-17T22:25:17.095652Z","shell.execute_reply.started":"2024-10-17T22:25:17.069621Z","shell.execute_reply":"2024-10-17T22:25:17.094643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Impute all values including 'sii' for train data\ntrain_data = apply_imputation(train_data, impute_sii=True)\n\n# Impute all values except 'sii' for test data\ntest_data = apply_imputation(test_data, impute_sii=False)\n\n# Check the results\nprint(\"Number of missing values after imputation:\")\nprint(\"BMI:\", train_data['Physical-BMI'].isna().sum())\nprint(\"Height:\", train_data['Physical-Height'].isna().sum())\nprint(\"Weight:\", train_data['Physical-Weight'].isna().sum())\nprint(\"Waist Circumference:\", train_data['Physical-Waist_Circumference'].isna().sum())\nprint(\"Body Fat Percentage:\", train_data['BIA-BIA_Fat'].isna().sum())\nprint(\"SDS Total Raw:\", train_data['SDS-SDS_Total_Raw'].isna().sum())\nprint(\"SDS Total T:\", train_data['SDS-SDS_Total_T'].isna().sum())\n\nprint(\"\\nSample of imputed values:\")\nimputed_sample = train_data[\n    (train_data['Physical-BMI'].notnull() | train_data['Physical-Height'].notnull() | \n     train_data['Physical-Weight'].notnull() | train_data['Physical-Waist_Circumference'].notnull() | \n     train_data['BIA-BIA_Fat'].notnull() | train_data['SDS-SDS_Total_Raw'].notnull() | \n     train_data['SDS-SDS_Total_T'].notnull()) &\n    ((train_data['Physical-BMI'].notnull() != train_data['Physical-BMI'].notnull().shift()) |\n     (train_data['Physical-Height'].notnull() != train_data['Physical-Height'].notnull().shift()) |\n     (train_data['Physical-Weight'].notnull() != train_data['Physical-Weight'].notnull().shift()) |\n     (train_data['Physical-Waist_Circumference'].notnull() != train_data['Physical-Waist_Circumference'].notnull().shift()) |\n     (train_data['BIA-BIA_Fat'].notnull() != train_data['BIA-BIA_Fat'].notnull().shift()) |\n     (train_data['SDS-SDS_Total_Raw'].notnull() != train_data['SDS-SDS_Total_Raw'].notnull().shift()) |\n     (train_data['SDS-SDS_Total_T'].notnull() != train_data['SDS-SDS_Total_T'].notnull().shift()))\n].sample(5)[['Basic_Demos-Age', 'Basic_Demos-Sex', 'Physical-BMI', 'Physical-Height', 'Physical-Weight',\n              'Physical-Waist_Circumference', 'BIA-BIA_Fat', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T']]\nprint(imputed_sample)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:25:17.096953Z","iopub.execute_input":"2024-10-17T22:25:17.097378Z","iopub.status.idle":"2024-10-17T22:25:20.051443Z","shell.execute_reply.started":"2024-10-17T22:25:17.097339Z","shell.execute_reply":"2024-10-17T22:25:20.050457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Additional analysis of imputation results\nprint(\"\\nImputation summary by age:\")\nage_summary = train_data.groupby('Basic_Demos-Age').agg({\n    'Physical-BMI': ['count', 'mean', 'std', 'min', 'max'],\n    'Physical-Height': ['count', 'mean', 'std', 'min', 'max'],\n    'Physical-Weight': ['count', 'mean', 'std', 'min', 'max'],\n    'Basic_Demos-Sex': 'count'\n})\nprint(age_summary)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:25:20.053139Z","iopub.execute_input":"2024-10-17T22:25:20.053562Z","iopub.status.idle":"2024-10-17T22:25:20.082053Z","shell.execute_reply.started":"2024-10-17T22:25:20.053518Z","shell.execute_reply":"2024-10-17T22:25:20.081113Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot of imputed vs. original data\nplt.figure(figsize=(12, 6))\nsns.scatterplot(data=train_data, x='Basic_Demos-Age', y='Physical-BMI', hue='Basic_Demos-Sex', alpha=0.5)\nplt.title('BMI vs Age (After Imputation)')\nplt.savefig('analysis_output/bmi_vs_age_imputed.png')\nplt.close()\n\nplt.figure(figsize=(12, 6))\nsns.scatterplot(data=train_data, x='Basic_Demos-Age', y='Physical-Height', hue='Basic_Demos-Sex', alpha=0.5)\nplt.title('Height vs Age (After Imputation)')\nplt.savefig('analysis_output/height_vs_age_imputed.png')\nplt.close()\n\nplt.figure(figsize=(12, 6))\nsns.scatterplot(data=train_data, x='Basic_Demos-Age', y='Physical-Weight', hue='Basic_Demos-Sex', alpha=0.5)\nplt.title('Weight vs Age (After Imputation)')\nplt.savefig('analysis_output/weight_vs_age_imputed.png')\nplt.close()\n\nprint(\"Imputation analysis plots saved in the 'analysis_output' folder.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:25:20.083035Z","iopub.execute_input":"2024-10-17T22:25:20.083319Z","iopub.status.idle":"2024-10-17T22:25:21.357307Z","shell.execute_reply.started":"2024-10-17T22:25:20.083287Z","shell.execute_reply":"2024-10-17T22:25:21.356379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Export train_data to CSV\ntrain_output_path = os.path.join(output_folder, 'train_data_imputed.csv')\ntrain_data.to_csv(train_output_path, index=False)\nprint(f\"Imputed train data exported to: {train_output_path}\")\n\n# Export test_data to CSV\ntest_output_path = os.path.join(output_folder, 'test_data_imputed.csv')\ntest_data.to_csv(test_output_path, index=False)\nprint(f\"Imputed test data exported to: {test_output_path}\")\n\nprint(\"Data export completed.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:25:21.358756Z","iopub.execute_input":"2024-10-17T22:25:21.359573Z","iopub.status.idle":"2024-10-17T22:25:21.988361Z","shell.execute_reply.started":"2024-10-17T22:25:21.359524Z","shell.execute_reply":"2024-10-17T22:25:21.987494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define categorical columns\ncategory_columns =[\n    'Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season',\n    'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season',\n    'PreInt_EduHx-Season'\n]\n\n# Enumerate categorical columns\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in category_columns:\n    mapping = create_mapping(col, train_data)\n    train_data[col] = train_data[col].replace(mapping).infer_objects(copy=False).astype(int)\n    test_data[col] = test_data[col].replace(create_mapping(col, test_data)).infer_objects(copy=False).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:25:21.989423Z","iopub.execute_input":"2024-10-17T22:25:21.989705Z","iopub.status.idle":"2024-10-17T22:25:22.053350Z","shell.execute_reply.started":"2024-10-17T22:25:21.989675Z","shell.execute_reply":"2024-10-17T22:25:22.052398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# High-risk vs. Low-risk analysis\n# Suppress RuntimeWarnings\nwarnings.filterwarnings(\"ignore\", category=RuntimeWarning)\n\n# List of columns to drop\ncolumns_to_drop = [\n    'FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', \n    'FGC-FGC_SRL', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', \n    'Fitness_Endurance-Time_Sec', 'stat_41', 'stat_42', 'stat_39', 'stat_92_boxcox', 'stat_92',\n    'stat_44', 'PCIAT-Season', 'PCIAT-PCIAT_Total', 'BIA-BIA_BMC', 'BIA-BIA_BMR', 'BIA-BIA_DEE', \n    'BIA-BIA_ECW', 'BIA-BIA_Fat', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_ICW', \n    'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_TBW', 'CGAS-CGAS_Score', 'stat_23', 'stat_35', \n    'stat_38', 'stat_40', 'stat_47', 'stat_54', 'stat_66', 'stat_78', 'stat_80', 'stat_88', 'stat_90'\n]\n\ntrain_data = train_data.drop(columns=columns_to_drop, errors='ignore')\n\n# Separate high-risk and low-risk groups\nhigh_risk = train_data[train_data['sii'] == 3]\nlow_risk = train_data[train_data['sii'] != 3]\n\n# Function to calculate summary statistics\ndef get_summary_stats(group):\n    return pd.DataFrame({\n        'mean': group.mean(),\n        'median': group.median(),\n        'std': group.std(),\n        'min': group.min(),\n        'max': group.max()\n    })\n\n# List of columns to analyze (excluding 'sii' and non-numeric columns)\ncolumns_to_analyze = train_data.select_dtypes(include=[np.number]).columns.drop('sii')\n\n# Calculate summary statistics for both groups\nhigh_risk_stats = get_summary_stats(high_risk[columns_to_analyze])\nlow_risk_stats = get_summary_stats(low_risk[columns_to_analyze])\n\n# Calculate the difference in means\nmean_diff = high_risk_stats['mean'] - low_risk_stats['mean']\n\n# Perform statistical tests to check for significant differences\ndef perform_statistical_test(col):\n    try:\n        # Try t-test first\n        t_stat, p_val = stats.ttest_ind(high_risk[col].dropna(), low_risk[col].dropna(), equal_var=False)\n        test_type = 't-test'\n    except Exception:\n        # If t-test fails, use Mann-Whitney U test\n        try:\n            u_stat, p_val = stats.mannwhitneyu(high_risk[col].dropna(), low_risk[col].dropna(), alternative='two-sided')\n            test_type = 'Mann-Whitney U'\n        except Exception:\n            # If both tests fail, return NaN values\n            return pd.Series({'statistic': np.nan, 'p_value': np.nan, 'test_type': 'Failed'})\n    \n    return pd.Series({'statistic': t_stat if test_type == 't-test' else u_stat, \n                      'p_value': p_val, \n                      'test_type': test_type})\n\ntest_results = pd.DataFrame({col: perform_statistical_test(col) for col in columns_to_analyze}).T\n\n# Combine results\ncomparison_results = pd.concat([\n    high_risk_stats.add_prefix('high_risk_'),\n    low_risk_stats.add_prefix('low_risk_'),\n    mean_diff.rename('mean_difference'),\n    test_results\n], axis=1)\n\n# Sort by absolute mean difference\ncomparison_results = comparison_results.sort_values('mean_difference', key=abs, ascending=False)\n\n# Display the top 20 most different features\nprint(comparison_results.head(20))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:25:22.054769Z","iopub.execute_input":"2024-10-17T22:25:22.055116Z","iopub.status.idle":"2024-10-17T22:25:22.674864Z","shell.execute_reply.started":"2024-10-17T22:25:22.055083Z","shell.execute_reply":"2024-10-17T22:25:22.673937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize distributions for top features\ndef plot_distribution(feature):\n    plt.figure(figsize=(10, 6))\n    sns.histplot(data=train_data, x=feature, hue='sii', kde=True, common_norm=False)\n    plt.title(f'Distribution of {feature} by Risk Group')\n    plt.savefig(f'analysis_output/distribution_{feature}.png')\n    plt.close()\n\nfor feature in comparison_results.head(20).index:\n    plot_distribution(feature)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:25:22.676255Z","iopub.execute_input":"2024-10-17T22:25:22.676586Z","iopub.status.idle":"2024-10-17T22:25:51.969638Z","shell.execute_reply.started":"2024-10-17T22:25:22.676554Z","shell.execute_reply":"2024-10-17T22:25:51.968820Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Correlation analysis\ndef get_correlations(group):\n    return group.corr().iloc[:, 0].sort_values(key=abs, ascending=False)\n\nhigh_risk_corr = get_correlations(high_risk[columns_to_analyze])\nlow_risk_corr = get_correlations(low_risk[columns_to_analyze])\n\n# Capture top 40 feature names for high-risk and low-risk correlations\ntop_high_risk_cor_features = high_risk_corr.head(40).index.tolist()\ntop_low_risk_cor_features = low_risk_corr.head(40).index.tolist()\n\nprint(\"\\nTop correlations in high-risk group:\")\nprint(high_risk_corr.head(40))\nprint(\"\\nTop correlations in low-risk group:\")\nprint(low_risk_corr.head(40))\n\n# Calculate correlation differences\ncorr_diff = high_risk_corr - low_risk_corr\ncorr_diff = corr_diff.sort_values(key=abs, ascending=False)\n\nprint(\"\\nTop correlation differences (high-risk minus low-risk):\")\nprint(corr_diff.head(20))\n\n# Save correlation results to CSV\npd.DataFrame({\n    'high_risk_corr': high_risk_corr,\n    'low_risk_corr': low_risk_corr,\n    'correlation_difference': corr_diff\n}).to_csv('analysis_output/correlation_comparison.csv')\n\n# Save results to CSV\ncomparison_results.to_csv('analysis_output/high_low_risk_comparison.csv')\n\nprint(\"\\nAnalysis complete. Results saved to 'analysis_output/high_low_risk_comparison.csv' and 'analysis_output/correlation_comparison.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:25:51.970942Z","iopub.execute_input":"2024-10-17T22:25:51.971302Z","iopub.status.idle":"2024-10-17T22:25:52.144294Z","shell.execute_reply.started":"2024-10-17T22:25:51.971265Z","shell.execute_reply":"2024-10-17T22:25:52.143221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize top correlation differences\nplt.figure(figsize=(16, 12))\ncorr_diff.head(20).plot(kind='bar')\nplt.title('Top 20 Correlation Differences (High-risk minus Low-risk)')\nplt.xlabel('Features')\nplt.ylabel('Correlation Difference')\nplt.tight_layout()\nplt.savefig('analysis_output/top_correlation_differences.png')\nplt.close()\n\nprint(\"Correlation difference plot saved to 'analysis_output/top_correlation_differences.png'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:25:52.145627Z","iopub.execute_input":"2024-10-17T22:25:52.146093Z","iopub.status.idle":"2024-10-17T22:25:52.600744Z","shell.execute_reply.started":"2024-10-17T22:25:52.146045Z","shell.execute_reply":"2024-10-17T22:25:52.599772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Identify common columns and train-only columns\ncommon_columns = list(set(train_data.columns) & set(test_data.columns))\ntrain_only_columns = list(set(train_data.columns) - set(test_data.columns))\n\n# Remove 'sii' from feature columns if present\nif 'sii' in common_columns:\n    common_columns.remove('sii')\nif 'sii' in train_only_columns:\n    train_only_columns.remove('sii')\n\nprint(f\"Number of common columns: {len(common_columns)}\")\nprint(f\"Number of train-only columns: {len(train_only_columns)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:25:52.601860Z","iopub.execute_input":"2024-10-17T22:25:52.602212Z","iopub.status.idle":"2024-10-17T22:25:52.608823Z","shell.execute_reply.started":"2024-10-17T22:25:52.602166Z","shell.execute_reply":"2024-10-17T22:25:52.607885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Separate features and target\nX = train_data[common_columns]\ny = train_data['sii']\nX_test = test_data[common_columns]\n\ny = y.fillna(y.mode()[0])\n\n# KNN Imputation\nimputer = KNNImputer(n_neighbors=5)\nX_imputed = pd.DataFrame(imputer.fit_transform(X), columns=X.columns)\nX_test_imputed = pd.DataFrame(imputer.transform(X_test), columns=X_test.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:25:52.609899Z","iopub.execute_input":"2024-10-17T22:25:52.610160Z","iopub.status.idle":"2024-10-17T22:26:03.597418Z","shell.execute_reply.started":"2024-10-17T22:25:52.610131Z","shell.execute_reply":"2024-10-17T22:26:03.596582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate correlation differences\nhigh_risk = X_imputed[y == 3]\nlow_risk = X_imputed[y != 3]\n\nhigh_risk_corr = high_risk.corr().abs()\nlow_risk_corr = low_risk.corr().abs()\n\ncorr_diff = (high_risk_corr - low_risk_corr).abs()\nmean_corr_diff = corr_diff.mean().sort_values(ascending=False)\ntop_corr_diff_features = mean_corr_diff.head(40).index.tolist() # Select the top 40 features\n\n# Main feature selection process\ncorrelation_threshold = 0.001\ncorr_with_target = X_imputed.corrwith(y).abs()\ncorr_selected_features = corr_with_target[corr_with_target > correlation_threshold].index.tolist()\n\n# T test\ndef calculate_t_test(feature):\n    high_risk_values = high_risk[feature]\n    low_risk_values = low_risk[feature]\n    _, p_value = stats.ttest_ind(high_risk_values, low_risk_values)\n    return p_value\n\nsignificant_difference_threshold = 0.005\ndiff_selected_features = [feature for feature in X_imputed.columns if calculate_t_test(feature) < significant_difference_threshold]\n\n# Statistical feature selection\nselector = SelectKBest(score_func=f_classif, k=50)\nselector.fit(X_imputed, y)\nstat_selected_features = X_imputed.columns[selector.get_support()].tolist()\n\nall_selected_features = list(set(corr_selected_features + \n                                 diff_selected_features + \n                                 top_corr_diff_features + \n                                 stat_selected_features))\n\ndomain_specific_features = ['PreInt_EduHx-computerinternet_hoursday', 'Basic_Demos-Age', 'Fitness_Endurance-Time_Mins', 'stat_88_boxcox', 'stat_84']\nall_selected_features += domain_specific_features\nall_selected_features = list(set(all_selected_features))\n\nprint(f\"Total number of selected features: {len(all_selected_features)}\")\nprint(\"Selected features:\", all_selected_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:33:51.187591Z","iopub.execute_input":"2024-10-17T22:33:51.188329Z","iopub.status.idle":"2024-10-17T22:33:51.675392Z","shell.execute_reply.started":"2024-10-17T22:33:51.188286Z","shell.execute_reply":"2024-10-17T22:33:51.674453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Final feature selection\n# ['BIA-BIA_Activity_Level_num', 'Basic_Demos-Sex', 'stat_32', 'stat_62', 'stat_26', 'stat_54_boxcox', 'Physical-Diastolic_BP', 'stat_59', 'Basic_Demos-Age', 'stat_63', 'stat_24', 'stat_58', 'FGC-FGC_CU_Zone', 'stat_47_boxcox', 'FGC-Season', 'PAQ_C-PAQ_C_Total', 'stat_49', 'PreInt_EduHx-Season', 'stat_46', 'Physical-Waist_Circumference', 'stat_45', 'stat_89', 'stat_93', 'stat_34', 'stat_28', 'BIA-BIA_FFMI', 'BIA-BIA_DEE_boxcox', 'stat_87', 'stat_88_boxcox', 'stat_38_boxcox', 'stat_40_boxcox', 'stat_25', 'stat_43', 'BIA-BIA_LST_boxcox', 'Physical-Weight', 'stat_23', 'Fitness_Endurance-Max_Stage', 'CGAS-CGAS_Score', 'stat_68', 'FGC-FGC_SR', 'stat_79', 'BIA-BIA_DEE', 'stat_51', 'stat_40', 'stat_78_boxcox', 'stat_21', 'stat_56', 'stat_47', 'Physical-Height', 'BIA-BIA_FFM_boxcox', 'combined_actigraphy_stat', 'stat_66', 'stat_36', 'stat_20', 'stat_22', 'stat_73', 'stat_80_boxcox', 'stat_70', 'BIA-BIA_ICW', 'Physical-HeartRate', 'stat_84', 'BIA-BIA_LDM_boxcox', 'BIA_BMI_X_PCIAT_Total', 'stat_66_boxcox', 'stat_16', 'Basic_Demos-Enroll_Season', 'stat_90', 'stat_80', 'stat_94', 'BIA-BIA_BMI', 'BIA_Fat_X_PCIAT_Total', 'stat_23_boxcox', 'BIA-BIA_TBW', 'stat_61', 'stat_86', 'stat_27', 'Fitness_Endurance-Season', 'stat_64', 'BIA-BIA_ICW_boxcox', 'stat_78', 'BIA-BIA_BMC', 'BIA-BIA_FMI_boxcox', 'BIA-BIA_ECW', 'Physical_Weight_X_BIA_BMI', 'SDS-SDS_Total_T', 'stat_38', 'stat_81', 'Physical-Systolic_BP', 'stat_83', 'PAQ_A-PAQ_A_Total', 'BIA-BIA_LDM', 'FGC-FGC_PU_Zone', 'stat_35', 'CGAS_Score_X_PCIAT_Total', 'stat_85', 'BIA-BIA_Frame_num', 'stat_31', 'PreInt_EduHx-computerinternet_hoursday', 'BIA-BIA_FFMI_boxcox', 'FGC-FGC_GS', 'stat_37', 'stat_91', 'BIA-Season', 'stat_50', 'Physical-Season', 'stat_18', 'stat_60', 'stat_72', 'BIA-BIA_Fat', 'stat_88', 'CGAS-Season', 'stat_77', 'stat_71', 'CGAS-CGAS_Score_boxcox', 'SDS-SDS_Total_Raw', 'stat_74', 'stat_95', 'stat_69', 'stat_55', 'PAQ_C-Season', 'PAQ_Total', 'stat_54', 'stat_52', 'BIA-BIA_BMR', 'stat_53', 'stat_90_boxcox', 'stat_29', 'stat_33', 'BIA-BIA_SMM', 'stat_82', 'SDS-Season', 'stat_30', 'BIA-BIA_BMC_boxcox', 'stat_75', 'BIA-BIA_FFM', 'stat_19', 'BIA-BIA_FMI', 'FGC-FGC_TL_Zone', 'BIA-BIA_BMR_boxcox', 'stat_35_boxcox', 'BIA_Fat_X_Physical_Weight', 'stat_65', 'BIA-BIA_ECW_boxcox', 'BIA-BIA_Fat_boxcox', 'stat_14', 'stat_57', 'Fitness_Endurance-Time_Mins', 'BIA-BIA_LST', 'stat_12', 'stat_67', 'Physical_Weight_X_PCIAT_Total', 'stat_13', 'BIA-BIA_TBW_boxcox', 'stat_15', 'FGC-FGC_TL', 'PAQ_A-Season', 'Physical-BMI', 'fitness_score', 'stat_76', 'stat_48', 'stat_17']\n\n# final_features = list(set(top_high_risk_cor_features + top_low_risk_cor_features + top_corr_diff_features))\n\nfinal_features = all_selected_features\n\n# final_features = [ 'BIA-BIA_Fat_boxcox', 'stat_88_boxcox', 'stat_38_boxcox', 'stat_80_boxcox', 'BIA-BIA_FMI_boxcox', 'stat_68', 'stat_32',\n#                    'stat_56', 'stat_80', 'stat_20', 'combined_actigraphy_stat', 'Physical_Weight_X_PCIAT_Total', 'BIA_Fat_X_Physical_Weight',\n#                    'BIA_Fat_X_PCIAT_Total', 'CGAS_Score_X_PCIAT_Total', 'Physical_Weight_X_BIA_BMI', 'BIA_BMI_X_PCIAT_Total', 'BIA-BIA_DEE',\n#                    'stat_90', 'BIA-BIA_BMR']\n\n# Prepare datasets\nX_selected = X_imputed[final_features]\nX_test_selected = X_test_imputed[final_features]\n\nX = X_selected\ny = y\nX_test = X_test_selected","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:33:58.362025Z","iopub.execute_input":"2024-10-17T22:33:58.362747Z","iopub.status.idle":"2024-10-17T22:33:58.372299Z","shell.execute_reply.started":"2024-10-17T22:33:58.362705Z","shell.execute_reply":"2024-10-17T22:33:58.371335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define modeling functions\ndef quadratic_weighted_kappa(y_true, y_pred):\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\ndef threshold_Rounder(oof_non_rounded, thresholds):\n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -quadratic_weighted_kappa(y_true, rounded_p)\n\n# Function to handle infinite values\ndef preprocess_data(X, X_test):\n    # Log transform columns with 'boxcox' in their name\n    boxcox_columns = [col for col in X.columns if 'boxcox' in col]\n    for col in boxcox_columns:\n        X[col] = np.log1p(X[col])\n        X_test[col] = np.log1p(X_test[col])\n    \n    # Replace inf with NaN\n    X = X.replace([np.inf, -np.inf], np.nan)\n    X_test = X_test.replace([np.inf, -np.inf], np.nan)\n    \n    # Fill NaN with median of the column\n    X = X.fillna(X.median())\n    X_test = X_test.fillna(X_test.median())\n    \n    # Function to cap values at 99th percentile\n    def cap_values(column):\n        cap_value = np.percentile(column, 99)\n        return np.minimum(column, cap_value)\n    \n    # Apply capping to all numeric columns\n    numeric_columns = X.select_dtypes(include=[np.number]).columns\n    X[numeric_columns] = X[numeric_columns].apply(cap_values)\n    X_test[numeric_columns] = X_test[numeric_columns].apply(cap_values)\n           \n    # Scale the data\n    scaler = StandardScaler()\n    X_scaled = pd.DataFrame(scaler.fit_transform(X), columns=X.columns)\n    X_test_scaled = pd.DataFrame(scaler.transform(X_test), columns=X_test.columns)\n    \n    return X_scaled, X_test_scaled","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:34:00.864035Z","iopub.execute_input":"2024-10-17T22:34:00.864756Z","iopub.status.idle":"2024-10-17T22:34:00.877217Z","shell.execute_reply.started":"2024-10-17T22:34:00.864711Z","shell.execute_reply":"2024-10-17T22:34:00.876103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Model parameters for tuning\n#def objective(trial, X, y, model_class):\n#    if model_class == LGBMClassifier:\n#        params = {\n#            'n_estimators': trial.suggest_int('n_estimators', 100, 1000),\n#            'learning_rate': trial.suggest_float('learning_rate', 1e-3, 0.1, log=True),\n#            'num_leaves': trial.suggest_int('num_leaves', 20, 1000),\n#            'max_depth': trial.suggest_int('max_depth', 3, 15),\n#            'min_child_samples': trial.suggest_int('min_child_samples', 5, 100),\n#            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.6, 1.0),\n#            'bagging_fraction': trial.suggest_float('bagging_fraction', 0.6, 1.0),\n#            'bagging_freq': trial.suggest_int('bagging_freq', 1, 10),\n#            'lambda_l1': trial.suggest_float('lambda_l1', 1e-8, 100.0, log=True),\n#            'lambda_l2': trial.suggest_float('lambda_l2', 1e-8, 100.0, log=True),\n#        }\n#    elif model_class == XGBClassifier:\n#        params = {\n#            'max_depth': trial.suggest_int('max_depth', 3, 10),\n#            'learning_rate': trial.suggest_float('learning_rate', 1e-3, 0.1, log=True),\n#            'n_estimators': trial.suggest_int('n_estimators', 100, 500),\n#            'min_child_weight': trial.suggest_int('min_child_weight', 1, 10),\n#            'subsample': trial.suggest_float('subsample', 0.6, 1.0),\n#            'colsample_bytree': trial.suggest_float('colsample_bytree', 0.6, 1.0),\n#            'reg_alpha': trial.suggest_float('reg_alpha', 1e-8, 10.0, log=True),\n#            'reg_lambda': trial.suggest_float('reg_lambda', 1e-8, 10.0, log=True),\n#            'tree_method': 'exact'\n#        }\n#    elif model_class == CatBoostClassifier:\n#        params = {\n#            'iterations': trial.suggest_int('iterations', 100, 500),\n#            'learning_rate': trial.suggest_float('learning_rate', 1e-3, 0.1, log=True),\n#            'depth': trial.suggest_int('depth', 4, 10),\n#            'l2_leaf_reg': trial.suggest_float('l2_leaf_reg', 1, 100, log=True),\n#            'bootstrap_type': trial.suggest_categorical('bootstrap_type', ['Bayesian', 'Bernoulli', 'MVS']),\n#            'random_strength': trial.suggest_float('random_strength', 1e-9, 10),\n#        }\n#        if params['bootstrap_type'] == 'Bayesian':\n#            params['bagging_temperature'] = trial.suggest_float('bagging_temperature', 0, 10)\n#        elif params['bootstrap_type'] == 'Bernoulli':\n#            params['subsample'] = trial.suggest_float('subsample', 0.1, 1)\n#    \n#    model = model_class(**params, random_state=42)\n#    score = cross_val_score(model, X, y, cv=5, scoring=make_scorer(quadratic_weighted_kappa))\n#    return score.mean()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hyperparameter tuning\n#def tune_hyperparameters(X, y):\n#    models = [LGBMClassifier, XGBClassifier, CatBoostClassifier]\n#    best_params = {}\n#\n#    for model_class in models:\n#        study = optuna.create_study(direction='maximize')\n#        study.optimize(lambda trial: objective(trial, X, y, model_class), n_trials=200)\n#        best_params[model_class.__name__] = study.best_params\n#\n#    return best_params\n\n# Feature importance\n#def feature_importance(X, y):\n#    rf = RandomForestClassifier(n_estimators=100, random_state=42)\n#    rf.fit(X, y)\n#    importances = pd.DataFrame({'feature': X.columns, 'importance': rf.feature_importances_})\n#    importances = importances.sort_values('importance', ascending=False)\n#    return importances","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#def TrainML(X, y, X_test, sample_submission, n_splits=5, SEED=42):\n#    # Preprocess the data\n#    X, X_test = preprocess_data(X, X_test)\n#    \n#    # Feature importance\n#    print(\"Calculating feature importance...\")\n#    importances = feature_importance(X, y)\n#    print(\"Top 45 important features:\")\n#    print(importances.head(45))\n#    \n#    # Select top 45 features\n#    top_features = importances['feature'][:45].tolist()\n#    X = X[top_features]\n#    X_test = X_test[top_features]\n    \n#    # Tune hyperparameters\n#    print(\"Tuning hyperparameters...\")\n#    best_params = tune_hyperparameters(X, y)\n    \n#    # Create models with tuned hyperparameters\n #   lgbm_model = LGBMClassifier(**best_params['LGBMClassifier'], random_state=SEED)\n #   xgb_model = XGBClassifier(**best_params['XGBClassifier'], random_state=SEED)\n#    catboost_model = CatBoostClassifier(**best_params['CatBoostClassifier'], random_state=SEED)\n#    \n#    # Create VotingClassifier\n#    voting_model = VotingClassifier([\n#        ('lgbm', lgbm_model),\n#        ('xgb', xgb_model),\n#        ('catboost', catboost_model)\n#    ], voting='soft')\n#    \n#    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n#    \n#    oof_preds = np.zeros((len(X), 4))  # 4 classes\n #   test_preds = np.zeros((len(X_test), 4))\n#\n#    for fold, (train_idx, val_idx) in enumerate(tqdm(SKF.split(X, y), total=n_splits, desc=\"Training folds\")):\n#        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n#        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n#        voting_model.fit(X_train, y_train)\n#\n#        oof_preds[val_idx] = voting_model.predict_proba(X_val)\n#        test_preds += voting_model.predict_proba(X_test) / n_splits\n#\n#        val_preds = np.argmax(oof_preds[val_idx], axis=1)\n#        val_score = quadratic_weighted_kappa(y_val, val_preds)\n#        print(f\"Fold {fold + 1} Validation QWK: {val_score:.4f}\")\n#\n#    oof_preds_class = np.argmax(oof_preds, axis=1)\n #   overall_score = quadratic_weighted_kappa(y, oof_preds_class)\n#    print(f\"Overall QWK Score: {overall_score:.4f}\")\n\n#    submission = pd.DataFrame({\n#        'id': sample_submission['id'],\n#        'sii': np.argmax(test_preds, axis=1)\n #   })\n\n    # Save optimal features and best parameters to JSON\n#    output_folder = 'output'\n#    os.makedirs(output_folder, exist_ok=True)\n    \n#    with open(os.path.join(output_folder, 'optimal_features.json'), 'w') as f:\n #       json.dump(top_features, f)\n    \n #   with open(os.path.join(output_folder, 'best_params.json'), 'w') as f:\n #       json.dump(best_params, f)\n\n #   print(f\"Optimal features saved to {os.path.join(output_folder, 'optimal_features.json')}\")\n #   print(f\"Best parameters saved to {os.path.join(output_folder, 'best_params.json')}\")\n\n #   return submission, overall_score, top_features, best_params","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load sample submission\nsample_submission = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:31:43.750514Z","iopub.execute_input":"2024-10-17T22:31:43.750957Z","iopub.status.idle":"2024-10-17T22:31:43.758152Z","shell.execute_reply.started":"2024-10-17T22:31:43.750916Z","shell.execute_reply":"2024-10-17T22:31:43.757390Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the model and to get optimal features and params\n#Submission, score, optimal_features, best_params = TrainML(X, y, X_test, sample_submission)\n\n# Save submission\n#Submission.to_csv('optimal_submission.csv', index=False)\n#print(\"Optimal submission saved to 'submission.csv'\")\n#print(\"Optimal features:\", optimal_features)\n#print(\"Best parameters:\", best_params)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Optimal features\noptimal_features = [\"BIA-BIA_Fat_boxcox\", \"stat_88_boxcox\", \"stat_38_boxcox\", \"stat_80_boxcox\", \"BIA-BIA_FMI_boxcox\", \"stat_68\",\n                    \"stat_32\", \"stat_56\", \"stat_20\", \"combined_actigraphy_stat\", \"BIA_Fat_X_Physical_Weight\", \"Physical_Weight_X_BIA_BMI\",\n                    \"stat_90_boxcox\", \"stat_43\", \"Physical-Weight\", \"stat_55\", \"stat_67\", \"stat_30\", \"stat_19\", \"Physical-Waist_Circumference\",\n                    \"SDS-SDS_Total_T\", \"SDS-SDS_Total_Raw\", \"Physical-BMI\", \"Physical-Height\", \"CGAS-CGAS_Score_boxcox\", \n                    \"Physical-HeartRate\", \"Physical-Systolic_BP\", \"Basic_Demos-Age\", \"Fitness_Endurance-Time_Mins\", \"CGAS-Season\", \n                    \"BIA-BIA_BMR_boxcox\", \"BIA-BIA_ECW_boxcox\", \"PAQ_A-PAQ_A_Total\", \"BIA-BIA_BMC_boxcox\", \"BIA-BIA_LDM_boxcox\", \n                    \"BIA-BIA_FFM_boxcox\", \"BIA-BIA_TBW_boxcox\", \"BIA-BIA_SMM\", \"BIA-BIA_FFMI_boxcox\", \"BIA-BIA_DEE_boxcox\", \"PAQ_C-Season\", \n                    \"BIA-BIA_ICW_boxcox\", \"stat_86\", \"BIA-BIA_LST_boxcox\", \"stat_49\", \"FGC-FGC_GS\", \"stat_78_boxcox\", \n                    \"stat_21\", \"stat_85\", \"stat_54_boxcox\", \"stat_12\", \"stat_27\", \"stat_40_boxcox\", \"stat_36\", \"stat_28\", \"stat_66_boxcox\", \n                    \"stat_31\", \"stat_84\"]\n\n# Best parameters\nbest_params = {\n    'LGBMClassifier': {\n        'n_estimators': 656,\n        'learning_rate': 0.015782890450877014,\n        'num_leaves': 1066,\n        'max_depth': 8,\n        'min_child_samples': 100,\n        'colsample_bytree': 0.940857616810904,\n        'bagging_fraction': 0.9991196168746357,\n        'bagging_freq': 6,\n        'lambda_l1': 6.332269679324449e-06,\n        'lambda_l2': 2.1839610820602853e-06\n    },\n    'XGBClassifier': {\n        'max_depth': 11,\n        'learning_rate': 0.06582943935263044,\n        'n_estimators': 446,\n        'min_child_weight': 8,\n        'subsample': 0.9993341826573182,\n        'colsample_bytree': 0.888783613042681,\n        'reg_alpha': 0.001075694456715294,\n        'reg_lambda': 0.0012448708989684465\n    },\n    'CatBoostClassifier': {\n        'iterations': 490,\n        'learning_rate': 0.09077165664505828,\n        'depth': 5,\n        'l2_leaf_reg': 2.452302163739088,\n        'bootstrap_type': \"Bernoulli\",\n        'random_strength': 0.21138971422950448,\n        'subsample': 0.612028714191788\n    }\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:35:20.180061Z","iopub.execute_input":"2024-10-17T22:35:20.180583Z","iopub.status.idle":"2024-10-17T22:35:20.191036Z","shell.execute_reply.started":"2024-10-17T22:35:20.180539Z","shell.execute_reply":"2024-10-17T22:35:20.190003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def TrainMLOptimal(X, y, X_test, sample_submission, optimal_features, best_params, n_splits=5, SEED=42):\n    # Preprocess the data\n    X, X_test = preprocess_data(X, X_test)\n    \n    # Use optimal features\n    X = X[optimal_features]\n    X_test = X_test[optimal_features]\n    \n    # Create models with best parameters\n    lgbm_model = LGBMClassifier(**best_params['LGBMClassifier'], random_state=SEED)\n    xgb_model = XGBClassifier(**best_params['XGBClassifier'], random_state=SEED)\n    catboost_model = CatBoostClassifier(**best_params['CatBoostClassifier'], random_state=SEED)\n    \n    # Create VotingClassifier\n    voting_model = VotingClassifier([\n        ('lgbm', lgbm_model),\n        ('xgb', xgb_model),\n        ('catboost', catboost_model)\n    ], voting='soft')\n    \n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=SEED)\n    \n    oof_preds = np.zeros((len(X), 4))  # 4 classes\n    test_preds = np.zeros((len(X_test), 4))\n\n    for fold, (train_idx, val_idx) in enumerate(tqdm(SKF.split(X, y), total=n_splits, desc=\"Training folds\")):\n        X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n        y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n        voting_model.fit(X_train, y_train)\n\n        oof_preds[val_idx] = voting_model.predict_proba(X_val)\n        test_preds += voting_model.predict_proba(X_test) / n_splits\n\n        val_preds = np.argmax(oof_preds[val_idx], axis=1)\n        val_score = quadratic_weighted_kappa(y_val, val_preds)\n        print(f\"Fold {fold + 1} Validation QWK: {val_score:.4f}\")\n\n    oof_preds_class = np.argmax(oof_preds, axis=1)\n    overall_score = quadratic_weighted_kappa(y, oof_preds_class)\n    print(f\"Overall QWK Score: {overall_score:.4f}\")\n\n    submission = pd.DataFrame({\n        'id': sample_submission['id'],\n        'sii': np.argmax(test_preds, axis=1)\n    })\n\n    return submission, overall_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:34:12.381865Z","iopub.execute_input":"2024-10-17T22:34:12.382270Z","iopub.status.idle":"2024-10-17T22:34:12.393907Z","shell.execute_reply.started":"2024-10-17T22:34:12.382230Z","shell.execute_reply":"2024-10-17T22:34:12.392880Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the optimal model and get predictions\noptimal_submission, optimal_score = TrainMLOptimal(X, y, X_test, sample_submission, optimal_features, best_params)\n\n# Save the optimal submission\noptimal_submission.to_csv('submission.csv', index=False)\nprint(\"Optimal submission saved to 'submission.csv'\")\nprint(f\"Optimal score: {optimal_score:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-17T22:35:24.360501Z","iopub.execute_input":"2024-10-17T22:35:24.360858Z","iopub.status.idle":"2024-10-17T22:36:45.485149Z","shell.execute_reply.started":"2024-10-17T22:35:24.360825Z","shell.execute_reply":"2024-10-17T22:36:45.484170Z"}},"outputs":[],"execution_count":null}]}