{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":9541977,"sourceType":"datasetVersion","datasetId":5812822},{"sourceId":9554556,"sourceType":"datasetVersion","datasetId":5821860}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-14T19:46:20.464394Z","iopub.execute_input":"2024-10-14T19:46:20.464786Z","iopub.status.idle":"2024-10-14T19:46:20.471062Z","shell.execute_reply.started":"2024-10-14T19:46:20.464745Z","shell.execute_reply":"2024-10-14T19:46:20.469708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pathlib import Path\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:20.476512Z","iopub.execute_input":"2024-10-14T19:46:20.476836Z","iopub.status.idle":"2024-10-14T19:46:21.406011Z","shell.execute_reply.started":"2024-10-14T19:46:20.476800Z","shell.execute_reply":"2024-10-14T19:46:21.404999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df=pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ntest_df=pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:21.407273Z","iopub.execute_input":"2024-10-14T19:46:21.407731Z","iopub.status.idle":"2024-10-14T19:46:21.466552Z","shell.execute_reply.started":"2024-10-14T19:46:21.407694Z","shell.execute_reply":"2024-10-14T19:46:21.465521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:21.469092Z","iopub.execute_input":"2024-10-14T19:46:21.469494Z","iopub.status.idle":"2024-10-14T19:46:21.508913Z","shell.execute_reply.started":"2024-10-14T19:46:21.469455Z","shell.execute_reply":"2024-10-14T19:46:21.507890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.columns","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:21.510167Z","iopub.execute_input":"2024-10-14T19:46:21.510492Z","iopub.status.idle":"2024-10-14T19:46:21.518018Z","shell.execute_reply.started":"2024-10-14T19:46:21.510458Z","shell.execute_reply":"2024-10-14T19:46:21.516961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop rows where 'sii' is missing\ntrain_df = train_df.dropna(subset=['sii'])","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:21.519358Z","iopub.execute_input":"2024-10-14T19:46:21.519705Z","iopub.status.idle":"2024-10-14T19:46:21.530396Z","shell.execute_reply.started":"2024-10-14T19:46:21.519668Z","shell.execute_reply":"2024-10-14T19:46:21.529315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_rows', None)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:21.531789Z","iopub.execute_input":"2024-10-14T19:46:21.532201Z","iopub.status.idle":"2024-10-14T19:46:21.538235Z","shell.execute_reply.started":"2024-10-14T19:46:21.532136Z","shell.execute_reply":"2024-10-14T19:46:21.537227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List of columns to drop from merged_df_train\ncolumns_to_drop = ['PCIAT-Season', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_15', 'PCIAT-PCIAT_19', \n                   'PCIAT-PCIAT_20', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_05', \n                   'PCIAT-PCIAT_13', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_02', \n                   'PCIAT-PCIAT_10', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_07', 'PCIAT-PCIAT_11', \n                   'PCIAT-PCIAT_12', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_16', \n                   'PCIAT-PCIAT_06', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_17','PCIAT-PCIAT_Total']\n\ntrain_df = train_df.drop(columns=columns_to_drop)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:21.539651Z","iopub.execute_input":"2024-10-14T19:46:21.540191Z","iopub.status.idle":"2024-10-14T19:46:21.550118Z","shell.execute_reply.started":"2024-10-14T19:46:21.540118Z","shell.execute_reply":"2024-10-14T19:46:21.549213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_percentage = train_df.isnull().mean() * 100\n\n# Print the percentage of missing values for each column\nprint(missing_percentage)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:21.551515Z","iopub.execute_input":"2024-10-14T19:46:21.552363Z","iopub.status.idle":"2024-10-14T19:46:21.565321Z","shell.execute_reply.started":"2024-10-14T19:46:21.552316Z","shell.execute_reply":"2024-10-14T19:46:21.564273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_percentage = test_df.isnull().mean() * 100\n\n# Print the percentage of missing values for each column\nprint(missing_percentage)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:21.566693Z","iopub.execute_input":"2024-10-14T19:46:21.567024Z","iopub.status.idle":"2024-10-14T19:46:21.576339Z","shell.execute_reply.started":"2024-10-14T19:46:21.566989Z","shell.execute_reply":"2024-10-14T19:46:21.575118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:21.577825Z","iopub.execute_input":"2024-10-14T19:46:21.578323Z","iopub.status.idle":"2024-10-14T19:46:21.597488Z","shell.execute_reply.started":"2024-10-14T19:46:21.578274Z","shell.execute_reply":"2024-10-14T19:46:21.596476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the id column before dropping\ntest_ids = test_df['id']","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:21.598901Z","iopub.execute_input":"2024-10-14T19:46:21.599323Z","iopub.status.idle":"2024-10-14T19:46:21.603641Z","shell.execute_reply.started":"2024-10-14T19:46:21.599286Z","shell.execute_reply":"2024-10-14T19:46:21.602671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List of columns to drop\ncolumns_to_drop = [\n    'CGAS-CGAS_Score', 'BIA-Season', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR',\n    'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n    'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total',\n    'PAQ_C-Season', 'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', \n    'Physical-Waist_Circumference', 'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage', \n    'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', \n    'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total','id'\n]\n\n# Drop columns from train and test dataframes\ntrain_df = train_df.drop(columns=columns_to_drop)\ntest_df = test_df.drop(columns=columns_to_drop)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:21.608775Z","iopub.execute_input":"2024-10-14T19:46:21.609261Z","iopub.status.idle":"2024-10-14T19:46:21.618535Z","shell.execute_reply.started":"2024-10-14T19:46:21.609203Z","shell.execute_reply":"2024-10-14T19:46:21.617497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Step 1: Identify and separate numeric columns\nnumeric_cols = train_df.select_dtypes(include=['number']).columns\ncategorical_cols = train_df.select_dtypes(include=['object']).columns\n\n# Step 2: Fill missing values only in numeric columns\ntrain_df[numeric_cols] = train_df[numeric_cols].fillna(train_df[numeric_cols].mean())\n\n# Step 3: Optional: Handle categorical columns (e.g., fill with mode or use one-hot encoding)\n# For simplicity, let's fill missing categorical values with the mode (most frequent value)\nfor col in categorical_cols:\n    train_df[col].fillna(train_df[col].mode()[0], inplace=True)\n\n# Step 4: Convert categorical variables to dummy/indicator variables if necessary\ntrain_df = pd.get_dummies(train_df, columns=categorical_cols, drop_first=True)\n\n# Step 5: Separate target (sii) and features\nX = train_df.drop(columns=['sii'])  # Features\ny = train_df['sii']  # Target\n\n# Step 6: Split data into train and test sets (optional, but recommended)\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)\n\n# Step 7: Train a RandomForest Classifier\nrf_model = RandomForestClassifier(random_state=42)\nrf_model.fit(X_train, y_train)\n\n# Step 8: Get Feature Importances\nimportances = rf_model.feature_importances_\nfeature_names = X.columns\n\n# Step 9: Create a DataFrame for better visualization\nfeature_importances = pd.DataFrame({'feature': feature_names, 'importance': importances})\nfeature_importances = feature_importances.sort_values(by='importance', ascending=False)\n\n# Step 10: Visualize the Top 10 Important Features\nplt.figure(figsize=(10, 6))\nsns.barplot(x='importance', y='feature', data=feature_importances.head(20))\nplt.title('Top 10 Important Features')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:21.619703Z","iopub.execute_input":"2024-10-14T19:46:21.620016Z","iopub.status.idle":"2024-10-14T19:46:22.799057Z","shell.execute_reply.started":"2024-10-14T19:46:21.619983Z","shell.execute_reply":"2024-10-14T19:46:22.797877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n\n# # Updated list of columns to drop (including the new ones with high missing values)\n# columns_to_drop = [\n#     'id',\n#     'Fitness_Endurance-Max_Stage',\n#     'Fitness_Endurance-Time_Sec',\n#     'Physical-Waist_Circumference',\n#     'BIA-BIA_Fat',\n#     'Basic_Demos-Sex',\n#     'BIA-BIA_BMC',\n#     'FGC-FGC_SRR',\n#     'BIA-BIA_LDM',\n#     'PAQ_C-PAQ_C_Total',\n#     'FGC-FGC_SRL',\n#     'FGC-FGC_SRR_Zone',\n#     'FGC-FGC_PU_Zone',\n#     'FGC-FGC_SRL_Zone',\n#     'FGC-FGC_TL_Zone',\n#     'BIA-Season',\n#     'BIA-BIA_FMI',\n#     'Physical-Season',\n#     'CGAS-Season',\n#     'PAQ_A-Season',\n#     'PAQ_A-PAQ_A_Total',\n#     'PAQ_C-Season',\n#     'FGC-FGC_GSND',               # 68.13%\n#     'FGC-FGC_GSND_Zone',          # 68.42%\n#     'FGC-FGC_GSD',                # 68.17%\n#     'FGC-FGC_GSD_Zone',           # 68.42%\n#     'Fitness_Endurance-Season',    # 53.95%\n#     'Fitness_Endurance-Time_Mins'  # 73.39%\n# ]\n\n# # Dropping columns from train_df and test_df\n# train_df = train_df.drop(columns=columns_to_drop, errors='ignore')\n# test_df = test_df.drop(columns=columns_to_drop, errors='ignore')\n\n# # Print confirmation\n# print(f\"Columns dropped from train_df: {', '.join(columns_to_drop)}\")\n# print(f\"Columns dropped from test_df: {', '.join(columns_to_drop)}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:22.800482Z","iopub.execute_input":"2024-10-14T19:46:22.800819Z","iopub.status.idle":"2024-10-14T19:46:22.806248Z","shell.execute_reply.started":"2024-10-14T19:46:22.800783Z","shell.execute_reply":"2024-10-14T19:46:22.805372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Define the columns to keep for train and test\n# columns_to_keep_train = [\n#     'Basic_Demos-Age',\n#     'CGAS-CGAS_Score',\n#     'Physical-BMI',\n#     'Physical-Height',\n#     'Physical-Weight',\n#     'Physical-Diastolic_BP',\n#     'Physical-HeartRate',\n#     'Physical-Systolic_BP',\n#     'BIA-BIA_BMR',\n#     'BIA-BIA_DEE',\n#     'BIA-BIA_SMM',\n#     'SDS-SDS_Total_Raw',\n#     'SDS-SDS_Total_T',\n#     'PreInt_EduHx-computerinternet_hoursday',\n#     'sii'  # Target feature\n# ]\n\n# columns_to_keep_test = [\n#     'Basic_Demos-Age',\n#     'CGAS-CGAS_Score',\n#     'Physical-BMI',\n#     'Physical-Height',\n#     'Physical-Weight',\n#     'Physical-Diastolic_BP',\n#     'Physical-HeartRate',\n#     'Physical-Systolic_BP',\n#     'BIA-BIA_BMR',\n#     'BIA-BIA_DEE',\n#     'BIA-BIA_SMM',\n#     'SDS-SDS_Total_Raw',\n#     'SDS-SDS_Total_T',\n#     'PreInt_EduHx-computerinternet_hoursday'\n# ]\n\n# # Drop other columns for train_df and test_df\n# train_df = train_df[columns_to_keep_train]\n# test_df = test_df[columns_to_keep_test]\n\n# # Check the updated DataFrames\n# print(train_df.head())\n# print(test_df.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:22.807524Z","iopub.execute_input":"2024-10-14T19:46:22.808070Z","iopub.status.idle":"2024-10-14T19:46:22.820279Z","shell.execute_reply.started":"2024-10-14T19:46:22.808034Z","shell.execute_reply":"2024-10-14T19:46:22.819322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:22.821769Z","iopub.execute_input":"2024-10-14T19:46:22.822606Z","iopub.status.idle":"2024-10-14T19:46:22.844190Z","shell.execute_reply.started":"2024-10-14T19:46:22.822555Z","shell.execute_reply":"2024-10-14T19:46:22.843065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:22.845394Z","iopub.execute_input":"2024-10-14T19:46:22.845725Z","iopub.status.idle":"2024-10-14T19:46:22.857921Z","shell.execute_reply.started":"2024-10-14T19:46:22.845689Z","shell.execute_reply":"2024-10-14T19:46:22.856777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Fill missing values for numerical columns using the mean\ntry:\n    numerical_columns = train_df.select_dtypes(include=['float64', 'int64']).columns\n    # Check if numerical columns exist in both train_df and test_df\n    common_numerical_columns = [col for col in numerical_columns if col in test_df.columns]\n    if common_numerical_columns:\n        train_df[common_numerical_columns] = train_df[common_numerical_columns].fillna(train_df[common_numerical_columns].mean())\n        test_df[common_numerical_columns] = test_df[common_numerical_columns].fillna(test_df[common_numerical_columns].mean())\n    else:\n        print(\"No common numerical columns found to fill in train_df and test_df.\")\nexcept Exception as e:\n    print(f\"Error occurred while filling missing values in numerical columns: {e}\")\n\n# Fill missing values for categorical columns using the mode\ntry:\n    categorical_columns = train_df.select_dtypes(include=['object']).columns\n    # Check if categorical columns exist in both train_df and test_df\n    common_categorical_columns = [col for col in categorical_columns if col in test_df.columns]\n    if common_categorical_columns:\n        train_df[common_categorical_columns] = train_df[common_categorical_columns].fillna(train_df[common_categorical_columns].mode().iloc[0])\n        test_df[common_categorical_columns] = test_df[common_categorical_columns].fillna(test_df[common_categorical_columns].mode().iloc[0])\n    else:\n        print(\"No common categorical columns found to fill in train_df and test_df.\")\nexcept Exception as e:\n    print(f\"Error occurred while filling missing values in categorical columns: {e}\")\n\n# Print confirmation\nprint(\"Missing values filled for both numerical and categorical columns in train_df and test_df.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:22.859284Z","iopub.execute_input":"2024-10-14T19:46:22.860275Z","iopub.status.idle":"2024-10-14T19:46:22.892195Z","shell.execute_reply.started":"2024-10-14T19:46:22.860233Z","shell.execute_reply":"2024-10-14T19:46:22.891197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:22.893457Z","iopub.execute_input":"2024-10-14T19:46:22.893789Z","iopub.status.idle":"2024-10-14T19:46:22.920921Z","shell.execute_reply.started":"2024-10-14T19:46:22.893754Z","shell.execute_reply":"2024-10-14T19:46:22.919927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:22.922228Z","iopub.execute_input":"2024-10-14T19:46:22.922569Z","iopub.status.idle":"2024-10-14T19:46:22.950441Z","shell.execute_reply.started":"2024-10-14T19:46:22.922535Z","shell.execute_reply":"2024-10-14T19:46:22.949384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.columns","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:22.951673Z","iopub.execute_input":"2024-10-14T19:46:22.951996Z","iopub.status.idle":"2024-10-14T19:46:22.958689Z","shell.execute_reply.started":"2024-10-14T19:46:22.951963Z","shell.execute_reply":"2024-10-14T19:46:22.957643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom imblearn.over_sampling import SMOTE\nfrom sklearn.model_selection import train_test_split\n\n# Assuming your DataFrame is already named train_df and contains categorical variables\n\n# Define the target variable and features\ntarget = 'sii'\nX = train_df.drop(columns=[target])  # Features\ny = train_df[target]  # Target\n\n# Identify categorical columns\ncategorical_cols = X.select_dtypes(include=['object']).columns.tolist()\n\n# Perform One-Hot Encoding\nX_encoded = pd.get_dummies(X, columns=categorical_cols, drop_first=True)\n\n# Split the data into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(X_encoded, y, test_size=0.2, random_state=42, stratify=y)\n\n# Initialize SMOTE\nsmote = SMOTE(random_state=42)\n\ntry:\n    # Apply SMOTE to the training data\n    X_resampled, y_resampled = smote.fit_resample(X_train, y_train)\n    \n    # Convert resampled arrays back to DataFrames\n    train_df_resampled = pd.DataFrame(X_resampled, columns=X_encoded.columns)\n    train_df_resampled[target] = y_resampled\n\n    # Print the original and new class distribution\n    print(\"Original class distribution:\")\n    print(y.value_counts())\n\n    print(\"\\nResampled class distribution:\")\n    print(y_resampled.value_counts())\n\n    # If you want to also see the new shape of the resampled DataFrame\n    print(f\"\\nResampled DataFrame shape: {train_df_resampled.shape}\")\n\nexcept ValueError as e:\n    print(f\"Error occurred while applying SMOTE: {e}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:22.960136Z","iopub.execute_input":"2024-10-14T19:46:22.960577Z","iopub.status.idle":"2024-10-14T19:46:23.072667Z","shell.execute_reply.started":"2024-10-14T19:46:22.960531Z","shell.execute_reply":"2024-10-14T19:46:23.071535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" missing_percentage = train_df_resampled.isnull().mean() * 100\n\n# Print the percentage of missing values for each column\nprint(missing_percentage)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:23.073989Z","iopub.execute_input":"2024-10-14T19:46:23.074631Z","iopub.status.idle":"2024-10-14T19:46:23.088050Z","shell.execute_reply.started":"2024-10-14T19:46:23.074582Z","shell.execute_reply":"2024-10-14T19:46:23.087131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\n# Suppress warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:23.089314Z","iopub.execute_input":"2024-10-14T19:46:23.089644Z","iopub.status.idle":"2024-10-14T19:46:23.095139Z","shell.execute_reply.started":"2024-10-14T19:46:23.089610Z","shell.execute_reply":"2024-10-14T19:46:23.093660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.ensemble import RandomForestClassifier\n# from sklearn.model_selection import GridSearchCV\n# from sklearn.metrics import classification_report, confusion_matrix, accuracy_score, cohen_kappa_score\n\n# # Define the parameter grid for GridSearchCV\n# param_grid = {\n#     'n_estimators': [100, 200, 300],\n#     'max_features': ['auto', 'sqrt', 'log2'],\n#     'max_depth': [None, 10, 20, 30],\n#     'min_samples_split': [2, 5, 10],\n#     'min_samples_leaf': [1, 2, 4],\n# }\n\n# # Initialize the model\n# rf_model = RandomForestClassifier(random_state=42)\n\n# # Initialize GridSearchCV\n# grid_search = GridSearchCV(estimator=rf_model, param_grid=param_grid, \n#                            cv=5, n_jobs=-1, verbose=2, scoring='accuracy')\n\n# # Fit the model on the resampled data using GridSearchCV\n# grid_search.fit(X_resampled, y_resampled)\n\n# # Get the best model from GridSearchCV\n# best_rf_model = grid_search.best_estimator_\n\n# # Make predictions on the validation set\n# y_val_pred = best_rf_model.predict(X_val)\n\n# # Calculate accuracy\n# accuracy = accuracy_score(y_val, y_val_pred)\n# print(f\"Accuracy: {accuracy:.4f}\")\n\n# # Calculate weighted quadratic Cohen's kappa score\n# kappa_score = cohen_kappa_score(y_val, y_val_pred, weights='quadratic')\n# print(f\"Weighted Quadratic Cohen's Kappa Score: {kappa_score:.4f}\")\n\n# # Print classification report\n# print(\"Classification Report:\")\n# print(classification_report(y_val, y_val_pred))\n\n# # Print confusion matrix\n# print(\"Confusion Matrix:\")\n# print(confusion_matrix(y_val, y_val_pred))\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T19:46:23.609423Z","iopub.execute_input":"2024-10-14T19:46:23.610371Z","iopub.status.idle":"2024-10-14T20:05:35.762457Z","shell.execute_reply.started":"2024-10-14T19:46:23.610325Z","shell.execute_reply":"2024-10-14T20:05:35.760791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import xgboost as xgb\n# from sklearn.model_selection import GridSearchCV\n# from sklearn.metrics import classification_report, confusion_matrix, accuracy_score, cohen_kappa_score\n\n# # Define the parameter grid for GridSearchCV\n# param_grid = {\n#     'n_estimators': [100, 200, 300],\n#     'max_depth': [3, 5, 7],\n#     'learning_rate': [0.01, 0.1, 0.2],\n#     'subsample': [0.6, 0.8, 1.0],\n#     'colsample_bytree': [0.6, 0.8, 1.0],\n# }\n\n# # Initialize the XGBoost model\n# xgb_model = xgb.XGBClassifier(random_state=42, use_label_encoder=False, eval_metric='logloss')\n\n# # Initialize GridSearchCV\n# grid_search = GridSearchCV(estimator=xgb_model, param_grid=param_grid, \n#                            cv=5, n_jobs=-1, verbose=2, scoring='accuracy')\n\n# # Fit the model on the resampled data using GridSearchCV\n# grid_search.fit(X_resampled, y_resampled)\n\n# # Get the best model from GridSearchCV\n# best_xgb_model = grid_search.best_estimator_\n\n# # Make predictions on the validation set\n# y_val_pred = best_xgb_model.predict(X_val)\n\n# # Calculate accuracy\n# accuracy = accuracy_score(y_val, y_val_pred)\n# print(f\"Accuracy: {accuracy:.4f}\")\n\n# # Calculate weighted quadratic Cohen's kappa score\n# kappa_score = cohen_kappa_score(y_val, y_val_pred, weights='quadratic')\n# print(f\"Weighted Quadratic Cohen's Kappa Score: {kappa_score:.4f}\")\n\n# # Print classification report\n# print(\"Classification Report:\")\n# print(classification_report(y_val, y_val_pred))\n\n# # Print confusion matrix\n# print(\"Confusion Matrix:\")\n# print(confusion_matrix(y_val, y_val_pred))\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T14:53:45.652585Z","iopub.execute_input":"2024-10-14T14:53:45.653003Z","iopub.status.idle":"2024-10-14T15:18:09.081175Z","shell.execute_reply.started":"2024-10-14T14:53:45.652939Z","shell.execute_reply":"2024-10-14T15:18:09.079916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import classification_report, confusion_matrix, accuracy_score, cohen_kappa_score\n\n# Define the parameter grid for GridSearchCV\nparam_grid = {\n    'n_estimators': [100, 200, 300],\n    'max_depth': [-1, 10, 20, 30],\n    'learning_rate': [0.01, 0.1, 0.2],\n    'num_leaves': [31, 50, 70],\n    'min_data_in_leaf': [20, 40, 60],\n}\n\n# Initialize the LightGBM model\nlgb_model = lgb.LGBMClassifier(random_state=42)\n\n# Initialize GridSearchCV\ngrid_search = GridSearchCV(estimator=lgb_model, param_grid=param_grid, \n                           cv=5, n_jobs=-1, verbose=2, scoring='accuracy')\n\n# Fit the model on the resampled data using GridSearchCV\ngrid_search.fit(X_resampled, y_resampled)\n\n# Get the best model from GridSearchCV\nbest_lgb_model = grid_search.best_estimator_\n\n# Make predictions on the validation set\ny_val_pred = best_lgb_model.predict(X_val)\n\n# Calculate accuracy\naccuracy = accuracy_score(y_val, y_val_pred)\nprint(f\"Accuracy: {accuracy:.4f}\")\n\n# Calculate weighted quadratic Cohen's kappa score\nkappa_score = cohen_kappa_score(y_val, y_val_pred, weights='quadratic')\nprint(f\"Weighted Quadratic Cohen's Kappa Score: {kappa_score:.4f}\")\n\n# Print classification report\nprint(\"Classification Report:\")\nprint(classification_report(y_val, y_val_pred))\n\n# Print confusion matrix\nprint(\"Confusion Matrix:\")\nprint(confusion_matrix(y_val, y_val_pred))","metadata":{"execution":{"iopub.status.busy":"2024-10-14T20:17:15.827488Z","iopub.execute_input":"2024-10-14T20:17:15.828450Z","iopub.status.idle":"2024-10-14T20:17:25.618212Z","shell.execute_reply.started":"2024-10-14T20:17:15.828397Z","shell.execute_reply":"2024-10-14T20:17:25.616157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import xgboost as xgb\n# from sklearn.model_selection import GridSearchCV, train_test_split, KFold\n# from sklearn.metrics import mean_squared_error, r2_score\n# from sklearn.preprocessing import StandardScaler\n# from xgboost import XGBRegressor\n# import numpy as np\n\n# # Define features and target variable\n# X = train_df.drop(columns=['PCIAT-PCIAT_Total'])  # Drop the target column\n# y = train_df['PCIAT-PCIAT_Total']\n\n# # Scale features (optional but often beneficial)\n# scaler = StandardScaler()\n# X_scaled = scaler.fit_transform(X)\n\n# # Split the data into training and validation sets\n# X_train, X_val, y_train, y_val = train_test_split(X_scaled, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:13.934753Z","iopub.execute_input":"2024-10-14T15:18:13.935368Z","iopub.status.idle":"2024-10-14T15:18:13.940964Z","shell.execute_reply.started":"2024-10-14T15:18:13.935326Z","shell.execute_reply":"2024-10-14T15:18:13.939775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Create the XGBRegressor object\n# xgb_model = XGBRegressor(objective='reg:squarederror', eval_metric='rmse', silent=True)\n\n# # Hyperparameter grid for tuning\n# param_grid = {\n#     'n_estimators': [100, 200, 300],  # Increased range\n#     'learning_rate': [0.01, 0.05, 0.1, 0.2],  # Adding smaller learning rates for better convergence\n#     'max_depth': [3, 4, 5, 6],\n#     'subsample': [0.6, 0.8, 1.0],\n#     'colsample_bytree': [0.6, 0.8, 1.0],\n#     'reg_alpha': [0, 0.01, 0.1],  # L1 regularization\n#     'reg_lambda': [0, 0.01, 0.1]  # L2 regularization\n# }\n\n# # Set up GridSearchCV with K-Fold Cross-Validation\n# kf = KFold(n_splits=5, shuffle=True, random_state=42)  # Using K-Fold Cross-Validation\n# grid_search = GridSearchCV(estimator=xgb_model, param_grid=param_grid, \n#                            scoring='neg_mean_squared_error', \n#                            cv=kf, n_jobs=-1, verbose=2)\n\n# # Fit the model\n# grid_search.fit(X_train, y_train)\n\n# # Get the best parameters and best score\n# best_params = grid_search.best_params_\n# best_score = -grid_search.best_score_  # negate because we used neg_mean_squared_error\n\n# print(f'Best Parameters: {best_params}')\n# print(f'Best CV MSE: {best_score}')","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:13.942549Z","iopub.execute_input":"2024-10-14T15:18:13.942998Z","iopub.status.idle":"2024-10-14T15:18:13.958524Z","shell.execute_reply.started":"2024-10-14T15:18:13.942928Z","shell.execute_reply":"2024-10-14T15:18:13.957447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Create a model with the best parameters\n# best_model = XGBRegressor(**best_params, objective='reg:squarederror', eval_metric='rmse')\n\n# # Fit the model with early stopping to avoid overfitting\n# best_model.fit(X_train, y_train, \n#                eval_set=[(X_val, y_val)], \n#                early_stopping_rounds=25,  # More balanced early stopping\n#                verbose=True)\n\n# # Make predictions on the validation set\n# y_pred = best_model.predict(X_val)\n\n# # Calculate and print the MSE, R² score, and RMSE\n# mse = mean_squared_error(y_val, y_pred)\n# r2 = r2_score(y_val, y_pred)\n# rmse = np.sqrt(mse)\n\n# print(f'Validation MSE: {mse}')\n# print(f'Validation RMSE: {rmse}')\n# print(f'Validation R²: {r2}')","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:13.963059Z","iopub.execute_input":"2024-10-14T15:18:13.963504Z","iopub.status.idle":"2024-10-14T15:18:13.969788Z","shell.execute_reply.started":"2024-10-14T15:18:13.963448Z","shell.execute_reply":"2024-10-14T15:18:13.968516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n# from sklearn.model_selection import train_test_split, GridSearchCV\n# from sklearn.ensemble import RandomForestClassifier\n# from sklearn.metrics import accuracy_score, classification_report\n# import numpy as np\n# from sklearn.metrics import confusion_matrix\n\n# # Function to calculate Quadratic Weighted Kappa (QWK)\n# def quadratic_weighted_kappa(y_true, y_pred, N):\n#     O = confusion_matrix(y_true, y_pred, labels=np.arange(N))\n\n#     W = np.zeros((N, N))\n#     for i in range(N):\n#         for j in range(N):\n#             W[i, j] = ((i - j) ** 2) / ((N - 1) ** 2)\n\n#     actual_hist = np.bincount(y_true, minlength=N)\n#     predicted_hist = np.bincount(y_pred, minlength=N)\n#     E = np.outer(actual_hist, predicted_hist) / np.sum(actual_hist)\n\n#     numerator = np.sum(W * O)\n#     denominator = np.sum(W * E)\n    \n#     kappa = 1 - (numerator / denominator)\n#     return kappa\n\n# # Step 1: Define the feature set (X) and target variable (y)\n# X = train_df.drop(columns=['sii'])  # Drop the target column 'sii' from features\n# y = train_df['sii']                 # Target column is 'sii'\n\n# # Step 2: Split the data into training and testing sets\n# X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# # Step 3: Initialize the RandomForestClassifier\n# rf_clf = RandomForestClassifier(random_state=42)\n\n# # Step 4: Set up hyperparameter grid for tuning\n# param_grid = {\n#     'n_estimators': [50, 100, 200],         # Number of trees in the forest\n#     'max_depth': [5, 10, 15, None],         # Maximum depth of the tree\n#     'min_samples_split': [2, 5, 10],        # Minimum number of samples required to split a node\n#     'min_samples_leaf': [1, 2, 4],          # Minimum number of samples required at each leaf node\n#     'bootstrap': [True, False]              # Method to sample data points (with or without replacement)\n# }\n\n# # Step 5: Use GridSearchCV for hyperparameter tuning\n# grid_search = GridSearchCV(estimator=rf_clf, param_grid=param_grid, \n#                            cv=3, n_jobs=-1, verbose=2, scoring='accuracy')\n\n# # Step 6: Fit the model with the best parameters found from GridSearch\n# grid_search.fit(X_train, y_train)\n\n# # Step 7: Get the best parameters and make predictions\n# best_params = grid_search.best_params_\n# best_rf_clf = grid_search.best_estimator_\n\n# # Step 8: Make predictions on the test set using the best estimator\n# y_pred = best_rf_clf.predict(X_test)\n\n# # Step 9: Convert predictions to integer (if needed)\n# y_pred = np.round(y_pred).astype(int)\n\n# # Step 10: Evaluate the model\n# accuracy = accuracy_score(y_test, y_pred)\n# report = classification_report(y_test, y_pred, zero_division=0)  # Add zero_division parameter\n\n# # Calculate the Quadratic Weighted Kappa (QWK)\n# unique_classes = len(np.unique(y))  # Get the number of unique classes in the target\n# qwk_score = quadratic_weighted_kappa(y_test, y_pred, unique_classes)\n\n# # Output results\n# print(f'Best Parameters: {best_params}')\n# print(f'Accuracy: {accuracy}')\n# print('Classification Report:')\n# print(report)\n# print(f'Quadratic Weighted Kappa (QWK): {qwk_score}')\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:13.971332Z","iopub.execute_input":"2024-10-14T15:18:13.971816Z","iopub.status.idle":"2024-10-14T15:18:13.985510Z","shell.execute_reply.started":"2024-10-14T15:18:13.971748Z","shell.execute_reply":"2024-10-14T15:18:13.984351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n# from sklearn.model_selection import train_test_split, GridSearchCV\n# from xgboost import XGBClassifier  # Import XGBClassifier\n# from sklearn.metrics import accuracy_score, classification_report\n# import numpy as np\n# from sklearn.metrics import confusion_matrix\n\n# # Function to calculate Quadratic Weighted Kappa (QWK)\n# def quadratic_weighted_kappa(y_true, y_pred, N):\n#     O = confusion_matrix(y_true, y_pred, labels=np.arange(N))\n\n#     W = np.zeros((N, N))\n#     for i in range(N):\n#         for j in range(N):\n#             W[i, j] = ((i - j) ** 2) / ((N - 1) ** 2)\n\n#     actual_hist = np.bincount(y_true, minlength=N)\n#     predicted_hist = np.bincount(y_pred, minlength=N)\n#     E = np.outer(actual_hist, predicted_hist) / np.sum(actual_hist)\n\n#     numerator = np.sum(W * O)\n#     denominator = np.sum(W * E)\n    \n#     kappa = 1 - (numerator / denominator)\n#     return kappa\n\n# # Step 1: Define the feature set (X) and target variable (y)\n# X = train_df.drop(columns=['sii'])  # Drop the target column 'sii' from features\n# y = train_df['sii']                 # Target column is 'sii'\n\n# # Step 2: Split the data into training and testing sets\n# X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# # Step 3: Initialize the XGBClassifier\n# xgb_clf = XGBClassifier(use_label_encoder=False, eval_metric='mlogloss', random_state=42)\n\n# # Step 4: Set up hyperparameter grid for tuning\n# param_grid = {\n#     'n_estimators': [50, 100, 200],         # Number of trees in the forest\n#     'max_depth': [5, 10, 15],               # Maximum depth of the tree\n#     'min_child_weight': [1, 3, 5],          # Minimum sum of instance weight (hessian) needed in a child\n#     'learning_rate': [0.01, 0.1, 0.2],      # Step size shrinkage\n#     'subsample': [0.6, 0.8, 1.0]            # Fraction of samples used for fitting the individual base learners\n# }\n\n# # Step 5: Use GridSearchCV for hyperparameter tuning\n# grid_search = GridSearchCV(estimator=xgb_clf, param_grid=param_grid, \n#                            cv=3, n_jobs=-1, verbose=2, scoring='accuracy')\n\n# # Step 6: Fit the model with the best parameters found from GridSearch\n# grid_search.fit(X_train, y_train)\n\n# # Step 7: Get the best parameters and make predictions\n# best_params = grid_search.best_params_\n# best_xgb_clf = grid_search.best_estimator_\n\n# # Step 8: Make predictions on the test set using the best estimator\n# y_pred = best_xgb_clf.predict(X_test)\n\n# # Step 9: Convert predictions to integer (if needed)\n# y_pred = np.round(y_pred).astype(int)\n\n# # Step 10: Evaluate the model\n# accuracy = accuracy_score(y_test, y_pred)\n# report = classification_report(y_test, y_pred, zero_division=0)  # Add zero_division parameter\n\n# # Calculate the Quadratic Weighted Kappa (QWK)\n# unique_classes = len(np.unique(y))  # Get the number of unique classes in the target\n# qwk_score = quadratic_weighted_kappa(y_test, y_pred, unique_classes)\n\n# # Output results\n# print(f'Best Parameters: {best_params}')\n# print(f'Accuracy: {accuracy}')\n# print('Classification Report:')\n# print(report)\n# print(f'Quadratic Weighted Kappa (QWK): {qwk_score}')\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:13.987282Z","iopub.execute_input":"2024-10-14T15:18:13.987788Z","iopub.status.idle":"2024-10-14T15:18:14.003113Z","shell.execute_reply.started":"2024-10-14T15:18:13.987734Z","shell.execute_reply":"2024-10-14T15:18:14.001886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n# from sklearn.model_selection import train_test_split\n# from xgboost import XGBRegressor\n# from sklearn.metrics import mean_squared_error, r2_score\n\n# # Assuming df is your DataFrame and 'sii' is the target column\n# # Step 1: Define the feature set (X) and target variable (y)\n# X = train_df.drop(columns=['sii'])  # Drop the target column from features\n# y = train_df['sii']                 # Target column is 'sii'\n\n# # Step 2: Split the data into training and testing sets\n# X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# # Step 3: Initialize the XGBoost Regressor\n# xgb_reg = XGBRegressor(objective='reg:squarederror', n_estimators=100, learning_rate=0.1)\n\n# # Step 4: Train the model on the training data\n# xgb_reg.fit(X_train, y_train)\n\n# # Step 5: Make predictions on the test data\n# y_pred = xgb_reg.predict(X_test)\n\n# # Step 6: Evaluate the model\n# mse = mean_squared_error(y_test, y_pred)\n# r2 = r2_score(y_test, y_pred)\n\n# print(f'Mean Squared Error: {mse}')\n# print(f'R^2 Score: {r2}')\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.004797Z","iopub.execute_input":"2024-10-14T15:18:14.005303Z","iopub.status.idle":"2024-10-14T15:18:14.019287Z","shell.execute_reply.started":"2024-10-14T15:18:14.005249Z","shell.execute_reply":"2024-10-14T15:18:14.017906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import GridSearchCV\n\n# # Define hyperparameters to tune\n# param_grid = {\n#     'n_estimators': [100, 200, 300],\n#     'learning_rate': [0.01, 0.05, 0.1],\n#     'max_depth': [3, 5, 7],\n#     'subsample': [0.7, 0.8, 1.0],\n#     'colsample_bytree': [0.7, 0.8, 1.0]\n# }\n\n# # Initialize the XGBoost Regressor\n# xgb_reg = XGBRegressor(objective='reg:squarederror')\n\n# # Use GridSearchCV to search for the best parameters\n# grid_search = GridSearchCV(estimator=xgb_reg, param_grid=param_grid, cv=3, scoring='r2', verbose=1)\n\n# # Train the model with hyperparameter tuning\n# grid_search.fit(X_train, y_train)\n\n# # Get the best parameters\n# print(\"Best Parameters:\", grid_search.best_params_)\n\n# # Make predictions using the best model\n# y_pred_best = grid_search.best_estimator_.predict(X_test)\n\n# # Evaluate the tuned model\n# mse_best = mean_squared_error(y_test, y_pred_best)\n# r2_best = r2_score(y_test, y_pred_best)\n\n# print(f'Mean Squared Error (Tuned): {mse_best}')\n# print(f'R^2 Score (Tuned): {r2_best}')\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.020809Z","iopub.execute_input":"2024-10-14T15:18:14.021236Z","iopub.status.idle":"2024-10-14T15:18:14.036394Z","shell.execute_reply.started":"2024-10-14T15:18:14.021193Z","shell.execute_reply":"2024-10-14T15:18:14.035155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Plot feature importance\n# import matplotlib.pyplot as plt\n# import xgboost as xgb\n\n# # Plot feature importance\n# xgb.plot_importance(grid_search.best_estimator_)\n# plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.038010Z","iopub.execute_input":"2024-10-14T15:18:14.038526Z","iopub.status.idle":"2024-10-14T15:18:14.051016Z","shell.execute_reply.started":"2024-10-14T15:18:14.038467Z","shell.execute_reply":"2024-10-14T15:18:14.049834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import RandomizedSearchCV\n\n# # Define the parameter grid for RandomizedSearch\n# param_dist = {\n#     'n_estimators': [50, 100, 200, 300],\n#     'learning_rate': [0.01, 0.05, 0.1, 0.2],\n#     'max_depth': [3, 5, 7, 9],\n#     'subsample': [0.6, 0.7, 0.8, 1.0],\n#     'colsample_bytree': [0.6, 0.7, 0.8, 1.0],\n#     'min_child_weight': [1, 3, 5],\n#     'gamma': [0, 0.1, 0.2],\n#     'reg_alpha': [0, 0.1, 1],\n#     'reg_lambda': [0, 1, 10]\n# }\n\n# # Use RandomizedSearchCV to find the best parameters\n# random_search = RandomizedSearchCV(estimator=xgb_reg, param_distributions=param_dist, n_iter=100, cv=3, verbose=1, random_state=42, scoring='r2')\n\n# # Fit the model\n# random_search.fit(X_train, y_train)\n\n# # Best parameters\n# print(\"Best Parameters (RandomizedSearch):\", random_search.best_params_)\n\n# # Predict and evaluate\n# y_pred_random = random_search.best_estimator_.predict(X_test)\n# mse_random = mean_squared_error(y_test, y_pred_random)\n# r2_random = r2_score(y_test, y_pred_random)\n\n# print(f'Mean Squared Error (Randomized): {mse_random}')\n# print(f'R^2 Score (Randomized): {r2_random}')\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.052814Z","iopub.execute_input":"2024-10-14T15:18:14.053625Z","iopub.status.idle":"2024-10-14T15:18:14.064246Z","shell.execute_reply.started":"2024-10-14T15:18:14.053568Z","shell.execute_reply":"2024-10-14T15:18:14.063044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Function to transform with handling unseen labels\n# def safe_transform(column, le):\n#     # Map unseen labels to -1\n#     return column.apply(lambda x: le.transform([x])[0] if x in le.classes_ else -1)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.065809Z","iopub.execute_input":"2024-10-14T15:18:14.066214Z","iopub.status.idle":"2024-10-14T15:18:14.081982Z","shell.execute_reply.started":"2024-10-14T15:18:14.066173Z","shell.execute_reply":"2024-10-14T15:18:14.080721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Apply the same label encoding to the test data\n# print(\"Applying label encoding to test data...\")\n# for column, le in label_encoders.items():\n#     if column in test_df.columns:\n#         test_df[column] = safe_transform(test_df[column].astype(str), le)\n#         print(f\"Transformed column: {column}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.083465Z","iopub.execute_input":"2024-10-14T15:18:14.083825Z","iopub.status.idle":"2024-10-14T15:18:14.095023Z","shell.execute_reply.started":"2024-10-14T15:18:14.083788Z","shell.execute_reply":"2024-10-14T15:18:14.093793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_percentage = test_df.isnull().mean() * 100\n\n# Print the percentage of missing values for each column\nprint(missing_percentage)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.096370Z","iopub.execute_input":"2024-10-14T15:18:14.096809Z","iopub.status.idle":"2024-10-14T15:18:14.116490Z","shell.execute_reply.started":"2024-10-14T15:18:14.096768Z","shell.execute_reply":"2024-10-14T15:18:14.115062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Assuming you have your training data in 'X' and your test data in 'test_df'\n\n# Identify categorical columns in the training dataset\ncategorical_cols = X.select_dtypes(include=['object']).columns.tolist()\n\n# Perform One-Hot Encoding on the training dataset\nX_encoded = pd.get_dummies(X, columns=categorical_cols, drop_first=True)\n\n# Perform One-Hot Encoding on the test dataset\ntest_encoded = pd.get_dummies(test_df, columns=categorical_cols, drop_first=True)\n\n# Align the columns of the test dataset with the training dataset\n# This will ensure both datasets have the same columns by adding missing columns with zeroes\ntest_encoded = test_encoded.reindex(columns=X_encoded.columns, fill_value=0)\n\n# Now you have X_encoded for training and test_encoded for testing with aligned columns\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.117887Z","iopub.execute_input":"2024-10-14T15:18:14.118323Z","iopub.status.idle":"2024-10-14T15:18:14.147872Z","shell.execute_reply.started":"2024-10-14T15:18:14.118280Z","shell.execute_reply":"2024-10-14T15:18:14.146754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Predict using the tuned model\n# predictions_rf = best_rf_model.predict(test_encoded)\n# # If you want to see the predicted classes\n# print(\"Predicted classes for the test set:\", predictions_rf)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.149364Z","iopub.execute_input":"2024-10-14T15:18:14.149752Z","iopub.status.idle":"2024-10-14T15:18:14.177111Z","shell.execute_reply.started":"2024-10-14T15:18:14.149712Z","shell.execute_reply":"2024-10-14T15:18:14.175933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Predict using the tuned model\n# predictions_xgbr = best_xgbr_model.predict(test_encoded)\n# # If you want to see the predicted classes\n# print(\"Predicted classes for the test set:\", predictions_xgbr)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.178678Z","iopub.execute_input":"2024-10-14T15:18:14.179054Z","iopub.status.idle":"2024-10-14T15:18:14.195475Z","shell.execute_reply.started":"2024-10-14T15:18:14.179013Z","shell.execute_reply":"2024-10-14T15:18:14.194376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict using the tuned model\npredictions_lgbm = best_lgb_model.predict(test_encoded)\n# If you want to see the predicted classes\nprint(\"Predicted classes for the test set:\", predictions_lgbm)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Make predictions on the test set using the trained model\n# predictions = lgb_model2.predict(test_encoded)\n\n# # Get the class with the highest probability\n# predictions_lgbm = predictions.argmax(axis=1)\n\n# # If you want to see the predicted classes\n# print(\"Predicted classes for the test set:\", predictions_lgbm)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.203875Z","iopub.execute_input":"2024-10-14T15:18:14.204341Z","iopub.status.idle":"2024-10-14T15:18:14.213305Z","shell.execute_reply.started":"2024-10-14T15:18:14.204297Z","shell.execute_reply":"2024-10-14T15:18:14.212057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_test_scaled = scaler.transform(test_df)  # Apply the same scaling to test data","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.215061Z","iopub.execute_input":"2024-10-14T15:18:14.215488Z","iopub.status.idle":"2024-10-14T15:18:14.224287Z","shell.execute_reply.started":"2024-10-14T15:18:14.215445Z","shell.execute_reply":"2024-10-14T15:18:14.223061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Predict using the tuned XGBoost model\n# predictions = best_model.predict(X_test_scaled)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.225878Z","iopub.execute_input":"2024-10-14T15:18:14.226309Z","iopub.status.idle":"2024-10-14T15:18:14.237056Z","shell.execute_reply.started":"2024-10-14T15:18:14.226266Z","shell.execute_reply":"2024-10-14T15:18:14.235661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import numpy as np\n# # Convert the values based on the specified conditions\n# def convert_values(arr):\n#     converted = np.zeros(arr.shape, dtype=int)  # Default to 0\n#     converted[(arr >= 30) & (arr < 50)] = 1\n#     converted[(arr >= 50) & (arr < 80)] = 2\n#     converted[(arr >= 80) & (arr <= 100)] = 3\n#     return converted\n\n# predictions = convert_values(predictions)","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.241579Z","iopub.execute_input":"2024-10-14T15:18:14.242059Z","iopub.status.idle":"2024-10-14T15:18:14.251216Z","shell.execute_reply.started":"2024-10-14T15:18:14.241997Z","shell.execute_reply":"2024-10-14T15:18:14.250061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import numpy as np\n# import pandas as pd\n\n# # Assuming you already have the predictions from the three models\n# # Random Forest Predictions\n# predictions_rf = best_rf_model.predict(test_encoded).astype(int)\n\n# # XGBoost Predictions\n# predictions_xgbr = best_xgb_model.predict(test_encoded).astype(int)\n\n# # LightGBM Predictions\n# predictions_lgbm = lgb_model2.predict(test_encoded).argmax(axis=1)\n\n# # Stack the predictions from all models\n# predictions_stack = np.vstack([predictions_rf, predictions_xgbr, predictions_lgbm])\n\n# # Calculate the majority vote across the three models\n# majority_vote_predictions = np.apply_along_axis(lambda x: np.bincount(x).argmax(), axis=0, arr=predictions_stack)\n\n# # If 'id' column is part of your test set or dataset, adjust accordingly\n# submission_df = pd.DataFrame({\n#     'id': test_ids,  # Replace with your actual test IDs if available\n#     'sii': majority_vote_predictions\n# })\n\n# # Save the submission as a CSV file\n# submission_df.to_csv('submission.csv', index=False)\n\n# print(\"Submission file created: submission.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:34:34.993776Z","iopub.execute_input":"2024-10-14T15:34:34.994287Z","iopub.status.idle":"2024-10-14T15:34:35.036531Z","shell.execute_reply.started":"2024-10-14T15:34:34.994242Z","shell.execute_reply":"2024-10-14T15:34:35.035416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission_df","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:34:37.838093Z","iopub.execute_input":"2024-10-14T15:34:37.839581Z","iopub.status.idle":"2024-10-14T15:34:37.854027Z","shell.execute_reply.started":"2024-10-14T15:34:37.839527Z","shell.execute_reply":"2024-10-14T15:34:37.852876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Create a submission DataFrame with the test 'id' and predictions for 'sii'\n# submission_df_rf = pd.DataFrame({\n#     'id': test_ids,   # Use saved id column\n#     'sii': predictions_rf  # Predictions for the target variable\n# })","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.252835Z","iopub.execute_input":"2024-10-14T15:18:14.253364Z","iopub.status.idle":"2024-10-14T15:18:14.269718Z","shell.execute_reply.started":"2024-10-14T15:18:14.253321Z","shell.execute_reply":"2024-10-14T15:18:14.268048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Create a submission DataFrame with the test 'id' and predictions for 'sii'\n# submission_df_xgbr = pd.DataFrame({\n#     'id': test_ids,   # Use saved id column\n#     'sii': predictions_xgbr  # Predictions for the target variable\n# })","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.272640Z","iopub.execute_input":"2024-10-14T15:18:14.273128Z","iopub.status.idle":"2024-10-14T15:18:14.283422Z","shell.execute_reply.started":"2024-10-14T15:18:14.273084Z","shell.execute_reply":"2024-10-14T15:18:14.282156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Create a submission DataFrame with the test 'id' and predictions for 'sii'\n# submission_df_lgbm = pd.DataFrame({\n#     'id': test_ids,   # Use saved id column\n#     'sii': predictions_lgbm  # Predictions for the target variable\n# })","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.284859Z","iopub.execute_input":"2024-10-14T15:18:14.285379Z","iopub.status.idle":"2024-10-14T15:18:14.295837Z","shell.execute_reply.started":"2024-10-14T15:18:14.285321Z","shell.execute_reply":"2024-10-14T15:18:14.294332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a submission DataFrame with the test 'id' and predictions for 'sii'\nsubmission_df = pd.DataFrame({\n    'id': test_ids,   # Use saved id column\n    'sii': predictions_lgbm  # Predictions for the target variable\n})\n\n# Save the predictions to a CSV file\nsubmission_df.to_csv('submission.csv', index=False)\n\nprint(\"Predictions saved to submission.csv!\")","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.297996Z","iopub.execute_input":"2024-10-14T15:18:14.298624Z","iopub.status.idle":"2024-10-14T15:18:14.308827Z","shell.execute_reply.started":"2024-10-14T15:18:14.298572Z","shell.execute_reply":"2024-10-14T15:18:14.307553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2024-10-14T15:18:14.310586Z","iopub.execute_input":"2024-10-14T15:18:14.311002Z","iopub.status.idle":"2024-10-14T15:18:14.320051Z","shell.execute_reply.started":"2024-10-14T15:18:14.310944Z","shell.execute_reply":"2024-10-14T15:18:14.318775Z"},"trusted":true},"execution_count":null,"outputs":[]}]}