{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"background-color: #FF5733; padding: 20px; border: 1px solid #FF5733; border-radius: 10px; box-shadow: 0 0 10px rgba(0, 0, 0, 0.1); text-align: center;\">  \n  <h2 style=\"color: #ffffff;\">CMI-PIU</h2>  \n  <h2 style=\"color: #ffffff;\">Child Mind Institute — Problematic Internet Use</h2>  \n</div>\n","metadata":{}},{"cell_type":"markdown","source":"<span style=\"color:#6B33FF ; font-size: 24px;\">Importing required libraries...</span>\n\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report\nfrom sklearn.ensemble import RandomForestClassifier\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.utils.data as data\nimport torchvision.transforms as transforms\n\nimport tensorflow as tf\nfrom tensorflow import keras\n\nimport xgboost as xgb\nimport lightgbm as lgb\nimport catboost as cat\n","metadata":{"execution":{"iopub.status.busy":"2024-10-15T14:20:46.994327Z","iopub.execute_input":"2024-10-15T14:20:46.994742Z","iopub.status.idle":"2024-10-15T14:20:47.001461Z","shell.execute_reply.started":"2024-10-15T14:20:46.994693Z","shell.execute_reply":"2024-10-15T14:20:47.000482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!DOCTYPE html>\n<html lang=\"en\">\n<head>\n    <meta charset=\"UTF-8\">\n    <meta name=\"viewport\" content=\"width=device-width, initial-scale=1.0\">\n    <title>Styled Task Output</title>\n</head>\n<body>\n    <div style=\"background-color: #FFDD44; \n                border: 5px solid black; \n                border-radius: 10px; \n                padding: 20px; \n                text-align: center; \n                max-width: 600px; \n                margin: 50px auto; \n                box-shadow: 0 4px 8px rgba(0, 0, 0, 0.2); \n                font-family: Arial, sans-serif; \n                font-size: 18px;\">\n        <span style=\"color: #FF0000; font-size: 36px;\">Understanding the task</span>\n    </div>\n</body>\n</html>\n","metadata":{}},{"cell_type":"markdown","source":"The aim of this competition is to predict the **Severity Impairment Index (sii)**, which measures the level of problematic internet use among children and adolescents, based on physical activity data and other features.\n<br><br>\n> sii is derived from PCIAT-PCIAT_Total, the sum of scores from the Parent-Child Internet Addiction Test (PCIAT: 20 questions, scored 0-5).\n\n<br><br>\nTarget Variable (sii) is defined as:\n<br><br>\n> 0: None (PCIAT-PCIAT_Total from 0 to 30)<br>\n> 1: Mild (PCIAT-PCIAT_Total from 31 to 49)<br>\n> 2: Moderate (PCIAT-PCIAT_Total from 50 to 79)<br>\n> 3: Severe (PCIAT-PCIAT_Total 80 and more)<br>\n\nThis makes sii an ordinal categorical variable with four levels, where the order of categories is meaningful.<br>\n<br>\nType of Machine Learning Problem we can use with sii as a target:<br><br>\n\n1. Ordinal classification (ordinal logistic regression, models with custom ordinal loss functions)<br>\n2. Multiclass classification (treat sii as a nominal categorical variable without considering the order)<br>\n3. Regression (ignore the discrete nature of categories and treat sii as a continuous variable, then round prediction)<br>\n4. Custom (e.g. loss functions that penalize errors based on the distance between categories)<br>\n5. We can also use PCIAT-PCIAT_Total as a continuous target variable, and implement regression on PCIAT-PCIAT_Total and then map predictions to sii categories.<br><br>\n<br>\nFinally, another strategy involves predicting responses to each question of the Parent-Child Internet Addiction Test: i.e. pedict individual question scores as separate targets, sum the predicted scores to get the PCIAT-PCIAT_Total and map predictions to the corresponding sii category.<br><br>\n\n<span style=\"color:#B733FF ; font-size: 22px;\">let's make some exploratory data analysis:</span>\n","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndata_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\nsubmission_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\nwarnings.filterwarnings('ignore', category=FutureWarning)\n\npd.set_option('display.max_columns', None)\ndisplay(train.head(2))\nprint(f\"Train shape: {train.shape}\") ","metadata":{"execution":{"iopub.status.busy":"2024-10-15T14:20:47.029431Z","iopub.execute_input":"2024-10-15T14:20:47.029743Z","iopub.status.idle":"2024-10-15T14:20:47.133926Z","shell.execute_reply.started":"2024-10-15T14:20:47.029711Z","shell.execute_reply":"2024-10-15T14:20:47.132967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)\ndisplay(test.head(2))\nprint(f\"Test shape: {test.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-15T14:20:47.135624Z","iopub.execute_input":"2024-10-15T14:20:47.136126Z","iopub.status.idle":"2024-10-15T14:20:47.180392Z","shell.execute_reply.started":"2024-10-15T14:20:47.136085Z","shell.execute_reply":"2024-10-15T14:20:47.179550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**as per submission requirement** \n<br><br>\ndropping the pciat-season columns","metadata":{}},{"cell_type":"code","source":"train.drop(['PCIAT-Season', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02',\n       'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06',\n       'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10',\n       'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14',\n       'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18',\n       'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20', 'PCIAT-PCIAT_Total'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T14:20:47.181294Z","iopub.execute_input":"2024-10-15T14:20:47.181553Z","iopub.status.idle":"2024-10-15T14:20:47.188085Z","shell.execute_reply.started":"2024-10-15T14:20:47.181524Z","shell.execute_reply":"2024-10-15T14:20:47.187133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isna().sum()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-10-15T14:20:47.190167Z","iopub.execute_input":"2024-10-15T14:20:47.190454Z","iopub.status.idle":"2024-10-15T14:20:47.202659Z","shell.execute_reply.started":"2024-10-15T14:20:47.190424Z","shell.execute_reply":"2024-10-15T14:20:47.201736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"styled_df = train.describe().style.background_gradient(cmap='viridis')\nstyled_df","metadata":{"execution":{"iopub.status.busy":"2024-10-15T14:20:47.204087Z","iopub.execute_input":"2024-10-15T14:20:47.204467Z","iopub.status.idle":"2024-10-15T14:20:47.365479Z","shell.execute_reply.started":"2024-10-15T14:20:47.204423Z","shell.execute_reply":"2024-10-15T14:20:47.364657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1 = train.copy()\ntrain1.drop(['id'], axis=1, inplace=True)\ncategorical_columns = [column for column in train1.columns if train1[column].dtype == 'object']\nnum_cols = 2  \nnum_rows = (len(categorical_columns) + num_cols - 1) // num_cols  \nfig, axes = plt.subplots(num_rows, num_cols, figsize=(15, num_rows * 4))  \naxes = axes.flatten()\nfor i, column in enumerate(categorical_columns):\n    sns.countplot(y=train1[column], order=train1[column].value_counts().index, palette='viridis', ax=axes[i])\n    axes[i].set_title(f'Value Counts of {column}')\n    axes[i].set_xlabel('Count')\nfor j in range(i + 1, len(axes)):\n    fig.delaxes(axes[j])\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-15T14:20:47.366790Z","iopub.execute_input":"2024-10-15T14:20:47.367338Z","iopub.status.idle":"2024-10-15T14:20:49.665488Z","shell.execute_reply.started":"2024-10-15T14:20:47.367297Z","shell.execute_reply":"2024-10-15T14:20:49.664520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color: #2a8000; font-size: 36px;\">making preprocessing class for Data:</span>\n","metadata":{}},{"cell_type":"markdown","source":"<div style=\"background-color: #ff69b4; padding: 20px; border-radius: 10px; box-shadow: 0 0 10px rgba(0, 0, 0, 0.2); width: 800px; height: 70px; text-align: center; font-family: Arial, sans-serif; font-size: 36px; margin: 0 auto;\">  \n  <p>Multiple Imputation by Chained Equations (MICE)</p>  \n</div>\n","metadata":{}},{"cell_type":"markdown","source":"**implementing in target column as there are almost 1/3 missing value**","metadata":{}},{"cell_type":"markdown","source":"Multiple Imputation by Chained Equations (MICE) is an advanced technique that handles missing data by generating multiple imputations (sets of plausible values) rather than a single imputation. This approach is particularly useful when uncertainty about the missing values should be preserved.\n\n**How MICE Works:**\n1. Multiple Imputation:\n\nMICE creates several different imputed datasets by filling in missing values multiple times based on the other observed data.\nIt uses a regression model to predict the missing values, where the observed features are used to predict the missing values in one variable at a time, cycling through the dataset multiple times.<br>\n2. Chained Equations:\n\nMICE iterates through each variable with missing values, using different sets of predictor variables for each.\nFor the target column, the algorithm would use the non-null part of the column and other features in the dataset to predict the missing values.<br>\n3. Pooling Results:\n\nAfter the imputation process, you will have multiple datasets. The results from analyses (like model predictions) on these datasets are combined to give a final result that reflects the uncertainty of the missing data.<br>\nRead Article:  https://www.ncbi.nlm.nih.gov/pmc/articles/PMC3074241/","metadata":{}},{"cell_type":"code","source":"X = train1.drop(columns=['sii'],axis=1)  \ny = train1['sii']\n\n# difining different type of column names\ncategorical_columns = X.select_dtypes(include=['object']).columns.tolist()\nbinary_columns = [col for col in X.columns if X[col].nunique() == 2]\nordinal_columns = [col for col in X.columns if X[col].nunique() >= 3 and X[col].nunique() <= 20 and \n       (X[col].dtype in ['int64', 'float64']) ]\nall_columns = X.columns.tolist()\ncontinuous_columns = [col for col in all_columns if col not in binary_columns + categorical_columns + ordinal_columns]","metadata":{"execution":{"iopub.status.busy":"2024-10-15T14:20:49.667048Z","iopub.execute_input":"2024-10-15T14:20:49.667342Z","iopub.status.idle":"2024-10-15T14:20:49.707088Z","shell.execute_reply.started":"2024-10-15T14:20:49.667310Z","shell.execute_reply":"2024-10-15T14:20:49.706342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color: #2a8000; font-size: 36px;\">Data Preprocessing...</span>","metadata":{}},{"cell_type":"code","source":"from sklearn.base import TransformerMixin, BaseEstimator\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler, OrdinalEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.experimental import enable_iterative_imputer  # Enables IterativeImputer\nfrom sklearn.impute import IterativeImputer\n\nclass CustomPreprocessor_MICE(BaseEstimator, TransformerMixin):\n    def __init__(self, continuous_columns, binary_columns, ordinal_columns, categorical_columns):\n        self.continuous_columns = continuous_columns\n        self.binary_columns = binary_columns\n        self.ordinal_columns = ordinal_columns\n        self.categorical_columns = categorical_columns\n        \n        self.one_hot_encoder = OneHotEncoder(sparse=False, handle_unknown='ignore')\n        self.scaler = StandardScaler()\n        self.iterative_imputer = IterativeImputer()\n\n    def fit(self, X, y=None):\n        # Fit the one-hot encoder on categorical columns\n        self.one_hot_encoder.fit(X[self.categorical_columns])\n        \n        # Fit the scaler only on continuous columns\n        self.scaler.fit(X[self.continuous_columns])\n        \n        return self\n\n    def transform(self, X, y=None):\n        # Copy the data\n        X_transformed = X.copy()\n\n        # Handle outliers in different types of columns\n        self.handle_outliers(X_transformed)\n\n        # Fill missing values\n        self.fill_missing_values(X_transformed)\n\n        # Handle encoding for categorical columns\n        X_transformed = self.handle_encoding(X_transformed)\n\n        # Scale the continuous columns\n        X_transformed[self.continuous_columns] = self.scaler.transform(X_transformed[self.continuous_columns])\n\n        return X_transformed\n\n    def handle_outliers(self, X):\n        # Detect and handle outliers in continuous columns (example using z-score)\n        for col in self.continuous_columns:\n            z_scores = np.abs((X[col] - X[col].mean()) / X[col].std())\n            outliers = z_scores > 2.7\n            # Handling outliers by capping them at 3 standard deviations\n            X.loc[outliers, col] = X[col].mean()\n\n        # Handle outliers in ordinal numeric columns (example logic)\n        for col in self.ordinal_columns:\n            lower_bound, upper_bound = X[col].quantile([0.01, 0.95])\n            X[col] = np.clip(X[col], lower_bound, upper_bound)\n\n    def fill_missing_values(self, X):\n        # Fill missing values for continuous columns with median\n        for col in self.continuous_columns:\n            X[col].fillna(X[col].median(), inplace=True)\n\n        # Fill missing values for ordinal columns with median\n        for col in self.ordinal_columns:\n            X[col].fillna(X[col].median(), inplace=True)\n\n        # Fill missing values for binary columns with mode (most common value)\n        for col in self.binary_columns:\n            X[col].fillna(X[col].mode()[0], inplace=True)\n\n        # Handle missing values in categorical columns\n        for col in self.categorical_columns:\n            X[col].fillna('Missing', inplace=True)\n\n    def handle_encoding(self, X):\n        # One-hot encoding for categorical columns\n        categorical_encoded = pd.DataFrame(self.one_hot_encoder.transform(X[self.categorical_columns]),\n                                           columns=self.one_hot_encoder.get_feature_names_out(self.categorical_columns),\n                                           index=X.index)\n\n        # Drop the original categorical columns and concatenate the encoded ones\n        X = X.drop(columns=self.categorical_columns)\n        X = pd.concat([X, categorical_encoded], axis=1)\n\n        return X\n    \n    def impute_y_based_on_X(self, X, y):\n        \"\"\"\n        Impute missing values in target column y based on the transformed X features.\n        Ensure y only contains the specified categories: 0, 1, 2, 3.\n        \"\"\"\n        # Filter y to keep only valid classes\n        valid_classes = [0, 1, 2, 3]\n        y_filtered = y[y.isin(valid_classes)]\n\n        # Combine filtered y with X\n        combined_data = pd.concat([X, y_filtered], axis=1)\n\n        # Impute missing values for the target column\n        combined_data_imputed = pd.DataFrame(self.iterative_imputer.fit_transform(combined_data),\n                                             columns=combined_data.columns)\n        \n        # Extract the imputed target column\n        y_imputed = combined_data_imputed[y.name]\n\n        # Map any other values to the nearest valid category if needed\n        # For example, if you want to convert float values to integers, you could do:\n        y_imputed = y_imputed.round().astype(int).clip(lower=0, upper=3)\n\n        return y_imputed\n\n    def split_and_transform(self, df, target_column, test_size=0.2, random_state=42):\n        \"\"\"\n        Perform preprocessing, impute missing values in y based on X,\n        then split into train-test sets and return X_train, X_test, y_train, y_test.\n        \"\"\"\n        # Split df into X (features) and y (target)\n        X = df.drop(columns=[target_column])\n        y = df[target_column]\n\n        # **Fit** the preprocessor before transforming\n        self.fit(X)\n\n        # Preprocess X\n        X_transformed = self.transform(X)\n\n        # Impute missing values in y based on X\n        y_imputed = self.impute_y_based_on_X(X_transformed, y)\n\n        # Ensure y_imputed is categorical and only contains 0, 1, 2, 3\n        y_imputed = y_imputed.astype('category')\n\n        # Split into train and test sets\n        X_train, X_test, y_train, y_test = train_test_split(X_transformed, y_imputed, test_size=test_size, random_state=random_state)\n\n        return X_train, X_test, y_train, y_test\n","metadata":{"execution":{"iopub.status.busy":"2024-10-15T14:20:49.710659Z","iopub.execute_input":"2024-10-15T14:20:49.711204Z","iopub.status.idle":"2024-10-15T14:20:49.733917Z","shell.execute_reply.started":"2024-10-15T14:20:49.711154Z","shell.execute_reply":"2024-10-15T14:20:49.733052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<span style=\"color: #2a8000; font-size: 36px;\">Finding the best parameters...</span>","metadata":{}},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.ensemble import RandomForestClassifier  \nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\n\n# Assuming you have your DataFrame 'df' and target_column defined\ntarget_column = 'sii'\n\n# Create an instance of your custom preprocessor\npreprocessor = CustomPreprocessor_MICE(\n    continuous_columns,  \n    binary_columns,         \n    ordinal_columns,        \n    categorical_columns  \n)\n\n# Split your data\n# Split and transform your dataset\nX_train, X_test, y_train, y_test = preprocessor.split_and_transform(train1, target_column)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T14:20:49.734996Z","iopub.execute_input":"2024-10-15T14:20:49.735353Z","iopub.status.idle":"2024-10-15T14:21:23.768576Z","shell.execute_reply.started":"2024-10-15T14:20:49.735312Z","shell.execute_reply":"2024-10-15T14:21:23.767379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nimport xgboost as xgb\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import accuracy_score\n\n# Define the parameter grid for GridSearchCV\nparam_grid = {\n    'n_estimators': [200,300],\n    'max_depth': [5,6,7],\n    'learning_rate': [0.07, 0.1],\n    'subsample': [0.8, 1.0],\n    'tree_method': ['gpu_hist'],  # Use GPU for training\n}\n\n# Initialize the XGBoost model with GPU parameters\nxgb_model = xgb.XGBClassifier(use_label_encoder=False, eval_metric='mlogloss', tree_method='gpu_hist')\n\n# Set up GridSearchCV\ngrid_search = GridSearchCV(estimator=xgb_model, param_grid=param_grid, scoring='accuracy', cv=3, verbose=2, n_jobs=-1)\n\n# Fit GridSearchCV\ngrid_search.fit(X_train, y_train)\n\n# Print the best parameters and best score\nprint(f\"Best Parameters: {grid_search.best_params_}\")\nprint(f\"Best Score: {grid_search.best_score_}\")\n\n# Make predictions on the test set\ny_pred = grid_search.predict(X_test)\naccuracy = accuracy_score(y_test, y_pred)\nprint(f\"Test Accuracy: {accuracy}\")\n'''","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-10-15T14:21:23.770848Z","iopub.execute_input":"2024-10-15T14:21:23.772364Z","iopub.status.idle":"2024-10-15T14:21:23.788566Z","shell.execute_reply.started":"2024-10-15T14:21:23.772315Z","shell.execute_reply":"2024-10-15T14:21:23.787430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!DOCTYPE html>\r\n<html lang=\"en\">\r\n<head>\r\n    <meta charset=\"UTF-8\">\r\n    <meta name=\"viewport\" content=\"width=device-width, initial-scale=1.0\">\r\n</head>\r\n<body>\r\n    <div style=\"background-color: #FFDD44; \r\n                border: 5px solid black; \r\n                border-radius: 10px; \r\n                padding: 20px; \r\n                text-align: center; \r\n                max-width: 600px; \r\n                margin: 50px auto; \r\n                box-shadow: 0 4px 8px rgba(0, 0, 0, 0.2); \r\n                font-family: Arial, sans-serif; \r\n                font-size: 18px;\">\r\n        After running the above code<br>\r\n        Best Parameters: {'learning_rate': 0.1, 'max_depth': 7, 'n_estimators': 300, 'subsample': 0.8, 'tree_method': 'gpu_hist'}<br>\r\n        Best Score: 0.8554930360769597<br>\r\n        Test Accuracy: 0.699494949494949597\r\n    </div>\r\n</body>\r\n</html>\r\n","metadata":{}},{"cell_type":"markdown","source":"# Final Model building","metadata":{}},{"cell_type":"code","source":"test1 = test.drop(['id'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T14:21:23.792455Z","iopub.execute_input":"2024-10-15T14:21:23.794471Z","iopub.status.idle":"2024-10-15T14:21:23.807421Z","shell.execute_reply.started":"2024-10-15T14:21:23.794337Z","shell.execute_reply":"2024-10-15T14:21:23.805423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.base import TransformerMixin, BaseEstimator\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler\nfrom sklearn.experimental import enable_iterative_imputer\nfrom sklearn.impute import IterativeImputer\nimport pandas as pd\nimport numpy as np\n\nclass UnifiedPreprocessor(BaseEstimator, TransformerMixin):\n    def __init__(self, continuous_columns, binary_columns, ordinal_columns, categorical_columns):\n        # Column types\n        self.continuous_columns = continuous_columns\n        self.binary_columns = binary_columns\n        self.ordinal_columns = ordinal_columns\n        self.categorical_columns = categorical_columns\n\n        # Encoders and scalers\n        self.one_hot_encoder = OneHotEncoder(sparse=False, handle_unknown='ignore')\n        self.scaler = StandardScaler()\n        self.iterative_imputer = IterativeImputer()\n\n    def fit(self, X, y=None):\n        \"\"\"\n        Fit preprocessing steps on the training data (fit encoders, scalers, imputers).\n        \"\"\"\n        # Fit the one-hot encoder on categorical columns\n        self.one_hot_encoder.fit(X[self.categorical_columns])\n\n        # Fit the scaler on continuous columns\n        self.scaler.fit(X[self.continuous_columns])\n\n        # Fit the imputer on all columns (for continuous, binary, and ordinal)\n        self.iterative_imputer.fit(X[self.continuous_columns + self.binary_columns + self.ordinal_columns])\n\n        return self\n\n    def transform(self, X, is_train=True):\n        \"\"\"\n        Apply transformations to the dataset.\n        - For training: Fits imputers and encoders.\n        - For testing: Only applies transformations without fitting new ones.\n        \"\"\"\n        X_transformed = X.copy()\n\n        # Fill missing values\n        X_transformed = self.fill_missing_values(X_transformed, is_train)\n\n        # Encode categorical columns using pre-fitted encoder\n        if self.categorical_columns:\n            X_transformed = self.handle_encoding(X_transformed)\n\n        # Scale continuous columns using pre-fitted scaler\n        if self.continuous_columns:\n            X_transformed[self.continuous_columns] = self.scaler.transform(X_transformed[self.continuous_columns])\n\n        return X_transformed\n\n    def fill_missing_values(self, X, is_train):\n        \"\"\"\n        Fill missing values in continuous, binary, and ordinal columns.\n        \"\"\"\n        # Use the pre-fitted iterative imputer only for training data\n        if is_train:\n            X[self.continuous_columns + self.binary_columns + self.ordinal_columns] = self.iterative_imputer.transform(\n                X[self.continuous_columns + self.binary_columns + self.ordinal_columns])\n        else:\n            # For test data, apply the imputer without fitting again\n            for col in self.continuous_columns:\n                X[col].fillna(X[col].median(), inplace=True)\n            for col in self.ordinal_columns:\n                X[col].fillna(X[col].median(), inplace=True)\n            for col in self.binary_columns:\n                X[col].fillna(X[col].mode()[0], inplace=True)\n\n        # Handle missing categorical values\n        for col in self.categorical_columns:\n            X[col].fillna('Missing', inplace=True)\n\n        return X\n\n    def handle_encoding(self, X):\n        \"\"\"\n        Apply one-hot encoding to categorical columns using the pre-fitted encoder.\n        \"\"\"\n        categorical_encoded = pd.DataFrame(\n            self.one_hot_encoder.transform(X[self.categorical_columns]),\n            columns=self.one_hot_encoder.get_feature_names_out(self.categorical_columns),\n            index=X.index\n        )\n\n        # Drop original categorical columns and concatenate encoded ones\n        X = X.drop(columns=self.categorical_columns)\n        X = pd.concat([X, categorical_encoded], axis=1)\n\n        return X\n\n    def impute_y_based_on_X(self, X, y):\n        \"\"\"\n        Impute missing values in target column y based on the transformed X features.\n        Ensure y only contains the specified categories: 0, 1, 2, 3.\n        \"\"\"\n        # Filter y to keep only valid classes\n        valid_classes = [0, 1, 2, 3]\n        y_filtered = y[y.isin(valid_classes)]\n\n        # Combine filtered y with X\n        combined_data = pd.concat([X, y_filtered], axis=1)\n\n        # Impute missing values for the target column\n        combined_data_imputed = pd.DataFrame(self.iterative_imputer.fit_transform(combined_data),\n                                             columns=combined_data.columns)\n        \n        # Extract the imputed target column\n        y_imputed = combined_data_imputed[y.name]\n\n        # Map any other values to the nearest valid category if needed\n        y_imputed = y_imputed.round().astype(int).clip(lower=0, upper=3)\n\n        return y_imputed\n\n    def fit_transform(self, X, y=None):\n        \"\"\"\n        Fit the transformer and apply the transformations to the training data.\n        \"\"\"\n        self.fit(X)\n        return self.transform(X, is_train=True)\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-10-15T14:21:23.810765Z","iopub.execute_input":"2024-10-15T14:21:23.813104Z","iopub.status.idle":"2024-10-15T14:21:23.859755Z","shell.execute_reply.started":"2024-10-15T14:21:23.813030Z","shell.execute_reply":"2024-10-15T14:21:23.858720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessor = UnifiedPreprocessor(\n    continuous_columns=continuous_columns,\n    binary_columns=binary_columns,\n    ordinal_columns=ordinal_columns,\n    categorical_columns=categorical_columns\n)\n\n# Fit and transform the training data\nX_train_transformed = preprocessor.fit_transform(train1)\n# Transform test data (without refitting)\nX_test_transformed = preprocessor.transform(test1, is_train=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-15T14:21:23.861036Z","iopub.execute_input":"2024-10-15T14:21:23.861378Z","iopub.status.idle":"2024-10-15T14:21:41.954020Z","shell.execute_reply.started":"2024-10-15T14:21:23.861343Z","shell.execute_reply":"2024-10-15T14:21:41.953223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Concatenate X data\nX_full = pd.concat([X_train, X_test], axis=0)\n\n# Concatenate y data\ny_full = pd.concat([y_train, y_test], axis=0)\n\n# Reset index to ensure alignment if the index order has been shuffled\nX_full = X_full.reset_index(drop=True)\ny_full = y_full.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-15T14:21:41.955189Z","iopub.execute_input":"2024-10-15T14:21:41.955476Z","iopub.status.idle":"2024-10-15T14:21:41.966737Z","shell.execute_reply.started":"2024-10-15T14:21:41.955446Z","shell.execute_reply":"2024-10-15T14:21:41.965883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import xgboost as xgb\nfrom sklearn.metrics import accuracy_score\n\n# Initialize XGBoost classifier with the best hyperparameters\nclf = xgb.XGBClassifier(\n    learning_rate=0.1,\n    max_depth=7,\n    n_estimators=300,\n    subsample=0.8,\n    tree_method='gpu_hist',  # Use GPU for faster computation if available\n    use_label_encoder=False,  # To avoid warnings with newer XGBoost versions\n    eval_metric='logloss'     # Common choice for classification tasks\n)\n\n# Fit the model on the training data\nclf.fit(X_full, y_full)\n\n# Predict on the test data\ny_pred = clf.predict(X_test_transformed)\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-10-15T14:21:41.968033Z","iopub.execute_input":"2024-10-15T14:21:41.968321Z","iopub.status.idle":"2024-10-15T14:21:45.098529Z","shell.execute_reply.started":"2024-10-15T14:21:41.968288Z","shell.execute_reply":"2024-10-15T14:21:45.097523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n    'id': test['id'],\n    'sii': y_pred\n})\n\n# Optionally, save the submission to a CSV file\nsubmission.to_csv('submission.csv', index=False)\n\n# Display the first few rows of the submission\nprint(submission.sii.value_counts())","metadata":{"execution":{"iopub.status.busy":"2024-10-15T14:21:45.099997Z","iopub.execute_input":"2024-10-15T14:21:45.100338Z","iopub.status.idle":"2024-10-15T14:21:45.109137Z","shell.execute_reply.started":"2024-10-15T14:21:45.100303Z","shell.execute_reply":"2024-10-15T14:21:45.108162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"background-color: #FF5733; padding: 20px; border: 1px solid #FF5733; border-radius: 20px; box-shadow: 0 0 10px rgba(0, 0, 0, 0.1); text-align: center;\">  \n  <h2 style=\"color: #ffffff;\">...Thank you for observing my notebook...</h2>  \n  <h2 style=\"color: #ffffff;\">If You find anything helpful then please Upvote</h2>  \n</div>","metadata":{}}]}