{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:06:43.658257Z","iopub.execute_input":"2024-12-05T13:06:43.659129Z","iopub.status.idle":"2024-12-05T13:06:45.400372Z","shell.execute_reply.started":"2024-12-05T13:06:43.659087Z","shell.execute_reply":"2024-12-05T13:06:45.399068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import libraries and models\nimport torch\nimport numpy as np\nimport pandas as pd\nimport os\nimport re\nfrom sklearn.base import clone, BaseEstimator, RegressorMixin\nfrom sklearn.metrics import cohen_kappa_score, accuracy_score, mean_squared_error, mean_absolute_error, mean_absolute_percentage_error\nfrom sklearn.model_selection import StratifiedKFold, train_test_split, KFold\nfrom scipy.optimize import minimize\nfrom sklearn.decomposition import PCA\nfrom concurrent.futures import ThreadPoolExecutor\nimport random\nfrom tqdm import tqdm\nimport polars as pl\nimport polars.selectors as cs\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import MaxNLocator, FormatStrFormatter, PercentFormatter\nimport seaborn as sns\nfrom torch.utils.data import Dataset, DataLoader\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, RobustScaler\nimport matplotlib.pyplot as plt\nfrom keras.models import Model\nfrom keras.layers import Input, Dense\nfrom keras.optimizers import Adam\nimport torch.nn as nn\nimport torch.optim as optim\n\nfrom colorama import Fore, Style\nfrom IPython.display import clear_output\nimport warnings\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.pipeline import Pipeline\nfrom matplotlib.ticker import MaxNLocator\nimport seaborn as sns\n\nwarnings.filterwarnings('ignore')\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n# pd.options.display.max_columns = None\n\n# seed: to keep the value unchanged after each loop\n# n_splits: splits the data into 4 parts\nseed = 42\nn_splits = 4","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:06:45.402707Z","iopub.execute_input":"2024-12-05T13:06:45.403176Z","iopub.status.idle":"2024-12-05T13:06:45.415523Z","shell.execute_reply.started":"2024-12-05T13:06:45.403126Z","shell.execute_reply":"2024-12-05T13:06:45.414238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\n\n# Ensure data consistency across iterations\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\n\nseed_everything(2024)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:06:45.417276Z","iopub.execute_input":"2024-12-05T13:06:45.417681Z","iopub.status.idle":"2024-12-05T13:06:45.438035Z","shell.execute_reply.started":"2024-12-05T13:06:45.417644Z","shell.execute_reply":"2024-12-05T13:06:45.436575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_data():\n    train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n    test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n    sample =  pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n    return train, test, sample","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:06:45.441780Z","iopub.execute_input":"2024-12-05T13:06:45.442166Z","iopub.status.idle":"2024-12-05T13:06:45.456303Z","shell.execute_reply.started":"2024-12-05T13:06:45.442132Z","shell.execute_reply":"2024-12-05T13:06:45.455095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train, test, sample = load_data()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:06:45.457972Z","iopub.execute_input":"2024-12-05T13:06:45.458452Z","iopub.status.idle":"2024-12-05T13:06:45.513791Z","shell.execute_reply.started":"2024-12-05T13:06:45.458402Z","shell.execute_reply":"2024-12-05T13:06:45.511978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\n\n# Handle missing data\nimputer = KNNImputer(n_neighbors=5)\n\n# Numeric columns\nnumeric_cols = train.select_dtypes(include=['int32', 'int64', 'float32', 'float64']).columns\n\n# Select numeric columns from the train dataset and fill missing values\n# using mean, median, or the most frequent value\nimputed_data = imputer.fit_transform(train[numeric_cols])\ntrain_imputed = pd.DataFrame(imputed_data, columns=numeric_cols)\ntrain_imputed['sii'] = train_imputed['sii'].round().astype(int)\n\n# Add non-numeric columns back to the train dataset\nfor col in train.columns:\n    if col not in numeric_cols:\n        train_imputed[col] = train[col]\ntrain = train_imputed\n\n# Analyze and drop unnecessary columns\ndef feature_engineering(df: pd.DataFrame) -> pd.DataFrame:\n    # Drop columns related to 'season'\n    season_cols = [col for col in df.columns if 'season' in col.lower()]\n    df = df.drop(columns=season_cols, axis=1)\n    \n    # Create new feature columns\n    df['BMI_Age'] = df['Physical-BMI'] * df['Basic_Demos-Age']\n    df['Internet_Hours_Age'] = df['PreInt_EduHx-computerinternet_hoursday'] * df['Basic_Demos-Age']\n    df['BMI_Internet_Hours'] = df['Physical-BMI'] * df['PreInt_EduHx-computerinternet_hoursday']\n    df['BFP_BMI'] = df['BIA-BIA_Fat'] / df['BIA-BIA_BMI']\n    df['FFMI_BFP'] = df['BIA-BIA_FFMI'] / df['BIA-BIA_Fat']\n    df['FMI_BFP'] = df['BIA-BIA_FMI'] / df['BIA-BIA_Fat']\n    df['LST_TBW'] = df['BIA-BIA_LST'] / df['BIA-BIA_TBW']\n    df['BFP_BMR'] = df['BIA-BIA_Fat'] * df['BIA-BIA_BMR']\n    df['BFP_DEE'] = df['BIA-BIA_Fat'] * df['BIA-BIA_DEE']\n    df['BMR_Weight'] = df['BIA-BIA_BMR'] / df['Physical-Weight']\n    df['DEE_Weight'] = df['BIA-BIA_DEE'] / df['Physical-Weight']\n    df['SMM_Height'] = df['BIA-BIA_SMM'] / df['Physical-Height']\n    df['Muscle_to_Fat'] = df['BIA-BIA_SMM'] / df['BIA-BIA_FMI']\n    df['Hydration_Status'] = df['BIA-BIA_TBW'] / df['Physical-Weight']\n    df['ICW_TBW'] = df['BIA-BIA_ICW'] / df['BIA-BIA_TBW']\n    df['BMI_PHR'] = df['Physical-BMI'] * df['Physical-HeartRate']\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:06:45.515543Z","iopub.execute_input":"2024-12-05T13:06:45.516050Z","iopub.status.idle":"2024-12-05T13:06:53.388290Z","shell.execute_reply.started":"2024-12-05T13:06:45.515998Z","shell.execute_reply":"2024-12-05T13:06:53.387029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Run the function to remove unnecessary columns\ndef preprocess_tabular_data(df: pd.DataFrame, test: bool = False) -> pd.DataFrame:\n    df = feature_engineering(df)\n    df = df.dropna(thresh = 10)  # Drop rows with less than 10 non-null values\n    return df\n\n# Update the 2 files, train and test\ntrain = preprocess_tabular_data(train)\ntest = feature_engineering(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:06:53.389620Z","iopub.execute_input":"2024-12-05T13:06:53.390073Z","iopub.status.idle":"2024-12-05T13:06:53.420828Z","shell.execute_reply.started":"2024-12-05T13:06:53.390027Z","shell.execute_reply":"2024-12-05T13:06:53.419250Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.drop('id', axis = 1)\ntest.drop('id', axis = 1)\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:06:53.422256Z","iopub.execute_input":"2024-12-05T13:06:53.422676Z","iopub.status.idle":"2024-12-05T13:06:53.455022Z","shell.execute_reply.started":"2024-12-05T13:06:53.422594Z","shell.execute_reply":"2024-12-05T13:06:53.453706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Processing features based on columns in the train dataset\nfeaturesCols = ['Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-CGAS_Score', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total',\n                'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii', 'BMI_Age', 'Internet_Hours_Age', 'BMI_Internet_Hours',\n                'BFP_BMI', 'FFMI_BFP', 'FMI_BFP', 'LST_TBW', 'BFP_BMR', 'BFP_DEE', 'BMR_Weight', 'DEE_Weight',\n                'SMM_Height', 'Muscle_to_Fat', 'Hydration_Status', 'ICW_TBW', 'BMI_PHR']\n\n# Check the number of features\nlen(featuresCols)\n\n# Select relevant columns in the train dataset\ntrain = train[featuresCols]\n\n# Drop rows with missing values in the 'sii' column\ntrain = train.dropna(subset='sii')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:06:53.456664Z","iopub.execute_input":"2024-12-05T13:06:53.457128Z","iopub.status.idle":"2024-12-05T13:06:53.469758Z","shell.execute_reply.started":"2024-12-05T13:06:53.457078Z","shell.execute_reply":"2024-12-05T13:06:53.468556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Export the train dataset to check\nlen(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:06:53.474225Z","iopub.execute_input":"2024-12-05T13:06:53.474714Z","iopub.status.idle":"2024-12-05T13:06:53.490021Z","shell.execute_reply.started":"2024-12-05T13:06:53.474648Z","shell.execute_reply":"2024-12-05T13:06:53.488608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:06:53.491355Z","iopub.execute_input":"2024-12-05T13:06:53.492290Z","iopub.status.idle":"2024-12-05T13:06:53.522512Z","shell.execute_reply.started":"2024-12-05T13:06:53.492255Z","shell.execute_reply":"2024-12-05T13:06:53.521067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for infinite values only in numeric columns\ntrain_numeric = train.select_dtypes(include=[np.number])\n\n# Apply np.isinf to the numeric columns\nif np.any(np.isinf(train_numeric)):\n    # Replace infinities with NaN in the numeric columns\n    train_numeric = train_numeric.replace([np.inf, -np.inf], np.nan)\n\n    # Optionally, merge the cleaned numeric columns back into the original DataFrame\n    train[train_numeric.columns] = train_numeric","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:06:53.524194Z","iopub.execute_input":"2024-12-05T13:06:53.524650Z","iopub.status.idle":"2024-12-05T13:06:53.551957Z","shell.execute_reply.started":"2024-12-05T13:06:53.524612Z","shell.execute_reply":"2024-12-05T13:06:53.550834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to calculate the similarity (QWK) between two datasets\ndef QWK(y_true, y_pred): \n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')\n\n# Function to convert continuous values -> classification classes\ndef threshold_Rounder(oof_non_rounded, thresholds): \n    oof_non_rounded = np.array(oof_non_rounded)\n    \n    return np.where(oof_non_rounded < thresholds[0], 0,\n                    np.where(oof_non_rounded < thresholds[1], 1,\n                             np.where(oof_non_rounded < thresholds[2], 2, 3)))\n\n# Calculate the QWK value for predictions rounded based on thresholds\ndef evaluate_predictions(thresholds, y_true, oof_non_rounded):\n    rounded_p = threshold_Rounder(oof_non_rounded, thresholds)\n    return -QWK(y_true, rounded_p)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:06:53.553198Z","iopub.execute_input":"2024-12-05T13:06:53.553610Z","iopub.status.idle":"2024-12-05T13:06:53.560956Z","shell.execute_reply.started":"2024-12-05T13:06:53.553575Z","shell.execute_reply":"2024-12-05T13:06:53.559655Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = test.drop('id', axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:06:53.562429Z","iopub.execute_input":"2024-12-05T13:06:53.562817Z","iopub.status.idle":"2024-12-05T13:06:53.577783Z","shell.execute_reply.started":"2024-12-05T13:06:53.562783Z","shell.execute_reply":"2024-12-05T13:06:53.576570Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to train the model\ndef TrainML(model_class, test_data): \n    X = train.drop('sii', axis=1)\n    y = train['sii']\n    \n    SKF = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=seed)\n    \n    train_Score = []\n    test_Score = []\n    oof_non_rounded = np.zeros(len(y), dtype=float)\n    oof_rounded = np.zeros(len(y), dtype=int)\n    test_preds = np.zeros((len(test_data), n_splits))\n    \n    for fold, (train_idx, valid_idx) in enumerate(tqdm(SKF.split(X, y), desc='Training Folds', total=n_splits)):\n        X_train, X_valid = X.iloc[train_idx], X.iloc[valid_idx]\n        y_train, y_valid = y.iloc[train_idx], y.iloc[valid_idx]\n        \n        model = clone(model_class)\n        model.fit(X_train, y_train)\n        \n        y_train_pred = model.predict(X_train)\n        y_valid_pred = model.predict(X_valid)\n        \n        oof_non_rounded[valid_idx] = y_valid_pred\n        y_valid_pred_rounded = y_valid_pred.round(0).astype(int)\n        oof_rounded[valid_idx] = y_valid_pred_rounded\n        \n        train_kappa = QWK(y_train, y_train_pred.round(0).astype(int))\n        valid_kappa = QWK(y_valid, y_valid_pred_rounded)\n        \n        train_Score.append(train_kappa)\n        test_Score.append(valid_kappa)\n        \n        test_preds[:, fold] = model.predict(test_data)\n        print(f\"Fold {fold + 1} - Train Kappa: {train_kappa:.6f}, Valid Kappa: {valid_kappa:.6f}\")\n        clear_output(wait=True)\n    \n    print(f'Mean Train Kappa: {np.mean(train_Score):.6f}, Mean Valid Kappa: {np.mean(test_Score):.6f}')\n    print(f'Min Train Kappa: {np.min(train_Score):.6f}, Min Valid Kappa: {np.min(test_Score):.6f}')\n    print(f'Max Train Kappa: {np.max(train_Score):.6f}, Max Valid Kappa: {np.max(test_Score):.6f}')\n    \n    KappaOptimizer = minimize(evaluate_predictions, x0=[0.5, 1.5, 2.5], args=(y, oof_non_rounded), method='Nelder-Mead')\n    assert KappaOptimizer.success, KappaOptimizer.message\n    \n    oof_tuned = threshold_Rounder(oof_non_rounded, KappaOptimizer.x)\n    tKappa = QWK(y, oof_tuned)\n    print(f'Tuned Thresholds: {KappaOptimizer.x}, Optimized QWK Score: {tKappa:.6f}')\n    \n    tpm = test_preds.mean(axis=1)\n    tp_rounded = threshold_Rounder(tpm, KappaOptimizer.x)\n    \n    submission = pd.DataFrame({'id': sample['id'], 'sii': tp_rounded})\n    \n    return submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:06:53.579034Z","iopub.execute_input":"2024-12-05T13:06:53.579342Z","iopub.status.idle":"2024-12-05T13:06:53.596062Z","shell.execute_reply.started":"2024-12-05T13:06:53.579313Z","shell.execute_reply":"2024-12-05T13:06:53.594623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.base import clone\nfrom sklearn.ensemble import VotingRegressor\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\n\n# Example: Load or define your data\n# X_train could be a pandas DataFrame or NumPy array (features)\n# y_train could be a pandas Series or NumPy array (target)\nX_train = pd.DataFrame(np.random.rand(100, 10))  # 100 samples, 10 features\ny_train = pd.Series(np.random.rand(100))         # 100 target values\ntest_data = pd.DataFrame(np.random.rand(10, 10))  # 10 test samples, 10 features\n\n# Check if y_train is a 1D array\nprint(\"Shape of y_train:\", y_train.shape)  # Should print something like (100,)\n\n# Ensure y_train is 1D\ny_train = np.ravel(y_train)  # Flatten y_train if it's 2D\n\n# Set the parameters for the models\nLGBParams = {\n    'learning_rate': 0.066,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.666,\n    'bagging_fraction': 0.666,\n    'bagging_freq': 4,\n    'lambda_l1': 10,\n    'lambda_l2': 0.06\n}\n\nXGB_Params = {\n    'learning_rate': 0.0666,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.6,\n    'colsample_bytree': 0.6,\n    'reg_alpha': 1,\n    'reg_lambda': 5,\n    'random_state': 42  # Or any value for seed\n}\n\nCatBoost_Params = {\n    'learning_rate': 0.06,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': 42,  # Same seed value as for XGBoost\n    'verbose': 0,\n    'l2_leaf_reg': 10\n}\n\n# Create LightGBM, XGBoost, and CatBoost models\nLight = LGBMRegressor(**LGBParams, random_state=42, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Create the voting model\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n])\n\n# Train function\ndef TrainML(model_class, X_train, y_train, test_data):\n    model = clone(model_class)  # Clone model to avoid refitting the same model\n    model.fit(X_train, y_train)  # Fit the model using the training data\n    \n    # Return predictions\n    return model.predict(test_data)\n\n# Train the model and make predictions\nSubmission1 = TrainML(voting_model, X_train, y_train, test_data)\n\n# Print the predictions\nprint(\"Predictions for the test data:\", Submission1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:06:53.598225Z","iopub.execute_input":"2024-12-05T13:06:53.598811Z","iopub.status.idle":"2024-12-05T13:06:53.943863Z","shell.execute_reply.started":"2024-12-05T13:06:53.598728Z","shell.execute_reply":"2024-12-05T13:06:53.942682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install xgboost catboost lightgbm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:06:53.945455Z","iopub.execute_input":"2024-12-05T13:06:53.946034Z","iopub.status.idle":"2024-12-05T13:07:04.374100Z","shell.execute_reply.started":"2024-12-05T13:06:53.945984Z","shell.execute_reply":"2024-12-05T13:07:04.372645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming Submission1 is a NumPy array (e.g., predictions)\n# Convert Submission1 to a pandas DataFrame\nSubmission1_df = pd.DataFrame(Submission1, columns=[\"Prediction\"])\n\n# Save the results to a CSV file\nSubmission1_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:07:04.376180Z","iopub.execute_input":"2024-12-05T13:07:04.376623Z","iopub.status.idle":"2024-12-05T13:07:04.386179Z","shell.execute_reply.started":"2024-12-05T13:07:04.376572Z","shell.execute_reply":"2024-12-05T13:07:04.384930Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Process the data for the second time before training the model\ntrain, test, sample = load_data()\n\ntrain = train.drop('id', axis=1)\ntest = test.drop('id', axis=1)   \n\nfeaturesCols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex',\n                'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI',\n                'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference',\n                'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP',\n                'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage',\n                'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec',\n                'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND',\n                'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU',\n                'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR',\n                'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season',\n                'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI',\n                'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM',\n                'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num',\n                'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM',\n                'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season',\n                'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw',\n                'SDS-SDS_Total_T', 'PreInt_EduHx-Season',\n                'PreInt_EduHx-computerinternet_hoursday', 'sii']\n\ntrain = train[featuresCols]\ntrain = train.dropna(subset='sii')\n\ncat_c = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n          'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', \n          'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n\ndef update(df):\n    global cat_c\n    for c in cat_c: \n        df[c] = df[c].fillna('Missing')\n        df[c] = df[c].astype('category')\n    return df\n        \ntrain = update(train)\ntest = update(test)\n\ndef create_mapping(column, dataset):\n    unique_values = dataset[column].unique()\n    return {value: idx for idx, value in enumerate(unique_values)}\n\nfor col in cat_c:\n    mapping = create_mapping(col, train)\n    mappingTe = create_mapping(col, test)\n    \n    train[col] = train[col].replace(mapping).astype(int)\n    test[col] = test[col].replace(mappingTe).astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:07:04.388132Z","iopub.execute_input":"2024-12-05T13:07:04.388726Z","iopub.status.idle":"2024-12-05T13:07:04.536005Z","shell.execute_reply.started":"2024-12-05T13:07:04.388682Z","shell.execute_reply":"2024-12-05T13:07:04.534565Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.base import clone\nfrom sklearn.ensemble import VotingRegressor\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\n\n# Set a seed for reproducibility\nseed = 42  # You can choose any integer value\n\n# Example: Load or define your data\nX_train = pd.DataFrame(np.random.rand(100, 10))  # 100 samples, 10 features\ny_train = pd.Series(np.random.rand(100))         # 100 target values\ntest_data = pd.DataFrame(np.random.rand(10, 10))  # 10 test samples, 10 features\n\n# Check if y_train is a 1D array\nprint(\"Shape of y_train:\", y_train.shape)  # Should print something like (100,)\n\n# Ensure y_train is 1D\ny_train = np.ravel(y_train)  # Flatten y_train if it's 2D\n\n# Set the parameters for the models\nLGBParams = {\n    'learning_rate': 0.066,\n    'max_depth': 12,\n    'num_leaves': 478,\n    'min_data_in_leaf': 13,\n    'feature_fraction': 0.666,\n    'bagging_fraction': 0.666,\n    'bagging_freq': 4,\n    'lambda_l1': 10,\n    'lambda_l2': 0.06\n}\n\nXGB_Params = {\n    'learning_rate': 0.0666,\n    'max_depth': 6,\n    'n_estimators': 200,\n    'subsample': 0.6,\n    'colsample_bytree': 0.6,\n    'reg_alpha': 1,\n    'reg_lambda': 5,\n    'random_state': seed\n}\n\nCatBoost_Params = {\n    'learning_rate': 0.06,\n    'depth': 6,\n    'iterations': 200,\n    'random_seed': seed,\n    'verbose': 0,\n    'l2_leaf_reg': 10\n}\n\n# Create LightGBM, XGBoost, and CatBoost models\nLight = LGBMRegressor(**LGBParams, random_state=seed, n_estimators=300)\nXGB_Model = XGBRegressor(**XGB_Params)\nCatBoost_Model = CatBoostRegressor(**CatBoost_Params)\n\n# Create the voting model\nvoting_model = VotingRegressor(estimators=[\n    ('lightgbm', Light),\n    ('xgboost', XGB_Model),\n    ('catboost', CatBoost_Model),\n])\n\n# Train function\ndef TrainML(model_class, X_train, y_train, test_data):\n    model = clone(model_class)  # Clone model to avoid refitting the same model\n    model.fit(X_train, y_train)  # Fit the model using the training data\n    \n    # Return predictions\n    return model.predict(test_data)\n\n# Train the model and make predictions\nSubmission2 = TrainML(voting_model, X_train, y_train, test_data)\n\n# Print the predictions\nprint(\"Predictions for the test data:\", Submission2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:07:04.537554Z","iopub.execute_input":"2024-12-05T13:07:04.537921Z","iopub.status.idle":"2024-12-05T13:07:04.973666Z","shell.execute_reply.started":"2024-12-05T13:07:04.537889Z","shell.execute_reply":"2024-12-05T13:07:04.972480Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming Submission1 is a NumPy array (e.g., predictions)\n# Convert Submission1 to a pandas DataFrame\nSubmission2_df = pd.DataFrame(Submission2, columns=[\"Prediction\"])\n\n# Save the results to a CSV file\nSubmission2_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:07:04.977162Z","iopub.execute_input":"2024-12-05T13:07:04.977687Z","iopub.status.idle":"2024-12-05T13:07:04.986271Z","shell.execute_reply.started":"2024-12-05T13:07:04.977637Z","shell.execute_reply":"2024-12-05T13:07:04.984800Z"}},"outputs":[],"execution_count":null}]}