{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":125192,"databundleVersionId":15408205,"sourceType":"competition"},{"sourceId":477177,"sourceType":"datasetVersion","datasetId":216167}],"dockerImageVersionId":31259,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"IMPORT PACKAGES\n","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport os\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nimport xgboost as xgb\nfrom lightgbm import LGBMClassifier\nimport lightgbm as lgb\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.linear_model import LogisticRegression\nimport catboost as cb\nfrom sklearn.metrics import classification_report, roc_auc_score\nfrom xgboost import XGBClassifier\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score\nimport warnings\nwarnings.filterwarnings('ignore')\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T15:59:56.725268Z","iopub.execute_input":"2026-02-07T15:59:56.725758Z","iopub.status.idle":"2026-02-07T16:00:04.249845Z","shell.execute_reply.started":"2026-02-07T15:59:56.725711Z","shell.execute_reply":"2026-02-07T16:00:04.249254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for dirname , _ , filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n       print( os.path.join(dirname , filename))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:00:04.250953Z","iopub.execute_input":"2026-02-07T16:00:04.251564Z","iopub.status.idle":"2026-02-07T16:00:04.261624Z","shell.execute_reply.started":"2026-02-07T16:00:04.25154Z","shell.execute_reply":"2026-02-07T16:00:04.260985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nDATA_PATH = \"/kaggle/input/playground-series-s6e2/\"\nN_FOLDS = 10\nRANDOM_STATE = 42\n\ntrain_df = pd.read_csv('/kaggle/input/playground-series-s6e2/train.csv')\ntest_df = pd.read_csv('/kaggle/input/playground-series-s6e2/test.csv')\nsubmission_df = pd.read_csv('/kaggle/input/playground-series-s6e2/sample_submission.csv')\n\ntrain_df.drop(columns = 'id' , inplace = True)\ntest_df.drop(columns = 'id' , inplace = True)\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:00:04.262399Z","iopub.execute_input":"2026-02-07T16:00:04.262666Z","iopub.status.idle":"2026-02-07T16:00:05.30923Z","shell.execute_reply.started":"2026-02-07T16:00:04.26264Z","shell.execute_reply":"2026-02-07T16:00:05.30863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:00:05.310157Z","iopub.execute_input":"2026-02-07T16:00:05.31048Z","iopub.status.idle":"2026-02-07T16:00:05.364124Z","shell.execute_reply.started":"2026-02-07T16:00:05.310429Z","shell.execute_reply":"2026-02-07T16:00:05.363578Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n  EXPLORATORY DATA ANALYSIS","metadata":{}},{"cell_type":"code","source":"sns.countplot(x  ='Heart Disease' , data = train_df)\nplt.title('Heart Disease aka target feature distribution!')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:00:05.365966Z","iopub.execute_input":"2026-02-07T16:00:05.366201Z","iopub.status.idle":"2026-02-07T16:00:06.571427Z","shell.execute_reply.started":"2026-02-07T16:00:05.36618Z","shell.execute_reply":"2026-02-07T16:00:06.570798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.histplot(data = train_df , x = 'Age' , hue = 'Heart Disease' , multiple = 'stack' , kde = True , bins = 30)\nplt.title('Age Distribution by Heart Disease')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:00:06.572204Z","iopub.execute_input":"2026-02-07T16:00:06.572468Z","iopub.status.idle":"2026-02-07T16:00:09.954372Z","shell.execute_reply.started":"2026-02-07T16:00:06.572446Z","shell.execute_reply":"2026-02-07T16:00:09.953779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def age_risk_category(age):\n    if age < 40 :\n        return 0\n\n    elif age < 50:\n        return 2\n\n    elif 50 <= age < 65:\n        return 3\n\n    else :\n        return 1\n\n\ntrain_df['Age risk category'] = train_df['Age'].apply(age_risk_category)\ntest_df['Age risk category'] = test_df['Age'].apply(age_risk_category)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:00:09.955336Z","iopub.execute_input":"2026-02-07T16:00:09.955813Z","iopub.status.idle":"2026-02-07T16:00:10.265419Z","shell.execute_reply.started":"2026-02-07T16:00:09.955791Z","shell.execute_reply":"2026-02-07T16:00:10.264818Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.histplot(data = train_df , hue = 'Heart Disease' , x = 'Age risk category' , kde = True , bins = 30 , multiple = 'stack')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:00:10.266403Z","iopub.execute_input":"2026-02-07T16:00:10.266684Z","iopub.status.idle":"2026-02-07T16:00:13.708744Z","shell.execute_reply.started":"2026-02-07T16:00:10.266654Z","shell.execute_reply":"2026-02-07T16:00:13.708071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.countplot(x = 'Sex', hue = 'Heart Disease' , data = train_df)\nplt.title('Heart Disease distribution with Sex')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:00:13.709717Z","iopub.execute_input":"2026-02-07T16:00:13.709985Z","iopub.status.idle":"2026-02-07T16:00:15.242285Z","shell.execute_reply.started":"2026-02-07T16:00:13.709964Z","shell.execute_reply":"2026-02-07T16:00:15.241645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.countplot(x=\"Chest pain type\", hue=\"Heart Disease\", data=train_df)\nplt.title(\"Chest Pain Type vs Heart Disease\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:00:15.243146Z","iopub.execute_input":"2026-02-07T16:00:15.243459Z","iopub.status.idle":"2026-02-07T16:00:16.957813Z","shell.execute_reply.started":"2026-02-07T16:00:15.243426Z","shell.execute_reply":"2026-02-07T16:00:16.957182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.boxplot(x = 'Heart Disease' , y  = 'Cholesterol' , data = train_df)\nplt.title('Cholestrol level by Heart Disease')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:00:16.958708Z","iopub.execute_input":"2026-02-07T16:00:16.958995Z","iopub.status.idle":"2026-02-07T16:00:18.062284Z","shell.execute_reply.started":"2026-02-07T16:00:16.958972Z","shell.execute_reply":"2026-02-07T16:00:18.061661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.boxplot(x=\"Heart Disease\", y=\"BP\", data=train_df)\nplt.title(\"Blood Pressure by Heart Disease\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:00:18.063296Z","iopub.execute_input":"2026-02-07T16:00:18.063574Z","iopub.status.idle":"2026-02-07T16:00:19.195712Z","shell.execute_reply.started":"2026-02-07T16:00:18.06355Z","shell.execute_reply":"2026-02-07T16:00:19.195097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.histplot(data = train_df , x = 'Max HR' , hue = 'Heart Disease' , kde = True)\nplt.title('Max HR vs Heart Disease')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:00:19.196652Z","iopub.execute_input":"2026-02-07T16:00:19.196894Z","iopub.status.idle":"2026-02-07T16:00:23.116378Z","shell.execute_reply.started":"2026-02-07T16:00:19.196873Z","shell.execute_reply":"2026-02-07T16:00:23.115726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.scatterplot( x= 'ST depression' , y = 'Slope of ST' , hue = 'Heart Disease', data = train_df  )\nplt.title('ST depression vs Slope of ST')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:00:23.119143Z","iopub.execute_input":"2026-02-07T16:00:23.119564Z","iopub.status.idle":"2026-02-07T16:00:45.003033Z","shell.execute_reply.started":"2026-02-07T16:00:23.119531Z","shell.execute_reply":"2026-02-07T16:00:45.002398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.countplot(x=\"Thallium\", hue=\"Heart Disease\", data=train_df)\nplt.title(\"Thallium Test Results vs Heart Disease\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:00:45.003881Z","iopub.execute_input":"2026-02-07T16:00:45.004132Z","iopub.status.idle":"2026-02-07T16:00:46.536196Z","shell.execute_reply.started":"2026-02-07T16:00:45.004112Z","shell.execute_reply":"2026-02-07T16:00:46.535593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lenc = LabelEncoder()\ntrain_df['Heart Disease'] = lenc.fit_transform(train_df['Heart Disease'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:00:46.537051Z","iopub.execute_input":"2026-02-07T16:00:46.537358Z","iopub.status.idle":"2026-02-07T16:00:46.63138Z","shell.execute_reply.started":"2026-02-07T16:00:46.537336Z","shell.execute_reply":"2026-02-07T16:00:46.630795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corr = train_df.corr()\nsns.heatmap(corr , annot = False , cmap='coolwarm')\nplt.title('Feature Correlation Heatmap')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:00:46.632136Z","iopub.execute_input":"2026-02-07T16:00:46.632421Z","iopub.status.idle":"2026-02-07T16:00:47.30991Z","shell.execute_reply.started":"2026-02-07T16:00:46.632399Z","shell.execute_reply":"2026-02-07T16:00:47.309308Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"\nHeart Disease Prediction - Kaggle Notebook Version\nPlayground Series S6E2 - Optimized for Kaggle Environment\n\nJust copy this entire script into a Kaggle notebook and run!\n\"\"\"\n\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Install packages (Kaggle usually has these, but just in case)\nimport sys\n!{sys.executable} -m pip install -q xgboost lightgbm catboost\n\nimport xgboost as xgb\nimport lightgbm as lgb\nimport catboost as cb\nfrom sklearn.linear_model import LogisticRegression\n\n# ============================================================================\n# CONFIGURATION FOR KAGGLE\n# ============================================================================\nDATA_PATH = \"/kaggle/input/playground-series-s6e2/\"\nN_FOLDS = 10\nRANDOM_STATE = 42\n\nprint(\"🚀 Starting Heart Disease Prediction Pipeline\")\nprint(\"=\"*70)\n\n# ============================================================================\n# FEATURE ENGINEERING\n# ============================================================================\ndef create_features(df):\n    \"\"\"Advanced feature engineering based on EDA insights\"\"\"\n    df = df.copy()\n    \n    print(f\"Creating features from {df.shape[1]} columns...\")\n    \n    # Age-based features\n    df['Age_squared'] = df['Age'] ** 2\n    df['Age_cubed'] = df['Age'] ** 3\n    df['Age_risk_category'] = pd.cut(df['Age'], bins=[0, 40, 50, 60, 70, 100], labels=[0, 1, 2, 3, 4])\n    df['Age_risk_category'] = df['Age_risk_category'].astype(int)\n    \n    # Heart rate features (strong predictor from EDA!)\n    df['Max_HR_squared'] = df['Max HR'] ** 2\n    df['HR_Age_ratio'] = df['Max HR'] / (df['Age'] + 1)\n    df['HR_below_120'] = (df['Max HR'] < 120).astype(int)\n    df['HR_above_160'] = (df['Max HR'] > 160).astype(int)\n    \n    # BP features\n    df['BP_squared'] = df['BP'] ** 2\n    df['BP_Age_ratio'] = df['BP'] / (df['Age'] + 1)\n    df['High_BP'] = (df['BP'] > 140).astype(int)\n    df['Low_BP'] = (df['BP'] < 110).astype(int)\n    \n    # Cholesterol features\n    df['Cholesterol_squared'] = df['Cholesterol'] ** 2\n    df['Chol_Age_ratio'] = df['Cholesterol'] / (df['Age'] + 1)\n    df['High_Cholesterol'] = (df['Cholesterol'] > 240).astype(int)\n    \n    # ST Depression features\n    df['ST_depression_squared'] = df['ST depression'] ** 2\n    df['ST_depression_cubed'] = df['ST depression'] ** 3\n    df['ST_depression_high'] = (df['ST depression'] > 1.5).astype(int)\n    \n    # Interaction features (based on correlation insights)\n    df['Sex_ChestPain'] = df['Sex'] * df['Chest pain type']\n    df['Age_Sex'] = df['Age'] * df['Sex']\n    df['Thallium_ChestPain'] = df['Thallium'] * df['Chest pain type']\n    df['MaxHR_ExerciseAngina'] = df['Max HR'] * df['Exercise angina']\n    df['ST_Slope'] = df['ST depression'] * df['Slope of ST']\n    df['Vessels_Thallium'] = df['Number of vessels fluro'] * df['Thallium']\n    \n    # Cardiovascular risk score (domain knowledge)\n    df['CV_risk_score'] = (\n        df['Age'] * 0.1 +\n        df['Sex'] * 10 +\n        df['BP'] * 0.1 +\n        df['Cholesterol'] * 0.05 +\n        df['FBS over 120'] * 5 +\n        df['Exercise angina'] * 8 +\n        df['ST depression'] * 3\n    )\n    \n    # Thallium-based features (VERY strong predictor from EDA!)\n    df['Thallium_is_7'] = (df['Thallium'] == 7).astype(int)\n    df['Thallium_is_3'] = (df['Thallium'] == 3).astype(int)\n    df['Thallium_is_6'] = (df['Thallium'] == 6).astype(int)\n    \n    # Chest pain type features (strong predictor from EDA!)\n    df['ChestPain_type4'] = (df['Chest pain type'] == 4).astype(int)\n    df['ChestPain_type3'] = (df['Chest pain type'] == 3).astype(int)\n    df['ChestPain_type2'] = (df['Chest pain type'] == 2).astype(int)\n    \n    # EKG abnormality\n    df['EKG_abnormal'] = (df['EKG results'] > 0).astype(int)\n    \n    # Vessel disease severity\n    df['Severe_vessel_disease'] = (df['Number of vessels fluro'] >= 2).astype(int)\n    df['No_vessel_disease'] = (df['Number of vessels fluro'] == 0).astype(int)\n    \n    # Combined risk factors\n    df['Multiple_risks'] = (\n        (df['Age'] > 60).astype(int) +\n        (df['Sex'] == 1).astype(int) +\n        (df['BP'] > 140).astype(int) +\n        (df['Cholesterol'] > 240).astype(int) +\n        (df['FBS over 120'] == 1).astype(int) +\n        (df['Exercise angina'] == 1).astype(int)\n    )\n    \n    # Health score\n    df['Health_score'] = (\n        df['Max HR'] / 200 * 100 -\n        df['ST depression'] * 10 -\n        df['Exercise angina'] * 15 -\n        df['Number of vessels fluro'] * 10\n    )\n    \n    # Additional ratio features\n    df['ST_HR_ratio'] = df['ST depression'] / (df['Max HR'] + 1)\n    df['Vessels_Age_ratio'] = df['Number of vessels fluro'] / (df['Age'] + 1)\n    \n    print(f\"✅ Features created! Total: {df.shape[1]} columns\")\n    return df\n\n# ============================================================================\n# DATA LOADING\n# ============================================================================\nprint(\"\\n📂 Loading data from Kaggle input...\")\ntrain = pd.read_csv(f\"{DATA_PATH}train.csv\")\ntest = pd.read_csv(f\"{DATA_PATH}test.csv\")\n\nprint(f\"Train shape: {train.shape}\")\nprint(f\"Test shape: {test.shape}\")\n\n# Check target distribution\nprint(f\"\\nTarget distribution:\")\nprint(train['Heart Disease'].value_counts(normalize=True))\n\n# Prepare data\n# Convert target to binary (0 and 1)\ny = (train['Heart Disease'] == 'Presence').astype(int).values\ntest_ids = test['id'].values\n\nprint(f\"\\nTarget converted to binary:\")\nprint(f\"  0 (Absence): {(y==0).sum()}\")\nprint(f\"  1 (Presence): {(y==1).sum()}\")\n\ntrain = train.drop(['id', 'Heart Disease'], axis=1)\ntest = test.drop(['id'], axis=1)\n\n# Create features\nprint(\"\\n🔧 Feature Engineering...\")\ntrain = create_features(train)\ntest = create_features(test)\n\n# ============================================================================\n# MODEL TRAINING WITH CROSS-VALIDATION\n# ============================================================================\nprint(f\"\\n🎯 Training Models with {N_FOLDS}-Fold CV...\")\nprint(\"=\"*70)\n\n# Initialize prediction arrays\nxgb_oof = np.zeros(len(train))\nxgb_test = np.zeros(len(test))\n\nlgb_oof = np.zeros(len(train))\nlgb_test = np.zeros(len(test))\n\ncb_oof = np.zeros(len(train))\ncb_test = np.zeros(len(test))\n\n# Cross-validation\nskf = StratifiedKFold(n_splits=N_FOLDS, shuffle=True, random_state=RANDOM_STATE)\n\nfor fold, (train_idx, val_idx) in enumerate(skf.split(train, y)):\n    print(f\"\\n{'='*70}\")\n    print(f\"📊 Fold {fold + 1}/{N_FOLDS}\")\n    print(f\"{'='*70}\")\n    \n    X_train, X_val = train.iloc[train_idx], train.iloc[val_idx]\n    y_train, y_val = y[train_idx], y[val_idx]\n    \n    # -------------------------------------------------------------------\n    # XGBoost\n    # -------------------------------------------------------------------\n    print(\"\\n🌳 Training XGBoost...\")\n    xgb_params = {\n        'objective': 'binary:logistic',\n        'eval_metric': 'auc',\n        'max_depth': 6,\n        'learning_rate': 0.02,\n        'subsample': 0.8,\n        'colsample_bytree': 0.8,\n        'min_child_weight': 3,\n        'gamma': 0.1,\n        'reg_alpha': 0.1,\n        'reg_lambda': 1.0,\n        'random_state': RANDOM_STATE,\n        'tree_method': 'hist',\n        'device': 'cuda',  # For XGBoost 3.x (changed from gpu_id)\n    }\n    \n    xgb_model = xgb.XGBClassifier(**xgb_params, n_estimators=3000, early_stopping_rounds=100)\n    xgb_model.fit(\n        X_train, y_train,\n        eval_set=[(X_val, y_val)],\n        verbose=False\n    )\n    \n    xgb_oof[val_idx] = xgb_model.predict_proba(X_val)[:, 1]\n    xgb_test += xgb_model.predict_proba(test)[:, 1] / N_FOLDS\n    \n    xgb_score = roc_auc_score(y_val, xgb_oof[val_idx])\n    print(f\"   XGBoost Fold {fold+1} AUC: {xgb_score:.6f}\")\n    \n    # -------------------------------------------------------------------\n    # LightGBM\n    # -------------------------------------------------------------------\n    print(\"\\n💡 Training LightGBM...\")\n    lgb_params = {\n        'objective': 'binary',\n        'metric': 'auc',\n        'boosting_type': 'gbdt',\n        'max_depth': 6,\n        'learning_rate': 0.02,\n        'num_leaves': 31,\n        'feature_fraction': 0.8,\n        'bagging_fraction': 0.8,\n        'bagging_freq': 5,\n        'min_child_samples': 20,\n        'reg_alpha': 0.1,\n        'reg_lambda': 1.0,\n        'random_state': RANDOM_STATE,\n        'verbose': -1,\n        'device': 'gpu',  # Use GPU!\n    }\n    \n    lgb_model = lgb.LGBMClassifier(**lgb_params, n_estimators=3000)\n    lgb_model.fit(\n        X_train, y_train,\n        eval_set=[(X_val, y_val)],\n        callbacks=[lgb.early_stopping(100), lgb.log_evaluation(0)]\n    )\n    \n    lgb_oof[val_idx] = lgb_model.predict_proba(X_val)[:, 1]\n    lgb_test += lgb_model.predict_proba(test)[:, 1] / N_FOLDS\n    \n    lgb_score = roc_auc_score(y_val, lgb_oof[val_idx])\n    print(f\"   LightGBM Fold {fold+1} AUC: {lgb_score:.6f}\")\n    \n    # -------------------------------------------------------------------\n    # CatBoost\n    # -------------------------------------------------------------------\n    print(\"\\n🐱 Training CatBoost...\")\n    cb_params = {\n        'objective': 'Logloss',\n        'eval_metric': 'AUC',\n        'depth': 6,\n        'learning_rate': 0.02,\n        'l2_leaf_reg': 3,\n        'random_state': RANDOM_STATE,\n        'verbose': False,\n        'task_type': 'GPU',  # Use GPU!\n        'devices': '0',\n    }\n    \n    cb_model = cb.CatBoostClassifier(**cb_params, iterations=3000)\n    cb_model.fit(\n        X_train, y_train,\n        eval_set=(X_val, y_val),\n        early_stopping_rounds=100,\n        verbose=False\n    )\n    \n    cb_oof[val_idx] = cb_model.predict_proba(X_val)[:, 1]\n    cb_test += cb_model.predict_proba(test)[:, 1] / N_FOLDS\n    \n    cb_score = roc_auc_score(y_val, cb_oof[val_idx])\n    print(f\"   CatBoost Fold {fold+1} AUC: {cb_score:.6f}\")\n\n# ============================================================================\n# RESULTS & ENSEMBLE\n# ============================================================================\nprint(\"\\n\" + \"=\"*70)\nprint(\"📈 CROSS-VALIDATION RESULTS\")\nprint(\"=\"*70)\n\nxgb_cv_score = roc_auc_score(y, xgb_oof)\nlgb_cv_score = roc_auc_score(y, lgb_oof)\ncb_cv_score = roc_auc_score(y, cb_oof)\n\nprint(f\"\\nIndividual Model Scores:\")\nprint(f\"  XGBoost CV AUC:  {xgb_cv_score:.6f}\")\nprint(f\"  LightGBM CV AUC: {lgb_cv_score:.6f}\")\nprint(f\"  CatBoost CV AUC: {cb_cv_score:.6f}\")\n\n# Weighted ensemble\nw1, w2, w3 = 0.35, 0.35, 0.30\n\noof_ensemble = w1 * xgb_oof + w2 * lgb_oof + w3 * cb_oof\ntest_ensemble = w1 * xgb_test + w2 * lgb_test + w3 * cb_test\n\nensemble_score = roc_auc_score(y, oof_ensemble)\nprint(f\"\\n🎯 Ensemble CV AUC: {ensemble_score:.6f}\")\n\n# Stacking with Logistic Regression\nprint(\"\\n🔗 Training Stacking Meta-Model...\")\nmeta_train = np.column_stack([xgb_oof, lgb_oof, cb_oof])\nmeta_test = np.column_stack([xgb_test, lgb_test, cb_test])\n\nmeta_oof = np.zeros(len(train))\nmeta_test_preds = np.zeros(len(test))\n\nfor fold, (train_idx, val_idx) in enumerate(skf.split(meta_train, y)):\n    X_meta_train, X_meta_val = meta_train[train_idx], meta_train[val_idx]\n    y_meta_train, y_meta_val = y[train_idx], y[val_idx]\n    \n    meta_model = LogisticRegression(random_state=RANDOM_STATE, max_iter=1000)\n    meta_model.fit(X_meta_train, y_meta_train)\n    \n    meta_oof[val_idx] = meta_model.predict_proba(X_meta_val)[:, 1]\n    meta_test_preds += meta_model.predict_proba(meta_test)[:, 1] / N_FOLDS\n\nstacking_score = roc_auc_score(y, meta_oof)\nprint(f\"  Stacking CV AUC: {stacking_score:.6f}\")\n\n# ============================================================================\n# CHOOSE BEST METHOD & CREATE SUBMISSION\n# ============================================================================\nprint(\"\\n\" + \"=\"*70)\nprint(\"🏆 FINAL SUBMISSION\")\nprint(\"=\"*70)\n\nbest_method = max(\n    [(\"Ensemble\", ensemble_score, test_ensemble),\n     (\"Stacking\", stacking_score, meta_test_preds)],\n    key=lambda x: x[1]\n)\n\nprint(f\"\\nBest Method: {best_method[0]}\")\nprint(f\"Best CV AUC: {best_method[1]:.6f}\")\nprint(f\"Target to Beat: 0.95394\")\n\nfinal_predictions = best_method[2]\n\n# Create submission\nsubmission = pd.DataFrame({\n    'id': test_ids,\n    'Heart Disease': final_predictions\n})\n\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"\\n✅ Submission file created successfully!\")\nprint(f\"📊 Final CV Score: {best_method[1]:.6f}\")\n\nif best_method[1] > 0.95394:\n    print(\"🎉 EXCELLENT! CV score beats the target of 0.95394!\")\n    print(\"🏆 You're likely to rank in the TOP positions!\")\nelse:\n    print(f\"⚠️  CV score is {0.95394 - best_method[1]:.6f} below target\")\n    print(\"💡 Consider: More feature engineering or hyperparameter tuning\")\n\nprint(\"\\n\" + \"=\"*70)\nprint(\"🎯 Ready to submit! Click 'Submit to Competition' in Kaggle.\")\nprint(\"=\"*70)\n\n# Display first few predictions\nprint(\"\\nFirst 10 predictions:\")\nprint(submission.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:00:47.310878Z","iopub.execute_input":"2026-02-07T16:00:47.311083Z","iopub.status.idle":"2026-02-07T16:26:46.671517Z","shell.execute_reply.started":"2026-02-07T16:00:47.311063Z","shell.execute_reply":"2026-02-07T16:26:46.670568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nfrom scipy.optimize import minimize\nimport numpy as np\n\n# You should already have these from your notebook:\n# xgb_oof, lgb_oof, cb_oof (out-of-fold predictions)\n# xgb_test, lgb_test, cb_test (test predictions)\n# y (target variable)\n\nprint(\"Current ensemble weights: 0.35, 0.35, 0.30\")\n\n# Check individual model scores\nxgb_score = roc_auc_score(y, xgb_oof)\nlgb_score = roc_auc_score(y, lgb_oof)\ncb_score = roc_auc_score(y, cb_oof)\n\nprint(f\"\\nIndividual CV Scores:\")\nprint(f\"  XGBoost:  {xgb_score:.6f}\")\nprint(f\"  LightGBM: {lgb_score:.6f}\")\nprint(f\"  CatBoost: {cb_score:.6f}\")\n\n# Current ensemble\ncurrent_ensemble = 0.35 * xgb_oof + 0.35 * lgb_oof + 0.30 * cb_oof\ncurrent_score = roc_auc_score(y, current_ensemble)\nprint(f\"\\nCurrent Ensemble: {current_score:.6f}\")\n\n# ================================================================\n# OPTIMIZE WEIGHTS\n# ================================================================\n\ndef objective(weights):\n    \"\"\"Objective function to minimize (negative AUC)\"\"\"\n    w1, w2, w3 = weights\n    ensemble_pred = w1 * xgb_oof + w2 * lgb_oof + w3 * cb_oof\n    return -roc_auc_score(y, ensemble_pred)\n\n# Constraints: weights must sum to 1\nconstraints = ({'type': 'eq', 'fun': lambda w: np.sum(w) - 1.0})\n\n# Bounds: each weight between 0 and 1\nbounds = [(0.0, 1.0), (0.0, 1.0), (0.0, 1.0)]\n\n# Initial guess (current weights)\ninitial_weights = np.array([0.35, 0.35, 0.30])\n\n# Optimize\nprint(\"\\nOptimizing weights...\")\nresult = minimize(\n    objective,\n    initial_weights,\n    method='SLSQP',\n    bounds=bounds,\n    constraints=constraints,\n    options={'maxiter': 1000, 'ftol': 1e-9}\n)\n\noptimal_weights = result.x\noptimal_score = -result.fun\n\nprint(f\"\\n{'='*60}\")\nprint(\"OPTIMIZATION RESULTS\")\nprint(f\"{'='*60}\")\nprint(f\"\\nOptimal Weights:\")\nprint(f\"  XGBoost:  {optimal_weights[0]:.4f}\")\nprint(f\"  LightGBM: {optimal_weights[1]:.4f}\")\nprint(f\"  CatBoost: {optimal_weights[2]:.4f}\")\nprint(f\"  Sum: {optimal_weights.sum():.4f}\")\n\nprint(f\"\\n📊 CV Scores:\")\nprint(f\"  Current (0.35, 0.35, 0.30): {current_score:.6f}\")\nprint(f\"  Optimized:                  {optimal_score:.6f}\")\nprint(f\"  Improvement:                +{optimal_score - current_score:.6f}\")\n\n# ================================================================\n# CREATE OPTIMIZED SUBMISSION\n# ================================================================\n\n# Apply optimal weights to test predictions\noptimized_test_pred = (\n    optimal_weights[0] * xgb_test +\n    optimal_weights[1] * lgb_test +\n    optimal_weights[2] * cb_test\n)\n\n# Create submission\nsubmission_optimized = pd.DataFrame({\n    'id': test_ids,\n    'Heart Disease': optimized_test_pred\n})\n\nsubmission_optimized.to_csv('/kaggle/working/submission_optimized.csv', index=False)\n\nprint(\"\\n✅ Optimized submission created!\")\nprint(f\"\\n🎯 Expected LB score: ~{optimal_score:.5f}\")\nprint(\"   (Should be 0.0002-0.0005 better than 0.95331)\")\n\n# ================================================================\n# ALSO TRY: DIFFERENT ENSEMBLE METHODS\n# ================================================================\n\n# Method 2: Rank averaging\nfrom scipy.stats import rankdata\n\nrank_ensemble = (\n    rankdata(xgb_test) / len(xgb_test) +\n    rankdata(lgb_test) / len(lgb_test) +\n    rankdata(cb_test) / len(cb_test)\n) / 3\n\n# Method 3: Power mean (geometric-like)\npower_ensemble = (\n    (xgb_test ** 2 + lgb_test ** 2 + cb_test ** 2) / 3\n) ** 0.5\n\n# Evaluate on OOF\nrank_oof = (rankdata(xgb_oof) / len(xgb_oof) + \n            rankdata(lgb_oof) / len(lgb_oof) + \n            rankdata(cb_oof) / len(cb_oof)) / 3\nrank_score = roc_auc_score(y, rank_oof)\n\npower_oof = ((xgb_oof**2 + lgb_oof**2 + cb_oof**2) / 3) ** 0.5\npower_score = roc_auc_score(y, power_oof)\n\nprint(f\"\\n{'='*60}\")\nprint(\"ALTERNATIVE ENSEMBLE METHODS\")\nprint(f\"{'='*60}\")\nprint(f\"Weighted Optimal: {optimal_score:.6f}\")\nprint(f\"Rank Averaging:   {rank_score:.6f}\")\nprint(f\"Power Mean:       {power_score:.6f}\")\n\n# Choose the best\nbest_method = max(\n    [(\"Weighted\", optimal_score, optimized_test_pred),\n     (\"Rank\", rank_score, rank_ensemble),\n     (\"Power\", power_score, power_ensemble)],\n    key=lambda x: x[1]\n)\n\nprint(f\"\\n🏆 Best method: {best_method[0]} with CV {best_method[1]:.6f}\")\n\n# Save the best\nfinal_submission = pd.DataFrame({\n    'id': test_ids,\n    'Heart Disease': best_method[2]\n})\n\nfinal_submission.to_csv('/kaggle/working/submission_best.csv', index=False)\nprint(\"✅ Best submission saved as submission_best.csv\")\n\nprint(f\"\\n🎯 EXPECTED IMPROVEMENT:\")\nprint(f\"   From: 0.95331\")\nprint(f\"   To:   ~{best_method[1]:.5f}\")\nprint(f\"   Gain: +{best_method[1] - 0.95331:.5f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T16:29:50.816477Z","iopub.execute_input":"2026-02-07T16:29:50.81738Z","iopub.status.idle":"2026-02-07T16:29:55.876707Z","shell.execute_reply.started":"2026-02-07T16:29:50.817323Z","shell.execute_reply":"2026-02-07T16:29:55.875985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}