{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### When only the last statement is considered\n- The best threshold for deciding whether it's gonna default or not is 0.3518\n- Accuracy = 0.8871\n- Precision = 0.7426\n- Recall = 0.8696\n- F1 Score = 0.8011\n- Amex Score = 0.7728\n- AUC = 0.9552\n- Normalized Weighted Gini = 0.9103\n- Percentage of total defaulters captured in Top Four Percent= 0.6354\n  \n  Note that these metrics are on validation data.","metadata":{}},{"cell_type":"markdown","source":"#### When all the statements are aggregated\n- The best threshold for deciding whether it's gonna default or not is 0.3518\n- Accuracy = 0.8971\n- Precision = 0.7827\n- Recall = 0.8394\n- F1 Score = 0.8101\n- Amex Score = 0.7754\n- AUC = 0.9565\n- Normalized Weighted Gini = 0.9103\n- Percentage of total defaulters captured in Top Four Percent= 0.6378\n  \n  Note that these metrics are on validation data.","metadata":{}},{"cell_type":"markdown","source":"## Import necessary libraries","metadata":{}},{"cell_type":"code","source":"#Import necessary libraries\nimport pandas as pd\nimport polars as pl\nimport numpy as np\nimport xgboost as xgb\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.metrics import balanced_accuracy_score, roc_auc_score, make_scorer\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T13:29:54.680191Z","iopub.execute_input":"2025-12-21T13:29:54.680448Z","iopub.status.idle":"2025-12-21T13:29:54.684908Z","shell.execute_reply.started":"2025-12-21T13:29:54.680426Z","shell.execute_reply":"2025-12-21T13:29:54.684175Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load the data in csv","metadata":{}},{"cell_type":"code","source":"ss_df = pd.read_csv('/kaggle/input/amex-default-prediction/sample_submission.csv', engine = 'python')\ntrain_labels = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv', engine = 'python')\ntrain_df1 = pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv', nrows=200_000) #contains header and first 200_000 data rows\ntrain_df2 = pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv', skiprows=range(1,200_000+1), nrows = 200_000) #contains header and data rows from 200_001 till and including 400_000\ntrain_df3 = pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv', skiprows=range(1,400_000+1), nrows=58913) #contains header and data rows from 400_001 till and including 458_913\ntrain_df1_2 = pd.concat([train_df1,train_df2], ignore_index = True)\ntrain_df = pd.DataFrame(pd.concat([train_df1_2, train_df3], ignore_index=True))\ncat_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T13:27:45.442358Z","iopub.execute_input":"2025-12-21T13:27:45.442729Z","iopub.status.idle":"2025-12-21T13:28:33.097047Z","shell.execute_reply.started":"2025-12-21T13:27:45.442708Z","shell.execute_reply":"2025-12-21T13:28:33.096359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# test_df1 = pd.read_csv('/kaggle/input/amex-default-prediction/test_data.csv', nrows=200_000)\n# test_df2 = pd.read_csv('/kaggle/input/amex-default-prediction/test_data.csv', skiprows=range(1,200_000+1), nrows=200_000)\n# test_df3 = pd.read_csv('/kaggle/input/amex-default-prediction/test_data.csv', skiprows=range(1,400_000+1), nrows=200_000)\n# test_df4 = pd.read_csv('/kaggle/input/amex-default-prediction/test_data.csv', skiprows=range(1,600_000+1), nrows=324_621)\n# test_df1_2 = pd.concat([test_df1, test_df2], ignore_index = True)\n# test_df3_4 = pd.concat([test_df3, test_df4], ignore_index = True)\n# test_df = pd.concat([test_df1_2, test_df3_4], ignore_index = True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Denoising function","metadata":{}},{"cell_type":"code","source":"def denoise_numeric(df):\n\n    df = df.copy()\n\n    # numeric columns only\n    num_cols = df.select_dtypes(include=['float', 'int']).columns\n\n    for col in num_cols:\n\n        # Remove infinities\n        #df[col].replace([np.inf, -np.inf], np.nan, inplace=True)\n\n        # Clip extreme outliers (standard Kaggle practice)\n        q1, q99 = df[col].quantile([0.01, 0.99])\n        df[col] = df[col].clip(q1, q99)\n\n        # Apply rounding / bucketing to reduce noise\n        # floor(x * 100) → keeps 2 decimal places\n        # You can change 100 to 1000 if you want more resolution.\n        df[col] = np.floor(df[col] * 100) / 100.0\n\n        # NA values remain NA — untouched\n\n    return df\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T13:28:33.098312Z","iopub.execute_input":"2025-12-21T13:28:33.098576Z","iopub.status.idle":"2025-12-21T13:28:33.103673Z","shell.execute_reply.started":"2025-12-21T13:28:33.098556Z","shell.execute_reply":"2025-12-21T13:28:33.103023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_merged = train_df.merge(train_labels, on='customer_ID', how='left')\ntrain_merged.shape\ntrain_merged['target'].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T17:57:22.614866Z","iopub.execute_input":"2025-12-19T17:57:22.615228Z","iopub.status.idle":"2025-12-19T17:57:23.430793Z","shell.execute_reply.started":"2025-12-19T17:57:22.615148Z","shell.execute_reply":"2025-12-19T17:57:23.430207Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Dealing with missing data\nWe will remove the columns for which the missing values are more than 50% using and also the 'customer","metadata":{}},{"cell_type":"code","source":"def CleanMissing(df, threshold):\n    # Drop the columns with missing fraction > criteria\n    cols_to_drop = df.columns[df.isna().mean() > threshold].tolist()\n    df_clear = df.drop(columns=cols_to_drop)\n    return df_clear","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T12:25:37.207586Z","iopub.execute_input":"2025-12-20T12:25:37.207803Z","iopub.status.idle":"2025-12-20T12:25:37.231130Z","shell.execute_reply.started":"2025-12-20T12:25:37.207787Z","shell.execute_reply":"2025-12-20T12:25:37.230385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Drop the columns with missing fraction > 0.5\ncols_to_drop = train_merged.columns[train_merged.isna().mean() > 0.5].tolist()\n\n# Add customer_ID and S_2\ncols_to_drop += ['customer_ID', 'S_2']\n\n# Create the modified dataframe\ntrain_merged = train_merged.drop(columns=cols_to_drop)\ncat_features = [c for c in cat_features if c not in cols_to_drop]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T17:54:37.138048Z","iopub.execute_input":"2025-12-19T17:54:37.138755Z","iopub.status.idle":"2025-12-19T17:54:37.640034Z","shell.execute_reply.started":"2025-12-19T17:54:37.138730Z","shell.execute_reply":"2025-12-19T17:54:37.639483Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Amex Metric Function","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)\n\ndef amex_metric_xgb(preds, dtrain):\n    y_true = pd.DataFrame({\"target\": dtrain.get_label()})\n    y_pred = pd.DataFrame({\"prediction\": preds})\n    return \"amex\", amex_metric(y_true, y_pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T12:25:37.232218Z","iopub.execute_input":"2025-12-20T12:25:37.232436Z","iopub.status.idle":"2025-12-20T12:25:37.250779Z","shell.execute_reply.started":"2025-12-20T12:25:37.232421Z","shell.execute_reply":"2025-12-20T12:25:37.250207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#test_df = denoise_numeric(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T04:19:03.201981Z","iopub.execute_input":"2025-11-19T04:19:03.202662Z","iopub.status.idle":"2025-11-19T04:19:12.443069Z","shell.execute_reply.started":"2025-11-19T04:19:03.202635Z","shell.execute_reply":"2025-11-19T04:19:12.442310Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels['customer_ID'].unique().size","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T17:54:47.683450Z","iopub.execute_input":"2025-12-19T17:54:47.684051Z","iopub.status.idle":"2025-12-19T17:54:47.804614Z","shell.execute_reply.started":"2025-12-19T17:54:47.684027Z","shell.execute_reply":"2025-12-19T17:54:47.804001Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Drop the columns you dropped in training dataset and declare the categorical varibles","metadata":{}},{"cell_type":"code","source":"cols_to_keep = X_train.columns\nX_test = test_df[cols_to_keep].copy()\n\nfor c in cat_features:\n    X_test[c] = X_test[c].astype('category')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T17:54:53.149494Z","iopub.execute_input":"2025-12-19T17:54:53.150068Z","iopub.status.idle":"2025-12-19T17:54:53.219422Z","shell.execute_reply.started":"2025-12-19T17:54:53.150045Z","shell.execute_reply":"2025-12-19T17:54:53.218557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -----------------------\n# 1. Align test columns\n# -----------------------\ncols_to_keep = X_train.columns\nX_test = test_df[cols_to_keep]\n\nfor c in cat_features:\n    X_test[c] = X_test[c].astype(\"category\")\n\n# -----------------------\n# 2. Predict on test\n# -----------------------\ntest_df['prediction'] = best_xgb.predict_proba(X_test)[:, 1]\n\n# -----------------------\n# 3. Aggregate per customer_ID\n# -----------------------\nfinal_pred = (\n    test_df.groupby('customer_ID')['prediction']\n    .tail(1)\n    .reset_index()\n)\n\n# -----------------------\n# 4. Merge with sample submission to ensure correct order\n# -----------------------\nsubmission = ss_df[['customer_ID']].merge(\n    final_pred,\n    on='customer_ID',\n    how='right'\n)\n\n# -----------------------\n# 5. Safety: fill missing\n# -----------------------\nsubmission['prediction'].fillna(0.0, inplace=True)\n\n# -----------------------\n# 6. Save file\n# -----------------------\nsubmission.to_csv(\"submission.csv\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T07:36:42.436362Z","iopub.execute_input":"2025-11-19T07:36:42.436782Z","iopub.status.idle":"2025-11-19T07:36:42.450929Z","shell.execute_reply.started":"2025-11-19T07:36:42.436753Z","shell.execute_reply":"2025-11-19T07:36:42.449535Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The amex metric for the test data is around 0.04, which means that our model is not performing good on the test data like it was doing on the validation data. The reason is that the validation data had the data for the same customer_IDs as were there in the training dataset and they must be correlated somehow as they are the data form a time series. The test data also has a time series data as there are 924621 entries, but only 75231 unique columns. Let's train only the on the the last statement of the training dataset and predict usingthe last statement of the dataset and fill the same target values for the all the statements for that customer - this sounds wrong because initially a customer's features might be goodand could get worse so the prediction in the test and validation data should be according to each statement","metadata":{}},{"cell_type":"code","source":"train_last = (\n    train_df\n    .groupby(\"customer_ID\")\n    .last()\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T13:33:15.196234Z","iopub.execute_input":"2025-12-21T13:33:15.196628Z","iopub.status.idle":"2025-12-21T13:33:15.828839Z","shell.execute_reply.started":"2025-12-21T13:33:15.196583Z","shell.execute_reply":"2025-12-21T13:33:15.827980Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_last","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T13:33:21.637166Z","iopub.execute_input":"2025-12-21T13:33:21.637452Z","iopub.status.idle":"2025-12-21T13:33:21.691768Z","shell.execute_reply.started":"2025-12-21T13:33:21.637425Z","shell.execute_reply":"2025-12-21T13:33:21.690949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_last_merged = train_last.merge(train_labels, on='customer_ID', how='left')\ntrain_last_merged.shape\ntrain_last_merged['target'].isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T17:57:39.952242Z","iopub.execute_input":"2025-12-19T17:57:39.952772Z","iopub.status.idle":"2025-12-19T17:57:40.080041Z","shell.execute_reply.started":"2025-12-19T17:57:39.952751Z","shell.execute_reply":"2025-12-19T17:57:40.079403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop the columns with missing fraction > 0.5\ncols_to_drop = train_last_merged.columns[train_last_merged.isna().mean() > 0.5].tolist()\n\n# Add customer_ID and S_2\ncols_to_drop += ['customer_ID', 'S_2']\n\n# Create the modified dataframe\ntrain_last_merged = train_last_merged.drop(columns=cols_to_drop)\ncat_features = [c for c in cat_features if c not in cols_to_drop]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T17:57:42.246243Z","iopub.execute_input":"2025-12-19T17:57:42.246509Z","iopub.status.idle":"2025-12-19T17:57:42.294790Z","shell.execute_reply.started":"2025-12-19T17:57:42.246489Z","shell.execute_reply":"2025-12-19T17:57:42.294033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train_last_merged.drop(columns=['target']).copy()\nfor c in cat_features:\n    X[c] = X[c].astype('category')\ny = train_last_merged['target'].copy()\nsum(y)/len(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T17:57:44.513256Z","iopub.execute_input":"2025-12-19T17:57:44.513780Z","iopub.status.idle":"2025-12-19T17:57:44.574691Z","shell.execute_reply.started":"2025-12-19T17:57:44.513757Z","shell.execute_reply":"2025-12-19T17:57:44.573905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(X, y, random_state=42, stratify = y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T17:57:47.860391Z","iopub.execute_input":"2025-12-19T17:57:47.860643Z","iopub.status.idle":"2025-12-19T17:57:47.910751Z","shell.execute_reply.started":"2025-12-19T17:57:47.860625Z","shell.execute_reply":"2025-12-19T17:57:47.909840Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#denoise the data\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import optuna\n\ntrial_history = []\n\ndef objective(trial):\n\n    params = {\n        \"objective\": \"binary:logistic\",\n        \"tree_method\": \"hist\",\n        \"device\": \"cuda\",\n        \"eval_metric\": \"auc\",\n        \"missing\": np.nan,\n        \"enable_categorical\": True,\n        \"n_estimators\": 400,\n\n        # ---- Parameters to tune ----\n        \"max_depth\": trial.suggest_int(\"max_depth\", 8, 80, step=8),\n        \"min_child_weight\": trial.suggest_int(\"min_child_weight\", 10, 100, step=10),\n        \"learning_rate\": trial.suggest_categorical(\"learning_rate\", [0.05, 0.1]),\n        \"subsample\": trial.suggest_categorical(\"subsample\", [0.3, 0.4, 0.5, 0.8, 1.0]),\n        \"colsample_bytree\": trial.suggest_categorical(\"colsample_bytree\", [0.5, 0.8, 1.0]),\n        \n\n        # ---- Regularization tuning ----\n        \"lambda\": trial.suggest_float(\"lambda\", 1e-3, 10, log=True),\n        \"alpha\": trial.suggest_float(\"alpha\", 1e-3, 5, log=True),\n        \"gamma\": trial.suggest_float(\"gamma\", 0, 10),\n    }\n\n    model = xgb.XGBClassifier(**params)\n\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_valid, y_valid)],\n        early_stopping_rounds=30,\n        verbose=False\n    )\n\n    # predictions\n    preds = model.predict_proba(X_valid)[:, 1]\n\n    # Compute AMEX\n    y_true_df = pd.DataFrame({'target': y_valid}).reset_index(drop=True)\n    y_pred_df = pd.DataFrame({'prediction': preds}).reset_index(drop=True)\n\n    score = amex_metric(y_true_df, y_pred_df)\n    \n    # ----------------------------\n    # SAVE TRIAL RESULTS SAFELY\n    # ----------------------------\n    trial_history.append({\n        \"params\": params.copy(),              # <-- IMPORTANT FIX\n        \"best_iteration\": model.best_iteration,\n        \"amex_score\": score\n    })\n    \n    return score  # Optuna will maximize the AMEX metric\n\n\n# Run study\nstudy = optuna.create_study(\n    study_name=\"xgb_amex_gpu\",\n    direction=\"maximize\",\n    storage=\"sqlite:///optuna_xgb.db\",\n    load_if_exists=True\n)\n\nstudy.optimize(objective, n_trials=300)  # adds 300 more\n\n# Convert to DataFrame\ntrial_df = pd.DataFrame(trial_history)\n\n# Flatten params dict into columns (optional but very useful!)\ntrial_df = trial_df.join(trial_df[\"params\"].apply(pd.Series)).drop(columns=[\"params\"])\n\ntrial_df.sort_values(\"amex_score\", ascending=False, inplace=True)\n\nprint(\"Best AMEX:\", study.best_trial.value)\nprint(\"Best Params:\", study.best_trial.params)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T18:30:33.715940Z","iopub.execute_input":"2025-12-19T18:30:33.716726Z","iopub.status.idle":"2025-12-19T18:53:57.357523Z","shell.execute_reply.started":"2025-12-19T18:30:33.716699Z","shell.execute_reply":"2025-12-19T18:53:57.356671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert to DataFrame\ntrial_df = pd.DataFrame(trial_history)\n\n# Flatten params dict into columns (optional but very useful!)\ntrial_df = trial_df.join(trial_df[\"params\"].apply(pd.Series)).drop(columns=[\"params\"])\n\ntrial_df.sort_values(\"amex_score\", ascending=False, inplace=True)\n\nprint(\"Best AMEX:\", study.best_trial.value)\nprint(\"Best Params:\", study.best_trial.params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T18:59:43.636652Z","iopub.execute_input":"2025-12-19T18:59:43.637111Z","iopub.status.idle":"2025-12-19T18:59:43.701176Z","shell.execute_reply.started":"2025-12-19T18:59:43.637087Z","shell.execute_reply":"2025-12-19T18:59:43.700570Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nfor col in ['max_depth', 'min_child_weight', 'learning_rate', 'subsample', \n            'colsample_bytree', 'n_estimators', 'lambda', 'alpha', 'gamma']:\n    \n    plt.figure(figsize=(6,4))\n    sns.scatterplot(data=trial_df, x=col, y='amex_score')\n    plt.title(f'{col} vs AMEX')\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T18:59:52.431801Z","iopub.execute_input":"2025-12-19T18:59:52.432400Z","iopub.status.idle":"2025-12-19T18:59:53.792064Z","shell.execute_reply.started":"2025-12-19T18:59:52.432376Z","shell.execute_reply":"2025-12-19T18:59:53.791370Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# The following values of bes","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import optuna\n\n# trial_history2 = []\n\n# def objective(trial):\n\n#     params = {\n#         \"objective\": \"binary:logistic\",\n#         \"tree_method\": \"hist\",\n#         \"eval_metric\": \"auc\",\n#         \"missing\": np.nan,\n#         \"enable_categorical\": True,\n\n#         # ---- tuned params ----\n#         \"max_depth\": 40,\n#         \"min_child_weight\": 30,\n#         \"learning_rate\":0.05,\n#         \"subsample\": 1.0,\n#         \"colsample_bytree\": 0.5,\n#         \"n_estimators\": 400,\n#         \"alpha\": 0,\n\n#         # ---- Regularization tuning ----\n#         \"lambda\": trial.suggest_float(\"lambda\", 1e-3, 7, log=True),\n#         \"gamma\": trial.suggest_float(\"gamma\", 0, 4),\n#     }\n\n#     model = xgb.XGBClassifier(**params)\n\n#     model.fit(\n#         X_train, y_train,\n#         eval_set=[(X_valid, y_valid)],\n#         early_stopping_rounds=30,\n#         verbose=False\n#     )\n\n#     # predictions\n#     preds = model.predict_proba(X_valid)[:, 1]\n\n#     # Compute AMEX\n#     y_true_df = pd.DataFrame({'target': y_valid}).reset_index(drop=True)\n#     y_pred_df = pd.DataFrame({'prediction': preds}).reset_index(drop=True)\n\n#     score = amex_metric(y_true_df, y_pred_df)\n    \n#     # ----------------------------\n#     # SAVE TRIAL RESULTS SAFELY\n#     # ----------------------------\n#     trial_history2.append({\n#         \"params\": params.copy(),              # <-- IMPORTANT FIX\n#         \"best_iteration\": model.best_iteration,\n#         \"amex_score\": score\n#     })\n    \n#     return score  # Optuna will maximize the AMEX metric\n\n\n# # Run study\n# study = optuna.create_study(direction=\"maximize\")\n# study.optimize(objective, n_trials=50)\n\n# # Convert to DataFrame\n# trial2_df = pd.DataFrame(trial_history2)\n\n# # Flatten params dict into columns (optional but very useful!)\n# trial2_df = trial2_df.join(trial2_df[\"params\"].apply(pd.Series)).drop(columns=[\"params\"])\n\n# trial2_df.sort_values(\"amex_score\", ascending=False, inplace=True)\n\n# print(\"Best AMEX:\", study.best_trial.value)\n# print(\"Best Params:\", study.best_trial.params)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T04:49:06.549957Z","iopub.execute_input":"2025-11-19T04:49:06.551652Z","iopub.status.idle":"2025-11-19T04:54:43.023356Z","shell.execute_reply.started":"2025-11-19T04:49:06.551613Z","shell.execute_reply":"2025-11-19T04:54:43.022464Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"alpha is cl","metadata":{}},{"cell_type":"code","source":"from xgboost.callback import EarlyStopping\n\nbest_params = {\n    \"objective\": \"binary:logistic\",\n    \"tree_method\": \"hist\",\n    \"device\": \"cuda\",\n    \"missing\": np.nan,\n    \"enable_categorical\": True,\n\n    # tuned params\n    \"max_depth\": 80,\n    \"min_child_weight\": 40,\n    \"learning_rate\": 0.05,\n    \"subsample\": 0.8,\n    \"colsample_bytree\": 0.5,\n    \"n_estimators\": 1000,   # LARGE ON PURPOSE\n    \"lambda\": 0.007670118758441345,\n    \"alpha\": 0.09157175965761555,\n    \"gamma\": 1.028865262384579,\n}\n\nbest_xgb = xgb.XGBClassifier(**best_params)\n\nbest_xgb.fit(\n    X_train,\n    y_train,\n    eval_set=[(X_valid, y_valid)],\n    eval_metric=\"auc\",  # REQUIRED for early stopping\n    callbacks=[\n        EarlyStopping(\n            rounds=50,           # patience\n            save_best=True,      # keep best trees\n            maximize=True\n        )\n    ],\n    verbose=False\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T19:11:28.615945Z","iopub.execute_input":"2025-12-19T19:11:28.616344Z","iopub.status.idle":"2025-12-19T19:11:34.618733Z","shell.execute_reply.started":"2025-12-19T19:11:28.616319Z","shell.execute_reply":"2025-12-19T19:11:34.615714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds = best_xgb.predict_proba(X_valid)[:,-1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T19:11:53.452458Z","iopub.execute_input":"2025-12-19T19:11:53.453053Z","iopub.status.idle":"2025-12-19T19:11:53.543501Z","shell.execute_reply.started":"2025-12-19T19:11:53.453030Z","shell.execute_reply":"2025-12-19T19:11:53.542887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    df = (pd.concat([y_true, y_pred], axis='columns')\n          .sort_values('prediction', ascending=False))\n    df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n    four_pct_cutoff = int(0.04 * df['weight'].sum())\n    df['weight_cumsum'] = df['weight'].cumsum()\n    df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n    return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n    \ndef weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    df = (pd.concat([y_true, y_pred], axis='columns')\n          .sort_values('prediction', ascending=False))\n    df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n    df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n    total_pos = (df['target'] * df['weight']).sum()\n    df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n    df['lorentz'] = df['cum_pos_found'] / total_pos\n    df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n    return df['gini'].sum()\n\ndef normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    y_true_pred = y_true.rename(columns={'target': 'prediction'})\n    return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T15:20:21.060595Z","iopub.execute_input":"2025-12-20T15:20:21.061364Z","iopub.status.idle":"2025-12-20T15:20:21.069003Z","shell.execute_reply.started":"2025-12-20T15:20:21.061335Z","shell.execute_reply":"2025-12-20T15:20:21.068407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = pd.DataFrame({'prediction': preds}).reset_index(drop=True)\ny_true = pd.DataFrame({'target': y_valid}).reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T19:11:57.344065Z","iopub.execute_input":"2025-12-19T19:11:57.344367Z","iopub.status.idle":"2025-12-19T19:11:57.349095Z","shell.execute_reply.started":"2025-12-19T19:11:57.344345Z","shell.execute_reply":"2025-12-19T19:11:57.348496Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"g = normalized_weighted_gini(y_true, y_pred)\nd = top_four_percent_captured(y_true, y_pred)\n\nprint(g, d, 0.5 * (g + d))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T19:11:59.071757Z","iopub.execute_input":"2025-12-19T19:11:59.072480Z","iopub.status.idle":"2025-12-19T19:11:59.097772Z","shell.execute_reply.started":"2025-12-19T19:11:59.072455Z","shell.execute_reply":"2025-12-19T19:11:59.097181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\n\n# 1. Predictions from your trained model\npreds = best_xgb.predict_proba(X_valid)[:, 1]\n\n# 2. Choose threshold\nthreshold = 0.5  # or any value you want (tuned threshold)\n\n# 3. Convert probabilities → binary labels\ny_pred_label = (preds >= threshold).astype(int)\n\n# 4. Compute confusion matrix\ncm = confusion_matrix(y_valid, y_pred_label)\n\n# 5. Plot\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm)\ndisp.plot(cmap=\"Blues\")\nplt.title(f\"Confusion Matrix (threshold={threshold})\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T19:12:10.154297Z","iopub.execute_input":"2025-12-19T19:12:10.154561Z","iopub.status.idle":"2025-12-19T19:12:10.402614Z","shell.execute_reply.started":"2025-12-19T19:12:10.154541Z","shell.execute_reply":"2025-12-19T19:12:10.401965Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds = best_xgb.predict_proba(X_valid)[:, 1]\ny_true = y_valid  # 0/1 labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T19:12:24.862289Z","iopub.execute_input":"2025-12-19T19:12:24.862837Z","iopub.status.idle":"2025-12-19T19:12:24.952031Z","shell.execute_reply.started":"2025-12-19T19:12:24.862815Z","shell.execute_reply":"2025-12-19T19:12:24.951311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics import f1_score\n\nthresholds = np.linspace(0.0, 1.0, 200)\nscores = []\n\nfor t in thresholds:\n    y_pred_label = (preds >= t).astype(int)\n    scores.append(f1_score(y_true, y_pred_label))\n\nbest_idx = np.argmax(scores)\nbest_threshold = thresholds[best_idx]\nbest_f1 = scores[best_idx]\n\nprint(\"Best threshold:\", best_threshold)\nprint(\"Best F1:\", best_f1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T19:14:27.094136Z","iopub.execute_input":"2025-12-19T19:14:27.094728Z","iopub.status.idle":"2025-12-19T19:14:28.286912Z","shell.execute_reply.started":"2025-12-19T19:14:27.094702Z","shell.execute_reply":"2025-12-19T19:14:28.286217Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import precision_recall_curve\nimport matplotlib.pyplot as plt\n\nprecision, recall, pr_thresholds = precision_recall_curve(y_true, preds)\n\nplt.figure(figsize=(7,5))\nplt.plot(recall, precision)\nplt.xlabel(\"Recall\")\nplt.ylabel(\"Precision\")\nplt.title(\"Precision–Recall Curve\")\nplt.grid()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T19:14:29.848674Z","iopub.execute_input":"2025-12-19T19:14:29.849386Z","iopub.status.idle":"2025-12-19T19:14:30.006847Z","shell.execute_reply.started":"2025-12-19T19:14:29.849359Z","shell.execute_reply":"2025-12-19T19:14:30.006205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\n\nfpr, tpr, roc_thresholds = roc_curve(y_true, preds)\nroc_auc = auc(fpr, tpr)\n\nplt.figure(figsize=(7,5))\nplt.plot(fpr, tpr, label=f\"AUC = {roc_auc:.4f}\")\nplt.xlabel(\"False Positive Rate\")\nplt.ylabel(\"True Positive Rate\")\nplt.title(\"ROC Curve\")\nplt.legend()\nplt.grid()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T19:14:33.198404Z","iopub.execute_input":"2025-12-19T19:14:33.198936Z","iopub.status.idle":"2025-12-19T19:14:33.362186Z","shell.execute_reply.started":"2025-12-19T19:14:33.198915Z","shell.execute_reply":"2025-12-19T19:14:33.361630Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import precision_score, recall_score, f1_score\n\nprecisions = []\nrecalls = []\nf1s = []\n\nfor t in thresholds:\n    y_pred_label = (preds >= t).astype(int)\n    precisions.append(precision_score(y_true, y_pred_label))\n    recalls.append(recall_score(y_true, y_pred_label))\n    f1s.append(f1_score(y_true, y_pred_label))\n\nplt.figure(figsize=(8,6))\nplt.plot(thresholds, precisions, label=\"Precision\")\nplt.plot(thresholds, recalls, label=\"Recall\")\nplt.plot(thresholds, f1s, label=\"F1 Score\")\nplt.axvline(best_threshold, color='black', linestyle='--', label=f\"Best t={best_threshold:.3f}\")\nplt.xlabel(\"Threshold\")\nplt.ylabel(\"Score\")\nplt.title(\"Threshold Tuning Curve\")\nplt.legend()\nplt.grid()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T19:14:35.887914Z","iopub.execute_input":"2025-12-19T19:14:35.888483Z","iopub.status.idle":"2025-12-19T19:14:39.693000Z","shell.execute_reply.started":"2025-12-19T19:14:35.888461Z","shell.execute_reply":"2025-12-19T19:14:39.692405Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\n\ny_pred_best = (preds >= best_threshold).astype(int)\n\ncm = confusion_matrix(y_true, y_pred_best)\n\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm)\ndisp.plot(cmap=\"Blues\")\nplt.title(f\"Confusion Matrix (Best threshold = {best_threshold:.3f})\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T19:14:42.561423Z","iopub.execute_input":"2025-12-19T19:14:42.561685Z","iopub.status.idle":"2025-12-19T19:14:42.711642Z","shell.execute_reply.started":"2025-12-19T19:14:42.561665Z","shell.execute_reply":"2025-12-19T19:14:42.710998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\n# Convert probabilities to labels\ny_pred_best = (preds >= best_threshold).astype(int)\n\nacc = accuracy_score(y_true, y_pred_best)\nprec = precision_score(y_true, y_pred_best)\nrec = recall_score(y_true, y_pred_best)\nf1 = f1_score(y_true, y_pred_best)\n\nprint(f\"Best Threshold: {best_threshold:.4f}\")\nprint(f\"Accuracy:  {acc:.4f}\")\nprint(f\"Precision: {prec:.4f}\")\nprint(f\"Recall:    {rec:.4f}\")\nprint(f\"F1 Score:  {f1:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T19:14:52.501750Z","iopub.execute_input":"2025-12-19T19:14:52.502426Z","iopub.status.idle":"2025-12-19T19:14:52.529801Z","shell.execute_reply.started":"2025-12-19T19:14:52.502401Z","shell.execute_reply":"2025-12-19T19:14:52.529196Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### When only the last statement is considered\n- The best threshold for deciding whether it's gonna default or not is 0.3518\n- Accuracy = 0.8871\n- Precision = 0.7426\n- Recall = 0.8696\n- F1 Score = 0.8011\n- Amex Score = 0.7728\n- Normalized Weighted Gini = 0.9103\n- Percentage of total defaulters captured in Top Four Percent= 0.6354\n  \n  Note that these metrics are on validation data.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import precision_score, recall_score, f1_score\n\nthresholds = np.linspace(0.0, 1.0, 101)\n\nprecision_list = []\nrecall_list = []\nf1_list = []\n\nfor t in thresholds:\n    y_pred = (preds > t).astype(int)\n    precision_list.append(precision_score(y_valid, y_pred))\n    recall_list.append(recall_score(y_valid, y_pred))\n    f1_list.append(f1_score(y_valid, y_pred))\n\nplt.figure(figsize=(8, 6))\nplt.plot(thresholds, precision_list, label=\"Precision\")\nplt.plot(thresholds, recall_list, label=\"Recall\")\nplt.plot(thresholds, f1_list, label=\"F1 Score\")\nplt.xlabel(\"Threshold\")\nplt.ylabel(\"Score\")\nplt.title(\"Threshold Tuning Curve\")\nplt.legend()\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T19:15:04.842600Z","iopub.execute_input":"2025-12-19T19:15:04.843147Z","iopub.status.idle":"2025-12-19T19:15:06.845893Z","shell.execute_reply.started":"2025-12-19T19:15:04.843124Z","shell.execute_reply":"2025-12-19T19:15:06.845220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# test_last = (\n#     test_df\n#     .groupby(\"customer_ID\")\n#     .apply(lambda x: x.sort_values(\"S_2\").iloc[-1])\n#     .reset_index(drop=True)\n# )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T12:43:34.564690Z","iopub.execute_input":"2025-12-20T12:43:34.565378Z","iopub.status.idle":"2025-12-20T12:43:34.577349Z","shell.execute_reply.started":"2025-12-20T12:43:34.565353Z","shell.execute_reply":"2025-12-20T12:43:34.576512Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['customer_ID'].unique().sum()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -----------------------\n# 1. Align test columns\n# -----------------------\ncols_to_keep = X_train.columns\nX_test = test_last[cols_to_keep].copy()\n\nfor c in cat_features:\n    X_test[c] = X_test[c].astype(\"category\")\n\n# -----------------------\n# 2. Predict on test\n# -----------------------\ntest_last['prediction'] = best_xgb.predict_proba(X_test)[:, 1]\n\n# -----------------------\n# 3. Merge with sample submission (correct order)\n# -----------------------\nsubmission = ss_df[['customer_ID']].merge(\n    test_last[['customer_ID', 'prediction']],\n    on='customer_ID',\n    how='left'\n)\n\n# -----------------------\n# 4. Safety\n# -----------------------\nsubmission['prediction'].fillna(0.0, inplace=True)\n\n# -----------------------\n# 5. Save file\n# -----------------------\nsubmission.to_csv(\"submission.csv\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T19:15:46.645542Z","iopub.execute_input":"2025-12-19T19:15:46.646148Z","iopub.status.idle":"2025-12-19T19:15:46.659203Z","shell.execute_reply.started":"2025-12-19T19:15:46.646123Z","shell.execute_reply":"2025-12-19T19:15:46.658106Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Aggregating all the columns for a customer","metadata":{}},{"cell_type":"code","source":"train_df['customer_ID']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-21T13:28:33.104544Z","iopub.execute_input":"2025-12-21T13:28:33.104834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = train_df.copy()\nuninclude_cols = [\"customer_ID\", \"S_2\"]\n\n# -----------------------------------\n# Explicit categorical columns\n# -----------------------------------\ncat_cols = [\n    'B_30', 'B_38', 'D_114', 'D_116', 'D_117', \n    'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68'\n]\n\n# Everything else except customer_ID, S_2, categorical = numeric\nnum_cols = [c for c in df.columns if c not in cat_cols and c not in uninclude_cols]\n\n# Convert numericals safely\nfor c in num_cols:\n    df[c] = pd.to_numeric(df[c], errors='coerce')\n\n# -----------------------------------\n# Aggregation specs\n# -----------------------------------\nnum_aggs = ['mean', 'std', 'min', 'max', 'last']\n\ncat_aggs = [\n    ('last', 'last'),\n    ('mode', lambda x: x.mode().iloc[0] if len(x.mode()) > 0 else np.nan),\n    ('nunique', 'nunique')\n]\n\n# Build agg dict\nagg_dict = {}\n\nfor col in num_cols:\n    agg_dict[col] = num_aggs\n\nfor col in cat_cols:\n    agg_dict[col] = cat_aggs\n\n# -----------------------------------\n# Groupby aggregation\n# -----------------------------------\ndf_sorted = df.sort_values([\"customer_ID\", \"S_2\"])\ndf_agg = df_sorted.groupby(\"customer_ID\").agg(agg_dict)\n\n\n\n# flatten names\ndf_agg.columns = [f\"{c[0]}_{c[1]}\" for c in df_agg.columns]\ndf_agg.reset_index(inplace=True)\n\n# need to make category datatype for the appropriate aggregated columns\nagg_cat_cols = [\n    c for c in df_agg.columns\n    if any(cat in c for cat in cat_cols)\n    and not c.endswith(\"_nunique\")\n]\n\nfor col in agg_cat_cols:\n    df_agg[col] = df_agg[col].astype(\"category\")\n\nprint(df_agg.shape)\nprint([c for c in df_agg.columns if \"mode\" in c])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T12:25:50.005941Z","iopub.execute_input":"2025-12-20T12:25:50.006516Z","iopub.status.idle":"2025-12-20T12:26:54.358257Z","shell.execute_reply.started":"2025-12-20T12:25:50.006493Z","shell.execute_reply":"2025-12-20T12:26:54.357363Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Finding the importance of feature before dropping the features with a lot of missing values","metadata":{}},{"cell_type":"code","source":"df_agg_merged = df_agg.merge(train_labels, on = \"customer_ID\", how =\"left\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T12:26:54.359393Z","iopub.execute_input":"2025-12-20T12:26:54.359691Z","iopub.status.idle":"2025-12-20T12:26:54.625621Z","shell.execute_reply.started":"2025-12-20T12:26:54.359663Z","shell.execute_reply":"2025-12-20T12:26:54.625001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train validation split\nX = df_agg_merged.drop(columns=[\"target\", \"customer_ID\"])\ny = df_agg_merged [\"target\"]\n\nX_train, X_valid, y_train, y_valid = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T12:53:59.641821Z","iopub.execute_input":"2025-12-20T12:53:59.642568Z","iopub.status.idle":"2025-12-20T12:53:59.942273Z","shell.execute_reply.started":"2025-12-20T12:53:59.642539Z","shell.execute_reply":"2025-12-20T12:53:59.941465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_frac = X_train.isna().mean().sort_values(ascending=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T12:52:10.677963Z","iopub.execute_input":"2025-12-20T12:52:10.678645Z","iopub.status.idle":"2025-12-20T12:52:10.770307Z","shell.execute_reply.started":"2025-12-20T12:52:10.678617Z","shell.execute_reply":"2025-12-20T12:52:10.769765Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#This model is only to find the importance of features\n\nbase_params = {\n    \"objective\": \"binary:logistic\",\n    \"tree_method\": \"hist\",\n    \"missing\": np.nan,\n    \"enable_categorical\": True,\n    \"device\": \"cuda\",\n\n    \"learning_rate\": 0.05,\n    \"max_depth\": 8,\n    \"subsample\": 0.8,\n    \"colsample_bytree\": 0.8,\n    \"eval_metric\": \"auc\"\n}\n\nbase_model = xgb.XGBClassifier(\n    **base_params,\n    n_estimators=500\n)\n\nbase_model.fit(\n    X_train, y_train,\n    eval_set=[(X_valid, y_valid)],\n    early_stopping_rounds=50,\n    verbose=True\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T12:54:07.826288Z","iopub.execute_input":"2025-12-20T12:54:07.826638Z","iopub.status.idle":"2025-12-20T12:55:01.947604Z","shell.execute_reply.started":"2025-12-20T12:54:07.826608Z","shell.execute_reply":"2025-12-20T12:55:01.946917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"importance = (\n    pd.Series(base_model.get_booster().get_score(importance_type=\"gain\"))\n    .rename(\"gain\")\n)\n\nimportance = importance / importance.sum()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T12:55:01.948820Z","iopub.execute_input":"2025-12-20T12:55:01.949044Z","iopub.status.idle":"2025-12-20T12:55:01.956285Z","shell.execute_reply.started":"2025-12-20T12:55:01.949027Z","shell.execute_reply":"2025-12-20T12:55:01.955668Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"feature_stats = pd.DataFrame({\n    \"missing_frac\": missing_frac,\n    \"gain\": importance\n}).fillna(0)\n\nfeature_stats.sort_values(\n    [\"gain\", \"missing_frac\"],\n    ascending=[False, True],\n    inplace=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T12:55:01.956973Z","iopub.execute_input":"2025-12-20T12:55:01.957665Z","iopub.status.idle":"2025-12-20T12:55:01.976094Z","shell.execute_reply.started":"2025-12-20T12:55:01.957644Z","shell.execute_reply":"2025-12-20T12:55:01.975505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# DROP_MISSING = 0.9\n# MIN_GAIN = 0.001  # 0.1%\n\n# features_to_keep = feature_stats[\n#     ~(\n#         (feature_stats[\"missing_frac\"] > DROP_MISSING) &\n#         (feature_stats[\"gain\"] < MIN_GAIN)\n#     )\n# ].index.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T20:47:21.304814Z","iopub.status.idle":"2025-12-19T20:47:21.305085Z","shell.execute_reply.started":"2025-12-19T20:47:21.304977Z","shell.execute_reply":"2025-12-19T20:47:21.304988Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# X_train = X_train[features_to_keep]\n# X_valid = X_valid[features_to_keep]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T20:40:43.982670Z","iopub.execute_input":"2025-12-19T20:40:43.983300Z","iopub.status.idle":"2025-12-19T20:40:44.066519Z","shell.execute_reply.started":"2025-12-19T20:40:43.983276Z","shell.execute_reply":"2025-12-19T20:40:44.065894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# feature_stats must already exist\n# Columns expected:\n#   - missing_frac\n#   - gain\n\nfeature_stats = feature_stats.copy()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T12:55:01.977740Z","iopub.execute_input":"2025-12-20T12:55:01.978005Z","iopub.status.idle":"2025-12-20T12:55:01.992439Z","shell.execute_reply.started":"2025-12-20T12:55:01.977984Z","shell.execute_reply":"2025-12-20T12:55:01.991934Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Finding the best values of DROP_MISSING and MIN_GAIN using hypertuning","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from optuna.exceptions import TrialPruned\nimport gc, torch, tqdm\n\ndef expand_cat_features(cat_cols, all_columns):\n    expanded = []\n    for c in cat_cols:\n        expanded += [col for col in all_columns if col.startswith(f\"{c}_\")]\n    return expanded\n\nfrom optuna.exceptions import TrialPruned\n\ndef objective(trial):\n\n    drop_missing = trial.suggest_float(\"drop_missing\", 0.6, 0.95)\n    min_gain = trial.suggest_float(\"min_gain\", 1e-5, 1e-2, log=True)\n\n    # Feature selection\n    selected_features = feature_stats[\n        ~(\n            (feature_stats[\"missing_frac\"] > drop_missing) &\n            (feature_stats[\"gain\"] < min_gain)\n        )\n    ].index.tolist()\n\n    # Expand categorical features properly\n    cat_features_expanded = expand_cat_features(cat_cols, X_train.columns)\n\n    # Force keep expanded categorical features\n    selected_features = list(set(selected_features).union(cat_features_expanded))\n\n    if len(selected_features) < 50:\n        raise TrialPruned()\n\n    X_tr = X_train[selected_features]\n    X_va = X_valid[selected_features]\n    \n\n    model = xgb.XGBClassifier(\n        objective=\"binary:logistic\",\n        tree_method=\"hist\",\n        device=\"cuda\",\n        enable_categorical=True,\n        learning_rate=0.05,\n        max_depth=24,\n        subsample=0.8,\n        colsample_bytree=0.8,\n        n_estimators=100,\n        eval_metric=\"auc\"\n    )\n\n    model.fit(\n        X_tr, y_train,\n        eval_set=[(X_va, y_valid)],\n        early_stopping_rounds=20,\n        verbose=True\n    )\n\n    preds = model.predict_proba(X_va)[:, 1]\n\n    # -------------------------\n    # GPU CLEANUP (CRITICAL)\n    # -------------------------\n    del model\n    gc.collect()\n    torch.cuda.empty_cache()\n    y_valid_df = pd.DataFrame(y_valid, columns = [\"target\"])\n    preds_df = pd.DataFrame(preds, columns = [\"prediction\"])\n    return amex_metric(y_valid_df, preds_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T13:01:58.954088Z","iopub.execute_input":"2025-12-20T13:01:58.954593Z","iopub.status.idle":"2025-12-20T13:01:58.962486Z","shell.execute_reply.started":"2025-12-20T13:01:58.954567Z","shell.execute_reply":"2025-12-20T13:01:58.961767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\nimport torch\n\n# Remove references to Optuna internals\ndel study\n\n# Clean Python memory\ngc.collect()\n\n# Release CUDA memory\ntorch.cuda.empty_cache()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T13:02:02.210025Z","iopub.execute_input":"2025-12-20T13:02:02.210885Z","iopub.status.idle":"2025-12-20T13:02:02.384671Z","shell.execute_reply.started":"2025-12-20T13:02:02.210848Z","shell.execute_reply":"2025-12-20T13:02:02.383931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import optuna, tqdm, time\n\nstudy = optuna.create_study(direction=\"maximize\")\n\nstudy.optimize(\n    objective,\n    n_trials=10,\n    show_progress_bar=True\n)\n\nbest_drop_missing = study.best_params[\"drop_missing\"]\nbest_min_gain = study.best_params[\"min_gain\"]\n\nprint(best_drop_missing, best_min_gain)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T13:02:05.433156Z","iopub.execute_input":"2025-12-20T13:02:05.433987Z","iopub.status.idle":"2025-12-20T13:19:17.868194Z","shell.execute_reply.started":"2025-12-20T13:02:05.433955Z","shell.execute_reply":"2025-12-20T13:19:17.867526Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Build mapping once\ncat_feature_map = {}\n\nfor c in cat_cols:\n    cat_feature_map[c] = [\n        col for col in X_train.columns\n        if col.startswith(f\"{c}_\")\n    ]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T13:31:48.646458Z","iopub.execute_input":"2025-12-20T13:31:48.646787Z","iopub.status.idle":"2025-12-20T13:31:48.653466Z","shell.execute_reply.started":"2025-12-20T13:31:48.646754Z","shell.execute_reply":"2025-12-20T13:31:48.652515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"forced_cat_features = []\n\nfor c in cat_cols:\n    forced_cat_features.extend(cat_feature_map.get(c, []))\n\nselected_features = list(set(selected_features).union(forced_cat_features))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T13:32:04.886131Z","iopub.execute_input":"2025-12-20T13:32:04.886413Z","iopub.status.idle":"2025-12-20T13:32:04.890606Z","shell.execute_reply.started":"2025-12-20T13:32:04.886392Z","shell.execute_reply":"2025-12-20T13:32:04.889995Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_features = [\n    f for f in selected_features\n    if f in X_train.columns\n]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T13:32:23.306778Z","iopub.execute_input":"2025-12-20T13:32:23.307471Z","iopub.status.idle":"2025-12-20T13:32:23.311561Z","shell.execute_reply.started":"2025-12-20T13:32:23.307447Z","shell.execute_reply":"2025-12-20T13:32:23.310989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_tr = X_train[selected_features]\nX_va = X_valid[selected_features]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T13:32:55.827717Z","iopub.execute_input":"2025-12-20T13:32:55.828395Z","iopub.status.idle":"2025-12-20T13:32:55.842746Z","shell.execute_reply.started":"2025-12-20T13:32:55.828372Z","shell.execute_reply":"2025-12-20T13:32:55.841781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_features = feature_stats[\n    ~(\n        (feature_stats[\"missing_frac\"] > best_drop_missing) &\n        (feature_stats[\"gain\"] < best_min_gain)\n    )\n].index.tolist()\n\nselected_features = list(set(selected_features).union(cat_cols))\n\nX_tr = X_train[selected_features]\nX_va = X_valid[selected_features]\n\nprint(\"Final feature count:\", len(selected_features))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T13:32:36.456646Z","iopub.execute_input":"2025-12-20T13:32:36.457392Z","iopub.status.idle":"2025-12-20T13:32:36.474658Z","shell.execute_reply.started":"2025-12-20T13:32:36.457364Z","shell.execute_reply":"2025-12-20T13:32:36.473754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# SHAP based keep/drop logic\nref_model = xgb.XGBClassifier(\n    objective=\"binary:logistic\",\n    tree_method=\"hist\",\n    device=\"cuda\",\n    enable_categorical=True,\n    learning_rate=0.05,\n    max_depth=8,\n    subsample=0.8,\n    colsample_bytree=0.8,\n    n_estimators=300,\n    eval_metric=\"auc\"\n)\n\nref_model.fit(\n    X_train[selected_features],\n    y_train,\n    eval_set=[(X_valid[selected_features], y_valid)],\n    early_stopping_rounds=50,\n    verbose=False\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T12:10:00.287696Z","iopub.execute_input":"2025-12-20T12:10:00.288429Z","iopub.status.idle":"2025-12-20T12:10:00.299105Z","shell.execute_reply.started":"2025-12-20T12:10:00.288402Z","shell.execute_reply":"2025-12-20T12:10:00.298104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import optuna\n\ntrial_history = []\n\ndef objective(trial):\n\n    params = {\n        \"objective\": \"binary:logistic\",\n        \"tree_method\": \"hist\",\n        \"device\": \"cuda\",\n        \"eval_metric\": \"auc\",\n        \"missing\": np.nan,\n        \"enable_categorical\": True,\n        \"n_estimators\": 500,\n\n        # ---- Parameters to tune ----\n        \"max_depth\": trial.suggest_int(\"max_depth\", 8, 80, step=8),\n        \"min_child_weight\": trial.suggest_int(\"min_child_weight\", 10, 100, step=10),\n        \"learning_rate\": trial.suggest_categorical(\"learning_rate\", [0.05, 0.1]),\n        \"subsample\": trial.suggest_categorical(\"subsample\", [0.3, 0.4, 0.5, 0.8, 1.0]),\n        \"colsample_bytree\": trial.suggest_categorical(\"colsample_bytree\", [0.5, 0.8, 1.0]),\n        \n\n        # ---- Regularization tuning ----\n        \"lambda\": trial.suggest_float(\"lambda\", 1e-3, 10, log=True),\n        \"alpha\": trial.suggest_float(\"alpha\", 1e-3, 5, log=True),\n        \"gamma\": trial.suggest_float(\"gamma\", 0, 10),\n    }\n\n    model = xgb.XGBClassifier(**params)\n\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_valid, y_valid)],\n        early_stopping_rounds=30,\n        verbose=False\n    )\n\n    # predictions\n    preds = model.predict_proba(X_valid)[:, 1]\n\n    # Compute AMEX\n    y_true_df = pd.DataFrame({'target': y_valid}).reset_index(drop=True)\n    y_pred_df = pd.DataFrame({'prediction': preds}).reset_index(drop=True)\n\n    score = amex_metric(y_true_df, y_pred_df)\n    \n    # ----------------------------\n    # SAVE TRIAL RESULTS SAFELY\n    # ----------------------------\n    trial_history.append({\n        \"params\": params.copy(),              # <-- IMPORTANT FIX\n        \"best_iteration\": model.best_iteration,\n        \"amex_score\": score\n    })\n    \n    return score  # Optuna will maximize the AMEX metric\n\n\n# Run study\nstudy = optuna.create_study(\n    study_name=\"xgb_amex_aggregated\",\n    direction=\"maximize\",\n    storage=\"sqlite:///optuna_xgb.db\",\n    load_if_exists=True\n)\n\nstudy.optimize(objective, n_trials=50)  # adds 300 more\n\n# Convert to DataFrame\ntrial_df = pd.DataFrame(trial_history)\n\n# Flatten params dict into columns (optional but very useful!)\ntrial_df = trial_df.join(trial_df[\"params\"].apply(pd.Series)).drop(columns=[\"params\"])\n\ntrial_df.sort_values(\"amex_score\", ascending=False, inplace=True)\n\nprint(\"Best AMEX:\", study.best_trial.value)\nprint(\"Best Params:\", study.best_trial.params)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T20:22:14.337766Z","iopub.execute_input":"2025-12-19T20:22:14.338025Z","iopub.status.idle":"2025-12-19T20:22:14.993855Z","shell.execute_reply.started":"2025-12-19T20:22:14.338007Z","shell.execute_reply":"2025-12-19T20:22:14.992840Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_params = {\n    \"objective\": \"binary:logistic\",\n    \"tree_method\": \"hist\",\n    \"missing\": np.nan,\n    \"enable_categorical\": True,\n    \"device\": \"cuda\",\n\n    \"max_depth\": 80,\n    \"min_child_weight\": 40,\n    \"learning_rate\": 0.05,\n    \"subsample\": 0.8,\n    \"colsample_bytree\": 0.5,\n\n    \"lambda\": 0.007670118758441345,\n    \"alpha\": 0.09157175965761555,\n    \"gamma\": 1.028865262384579,\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T20:16:12.701547Z","iopub.execute_input":"2025-12-19T20:16:12.702329Z","iopub.status.idle":"2025-12-19T20:16:12.706493Z","shell.execute_reply.started":"2025-12-19T20:16:12.702304Z","shell.execute_reply":"2025-12-19T20:16:12.705632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_model = xgb.XGBClassifier(\n    **best_params,\n    n_estimators=500  # intentionally large\n)\n\nfinal_model.fit(\n    X_train,\n    y_train,\n    eval_set=[(X_valid, y_valid)],\n    eval_metric=amex_metric_xgb,\n    verbose=True\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-19T20:16:15.235039Z","iopub.execute_input":"2025-12-19T20:16:15.235710Z","iopub.status.idle":"2025-12-19T20:17:03.022454Z","shell.execute_reply.started":"2025-12-19T20:16:15.235685Z","shell.execute_reply":"2025-12-19T20:17:03.021831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop the columns with missing fraction > 0.5\ncols_to_drop = df_agg.columns[df_agg.isna().mean() > 0.83].tolist()\n# Create the modified dataframe\ndf_agg= df_agg.drop(columns=cols_to_drop)\n\n#denoise\nimport warnings\nwarnings.filterwarnings(\"ignore\")\ndf_agg = denoise_numeric(df_agg)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T13:34:42.171316Z","iopub.execute_input":"2025-12-20T13:34:42.171964Z","iopub.status.idle":"2025-12-20T13:34:45.428573Z","shell.execute_reply.started":"2025-12-20T13:34:42.171940Z","shell.execute_reply":"2025-12-20T13:34:45.427964Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"agg_cat_cols = [col for col in df_agg.columns if any(col.startswith(cat) for cat in cat_cols)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T13:35:03.877892Z","iopub.execute_input":"2025-12-20T13:35:03.878747Z","iopub.status.idle":"2025-12-20T13:35:03.884501Z","shell.execute_reply.started":"2025-12-20T13:35:03.878690Z","shell.execute_reply":"2025-12-20T13:35:03.883627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_agg_merged","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T13:35:06.022661Z","iopub.execute_input":"2025-12-20T13:35:06.023238Z","iopub.status.idle":"2025-12-20T13:35:06.070307Z","shell.execute_reply.started":"2025-12-20T13:35:06.023212Z","shell.execute_reply":"2025-12-20T13:35:06.069470Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for c in agg_cat_cols:\n    df_agg_merged[c] = df_agg_merged[c].astype('category')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T13:35:09.434559Z","iopub.execute_input":"2025-12-20T13:35:09.434899Z","iopub.status.idle":"2025-12-20T13:35:09.455065Z","shell.execute_reply.started":"2025-12-20T13:35:09.434875Z","shell.execute_reply":"2025-12-20T13:35:09.454282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df_agg_merged.drop(['customer_ID', 'target'], axis=1)\ny = df_agg_merged['target']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T13:35:10.550656Z","iopub.execute_input":"2025-12-20T13:35:10.551462Z","iopub.status.idle":"2025-12-20T13:35:10.614439Z","shell.execute_reply.started":"2025-12-20T13:35:10.551435Z","shell.execute_reply":"2025-12-20T13:35:10.613847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_valid, y_train, y_valid = train_test_split(X, y, random_state=42, stratify = y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T13:35:12.271000Z","iopub.execute_input":"2025-12-20T13:35:12.271291Z","iopub.status.idle":"2025-12-20T13:35:12.509078Z","shell.execute_reply.started":"2025-12-20T13:35:12.271269Z","shell.execute_reply":"2025-12-20T13:35:12.508166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import optuna\n\ntrial_history = []\n\ndef objective(trial):\n\n    params = {\n        \"objective\": \"binary:logistic\",\n        \"tree_method\": \"hist\",\n        \"eval_metric\": \"auc\",\n        \"missing\": np.nan,\n        \"enable_categorical\": True,\n\n        # ---- Parameters to tune ----\n        \"max_depth\": trial.suggest_int(\"max_depth\", 8, 48, step=8),\n        \"min_child_weight\": trial.suggest_int(\"min_child_weight\", 20, 50, step=10),\n        \"learning_rate\": trial.suggest_categorical(\"learning_rate\", [0.05, 0.1]),\n        \"subsample\": trial.suggest_categorical(\"subsample\", [0.5, 0.8, 1.0]),\n        \"colsample_bytree\": trial.suggest_categorical(\"colsample_bytree\", [0.5, 0.8, 1.0]),\n        \"n_estimators\": trial.suggest_categorical(\"n_estimators\", [200, 400]),\n\n        # ---- Regularization tuning ----\n        \"lambda\": trial.suggest_float(\"lambda\", 1e-3, 10, log=True),\n        \"alpha\": trial.suggest_float(\"alpha\", 1e-3, 5, log=True),\n        \"gamma\": trial.suggest_float(\"gamma\", 0, 10),\n    }\n\n    model = xgb.XGBClassifier(**params)\n\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_valid, y_valid)],\n        early_stopping_rounds=30,\n        verbose=False\n    )\n\n    # predictions\n    preds = model.predict_proba(X_valid)[:, 1]\n\n    # Compute AMEX\n    y_true_df = pd.DataFrame({'target': y_valid}).reset_index(drop=True)\n    y_pred_df = pd.DataFrame({'prediction': preds}).reset_index(drop=True)\n\n    score = amex_metric(y_true_df, y_pred_df)\n    \n    # ----------------------------\n    # SAVE TRIAL RESULTS SAFELY\n    # ----------------------------\n    trial_history.append({\n        \"params\": params.copy(),              # <-- IMPORTANT FIX\n        \"best_iteration\": model.best_iteration,\n        \"amex_score\": score\n    })\n    \n    return score  # Optuna will maximize the AMEX metric\n\n\n# Run study\nstudy = optuna.create_study(direction=\"maximize\")\nstudy.optimize(objective, n_trials=50)\n\n# Convert to DataFrame\ntrial_df = pd.DataFrame(trial_history)\n\n# Flatten params dict into columns (optional but very useful!)\ntrial_df = trial_df.join(trial_df[\"params\"].apply(pd.Series)).drop(columns=[\"params\"])\n\ntrial_df.sort_values(\"amex_score\", ascending=False, inplace=True)\n\nprint(\"Best AMEX:\", study.best_trial.value)\nprint(\"Best Params:\", study.best_trial.params)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T13:35:14.023244Z","iopub.execute_input":"2025-12-20T13:35:14.023790Z","iopub.status.idle":"2025-12-20T14:50:16.286445Z","shell.execute_reply.started":"2025-12-20T13:35:14.023766Z","shell.execute_reply":"2025-12-20T14:50:16.285524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nfor col in ['max_depth', 'min_child_weight', 'learning_rate', 'subsample', \n            'colsample_bytree', 'n_estimators', 'lambda', 'alpha', 'gamma']:\n    \n    plt.figure(figsize=(6,4))\n    sns.scatterplot(data=trial_df, x=col, y='amex_score')\n    plt.title(f'{col} vs AMEX')\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-19T12:11:44.743801Z","iopub.execute_input":"2025-11-19T12:11:44.744158Z","iopub.status.idle":"2025-11-19T12:11:46.960625Z","shell.execute_reply.started":"2025-11-19T12:11:44.744134Z","shell.execute_reply":"2025-11-19T12:11:46.959614Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T15:06:37.856047Z","iopub.execute_input":"2025-12-20T15:06:37.856489Z","iopub.status.idle":"2025-12-20T15:06:37.861371Z","shell.execute_reply.started":"2025-12-20T15:06:37.856456Z","shell.execute_reply":"2025-12-20T15:06:37.860509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_params = {\n        \"objective\": \"binary:logistic\",\n        \"tree_method\": \"hist\",\n        \"missing\": np.nan,\n        \"enable_categorical\": True,\n\n        # ---- tuned params ----\n        \"max_depth\": 48,\n        \"min_child_weight\": 30,\n        \"learning_rate\":0.05,\n        \"subsample\": 1.0,\n        \"colsample_bytree\": 0.8,\n        \"n_estimators\": 10000,\n        \"lambda\": 1.2899531383219924, \n        \"alpha\": 1.3441815317991856, \n        \"gamma\": 3.066108845170049,\n        \"eval_metric\": \"auc\"\n    }\n\nbest_xgb = xgb.XGBClassifier(**best_params)\n\nbest_xgb.fit(\n        X_train, y_train,\n        eval_set=[(X_valid, y_valid)],\n        early_stopping_rounds=150,\n\n        verbose=True\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T15:08:41.008217Z","iopub.execute_input":"2025-12-20T15:08:41.008934Z","iopub.status.idle":"2025-12-20T15:10:58.560034Z","shell.execute_reply.started":"2025-12-20T15:08:41.008908Z","shell.execute_reply":"2025-12-20T15:10:58.559395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T15:11:30.818644Z","iopub.execute_input":"2025-12-20T15:11:30.819340Z","iopub.status.idle":"2025-12-20T15:11:30.822673Z","shell.execute_reply.started":"2025-12-20T15:11:30.819317Z","shell.execute_reply":"2025-12-20T15:11:30.822120Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics import f1_score\n\nthresholds = np.linspace(0.0, 1.0, 200)\nscores = []\n\nfor t in thresholds:\n    y_pred_label = (preds >= t).astype(int)\n    scores.append(f1_score(y_true, y_pred_label))\n\nbest_idx = np.argmax(scores)\nbest_threshold = thresholds[best_idx]\nbest_f1 = scores[best_idx]\n\nprint(\"Best threshold:\", best_threshold)\nprint(\"Best F1:\", best_f1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T15:17:04.938426Z","iopub.execute_input":"2025-12-20T15:17:04.939092Z","iopub.status.idle":"2025-12-20T15:17:05.993806Z","shell.execute_reply.started":"2025-12-20T15:17:04.939064Z","shell.execute_reply":"2025-12-20T15:17:05.992970Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import precision_recall_curve\nimport matplotlib.pyplot as plt\n\nprecision, recall, pr_thresholds = precision_recall_curve(y_true, preds)\n\nplt.figure(figsize=(7,5))\nplt.plot(recall, precision)\nplt.xlabel(\"Recall\")\nplt.ylabel(\"Precision\")\nplt.title(\"Precision–Recall Curve\")\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T15:17:10.289641Z","iopub.execute_input":"2025-12-20T15:17:10.290219Z","iopub.status.idle":"2025-12-20T15:17:10.475510Z","shell.execute_reply.started":"2025-12-20T15:17:10.290191Z","shell.execute_reply":"2025-12-20T15:17:10.474904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\n\nfpr, tpr, roc_thresholds = roc_curve(y_true, preds)\nroc_auc = auc(fpr, tpr)\n\nplt.figure(figsize=(7,5))\nplt.plot(fpr, tpr, label=f\"AUC = {roc_auc:.4f}\")\nplt.xlabel(\"False Positive Rate\")\nplt.ylabel(\"True Positive Rate\")\nplt.title(\"ROC Curve\")\nplt.legend()\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T15:17:14.514403Z","iopub.execute_input":"2025-12-20T15:17:14.515010Z","iopub.status.idle":"2025-12-20T15:17:14.678108Z","shell.execute_reply.started":"2025-12-20T15:17:14.514987Z","shell.execute_reply":"2025-12-20T15:17:14.677499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import precision_score, recall_score, f1_score\n\nprecisions = []\nrecalls = []\nf1s = []\n\nfor t in thresholds:\n    y_pred_label = (preds >= t).astype(int)\n    precisions.append(precision_score(y_true, y_pred_label))\n    recalls.append(recall_score(y_true, y_pred_label))\n    f1s.append(f1_score(y_true, y_pred_label))\n\nplt.figure(figsize=(8,6))\nplt.plot(thresholds, precisions, label=\"Precision\")\nplt.plot(thresholds, recalls, label=\"Recall\")\nplt.plot(thresholds, f1s, label=\"F1 Score\")\nplt.axvline(best_threshold, color='black', linestyle='--', label=f\"Best t={best_threshold:.3f}\")\nplt.xlabel(\"Threshold\")\nplt.ylabel(\"Score\")\nplt.title(\"Threshold Tuning Curve\")\nplt.legend()\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T15:17:19.098461Z","iopub.execute_input":"2025-12-20T15:17:19.098791Z","iopub.status.idle":"2025-12-20T15:17:22.381440Z","shell.execute_reply.started":"2025-12-20T15:17:19.098767Z","shell.execute_reply":"2025-12-20T15:17:22.380699Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\n\ny_pred_best = (preds >= best_threshold).astype(int)\n\ncm = confusion_matrix(y_true, y_pred_best)\n\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm)\ndisp.plot(cmap=\"Blues\")\nplt.title(f\"Confusion Matrix (Best threshold = {best_threshold:.3f})\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T15:17:25.929926Z","iopub.execute_input":"2025-12-20T15:17:25.930404Z","iopub.status.idle":"2025-12-20T15:17:26.112980Z","shell.execute_reply.started":"2025-12-20T15:17:25.930379Z","shell.execute_reply":"2025-12-20T15:17:26.112137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\n# Convert probabilities to labels\ny_pred_best = (preds >= best_threshold).astype(int)\n\nacc = accuracy_score(y_true, y_pred_best)\nprec = precision_score(y_true, y_pred_best)\nrec = recall_score(y_true, y_pred_best)\nf1 = f1_score(y_true, y_pred_best)\n\nprint(f\"Best Threshold: {best_threshold:.4f}\")\nprint(f\"Accuracy:  {acc:.4f}\")\nprint(f\"Precision: {prec:.4f}\")\nprint(f\"Recall:    {rec:.4f}\")\nprint(f\"F1 Score:  {f1:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T15:17:34.696097Z","iopub.execute_input":"2025-12-20T15:17:34.696622Z","iopub.status.idle":"2025-12-20T15:17:34.720383Z","shell.execute_reply.started":"2025-12-20T15:17:34.696600Z","shell.execute_reply":"2025-12-20T15:17:34.719806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = pd.DataFrame({'prediction': preds}).reset_index(drop=True)\ny_true = pd.DataFrame({'target': y_valid}).reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T15:21:18.931816Z","iopub.execute_input":"2025-12-20T15:21:18.932081Z","iopub.status.idle":"2025-12-20T15:21:18.937223Z","shell.execute_reply.started":"2025-12-20T15:21:18.932061Z","shell.execute_reply":"2025-12-20T15:21:18.936581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"g = normalized_weighted_gini(y_true, y_pred)\nd = top_four_percent_captured(y_true, y_pred)\n\nprint(g, d, 0.5 * (g + d))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T15:21:20.426948Z","iopub.execute_input":"2025-12-20T15:21:20.427212Z","iopub.status.idle":"2025-12-20T15:21:20.452657Z","shell.execute_reply.started":"2025-12-20T15:21:20.427192Z","shell.execute_reply":"2025-12-20T15:21:20.452040Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### When all the statements are aggregated\n- The best threshold for deciding whether it's gonna default or not is 0.3518\n- Accuracy = 0.8971\n- Precision = 0.7827\n- Recall = 0.8394\n- F1 Score = 0.8101\n- Amex Score = 0.7754\n- AUC = 0.9565\n- Normalized Weighted Gini = 0.9103\n- Percentage of total defaulters captured in Top Four Percent= 0.6378\n  \n  Note that these metrics are on validation data.","metadata":{}},{"cell_type":"markdown","source":"#### When only the last statement is considered\n- The best threshold for deciding whether it's gonna default or not is 0.3518\n- Accuracy = 0.8871\n- Precision = 0.7426\n- Recall = 0.8696\n- F1 Score = 0.8011\n- Amex Score = 0.7728\n- AUC = 0.9552\n- Normalized Weighted Gini = 0.9103\n- Percentage of total defaulters captured in Top Four Percent= 0.6354\n  \n  Note that these metrics are on validation data.","metadata":{}}]}