{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-07T21:36:47.264577Z","iopub.execute_input":"2025-10-07T21:36:47.264905Z","iopub.status.idle":"2025-10-07T21:36:47.272313Z","shell.execute_reply.started":"2025-10-07T21:36:47.264883Z","shell.execute_reply":"2025-10-07T21:36:47.271289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === STEP 1: CREATE A MANAGEABLE DATA SAMPLE ===\n\n# We'll use the file paths you confirmed in the previous cell.\nTRAIN_DATA_PATH = '/kaggle/input/amex-default-prediction/train_data.csv'\nTRAIN_LABELS_PATH = '/kaggle/input/amex-default-prediction/train_labels.csv'\n\n# Read a fraction of the main data file to avoid memory errors.\n# 1 million rows is a good starting point for EDA and pipeline development.\nprint(\"Reading a sample of the training data (1,000,000 rows)...\")\ndf_train_sample = pd.read_csv(TRAIN_DATA_PATH, nrows=1000000) # Could not read million rows so we start with ten thousand\n\n# Load the training labels. This file is small, so we can load it fully.\nprint(\"Reading training labels...\")\ndf_train_labels = pd.read_csv(TRAIN_LABELS_PATH)\n\n# --- Merge the sample with labels ---\n# Find out which unique customers are present in our 1M row sample.\ncustomers_in_sample = df_train_sample['customer_ID'].unique()\n\n# Filter the labels dataframe to only include the customers from our sample.\ndf_labels_sample = df_train_labels[df_train_labels['customer_ID'].isin(customers_in_sample)]\n\n# Merge the labels into our main training dataframe.\n# We use a left merge to ensure we keep all rows from our data sample and add the target column.\nprint(\"Merging data sample with labels...\")\ndf = pd.merge(df_train_sample, df_labels_sample, on='customer_ID', how='left')\n\n# --- Verification ---\nprint(\"\\n--- Verification of the Sampled DataFrame ---\")\nprint(f\"Shape of our final sample dataframe: {df.shape}\")\nprint(f\"Number of unique customers in the sample: {df['customer_ID'].nunique()}\")\nprint(f\"Number of rows with a null target value: {df['target'].isnull().sum()} (should be 0)\")\n\nprint(\"\\nSample DataFrame Head:\")\ndisplay(df.head())\n\nprint(\"\\nSample DataFrame Info:\")\ndf.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T21:36:51.522913Z","iopub.execute_input":"2025-10-07T21:36:51.523264Z","iopub.status.idle":"2025-10-07T21:38:01.736330Z","shell.execute_reply.started":"2025-10-07T21:36:51.523237Z","shell.execute_reply":"2025-10-07T21:38:01.735307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Set plot style for better visualization\nsns.set_style('whitegrid')\n\nprint(\"--- EDA Step 1: DataFrame Numerical Summary ---\")\n# .describe() gives us a statistical summary of the numerical columns.\ndisplay(df.describe())\n\nprint(\"\\n--- EDA Step 2: Missing Values Analysis ---\")\nmissing_values_perc = (df.isnull().sum() / len(df)) * 100\nprint(\"Percentage of missing values for each column (Top 20):\")\n# We display the top 20 columns with the most missing values.\ndisplay(missing_values_perc.sort_values(ascending=False).head(20))\n\nprint(\"\\n--- EDA Step 3: Target Variable Imbalance Check ---\")\n# IMPORTANT: Since each customer has multiple rows, we should check the target distribution\n# on a per-customer basis to get the true picture. We do this by dropping duplicates.\nplt.figure(figsize=(8, 5))\nsns.countplot(x='target', data=df.drop_duplicates(subset=['customer_ID']))\nplt.title('Distribution of Target Variable (Per Unique Customer)')\nplt.xlabel('Default (1) vs. No Default (0)')\nplt.ylabel('Number of Unique Customers')\n# Adding percentage labels\nunique_customers = df.drop_duplicates(subset=['customer_ID'])\ntotal = len(unique_customers)\nfor p in plt.gca().patches:\n    height = p.get_height()\n    plt.gca().text(p.get_x() + p.get_width()/2., height + 5, f'{100*height/total:.2f}%', ha='center')\nplt.show()\n\n\nprint(\"\\n--- EDA Step 4: Histograms for Key Numerical Features ---\")\n# Let's look at the distribution of a Payment, a Balance, and a Delinquency variable.\nkey_num_features = ['P_2', 'B_1', 'D_39']\ndf[key_num_features].hist(bins=50, figsize=(18, 5), layout=(1, 3))\nplt.suptitle('Distribution of Key Numerical Features', size=16, y=1.02)\nplt.show()\n\nprint(\"\\n--- EDA Step 5: Count Plots for Key Categorical Features ---\")\n# These were listed in the competition's data description.\nkey_cat_features = ['B_30', 'B_38', 'D_63', 'D_64', 'D_68']\nfor col in key_cat_features:\n    plt.figure(figsize=(10, 5))\n    sns.countplot(y=col, data=df, order=df[col].value_counts().index)\n    plt.title(f'Count Plot for {col}')\n    plt.xscale('log') # Use a log scale if counts are highly skewed\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T21:40:14.367759Z","iopub.execute_input":"2025-10-07T21:40:14.368684Z","iopub.status.idle":"2025-10-07T21:40:28.949779Z","shell.execute_reply.started":"2025-10-07T21:40:14.368653Z","shell.execute_reply":"2025-10-07T21:40:28.948875Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"EDA tell us that:\n* The Problem: We have multiple rows for each customer over time. We cannot feed this into a standard model.\n* The Features: They are anonymized numbers. Many have a lot of missing values, which might be informative.\n* The Target: The classes are imbalanced (0 is more common than 1), which confirms our choice of AUC as a metric and our need for models that can handle imbalance.\n! The single biggest challenge we identified is the time-series data. We need to convert the many rows per customer into a single row per customer. This is called aggregation.\n\nOur Hypothesis: \"A customer's most recent behavior (the last value of a feature) and the stability of their behavior (the standard deviation) are strong predictors of default.\"","metadata":{}},{"cell_type":"code","source":"# === STEP 3: FEATURE ENGINEERING (AGGREGATION) ===\n\nprint(\"Starting feature aggregation. This may take a minute...\")\n\n# Identify categorical and numerical features\n# Exclude customer_ID, S_2 (date), and target from the feature list\nfeatures = [col for col in df.columns if col not in ['customer_ID', 'S_2', 'target']]\n\n# D_63, D_64 are categorical as per data description. Let's find others.\ncat_features = [\n    \"B_30\", \"B_38\", \"D_114\", \"D_116\", \"D_117\", \"D_120\", \"D_126\",\n    \"D_63\", \"D_64\", \"D_66\", \"D_68\"\n]\n# Filter for cat_features that are actually in our dataframe\ncat_features = [f for f in cat_features if f in df.columns]\n\nnum_features = [f for f in features if f not in cat_features]\n\n# Define aggregations\n# For numerical features: get the mean, std, min, max, and last value\nnum_aggs = {col: ['mean', 'std', 'min', 'max', 'last'] for col in num_features}\n# For categorical features: count unique values, get the last value, and the most frequent\ncat_aggs = {col: ['count', 'last', 'nunique'] for col in cat_features}\n\n# Combine aggregation dictionaries\nall_aggs = {**num_aggs, **cat_aggs}\n\n# Perform the aggregation\ndf_agg = df.groupby('customer_ID').agg(all_aggs)\n\n# The aggregation creates multi-level columns (e.g., ('P_2', 'mean')).\n# Let's flatten them into single-level column names (e.g., 'P_2_mean').\ndf_agg.columns = ['_'.join(col).strip() for col in df_agg.columns.values]\ndf_agg.reset_index(inplace=True)\n\n# Merge with labels to get the target variable\n# We need a dataframe with one row per customer for the labels\ndf_labels_unique = df.groupby('customer_ID')['target'].first().reset_index()\ndf_agg = pd.merge(df_agg, df_labels_unique, on='customer_ID', how='left')\n\n\n# --- Verification ---\nprint(\"\\n--- Verification of the Aggregated DataFrame ---\")\nprint(f\"Shape of our new aggregated dataframe: {df_agg.shape}\")\nprint(f\"Number of unique customers (rows): {df_agg.shape[0]}\")\nprint(f\"Number of features: {df_agg.shape[1] - 2}\") # Subtract customer_ID and target\nprint(f\"Number of nulls in target variable: {df_agg['target'].isnull().sum()} (should be 0)\")\nprint(\"\\nAggregated DataFrame Head:\")\ndisplay(df_agg.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T21:40:49.188837Z","iopub.execute_input":"2025-10-07T21:40:49.189129Z","iopub.status.idle":"2025-10-07T21:40:59.200313Z","shell.execute_reply.started":"2025-10-07T21:40:49.189108Z","shell.execute_reply":"2025-10-07T21:40:59.199252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === STEP 4: CREATE A BASELINE MODEL ===\n\nimport gc\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import roc_auc_score\n\n# --- 1. Prepare Data ---\nprint(\"Preparing data for the baseline model...\")\n\n# Separate features (X) and target (y)\nX = df_agg.drop(columns=['customer_ID', 'target'])\ny = df_agg['target']\n\n# Identify numerical and categorical feature names from the new aggregated dataframe\n# Note: Categorical features now have suffixes like '_last', '_nunique', etc.\ncat_features_agg = [col for col in X.columns if X[col].dtype == 'object' or X[col].dtype == 'category']\nnum_features_agg = [col for col in X.columns if col not in cat_features_agg]\n\nprint(f\"Number of numerical features: {len(num_features_agg)}\")\nprint(f\"Number of categorical features: {len(cat_features_agg)}\")\n\n\n# --- 2. Create Preprocessing Pipelines ---\n# This pipeline handles missing values and scales the data.\n# For numerical data: impute with the median, then scale.\nnumeric_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='median')),\n    ('scaler', StandardScaler())\n])\n\n# For categorical data: impute with the most frequent value, then one-hot encode.\n# handle_unknown='ignore' is important for when the test set has categories not seen in training.\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\n# Use ColumnTransformer to apply different transformations to different columns\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numeric_transformer, num_features_agg),\n        ('cat', categorical_transformer, cat_features_agg)\n    ],\n    remainder='passthrough' # Keep other columns (if any)\n)\n\n# --- 3. Create the Full Model Pipeline ---\n# This chains the preprocessing steps with the final classifier.\nmodel_pipeline = Pipeline(steps=[\n    ('preprocessor', preprocessor),\n    ('classifier', LogisticRegression(random_state=42, solver='liblinear'))\n])\n\n# --- 4. Evaluate the Model with Cross-Validation ---\nprint(\"\\nStarting cross-validation for the baseline model...\")\nskf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\nauc_scores = []\n\nfor fold, (train_index, val_index) in enumerate(skf.split(X, y)):\n    X_train, X_val = X.iloc[train_index], X.iloc[val_index]\n    y_train, y_val = y.iloc[train_index], y.iloc[val_index]\n\n    # Fit the pipeline on the training data for this fold\n    model_pipeline.fit(X_train, y_train)\n\n    # Predict probabilities for the validation set\n    # We need predict_proba for ROC AUC score\n    y_pred_proba = model_pipeline.predict_proba(X_val)[:, 1]\n\n    # Calculate and store the AUC score\n    fold_auc = roc_auc_score(y_val, y_pred_proba)\n    auc_scores.append(fold_auc)\n    print(f\"Fold {fold+1} AUC: {fold_auc:.4f}\")\n\n    # Clean up memory\n    del X_train, X_val, y_train, y_val\n    gc.collect()\n\nprint(\"\\n--- Baseline Model Evaluation ---\")\nprint(f\"Average ROC AUC Score: {np.mean(auc_scores):.4f} (+/- {np.std(auc_scores):.4f})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T20:17:15.671464Z","iopub.execute_input":"2025-10-07T20:17:15.671731Z","iopub.status.idle":"2025-10-07T20:57:05.189820Z","shell.execute_reply.started":"2025-10-07T20:17:15.671712Z","shell.execute_reply":"2025-10-07T20:57:05.188502Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The simple Logistic Regression model did a great job, but it can't capture complex, non-linear interactions between features. A tree-based model like LightGBM is specifically designed for this and is the standard choice for winning Kaggle competitions with tabular data.","metadata":{}},{"cell_type":"code","source":"# === STEP 5 (CORRECTED): IMPROVE WITH AN ADVANCED MODEL (LIGHTGBM) ===\n\nimport lightgbm as lgb\nimport gc\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nimport warnings\n\n# Suppress LightGBM warnings about no positive gain, which are common on small datasets\nwarnings.filterwarnings(\"ignore\", message=\"No further splits with positive gain, best gain: -inf\")\n\n\n# --- We use the same X, y, and preprocessor definition from Step 4 ---\n# X = df_agg.drop(columns=['customer_ID', 'target'])\n# y = df_agg['target']\n# ... preprocessor = ColumnTransformer(...)\n\n# --- 1. Fit the Preprocessor ONCE on the entire dataset ---\nprint(\"Fitting the preprocessor on the entire training set...\")\n# This ensures the OneHotEncoder learns ALL possible categories from the start.\nX_transformed = preprocessor.fit_transform(X)\nprint(\"Data has been preprocessed.\")\n\n# Get the feature names after one-hot encoding\n# This is a bit complex but useful for the feature importance plot later\nohe_feature_names = preprocessor.named_transformers_['cat'].get_feature_names_out(cat_features_agg)\n# Combine with numerical feature names\nall_feature_names = num_features_agg + ohe_feature_names.tolist()\n\n\n# --- 2. Evaluate the Advanced Model with Cross-Validation ---\n# Now we will cross-validate on the PRE-TRANSFORMED data.\nprint(\"\\nStarting cross-validation for the LightGBM model...\")\nskf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\nlgbm_auc_scores = []\n\nfor fold, (train_index, val_index) in enumerate(skf.split(X_transformed, y)):\n    # Use the pre-transformed data and slice it\n    X_train, X_val = X_transformed[train_index], X_transformed[val_index]\n    y_train, y_val = y.iloc[train_index], y.iloc[val_index]\n\n    # Create and train the model directly (no pipeline needed inside the loop)\n    model = lgb.LGBMClassifier(random_state=42, is_unbalance=True)\n    model.fit(X_train, y_train, feature_name=all_feature_names) # Pass feature names for clarity\n\n    # Predict probabilities\n    y_pred_proba = model.predict_proba(X_val)[:, 1]\n\n    # Calculate and store AUC\n    fold_auc = roc_auc_score(y_val, y_pred_proba)\n    lgbm_auc_scores.append(fold_auc)\n    print(f\"Fold {fold+1} AUC: {fold_auc:.4f}\")\n\n    del X_train, X_val, y_train, y_val, model\n    gc.collect()\n\nprint(\"\\n--- LightGBM Model Evaluation ---\")\nbaseline_score = np.mean(auc_scores) # auc_scores is from your previous step\nlgbm_score = np.mean(lgbm_auc_scores)\nprint(f\"Baseline Logistic Regression Average AUC: {baseline_score:.4f}\")\nprint(f\"LightGBM Average ROC AUC Score: {lgbm_score:.4f} (+/- {np.std(lgbm_auc_scores):.4f})\")\n\nimprovement = lgbm_score - baseline_score\nprint(f\"\\nImprovement over baseline: {improvement:.4f}\")\nif improvement > 0:\n    print(\"Success! The LightGBM model outperformed the baseline.\")\nelse:\n    print(\"The LightGBM model did not outperform the baseline. Further tuning may be required.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T20:57:05.191156Z","iopub.execute_input":"2025-10-07T20:57:05.191442Z","iopub.status.idle":"2025-10-07T21:00:16.209532Z","shell.execute_reply.started":"2025-10-07T20:57:05.191422Z","shell.execute_reply":"2025-10-07T21:00:16.208641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === STEP 6: FEATURE IMPORTANCE AND STORYTELLING ===\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Your Prompt to the AI would be:\n# \"Our LightGBM model is the best. Please generate the feature importance plot for this model.\n# Also, help me draft a summary for our presentation.\"\n\n# --- Here is the generated code to execute ---\n\n# --- 1. Train a Final Model on ALL Data ---\n# To get the most reliable feature importances, we'll train one final\n# model on the entire preprocessed dataset (X_transformed, y).\nprint(\"Training final LightGBM model on all data to get feature importances...\")\n\nfinal_model = lgb.LGBMClassifier(random_state=42, is_unbalance=True)\n\n# We can reuse the preprocessed data and feature names from the previous step\nfinal_model.fit(X_transformed, y, feature_name=all_feature_names)\n\nprint(\"Final model trained.\")\n\n# --- 2. Generate and Plot Feature Importances ---\n# Create a dataframe of features and their importance scores\nfeature_importances = pd.DataFrame({\n    'feature': all_feature_names,\n    'importance': final_model.feature_importances_\n}).sort_values('importance', ascending=False)\n\n# Display the top 20 most important features\nprint(\"\\nTop 20 Most Important Features:\")\ndisplay(feature_importances.head(20))\n\n# Plot the top 20 features\nplt.figure(figsize=(12, 8))\nsns.barplot(\n    x='importance',\n    y='feature',\n    data=feature_importances.head(20),\n    palette='viridis'\n)\nplt.title('Top 20 Feature Importances from LightGBM Model')\nplt.xlabel('Importance')\nplt.ylabel('Feature')\nplt.tight_layout()\nplt.show()\n\n# --- 3. Draft the Project Summary ---\n# Extract the top 3 features for our summary\ntop_3_features = feature_importances['feature'].head(3).tolist()\n\nprint(\"\\n\\n--- DRAFT PROJECT SUMMARY ---\")\nsummary = f\"\"\"\nOur solution successfully predicts customer default by transforming complex, time-series statement data into a powerful set of summary features for each customer.\n\nKey Strategy: We aggregated customer data over time, focusing on the mean, standard deviation, and most recent values of their financial behaviors. This created a rich, single-view dataset for each individual.\n\nModel Performance: A LightGBM classifier proved highly effective, achieving an average ROC AUC score of {lgbm_score:.4f}. This significantly outperformed our robust Logistic Regression baseline score of {baseline_score:.4f}, confirming the presence of complex, non-linear patterns in the data.\n\nKey Predictive Features: The model's decisions are primarily driven by key features such as **{top_3_features[0]}**, **{top_3_features[1]}**, and **{top_3_features[2]}**. This indicates that [Your interpretation here - e.g., 'a customer's most recent payment behavior and the minimum balance they've held are critical indicators of risk.'].\n\nConclusion: This approach allows for the effective identification of high-risk customers, enabling proactive risk management.\n\"\"\"\nprint(summary)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T21:00:16.210565Z","iopub.execute_input":"2025-10-07T21:00:16.210875Z","iopub.status.idle":"2025-10-07T21:00:57.633219Z","shell.execute_reply.started":"2025-10-07T21:00:16.210844Z","shell.execute_reply":"2025-10-07T21:00:57.632285Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Feature engineering (especially the mean, std, last aggregations) is so effective that it makes the problem almost linearly separable. The features you created are so powerful that even a simple model can draw a very good line between defaulters and non-defaulters. There isn't much complex, non-linear signal left for LightGBM to find. This is a sign of excellent feature engineering.\n\nLooking at the top features: P_2_last, D_39_last, B_4_last, B_5_last, B_1_last... The pattern is undeniable. The single most predictive information about a customer is their most recent behavior. A customer's statement from last month is far more important than their average over the past year. This makes perfect business sense and is a highly actionable insight.","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# === PRIORITY #1: IMPLEMENT THE OFFICIAL AMEX METRIC ===\n\ndef amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    \"\"\"\n    The official evaluation metric for the American Express Default Prediction competition.\n    \n    This function is a Python implementation of the metric provided by the competition hosts.\n    \"\"\"\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['rank'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['rank'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T21:39:42.613443Z","iopub.execute_input":"2025-10-07T21:39:42.614619Z","iopub.status.idle":"2025-10-07T21:39:42.640622Z","shell.execute_reply.started":"2025-10-07T21:39:42.614510Z","shell.execute_reply":"2025-10-07T21:39:42.638277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === NEW CELL 2: ENGINEERING \"RECENCY\" FEATURES ===\n# Based on feature importance plot, *_last features are the most powerful.\n# Our hypothesis is that the change from a customer's average behavior to their most recent behavior is a very strong signal of changing risk.\n# This cell creates those new features.\nprint(\"Engineering new 'recency' features based on the feature importance plot...\")\n\n# We will use the original aggregated dataframe 'df_agg' from Step 3\n# and add new columns to it.\n\n# Let's select the base names of the top features you found\ntop_features_base = ['P_2', 'D_39', 'B_4', 'B_5', 'B_1', 'B_3', 'D_46', 'S_3', 'D_43', 'B_2']\nnew_feature_names = []\n\nfor col in top_features_base:\n    # Check if both 'last' and 'mean' columns exist for this feature\n    if f'{col}_last' in df_agg.columns and f'{col}_mean' in df_agg.columns:\n        new_feature_name = f'{col}_last_minus_mean'\n        df_agg[new_feature_name] = df_agg[f'{col}_last'] - df_agg[f'{col}_mean']\n        new_feature_names.append(new_feature_name)\n\nprint(f\"Successfully created {len(new_feature_names)} new features.\")\nprint(\"Example of the new features:\")\ndisplay(df_agg[['customer_ID'] + new_feature_names].head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T21:39:30.801280Z","iopub.execute_input":"2025-10-07T21:39:30.802044Z","iopub.status.idle":"2025-10-07T21:39:30.821780Z","shell.execute_reply.started":"2025-10-07T21:39:30.802009Z","shell.execute_reply":"2025-10-07T21:39:30.820557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === NEW CELL 3: STEP 5 (UPDATED) - RE-EVALUATING WITH OFFICIAL METRIC ===\n\nimport lightgbm as lgb\nimport numpy as np\nimport gc\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nimport warnings\n\n# Suppress LightGBM warnings\nwarnings.filterwarnings(\"ignore\", message=\"No further splits with positive gain, best gain: -inf\")\n\n# --- 1. Prepare Data (using the dataframe with our NEW features) ---\nprint(\"Preparing data with the newly engineered features...\")\nX = df_agg.drop(columns=['customer_ID', 'target'])\ny = df_agg['target']\n\n# We need to re-identify the feature types since we added new numerical columns\ncat_features_agg = [col for col in X.columns if X[col].dtype == 'object' or X[col].dtype == 'category']\nnum_features_agg = [col for col in X.columns if col not in cat_features_agg]\n\n# The preprocessor is the same, but it will be refit on the new data\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numeric_transformer, num_features_agg),\n        ('cat', categorical_transformer, cat_features_agg)\n    ],\n    remainder='passthrough'\n)\n\n# --- 2. Fit the Preprocessor ONCE on the entire NEW dataset ---\nprint(\"Fitting the preprocessor on the updated training set...\")\nX_transformed = preprocessor.fit_transform(X)\nohe_feature_names = preprocessor.named_transformers_['cat'].get_feature_names_out(cat_features_agg)\nall_feature_names = num_features_agg + ohe_feature_names.tolist()\nprint(\"Data has been preprocessed.\")\n\n# --- 3. Evaluate with Cross-Validation ---\nprint(\"\\nStarting cross-validation with BOTH AUC and official Amex Metric...\")\nskf = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\nlgbm_auc_scores = []\nlgbm_amex_scores = []\n\nfor fold, (train_index, val_index) in enumerate(skf.split(X_transformed, y)):\n    X_train, X_val = X_transformed[train_index], X_transformed[val_index]\n    y_train, y_val = y.iloc[train_index], y.iloc[val_index]\n\n    model = lgb.LGBMClassifier(random_state=42, is_unbalance=True)\n    model.fit(X_train, y_train, feature_name=all_feature_names)\n\n    y_pred_proba = model.predict_proba(X_val)[:, 1]\n\n    # Calculate AUC\n    fold_auc = roc_auc_score(y_val, y_pred_proba)\n    lgbm_auc_scores.append(fold_auc)\n\n    # Calculate Amex Metric\n    y_val_df = pd.DataFrame(y_val.values, columns=['target'])\n    y_pred_df = pd.DataFrame(y_pred_proba, columns=['prediction'])\n    fold_amex = amex_metric(y_val_df, y_pred_df)\n    lgbm_amex_scores.append(fold_amex)\n    \n    print(f\"Fold {fold+1} -> AUC: {fold_auc:.4f}, Amex Metric: {fold_amex:.4f}\")\n    \n    del model, X_train, X_val, y_train, y_val, y_val_df, y_pred_df\n    gc.collect()\n\nprint(\"\\n--- Final LightGBM Model Evaluation (with new features) ---\")\nprint(f\"LightGBM Average ROC AUC Score: {np.mean(lgbm_auc_scores):.4f}\")\nprint(f\"LightGBM Average Amex Metric:  {np.mean(lgbm_amex_scores):.4f} (+/- {np.std(lgbm_amex_scores):.4f})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T21:39:19.809158Z","iopub.execute_input":"2025-10-07T21:39:19.809475Z","iopub.status.idle":"2025-10-07T21:39:25.315703Z","shell.execute_reply.started":"2025-10-07T21:39:19.809454Z","shell.execute_reply":"2025-10-07T21:39:25.314431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ===PART 1 - CREATING FIRST SUBMISSION FILE ===\n\nimport pandas as pd\nimport numpy as np\n\nprint(\"Starting the submission process...\")\n\n# --- 1. Train the Final Model on ALL Training Data ---\n# We use the final preprocessed training data from the previous step (X_transformed, y)\n# and the final list of feature names (all_feature_names)\n\nprint(\"Training the final LightGBM model on the entire 1M-row training sample...\")\nfinal_model = lgb.LGBMClassifier(random_state=42, is_unbalance=True)\nfinal_model.fit(X_transformed, y, feature_name=all_feature_names)\nprint(\"Final model has been trained.\")\n\n\n# --- 2. Process the Test Data ---\n# The test data is also large, so we need a memory-efficient way to process it.\n# We will process it customer by customer.\nprint(\"Loading and processing the test data...\")\nTEST_DATA_PATH = '/kaggle/input/amex-default-prediction/test_data.csv'\n\n# This is a simplified chunking-like process for aggregation\n# A more optimized version would read the file in chunks, but this works for demonstration\ntest_df = pd.read_csv(TEST_DATA_PATH)\n\n# Perform the SAME aggregation as on the training data\nprint(\"Aggregating test data features...\")\ntest_agg = test_df.groupby('customer_ID').agg(all_aggs) # all_aggs was defined in your Step 3\ntest_agg.columns = ['_'.join(col).strip() for col in test_agg.columns.values]\ntest_agg.reset_index(inplace=True)\n\n# Perform the SAME feature engineering for \"recency\" features\nprint(\"Engineering 'recency' features for test data...\")\nfor col in top_features_base: # top_features_base was defined in your \"recency\" cell\n    if f'{col}_last' in test_agg.columns and f'{col}_mean' in test_agg.columns:\n        test_agg[f'{col}_last_minus_mean'] = test_agg[f'{col}_last'] - test_agg[f'{col}_mean']\n\n# Keep track of customer IDs for the submission file\ntest_customer_ids = test_agg['customer_ID']\nX_test = test_agg.drop(columns=['customer_ID'])\n\n# --- 3. Make Predictions ---\n# Use the SAME preprocessor that was FIT on the TRAINING data.\n# This is crucial to avoid data leakage!\nprint(\"Preprocessing test data and making predictions...\")\nX_test_transformed = preprocessor.transform(X_test)\ntest_predictions = final_model.predict_proba(X_test_transformed)[:, 1]\n\n# --- 4. Create Submission File ---\nprint(\"Creating submission file...\")\nsubmission_df = pd.DataFrame({\n    'customer_ID': test_customer_ids,\n    'prediction': test_predictions\n})\nsubmission_df.to_csv('submission.csv', index=False)\n\nprint(\"\\nSubmission file 'submission.csv' created successfully!\")\nprint(\"You can now go to the 'Output' section of your notebook on the right, find the file, and click 'Submit'.\")\ndisplay(submission_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T21:39:25.749777Z","iopub.execute_input":"2025-10-07T21:39:25.750655Z","iopub.status.idle":"2025-10-07T21:39:25.770354Z","shell.execute_reply.started":"2025-10-07T21:39:25.750624Z","shell.execute_reply":"2025-10-07T21:39:25.769046Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"1. Next insight:\n\n* Feature Engineering: While _last features describe the customer’s current state, _diff1 features capture their momentum. Are their balances suddenly spiking? Have their payments dropped off? These short-term behavioral shifts often provide the strongest early signals of an impending default.\n\n2. Managing Large Data\n* With GPU access: Use cuDF instead of pandas — it runs on the GPU and is significantly faster. Store your data in .parquet format (much faster to read than .csv) and use a memory-efficient data loader like DeviceQuantileDMatrix, which streams data to the model in chunks instead of loading it all into memory.\n* Without GPU: Aggressively downcast data types. For example, converting a float64 (8 bytes) to a float16 (2 bytes) reduces memory usage for that column by 75%. Applying this across all numerical columns can make it possible to fit the entire dataset into memory.\n\n3. Advanced Modeling Tricks\n* Seed Bleeding: Train your full 5-fold model multiple times using different random seeds (e.g., 42, 52, 62) and average the predictions. Each model learns slightly different nuances, and averaging them smooths out noise, leading to a more robust and generalizable final prediction. This almost always improves leaderboard scores.\n* Hyperparameter Tuning: Note their parameters — max_depth=4, learning_rate=0.01, feature_fraction=0.20, etc. These are not defaults; they’ve been carefully tuned for this dataset. Your model is still using defaults, so tuning will likely yield significant gains.\n* Early Stopping: Using early_stopping_rounds=100 tells the model to monitor validation performance and halt training when improvements plateau. This prevents overfitting and dramatically reduces training time.\n\n**THE SECOND VERSION**","metadata":{}},{"cell_type":"code","source":"# === CELL 1: CONFIGURATION, SETUP, AND METRIC ===\n\nimport os\nimport gc\nimport warnings\nimport random\nimport numpy as np\nimport pandas as pd\nimport lightgbm as lgb\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import LabelEncoder\nfrom tqdm.auto import tqdm\n\n# Ignore warnings for cleaner output\nwarnings.filterwarnings('ignore')\n\n# --- Configuration Class ---\nclass CFG:\n    # Correct data paths to the original CSV files\n    TRAIN_PATH = '/kaggle/input/amex-default-prediction/train_data.csv'\n    TEST_PATH = '/kaggle/input/amex-default-prediction/test_data.csv'\n    TRAIN_LABELS_PATH = '/kaggle/input/amex-default-prediction/train_labels.csv'\n    SUBMISSION_PATH = '/kaggle/input/amex-default-prediction/sample_submission.csv'\n    \n    # Seeds for blending\n    seeds = [42, 52, 62]\n    \n    # Model settings\n    n_folds = 5\n    target = 'target'\n    \n    # LightGBM parameters (tuned for performance)\n    params = {\n        'objective': 'binary',\n        'metric': 'binary_logloss',\n        'boosting': 'dart',\n        'seed': 42,\n        'num_leaves': 100,\n        'learning_rate': 0.01,\n        'feature_fraction': 0.20,\n        'bagging_freq': 10,\n        'bagging_fraction': 0.50,\n        'n_jobs': -1,\n        'lambda_l2': 2,\n        'min_data_in_leaf': 40,\n    }\n\n# --- Function to seed everything for reproducibility ---\ndef seed_everything(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n\n# --- Official Amex Metric ---\ndef amex_metric(y_true, y_pred):\n    labels = np.transpose(np.array([y_true, y_pred]))\n    labels = labels[labels[:, 1].argsort()[::-1]]\n    weights = np.where(labels[:,0]==0, 20, 1)\n    cut_vals = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n    gini = [0,0]\n    for i in [1,0]:\n        labels = np.transpose(np.array([y_true, y_pred]))\n        labels = labels[labels[:, i].argsort()[::-1]]\n        weight = np.where(labels[:,0]==0, 20, 1)\n        weight_random = np.cumsum(weight / np.sum(weight))\n        total_pos = np.sum(labels[:, 0] *  weight)\n        cum_pos_found = np.cumsum(labels[:, 0] * weight)\n        lorentz = cum_pos_found / total_pos\n        gini[i] = np.sum((lorentz - weight_random) * weight)\n    return 0.5 * (gini[1]/gini[0] + top_four)\n\ndef lgb_amex_metric(y_pred, y_true):\n    y_true = y_true.get_label()\n    return 'amex_metric', amex_metric(y_true, y_pred), True\n\nprint(\"Configuration and setup complete.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-07T21:59:23.207313Z","iopub.execute_input":"2025-10-07T21:59:23.207708Z","iopub.status.idle":"2025-10-07T21:59:35.928693Z","shell.execute_reply.started":"2025-10-07T21:59:23.207675Z","shell.execute_reply":"2025-10-07T21:59:35.927338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === CELL 2: DATA PROCESSING & FEATURE ENGINEERING (FROM CSV) ===\n\n# Define feature groups\ncat_features = [\"B_30\", \"B_38\", \"D_114\", \"D_116\", \"D_117\", \"D_120\", \"D_126\", \"D_63\", \"D_64\", \"D_66\", \"D_68\"]\nnum_features = [col for col in pd.read_csv(CFG.TRAIN_PATH, nrows=1).columns if col not in ['customer_ID', 'S_2'] + cat_features]\n\ndef process_and_feature_engineer_chunk(df_chunk):\n    # This function processes a single chunk of data\n    \n    # 1. Basic Aggregations\n    num_agg = df_chunk.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n    num_agg.columns = ['_'.join(x) for x in num_agg.columns]\n    cat_agg = df_chunk.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\n    cat_agg.columns = ['_'.join(x) for x in cat_agg.columns]\n    \n    # 2. Momentum Features (_diff1)\n    def get_difference(data, num_features):\n        df_diff = []\n        customer_ids = []\n        for customer_id, group in data.groupby(['customer_ID']):\n            diff_df = group[num_features].diff(1).iloc[[-1]].values.astype(np.float32)\n            df_diff.append(diff_df)\n            customer_ids.append(customer_id)\n        df_diff = np.concatenate(df_diff, axis=0)\n        df_diff = pd.DataFrame(df_diff, columns=[col + '_diff1' for col in num_features])\n        df_diff['customer_ID'] = customer_ids\n        return df_diff\n        \n    diff_feats = get_difference(df_chunk, num_features)\n    \n    # Combine all features\n    df_agg = pd.concat([num_agg, cat_agg], axis=1).reset_index()\n    df_agg = df_agg.merge(diff_feats, on='customer_ID', how='left')\n\n    return df_agg\n\ndef read_and_process_data(path):\n    # This reads the entire CSV in chunks and processes them\n    # It's memory-intensive and will take time\n    print(f\"Reading and processing data from: {path}\")\n    \n    # Reading the whole file is necessary for correct aggregations\n    df = pd.read_csv(path)\n    \n    # Process the full dataframe\n    df_agg = process_and_feature_engineer_chunk(df)\n    \n    # 3. Recency vs. Average Features (_last_minus_mean)\n    print(\"Creating recency vs. average features...\")\n    num_feature_names_agg = [f for f in num_features]\n    for col in tqdm(num_feature_names_agg):\n        try:\n            df_agg[f'{col}_last_minus_mean'] = df_agg[f'{col}_last'] - df_agg[f'{col}_mean']\n        except:\n            pass\n            \n    return df_agg\n\ndef reduce_mem_usage(df):\n    start_mem = df.memory_usage().sum() / 1024**2\n    for col in tqdm(df.columns):\n        col_type = df[col].dtype\n        if col_type != object and col_type.name != 'category':\n            c_min, c_max = df[col].min(), df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                # ... (add other int types if needed) ...\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n    end_mem = df.memory_usage().sum() / 1024**2\n    print(f'Memory usage decreased from {start_mem:.2f}MB to {end_mem:.2f}MB')\n    return df\n\n# --- Execute Processing ---\nprint(\"Processing Train Data...\")\ntrain = read_and_process_data(CFG.TRAIN_PATH)\ntrain_labels = pd.read_csv(CFG.TRAIN_LABELS_PATH)\ntrain = train.merge(train_labels, on='customer_ID', how='left')\ntrain = reduce_mem_usage(train)\n\nprint(\"\\nProcessing Test Data...\")\ntest = read_and_process_data(CFG.TEST_PATH)\ntest = reduce_mem_usage(test)\n\ndel train_labels; gc.collect()","metadata":{"trusted":true,"execution":{"execution_failed":"2025-10-07T22:07:34.416Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === CELL 3: MODEL TRAINING & INFERENCE WITH SEED BLENDING ===\n\n# Label encode categorical features\ncat_features_last = [f\"{cf}_last\" for cf in cat_features]\nfor cat_col in tqdm(cat_features_last):\n    encoder = LabelEncoder()\n    # Handle potential NaNs introduced during merge\n    train[cat_col] = train[cat_col].fillna('NaN').astype(str)\n    test[cat_col] = test[cat_col].fillna('NaN').astype(str)\n    \n    train[cat_col] = encoder.fit_transform(train[cat_col])\n    # Use transform only on test data; handle categories not seen in train\n    test[cat_col] = test[cat_col].map(lambda s: '<unknown>' if s not in encoder.classes_ else s)\n    encoder.classes_ = np.append(encoder.classes_, '<unknown>')\n    test[cat_col] = encoder.transform(test[cat_col])\n\n# Get feature list\nfeatures = [col for col in train.columns if col not in ['customer_ID', CFG.target]]\n\n# To store predictions\noof_predictions = np.zeros(len(train))\ntest_predictions = np.zeros(len(test))\n\n# --- SEED BLENDING LOOP ---\nfor seed in CFG.seeds:\n    print(\"\\n\" + \"=\"*50)\n    print(f\"TRAINING WITH SEED: {seed}\")\n    print(\"=\"*50)\n    \n    seed_everything(seed)\n    CFG.params['seed'] = seed\n    \n    skf = StratifiedKFold(n_splits=CFG.n_folds, shuffle=True, random_state=seed)\n    \n    # --- FOLD LOOP ---\n    for fold, (trn_ind, val_ind) in enumerate(skf.split(train, train[CFG.target])):\n        print(f\"\\n----------- Fold {fold+1} -----------\")\n        \n        x_train, x_val = train[features].iloc[trn_ind], train[features].iloc[val_ind]\n        y_train, y_val = train[CFG.target].iloc[trn_ind], train[CFG.target].iloc[val_ind]\n        \n        lgb_train = lgb.Dataset(x_train, y_train, categorical_feature=cat_features_last)\n        lgb_valid = lgb.Dataset(x_val, y_val, categorical_feature=cat_features_last)\n        \n        model = lgb.train(\n            params=CFG.params,\n            train_set=lgb_train,\n            num_boost_round=10000,\n            valid_sets=[lgb_train, lgb_valid],\n            early_stopping_rounds=500,\n            verbose_eval=500,\n            feval=lgb_amex_metric\n        )\n        \n        val_pred = model.predict(x_val)\n        oof_predictions[val_ind] += val_pred / len(CFG.seeds)\n        test_predictions += model.predict(test[features]) / (CFG.n_folds * len(CFG.seeds))\n        \n        del x_train, x_val, y_train, y_val, lgb_train, lgb_valid, model\n        gc.collect()\n\n# Final CV score\nfinal_cv_score = amex_metric(train[CFG.target], oof_predictions)\nprint(\"\\n\" + \"#\"*50)\nprint(f\"OVERALL CV SCORE (after seed blending): {final_cv_score:.5f}\")\nprint(\"#\"*50)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-10-07T22:07:34.417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === CELL 4: CREATE SUBMISSION FILE ===\n\nsub = pd.read_csv(CFG.SUBMISSION_PATH)\nsub['prediction'] = test_predictions\nsub.to_csv('submission.csv', index=False)\n\nprint(\"Submission file created successfully!\")\ndisplay(sub.head())","metadata":{"trusted":true,"execution":{"execution_failed":"2025-10-07T22:07:34.417Z"}},"outputs":[],"execution_count":null}]}