{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"}],"dockerImageVersionId":31234,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# STEP 2: Import required libraries\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Make plots look clean\nsns.set_style(\"whitegrid\")\n\nprint(\"Libraries loaded successfully\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:46:22.760879Z","iopub.execute_input":"2025-12-17T17:46:22.761226Z","iopub.status.idle":"2025-12-17T17:46:25.882541Z","shell.execute_reply.started":"2025-12-17T17:46:22.761195Z","shell.execute_reply":"2025-12-17T17:46:25.881586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 3: Load the dataset (small sample)\n\nprint(\"Loading data...\")\n\n# Load first 50,000 rows only (safe for Kaggle)\ntrain_data = pd.read_csv(\n    \"/kaggle/input/amex-default-prediction/train_data.csv\",\n    nrows=50000\n)\n\n# Load labels\ntrain_labels = pd.read_csv(\n    \"/kaggle/input/amex-default-prediction/train_labels.csv\"\n)\n\nprint(\"Train data shape:\", train_data.shape)\nprint(\"Train labels shape:\", train_labels.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:47:22.441528Z","iopub.execute_input":"2025-12-17T17:47:22.441964Z","iopub.status.idle":"2025-12-17T17:47:26.657575Z","shell.execute_reply.started":"2025-12-17T17:47:22.441936Z","shell.execute_reply":"2025-12-17T17:47:26.656718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 4: Merge train data with labels\n\nprint(\"Merging data...\")\n\ndf = train_data.merge(\n    train_labels,\n    on=\"customer_ID\",\n    how=\"left\"\n)\n\nprint(\"Merged dataset shape:\", df.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:49:09.325249Z","iopub.execute_input":"2025-12-17T17:49:09.325624Z","iopub.status.idle":"2025-12-17T17:49:09.412523Z","shell.execute_reply.started":"2025-12-17T17:49:09.325595Z","shell.execute_reply":"2025-12-17T17:49:09.411507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 5: Quick data check\n\nprint(\"Column names:\")\nprint(df.columns)\n\nprint(\"\\nFirst 5 rows of the dataset:\")\ndf.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:49:54.404950Z","iopub.execute_input":"2025-12-17T17:49:54.405579Z","iopub.status.idle":"2025-12-17T17:49:54.440752Z","shell.execute_reply.started":"2025-12-17T17:49:54.405548Z","shell.execute_reply":"2025-12-17T17:49:54.440124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 6: Check missing values\n\ntotal_missing = df.isnull().sum().sum()\nprint(\"Total missing values in dataset:\", total_missing)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:50:56.369303Z","iopub.execute_input":"2025-12-17T17:50:56.369920Z","iopub.status.idle":"2025-12-17T17:50:56.399917Z","shell.execute_reply.started":"2025-12-17T17:50:56.369891Z","shell.execute_reply":"2025-12-17T17:50:56.399126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 7: Analyse the target variable\n\ntarget_distribution = df[\"target\"].value_counts()\ntarget_percentage = df[\"target\"].value_counts(normalize=True) * 100\n\nprint(\"Target counts:\")\nprint(target_distribution)\n\nprint(\"\\nTarget percentage:\")\nprint(target_percentage)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:51:53.396927Z","iopub.execute_input":"2025-12-17T17:51:53.397609Z","iopub.status.idle":"2025-12-17T17:51:53.410745Z","shell.execute_reply.started":"2025-12-17T17:51:53.397578Z","shell.execute_reply":"2025-12-17T17:51:53.409770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 8: Visualise default vs non-default\n\nplt.figure(figsize=(6, 4))\nsns.countplot(x=\"target\", data=df, palette=\"viridis\")\n\nplt.title(\"Distribution of Credit Defaults\")\nplt.xlabel(\"Customer Status (0 = Paid, 1 = Default)\")\nplt.ylabel(\"Number of Customers\")\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:52:56.709452Z","iopub.execute_input":"2025-12-17T17:52:56.710384Z","iopub.status.idle":"2025-12-17T17:52:57.014754Z","shell.execute_reply.started":"2025-12-17T17:52:56.710347Z","shell.execute_reply":"2025-12-17T17:52:57.013974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 8 (UPGRADED): Professional default distribution chart\n\nplt.figure(figsize=(7, 5))\n\nax = sns.countplot(\n    x=\"target\",\n    data=df,\n    hue=\"target\",\n    palette=[\"#1f77b4\", \"#d62728\"],\n    legend=False\n)\n\n# Titles and labels\nplt.title(\"Credit Default Distribution\", fontsize=14, weight=\"bold\")\nplt.xlabel(\"Customer Status (0 = Paid, 1 = Default)\", fontsize=11)\nplt.ylabel(\"Number of Customers\", fontsize=11)\n\n# Add numbers on top of bars\nfor p in ax.patches:\n    ax.annotate(\n        f\"{int(p.get_height())}\",\n        (p.get_x() + p.get_width() / 2., p.get_height()),\n        ha=\"center\",\n        va=\"bottom\",\n        fontsize=10\n    )\n\nsns.despine()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:54:25.936368Z","iopub.execute_input":"2025-12-17T17:54:25.936735Z","iopub.status.idle":"2025-12-17T17:54:26.187017Z","shell.execute_reply.started":"2025-12-17T17:54:25.936707Z","shell.execute_reply":"2025-12-17T17:54:26.186287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 8 (FINAL): Advanced & aesthetic default distribution chart\n\n# Calculate percentages for labels\ncounts = df[\"target\"].value_counts().sort_index()\npercentages = df[\"target\"].value_counts(normalize=True).sort_index() * 100\n\n# Create figure\nplt.figure(figsize=(8, 5))\n\nax = sns.barplot(\n    x=[\"Paid (0)\", \"Default (1)\"],\n    y=counts.values,\n    palette=[\"#4C72B0\", \"#DD8452\"]\n)\n\n# Title and labels\nplt.title(\"Customer Credit Default Distribution\", fontsize=15, weight=\"bold\", pad=15)\nplt.ylabel(\"Number of Customers\", fontsize=11)\nplt.xlabel(\"Customer Status\", fontsize=11)\n\n# Add value + percentage labels\nfor i, value in enumerate(counts.values):\n    ax.text(\n        i,\n        value + 500,\n        f\"{value:,}\\n({percentages.iloc[i]:.1f}%)\",\n        ha=\"center\",\n        va=\"bottom\",\n        fontsize=11,\n        weight=\"bold\"\n    )\n\n# Style cleanup\nsns.despine(left=True, bottom=True)\nax.grid(axis=\"y\", linestyle=\"--\", alpha=0.4)\nplt.tight_layout()\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T17:57:28.072426Z","iopub.execute_input":"2025-12-17T17:57:28.073290Z","iopub.status.idle":"2025-12-17T17:57:28.259472Z","shell.execute_reply.started":"2025-12-17T17:57:28.073246Z","shell.execute_reply":"2025-12-17T17:57:28.257374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 9: Which customers default? (Spending behaviour comparison)\n\nplt.figure(figsize=(8, 5))\n\nsns.boxplot(\n    x=\"target\",\n    y=\"S_3\",\n    data=df,\n    palette=[\"#4C72B0\", \"#DD8452\"]\n)\n\nplt.title(\n    \"Spending Behaviour by Customer Default Status\",\n    fontsize=15,\n    weight=\"bold\",\n    pad=15\n)\n\nplt.xlabel(\"Customer Status (0 = Paid, 1 = Default)\", fontsize=11)\nplt.ylabel(\"Spending Metric (S_3)\", fontsize=11)\n\nsns.despine()\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T18:01:48.052231Z","iopub.execute_input":"2025-12-17T18:01:48.052545Z","iopub.status.idle":"2025-12-17T18:01:48.341837Z","shell.execute_reply.started":"2025-12-17T18:01:48.052518Z","shell.execute_reply":"2025-12-17T18:01:48.340986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 9 (EXECUTIVE VERSION): Who defaults? Spend vs Payment\n\nplt.figure(figsize=(9, 6))\n\nscatter = sns.scatterplot(\n    data=df,\n    x=\"S_3\",\n    y=\"P_2\",\n    hue=\"target\",\n    palette={0: \"#4C72B0\", 1: \"#C44E52\"},\n    alpha=0.6\n)\n\nplt.title(\n    \"Customer Behaviour Segmentation: Spend vs Payment\",\n    fontsize=15,\n    weight=\"bold\",\n    pad=15\n)\n\nplt.xlabel(\"Spending Behaviour (Higher = More Spend)\", fontsize=11)\nplt.ylabel(\"Payment Behaviour (Higher = Better Repayment)\", fontsize=11)\n\n# Improve legend\nplt.legend(\n    title=\"Customer Outcome\",\n    labels=[\"Paid (0)\", \"Defaulted (1)\"],\n    frameon=False\n)\n\nsns.despine()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T18:09:33.712611Z","iopub.execute_input":"2025-12-17T18:09:33.712981Z","iopub.status.idle":"2025-12-17T18:09:35.129506Z","shell.execute_reply.started":"2025-12-17T18:09:33.712951Z","shell.execute_reply":"2025-12-17T18:09:35.128392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 9 (FINAL): Executive segmentation – Default rate by customer type\n\n# Create simple behavioural segments using medians\nspend_median = df[\"S_3\"].median()\npayment_median = df[\"P_2\"].median()\n\ndef classify_customer(row):\n    if row[\"S_3\"] <= spend_median and row[\"P_2\"] > payment_median:\n        return \"Low Spend / High Payment (Safe)\"\n    elif row[\"S_3\"] <= spend_median and row[\"P_2\"] <= payment_median:\n        return \"Low Spend / Low Payment\"\n    elif row[\"S_3\"] > spend_median and row[\"P_2\"] > payment_median:\n        return \"High Spend / High Payment\"\n    else:\n        return \"High Spend / Low Payment (High Risk)\"\n\ndf[\"Customer_Type\"] = df.apply(classify_customer, axis=1)\n\n# Calculate default rate per segment\nsegment_default_rate = (\n    df.groupby(\"Customer_Type\")[\"target\"]\n    .mean()\n    .sort_values(ascending=False) * 100\n)\n\n# Plot\nplt.figure(figsize=(9, 5))\n\nax = sns.barplot(\n    x=segment_default_rate.values,\n    y=segment_default_rate.index,\n    palette=\"Reds_r\"\n)\n\nplt.title(\n    \"Which Customer Types Are Most Likely to Default?\",\n    fontsize=15,\n    weight=\"bold\",\n    pad=15\n)\n\nplt.xlabel(\"Default Rate (%)\", fontsize=11)\nplt.ylabel(\"Customer Behaviour Segment\", fontsize=11)\n\n# Add labels\nfor i, v in enumerate(segment_default_rate.values):\n    ax.text(v + 0.5, i, f\"{v:.1f}%\", va=\"center\", fontsize=11)\n\nsns.despine()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T18:12:54.320187Z","iopub.execute_input":"2025-12-17T18:12:54.321127Z","iopub.status.idle":"2025-12-17T18:12:55.617868Z","shell.execute_reply.started":"2025-12-17T18:12:54.321088Z","shell.execute_reply":"2025-12-17T18:12:55.617056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 10: Early warning signal – Payment behaviour comparison\n\nplt.figure(figsize=(9, 5))\n\n# Calculate average payment behaviour by outcome\npayment_by_target = df.groupby(\"target\")[\"P_2\"].mean()\n\nax = sns.barplot(\n    x=[\"Paid (0)\", \"Defaulted (1)\"],\n    y=payment_by_target.values,\n    palette=[\"#4C72B0\", \"#C44E52\"]\n)\n\nplt.title(\n    \"Early Warning Signal: Decline in Payment Behaviour Before Default\",\n    fontsize=15,\n    weight=\"bold\",\n    pad=15\n)\n\nplt.xlabel(\"Customer Outcome\", fontsize=11)\nplt.ylabel(\"Average Payment Behaviour (P_2)\", fontsize=11)\n\n# Add values on bars\nfor i, v in enumerate(payment_by_target.values):\n    ax.text(i, v + 0.01, f\"{v:.2f}\", ha=\"center\", fontsize=11, weight=\"bold\")\n\nsns.despine()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T18:22:35.861830Z","iopub.execute_input":"2025-12-17T18:22:35.862775Z","iopub.status.idle":"2025-12-17T18:22:36.147411Z","shell.execute_reply.started":"2025-12-17T18:22:35.862740Z","shell.execute_reply":"2025-12-17T18:22:36.146526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 10 (FINAL): Early warning signal – Default rate by payment behaviour level\n\n# Create payment behaviour bands (early warning levels)\ndf[\"Payment_Level\"] = pd.qcut(\n    df[\"P_2\"],\n    q=4,\n    labels=[\n        \"Very Low Payment\",\n        \"Low Payment\",\n        \"Moderate Payment\",\n        \"High Payment\"\n    ]\n)\n\n# Calculate default rate per payment level\npayment_default_rate = (\n    df.groupby(\"Payment_Level\")[\"target\"]\n    .mean()\n    .sort_values(ascending=False) * 100\n)\n\n# Plot\nplt.figure(figsize=(9, 5))\n\nax = sns.barplot(\n    x=payment_default_rate.values,\n    y=payment_default_rate.index,\n    palette=\"Reds_r\"\n)\n\nplt.title(\n    \"Early Warning Signal: Default Risk by Payment Behaviour Level\",\n    fontsize=15,\n    weight=\"bold\",\n    pad=15\n)\n\nplt.xlabel(\"Default Rate (%)\", fontsize=11)\nplt.ylabel(\"Payment Behaviour Level\", fontsize=11)\n\n# Add labels\nfor i, v in enumerate(payment_default_rate.values):\n    ax.text(v + 0.5, i, f\"{v:.1f}%\", va=\"center\", fontsize=11)\n\nsns.despine()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T18:26:23.295561Z","iopub.execute_input":"2025-12-17T18:26:23.296535Z","iopub.status.idle":"2025-12-17T18:26:23.498001Z","shell.execute_reply.started":"2025-12-17T18:26:23.296500Z","shell.execute_reply":"2025-12-17T18:26:23.497131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 10 (ALTERNATIVE): Early warning funnel – Payment deterioration to default risk\n\n# Create ordered payment behaviour bands (early to late warning)\ndf[\"Payment_Band\"] = pd.qcut(\n    df[\"P_2\"],\n    q=4,\n    labels=[\n        \"Strong Payment Behaviour\",\n        \"Moderate Payment Behaviour\",\n        \"Weak Payment Behaviour\",\n        \"Critical Payment Behaviour\"\n    ]\n)\n\n# Calculate default rate\nfunnel_data = (\n    df.groupby(\"Payment_Band\")[\"target\"]\n    .mean()\n    .reindex([\n        \"Strong Payment Behaviour\",\n        \"Moderate Payment Behaviour\",\n        \"Weak Payment Behaviour\",\n        \"Critical Payment Behaviour\"\n    ]) * 100\n)\n\n# Plot funnel-style horizontal bars\nplt.figure(figsize=(9, 5))\n\nax = sns.barplot(\n    x=funnel_data.values,\n    y=funnel_data.index,\n    palette=[\"#4CAF50\", \"#FFC107\", \"#FF9800\", \"#D32F2F\"]\n)\n\nplt.title(\n    \"Early Warning Funnel: Payment Behaviour Deterioration and Default Risk\",\n    fontsize=15,\n    weight=\"bold\",\n    pad=15\n)\n\nplt.xlabel(\"Default Rate (%)\", fontsize=11)\nplt.ylabel(\"Payment Behaviour Stage\", fontsize=11)\n\n# Add labels\nfor i, v in enumerate(funnel_data.values):\n    ax.text(v + 0.5, i, f\"{v:.1f}%\", va=\"center\", fontsize=11, weight=\"bold\")\n\nsns.despine()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T18:29:36.882982Z","iopub.execute_input":"2025-12-17T18:29:36.883363Z","iopub.status.idle":"2025-12-17T18:29:37.088712Z","shell.execute_reply.started":"2025-12-17T18:29:36.883337Z","shell.execute_reply":"2025-12-17T18:29:37.087755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 10 (FINAL Q2): Early warning signal – Risk escalation line chart\n\n# Create ordered payment behaviour bands\ndf[\"Payment_Band\"] = pd.qcut(\n    df[\"P_2\"],\n    q=5,\n    labels=[\n        \"Very Strong\",\n        \"Strong\",\n        \"Moderate\",\n        \"Weak\",\n        \"Critical\"\n    ]\n)\n\n# Calculate default rate per band\nrisk_curve = (\n    df.groupby(\"Payment_Band\")[\"target\"]\n    .mean()\n    .reindex([\n        \"Very Strong\",\n        \"Strong\",\n        \"Moderate\",\n        \"Weak\",\n        \"Critical\"\n    ]) * 100\n)\n\n# Plot line chart\nplt.figure(figsize=(9, 5))\n\nplt.plot(\n    risk_curve.index,\n    risk_curve.values,\n    marker=\"o\",\n    linewidth=3,\n    color=\"#C44E52\"\n)\n\nplt.fill_between(\n    risk_curve.index,\n    risk_curve.values,\n    alpha=0.15,\n    color=\"#C44E52\"\n)\n\nplt.title(\n    \"Early Warning Signal: Default Risk Escalates as Payment Behaviour Deteriorates\",\n    fontsize=15,\n    weight=\"bold\",\n    pad=15\n)\n\nplt.xlabel(\"Payment Behaviour (Best → Worst)\", fontsize=11)\nplt.ylabel(\"Default Rate (%)\", fontsize=11)\n\n# Add point labels\nfor i, v in enumerate(risk_curve.values):\n    plt.text(i, v + 0.8, f\"{v:.1f}%\", ha=\"center\", fontsize=11, weight=\"bold\")\n\nsns.despine()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T18:31:24.850639Z","iopub.execute_input":"2025-12-17T18:31:24.851019Z","iopub.status.idle":"2025-12-17T18:31:25.042407Z","shell.execute_reply.started":"2025-12-17T18:31:24.850989Z","shell.execute_reply":"2025-12-17T18:31:25.041446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 10 (CORRECTED): Early warning signal – intuitive risk escalation line chart\n\n# Create payment behaviour bands (LOW P_2 = WORST payment)\ndf[\"Payment_Band\"] = pd.qcut(\n    df[\"P_2\"],\n    q=5,\n    labels=[\n        \"Critical (Worst Payment)\",\n        \"Weak\",\n        \"Moderate\",\n        \"Strong\",\n        \"Very Strong (Best Payment)\"\n    ]\n)\n\n# Calculate default rate per band\nrisk_curve = (\n    df.groupby(\"Payment_Band\")[\"target\"]\n    .mean()\n    .reindex([\n        \"Critical (Worst Payment)\",\n        \"Weak\",\n        \"Moderate\",\n        \"Strong\",\n        \"Very Strong (Best Payment)\"\n    ]) * 100\n)\n\n# Plot line chart\nplt.figure(figsize=(9, 5))\n\nplt.plot(\n    risk_curve.index,\n    risk_curve.values,\n    marker=\"o\",\n    linewidth=3,\n    color=\"#C44E52\"\n)\n\nplt.fill_between(\n    range(len(risk_curve)),\n    risk_curve.values,\n    alpha=0.15,\n    color=\"#C44E52\"\n)\n\nplt.title(\n    \"Early Warning Signal: Default Risk Decreases as Payment Behaviour Improves\",\n    fontsize=15,\n    weight=\"bold\",\n    pad=15\n)\n\nplt.xlabel(\"Payment Behaviour (Worst → Best)\", fontsize=11)\nplt.ylabel(\"Default Rate (%)\", fontsize=11)\n\n# Add value labels\nfor i, v in enumerate(risk_curve.values):\n    plt.text(i, v + 1, f\"{v:.1f}%\", ha=\"center\", fontsize=11, weight=\"bold\")\n\nsns.despine()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T18:53:11.596051Z","iopub.execute_input":"2025-12-17T18:53:11.596850Z","iopub.status.idle":"2025-12-17T18:53:11.788541Z","shell.execute_reply.started":"2025-12-17T18:53:11.596809Z","shell.execute_reply":"2025-12-17T18:53:11.787565Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Q3: Business impact of targeting high-risk groups\n\n# Create risk tiers based on payment behaviour\ndf[\"Risk_Tier\"] = pd.qcut(\n    df[\"P_2\"],\n    q=3,\n    labels=[\n        \"High Risk\",\n        \"Medium Risk\",\n        \"Low Risk\"\n    ]\n)\n\n# Calculate customer share and default share\nsummary = (\n    df.groupby(\"Risk_Tier\")\n    .agg(\n        Customers=(\"customer_ID\", \"count\"),\n        Defaults=(\"target\", \"sum\")\n    )\n)\n\nsummary[\"Customer Share (%)\"] = summary[\"Customers\"] / summary[\"Customers\"].sum() * 100\nsummary[\"Default Share (%)\"] = summary[\"Defaults\"] / summary[\"Defaults\"].sum() * 100\n\n# Reorder for presentation\nsummary = summary.loc[[\"High Risk\", \"Medium Risk\", \"Low Risk\"]]\n\n# Plot\nplt.figure(figsize=(9, 5))\n\nx = range(len(summary))\n\nplt.bar(\n    x,\n    summary[\"Customer Share (%)\"],\n    width=0.4,\n    label=\"Customer Share (%)\",\n    color=\"#4C72B0\"\n)\n\nplt.bar(\n    [i + 0.4 for i in x],\n    summary[\"Default Share (%)\"],\n    width=0.4,\n    label=\"Default Share (%)\",\n    color=\"#C44E52\"\n)\n\nplt.xticks(\n    [i + 0.2 for i in x],\n    summary.index\n)\n\nplt.title(\n    \"Business Impact of Targeting High-Risk Customers\",\n    fontsize=15,\n    weight=\"bold\",\n    pad=15\n)\n\nplt.ylabel(\"Percentage (%)\", fontsize=11)\nplt.xlabel(\"Customer Risk Segment\", fontsize=11)\n\nplt.legend(frameon=False)\nsns.despine()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T19:18:30.847550Z","iopub.execute_input":"2025-12-17T19:18:30.847883Z","iopub.status.idle":"2025-12-17T19:18:31.042125Z","shell.execute_reply.started":"2025-12-17T19:18:30.847856Z","shell.execute_reply":"2025-12-17T19:18:31.041418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 12 (FIXED): Q3 – Business impact of targeting high-risk customers (Pareto curve)\n\n# Create a risk score proxy (lower P_2 = higher risk)\ndf_q3 = df[[\"P_2\", \"target\"]].dropna().copy()\ndf_q3[\"Risk_Score\"] = -df_q3[\"P_2\"]\n\n# Sort customers by risk (highest risk first)\ndf_q3 = df_q3.sort_values(\"Risk_Score\", ascending=False).reset_index(drop=True)\n\n# Cumulative calculations (FIXED)\ndf_q3[\"Cumulative_Customers\"] = (\n    (df_q3.index + 1) / len(df_q3) * 100\n)\n\ndf_q3[\"Cumulative_Defaults\"] = (\n    df_q3[\"target\"].cumsum() / df_q3[\"target\"].sum() * 100\n)\n\n# Plot Pareto curve\nplt.figure(figsize=(9, 6))\n\nplt.plot(\n    df_q3[\"Cumulative_Customers\"],\n    df_q3[\"Cumulative_Defaults\"],\n    linewidth=3,\n    color=\"#C44E52\",\n    label=\"Defaults Captured by Targeting\"\n)\n\n# Reference diagonal (random targeting)\nplt.plot(\n    [0, 100],\n    [0, 100],\n    linestyle=\"--\",\n    color=\"grey\",\n    alpha=0.6,\n    label=\"Random Targeting\"\n)\n\nplt.title(\n    \"Business Impact of Targeting High-Risk Customers\",\n    fontsize=15,\n    weight=\"bold\",\n    pad=15\n)\n\nplt.xlabel(\"Customers Targeted (%)\", fontsize=11)\nplt.ylabel(\"Defaults Captured (%)\", fontsize=11)\n\nplt.legend(frameon=False)\nsns.despine()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T19:22:13.591724Z","iopub.execute_input":"2025-12-17T19:22:13.592089Z","iopub.status.idle":"2025-12-17T19:22:13.854290Z","shell.execute_reply.started":"2025-12-17T19:22:13.592039Z","shell.execute_reply":"2025-12-17T19:22:13.853416Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 12 (GRAND VERSION): Lorenz-style risk concentration curve with impact shading\n\nimport numpy as np\n\n# Prepare data\ndf_q3 = df[[\"P_2\", \"target\"]].dropna().copy()\ndf_q3[\"Risk_Score\"] = -df_q3[\"P_2\"]  # higher score = higher risk\n\n# Sort by risk (highest first)\ndf_q3 = df_q3.sort_values(\"Risk_Score\", ascending=False).reset_index(drop=True)\n\n# Cumulative shares\ndf_q3[\"cum_customers\"] = (df_q3.index + 1) / len(df_q3)\ndf_q3[\"cum_defaults\"] = df_q3[\"target\"].cumsum() / df_q3[\"target\"].sum()\n\n# Convert to percentages\nx = df_q3[\"cum_customers\"] * 100\ny = df_q3[\"cum_defaults\"] * 100\n\n# Plot\nplt.figure(figsize=(10, 7))\n\n# Targeting curve\nplt.plot(\n    x, y,\n    color=\"#C62828\",\n    linewidth=3,\n    label=\"Targeted Risk Strategy\"\n)\n\n# Random baseline\nplt.plot(\n    [0, 100], [0, 100],\n    linestyle=\"--\",\n    color=\"gray\",\n    linewidth=2,\n    label=\"Untargeted (Random) Strategy\"\n)\n\n# Impact area shading\nplt.fill_between(\n    x, y, x,\n    where=(y > x),\n    color=\"#C62828\",\n    alpha=0.25,\n    label=\"Value Created by Targeting\"\n)\n\n# Titles & labels\nplt.title(\n    \"Business Impact of Targeting High-Risk Customers\",\n    fontsize=16,\n    weight=\"bold\",\n    pad=20\n)\n\nplt.xlabel(\"Customers Targeted (%)\", fontsize=12)\nplt.ylabel(\"Defaults Captured (%)\", fontsize=12)\n\nplt.xlim(0, 100)\nplt.ylim(0, 100)\n\nplt.legend(frameon=False, fontsize=11)\nsns.despine()\nplt.grid(alpha=0.2)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T19:26:00.599255Z","iopub.execute_input":"2025-12-17T19:26:00.599588Z","iopub.status.idle":"2025-12-17T19:26:01.010135Z","shell.execute_reply.started":"2025-12-17T19:26:00.599563Z","shell.execute_reply":"2025-12-17T19:26:01.009283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Q3: Business impact of targeting high-risk groups\n# Graph type: Stacked bar (default concentration)\n\n# Create risk segments using payment behaviour\ndf_q3 = df[[\"P_2\", \"target\"]].dropna().copy()\n\ndf_q3[\"Risk_Segment\"] = pd.qcut(\n    df_q3[\"P_2\"],\n    q=3,\n    labels=[\"High Risk\", \"Medium Risk\", \"Low Risk\"]\n)\n\n# Calculate default contribution by segment\ndefault_contribution = (\n    df_q3[df_q3[\"target\"] == 1]\n    .groupby(\"Risk_Segment\")\n    .size()\n    / df_q3[df_q3[\"target\"] == 1].shape[0]\n    * 100\n)\n\n# Prepare data for stacked bar\nsegments = default_contribution.index.tolist()\nvalues = default_contribution.values.tolist()\n\n# Plot\nplt.figure(figsize=(8, 4))\n\nleft = 0\ncolors = [\"#C62828\", \"#F9A825\", \"#2E7D32\"]\n\nfor segment, value, color in zip(segments, values, colors):\n    plt.barh(\n        y=\"Total Defaults\",\n        width=value,\n        left=left,\n        color=color,\n        label=f\"{segment} ({value:.1f}%)\"\n    )\n    left += value\n\nplt.title(\n    \"Business Impact of Targeting High-Risk Customers\",\n    fontsize=14,\n    weight=\"bold\",\n    pad=12\n)\n\nplt.xlabel(\"Share of Total Defaults (%)\")\nplt.ylabel(\"\")\nplt.xlim(0, 100)\n\nplt.legend(frameon=False, bbox_to_anchor=(1.02, 1), loc=\"upper left\")\nsns.despine(left=True)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T19:28:50.383190Z","iopub.execute_input":"2025-12-17T19:28:50.383589Z","iopub.status.idle":"2025-12-17T19:28:50.522629Z","shell.execute_reply.started":"2025-12-17T19:28:50.383558Z","shell.execute_reply":"2025-12-17T19:28:50.521812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 0: Load data and create df\n\nimport pandas as pd\n\nprint(\"Loading data...\")\n\n# Load labels\ntrain_labels = pd.read_csv(\"/kaggle/input/amex-default-prediction/train_labels.csv\")\n\n# Load a manageable sample of training data\ntrain_data = pd.read_csv(\n    \"/kaggle/input/amex-default-prediction/train_data.csv\",\n    nrows=50000\n)\n\n# Merge to create final dataframe\ndf = train_data.merge(train_labels, on=\"customer_ID\", how=\"left\")\n\nprint(\"Data loaded successfully.\")\nprint(\"df is now ready for analysis.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T20:11:50.175554Z","iopub.execute_input":"2025-12-17T20:11:50.175944Z","iopub.status.idle":"2025-12-17T20:11:55.363220Z","shell.execute_reply.started":"2025-12-17T20:11:50.175913Z","shell.execute_reply":"2025-12-17T20:11:55.362279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 1: Dataset Validation - Structure & Size\n\nprint(\"STEP 1: DATASET STRUCTURE & SIZE CHECK\\n\")\n\nrows, columns = df.shape\nprint(f\"Number of rows (records): {rows}\")\nprint(f\"Number of columns (features): {columns}\")\n\nmemory_usage = df.memory_usage(deep=True).sum() / (1024**2)\nprint(f\"Approximate memory usage: {memory_usage:.2f} MB\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T20:12:28.503442Z","iopub.execute_input":"2025-12-17T20:12:28.504461Z","iopub.status.idle":"2025-12-17T20:12:28.560584Z","shell.execute_reply.started":"2025-12-17T20:12:28.504423Z","shell.execute_reply":"2025-12-17T20:12:28.559804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 2: Data Types & Schema Check\n\nprint(\"STEP 2: DATA TYPES & SCHEMA CHECK\\n\")\n\n# Show data types\ndf_types = df.dtypes.value_counts()\nprint(\"Data type distribution:\")\nprint(df_types)\n\nprint(\"\\nSample of column data types:\")\nprint(df.dtypes.head(10))\n\n# Separate numeric and categorical columns\nnumeric_cols = df.select_dtypes(include=[\"int64\", \"float64\"]).columns\ncategorical_cols = df.select_dtypes(include=[\"object\"]).columns\n\nprint(f\"\\nNumber of numeric columns: {len(numeric_cols)}\")\nprint(f\"Number of categorical columns: {len(categorical_cols)}\")\n\nprint(\"\\nCategorical columns (sample):\")\nprint(list(categorical_cols[:10]))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T20:13:41.839552Z","iopub.execute_input":"2025-12-17T20:13:41.839949Z","iopub.status.idle":"2025-12-17T20:13:41.888602Z","shell.execute_reply.started":"2025-12-17T20:13:41.839919Z","shell.execute_reply":"2025-12-17T20:13:41.887629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 3: Missing Value Analysis (Data Quality Check)\n\nprint(\"STEP 3: MISSING VALUE ANALYSIS\\n\")\n\n# Total missing values\ntotal_missing = df.isnull().sum().sum()\nprint(f\"Total missing values in dataset: {total_missing}\")\n\n# Missing values per column (top 10)\nmissing_by_column = (\n    df.isnull()\n    .sum()\n    .sort_values(ascending=False)\n)\n\nprint(\"\\nTop 10 columns with missing values:\")\nprint(missing_by_column.head(10))\n\n# Percentage missing for top columns\nmissing_percentage = (missing_by_column / len(df)) * 100\n\nprint(\"\\nMissing percentage (Top 10 columns):\")\nprint(missing_percentage.head(10).round(2))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T20:15:03.578998Z","iopub.execute_input":"2025-12-17T20:15:03.579876Z","iopub.status.idle":"2025-12-17T20:15:03.636996Z","shell.execute_reply.started":"2025-12-17T20:15:03.579838Z","shell.execute_reply":"2025-12-17T20:15:03.636220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 4: Target Variable Validation\n\nprint(\"STEP 4: TARGET VARIABLE CHECK\\n\")\n\n# Count default vs non-default\ntarget_counts = df[\"target\"].value_counts()\ntarget_percent = df[\"target\"].value_counts(normalize=True) * 100\n\nprint(\"Target counts:\")\nprint(target_counts)\n\nprint(\"\\nTarget distribution (%):\")\nprint(target_percent.round(2))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T20:15:59.346753Z","iopub.execute_input":"2025-12-17T20:15:59.347074Z","iopub.status.idle":"2025-12-17T20:15:59.358415Z","shell.execute_reply.started":"2025-12-17T20:15:59.347050Z","shell.execute_reply":"2025-12-17T20:15:59.357651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 5: Descriptive Statistics & Initial EDA\n\nprint(\"STEP 5: DESCRIPTIVE STATISTICS & INITIAL EDA\\n\")\n\nkey_features = [\"P_2\", \"S_3\", \"B_1\", \"target\"]\n\n# Overall descriptive statistics\nprint(\"Overall Descriptive Statistics:\")\nprint(df[key_features].describe().round(2))\n\n# Descriptive statistics by default status\nprint(\"\\nDescriptive Statistics by Default Status:\")\ngrouped_stats = df[key_features].groupby(\"target\").describe().round(2)\nprint(grouped_stats)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T20:18:27.910417Z","iopub.execute_input":"2025-12-17T20:18:27.911145Z","iopub.status.idle":"2025-12-17T20:18:27.987580Z","shell.execute_reply.started":"2025-12-17T20:18:27.911111Z","shell.execute_reply":"2025-12-17T20:18:27.986815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 6: Visual EDA – Clean, Business-Ready Charts (FIXED)\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nsns.set_theme(style=\"whitegrid\")\n\n# Convert target to string to avoid palette issues\ndf[\"target_str\"] = df[\"target\"].astype(str)\n\n# -------- Chart 1: Payment Behaviour vs Default --------\nplt.figure(figsize=(8, 5))\n\nsns.boxplot(\n    x=\"target_str\",\n    y=\"P_2\",\n    hue=\"target_str\",\n    data=df,\n    palette={\"0\": \"#2E7D32\", \"1\": \"#C62828\"},\n    legend=False\n)\n\nplt.title(\n    \"Payment Behaviour by Default Status\",\n    fontsize=14,\n    weight=\"bold\",\n    pad=10\n)\n\nplt.xlabel(\"Customer Status (0 = Non-Default, 1 = Default)\")\nplt.ylabel(\"Payment Behaviour (P_2)\")\n\nplt.tight_layout()\nplt.show()\n\n\n# -------- Chart 2: Balance vs Default --------\nplt.figure(figsize=(8, 5))\n\nsns.boxplot(\n    x=\"target_str\",\n    y=\"B_1\",\n    hue=\"target_str\",\n    data=df,\n    palette={\"0\": \"#2E7D32\", \"1\": \"#C62828\"},\n    legend=False\n)\n\nplt.title(\n    \"Outstanding Balance by Default Status\",\n    fontsize=14,\n    weight=\"bold\",\n    pad=10\n)\n\nplt.xlabel(\"Customer Status (0 = Non-Default, 1 = Default)\")\nplt.ylabel(\"Balance Metric (B_1)\")\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T20:22:17.088596Z","iopub.execute_input":"2025-12-17T20:22:17.088963Z","iopub.status.idle":"2025-12-17T20:22:17.794410Z","shell.execute_reply.started":"2025-12-17T20:22:17.088934Z","shell.execute_reply":"2025-12-17T20:22:17.793586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 6 – Advanced Visual EDA\n# Graph 1: Payment Behaviour Distribution by Default Status\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nsns.set_theme(style=\"white\", font_scale=1.1)\n\nplt.figure(figsize=(9, 5))\n\nsns.kdeplot(\n    data=df[df[\"target\"] == 0],\n    x=\"P_2\",\n    fill=True,\n    alpha=0.5,\n    linewidth=2,\n    label=\"Non-Default\",\n    color=\"#2E7D32\"\n)\n\nsns.kdeplot(\n    data=df[df[\"target\"] == 1],\n    x=\"P_2\",\n    fill=True,\n    alpha=0.5,\n    linewidth=2,\n    label=\"Default\",\n    color=\"#C62828\"\n)\n\nplt.title(\n    \"Payment Behaviour Distribution by Default Risk\",\n    fontsize=15,\n    weight=\"bold\",\n    pad=12\n)\n\nplt.xlabel(\"Payment Behaviour (P₂)\")\nplt.ylabel(\"Density\")\nplt.legend(frameon=False)\nsns.despine()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T20:26:49.603729Z","iopub.execute_input":"2025-12-17T20:26:49.604107Z","iopub.status.idle":"2025-12-17T20:26:50.238351Z","shell.execute_reply.started":"2025-12-17T20:26:49.604079Z","shell.execute_reply":"2025-12-17T20:26:50.237436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 6 – Advanced Visual EDA\n# Graph 2: Default Risk Gradient vs Payment Behaviour\n\n# Create payment behaviour bins\ndf[\"P2_bin\"] = pd.qcut(df[\"P_2\"], q=20)\n\n# Calculate default rate per bin\nrisk_curve = df.groupby(\"P2_bin\", observed=True)[\"target\"].mean().reset_index()\n\n# Extract midpoints for plotting\nrisk_curve[\"P2_mid\"] = risk_curve[\"P2_bin\"].apply(lambda x: x.mid)\n\nplt.figure(figsize=(9, 5))\n\nsns.scatterplot(\n    x=\"P2_mid\",\n    y=\"target\",\n    data=risk_curve,\n    s=80,\n    color=\"#C62828\"\n)\n\nsns.lineplot(\n    x=\"P2_mid\",\n    y=\"target\",\n    data=risk_curve,\n    linewidth=3,\n    color=\"#C62828\"\n)\n\nplt.title(\n    \"Default Risk Escalation as Payment Behaviour Deteriorates\",\n    fontsize=15,\n    weight=\"bold\",\n    pad=12\n)\n\nplt.xlabel(\"Payment Behaviour (P₂)\")\nplt.ylabel(\"Default Rate\")\nsns.despine()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T20:27:32.468168Z","iopub.execute_input":"2025-12-17T20:27:32.468599Z","iopub.status.idle":"2025-12-17T20:27:32.704681Z","shell.execute_reply.started":"2025-12-17T20:27:32.468544Z","shell.execute_reply":"2025-12-17T20:27:32.703883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 6A: Ridgeline Plot (Joy Plot) – Payment Behaviour\n\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nsns.set_theme(style=\"white\")\n\n# Prepare data\ndf_ridge = df[[\"P_2\", \"target\"]].dropna()\ndf_ridge[\"target_label\"] = df_ridge[\"target\"].map({0: \"Non-Default\", 1: \"Default\"})\n\n# Plot\nplt.figure(figsize=(9, 5))\n\nfor i, label in enumerate([\"Non-Default\", \"Default\"]):\n    subset = df_ridge[df_ridge[\"target_label\"] == label][\"P_2\"]\n    density = sns.kdeplot(\n        subset,\n        fill=True,\n        linewidth=2,\n        alpha=0.8,\n        label=label\n    )\n    for artist in density.collections:\n        artist.set_alpha(0.6)\n\nplt.title(\n    \"Ridgeline View of Payment Behaviour by Risk Outcome\",\n    fontsize=15,\n    weight=\"bold\",\n    pad=12\n)\n\nplt.xlabel(\"Payment Behaviour (P₂)\")\nplt.ylabel(\"\")\nplt.yticks([])\nplt.legend(frameon=False)\nsns.despine(left=True)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T20:31:32.448198Z","iopub.execute_input":"2025-12-17T20:31:32.448621Z","iopub.status.idle":"2025-12-17T20:31:33.039619Z","shell.execute_reply.started":"2025-12-17T20:31:32.448554Z","shell.execute_reply":"2025-12-17T20:31:33.038795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 6B: Hexbin Risk Landscape – Payment vs Balance\n\nplt.figure(figsize=(8, 6))\n\nplt.hexbin(\n    df[\"P_2\"],\n    df[\"B_1\"],\n    C=df[\"target\"],\n    gridsize=40,\n    cmap=\"inferno\",\n    reduce_C_function=np.mean\n)\n\ncb = plt.colorbar()\ncb.set_label(\"Average Default Risk\")\n\nplt.title(\n    \"Customer Risk Landscape: Payment vs Balance\",\n    fontsize=15,\n    weight=\"bold\",\n    pad=12\n)\n\nplt.xlabel(\"Payment Behaviour (P₂)\")\nplt.ylabel(\"Balance Metric (B₁)\")\nsns.despine()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 6B: Hexbin Risk Landscape – Payment vs Balance\n\nplt.figure(figsize=(8, 6))\n\nplt.hexbin(\n    df[\"P_2\"],\n    df[\"B_1\"],\n    C=df[\"target\"],\n    gridsize=40,\n    cmap=\"inferno\",\n    reduce_C_function=np.mean\n)\n\ncb = plt.colorbar()\ncb.set_label(\"Average Default Risk\")\n\nplt.title(\n    \"Customer Risk Landscape: Payment vs Balance\",\n    fontsize=15,\n    weight=\"bold\",\n    pad=12\n)\n\nplt.xlabel(\"Payment Behaviour (P₂)\")\nplt.ylabel(\"Balance Metric (B₁)\")\nsns.despine()\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T20:32:06.715955Z","iopub.execute_input":"2025-12-17T20:32:06.716314Z","iopub.status.idle":"2025-12-17T20:32:07.016835Z","shell.execute_reply.started":"2025-12-17T20:32:06.716282Z","shell.execute_reply":"2025-12-17T20:32:07.015934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 6C: Styled Correlation Heatmap (Aesthetic)\n\nfeatures = [\"P_2\", \"S_3\", \"B_1\", \"target\"]\ncorr = df[features].corr()\n\nplt.figure(figsize=(7, 5))\n\nsns.heatmap(\n    corr,\n    annot=True,\n    fmt=\".2f\",\n    cmap=\"magma\",\n    linewidths=1,\n    linecolor=\"black\",\n    cbar_kws={\"shrink\": 0.8}\n)\n\nplt.title(\n    \"Correlation Structure of Key Risk Drivers\",\n    fontsize=14,\n    weight=\"bold\",\n    pad=12\n)\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T20:32:22.651816Z","iopub.execute_input":"2025-12-17T20:32:22.652447Z","iopub.status.idle":"2025-12-17T20:32:22.869162Z","shell.execute_reply.started":"2025-12-17T20:32:22.652416Z","shell.execute_reply":"2025-12-17T20:32:22.868464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 6D: Risk Contour Map – Default Topography\n\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom scipy.stats import gaussian_kde\n\n# Prepare data\ndf_contour = df[[\"P_2\", \"B_1\", \"target\"]].dropna()\n\n# Only defaulters for risk surface\nx = df_contour[df_contour[\"target\"] == 1][\"P_2\"]\ny = df_contour[df_contour[\"target\"] == 1][\"B_1\"]\n\n# Create grid\nxmin, xmax = x.min(), x.max()\nymin, ymax = y.min(), y.max()\nxx, yy = np.mgrid[xmin:xmax:200j, ymin:ymax:200j]\n\n# Kernel Density Estimation\npositions = np.vstack([xx.ravel(), yy.ravel()])\nvalues = np.vstack([x, y])\nkernel = gaussian_kde(values)\nf = np.reshape(kernel(positions).T, xx.shape)\n\n# Plot\nplt.figure(figsize=(9, 6))\nplt.contourf(xx, yy, f, levels=30, cmap=\"plasma\")\nplt.colorbar(label=\"Default Risk Density\")\n\nplt.title(\n    \"Default Risk Topography: Payment vs Balance\",\n    fontsize=16,\n    weight=\"bold\",\n    pad=15\n)\n\nplt.xlabel(\"Payment Behaviour (P₂)\")\nplt.ylabel(\"Balance Metric (B₁)\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T20:34:55.127662Z","iopub.execute_input":"2025-12-17T20:34:55.128703Z","iopub.status.idle":"2025-12-17T20:35:02.945078Z","shell.execute_reply.started":"2025-12-17T20:34:55.128666Z","shell.execute_reply":"2025-12-17T20:35:02.944251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 6E: Feature–Risk Network Graph\n\nimport networkx as nx\n\n# Select key features\nfeatures = [\"P_2\", \"S_3\", \"B_1\"]\ncorr = df[features + [\"target\"]].corr()[\"target\"].drop(\"target\")\n\n# Create graph\nG = nx.Graph()\n\n# Add target node\nG.add_node(\"Default Risk\", size=3000)\n\n# Add feature nodes\nfor feature in features:\n    G.add_node(feature, size=1500)\n    G.add_edge(\"Default Risk\", feature, weight=abs(corr[feature]))\n\n# Layout\npos = nx.spring_layout(G, seed=42)\n\n# Plot\nplt.figure(figsize=(8, 6))\n\n# Nodes\nnx.draw_networkx_nodes(\n    G, pos,\n    node_size=[G.nodes[n][\"size\"] for n in G.nodes],\n    node_color=[\"#C62828\" if n == \"Default Risk\" else \"#1565C0\" for n in G.nodes]\n)\n\n# Edges\nnx.draw_networkx_edges(\n    G, pos,\n    width=[G[u][v][\"weight\"] * 10 for u, v in G.edges],\n    alpha=0.7\n)\n\n# Labels\nnx.draw_networkx_labels(G, pos, font_size=11, font_color=\"white\")\n\nplt.title(\n    \"Network View of Feature Influence on Default Risk\",\n    fontsize=15,\n    weight=\"bold\",\n    pad=15\n)\n\nplt.axis(\"off\")\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T20:35:27.439419Z","iopub.execute_input":"2025-12-17T20:35:27.440199Z","iopub.status.idle":"2025-12-17T20:35:27.808029Z","shell.execute_reply.started":"2025-12-17T20:35:27.440167Z","shell.execute_reply":"2025-12-17T20:35:27.807362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 6F: Risk Galaxy – Artistic Scatter\n\nplt.figure(figsize=(9, 6))\n\nplt.scatter(\n    df[\"P_2\"],\n    df[\"S_3\"],\n    c=df[\"target\"],\n    cmap=\"coolwarm\",\n    alpha=0.25,\n    s=15\n)\n\nplt.title(\n    \"Customer Risk Galaxy: Payment vs Spend\",\n    fontsize=16,\n    weight=\"bold\",\n    pad=15\n)\n\nplt.xlabel(\"Payment Behaviour (P₂)\")\nplt.ylabel(\"Spending Behaviour (S₃)\")\nplt.colorbar(label=\"Default Risk\")\n\nplt.grid(False)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-17T20:35:38.823353Z","iopub.execute_input":"2025-12-17T20:35:38.824336Z","iopub.status.idle":"2025-12-17T20:35:39.963231Z","shell.execute_reply.started":"2025-12-17T20:35:38.824297Z","shell.execute_reply":"2025-12-17T20:35:39.962327Z"}},"outputs":[],"execution_count":null}]}