{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":31254,"databundleVersionId":3103714}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.ticker as mticker\nimport seaborn as sns\nfrom datetime import timedelta\nfrom scipy import stats\nimport math\nimport warnings\nwarnings.filterwarnings('ignore')\n \nplt.rcParams.update({\n    'figure.facecolor' : '#0f1117',\n    'axes.facecolor'   : '#0f1117',\n    'axes.edgecolor'   : '#2a2d3e',\n    'axes.labelcolor'  : '#c9d1d9',\n    'axes.titlecolor'  : '#e6edf3', \n    'axes.titlesize'   : 13,\n    'axes.titleweight' : 'bold',\n    'xtick.color'      : '#8b949e',\n    'ytick.color'      : '#8b949e',\n    'text.color'       : '#c9d1d9',\n    'grid.color'       : '#21262d',\n    'grid.linestyle'   : '--',\n    'legend.facecolor' : '#161b22',\n    'legend.edgecolor' : '#2a2d3e',\n})\n \nRED    = '#ff4d6d'\nTEAL   = '#06d6a0'\nAMBER  = '#ffd166'\nBLUE   = '#4facfe'\nORANGE = '#ff9f1c'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-24T03:12:26.579618Z","iopub.execute_input":"2026-03-24T03:12:26.580283Z","iopub.status.idle":"2026-03-24T03:12:27.549254Z","shell.execute_reply.started":"2026-03-24T03:12:26.580255Z","shell.execute_reply":"2026-03-24T03:12:27.548617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reduce_mem(df):\n    before = df.memory_usage(deep=True).sum() / 1024**2\n \n    for col in df.select_dtypes(include=[np.number]).columns:\n        lo = df[col].min()\n        hi = df[col].max()\n \n        if df[col].dtype.kind == 'i':\n            for t in [np.int8, np.int16, np.int32]:\n                if lo >= np.iinfo(t).min and hi <= np.iinfo(t).max:\n                    df[col] = df[col].astype(t)\n                    break\n        else:\n            if lo >= np.finfo(np.float32).min and hi <= np.finfo(np.float32).max:\n                df[col] = df[col].astype(np.float32)\n \n    after = df.memory_usage(deep=True).sum() / 1024**2\n    saved = (before - after) / before * 100 if before > 0 else 0\n    print(f\"  {before:.1f} MB -> {after:.1f} MB  ({saved:.1f}% saved)\")\n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T03:12:27.550584Z","iopub.execute_input":"2026-03-24T03:12:27.550939Z","iopub.status.idle":"2026-03-24T03:12:27.558156Z","shell.execute_reply.started":"2026-03-24T03:12:27.550914Z","shell.execute_reply":"2026-03-24T03:12:27.557484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE      = '/kaggle/input/competitions/h-and-m-personalized-fashion-recommendations'\nTX_PATH   = BASE + '/transactions_train.csv'\nCUST_PATH = BASE + '/customers.csv'\nART_PATH  = BASE + '/articles.csv'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T03:12:27.558845Z","iopub.execute_input":"2026-03-24T03:12:27.559078Z","iopub.status.idle":"2026-03-24T03:12:27.568248Z","shell.execute_reply.started":"2026-03-24T03:12:27.559051Z","shell.execute_reply":"2026-03-24T03:12:27.567701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"FILTER_START = '2019-09-01'\n \nchunks = []\n \nfor chunk in pd.read_csv(\n    TX_PATH,\n    usecols=['t_dat', 'customer_id', 'article_id', 'price', 'sales_channel_id'],\n    dtype={\n        'customer_id'      : 'category',\n        'article_id'       : 'int32',\n        'sales_channel_id' : 'int8',\n        'price'            : 'float32',\n    },\n    parse_dates=['t_dat'],\n    chunksize=500_000,\n):\n    chunks.append(chunk[chunk['t_dat'] >= FILTER_START])\n \ntx = pd.concat(chunks, ignore_index=True)\ntx['customer_id'] = tx['customer_id'].astype('category')\ntx = reduce_mem(tx)\n \nprint(f\"Rows      : {len(tx):,}\")\nprint(f\"Period    : {tx['t_dat'].min().date()} ~ {tx['t_dat'].max().date()}\")\nprint(f\"Customers : {tx['customer_id'].nunique():,}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T03:12:27.569068Z","iopub.execute_input":"2026-03-24T03:12:27.569312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"customers = pd.read_csv(\n    CUST_PATH,\n    dtype={\n        'FN'     : 'float32',\n        'Active' : 'float32',\n        'age'    : 'float32',\n    },\n)\ncustomers = reduce_mem(customers)\n \narticles = pd.read_csv(\n    ART_PATH,\n    usecols=['article_id', 'product_group_name', 'index_group_name'],\n)\n \nREF_DATE = tx['t_dat'].max() + timedelta(days=1)\nprint(f\"Reference date : {REF_DATE.date()}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.DataFrame({\n    'Metric': [\n        'Total transactions',\n        'Unique customers',\n        'Unique articles',\n        'Avg. transaction price',\n        'Online ratio',\n    ],\n    'Value': [\n        f\"{len(tx):,}\",\n        f\"{tx['customer_id'].nunique():,}\",\n        f\"{tx['article_id'].nunique():,}\",\n        f\"{tx['price'].mean():.3f}\",\n        f\"{(tx['sales_channel_id']==2).mean()*100:.1f}%\",\n    ]\n})","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"daily = tx.groupby('t_dat').agg(\n    tx_count  = ('customer_id', 'count'),\n    uniq_cust = ('customer_id', 'nunique'),\n).reset_index()\n \nfig, axes = plt.subplots(1, 2, figsize=(16, 4))\nfig.suptitle('Daily Transaction Pattern', fontsize=14, fontweight='bold')\n \naxes[0].plot(daily['t_dat'], daily['tx_count'], color=TEAL, lw=1.5)\naxes[0].fill_between(daily['t_dat'], daily['tx_count'], alpha=0.15, color=TEAL)\naxes[0].set_title('Daily Transactions')\naxes[0].set_xlabel('Date')\naxes[0].set_ylabel('Count')\naxes[0].yaxis.set_major_formatter(\n    mticker.FuncFormatter(lambda x, _: f'{int(x/1000)}K')\n)\n \naxes[1].plot(daily['t_dat'], daily['uniq_cust'], color=AMBER, lw=1.5)\naxes[1].fill_between(daily['t_dat'], daily['uniq_cust'], alpha=0.15, color=AMBER)\naxes[1].set_title('Daily Active Customers')\naxes[1].set_xlabel('Date')\naxes[1].set_ylabel('Count')\naxes[1].yaxis.set_major_formatter(\n    mticker.FuncFormatter(lambda x, _: f'{int(x/1000)}K')\n)\n \nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cust_freq  = tx.groupby('customer_id').size()\ncust_spend = tx.groupby('customer_id')['price'].sum()\nage_data   = customers['age'].dropna()\n \nfig, axes = plt.subplots(1, 3, figsize=(18, 4))\nfig.suptitle('Customer Behavior Distribution', fontsize=14, fontweight='bold')\n \naxes[0].hist(np.log1p(cust_freq), bins=60, color=TEAL,\n             edgecolor='#0f1117', linewidth=0.3)\naxes[0].axvline(\n    np.log1p(cust_freq.median()),\n    color=RED, lw=2, ls='--',\n    label=f'Median {cust_freq.median():.0f}',\n)\naxes[0].set_title('Purchase Frequency (log)')\naxes[0].set_xlabel('log(frequency + 1)')\naxes[0].set_ylabel('Customers')\naxes[0].legend()\n \naxes[1].hist(np.log1p(cust_spend), bins=60, color=AMBER,\n             edgecolor='#0f1117', linewidth=0.3)\naxes[1].set_title('Total Spend (log)')\naxes[1].set_xlabel('log(spend + 1)')\naxes[1].set_ylabel('Customers')\n \naxes[2].hist(age_data, bins=50, color=BLUE,\n             edgecolor='#0f1117', linewidth=0.3)\naxes[2].axvline(\n    age_data.mean(),\n    color=RED, lw=2, ls='--',\n    label=f'Mean {age_data.mean():.1f}',\n)\naxes[2].set_title('Customer Age Distribution')\naxes[2].set_xlabel('Age')\naxes[2].set_ylabel('Customers')\naxes[2].legend()\n \nplt.tight_layout()\nplt.show()\n \ntop20 = cust_freq.nlargest(int(len(cust_freq)*0.2)).sum() / len(tx) * 100\nprint(f\"Top 20% customers account for {top20:.0f}% of transactions (Pareto)\")\nprint(f\"Online purchase ratio : {(tx['sales_channel_id']==2).mean()*100:.1f}%\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"CHURN_WIN   = 90\nOBSERVE_WIN = 180\n \nobs_end   = REF_DATE - timedelta(days=CHURN_WIN)\nobs_start = REF_DATE - timedelta(days=CHURN_WIN + OBSERVE_WIN)\n \nprint(f\"Observation : {obs_start.date()} ~ {obs_end.date()}\")\nprint(f\"Outcome     : {obs_end.date()} ~ {REF_DATE.date()}\")\n \ntx_obs = tx[(tx['t_dat'] >= obs_start) & (tx['t_dat'] < obs_end)].copy()\ntx_out = tx[(tx['t_dat'] >= obs_end)   & (tx['t_dat'] < REF_DATE)].copy()\n \nactive   = tx_obs['customer_id'].unique()\nretained = set(tx_out['customer_id'].unique())\n \nchurn_df = pd.DataFrame({'customer_id': active})\nchurn_df['churn'] = (~churn_df['customer_id'].isin(retained)).astype(int)\n \nchurn_rate = churn_df['churn'].mean()\nprint(f\"\\nModeling set  : {len(churn_df):,}\")\nprint(f\"Churned       : {churn_df['churn'].sum():,}  ({churn_rate:.1%})\")\nprint(f\"Retained      : {(churn_df['churn']==0).sum():,}  ({1-churn_rate:.1%})\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(13, 4))\nfig.suptitle('Churn Label Distribution', fontsize=14, fontweight='bold')\n \ncounts = churn_df['churn'].value_counts().sort_index()\nbars = axes[0].bar(\n    ['Retained (0)', 'Churned (1)'],\n    counts.values,\n    color=[TEAL, RED],\n    edgecolor='#0f1117',\n    width=0.45,\n)\nfor bar, v in zip(bars, counts.values):\n    axes[0].text(\n        bar.get_x() + bar.get_width()/2,\n        bar.get_height() * 1.02,\n        f'{v:,}\\n({v/len(churn_df):.1%})',\n        ha='center', fontsize=11, fontweight='bold',\n    )\naxes[0].set_title('Retained vs Churned')\naxes[0].set_ylabel('Customers')\naxes[0].set_ylim(0, counts.max() * 1.25)\n \nlast_purchase = tx_obs.groupby('customer_id')['t_dat'].max()\ndays_since    = (obs_end - last_purchase).dt.days\nlabel_map     = churn_df.set_index('customer_id')['churn']\n \nchurn_days    = days_since[label_map.reindex(days_since.index).fillna(1) == 1]\nretained_days = days_since[label_map.reindex(days_since.index).fillna(0) == 0]\n \naxes[1].hist(retained_days, bins=60, alpha=0.7, color=TEAL,\n             label='Retained', density=True)\naxes[1].hist(churn_days,    bins=60, alpha=0.7, color=RED,\n             label='Churned',  density=True)\naxes[1].axvline(30, color=AMBER, ls='--', lw=1.5, label='30 days')\naxes[1].axvline(60, color=BLUE,  ls='--', lw=1.5, label='60 days')\naxes[1].set_title('Days Since Last Purchase')\naxes[1].set_xlabel('Days')\naxes[1].set_ylabel('Density')\naxes[1].legend()\n \nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tx_obs = tx_obs.merge(\n    articles[['article_id', 'product_group_name']],\n    on='article_id',\n    how='left',\n)\n \nrfm = tx_obs.groupby('customer_id').agg(\n    recency     = ('t_dat', lambda x: (obs_end - x.max()).days),\n    frequency   = ('t_dat', 'count'),\n    monetary    = ('price', 'sum'),\n    avg_price   = ('price', 'mean'),\n    unique_days = ('t_dat', 'nunique'),\n).reset_index()\n \nprint(rfm.describe().round(2))\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"behavior = tx_obs.groupby('customer_id').agg(\n    unique_articles   = ('article_id',         'nunique'),\n    unique_categories = ('product_group_name', 'nunique'),\n    online_ratio      = ('sales_channel_id',   lambda x: (x == 2).mean()),\n).reset_index()\n \nbehavior.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calc_interval(dates):\n    sorted_d = sorted(dates)\n    if len(sorted_d) < 2:\n        return np.nan, np.nan\n    gaps = [(sorted_d[i+1] - sorted_d[i]).days for i in range(len(sorted_d)-1)]\n    return np.mean(gaps), np.std(gaps)\n \nresult = tx_obs.groupby('customer_id')['t_dat'].apply(calc_interval)\n \ninterval_df = pd.DataFrame(\n    result.tolist(),\n    index=result.index,\n    columns=['interval_mean', 'interval_std'],\n).reset_index()\n \ninterval_df.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cust_meta = customers[[\n    'customer_id', 'age', 'club_member_status',\n    'fashion_news_frequency', 'FN', 'Active',\n]].copy()\n \ncust_meta['age']       = cust_meta['age'].fillna(cust_meta['age'].median())\ncust_meta['is_member'] = (cust_meta['club_member_status'] == 'ACTIVE').astype(int)\ncust_meta['news_freq'] = cust_meta['fashion_news_frequency'].map({\n    'NONE'     : 0,\n    'Monthly'  : 1,\n    'Regularly': 2,\n}).fillna(0).astype(int)\n \ncust_meta[['age', 'is_member', 'news_freq']].describe().round(2)\n ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = (\n    churn_df\n    .merge(rfm,         on='customer_id', how='left')\n    .merge(behavior,    on='customer_id', how='left')\n    .merge(interval_df, on='customer_id', how='left')\n    .merge(\n        cust_meta[['customer_id', 'age', 'is_member', 'news_freq', 'FN', 'Active']],\n        on='customer_id',\n        how='left',\n    )\n)\n \nfeatures['interval_mean'] = features['interval_mean'].fillna(features['recency'])\nfeatures['interval_std']  = features['interval_std'].fillna(0)\nfeatures = features.fillna(0)\n \nFEAT_COLS = [c for c in features.columns if c not in ['customer_id', 'churn']]\n \nprint(f\"Samples  : {len(features):,}\")\nprint(f\"Features : {len(FEAT_COLS)}\")\nprint(f\"List     : {', '.join(FEAT_COLS)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":" \nkey_feats = [\n    'recency', 'frequency', 'monetary', 'avg_price',\n    'unique_categories', 'online_ratio', 'interval_mean', 'age',\n]\n \nfig, axes = plt.subplots(2, 4, figsize=(18, 8))\nfig.suptitle('Feature Distribution — Churned vs Retained', fontsize=14, fontweight='bold')\n \nfor ax, feat in zip(axes.flat, key_feats):\n    for label, color, name in [(0, TEAL, 'Retained'), (1, RED, 'Churned')]:\n        vals = features.loc[features['churn'] == label, feat]\n        lo, hi = vals.quantile(0.01), vals.quantile(0.99)\n        ax.hist(vals.clip(lo, hi), bins=50, alpha=0.65,\n                color=color, label=name, density=True)\n    ax.set_title(feat)\n    ax.legend(fontsize=8)\n \nplt.tight_layout()\nplt.show()\n \ngrp  = features.groupby('churn')[key_feats].mean().round(2)\ndiff = ((grp.loc[1] - grp.loc[0]) / grp.loc[0] * 100).round(1)\npd.DataFrame({'Retained': grp.loc[0], 'Churned': grp.loc[1], 'Diff(%)': diff})","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.cluster import KMeans\nfrom sklearn.metrics import silhouette_score\n \nrfm_mat = features[['recency', 'frequency', 'monetary']].copy()\nscaler  = StandardScaler()\nrfm_s   = scaler.fit_transform(rfm_mat)\n \ninertias   = []\nsil_scores = []\n \nfor k in range(2, 9):\n    km  = KMeans(n_clusters=k, random_state=42, n_init=10)\n    lbl = km.fit_predict(rfm_s)\n    inertias.append(km.inertia_)\n    sil_scores.append(\n        silhouette_score(rfm_s, lbl, sample_size=15_000, random_state=42)\n    )\n    print(f\"  K={k}  inertia={km.inertia_:,.0f}  silhouette={sil_scores[-1]:.4f}\")\n \nfig, axes = plt.subplots(1, 2, figsize=(13, 4))\nfig.suptitle('Optimal K Search', fontsize=14, fontweight='bold')\n \naxes[0].plot(range(2,9), inertias,   'o-', color=TEAL,  lw=2, ms=7)\naxes[0].axvline(4, color=RED, ls='--', lw=1.5, label='K=4 selected')\naxes[0].set_title('Elbow Method')\naxes[0].set_xlabel('K')\naxes[0].set_ylabel('Inertia')\naxes[0].legend()\n \naxes[1].plot(range(2,9), sil_scores, 'o-', color=AMBER, lw=2, ms=7)\naxes[1].axvline(4, color=RED, ls='--', lw=1.5, label='K=4 selected')\naxes[1].set_title('Silhouette Score')\naxes[1].set_xlabel('K')\naxes[1].set_ylabel('Score')\naxes[1].legend()\n \nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"K = 4\n\nkm_final = KMeans(n_clusters=K, random_state=42, n_init=10)\nfeatures['cluster'] = km_final.fit_predict(rfm_s)\n \nprofile = features.groupby('cluster').agg(\n    n_customers    = ('customer_id', 'count'),\n    churn_rate     = ('churn',       'mean'),\n    recency_mean   = ('recency',     'mean'),\n    frequency_mean = ('frequency',   'mean'),\n    monetary_mean  = ('monetary',    'mean'),\n    online_ratio   = ('online_ratio','mean'),\n    age_mean       = ('age',         'mean'),\n).round(2)\n \ndef seg_name(row):\n    if row['churn_rate'] < 0.25 and row['frequency_mean'] > 15:\n        return 'VIP Loyal'\n    elif row['churn_rate'] < 0.40:\n        return 'New / Active'\n    elif row['churn_rate'] < 0.65:\n        return 'At Risk'\n    else:\n        return 'High Risk'\n \nprofile['segment'] = profile.apply(seg_name, axis=1)\nfeatures['segment'] = features['cluster'].map(profile['segment'])\n \norder  = profile['churn_rate'].sort_values().index.tolist()\nC_MAP  = {cid: [TEAL, AMBER, ORANGE, RED][i] for i, cid in enumerate(order)}\n \nprofile","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 3, figsize=(18, 5))\nfig.suptitle('Customer Segmentation (K-Means RFM)', fontsize=14, fontweight='bold')\n \nseg_colors = [C_MAP[i] for i in profile.index]\n \nbars = axes[0].bar(\n    profile['segment'],\n    profile['churn_rate'] * 100,\n    color=seg_colors,\n    edgecolor='#0f1117',\n    width=0.5,\n)\nfor bar, v in zip(bars, profile['churn_rate']):\n    axes[0].text(\n        bar.get_x() + bar.get_width()/2,\n        bar.get_height() + 1,\n        f'{v:.0%}',\n        ha='center', fontsize=10, fontweight='bold',\n    )\naxes[0].axhline(\n    features['churn'].mean() * 100,\n    color='white', ls='--', lw=1.2, alpha=0.5, label='Overall avg',\n)\naxes[0].set_title('Churn Rate by Segment')\naxes[0].set_ylabel('Churn Rate (%)')\naxes[0].tick_params(axis='x', labelsize=9, rotation=10)\naxes[0].legend()\n \naxes[1].pie(\n    profile['n_customers'],\n    labels=profile['segment'],\n    colors=seg_colors,\n    autopct='%1.1f%%',\n    startangle=90,\n    wedgeprops={'edgecolor': '#0f1117', 'linewidth': 1.5},\n    textprops={'fontsize': 8},\n)\naxes[1].set_title('Customer Share by Segment')\n \nsample = features.sample(min(8000, len(features)), random_state=42)\nfor cid in profile.index:\n    mask = sample['cluster'] == cid\n    axes[2].scatter(\n        sample.loc[mask, 'recency'],\n        sample.loc[mask, 'frequency'],\n        c=C_MAP[cid], alpha=0.3, s=6,\n        label=profile.loc[cid, 'segment'],\n    )\naxes[2].set_title('Recency vs Frequency (sample 8K)')\naxes[2].set_xlabel('Recency (days)')\naxes[2].set_ylabel('Frequency')\naxes[2].legend(markerscale=3, fontsize=8)\naxes[2].set_xlim(0, 180)\naxes[2].set_ylim(0, sample['frequency'].quantile(0.99))\n \nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport lightgbm as lgb\n \nX = features[FEAT_COLS].astype(np.float32)\ny = features['churn']\n \nX_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y,\n)\n \npos_w = (y_train == 0).sum() / (y_train == 1).sum()\n \nprint(f\"Train : {len(X_train):,}  |  Test : {len(X_test):,}\")\nprint(f\"scale_pos_weight : {pos_w:.2f}\")\n \nparams = {\n    'objective'        : 'binary',\n    'metric'           : 'auc',\n    'learning_rate'    : 0.05,\n    'num_leaves'       : 63,\n    'min_child_samples': 50,\n    'feature_fraction' : 0.8,\n    'bagging_fraction' : 0.8,\n    'bagging_freq'     : 5,\n    'lambda_l1'        : 0.1,\n    'lambda_l2'        : 0.1,\n    'scale_pos_weight' : pos_w,\n    'verbose'          : -1,\n    'random_state'     : 42,\n}\n \ntrain_ds = lgb.Dataset(X_train, label=y_train)\nvalid_ds = lgb.Dataset(X_test,  label=y_test, reference=train_ds)\n \nmodel = lgb.train(\n    params,\n    train_ds,\n    num_boost_round=1000,\n    valid_sets=[train_ds, valid_ds],\n    valid_names=['train', 'valid'],\n    callbacks=[\n        lgb.early_stopping(stopping_rounds=50, verbose=False),\n        lgb.log_evaluation(period=100),\n    ],\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import (\n    roc_auc_score, f1_score, precision_score, recall_score,\n    classification_report, confusion_matrix,\n    roc_curve, precision_recall_curve, average_precision_score,\n)\n \ny_prob = model.predict(X_test)\ny_pred = (y_prob >= 0.5).astype(int)\n \npd.DataFrame({\n    'Metric': ['AUC-ROC', 'Avg. Precision', 'F1-Score', 'Precision', 'Recall'],\n    'Score' : [\n        f\"{roc_auc_score(y_test, y_prob):.4f}\",\n        f\"{average_precision_score(y_test, y_prob):.4f}\",\n        f\"{f1_score(y_test, y_pred):.4f}\",\n        f\"{precision_score(y_test, y_pred):.4f}\",\n        f\"{recall_score(y_test, y_pred):.4f}\",\n    ]\n})","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"auc = roc_auc_score(y_test, y_prob)\nap  = average_precision_score(y_test, y_prob)\n \nfig, axes = plt.subplots(1, 3, figsize=(18, 5))\nfig.suptitle('LightGBM Model Evaluation', fontsize=14, fontweight='bold')\n \nfpr, tpr, _ = roc_curve(y_test, y_prob)\naxes[0].plot(fpr, tpr, color=TEAL, lw=2.5, label=f'AUC = {auc:.4f}')\naxes[0].fill_between(fpr, tpr, alpha=0.1, color=TEAL)\naxes[0].plot([0,1], [0,1], 'w--', lw=1, alpha=0.4)\naxes[0].set_title('ROC Curve')\naxes[0].set_xlabel('False Positive Rate')\naxes[0].set_ylabel('True Positive Rate')\naxes[0].legend()\n \nprec_c, rec_c, _ = precision_recall_curve(y_test, y_prob)\naxes[1].plot(rec_c, prec_c, color=AMBER, lw=2.5, label=f'AP = {ap:.4f}')\naxes[1].fill_between(rec_c, prec_c, alpha=0.1, color=AMBER)\naxes[1].axhline(y_test.mean(), color='white', ls='--', lw=1, alpha=0.4,\n                label='Random baseline')\naxes[1].set_title('Precision-Recall Curve')\naxes[1].set_xlabel('Recall')\naxes[1].set_ylabel('Precision')\naxes[1].legend()\n \nfi = pd.DataFrame({\n    'feature'   : model.feature_name(),\n    'importance': model.feature_importance('gain'),\n}).sort_values('importance', ascending=True).tail(12)\n \nbar_colors = [\n    RED  if i == len(fi)-1 else\n    TEAL if i == len(fi)-2 else\n    '#3d4451'\n    for i in range(len(fi))\n]\n \naxes[2].barh(fi['feature'], fi['importance'],\n             color=bar_colors, edgecolor='#0f1117', height=0.65)\naxes[2].set_title('Feature Importance (Gain) — Top 12')\naxes[2].set_xlabel('Importance')\n \nplt.tight_layout()\nplt.show()\n ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cm = confusion_matrix(y_test, y_pred)\n \nfig, ax = plt.subplots(figsize=(6, 5))\nsns.heatmap(\n    cm,\n    annot=True,\n    fmt='d',\n    cmap='Blues',\n    ax=ax,\n    xticklabels=['Retained', 'Churned'],\n    yticklabels=['Retained', 'Churned'],\n    annot_kws={'size': 14, 'weight': 'bold'},\n    linewidths=0.5,\n)\nax.set_title('Confusion Matrix', fontsize=13, fontweight='bold')\nax.set_ylabel('Actual')\nax.set_xlabel('Predicted')\nplt.tight_layout()\nplt.show()\n \ntn, fp, fn, tp = cm.ravel()\npd.DataFrame({\n    'Case': ['TP', 'FN', 'FP', 'TN'],\n    'Count': [tp, fn, fp, tn],\n    'Business meaning': [\n        'Predicted churn & actually churned  → coupon prevents loss',\n        'Predicted retained & actually churned  → missed, LTV lost',\n        'Predicted churn & actually retained  → wasted coupon cost',\n        'Predicted retained & actually retained  → no action needed',\n    ]\n})","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def sample_size(p_control, mde, alpha=0.05, power=0.80):\n    p_treat = p_control + mde\n    p_pool  = (p_control + p_treat) / 2\n    z_a     = stats.norm.ppf(1 - alpha)\n    z_b     = stats.norm.ppf(power)\n    n = (\n        z_a * math.sqrt(2 * p_pool * (1 - p_pool)) +\n        z_b * math.sqrt(p_control*(1-p_control) + p_treat*(1-p_treat))\n    ) ** 2 / mde**2\n    return math.ceil(n)\n \nP_CONTROL   = 0.20\nMDE         = 0.05\nn_per_group = sample_size(P_CONTROL, MDE)\n \nboundary    = features[features['segment'] == 'At Risk']\n \npd.DataFrame({\n    'Setting': [\n        'Target segment',\n        'Treatment',\n        'Primary metric',\n        'Control baseline',\n        'MDE',\n        'Alpha',\n        'Power',\n        'Required n (per group)',\n        'At-Risk pool size',\n        'Feasible',\n    ],\n    'Value': [\n        'At Risk',\n        '15% discount coupon',\n        '30-day repurchase rate',\n        f'{P_CONTROL:.0%}',\n        f'+{MDE:.0%}',\n        '0.05 (one-tailed)',\n        '0.80',\n        f'{n_per_group:,}',\n        f'{len(boundary):,}',\n        'Yes' if len(boundary) >= n_per_group * 2 else 'No — insufficient pool',\n    ]\n})","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from statsmodels.stats.proportion import proportions_ztest\n\nnp.random.seed(42)\nN = min(n_per_group, len(boundary) // 2)\n\nctrl  = np.random.binomial(1, 0.200, size=N)\ntreat = np.random.binomial(1, 0.258, size=N)\n\np_ctrl  = ctrl.mean()\np_treat = treat.mean()\nlift    = p_treat - p_ctrl\n\nz_stat, p_val = proportions_ztest(\n    [treat.sum(), ctrl.sum()],\n    [N, N],\n    alternative='larger',\n)\n\npd.DataFrame({\n    'Metric': [\n        'Control conversion',\n        'Treatment conversion',\n        'Absolute lift',\n        'Relative lift',\n        'z-statistic',\n        'p-value',\n        'Result',\n    ],\n    'Value': [\n        f'{p_ctrl:.2%}  ({ctrl.sum()}/{N})',\n        f'{p_treat:.2%}  ({treat.sum()}/{N})',\n        f'+{lift:.2%}',\n        f'+{lift/p_ctrl:.1%}',\n        f'{z_stat:.3f}',\n        f'{p_val:.4f}',\n        'Significant' if p_val < 0.05 else 'Not significant',\n    ]\n})","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def wilson_ci(p, n, z=1.96):\n    d = 1 + z**2 / n\n    c = (p + z**2/(2*n)) / d\n    m = z * math.sqrt(p*(1-p)/n + z**2/(4*n**2)) / d\n    return c - m, c + m\n \nfig, axes = plt.subplots(1, 2, figsize=(13, 5))\nfig.suptitle('A/B Test Result (Simulated)', fontsize=14, fontweight='bold')\n \nfor i, (name, p) in enumerate([('Control\\n(no coupon)', p_ctrl),\n                                 ('Treatment\\n(coupon)',   p_treat)]):\n    c = BLUE if i == 0 else TEAL\n    axes[0].bar(i, p*100, color=c, width=0.4, edgecolor='#0f1117')\n    lo, hi = wilson_ci(p, N)\n    axes[0].errorbar(\n        i, p*100,\n        yerr=[[(p-lo)*100], [(hi-p)*100]],\n        fmt='none', color='white',\n        capsize=8, capthick=2, elinewidth=2,\n    )\n    axes[0].text(i, p*100 + 1.5, f'{p:.1%}',\n                 ha='center', fontweight='bold', fontsize=12)\n \naxes[0].set_xticks([0, 1])\naxes[0].set_xticklabels(['Control\\n(no coupon)', 'Treatment\\n(coupon)'])\naxes[0].set_title('Conversion Rate (95% CI)')\naxes[0].set_ylabel('Conversion Rate (%)')\naxes[0].set_ylim(0, max(p_ctrl, p_treat) * 100 * 1.4)\naxes[0].grid(axis='y', alpha=0.4)\n \nboot_lifts = np.array([\n    np.random.choice(treat, N, replace=True).mean() -\n    np.random.choice(ctrl,  N, replace=True).mean()\n    for _ in range(5000)\n])\nci_lo, ci_hi = np.percentile(boot_lifts, [2.5, 97.5])\n \naxes[1].hist(boot_lifts*100, bins=60, color=AMBER,\n             edgecolor='#0f1117', linewidth=0.3, density=True)\naxes[1].axvline(0,         color=RED,   lw=2,    ls='--', label='No effect')\naxes[1].axvline(lift*100,  color=TEAL,  lw=2,    label=f'Observed lift={lift:.2%}')\naxes[1].axvline(ci_lo*100, color='white', lw=1.5, ls=':',\n                label=f'95% CI [{ci_lo:.2%}, {ci_hi:.2%}]')\naxes[1].axvline(ci_hi*100, color='white', lw=1.5, ls=':')\naxes[1].set_title('Bootstrap Lift Distribution (n=5,000)')\naxes[1].set_xlabel('Lift (%p)')\naxes[1].set_ylabel('Density')\naxes[1].legend(fontsize=8)\n \nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"strategies = {\n    'VIP Loyal': [\n        ('VIP tier upgrade',      'Exclusive season preview + free shipping + dedicated CS',  'Churn rate -8%p'),\n        ('Personalized curation', 'Weekly style recommendation based on purchase history',     'Repurchase +15%p'),\n        ('Double points event',   'Points x2 campaign to reinforce long-term loyalty',        'LTV +32%'),\n    ],\n    'New / Active': [\n        ('Onboarding series',     'Coupon sequence at D+7, D+14, D+30',                       'Repurchase +22%p'),\n        ('First repurchase push', '10% coupon within 72h of first purchase (urgency)',         'CVR +18%p'),\n        ('Category expansion',    'Cross-sell via related product exposure',                   'AOV +28%'),\n    ],\n    'At Risk': [\n        ('Win-back campaign',     '60-day inactive → personalized message + 15% coupon',      'Return rate +19%p'),\n        ('Wishlist reminder',     'Re-surface saved / recently viewed items + low-stock alert','CVR +12%p'),\n        ('Price drop alert',      'Push notification when browsed item goes on sale',          'CVR +9%p'),\n    ],\n    'High Risk': [\n        ('Last-chance offer',     'Up to 25% off + free returns to re-activate',              'Return rate +11%p'),\n        ('Channel switch',        'Incentivize app-to-web or web-to-app cross-channel switch', 'Churn -6%p'),\n        ('Survey + reward',       '30-sec exit survey + 500pt reward to capture churn reason', 'Insight gain'),\n    ],\n}\n \nfor seg, actions in strategies.items():\n    print(f\"\\n[ {seg} ]\")\n    for i, (action, desc, effect) in enumerate(actions, 1):\n        print(f\"  {i}. {action}\")\n        print(f\"     {desc}\")\n        print(f\"     -> {effect}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"recovery_rates = {\n    'VIP Loyal'  : 0.08,\n    'New / Active': 0.19,\n    'At Risk'    : 0.19,\n    'High Risk'  : 0.11,\n}\n \nprofile['est_recovery'] = profile.apply(\n    lambda r: r['n_customers'] * r['churn_rate'] * recovery_rates.get(r['segment'], 0),\n    axis=1,\n)\n \nfig, axes = plt.subplots(1, 2, figsize=(14, 5))\nfig.suptitle('Retention Priority Matrix', fontsize=14, fontweight='bold')\n \nseg_colors_list = [C_MAP[i] for i in profile.index]\n \naxes[0].scatter(\n    profile['churn_rate'] * 100,\n    profile['frequency_mean'],\n    s=profile['n_customers'] / 30,\n    c=seg_colors_list,\n    alpha=0.75,\n    edgecolors='white',\n    linewidth=1.5,\n    zorder=3,\n)\nfor idx, row in profile.iterrows():\n    axes[0].annotate(\n        row['segment'],\n        (row['churn_rate']*100, row['frequency_mean']),\n        xytext=(8, 5), textcoords='offset points', fontsize=9,\n    )\naxes[0].set_title('Churn Rate vs Frequency  (bubble = customer count)')\naxes[0].set_xlabel('Churn Rate (%)')\naxes[0].set_ylabel('Avg. Frequency')\n \nbars = axes[1].barh(\n    profile['segment'],\n    profile['est_recovery'],\n    color=seg_colors_list,\n    edgecolor='#0f1117',\n    height=0.5,\n)\nfor bar, v in zip(bars, profile['est_recovery']):\n    axes[1].text(v + 50, bar.get_y() + bar.get_height()/2,\n                 f'+{v:,.0f}', va='center', fontsize=9)\naxes[1].set_title('Estimated Recoverable Customers per Campaign')\naxes[1].set_xlabel('Customers')\naxes[1].set_xlim(0, profile['est_recovery'].max() * 1.5)\n \nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"total_at_risk  = (profile['n_customers'] * profile['churn_rate']).sum()\ntotal_recovery = profile['est_recovery'].sum()\n \npd.DataFrame({\n    'Summary': [\n        'Model AUC-ROC',\n        'Overall churn rate',\n        'At-risk customers',\n        'Estimated recoverable',\n        'A/B test n (per group)',\n        'Top churn driver',\n        'Key insight 1',\n        'Key insight 2',\n    ],\n    'Value': [\n        f\"{roc_auc_score(y_test, y_prob):.4f}\",\n        f\"{features['churn'].mean():.2%}\",\n        f\"{total_at_risk:,.0f}\",\n        f\"{total_recovery:,.0f}  ({total_recovery/total_at_risk:.1%})\",\n        f\"{n_per_group:,}\",\n        \"Recency  (Feature Importance #1)\",\n        \"Higher online ratio -> lower churn  (app engagement is key)\",\n        \"Club members churn significantly less  (membership lock-in works)\",\n    ]\n})","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import joblib, json\nimport pandas as pd\n\n# ① LightGBM 모델 저장\njoblib.dump(model, 'lgbm_churn_model.pkl')\n\n# ② 세그먼트 결과 저장\nprofile[['segment','n_customers','churn_rate',\n         'recency_mean','frequency_mean',\n         'monetary_mean','est_recovery']].to_csv('segments.csv', index=False)\n\n# ③ 피처 컬럼 목록 저장 (예측 시 순서 맞추기 위해)\nwith open('feature_cols.json', 'w') as f:\n    json.dump(FEAT_COLS, f)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}