{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":31254,"databundleVersionId":3103714}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# 导入必要的库\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\n\n# 解决中文乱码问题 (Kaggle环境)\nplt.rcParams['font.sans-serif'] = ['DejaVu Sans']\nplt.rcParams['axes.unicode_minus'] = False \n\n# 数据加载 (Kaggle路径，请确保路径正确)\n# 注意：请根据实际Kaggle比赛名称调整路径\narticles_df = pd.read_csv('/kaggle/input/competitions/h-and-m-personalized-fashion-recommendations/articles.csv')\ncustomers_df = pd.read_csv('/kaggle/input/competitions/h-and-m-personalized-fashion-recommendations/customers.csv')\ntransactions_df = pd.read_csv('/kaggle/input/competitions/h-and-m-personalized-fashion-recommendations/transactions_train.csv')\n\n# 日期格式转换 (关键步骤)\ntransactions_df['t_dat'] = pd.to_datetime(transactions_df['t_dat'])\n\n# 数据概览（EDA第一步）\nprint(\"交易表形状：\", transactions_df.shape)\nprint(\"用户表形状：\", customers_df.shape)\nprint(\"商品表形状：\", articles_df.shape)\n\n# 查看字段\nprint(\"\\n交易表字段：\", transactions_df.columns.tolist())\nprint(\"用户表字段：\", customers_df.columns.tolist())\nprint(\"商品表字段：\", articles_df.columns.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-17T08:02:38.191267Z","iopub.execute_input":"2026-04-17T08:02:38.191595Z","iopub.status.idle":"2026-04-17T08:03:55.764946Z","shell.execute_reply.started":"2026-04-17T08:02:38.191565Z","shell.execute_reply":"2026-04-17T08:03:55.763186Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"交易表：31788324行，5个字段（购买记录）\n用户表：1371980用户，7个字段（用户信息）\n商品表：105542商品，25个字段（服装属性）","metadata":{}},{"cell_type":"code","source":"# 看交易表前5行\nprint(\"交易表前5行：\")\nprint(transactions_df.head())\n\n# 看用户表前5行\nprint(\"\\n用户表前5行：\")\nprint(customers_df.head())\n\n# 看商品表前5行\nprint(\"\\n商品表前5行：\")\nprint(articles_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T11:01:16.786163Z","iopub.execute_input":"2026-04-08T11:01:16.786480Z","iopub.status.idle":"2026-04-08T11:01:16.810423Z","shell.execute_reply.started":"2026-04-08T11:01:16.786446Z","shell.execute_reply":"2026-04-08T11:01:16.809585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ---------------------- 1.1 Daily/Monthly Sales Line Chart ----------------------\n# 按天统计购买量\ndaily_sales = transactions_df.groupby('t_dat').size().reset_index(name='purchase_count')\n\n# 按月统计购买量\nmonthly_sales = transactions_df.groupby(transactions_df['t_dat'].dt.to_period('M')).size().reset_index(name='purchase_count')\nmonthly_sales['t_dat'] = monthly_sales['t_dat'].dt.to_timestamp()\n\n# 创建画布\nfig, (ax1, ax2) = plt.subplots(2, 1, figsize=(16, 10))\n\n# 每日购买量折线图\nsns.lineplot(data=daily_sales, x='t_dat', y='purchase_count', ax=ax1, color='#1f77b4')\nax1.set_title('Daily Purchase Quantity Trend (2018-2020)', fontsize=14, pad=20) # 标题：每日购买量趋势\nax1.set_xlabel('Date', fontsize=12) # 标签：日期\nax1.set_ylabel('Purchase Count', fontsize=12) # 标签：购买次数\nax1.grid(alpha=0.3) # 添加网格\n\n# 每月购买量折线图\nsns.lineplot(data=monthly_sales, x='t_dat', y='purchase_count', ax=ax2, color='#ff7f0e', marker='o')\nax2.set_title('Monthly Purchase Quantity Trend', fontsize=14, pad=20) # 标题：每月购买量趋势\nax2.set_xlabel('Month', fontsize=12) # 标签：月份\nax2.set_ylabel('Purchase Count', fontsize=12) # 标签：购买次数\nax2.grid(alpha=0.3) # 添加网格\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T11:01:16.811252Z","iopub.execute_input":"2026-04-08T11:01:16.811521Z","iopub.status.idle":"2026-04-08T11:01:20.811514Z","shell.execute_reply.started":"2026-04-08T11:01:16.811494Z","shell.execute_reply":"2026-04-08T11:01:20.810393Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"核心规律：用户购买行为呈现显著的年度季节性 + 周度周期性，年度维度 Q3-Q4 为销售旺季，2019 年 9 月达销量峰值，Q1-Q2 为淡季；2020 年受疫情影响出现销量低谷与后续反弹。\n异常洞察：日度图中的极端尖峰对应大促节点，大促日销量为平日 3-4 倍，是驱动短期销量爆发的核心因素。\n建模指导：时间是核心预测特征，需在模型中加入月份、是否大促等时间特征，捕捉周期与事件对购买转化的影响。","metadata":{}},{"cell_type":"code","source":"# ---------------------- 1.2 Weekday-Hour Heatmap (Periodicity) ----------------------\n# 提取星期特征 (0=周一，6=周日)\ntransactions_df['weekday'] = transactions_df['t_dat'].dt.dayofweek\n# 按星期聚合购买量\nweekday_agg = transactions_df.groupby('weekday').size().reset_index(name='purchase_count')\n# 为了在barplot中使用hue，我们创建一个临时的分类列（每个柱子作为一个类别）\nweekday_agg['weekday_label'] = weekday_agg['weekday'].apply(lambda x: f'Day {x}')\n\n# 修复后的代码：指定hue参数 + palette\nplt.figure(figsize=(10, 6))\n# hue='weekday_label' 为每个柱子分配不同颜色，消除警告\nsns.barplot(\n    x='weekday_label', \n    y='purchase_count', \n    hue='weekday_label',  # 关键修复：指定hue\n    data=weekday_agg, \n    palette='coolwarm',\n    legend=False  # 关闭图例，因为x和hue是同一列\n)\nplt.title('Purchase Count by Weekday', fontsize=14, pad=20)\nplt.xlabel('Weekday (0=Mon, 6=Sun)', fontsize=12)\nplt.ylabel('Purchase Count', fontsize=12)\nplt.xticks(ticks=range(7), labels=['Mon', 'Tue', 'Wed', 'Thu', 'Fri', 'Sat', 'Sun'])\nplt.grid(alpha=0.3, axis='y')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T11:01:20.813008Z","iopub.execute_input":"2026-04-08T11:01:20.813387Z","iopub.status.idle":"2026-04-08T11:01:22.697154Z","shell.execute_reply.started":"2026-04-08T11:01:20.813335Z","shell.execute_reply":"2026-04-08T11:01:22.696349Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"核心规律：用户在周六（Sat）购买量达到全周峰值，周三（Wed）、周四（Thu）紧随其后，呈现 “周末集中购物” 的消费节奏。\n业务洞察：周末是用户转化的黄金时段，用户休闲时间购物意愿最强；工作日（周一至周二）购买量平稳，属于日常消费期。\n建模指导：“星期几” 是核心预测特征，需在模型中加入周几、是否周末等特征，精准捕捉不同时段的购买差异。","metadata":{}},{"cell_type":"code","source":"# ---------------------- 2.1 High-Value Region Bar Chart ----------------------\n# 合并交易数据和用户数据\ncustomer_trans = transactions_df.merge(customers_df, on='customer_id', how='left')\n\n# 按邮编统计关键指标\npostal_stats = customer_trans.groupby('postal_code').agg(\n    user_count=('customer_id', 'nunique'), # 用户数\n    total_spend=('price', 'sum'), # 总消费金额\n    purchase_count=('t_dat', 'count') # 购买次数\n).reset_index()\n\n# 计算客单价\npostal_stats['avg_spend_per_user'] = postal_stats['total_spend'] / postal_stats['user_count']\n\n# 选取TOP 15高价值区域 (按总消费金额)\ntop_postal = postal_stats.nlargest(15, 'total_spend')\n\n# 绘制双轴柱状图\nfig, ax1 = plt.subplots(figsize=(14, 7))\n\n# 总消费金额 (左轴)\nsns.barplot(data=top_postal, x='postal_code', y='total_spend', ax=ax1, color='#1f77b4', alpha=0.8)\nax1.set_title('Top 15 High-Value Regions by Total Spend', fontsize=14, pad=20) # 标题：TOP 15高价值区域 (按总消费)\nax1.set_xlabel('Postal Code (Region)', fontsize=12) # 标签：邮编 (地域)\nax1.set_ylabel('Total Spend', fontsize=12, color='#1f77b4') # 标签：总消费金额\nax1.tick_params(axis='x', rotation=45) # 旋转X轴标签\n\n# 用户数 (右轴)\nax2 = ax1.twinx()\nsns.lineplot(data=top_postal, x='postal_code', y='user_count', ax=ax2, color='#ff7f0e', marker='o', linewidth=2)\nax2.set_ylabel('User Count', fontsize=12, color='#ff7f0e') # 标签：用户数\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T11:01:22.698175Z","iopub.execute_input":"2026-04-08T11:01:22.698442Z","iopub.status.idle":"2026-04-08T11:01:58.663008Z","shell.execute_reply.started":"2026-04-08T11:01:22.698416Z","shell.execute_reply":"2026-04-08T11:01:58.662014Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"核心规律：消费呈现极度集中的头部效应，TOP1 高价值区域贡献了远超其他区域的总消费额，其余 14 个区域消费规模极低，用户数也无明显差异，说明业务收入高度依赖少数核心区域。\n业务洞察：头部区域是品牌的核心高价值客群聚集地，是营收的核心支柱；其余区域用户基数相近但消费能力弱，属于低价值潜力市场。\n建模与业务指导：地域是强区分度特征，需对头部区域做单独特征标记，提升模型对高价值用户的识别能力；业务上可针对性深耕头部区域的用户运营，同时挖掘低消费区域的增长潜力。","metadata":{}},{"cell_type":"code","source":"# ---------------------- 3.1 User Age Distribution (Histogram) ----------------------\n# 用户年龄分布直方图\nplt.figure(figsize=(12, 6))\nsns.histplot(data=customers_df, x='age', bins=40, kde=True, color='#2ca02c')\nplt.title('User Age Distribution', fontsize=14, pad=20)  # 标题：用户年龄分布\nplt.xlabel('Age', fontsize=12)  # 标签：年龄\nplt.ylabel('User Count', fontsize=12)  # 标签：用户数\nplt.grid(alpha=0.3)\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T11:01:58.664155Z","iopub.execute_input":"2026-04-08T11:01:58.664423Z","iopub.status.idle":"2026-04-08T11:02:04.810863Z","shell.execute_reply.started":"2026-04-08T11:01:58.664397Z","shell.execute_reply":"2026-04-08T11:02:04.809758Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"核心规律：用户呈右偏分布，20-30 岁是绝对核心主力客群，人数占比最高；30-50 岁为次要客群，50 岁以上用户数量快速衰减。\n业务洞察：** 年轻群体（20-30 岁）** 是品牌基石，消费意愿与购买力最强；中老年用户为补充客群，需差异化匹配商品偏好。\n建模指导：年龄是核心分层特征，需按年龄段（如 25 岁以下、25-35 岁）做用户分群，提升推荐与转化的精准度。","metadata":{}},{"cell_type":"code","source":"# ---------------------- 3.2 User Activity & Spend (Boxplot & User Stratification) ----------------------\n# 计算每个用户的活跃度和总消费\nuser_profile = transactions_df.groupby('customer_id').agg(\n    purchase_freq=('t_dat', 'count'),  # 活跃度：购买次数\n    total_spend=('price', 'sum')  # 总消费金额\n).reset_index()\n\n# 合并用户年龄信息\nuser_profile = user_profile.merge(customers_df[['customer_id', 'age']], on='customer_id', how='left')\n\n# 用户分层 (基于年龄和消费金额)\nuser_profile['age_group'] = pd.cut(\n    user_profile['age'], \n    bins=[0, 25, 35, 45, 100], \n    labels=['<25', '25-35', '35-45', '>45']\n)\nuser_profile['spend_level'] = pd.cut(\n    user_profile['total_spend'], \n    bins=3, \n    labels=['Low', 'Medium', 'High']\n)\n\n# 绘制箱线图\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(18, 6))\n\n# 1. 不同年龄组的购买频率（已修复：指定hue+关闭图例）\nsns.boxplot(\n    data=user_profile, \n    x='age_group', \n    y='purchase_freq', \n    ax=ax1, \n    hue='age_group',  # 必选：指定hue，消除palette传参警告\n    palette='Set2', \n    legend=False      # 关闭图例，避免图表冗余\n)\nax1.set_title('Purchase Frequency by Age Group', fontsize=14, pad=20)  # 标题：不同年龄组的购买频率\nax1.set_xlabel('Age Group', fontsize=12)  # 标签：年龄组\nax1.set_ylabel('Purchase Frequency (Count)', fontsize=12)  # 标签：购买频率\nax1.grid(alpha=0.3)\n\n# 2. 不同消费层级的用户分布（修复：添加hue+关闭图例，彻底消除警告）\nspend_dist = user_profile['spend_level'].value_counts().reset_index()\nspend_dist.columns = ['spend_level', 'user_count']\nsns.barplot(\n    data=spend_dist, \n    x='spend_level', \n    y='user_count', \n    ax=ax2, \n    hue='spend_level',  # 新增：指定hue，匹配x轴变量\n    palette='pastel', \n    legend=False        # 新增：关闭图例\n)\nax2.set_title('User Count by Spend Level', fontsize=14, pad=20)  # 标题：不同消费层级的用户数\nax2.set_xlabel('Spend Level', fontsize=12)  # 标签：消费层级\nax2.set_ylabel('User Count', fontsize=12)  # 标签：用户数\nax2.grid(alpha=0.3, axis='y')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T11:02:04.812486Z","iopub.execute_input":"2026-04-08T11:02:04.812875Z","iopub.status.idle":"2026-04-08T11:02:19.659714Z","shell.execute_reply.started":"2026-04-08T11:02:04.812842Z","shell.execute_reply":"2026-04-08T11:02:19.658801Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"核心规律：\n购买频率：>45 岁群体购买频率中位数最高，其次是 35-45 岁、25-35 岁，<25 岁最低；全年龄段存在少量高频率购买（离群点），说明高活跃用户跨年龄段分布。\n消费层级：用户呈极端单峰分布，低消费（Low）用户占据绝对主体，中（Medium）、高（High）消费用户数量极少。\n业务洞察：\n中高龄群体（35 岁 +）是购买频率的核心贡献者，需重点维护其复购行为；年轻用户（<25 岁）消费频次待提升。\n业务以 “大众低消费” 为主，高消费客群稀缺，需通过精准运营挖掘高价值用户潜力。\n建模指导：需将年龄组、消费层级作为核心分类特征，针对低消费用户做转化提升，对高频率 / 高消费用户做精准留存。","metadata":{}},{"cell_type":"code","source":"# ---------------------- 4.1 Product Category Sales (Bar Chart) ----------------------\n# 合并交易数据和商品数据\nproduct_trans = transactions_df.merge(articles_df[['article_id', 'product_group_name', 'product_type_name']], on='article_id', how='left')\n\n# 按商品大类统计销量\ncategory_sales = product_trans['product_group_name'].value_counts().reset_index()\ncategory_sales.columns = ['product_group', 'sales_count']\ntop_categories = category_sales.head(10) # 取TOP 10大类\n\n# 绘制柱状图（修复警告：将x赋值给hue，设置legend=False，保持原视觉效果）\nplt.figure(figsize=(14, 6))\n# 修复点：hue=top_categories['product_group'], legend=False\nsns.barplot(data=top_categories, x='product_group', y='sales_count', hue=top_categories['product_group'], palette='viridis', legend=False)\nplt.title('Top 10 Product Category Sales Volume', fontsize=14, pad=20) # 标题：TOP 10商品大类销量\nplt.xlabel('Product Category', fontsize=12) # 标签：商品大类\nplt.ylabel('Sales Count', fontsize=12) # 标签：销量\nplt.xticks(rotation=45)\nplt.grid(alpha=0.3, axis='y')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T11:02:19.660977Z","iopub.execute_input":"2026-04-08T11:02:19.661291Z","iopub.status.idle":"2026-04-08T11:02:27.731698Z","shell.execute_reply.started":"2026-04-08T11:02:19.661256Z","shell.execute_reply":"2026-04-08T11:02:27.730654Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"核心规律：Garment Upper body（上装） 是绝对爆款类目，销量断层领先，远超第二名下装；销量呈头部集中、长尾分布，TOP3 类目（上装 / 下装 / 全身装）贡献了绝大多数销量，其余类目规模差距显著。\n业务洞察：上装是品牌营收核心支柱，下装、全身装为核心补充；泳装、内衣等为细分刚需类目，鞋履、配饰等为长尾品类。\n建模指导：商品类目是强特征，需对头部类目做重点建模，推荐系统优先承接上装等爆款的流量需求，同时挖掘长尾类目增长空间。","metadata":{}},{"cell_type":"code","source":"# ---------------------- 4.2 Price Distribution & Top Products (Pie Chart) ----------------------\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(16, 6))\n\n# 价格分布直方图（无警告，无需修改）\nsns.histplot(data=transactions_df, x='price', bins=50, kde=True, ax=ax1, color='#d62728')\nax1.set_title('Product Price Distribution', fontsize=14, pad=20) # 标题：商品价格分布\nax1.set_xlabel('Price', fontsize=12) # 标签：价格\nax1.set_ylabel('Frequency', fontsize=12) # 标签：频次\nax1.grid(alpha=0.3)\n\n# TOP 10热门商品饼图\ntop_products = product_trans['article_id'].value_counts().head(10).reset_index(name='sales_count')\ntop_products = top_products.merge(articles_df[['article_id', 'prod_name']], on='article_id', how='left')\ntop_products['short_name'] = top_products['prod_name'].str[:10] + '...' # 缩短名称\n\nax2.pie(\n    top_products['sales_count'], \n    labels=top_products['short_name'], \n    autopct='%1.1f%%', \n    startangle=90, \n    colors=sns.color_palette('pastel')\n)\nax2.set_title('Top 10 Best-Selling Products', fontsize=14, pad=20) # 标题：TOP 10热门商品\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-08T11:02:27.733006Z","iopub.execute_input":"2026-04-08T11:02:27.733370Z","iopub.status.idle":"2026-04-08T11:04:37.582840Z","shell.execute_reply.started":"2026-04-08T11:02:27.733312Z","shell.execute_reply":"2026-04-08T11:04:37.581813Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"核心规律：\n价格：呈右偏分布，绝大多数商品集中在低价区间，高价商品占比极少，用户消费以高性价比产品为主。\n爆款：TOP10 商品贡献了显著销量份额，头部商品（如 Jade HW Sk...）单类占比超 17%，形成绝对爆款效应。\n业务洞察：低价策略是核心基础，需保障刚需低价款的库存与供应；爆款商品是流量引擎，需重点维护其供应链与上新节奏，同时适度挖掘长尾商品满足差异化需求。\n建模指导：价格区间是核心特征，需分档处理；推荐系统需向爆款倾斜，同时平衡长尾推荐，避免商品同质化。","metadata":{}},{"cell_type":"code","source":"# ---------------------- 5.1 User Conversion Funnel (Browsing -> Buying) ----------------------\n# 注：H&M数据集无浏览/点击数据，我们基于购买行为构建漏斗\n# 定义虚拟环节：总用户 -> 有购买行为用户 -> 多次购买用户 (复购用户)\ntotal_users = customers_df['customer_id'].nunique() # 总触达用户数\nbuy_users = transactions_df['customer_id'].nunique() # 有购买行为用户数\nrepeat_buy_users = transactions_df[transactions_df.duplicated('customer_id', keep=False)]['customer_id'].nunique() # 复购用户数\n\n# 构建漏斗数据\nfunnel_data = pd.DataFrame({\n    'stage': ['Total Users (Reach)', 'Purchase Users', 'Repeat Purchase Users'],\n    'user_count': [total_users, buy_users, repeat_buy_users]\n})\n\n# 计算转化率\nfunnel_data['conversion_rate'] = funnel_data['user_count'] / funnel_data['user_count'].shift(1).fillna(total_users)\n\n# 绘制Plotly漏斗图 (交互式)\nfig = go.Figure(go.Funnel(\n    y=funnel_data['stage'],\n    x=funnel_data['user_count'],\n    textposition=\"inside\",\n    textinfo=\"value+percent previous\", # 显示数值和上一级转化率\n    marker={\"color\": [\"#1f77b4\", \"#ff7f0e\", \"#2ca02c\"]},\n    connector={\"line\": {\"color\": \"black\", \"dash\": \"dot\"}}\n))\n\nfig.update_layout(\n    title='User Conversion Funnel (Reach -> Purchase)', # 标题：用户转化漏斗\n    title_x=0.5,\n    width=800,\n    height=500\n)\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-17T08:04:50.020114Z","iopub.execute_input":"2026-04-17T08:04:50.021239Z","iopub.status.idle":"2026-04-17T08:05:14.405211Z","shell.execute_reply.started":"2026-04-17T08:04:50.021196Z","shell.execute_reply":"2026-04-17T08:05:14.404161Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"核心规律：三层转化环节留存率极高，触达用户到购买用户转化率 99%，购买到复购用户转化率 90%，整体转化链路顺畅，流失率极低。\n业务洞察：产品转化能力极强，用户从触达到购买、复购的信任度高；复购率 90% 显示用户忠诚度高，业务具备稳健的复购基本盘。\n建模指导：需重点关注复购用户特征建模，挖掘高留存、高复购的用户偏好，用于精准推荐与用户生命周期价值运营。","metadata":{}}]}