{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":31254,"databundleVersionId":3103714}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ============================================================\n# H&M Weekly Demand Forecasting by Category and Colour\n# Alberto Caretto — Student ID 25125771\n# MADSC201 — Machine Learning and AI — March 2026\n# ============================================================\n# This notebook reproduces all steps of the ML analysis\n# presented in Part 2 of the course project.\n# Dataset: H&M Group official Kaggle dataset (2022)\n# 31,788,324 real transactions — Sep 2018 to Sep 2020\n# ============================================================\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.tree import DecisionTreeRegressor, plot_tree\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_absolute_error, r2_score\nimport warnings\nwarnings.filterwarnings('ignore')\n\nPATH = '/kaggle/input/competitions/h-and-m-personalized-fashion-recommendations/'\n\ntransactions = pd.read_csv(PATH + 'transactions_train.csv', parse_dates=['t_dat'])\narticles     = pd.read_csv(PATH + 'articles.csv')\n\nprint(f'Transactions loaded: {len(transactions):,}')\nprint(f'Articles loaded:     {len(articles):,}')\nprint(f'Date range: {transactions[\"t_dat\"].min().date()} → {transactions[\"t_dat\"].max().date()}')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-20T19:39:30.915219Z","iopub.execute_input":"2026-03-20T19:39:30.915507Z","iopub.status.idle":"2026-03-20T19:40:11.201917Z","shell.execute_reply.started":"2026-03-20T19:39:30.915487Z","shell.execute_reply":"2026-03-20T19:40:11.201220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# STEP 1: Filter top 8 garment groups and top 3 colours\n# STEP 2: Aggregate transactions to weekly units per combination\n# ============================================================\n\nTOP8_GROUPS  = ['Jersey Fancy', 'Jersey Basic', 'Under-, Nightwear', 'Trousers',\n                'Swimwear', 'Blouses', 'Knitwear', 'Dresses Ladies']\nTOP3_COLOURS = ['Black', 'White', 'Dark Blue']\n\n# Filter articles to top 8 groups and top 3 colours only\narticles_slim = articles[\n    ['article_id', 'garment_group_name', 'colour_group_name']\n].copy()\narticles_slim = articles_slim[\n    articles_slim['garment_group_name'].isin(TOP8_GROUPS) &\n    articles_slim['colour_group_name'].isin(TOP3_COLOURS)\n]\n\n# Join transactions with filtered articles\ntx = transactions.merge(articles_slim, on='article_id', how='inner')\n\n# Aggregate to weekly units per garment group + colour\ntx['week'] = tx['t_dat'].dt.to_period('W').dt.start_time\nweekly = tx.groupby(\n    ['week', 'garment_group_name', 'colour_group_name']\n)['article_id'].count().reset_index()\nweekly.columns = ['week', 'garment_group', 'colour', 'units']\n\n# Create combined identifier: e.g. \"Trousers — Black\"\nweekly['group_colour'] = weekly['garment_group'] + ' — ' + weekly['colour']\n\ncombinations = sorted(weekly['group_colour'].unique())\ncombo_ids    = {c: i for i, c in enumerate(combinations)}\n\nprint(f'Combinations (group + colour): {len(combinations)}')\nprint(f'Total weekly observations:     {len(weekly):,}')\nprint()\nprint('All 24 combinations:')\nfor c in combinations:\n    total = weekly[weekly['group_colour'] == c]['units'].sum()\n    print(f'  {c}: {total:,} total units')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T19:40:11.203072Z","iopub.execute_input":"2026-03-20T19:40:11.203235Z","iopub.status.idle":"2026-03-20T19:40:16.737747Z","shell.execute_reply.started":"2026-03-20T19:40:11.203218Z","shell.execute_reply":"2026-03-20T19:40:16.737129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Build lag features using only information available\n# 6 months (26 weeks) before the target selling week.\n# This matches H&M's real procurement lead time.\n#\n# Features kept (all available at order time):\n#   lag_52       — sales same week last year\n#   roll_8w_mean — 8-week rolling average at order time\n#   yoy_ratio_6m — roll_8w_mean / lag_52 (trend vs last year)\n#   week_of_year — target week position in calendar\n#   month        — target month\n#   is_nov_dec   — Black Friday / Christmas period\n#   is_summer    — June to August period\n#   combo_id     — garment group + colour identifier\n#\n# Features excluded (NOT available 6 months before):\n#   lag_1, lag_2, lag_4  — require recent sales data\n#   roll_4w_mean         — requires recent sales data\n# ============================================================\n\nall_weeks = sorted(weekly['week'].unique())\n\nrows = []\nfor combo in combinations:\n    grp_data = weekly[\n        weekly['group_colour'] == combo\n    ].set_index('week')['units']\n\n    for i, week in enumerate(all_weeks):\n        if i < 52:\n            continue\n        if week not in grp_data.index:\n            continue\n\n        lag_52       = grp_data.get(all_weeks[i-52], 0)\n        roll_8w_mean = np.mean([grp_data.get(all_weeks[i-j], 0) for j in range(1, 9)])\n        yoy_ratio_6m = roll_8w_mean / max(lag_52, 1)\n\n        rows.append({\n            'group_colour': combo,\n            'week':         week,\n            'units':        grp_data.get(week, 0),\n            'lag_52':       lag_52,\n            'roll_8w_mean': roll_8w_mean,\n            'yoy_ratio_6m': yoy_ratio_6m,\n            'week_of_year': week.isocalendar()[1],\n            'month':        week.month,\n            'is_nov_dec':   1 if week.month in [11, 12] else 0,\n            'is_summer':    1 if week.month in [6, 7, 8] else 0,\n            'combo_id':     combo_ids[combo],\n        })\n\nfeat_df = pd.DataFrame(rows)\n\nFEATURES = ['week_of_year', 'month', 'is_nov_dec', 'is_summer',\n            'lag_52', 'roll_8w_mean', 'yoy_ratio_6m', 'combo_id']\n\n# Temporal train/test split — last 8 weeks = test\ncutoff   = sorted(feat_df['week'].unique())[-8]\ntrain_df = feat_df[feat_df['week'] <  cutoff]\ntest_df  = feat_df[feat_df['week'] >= cutoff]\nX_train  = train_df[FEATURES].values\nX_test   = test_df[FEATURES].values\ny_train  = train_df['units'].values\ny_test   = test_df['units'].values\n\nprint('Feature set (all available 6 months before target week):')\nfor f in FEATURES:\n    print(f'  {f}')\nprint()\nprint(f'Training observations: {len(X_train):,}  ({len(train_df[\"week\"].unique())} weeks x {len(combinations)} combinations)')\nprint(f'Test observations:     {len(X_test):,}  (last 8 weeks x {len(combinations)} combinations)')\nprint(f'Training period: {train_df[\"week\"].min().date()} → {train_df[\"week\"].max().date()}')\nprint(f'Test period:     {test_df[\"week\"].min().date()} → {test_df[\"week\"].max().date()}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T19:40:16.738495Z","iopub.execute_input":"2026-03-20T19:40:16.738694Z","iopub.status.idle":"2026-03-20T19:40:16.848041Z","shell.execute_reply.started":"2026-03-20T19:40:16.738676Z","shell.execute_reply":"2026-03-20T19:40:16.847217Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Three models are trained on the same feature set.\n# All use only information available 6 months before\n# the target selling week.\n# ============================================================\n\n# Model 1: Linear Regression\n# Simplest model — fully interpretable coefficients\n# Each coefficient = direct business meaning\nlr = LinearRegression()\nlr.fit(X_train, y_train)\nlr_pred = np.maximum(lr.predict(X_test), 0)\n\n# Model 2: Decision Tree (max_depth=5)\n# Non-linear model — captures interactions between features\n# max_depth=5 limits complexity given 1,104 training observations\ndt = DecisionTreeRegressor(max_depth=5, random_state=42)\ndt.fit(X_train, y_train)\ndt_pred = np.maximum(dt.predict(X_test), 0)\n\n# Model 3: Random Forest (100 trees, max_depth=8)\n# Ensemble of 100 Decision Trees — reduces overfitting via bagging\n# n_jobs=-1 uses all available CPU cores\nrf = RandomForestRegressor(n_estimators=100, max_depth=8,\n                            random_state=42, n_jobs=-1)\nrf.fit(X_train, y_train)\nrf_pred = np.maximum(rf.predict(X_test), 0)\n\n# Results summary\nprint('=' * 65)\nprint('MODEL RESULTS — 6-month forecast | 24 combinations | 192 test obs')\nprint('=' * 65)\nprint(f'{\"Model\":<25} {\"MAE\":>8} {\"R² test\":>10} {\"R² train\":>10}')\nprint('-' * 65)\nfor name, model, pred in [\n    ('Linear Regression',  lr, lr_pred),\n    ('Decision Tree',      dt, dt_pred),\n    ('Random Forest',      rf, rf_pred),\n]:\n    mae      = mean_absolute_error(y_test, pred)\n    r2_test  = r2_score(y_test, pred)\n    r2_train = r2_score(y_train, np.maximum(model.predict(X_train), 0))\n    print(f'{name:<25} {mae:>8,.0f} {r2_test:>+10.3f} {r2_train:>10.3f}')\nprint('-' * 65)\nprint('★ Best model: Linear Regression (lowest MAE, highest R² test)')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T19:40:16.849427Z","iopub.execute_input":"2026-03-20T19:40:16.849636Z","iopub.status.idle":"2026-03-20T19:40:17.102496Z","shell.execute_reply.started":"2026-03-20T19:40:16.849618Z","shell.execute_reply":"2026-03-20T19:40:17.101610Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# EXPLORATORY DATA ANALYSIS\n# Weekly demand patterns by category and colour\n# Source: H&M Group Kaggle official dataset (2022)\n# ============================================================\n\nfig, axes = plt.subplots(2, 1, figsize=(16, 10))\nfig.suptitle('Weekly Units Sold by Garment Group — Top 8 Categories\\n'\n             'H&M Group Kaggle Official Dataset | Sep 2018 – Sep 2020',\n             fontsize=13, fontweight='bold')\n\n# Panel 1: weekly units per garment group (aggregated across colours)\ntop8_weekly = tx.groupby(['week', 'garment_group_name'])['article_id'].count().reset_index()\ntop8_weekly.columns = ['week', 'garment_group', 'units']\ntop8_weekly = top8_weekly[top8_weekly['garment_group'].isin(TOP8_GROUPS)]\n\ncolors_8 = ['#1A3A6B','#2C5F9E','#5B8DD4','#0F6E56',\n            '#1D9E75','#BA7517','#A32D2D','#5A5A5A']\nfor grp, col in zip(TOP8_GROUPS, colors_8):\n    data = top8_weekly[top8_weekly['garment_group']==grp].sort_values('week')\n    axes[0].plot(data['week'], data['units'], lw=1.8, color=col,\n                 alpha=0.85, label=grp)\n\naxes[0].set_ylabel('Units sold per week', fontsize=10)\naxes[0].legend(fontsize=8, ncol=4, loc='upper left')\naxes[0].set_title('Weekly demand by garment group', fontsize=10, fontweight='bold')\n\n# Panel 2: revenue share by garment group\ntotal_by_group = top8_weekly.groupby('garment_group')['units'].sum()\ntotal_by_group = total_by_group.sort_values(ascending=True)\npct = (total_by_group / total_by_group.sum() * 100)\naxes[1].barh(pct.index, pct.values,\n             color=colors_8[::-1], alpha=0.85)\naxes[1].set_xlabel('Revenue share (%)', fontsize=10)\naxes[1].set_title('Revenue share by garment group (2018–2020)',\n                  fontsize=10, fontweight='bold')\nfor i, v in enumerate(pct.values):\n    axes[1].text(v+0.2, i, f'{v:.1f}%', va='center',\n                 fontsize=9, fontweight='bold')\n\nplt.tight_layout()\nplt.savefig('fig_eda_weekly.png', dpi=150, bbox_inches='tight')\nplt.show()\nprint('Saved: fig_eda_weekly.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T19:40:17.103240Z","iopub.execute_input":"2026-03-20T19:40:17.103455Z","iopub.status.idle":"2026-03-20T19:40:18.845226Z","shell.execute_reply.started":"2026-03-20T19:40:17.103431Z","shell.execute_reply":"2026-03-20T19:40:18.844504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# FIGURE 5 — Seasonal decomposition\n# Confirms the annual patterns that justify lag_52\n# ============================================================\n\nfig, axes = plt.subplots(2, 2, figsize=(16, 10))\nfig.suptitle('Figure 5 — Seasonal Decomposition of H&M Weekly Demand\\n'\n             'H&M Group Kaggle Official Dataset | Sep 2018 – Sep 2020',\n             fontsize=13, fontweight='bold')\n\n# Total weekly units across all 24 combinations\ntotal_weekly = weekly.groupby('week')['units'].sum().reset_index()\ntotal_weekly = total_weekly.sort_values('week')\nroll4 = total_weekly['units'].rolling(4, center=True).mean()\n\n# Panel 1: total weekly + rolling average\naxes[0,0].plot(total_weekly['week'], total_weekly['units'],\n               color='#5A5A5A', lw=1.5, alpha=0.7, label='Weekly units')\naxes[0,0].plot(total_weekly['week'], roll4,\n               color='#A32D2D', lw=2.5, label='4-week rolling average')\naxes[0,0].set_title('Total weekly units (all 24 combinations)',\n                     fontsize=10, fontweight='bold')\naxes[0,0].legend(fontsize=9)\naxes[0,0].set_ylabel('Units sold')\n\n# Panel 2: average by week of year\nweekly['week_of_year'] = weekly['week'].dt.isocalendar().week.astype(int)\navg_by_woy = weekly.groupby('week_of_year')['units'].mean()\naxes[0,1].bar(avg_by_woy.index, avg_by_woy.values,\n              color='#2C5F9E', alpha=0.75)\naxes[0,1].axvline(47, color='#A32D2D', lw=2, linestyle='--',\n                   label='Week 47 — Black Friday')\naxes[0,1].axvspan(24, 27, alpha=0.15, color='#0F6E56', label='Summer peak')\naxes[0,1].set_title('Average units by week of year\\n(confirms lag_52 seasonality)',\n                     fontsize=10, fontweight='bold')\naxes[0,1].set_xlabel('Week of year')\naxes[0,1].legend(fontsize=9)\n\n# Panel 3: sales by day of week\ntx['day_of_week'] = tx['t_dat'].dt.day_name()\nday_order = ['Monday','Tuesday','Wednesday','Thursday','Friday','Saturday','Sunday']\ndaily = tx.groupby('day_of_week')['article_id'].count().reindex(day_order)\naxes[1,0].bar(range(7), daily.values, color='#0F6E56', alpha=0.85)\naxes[1,0].set_xticks(range(7))\naxes[1,0].set_xticklabels(['Mon','Tue','Wed','Thu','Fri','Sat','Sun'])\naxes[1,0].set_title('Transactions by day of week\\nSaturday = highest volume',\n                     fontsize=10, fontweight='bold')\naxes[1,0].set_ylabel('Total transactions')\nfor i, v in enumerate(daily.values):\n    axes[1,0].text(i, v+200000, f'{v/1e6:.1f}M',\n                   ha='center', fontsize=8, fontweight='bold')\n\n# Panel 4: year-over-year 2018 vs 2019\nweekly['year'] = weekly['week'].dt.year\nfor yr, col, ls in [(2018,'#2C5F9E','-'),(2019,'#0F6E56','-')]:\n    yr_data = weekly[weekly['year']==yr].groupby('week_of_year')['units'].sum()\n    axes[1,1].plot(yr_data.index, yr_data.values, lw=2,\n                   color=col, label=str(yr), alpha=0.85, linestyle=ls)\naxes[1,1].set_title('Year-over-year comparison\\n2018 vs 2019',\n                     fontsize=10, fontweight='bold')\naxes[1,1].set_xlabel('Week of year')\naxes[1,1].set_ylabel('Total units')\naxes[1,1].legend(fontsize=9)\n\nplt.tight_layout()\nplt.savefig('figure_5_seasonal.png', dpi=150, bbox_inches='tight')\nplt.show()\nprint('Saved: figure_5_seasonal.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T19:40:18.846078Z","iopub.execute_input":"2026-03-20T19:40:18.846254Z","iopub.status.idle":"2026-03-20T19:40:22.344433Z","shell.execute_reply.started":"2026-03-20T19:40:18.846235Z","shell.execute_reply":"2026-03-20T19:40:22.343668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# FIGURE 5 — Seasonal decomposition\n# Confirms the annual patterns that justify lag_52\n# ============================================================\nfig, axes = plt.subplots(2, 2, figsize=(18, 12))\nfig.suptitle('Figure 5 — Seasonal Decomposition of H&M Weekly Demand\\n'\n             'H&M Group Kaggle Official Dataset | Sep 2018 – Sep 2020',\n             fontsize=13, fontweight='bold', y=0.98)\nfig.subplots_adjust(hspace=0.45, wspace=0.32)\n\n# Total weekly units across all 24 combinations\ntotal_weekly = weekly.groupby('week')['units'].sum().reset_index()\ntotal_weekly = total_weekly.sort_values('week')\nroll4 = total_weekly['units'].rolling(4, center=True).mean()\n\n# Panel 1: total weekly + rolling average\naxes[0,0].plot(total_weekly['week'], total_weekly['units'],\n               color='#5A5A5A', lw=1.5, alpha=0.7, label='Weekly units')\naxes[0,0].plot(total_weekly['week'], roll4,\n               color='#A32D2D', lw=2.5, label='4-week rolling average')\naxes[0,0].set_title('Total weekly units (all 24 combinations)',\n                     fontsize=11, fontweight='bold', pad=10)\naxes[0,0].legend(fontsize=10, loc='upper right')\naxes[0,0].set_ylabel('Units sold', fontsize=10)\naxes[0,0].tick_params(axis='x', labelrotation=15, labelsize=9)\n\n# Panel 2: average by week of year\nweekly['week_of_year'] = weekly['week'].dt.isocalendar().week.astype(int)\navg_by_woy = weekly.groupby('week_of_year')['units'].mean()\naxes[0,1].bar(avg_by_woy.index, avg_by_woy.values,\n              color='#2C5F9E', alpha=0.75)\naxes[0,1].axvline(47, color='#A32D2D', lw=2, linestyle='--')\naxes[0,1].axvspan(24, 27, alpha=0.15, color='#0F6E56')\naxes[0,1].set_title('Average units by week of year\\n(confirms lag_52 seasonality)',\n                     fontsize=11, fontweight='bold', pad=10)\naxes[0,1].set_xlabel('Week of year', fontsize=10)\naxes[0,1].set_ylabel('Avg units', fontsize=10)\n# Annotations separate dal legend per evitare sovrapposizioni\naxes[0,1].annotate('Black Friday\\n(week 47)',\n                    xy=(47, avg_by_woy.get(47, avg_by_woy.mean())),\n                    xytext=(38, avg_by_woy.max()*0.92),\n                    fontsize=9, color='#A32D2D', fontweight='bold',\n                    arrowprops=dict(arrowstyle='->', color='#A32D2D', lw=1.2),\n                    bbox=dict(boxstyle='round,pad=0.2', facecolor='#FFF0F0',\n                              edgecolor='#A32D2D', alpha=0.85))\naxes[0,1].annotate('Summer\\npeak',\n                    xy=(25, avg_by_woy.get(25, avg_by_woy.mean())),\n                    xytext=(10, avg_by_woy.max()*0.88),\n                    fontsize=9, color='#0F6E56', fontweight='bold',\n                    arrowprops=dict(arrowstyle='->', color='#0F6E56', lw=1.2),\n                    bbox=dict(boxstyle='round,pad=0.2', facecolor='#EAF3DE',\n                              edgecolor='#0F6E56', alpha=0.85))\n\n# Panel 3: sales by day of week\ntx['day_of_week'] = tx['t_dat'].dt.day_name()\nday_order = ['Monday','Tuesday','Wednesday','Thursday','Friday','Saturday','Sunday']\ndaily = tx.groupby('day_of_week')['article_id'].count().reindex(day_order)\naxes[1,0].bar(range(7), daily.values, color='#0F6E56', alpha=0.85)\naxes[1,0].set_xticks(range(7))\naxes[1,0].set_xticklabels(['Mon','Tue','Wed','Thu','Fri','Sat','Sun'], fontsize=10)\naxes[1,0].set_title('Transactions by day of week',\n                     fontsize=11, fontweight='bold', pad=10)\naxes[1,0].set_ylabel('Total transactions', fontsize=10)\naxes[1,0].set_ylim(0, daily.max() * 1.20)\nfor i, v in enumerate(daily.values):\n    axes[1,0].text(i, v + daily.max()*0.02,\n                   f'{v/1e6:.1f}M',\n                   ha='center', va='bottom', fontsize=10, fontweight='bold')\n\n# Panel 4: year-over-year 2018 vs 2019\nweekly['year'] = weekly['week'].dt.year\nfor yr, col in [(2018,'#2C5F9E'),(2019,'#0F6E56')]:\n    yr_data = weekly[weekly['year']==yr].groupby('week_of_year')['units'].sum()\n    axes[1,1].plot(yr_data.index, yr_data.values, lw=2,\n                   color=col, label=str(yr), alpha=0.85)\naxes[1,1].set_title('Year-over-year comparison\\n2018 vs 2019',\n                     fontsize=11, fontweight='bold', pad=10)\naxes[1,1].set_xlabel('Week of year', fontsize=10)\naxes[1,1].set_ylabel('Total units', fontsize=10)\naxes[1,1].legend(fontsize=10, loc='upper left')\n\nplt.savefig('figure_5_seasonal1.png', dpi=150, bbox_inches='tight')\nplt.show()\nprint('Saved: figure_5_seasonal1.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T19:40:22.345648Z","iopub.execute_input":"2026-03-20T19:40:22.345926Z","iopub.status.idle":"2026-03-20T19:40:25.851693Z","shell.execute_reply.started":"2026-03-20T19:40:22.345907Z","shell.execute_reply":"2026-03-20T19:40:25.850975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# FIGURE A — Model 1: Linear Regression\n# Predictions vs actual | Scatter | Residuals\n# ============================================================\n\nselected_4 = ['Trousers — Black', 'Trousers — Dark Blue',\n               'Jersey Fancy — Black', 'Knitwear — Black']\ncolors_4   = ['#1A3A6B','#2C5F9E','#BA7517','#0F6E56']\n\nfig, axes = plt.subplots(1, 3, figsize=(18, 6))\nfig.suptitle('Figure A — Model 1: Linear Regression\\n'\n             '6-month forecast | 8 features available at order time | MAE = 1,305 | R² = +0.686',\n             fontsize=12, fontweight='bold', y=1.02)\n\n# Panel 1: predictions vs actual for 4 combinations\nfor combo, col in zip(selected_4, colors_4):\n    ct     = test_df[test_df['group_colour'] == combo]\n    actual = ct['units'].values\n    pred   = np.maximum(lr.predict(ct[FEATURES].values), 0)\n    x      = range(len(actual))\n    axes[0].plot(x, actual, 'o-', color=col, lw=2, markersize=6, label=combo)\n    axes[0].plot(x, pred, 's--', color=col, lw=1.5, markersize=5, alpha=0.55)\n\naxes[0].set_title('Predictions vs Actual\\nsolid = actual  |  dashed = predicted',\n                   fontsize=10, fontweight='bold')\naxes[0].set_xticks(range(8))\naxes[0].set_xticklabels([f'Wk{i+1}' for i in range(8)], fontsize=9)\naxes[0].set_ylabel('Units sold', fontsize=10)\naxes[0].yaxis.set_major_formatter(\n    plt.FuncFormatter(lambda v,_: f'{v/1000:.0f}k'))\naxes[0].legend(fontsize=8, loc='upper right')\n\n# Panel 2: scatter predicted vs actual\nlr_pred_all = np.maximum(lr.predict(X_test), 0)\naxes[1].scatter(y_test, lr_pred_all, color='#2C5F9E', alpha=0.4, s=20)\nmaxv = max(y_test.max(), lr_pred_all.max())\naxes[1].plot([0, maxv], [0, maxv], color='red', lw=2,\n             linestyle='--', label='Perfect prediction')\naxes[1].set_xlabel('Actual units', fontsize=10)\naxes[1].set_ylabel('Predicted units', fontsize=10)\naxes[1].xaxis.set_major_formatter(\n    plt.FuncFormatter(lambda v,_: f'{v/1000:.0f}k'))\naxes[1].yaxis.set_major_formatter(\n    plt.FuncFormatter(lambda v,_: f'{v/1000:.0f}k'))\naxes[1].set_title('All 192 test observations\\nvs perfect prediction line',\n                   fontsize=10, fontweight='bold')\naxes[1].legend(fontsize=9)\naxes[1].text(0.05, 0.93,\n             f'MAE = {mean_absolute_error(y_test, lr_pred_all):,.0f}\\n'\n             f'R² = {r2_score(y_test, lr_pred_all):+.3f}',\n             transform=axes[1].transAxes, fontsize=10, fontweight='bold',\n             bbox=dict(boxstyle='round,pad=0.4', facecolor='#EEF4FF',\n                       edgecolor='#2C5F9E', alpha=0.9))\n\n# Panel 3: residuals\nresiduals = y_test - lr_pred_all\naxes[2].scatter(lr_pred_all, residuals, color='#2C5F9E', alpha=0.4, s=20)\naxes[2].axhline(0, color='red', lw=2, linestyle='--', label='Zero error')\naxes[2].axhline(residuals.mean(), color='#BA7517', lw=1.5, linestyle=':',\n                label=f'Mean = {residuals.mean():+,.0f}')\naxes[2].set_xlabel('Predicted units', fontsize=10)\naxes[2].set_ylabel('Residual (Actual − Predicted)', fontsize=10)\naxes[2].xaxis.set_major_formatter(\n    plt.FuncFormatter(lambda v,_: f'{v/1000:.0f}k'))\naxes[2].set_title('Residuals on test set\\nrandom scatter = no systematic bias',\n                   fontsize=10, fontweight='bold')\naxes[2].legend(fontsize=9)\n\nplt.tight_layout()\nplt.savefig('figure_A_linear_regression.png', dpi=150, bbox_inches='tight')\nplt.show()\nprint('Saved: figure_A_linear_regression.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T19:40:25.852321Z","iopub.execute_input":"2026-03-20T19:40:25.852534Z","iopub.status.idle":"2026-03-20T19:40:26.726422Z","shell.execute_reply.started":"2026-03-20T19:40:25.852516Z","shell.execute_reply":"2026-03-20T19:40:26.725618Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# FIGURE B — Model 2: Decision Tree Regressor\n# Tree structure + predictions vs actual\n# ============================================================\n\ndt_viz = DecisionTreeRegressor(max_depth=3, random_state=42)\ndt_viz.fit(X_train, y_train)\n\nfeature_names_readable = [\n    'Week of year', 'Month',\n    'Nov-Dec (Black Friday)', 'Summer period',\n    'Sales same week last year',\n    '8-week average',\n    'YoY ratio', 'Category + Colour ID'\n]\n\nfig, axes = plt.subplots(1, 2, figsize=(18, 8))\nfig.suptitle('Figure B — Model 2: Decision Tree Regressor\\n'\n             '6-month forecast | max_depth=5 | depth=3 shown for readability | MAE = 1,469 | R² = +0.629',\n             fontsize=12, fontweight='bold', y=1.02)\n\n# Panel 1: tree structure\nplot_tree(dt_viz, feature_names=feature_names_readable,\n          filled=True, rounded=True, fontsize=8,\n          ax=axes[0], impurity=False, label='root')\naxes[0].set_title('Tree structure (depth=3)\\nLEFT if condition TRUE  |  RIGHT if condition FALSE',\n                   fontsize=10, fontweight='bold')\n\n# Panel 2: predictions vs actual for 4 combinations\nfor combo, col in zip(selected_4, colors_4):\n    ct     = test_df[test_df['group_colour'] == combo]\n    actual = ct['units'].values\n    pred   = np.maximum(dt.predict(ct[FEATURES].values), 0)\n    x      = range(len(actual))\n    axes[1].plot(x, actual, 'o-', color=col, lw=2, markersize=6, label=combo)\n    axes[1].plot(x, pred, 's--', color=col, lw=1.5, markersize=5, alpha=0.55)\n\naxes[1].set_title('Predictions vs Actual\\nsolid = actual  |  dashed = predicted',\n                   fontsize=10, fontweight='bold')\naxes[1].set_xticks(range(8))\naxes[1].set_xticklabels([f'Wk{i+1}' for i in range(8)], fontsize=9)\naxes[1].set_ylabel('Units sold', fontsize=10)\naxes[1].yaxis.set_major_formatter(\n    plt.FuncFormatter(lambda v,_: f'{v/1000:.0f}k'))\naxes[1].legend(fontsize=8, loc='upper right')\naxes[1].text(0.03, 0.05,\n             f'MAE = {mean_absolute_error(y_test, np.maximum(dt.predict(X_test),0)):,.0f}\\n'\n             f'R² = {r2_score(y_test, np.maximum(dt.predict(X_test),0)):+.3f}',\n             transform=axes[1].transAxes, fontsize=10, fontweight='bold',\n             bbox=dict(boxstyle='round,pad=0.4', facecolor='#FFF9E6',\n                       edgecolor='#BA7517', alpha=0.9))\n\nplt.tight_layout()\nplt.savefig('figure_B_decision_tree.png', dpi=150, bbox_inches='tight')\nplt.show()\nprint('Saved: figure_B_decision_tree.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T19:40:26.727258Z","iopub.execute_input":"2026-03-20T19:40:26.727465Z","iopub.status.idle":"2026-03-20T19:40:27.756378Z","shell.execute_reply.started":"2026-03-20T19:40:26.727446Z","shell.execute_reply":"2026-03-20T19:40:27.755579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# FIGURE B v3 — Decision Tree con etichette leggibili nelle foglie\n# Stesso grafico di prima, solo testo nelle foglie migliorato\n# ============================================================\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as mpatches\nfrom sklearn.tree import DecisionTreeRegressor\nimport numpy as np\n\ndt_viz = DecisionTreeRegressor(max_depth=3, random_state=42)\ndt_viz.fit(X_train, y_train)\n\nfeature_names_readable = [\n    'Week of year', 'Month',\n    'Nov-Dec (Black Friday)', 'Summer period',\n    'Sales same week last year',\n    '8-week average',\n    'YoY ratio', 'Category + Colour ID'\n]\n\nfig, axes = plt.subplots(1, 2, figsize=(22, 10))\nfig.suptitle('Figure B — Model 2: Decision Tree Regressor\\n'\n             '6-month forecast | max_depth=5 | depth=3 shown for readability | MAE = 1,469 | R² = +0.629',\n             fontsize=12, fontweight='bold', y=1.02)\n\n# Panel 1: disegna albero con plot_tree\nfrom sklearn.tree import plot_tree\nartists = plot_tree(dt_viz,\n                    feature_names=feature_names_readable,\n                    filled=True,\n                    rounded=True,\n                    fontsize=10,\n                    ax=axes[0],\n                    impurity=False,\n                    label='root',\n                    proportion=False,\n                    precision=0)\naxes[0].set_title('Tree structure (depth=3)\\nLEFT if condition TRUE  |  RIGHT if condition FALSE',\n                   fontsize=11, fontweight='bold', pad=12)\n\n# Aggiungi prefissi leggibili al testo di ogni nodo\ntree = dt_viz.tree_\nfor i, text_obj in enumerate(axes[0].texts):\n    original = text_obj.get_text()\n    lines = original.split('\\n')\n    new_lines = []\n    for line in lines:\n        line = line.strip()\n        if line == '':\n            continue\n        # samples = numero osservazioni\n        if line.startswith('samples'):\n            val = line.replace('samples =', '').strip()\n            new_lines.append(f'Samples: {val}')\n        # value = previsione in unità\n        elif line.startswith('value'):\n            val = line.replace('value =', '').strip()\n            new_lines.append(f'Predict: {val} units/wk')\n        # condizione o altro — lascia invariato\n        else:\n            new_lines.append(line)\n    text_obj.set_text('\\n'.join(new_lines))\n\n# Panel 2: predictions vs actual (invariato)\nfor combo, col in zip(selected_4, colors_4):\n    ct     = test_df[test_df['group_colour'] == combo]\n    actual = ct['units'].values\n    pred   = np.maximum(dt.predict(ct[FEATURES].values), 0)\n    x      = range(len(actual))\n    axes[1].plot(x, actual, 'o-', color=col, lw=2, markersize=6, label=combo)\n    axes[1].plot(x, pred, 's--', color=col, lw=1.5, markersize=5, alpha=0.55)\n\naxes[1].set_title('Predictions vs Actual\\nsolid = actual  |  dashed = predicted',\n                   fontsize=11, fontweight='bold')\naxes[1].set_xticks(range(8))\naxes[1].set_xticklabels([f'Wk{i+1}' for i in range(8)], fontsize=10)\naxes[1].set_ylabel('Units sold', fontsize=10)\naxes[1].yaxis.set_major_formatter(\n    plt.FuncFormatter(lambda v,_: f'{v/1000:.0f}k'))\naxes[1].legend(fontsize=9, loc='upper right')\naxes[1].text(0.03, 0.05,\n             f'MAE = {mean_absolute_error(y_test, np.maximum(dt.predict(X_test),0)):,.0f}\\n'\n             f'R² = {r2_score(y_test, np.maximum(dt.predict(X_test),0)):+.3f}',\n             transform=axes[1].transAxes, fontsize=10, fontweight='bold',\n             bbox=dict(boxstyle='round,pad=0.4', facecolor='#FFF9E6',\n                       edgecolor='#BA7517', alpha=0.9))\n\nplt.tight_layout()\nplt.savefig('figure_B_decision_tree_v3.png', dpi=150, bbox_inches='tight')\nplt.show()\nprint('Saved: figure_B_decision_tree_v3.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T19:40:27.758237Z","iopub.execute_input":"2026-03-20T19:40:27.758471Z","iopub.status.idle":"2026-03-20T19:40:29.059743Z","shell.execute_reply.started":"2026-03-20T19:40:27.758452Z","shell.execute_reply":"2026-03-20T19:40:29.059063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# FIGURE C — Model 3: Random Forest Regressor\n# Feature importance + scatter + predictions vs actual\n# ============================================================\n\nfig, axes = plt.subplots(1, 3, figsize=(18, 6))\nfig.suptitle('Figure C — Model 3: Random Forest Regressor\\n'\n             '100 trees | max_depth=8 | 6-month forecast | MAE = 1,386 | R² = +0.655',\n             fontsize=12, fontweight='bold', y=1.02)\n\n# Panel 1: feature importance\nfeature_names_readable = [\n    'Week of year', 'Month',\n    'Nov-Dec (Black Friday)', 'Summer period',\n    'Sales same week\\nlast year',\n    '8-week average',\n    'YoY ratio',\n    'Category + Colour ID'\n]\nfeat_imp = pd.Series(rf.feature_importances_,\n                     index=feature_names_readable).sort_values(ascending=True)\ncolors_fi = ['#2C5F9E' if 'last year' in f or 'average' in f\n             else '#0F6E56' if 'YoY' in f\n             else '#BA7517' for f in feat_imp.index]\nfeat_imp.plot(kind='barh', ax=axes[0], color=colors_fi, alpha=0.85)\naxes[0].set_xlabel('Feature importance', fontsize=10)\naxes[0].set_title('Feature importance\\nwhat drives 6-month forecasts',\n                   fontsize=10, fontweight='bold')\nfor i, v in enumerate(feat_imp.values):\n    axes[0].text(v+0.005, i, f'{v:.1%}',\n                 va='center', fontsize=9, fontweight='bold')\naxes[0].set_xlim(0, feat_imp.max()*1.3)\n\nfrom matplotlib.patches import Patch\nlegend_els = [\n    Patch(color='#2C5F9E', alpha=0.85, label='Lag & rolling history'),\n    Patch(color='#0F6E56', alpha=0.85, label='YoY ratio'),\n    Patch(color='#BA7517', alpha=0.85, label='Calendar & structural'),\n]\naxes[0].legend(handles=legend_els, fontsize=8.5, loc='lower right')\n\n# Panel 2: scatter predicted vs actual\nrf_pred_all = np.maximum(rf.predict(X_test), 0)\naxes[1].scatter(y_test, rf_pred_all, color='#0F6E56', alpha=0.4, s=20)\nmaxv = max(y_test.max(), rf_pred_all.max())\naxes[1].plot([0, maxv], [0, maxv], color='red', lw=2,\n             linestyle='--', label='Perfect prediction')\naxes[1].set_xlabel('Actual units', fontsize=10)\naxes[1].set_ylabel('Predicted units', fontsize=10)\naxes[1].xaxis.set_major_formatter(\n    plt.FuncFormatter(lambda v,_: f'{v/1000:.0f}k'))\naxes[1].yaxis.set_major_formatter(\n    plt.FuncFormatter(lambda v,_: f'{v/1000:.0f}k'))\naxes[1].set_title('All 192 test observations\\nvs perfect prediction line',\n                   fontsize=10, fontweight='bold')\naxes[1].legend(fontsize=9)\naxes[1].text(0.05, 0.93,\n             f'MAE = {mean_absolute_error(y_test, rf_pred_all):,.0f}\\n'\n             f'R² = {r2_score(y_test, rf_pred_all):+.3f}',\n             transform=axes[1].transAxes, fontsize=10, fontweight='bold',\n             bbox=dict(boxstyle='round,pad=0.4', facecolor='#EAF3DE',\n                       edgecolor='#0F6E56', alpha=0.9))\n\n# Panel 3: predictions vs actual for 4 combinations\nfor combo, col in zip(selected_4, colors_4):\n    ct     = test_df[test_df['group_colour'] == combo]\n    actual = ct['units'].values\n    pred   = np.maximum(rf.predict(ct[FEATURES].values), 0)\n    x      = range(len(actual))\n    axes[2].plot(x, actual, 'o-', color=col, lw=2, markersize=6, label=combo)\n    axes[2].plot(x, pred, 's--', color=col, lw=1.5, markersize=5, alpha=0.55)\n\naxes[2].set_title('Predictions vs Actual\\nsolid = actual  |  dashed = predicted',\n                   fontsize=10, fontweight='bold')\naxes[2].set_xticks(range(8))\naxes[2].set_xticklabels([f'Wk{i+1}' for i in range(8)], fontsize=9)\naxes[2].set_ylabel('Units sold', fontsize=10)\naxes[2].yaxis.set_major_formatter(\n    plt.FuncFormatter(lambda v,_: f'{v/1000:.0f}k'))\naxes[2].legend(fontsize=8, loc='upper right')\n\nplt.tight_layout()\nplt.savefig('figure_C_random_forest.png', dpi=150, bbox_inches='tight')\nplt.show()\nprint('Saved: figure_C_random_forest.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T19:40:29.060655Z","iopub.execute_input":"2026-03-20T19:40:29.060825Z","iopub.status.idle":"2026-03-20T19:40:30.148104Z","shell.execute_reply.started":"2026-03-20T19:40:29.060808Z","shell.execute_reply":"2026-03-20T19:40:30.147458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# FIGURE 6 — Model comparison\n# MAE + R² test vs R² train for all three models\n# ============================================================\n\nmodels   = ['Linear\\nRegression', 'Decision\\nTree', 'Random\\nForest']\nmaes     = [mean_absolute_error(y_test, np.maximum(lr.predict(X_test), 0)),\n            mean_absolute_error(y_test, np.maximum(dt.predict(X_test), 0)),\n            mean_absolute_error(y_test, np.maximum(rf.predict(X_test), 0))]\nr2_test  = [r2_score(y_test, np.maximum(lr.predict(X_test), 0)),\n            r2_score(y_test, np.maximum(dt.predict(X_test), 0)),\n            r2_score(y_test, np.maximum(rf.predict(X_test), 0))]\nr2_train = [r2_score(y_train, np.maximum(lr.predict(X_train), 0)),\n            r2_score(y_train, np.maximum(dt.predict(X_train), 0)),\n            r2_score(y_train, np.maximum(rf.predict(X_train), 0))]\ncolors_m = ['#2C5F9E', '#BA7517', '#0F6E56']\n\nfig, axes = plt.subplots(1, 2, figsize=(14, 6))\nfig.suptitle('Figure 6 — Model Evaluation: Comparison of All Three Models\\n'\n             '6-month forecast | 24 combinations | 192 test observations',\n             fontsize=12, fontweight='bold', y=1.02)\n\n# Panel 1: MAE\nbars = axes[0].bar(models, maes, color=colors_m, alpha=0.85, width=0.5)\naxes[0].set_ylabel('MAE (units per week per combination)', fontsize=10)\naxes[0].set_title('Mean Absolute Error on test set\\nLower is better',\n                   fontsize=10, fontweight='bold')\naxes[0].set_ylim(0, max(maes)*1.35)\nfor bar, mae in zip(bars, maes):\n    axes[0].text(bar.get_x()+bar.get_width()/2,\n                 bar.get_height()+15,\n                 f'{mae:,.0f}',\n                 ha='center', fontsize=11, fontweight='bold')\naxes[0].annotate('★ Best model',\n                 xy=(0, maes[0]),\n                 xytext=(0.6, maes[0]+120),\n                 fontsize=10, color='#2C5F9E', fontweight='bold',\n                 arrowprops=dict(arrowstyle='->', color='#2C5F9E', lw=1.5))\n\n# Panel 2: R² test vs train\nx = np.arange(3)\nw = 0.35\nb1 = axes[1].bar(x-w/2, r2_test,  w, color=colors_m, alpha=0.85,\n                  label='R² test')\nb2 = axes[1].bar(x+w/2, r2_train, w, color=colors_m, alpha=0.40,\n                  label='R² train', hatch='//')\naxes[1].set_xticks(x)\naxes[1].set_xticklabels(models, fontsize=10)\naxes[1].set_ylabel('R² score', fontsize=10)\naxes[1].set_title('R² test vs R² train\\nGap = overfitting',\n                   fontsize=10, fontweight='bold')\naxes[1].set_ylim(0, 1.15)\naxes[1].legend(fontsize=9)\n\nfor i, (rt, rtr) in enumerate(zip(r2_test, r2_train)):\n    axes[1].text(i-w/2, rt+0.02, f'{rt:.3f}',\n                 ha='center', fontsize=9, fontweight='bold')\n    axes[1].text(i+w/2, rtr+0.02, f'{rtr:.3f}',\n                 ha='center', fontsize=9, color='#5A5A5A')\n    gap = rtr - rt\n    axes[1].annotate('', xy=(i+w/2, rt), xytext=(i+w/2, rtr),\n                     arrowprops=dict(arrowstyle='<->', color='red', lw=1.5))\n    axes[1].text(i+w/2+0.06, (rt+rtr)/2,\n                 f'Δ{gap:.3f}',\n                 fontsize=8.5, color='red', fontweight='bold', va='center')\n\naxes[1].text(0.5, -0.16,\n             'Smaller gap = less overfitting\\n'\n             'Linear Regression: smallest gap (Δ0.172) and best test performance',\n             transform=axes[1].transAxes, ha='center', fontsize=9,\n             color='#2C5F9E', fontweight='bold', style='italic')\n\nplt.tight_layout(rect=[0, 0.05, 1, 1])\nplt.savefig('figure_6_model_comparison.png', dpi=150, bbox_inches='tight')\nplt.show()\nprint('Saved: figure_6_model_comparison.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T19:40:30.148894Z","iopub.execute_input":"2026-03-20T19:40:30.149090Z","iopub.status.idle":"2026-03-20T19:40:30.969996Z","shell.execute_reply.started":"2026-03-20T19:40:30.149071Z","shell.execute_reply":"2026-03-20T19:40:30.969321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# FIGURE 7 — All three models: predictions vs actual\n# 6 representative category-colour combinations\n# Last 8 test weeks\n# ============================================================\n\nselected_6 = [\n    'Trousers — Black',\n    'Trousers — Dark Blue',\n    'Jersey Fancy — Black',\n    'Knitwear — Black',\n    'Swimwear — Black',\n    'Dresses Ladies — Black'\n]\n\ncolors_model = {\n    'Actual':             '#444444',\n    'Linear Regression':  '#2C5F9E',\n    'Decision Tree':      '#BA7517',\n    'Random Forest':      '#0F6E56'\n}\n\nfig, axes = plt.subplots(2, 3, figsize=(18, 10))\naxes = axes.flatten()\nfig.suptitle(\n    'Figure 7 — All Three Models: Predictions vs Actual by Category + Colour\\n'\n    'Last 8 test weeks | Only features available 6 months before order',\n    fontsize=13, fontweight='bold', y=1.02)\n\nfor i, combo in enumerate(selected_6):\n    ax = axes[i]\n    combo_test = test_df[test_df['group_colour'] == combo]\n    actual     = combo_test['units'].values\n    x          = range(len(actual))\n\n    ax.plot(x, actual, 'o-',\n            color=colors_model['Actual'], lw=2.5, markersize=8,\n            label='Actual', zorder=5)\n    ax.plot(x, np.maximum(lr.predict(combo_test[FEATURES].values), 0),\n            's--', color=colors_model['Linear Regression'],\n            lw=2, markersize=6, label='Linear Regression')\n    ax.plot(x, np.maximum(dt.predict(combo_test[FEATURES].values), 0),\n            '^:', color=colors_model['Decision Tree'],\n            lw=2, markersize=6, label='Decision Tree')\n    ax.plot(x, np.maximum(rf.predict(combo_test[FEATURES].values), 0),\n            'D-', color=colors_model['Random Forest'],\n            lw=2, markersize=6, alpha=0.85, label='Random Forest')\n\n    ax.set_title(combo, fontsize=11, fontweight='bold')\n    ax.set_xticks(range(8))\n    ax.set_xticklabels([f'Wk{j+1}' for j in range(8)], fontsize=9)\n    ax.set_ylabel('Units sold', fontsize=9)\n    ax.yaxis.set_major_formatter(\n        plt.FuncFormatter(lambda v,_: f'{v/1000:.1f}k'))\n    if i == 0:\n        ax.legend(fontsize=8.5, loc='upper right')\n\n    mae_lr = mean_absolute_error(\n        actual, np.maximum(lr.predict(combo_test[FEATURES].values), 0))\n    mae_rf = mean_absolute_error(\n        actual, np.maximum(rf.predict(combo_test[FEATURES].values), 0))\n    ax.text(0.02, 0.05,\n            f'MAE LR={mae_lr:,.0f}  RF={mae_rf:,.0f}',\n            transform=ax.transAxes, fontsize=8,\n            bbox=dict(boxstyle='round,pad=0.3',\n                      facecolor='white', edgecolor='#CCCCCC', alpha=0.9))\n\nplt.tight_layout()\nplt.savefig('figure_7_predictions.png', dpi=150, bbox_inches='tight')\nplt.show()\nprint('Saved: figure_7_predictions.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T19:40:30.970823Z","iopub.execute_input":"2026-03-20T19:40:30.971035Z","iopub.status.idle":"2026-03-20T19:40:32.595691Z","shell.execute_reply.started":"2026-03-20T19:40:30.971017Z","shell.execute_reply":"2026-03-20T19:40:32.594806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# FINAL SUMMARY\n# All results in one place for easy verification\n# ============================================================\n\nprint('=' * 70)\nprint('H&M WEEKLY DEMAND FORECASTING BY CATEGORY AND COLOUR')\nprint('Alberto Caretto — Student ID 25125771')\nprint('MADSC201 — Machine Learning and AI — March 2026')\nprint('=' * 70)\n\nprint('\\n📦 DATASET')\nprint(f'  Source:       H&M Group official Kaggle dataset (2022)')\nprint(f'  Transactions: 31,788,324 real purchases (Sep 2018 – Sep 2020)')\nprint(f'  Articles:     105,542 products')\nprint(f'  Combinations: {len(combinations)} (8 garment groups × 3 colours)')\n\nprint('\\n🔧 FEATURE SET (all available 6 months before target week)')\nfor f in FEATURES:\n    print(f'  {f}')\n\nprint(f'\\n📊 TRAIN/TEST SPLIT')\nprint(f'  Training: {len(X_train):,} observations ({len(train_df[\"week\"].unique())} weeks × {len(combinations)} combinations)')\nprint(f'  Test:     {len(X_test):,} observations (last 8 weeks × {len(combinations)} combinations)')\nprint(f'  Training period: {train_df[\"week\"].min().date()} → {train_df[\"week\"].max().date()}')\nprint(f'  Test period:     {test_df[\"week\"].min().date()} → {test_df[\"week\"].max().date()}')\n\nprint('\\n🏆 MODEL RESULTS')\nprint(f'  {\"Model\":<25} {\"MAE\":>8} {\"R² test\":>10} {\"R² train\":>10} {\"Overfitting gap\":>16}')\nprint('  ' + '-' * 72)\nfor name, model in [('Linear Regression ★', lr),\n                    ('Decision Tree',       dt),\n                    ('Random Forest',       rf)]:\n    pred     = np.maximum(model.predict(X_test), 0)\n    pred_tr  = np.maximum(model.predict(X_train), 0)\n    mae      = mean_absolute_error(y_test, pred)\n    r2t      = r2_score(y_test, pred)\n    r2tr     = r2_score(y_train, pred_tr)\n    gap      = r2tr - r2t\n    print(f'  {name:<25} {mae:>8,.0f} {r2t:>+10.3f} {r2tr:>10.3f} {gap:>16.3f}')\n\nprint('\\n✅ BEST MODEL: Linear Regression')\nprint('   Lowest MAE (1,305 units/week) | Highest R² test (+0.686)')\nprint('   Smallest overfitting gap (0.172) | Fully interpretable coefficients')\n\nprint('\\n🔑 KEY FINDING')\nprint('   lag_52 (sales same week last year) = 59.6% of predictive power')\nprint('   8-week average                     = 34.3% of predictive power')\nprint('   Together they explain 94% of 6-month demand predictability')\n\nprint('\\n📎 FIGURES GENERATED')\nfigures = [\n    ('Figure 4', 'figure_4_weekly.png',           'Weekly demand by garment group'),\n    ('Figure 5', 'figure_5_seasonal1.png',         'Seasonal decomposition'),\n    ('Figure A', 'figure_A_linear_regression.png', 'Linear Regression detail'),\n    ('Figure B', 'figure_B_decision_tree_v2.png',  'Decision Tree structure + predictions'),\n    ('Figure C', 'figure_C_random_forest.png',     'Random Forest feature importance + predictions'),\n    ('Figure 6', 'figure_6_model_comparison.png',  'Model comparison MAE + R²'),\n    ('Figure 7', 'figure_7_predictions.png',       'All 3 models vs actual (6 combinations)'),\n]\nfor fig_name, fname, desc in figures:\n    print(f'  {fig_name} — {desc}')\n    print(f'           Saved as: {fname}')\n\nprint('\\n' + '=' * 70)\nprint('All steps reproducible — run cells 1 to 12 in order')\nprint('=' * 70)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-20T19:40:32.596662Z","iopub.execute_input":"2026-03-20T19:40:32.596891Z","iopub.status.idle":"2026-03-20T19:40:32.685907Z","shell.execute_reply.started":"2026-03-20T19:40:32.596867Z","shell.execute_reply":"2026-03-20T19:40:32.685274Z"}},"outputs":[],"execution_count":null}]}