{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-09-02T09:04:38.373988Z","iopub.execute_input":"2026-09-02T09:04:38.374265Z","iopub.status.idle":"2026-09-02T09:04:40.585331Z","shell.execute_reply.started":"2026-09-02T09:04:38.37424Z","shell.execute_reply":"2026-09-02T09:04:40.583633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Downcast data types to optimize RAM usage\ndtypes = {\n    'customer_id': 'category',\n    'article_id': 'category',\n    'price': 'float32',\n    'sales_channel_id': 'int8'\n}\n\n# Read the transactions CSV with date parsing\ntransactions = pd.read_csv(\n    '/kaggle/input/competitions/h-and-m-personalized-fashion-recommendations/transactions_train.csv',\n    dtype=dtypes,\n    parse_dates=['t_dat']\n)\n\n# Read metadata files\narticles = pd.read_csv('/kaggle/input/competitions/h-and-m-personalized-fashion-recommendations/articles.csv')\ncustomers = pd.read_csv('/kaggle/input/competitions/h-and-m-personalized-fashion-recommendations/customers.csv')\n\nprint(f\"Transactions loaded: {len(transactions):,} rows\")\nprint(f\"Memory footprint: {transactions.memory_usage().sum() / 1024**2:.2f} MB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-02T09:04:40.585799Z","iopub.status.idle":"2026-09-02T09:04:40.586103Z","shell.execute_reply.started":"2026-09-02T09:04:40.585914Z","shell.execute_reply":"2026-09-02T09:04:40.585966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\n\ntransactions = pl.read_csv(\n    '/kaggle/input/competitions/h-and-m-personalized-fashion-recommendations/transactions_train.csv',\n    schema_overrides={\n        't_dat': pl.Date,\n        'customer_id': pl.Categorical,\n        'article_id': pl.Categorical,\n        'price': pl.Float32,\n        'sales_channel_id': pl.Int8\n    }\n)\n\nprint(transactions.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-02T09:04:40.587505Z","iopub.status.idle":"2026-09-02T09:04:40.58778Z","shell.execute_reply.started":"2026-09-02T09:04:40.587661Z","shell.execute_reply":"2026-09-02T09:04:40.587676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from datasets import load_dataset\n\n# Load the Amazon Fashion raw reviews subset directly\nds = load_dataset(\"McAuley-Lab/Amazon-Reviews-2023\", \"raw_review_Amazon_Fashion\", split=\"full\")\n\n# Convert to pandas DataFrame\ndf_amazon = ds.to_pandas()\ndf_amazon.to_parquet(\"/kaggle/working/amazon_fashion_reviews.parquet\", index=False)\nprint(f\"Successfully loaded {len(df_amazon):,} rows.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-02T09:04:40.588964Z","iopub.status.idle":"2026-09-02T09:04:40.589241Z","shell.execute_reply.started":"2026-09-02T09:04:40.589087Z","shell.execute_reply":"2026-09-02T09:04:40.589101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from datasets import load_dataset\n\nds = load_dataset(\n    \"McAuley-Lab/Amazon-Reviews-2023\",\n    \"raw_review_Amazon_Fashion\",\n    split=\"full\",\n    trust_remote_code=True\n)\n\ndf_amazon = ds.to_pandas()\ndf_amazon.to_parquet(\"/kaggle/working/amazon_fashion_reviews.parquet\", index=False)\nprint(f\"Successfully loaded {len(df_amazon):,} rows.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-02T09:04:40.590894Z","iopub.status.idle":"2026-09-02T09:04:40.59113Z","shell.execute_reply.started":"2026-09-02T09:04:40.59102Z","shell.execute_reply":"2026-09-02T09:04:40.591034Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from datasets import load_dataset\n\n# Load directly using the dataset maintainer's script\nds = load_dataset(\n    \"McAuley-Lab/Amazon-Reviews-2023\",\n    \"raw_review_Amazon_Fashion\",\n    split=\"full\",\n    trust_remote_code=True\n)\n\n# Convert to pandas DataFrame and save locally as parquet\ndf_amazon = ds.to_pandas()\ndf_amazon.to_parquet(\"/kaggle/working/amazon_fashion_reviews.parquet\", index=False)\n\nprint(f\"Successfully loaded {len(df_amazon):,} rows.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-02T09:04:40.592116Z","iopub.status.idle":"2026-09-02T09:04:40.592404Z","shell.execute_reply.started":"2026-09-02T09:04:40.592287Z","shell.execute_reply":"2026-09-02T09:04:40.592302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q \"datasets<3.0.0\" pyarrow pandas","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-02T09:04:40.59369Z","iopub.status.idle":"2026-09-02T09:04:40.594006Z","shell.execute_reply.started":"2026-09-02T09:04:40.593862Z","shell.execute_reply":"2026-09-02T09:04:40.59388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# -------------------------------------------------------------------------\n# Step 1: Memory-Efficient Loading\n# Downcast data types to keep memory usage under Kaggle RAM limits.\n# -------------------------------------------------------------------------\ntrans_dtypes = {\n    'article_id': 'int64',\n    'price': 'float32',\n    'sales_channel_id': 'int8'\n}\n\ntransactions = pd.read_csv(\n    '/kaggle/input/competitions/h-and-m-personalized-fashion-recommendations/transactions_train.csv',\n    dtype=trans_dtypes,\n    parse_dates=['t_dat']\n)\n\narticles = pd.read_csv(\n    '/kaggle/input/competitions/h-and-m-personalized-fashion-recommendations/articles.csv',\n    usecols=[\n        'article_id', \n        'product_type_name', \n        'garment_group_name', \n        'index_group_name', \n        'section_name'\n    ]\n)\n\ncustomers = pd.read_csv(\n    '/kaggle/input/competitions/h-and-m-personalized-fashion-recommendations/customers.csv',\n    usecols=['customer_id', 'age']\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-02T09:04:40.596164Z","iopub.status.idle":"2026-09-02T09:04:40.596653Z","shell.execute_reply.started":"2026-09-02T09:04:40.596412Z","shell.execute_reply":"2026-09-02T09:04:40.59643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -------------------------------------------------------------------------\n# Step 2: Merge Customer Demographics Pre-Aggregation\n# Attach age to transactions prior to grouping so we can aggregate \n# demographic profiles down to the product-period-channel level.\n# -------------------------------------------------------------------------\ntransactions = transactions.merge(customers, on='customer_id', how='left')\n\n# Convert daily dates to weekly boundaries to reduce time-series sparsity\ntransactions['week'] = transactions['t_dat'].dt.to_period('W').dt.to_timestamp()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-02T09:04:40.598393Z","iopub.status.idle":"2026-09-02T09:04:40.598979Z","shell.execute_reply.started":"2026-09-02T09:04:40.598753Z","shell.execute_reply":"2026-09-02T09:04:40.598797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -------------------------------------------------------------------------\n# Step 3: Aggregate to (article_id, week, sales_channel_id) Level\n# Map 'sales_channel_id' (1: Store, 2: Online) to replace missing geographic \n# data with channel-based segmentation.\n# -------------------------------------------------------------------------\nts_aggregated = transactions.groupby(\n    ['article_id', 'week', 'sales_channel_id'], \n    observed=True\n).agg(\n    sales_count=('price', 'count'),\n    avg_price=('price', 'mean'),\n    total_revenue=('price', 'sum'),\n    mean_customer_age=('age', 'mean')\n).reset_index()\n\n# Map explicit channel names\nts_aggregated['channel_name'] = ts_aggregated['sales_channel_id'].map({1: 'Store', 2: 'Online'})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-02T09:04:40.601781Z","iopub.status.idle":"2026-09-02T09:04:40.602247Z","shell.execute_reply.started":"2026-09-02T09:04:40.602077Z","shell.execute_reply":"2026-09-02T09:04:40.602094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -------------------------------------------------------------------------\n# Step 4: Join Article Category & Garment-Group Features\n# -------------------------------------------------------------------------\nts_dataset = ts_aggregated.merge(articles, on='article_id', how='left')\n\n# Categorical type conversions for tabular forecasting models (LightGBM/XGBoost)\ncategorical_cols = [\n    'product_type_name', \n    'garment_group_name', \n    'index_group_name', \n    'section_name', \n    'channel_name'\n]\nfor col in categorical_cols:\n    ts_dataset[col] = ts_dataset[col].astype('category')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-02T09:04:40.60342Z","iopub.status.idle":"2026-09-02T09:04:40.603723Z","shell.execute_reply.started":"2026-09-02T09:04:40.603592Z","shell.execute_reply":"2026-09-02T09:04:40.603615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -------------------------------------------------------------------------\n# Step 5: Time-Series & Demand Feature Engineering\n# Create rolling trends and lags required for WaveNet/temporal models \n# and SHAP explainability feature importances.\n# -------------------------------------------------------------------------\nts_dataset = ts_dataset.sort_values(['article_id', 'sales_channel_id', 'week'])\n\ngroup_cols = ['article_id', 'sales_channel_id']\n\n# Calendar features\nts_dataset['month'] = ts_dataset['week'].dt.month.astype('int8')\nts_dataset['week_of_year'] = ts_dataset['week'].dt.isocalendar().week.astype('int8')\n\n# Lagged demand features\nts_dataset['sales_count_lag1'] = ts_dataset.groupby(group_cols)['sales_count'].shift(1)\nts_dataset['sales_count_lag2'] = ts_dataset.groupby(group_cols)['sales_count'].shift(2)\n\n# Moving averages (4-week rolling window)\nts_dataset['sales_rolling_mean_4w'] = (\n    ts_dataset.groupby(group_cols)['sales_count']\n    .transform(lambda x: x.shift(1).rolling(4, min_periods=1).mean())\n)\n\n# Price elasticity indicator: price change from previous period\nts_dataset['price_change_lag1'] = ts_dataset['avg_price'] - ts_dataset.groupby(group_cols)['avg_price'].shift(1)\n\n# Save processed dataframe for downstream modeling\nts_dataset.to_parquet('/kaggle/working/hm_timeseries_aggregated.parquet', index=False)\n\nprint(f\"Aggregated Dataset Shape: {ts_dataset.shape}\")\nprint(ts_dataset.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-02T09:04:40.605147Z","iopub.status.idle":"2026-09-02T09:04:40.605738Z","shell.execute_reply.started":"2026-09-02T09:04:40.605546Z","shell.execute_reply":"2026-09-02T09:04:40.605566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport re\nimport nltk\nfrom nltk.sentiment.vader import SentimentIntensityAnalyzer\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.decomposition import NMF\n\n# Download NLTK VADER lexicon for sentiment scoring\nnltk.download('vader_lexicon', quiet=True)\n\n# -------------------------------------------------------------------------\n# Step 1: Ingest Local Parquet Cache\n# Load the saved Amazon Reviews dataset independently from Kaggle working space.\n# -------------------------------------------------------------------------\ndf_amazon = pd.read_parquet(\"/kaggle/working/amazon_fashion_reviews.parquet\")\n\n# Standardize text and rating column names across Amazon 2023 schemas\ntext_col = 'text' if 'text' in df_amazon.columns else 'review_text'\nrating_col = 'rating' if 'rating' in df_amazon.columns else 'overall'\n\n# Drop missing text entries\ndf_amazon = df_amazon.dropna(subset=[text_col]).copy()\n\n# -------------------------------------------------------------------------\n# Step 2: Text Preprocessing & Cleaning\n# Clean text for aspect extraction while preserving review context.\n# -------------------------------------------------------------------------\ndef clean_review_text(text):\n    text = str(text).lower()\n    text = re.sub(r'http\\S+|www\\S+|https\\S+', '', text, flags=re.MULTILINE)  # remove URLs\n    text = re.sub(r'[^a-zA-Z\\s]', '', text)  # keep letters only\n    text = re.sub(r'\\s+', ' ', text).strip()  # normalize whitespace\n    return text\n\ndf_amazon['cleaned_text'] = df_amazon[text_col].apply(clean_review_text)\n\n# -------------------------------------------------------------------------\n# Step 3: Sentiment Analysis & Negative Subset Isolation\n# Unmet customer needs (gap mining) primarily reside in low-rating (<= 3) \n# or negative-sentiment reviews.\n# -------------------------------------------------------------------------\nsia = SentimentIntensityAnalyzer()\n\n# Calculate VADER compound sentiment score (-1 = most negative, +1 = most positive)\ndf_amazon['sentiment_compound'] = df_amazon['cleaned_text'].apply(\n    lambda x: sia.polarity_scores(x)['compound'] if len(x) > 0 else 0.0\n)\n\n# Isolate negative reviews representing product gaps and customer friction points\nnegative_reviews = df_amazon[\n    (df_amazon[rating_col] <= 3) | (df_amazon['sentiment_compound'] < -0.05)\n].copy()\n\nprint(f\"Total Reviews Processed: {len(df_amazon):,}\")\nprint(f\"Negative/Friction Reviews Identified: {len(negative_reviews):,}\")\n\n# -------------------------------------------------------------------------\n# Step 4: Aspect & Topic Extraction using TF-IDF + NMF\n# Group negative review complaints into major friction topics \n# (e.g., Sizing & Fit, Material/Durability, Zipper/Hardware Quality).\n# -------------------------------------------------------------------------\ncustom_stopwords = list(TfidfVectorizer(stop_words='english').get_stop_words()) + [\n    'item', 'product', 'bought', 'ordered', 'amazon', 'returned', 'just', 'like', 'really', 'would'\n]\n\nvectorizer = TfidfVectorizer(\n    max_features=5000,\n    ngram_range=(1, 2),\n    stop_words=custom_stopwords,\n    min_df=5\n)\n\ntfidf_matrix = vectorizer.fit_transform(negative_reviews['cleaned_text'])\n\n# Train Non-Negative Matrix Factorization (NMF) to extract 5 key gap topics\nnum_topics = 5\nnmf_model = NMF(n_components=num_topics, random_state=42, init='nndsvd')\nnmf_topic_matrix = nmf_model.fit_transform(tfidf_matrix)\n\n# Map dominant topic back to each negative review\nnegative_reviews['dominant_topic_id'] = nmf_topic_matrix.argmax(axis=1)\n\n# Display top aspect keywords per topic cluster\nfeature_names = vectorizer.get_feature_names_out()\ntopic_summaries = {}\n\nfor topic_idx, topic in enumerate(nmf_model.components_):\n    top_features = [feature_names[i] for i in topic.argsort()[:-7:-1]]\n    topic_label = f\"Gap_Topic_{topic_idx+1}: \" + \", \".join(top_features)\n    topic_summaries[topic_idx] = topic_label\n    print(topic_label)\n\nnegative_reviews['topic_description'] = negative_reviews['dominant_topic_id'].map(topic_summaries)\n\n# -------------------------------------------------------------------------\n# Step 5: Category-Level Gap Aggregation\n# Aggregate complaint counts, average sentiment, and dominant gap topics by category.\n# -------------------------------------------------------------------------\ncategory_col = 'main_category' if 'main_category' in df_amazon.columns else 'parent_asin'\n\ncategory_gaps = negative_reviews.groupby(category_col).agg(\n    negative_review_count=(text_col, 'count'),\n    avg_negative_sentiment=('sentiment_compound', 'mean'),\n    top_complaint_topic=('dominant_topic_id', lambda x: x.mode()[0] if not x.empty else np.nan)\n).reset_index()\n\ncategory_gaps['top_complaint_description'] = category_gaps['top_complaint_topic'].map(topic_summaries)\n\n# Calculate overall sentiment per category across ALL reviews for relative comparison\noverall_category_sentiment = df_amazon.groupby(category_col).agg(\n    total_reviews=(text_col, 'count'),\n    overall_avg_sentiment=('sentiment_compound', 'mean')\n).reset_index()\n\n# Merge overall metrics with negative gap metrics\ncategory_gap_summary = overall_category_sentiment.merge(category_gaps, on=category_col, how='left')\ncategory_gap_summary['complaint_ratio'] = (\n    category_gap_summary['negative_review_count'] / category_gap_summary['total_reviews']\n).fillna(0)\n\n# -------------------------------------------------------------------------\n# Step 6: Export Pipeline Output\n# Save the category-level gap dataset to /kaggle/working/ for high-level \n# matching with H&M garment/product categories downstream.\n# -------------------------------------------------------------------------\ncategory_gap_summary.to_parquet(\"/kaggle/working/amazon_category_gaps.parquet\", index=False)\n\nprint(\"\\n--- Pipeline Execution Complete ---\")\nprint(category_gap_summary.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-02T09:04:40.60682Z","iopub.status.idle":"2026-09-02T09:04:40.60706Z","shell.execute_reply.started":"2026-09-02T09:04:40.606949Z","shell.execute_reply":"2026-09-02T09:04:40.606964Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n# -------------------------------------------------------------------------\n# Step 1: Ingest Pipeline Outputs\n# -------------------------------------------------------------------------\nhm_ts = pd.read_parquet(\"/kaggle/working/hm_timeseries_aggregated.parquet\")\namazon_gaps = pd.read_parquet(\"/kaggle/working/amazon_category_gaps.parquet\")\n\n# -------------------------------------------------------------------------\n# Step 2: Extract Category-Level Demand & Trend Insights (H&M Pipeline)\n# -------------------------------------------------------------------------\nlatest_date = hm_ts['week'].max()\nstart_window = latest_date - pd.Timedelta(weeks=8)\nrecent_hm = hm_ts[hm_ts['week'] >= start_window].copy()\n\n# Vectorized channel indicator (replaces slow Python lambda inside .agg)\nrecent_hm['is_online'] = (recent_hm['sales_channel_id'] == 2).astype(np.float32)\n\nhm_insights = recent_hm.groupby('garment_group_name', observed=True).agg(\n    avg_weekly_sales=('sales_count', 'mean'),\n    online_sales_share=('is_online', 'mean'),\n    avg_price=('avg_price', 'mean'),\n    price_trend=('price_change_lag1', 'mean')\n).reset_index()\n\n# -------------------------------------------------------------------------\n# Step 3: Define Taxonomy Mapping & Aggregate Amazon Gaps\n# -------------------------------------------------------------------------\ncategory_map = {\n    'Jersey Basic': 'Clothing_Shoes_and_Jewelry',\n    'Jersey Fancy': 'Clothing_Shoes_and_Jewelry',\n    'Knitwear': 'Clothing_Shoes_and_Jewelry',\n    'Trousers': 'Clothing_Shoes_and_Jewelry',\n    'Accessories': 'Amazon_Fashion',\n    'Under-, Nightwear': 'Clothing_Shoes_and_Jewelry',\n    'Dresses Outdoor': 'Clothing_Shoes_and_Jewelry'\n}\n\nhm_insights['unified_category'] = hm_insights['garment_group_name'].map(category_map)\n\namazon_gaps['unified_category'] = (\n    amazon_gaps['main_category'] \n    if 'main_category' in amazon_gaps.columns \n    else 'Clothing_Shoes_and_Jewelry'\n)\n\n# Aggregate Amazon gaps to 1 row per unified category to prevent Cartesian explode\namazon_summary = amazon_gaps.groupby('unified_category').agg(\n    complaint_ratio=('complaint_ratio', 'mean'),\n    top_complaint_description=('top_complaint_description', lambda x: x.mode()[0] if not x.dropna().empty else \"N/A\")\n).reset_index()\n\n# -------------------------------------------------------------------------\n# Step 4: Decision Layer - Merge Aggregated Summaries\n# -------------------------------------------------------------------------\nexmarket_summary = hm_insights.merge(\n    amazon_summary, \n    on='unified_category', \n    how='inner'\n)\n\n# -------------------------------------------------------------------------\n# Step 5: Fully Vectorized Opportunity Matrix & Decision Rules\n# -------------------------------------------------------------------------\n# Calculate dataset medians ONCE instead of recalculating per row\nsales_median = exmarket_summary['avg_weekly_sales'].median()\ncomplaint_median = exmarket_summary['complaint_ratio'].median()\n\nhigh_demand = exmarket_summary['avg_weekly_sales'] > sales_median\nhigh_friction = exmarket_summary['complaint_ratio'] > complaint_median\nonline_dominant = exmarket_summary['online_sales_share'] > 0.5\n\ncomplaint_desc = exmarket_summary['top_complaint_description'].astype(str)\nhigh_opp_msg = \"HIGH OPPORTUNITY: Redesign product line. High sales volume paired with top complaint: '\" + complaint_desc + \"'.\"\n\nconditions = [\n    high_demand & high_friction,\n    high_demand & ~high_friction & online_dominant,\n    high_demand & ~high_friction & ~online_dominant,\n    ~high_demand & high_friction,\n]\n\nchoices = [\n    high_opp_msg,\n    \"SCALE & OPTIMIZE: High online demand and low friction. Expand online inventory and marketing.\",\n    \"CHANNEL EXPANSION: Store demand is strong with low friction. Push online channel adoption.\",\n    \"PRODUCT REVIEW: Low demand compounded by customer complaints. Audit product specs or discontinue.\"\n]\n\nexmarket_summary['strategic_recommendation'] = np.select(\n    conditions, \n    choices, \n    default=\"MAINTAIN: Stable low-friction category.\"\n)\n\n# -------------------------------------------------------------------------\n# Step 6: Save & Display ExMarket Final Insights\n# -------------------------------------------------------------------------\noutput_cols = [\n    'garment_group_name',\n    'avg_weekly_sales',\n    'avg_price',\n    'complaint_ratio',\n    'top_complaint_description',\n    'strategic_recommendation'\n]\n\nexmarket_report = exmarket_summary[output_cols].sort_values(by='avg_weekly_sales', ascending=False)\nexmarket_report.to_csv(\"/kaggle/working/exmarket_final_decision_report.csv\", index=False)\n\nprint(\"ExMarket Decision Layer Report:\")\nprint(exmarket_report.head(10).to_string())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-02T09:04:40.607985Z","iopub.status.idle":"2026-09-02T09:04:40.608346Z","shell.execute_reply.started":"2026-09-02T09:04:40.608159Z","shell.execute_reply":"2026-09-02T09:04:40.608175Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nfrom IPython.display import display, Markdown\n\n# -------------------------------------------------------------------------\n# Step 1: Define Dataset Justification & Architectural Trade-offs Text\n# -------------------------------------------------------------------------\njustification_markdown = \"\"\"\n## Dataset Choice & Architectural Trade-off Justification\n\n### 1. Primary Transaction Backbone: H&M (Sept 2018 – Sept 2020) vs. Olist (2016 – 2018)\n* **Recency & Relevance:** H&M replaces the legacy Olist E-commerce dataset (2016–2018), advancing the core transactional baseline by two years to better reflect modern post-2018 fast-fashion dynamics.\n* **Authenticity & Price Variation:** Unlike synthetic benchmark datasets, H&M contains over 31 million authentic, production-grade retail transactions with genuine price distributions, promotional discounting cycles, and seasonal variations across 104,000 unique garment articles.\n\n### 2. Gap Mining Backbone: Amazon Reviews 2023 (McAuley Lab)\n* **Temporal Extension:** Integrating the Amazon Reviews 2023 dataset extends the platform's review-mining and market-gap capabilities directly to 2023.\n* **Separation of Concerns:** The Amazon dataset runs independently as Module 2's NLP aspect-extraction engine. Insights are bridged with H&M's quantitative sales trend data at the higher-level product category layer rather than via direct record linkage.\n\n### 3. Scope Redefinition for Module 3: Channel-Based vs. Regional Pricing\n* **Constraint:** H&M anonymizes geographic location by hashing postal codes into non-reversible string tokens (`customer_id` hash mapping without plaintext city/state names).\n* **Adaptation:** Module 3 scope is explicitly redefined from **Geographic-Regional Pricing** to **Channel-Based Pricing** (Online vs. Physical In-Store). This retains high-value pricing granularity across distinct purchasing contexts without making unsupported spatial assumptions.\n\"\"\"\n\n# -------------------------------------------------------------------------\n# Step 2: Render Structured Markdown directly in the Kaggle Notebook UI\n# -------------------------------------------------------------------------\ndisplay(Markdown(justification_markdown))\n\n# -------------------------------------------------------------------------\n# Step 3: Serialize Justification to /kaggle/working/ for Automated Reports\n# -------------------------------------------------------------------------\njustification_metadata = {\n    \"transaction_dataset\": {\n        \"name\": \"H&M Personalized Fashion Recommendations\",\n        \"timeframe\": \"Sept 2018 - Sept 2020\",\n        \"replaced_dataset\": \"Olist E-commerce (2016 - 2018)\",\n        \"rationale\": \"Improved recency, non-synthetic production data, real-world price variation.\"\n    },\n    \"nlp_dataset\": {\n        \"name\": \"McAuley Amazon Reviews 2023\",\n        \"timeframe\": \"Up to 2023\",\n        \"rationale\": \"Extends framework recency claim to 2023 for aspect extraction and sentiment gap mining.\"\n    },\n    \"module_3_scope_redefinition\": {\n        \"original_scope\": \"Geographic / Regional Pricing\",\n        \"redefined_scope\": \"Channel-Based Pricing (Online vs. Physical Store)\",\n        \"rationale\": \"H&M postal codes are hashed string tokens, rendering true spatial geographic modeling infeasible.\"\n    }\n}\n\nwith open(\"/kaggle/working/dataset_justification.json\", \"w\") as f:\n    json.dump(justification_metadata, f, indent=4)\n\nprint(\"Dataset justification rendered to notebook output and saved to /kaggle/working/dataset_justification.json\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-02T09:04:44.427025Z","iopub.execute_input":"2026-09-02T09:04:44.427467Z","iopub.status.idle":"2026-09-02T09:04:44.437371Z","shell.execute_reply.started":"2026-09-02T09:04:44.427435Z","shell.execute_reply":"2026-09-02T09:04:44.436575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}