{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:38:21.714908Z","iopub.execute_input":"2026-07-06T20:38:21.715254Z","iopub.status.idle":"2026-07-06T20:39:35.299789Z","shell.execute_reply.started":"2026-07-06T20:38:21.715225Z","shell.execute_reply":"2026-07-06T20:39:35.298525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import kagglehub\n\n# Download the latest version of the competition dataset\npath = kagglehub.competition_download(\n    \"h-and-m-personalized-fashion-recommendations\"\n)\n\nprint(\"Path to competition files:\", path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:39:35.301692Z","iopub.execute_input":"2026-07-06T20:39:35.301987Z","iopub.status.idle":"2026-07-06T20:39:35.845424Z","shell.execute_reply.started":"2026-07-06T20:39:35.301957Z","shell.execute_reply":"2026-07-06T20:39:35.844613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_PATH = \"/kaggle/input/competitions/h-and-m-personalized-fashion-recommendations\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:39:35.846436Z","iopub.execute_input":"2026-07-06T20:39:35.846908Z","iopub.status.idle":"2026-07-06T20:39:35.853483Z","shell.execute_reply.started":"2026-07-06T20:39:35.846870Z","shell.execute_reply":"2026-07-06T20:39:35.852387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load datasets\narticles = pd.read_csv(os.path.join(DATA_PATH, \"articles.csv\"))\ncustomers = pd.read_csv(os.path.join(DATA_PATH, \"customers.csv\"))\ntransactions = pd.read_csv(os.path.join(DATA_PATH, \"transactions_train.csv\"))\n\nprint(\"Articles shape:\", articles.shape)\nprint(\"Customers shape:\", customers.shape)\nprint(\"Transactions shape:\", transactions.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:39:35.856676Z","iopub.execute_input":"2026-07-06T20:39:35.857368Z","iopub.status.idle":"2026-07-06T20:40:21.798975Z","shell.execute_reply.started":"2026-07-06T20:39:35.857327Z","shell.execute_reply":"2026-07-06T20:40:21.797795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"datasets = {\n    \"Articles\": articles,\n    \"Customers\": customers,\n    \"Transactions\": transactions\n}\n\nfor name, df in datasets.items():\n    print(\"=\" * 70)\n    print(f\"{name} Dataset\")\n    print(\"=\" * 70)\n\n    print(f\"Shape: {df.shape}\")\n    print(\"\\nColumns:\")\n    print(df.columns.tolist())\n\n    print(\"\\nData Types:\")\n    print(df.dtypes)\n\n    print(\"\\nFirst Five Rows:\")\n    display(df.head())\n\n    print(\"\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:40:21.801170Z","iopub.execute_input":"2026-07-06T20:40:21.801508Z","iopub.status.idle":"2026-07-06T20:40:22.377177Z","shell.execute_reply.started":"2026-07-06T20:40:21.801479Z","shell.execute_reply":"2026-07-06T20:40:22.376044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Dataset Summary\n# ============================================================\n\nsummary = []\n\ndatasets = {\n    \"Articles\": articles,\n    \"Customers\": customers,\n    \"Transactions\": transactions\n}\n\nfor name, df in datasets.items():\n\n    summary.append({\n        \"Dataset\": name,\n        \"Rows\": df.shape[0],\n        \"Columns\": df.shape[1],\n        \"Missing Values\": df.isnull().sum().sum(),\n        \"Duplicate Rows\": df.duplicated().sum(),\n        \"Memory Usage (MB)\": round(df.memory_usage(deep=True).sum()/1024**2,2)\n    })\n\nsummary_df = pd.DataFrame(summary)\n\nsummary_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:40:22.378404Z","iopub.execute_input":"2026-07-06T20:40:22.378774Z","iopub.status.idle":"2026-07-06T20:41:04.395036Z","shell.execute_reply.started":"2026-07-06T20:40:22.378735Z","shell.execute_reply":"2026-07-06T20:41:04.394075Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Initial Data Quality Observations\n\nAfter loading the datasets, several important observations can be made:\n\n### Articles Dataset\n- Contains **105,542** unique fashion products.\n- Contains **25** descriptive attributes.\n- Only **416 missing values**, indicating excellent data quality.\n- No duplicate records were found.\n\n### Customers Dataset\n- Contains **1,371,980** customers.\n- Contains demographic and membership information.\n- Large number of missing values (1,840,560), primarily in:\n  - FN\n  - Active\n  - Age\n  - Club Member Status\n- No duplicate customer records.\n\n### Transactions Dataset\n- Contains over **31.7 million purchase transactions**, making this a large-scale retail dataset.\n- No missing values.\n- Approximately **2.97 million duplicate transactions** exist and should be investigated before modelling.\n- Dataset size exceeds **5.9 GB** in memory, requiring efficient processing techniques.\n\nOverall, the datasets are well structured and suitable for building an industrial-scale recommendation system.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Missing Value Analysis\n# ============================================================\n\nfor name, df in datasets.items():\n\n    print(\"=\"*70)\n    print(f\"{name} Dataset\")\n    print(\"=\"*70)\n\n    missing = pd.DataFrame({\n        \"Missing Count\": df.isnull().sum(),\n        \"Missing Percentage\": round(df.isnull().mean()*100,2)\n    })\n\n    missing = missing[missing[\"Missing Count\"]>0].sort_values(\n        by=\"Missing Count\",\n        ascending=False\n    )\n\n    if missing.empty:\n        print(\"No missing values.\\n\")\n    else:\n        display(missing)# ============================================================\n# Missing Value Analysis\n# ============================================================\n\nfor name, df in datasets.items():\n\n    print(\"=\"*70)\n    print(f\"{name} Dataset\")\n    print(\"=\"*70)\n\n    missing = pd.DataFrame({\n        \"Missing Count\": df.isnull().sum(),\n        \"Missing Percentage\": round(df.isnull().mean()*100,2)\n    })\n\n    missing = missing[missing[\"Missing Count\"]>0].sort_values(\n        by=\"Missing Count\",\n        ascending=False\n    )\n\n    if missing.empty:\n        print(\"No missing values.\\n\")\n    else:\n        display(missing)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:41:04.396264Z","iopub.execute_input":"2026-07-06T20:41:04.396609Z","iopub.status.idle":"2026-07-06T20:41:19.139655Z","shell.execute_reply.started":"2026-07-06T20:41:04.396572Z","shell.execute_reply":"2026-07-06T20:41:19.138647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Duplicate Records Analysis\n# ============================================================\n\nfor name, df in datasets.items():\n\n    duplicates = df.duplicated().sum()\n\n    print(\"=\" * 70)\n    print(f\"{name} Dataset\")\n    print(\"=\" * 70)\n    print(f\"Duplicate Rows: {duplicates:,}\")\n\n    if duplicates > 0:\n        print(\"\\nSample Duplicate Rows:\")\n        display(df[df.duplicated()].head())\n    else:\n        print(\"No duplicate rows found.\")\n\n    print()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:41:19.140765Z","iopub.execute_input":"2026-07-06T20:41:19.141104Z","iopub.status.idle":"2026-07-06T20:42:02.171038Z","shell.execute_reply.started":"2026-07-06T20:41:19.141075Z","shell.execute_reply":"2026-07-06T20:42:02.170260Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Duplicate Records Analysis\n\nThe Articles and Customers datasets contain no duplicate records.\n\nThe Transactions dataset contains **2,974,905 duplicate rows**.\n\nAt first glance, these appear to be duplicate transactions. However, in retail datasets this does **not necessarily indicate erroneous data**.\n\nA customer may legitimately purchase:\n\n- Multiple quantities of the same product\n- The same product on the same day\n- Identical items at the same price through the same sales channel\n\nFor example, purchasing three identical T-shirts in a single order would generate multiple records with identical values.\n\nTherefore, these records may represent **valid purchasing behaviour rather than data quality issues**.\n\nBefore deciding whether to remove duplicates, further business investigation is required.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Investigating Duplicate Transactions\n# ============================================================\n\nduplicate_transactions = transactions[\n    transactions.duplicated(keep=False)\n].sort_values(\n    by=[\"customer_id\", \"t_dat\", \"article_id\"]\n)\n\nprint(f\"Total Duplicate Records: {len(duplicate_transactions):,}\")\n\nduplicate_transactions.head(20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:42:02.172397Z","iopub.execute_input":"2026-07-06T20:42:02.172692Z","iopub.status.idle":"2026-07-06T20:42:28.934464Z","shell.execute_reply.started":"2026-07-06T20:42:02.172665Z","shell.execute_reply":"2026-07-06T20:42:28.933526Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We will keep these duplicate transaction records because they likely represent valid customer purchasing behaviour. Repeated purchases provide valuable signals for recommendation models, especially collaborative filtering and popularity-based recommenders.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Dataset Profiling\n# ============================================================\n\nfor name, df in datasets.items():\n\n    print(\"=\" * 80)\n    print(f\"{name} Dataset Profile\")\n    print(\"=\" * 80)\n\n    profile = pd.DataFrame({\n        \"Data Type\": df.dtypes,\n        \"Unique Values\": df.nunique(),\n        \"Missing Values\": df.isnull().sum()\n    })\n\n    display(profile)\n\n    print(\"\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:42:28.937746Z","iopub.execute_input":"2026-07-06T20:42:28.938081Z","iopub.status.idle":"2026-07-06T20:42:45.806886Z","shell.execute_reply.started":"2026-07-06T20:42:28.938052Z","shell.execute_reply":"2026-07-06T20:42:45.806030Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Key Insights from Data Profiling\n\nBased on the initial profiling of the datasets, several important business insights can be drawn:\n\n### Articles Dataset\n- The dataset contains **105,542 unique fashion products** described by 25 attributes.\n- Products belong to **19 major product groups** and **21 garment groups**.\n- There are **50 colour groups**, enabling colour-based recommendations.\n- Product descriptions are available for almost every product (only 0.39% missing).\n\n### Customers Dataset\n- The dataset contains over **1.37 million customers**.\n- Customer ages range across **84 unique values**.\n- More than **352,000 unique postal codes** suggest a geographically diverse customer base.\n- Membership and marketing-related variables contain missing values and will require preprocessing.\n\n### Transactions Dataset\n- The dataset contains **31.7 million purchase transactions** collected over **734 days**.\n- More than **1.36 million customers** have made purchases.\n- Customers purchased over **104,000 unique products**.\n- Sales were recorded through **two different sales channels** (online and offline).\n\nThese observations indicate that the dataset is highly suitable for developing a large-scale personalized recommendation system.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Verify Relationships Between Datasets\n# ============================================================\n\nprint(\"Unique Customers in Customers Dataset      :\", customers[\"customer_id\"].nunique())\nprint(\"Unique Customers in Transactions Dataset   :\", transactions[\"customer_id\"].nunique())\n\nprint()\n\nprint(\"Unique Articles in Articles Dataset        :\", articles[\"article_id\"].nunique())\nprint(\"Unique Articles in Transactions Dataset    :\", transactions[\"article_id\"].nunique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:42:45.808000Z","iopub.execute_input":"2026-07-06T20:42:45.808542Z","iopub.status.idle":"2026-07-06T20:42:55.676547Z","shell.execute_reply.started":"2026-07-06T20:42:45.808512Z","shell.execute_reply":"2026-07-06T20:42:55.675597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# ============================================================\n# Step 3.9: Verify Relationships Between Datasets\n# ============================================================\n\ncustomer_master = customers[\"customer_id\"].nunique()\ncustomer_transactions = transactions[\"customer_id\"].nunique()\n\narticle_master = articles[\"article_id\"].nunique()\narticle_transactions = transactions[\"article_id\"].nunique()\n\nprint(\"=\" * 70)\nprint(\"Customer Relationship\")\nprint(\"=\" * 70)\nprint(f\"Customers in Customer Dataset      : {customer_master:,}\")\nprint(f\"Customers in Transactions Dataset  : {customer_transactions:,}\")\nprint(f\"Customers with No Purchases        : {customer_master - customer_transactions:,}\")\n\nprint()\n\nprint(\"=\" * 70)\nprint(\"Article Relationship\")\nprint(\"=\" * 70)\nprint(f\"Articles in Articles Dataset       : {article_master:,}\")\nprint(f\"Articles in Transactions Dataset   : {article_transactions:,}\")\nprint(f\"Articles Never Purchased           : {article_master - article_transactions:,}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:42:55.677783Z","iopub.execute_input":"2026-07-06T20:42:55.678116Z","iopub.status.idle":"2026-07-06T20:43:05.517526Z","shell.execute_reply.started":"2026-07-06T20:42:55.678087Z","shell.execute_reply":"2026-07-06T20:43:05.516506Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Relationship Between the Datasets\n\nThe recommendation system is built by linking three datasets together.\n\n### Customer Relationship\n\n- The **Customers** dataset contains **1,371,980** unique customers.\n- The **Transactions** dataset contains purchases from **1,362,281** unique customers.\n- Therefore, **9,699 customers** have **no recorded purchase history**.\n\nThese customers present the **cold-start problem**, where a recommendation system has little or no historical information to generate personalized recommendations.\n\n---\n\n### Article Relationship\n\n- The **Articles** dataset contains **105,542** unique fashion products.\n- The **Transactions** dataset contains **104,547** purchased products.\n- Therefore, **995 products** were **never purchased** during the observation period.\n\nThese products may represent:\n\n- Newly introduced products\n- Discontinued inventory\n- Low-demand products\n- Products unavailable during the recorded transaction period\n\n---\n\n### Entity Relationship\n\nThe three datasets are connected through two primary keys.\n\n```text\n                 Customers\n              (customer_id)\n                    │\n                    │\n                    ▼\nTransactions (customer_id, article_id)\n                    ▲\n                    │\n                    │\n                 Articles\n               (article_id)\n```\n\nThe **Transactions** dataset acts as the bridge between customers and products, enabling the development of personalized recommendation systems using historical purchasing behaviour.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Step 4.1: Check Data Types\n# ============================================================\n\nfor name, df in datasets.items():\n\n    print(\"=\" * 70)\n    print(name)\n    print(\"=\" * 70)\n\n    display(df.dtypes.to_frame(\"Data Type\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:05.518954Z","iopub.execute_input":"2026-07-06T20:43:05.519356Z","iopub.status.idle":"2026-07-06T20:43:05.541313Z","shell.execute_reply.started":"2026-07-06T20:43:05.519317Z","shell.execute_reply":"2026-07-06T20:43:05.540420Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 4.1 – Data Type Inspection\n\nBefore cleaning the datasets, it is important to verify that each column has an appropriate data type.\n\nIncorrect data types can:\n\n- Increase memory consumption\n- Slow down computations\n- Cause incorrect statistical calculations\n- Affect machine learning model performance\n\nDuring this stage, each feature will be evaluated to determine whether it should remain as its current data type or be converted to a more appropriate format.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Memory Usage\n# ============================================================\n\nfor name, df in datasets.items():\n\n    memory = df.memory_usage(deep=True).sum() / 1024**2\n\n    print(f\"{name:<15}: {memory:.2f} MB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:05.542530Z","iopub.execute_input":"2026-07-06T20:43:05.542818Z","iopub.status.idle":"2026-07-06T20:43:21.665289Z","shell.execute_reply.started":"2026-07-06T20:43:05.542788Z","shell.execute_reply":"2026-07-06T20:43:21.664378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Step 4.3: Check for Invalid Values\n# ============================================================\n\nprint(\"=\" * 70)\nprint(\"Customer Age\")\nprint(\"=\" * 70)\n\nprint(f\"Minimum Age : {customers['age'].min()}\")\nprint(f\"Maximum Age : {customers['age'].max()}\")\n\nprint()\n\nprint(\"=\" * 70)\nprint(\"Transaction Price\")\nprint(\"=\" * 70)\n\nprint(f\"Minimum Price : {transactions['price'].min()}\")\nprint(f\"Maximum Price : {transactions['price'].max()}\")\n\nprint()\n\nprint(\"=\" * 70)\nprint(\"Sales Channel Distribution\")\nprint(\"=\" * 70)\n\nprint(transactions[\"sales_channel_id\"].value_counts().sort_index())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:21.666524Z","iopub.execute_input":"2026-07-06T20:43:21.666838Z","iopub.status.idle":"2026-07-06T20:43:21.951894Z","shell.execute_reply.started":"2026-07-06T20:43:21.666792Z","shell.execute_reply":"2026-07-06T20:43:21.950811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Memory Usage Analysis\n\nThe transaction dataset occupies several gigabytes of memory due to the large number of purchase records.\n\nOptimizing data types can significantly reduce memory consumption and improve computational efficiency, particularly when performing feature engineering and model training.","metadata":{}},{"cell_type":"markdown","source":"## Step 4.3 – Data Validation Observations\n\nThe data validation checks indicate that the key variables are within reasonable ranges.\n\n### Customer Age\n- Minimum Age: **16 years**\n- Maximum Age: **99 years**\n\nThe recorded ages appear realistic for H&M customers, and no invalid ages were identified.\n\n### Transaction Price\n- Minimum Price: **0.000017**\n- Maximum Price: **0.591525**\n\nThe prices are normalized rather than stored in actual currency values. This is expected for the H&M Kaggle competition dataset.\n\n### Sales Channels\nThe dataset contains transactions from **two sales channels**:\n\n- **Channel 1:** 9,408,462 transactions\n- **Channel 2:** 22,379,862 transactions\n\nChannel 2 accounts for the majority of purchases, indicating that most customer transactions occurred through this sales channel.\n\nOverall, no major data quality issues were identified during the validation process. Therefore, no records will be removed at this stage.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Step 4.4: Data Type Optimization\n# ============================================================\n\n# Convert transaction date to datetime\ntransactions[\"t_dat\"] = pd.to_datetime(transactions[\"t_dat\"])\n\n# Convert selected customer columns to category\ncustomers[\"club_member_status\"] = customers[\"club_member_status\"].astype(\"category\")\ncustomers[\"fashion_news_frequency\"] = customers[\"fashion_news_frequency\"].astype(\"category\")\n\n# Convert selected article columns to category\ncategory_columns = [\n    \"product_type_name\",\n    \"product_group_name\",\n    \"graphical_appearance_name\",\n    \"colour_group_name\",\n    \"perceived_colour_value_name\",\n    \"perceived_colour_master_name\",\n    \"department_name\",\n    \"index_code\",\n    \"index_name\",\n    \"index_group_name\",\n    \"section_name\",\n    \"garment_group_name\"\n]\n\nfor col in category_columns:\n    articles[col] = articles[col].astype(\"category\")\n\n# Convert sales channel to category\ntransactions[\"sales_channel_id\"] = transactions[\"sales_channel_id\"].astype(\"category\")\n\nprint(\"Data type optimization completed successfully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:21.953251Z","iopub.execute_input":"2026-07-06T20:43:21.953646Z","iopub.status.idle":"2026-07-06T20:43:26.807167Z","shell.execute_reply.started":"2026-07-06T20:43:21.953612Z","shell.execute_reply":"2026-07-06T20:43:26.806293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Step 4.5: Memory Usage After Optimization\n# ============================================================\n\nprint(\"=\" * 70)\nprint(\"Memory Usage After Data Type Optimization\")\nprint(\"=\" * 70)\n\nfor name, df in datasets.items():\n\n    memory = df.memory_usage(deep=True).sum() / 1024**2\n\n    print(f\"{name:<15}: {memory:.2f} MB\")# ============================================================\n# Step 4.5: Memory Usage After Optimization\n# ============================================================\n\nprint(\"=\" * 70)\nprint(\"Memory Usage After Data Type Optimization\")\nprint(\"=\" * 70)\n\nfor name, df in datasets.items():\n\n    memory = df.memory_usage(deep=True).sum() / 1024**2\n\n    print(f\"{name:<15}: {memory:.2f} MB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:26.808406Z","iopub.execute_input":"2026-07-06T20:43:26.808821Z","iopub.status.idle":"2026-07-06T20:43:42.877558Z","shell.execute_reply.started":"2026-07-06T20:43:26.808788Z","shell.execute_reply":"2026-07-06T20:43:42.876687Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 4.5 – Memory Optimization Results\n\nTo improve computational efficiency, several columns were converted from the `object` data type to the more memory-efficient `category` data type. In addition, the transaction date (`t_dat`) was converted from a string to the `datetime` data type.\n\n### Memory Usage Comparison\n\n| Dataset | Before (MB) | After (MB) | Reduction (MB) | Reduction (%) |\n|---------|------------:|-----------:|---------------:|--------------:|\n| Articles | 106.25 | 36.83 | **69.42** | **65.34%** |\n| Customers | 470.60 | 329.72 | **140.88** | **29.94%** |\n| Transactions | 5941.88 | 4183.57 | **1758.31** | **29.59%** |\n\n### Overall Improvement\n\nThe total memory usage decreased from **6518.73 MB** to **4550.12 MB**, resulting in a reduction of **1968.61 MB (30.20%)**.\n\nThis optimization significantly improves computational efficiency while preserving all original information, making the dataset more suitable for large-scale exploratory data analysis, feature engineering, and recommendation model development.","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ============================================================\n# Phase 5: Exploratory Data Analysis (EDA)\n# ============================================================","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# EDA 1: Articles Dataset Overview\n# ============================================================\n\nprint(\"=\" * 70)\nprint(\"Articles Dataset Overview\")\nprint(\"=\" * 70)\n\nprint(f\"Total Products           : {articles['article_id'].nunique():,}\")\nprint(f\"Unique Product Types     : {articles['product_type_name'].nunique():,}\")\nprint(f\"Unique Product Groups    : {articles['product_group_name'].nunique():,}\")\nprint(f\"Unique Departments       : {articles['department_name'].nunique():,}\")\nprint(f\"Unique Sections          : {articles['section_name'].nunique():,}\")\nprint(f\"Unique Garment Groups    : {articles['garment_group_name'].nunique():,}\")\nprint(f\"Unique Colours           : {articles['colour_group_name'].nunique():,}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:42.878904Z","iopub.execute_input":"2026-07-06T20:43:42.879452Z","iopub.status.idle":"2026-07-06T20:43:42.897877Z","shell.execute_reply.started":"2026-07-06T20:43:42.879411Z","shell.execute_reply":"2026-07-06T20:43:42.896970Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA 1 – Articles Dataset Overview\n\nThe Articles dataset contains detailed information about every fashion product available in the H&M catalogue.\n\nThis overview provides a high-level understanding of the product inventory by examining:\n\n- Total number of products\n- Product types\n- Product groups\n- Departments\n- Sections\n- Garment groups\n- Colour groups\n\nThese characteristics form the basis for content-based recommendation systems, where products are recommended based on similarities in their attributes.","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:42.899077Z","iopub.execute_input":"2026-07-06T20:43:42.899413Z","iopub.status.idle":"2026-07-06T20:43:42.904907Z","shell.execute_reply.started":"2026-07-06T20:43:42.899376Z","shell.execute_reply":"2026-07-06T20:43:42.903942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# EDA 2: Top 15 Product Types\n# ============================================================\n\nproduct_types = (\n    articles[\"product_type_name\"]\n    .value_counts()\n    .head(15)\n)\n\nplt.figure(figsize=(12,6))\n\nproduct_types.plot(kind=\"bar\")\n\nplt.title(\"Top 15 Product Types\")\nplt.xlabel(\"Product Type\")\nplt.ylabel(\"Number of Products\")\n\nplt.xticks(rotation=45, ha=\"right\")\n\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:42.906192Z","iopub.execute_input":"2026-07-06T20:43:42.906517Z","iopub.status.idle":"2026-07-06T20:43:43.341678Z","shell.execute_reply.started":"2026-07-06T20:43:42.906490Z","shell.execute_reply":"2026-07-06T20:43:43.340352Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA 2 – Top 15 Product Types\n\nThis visualization shows the fifteen most common product types available in the H&M product catalogue.\n\nUnderstanding the distribution of product types helps identify the dominant product categories and provides insight into H&M's merchandising strategy.\n\nThese product categories will later play an important role in building content-based recommendation models, where products with similar characteristics are recommended to customers.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# EDA 3: Top Product Groups\n# ============================================================\n\nproduct_groups = (\n    articles[\"product_group_name\"]\n    .value_counts()\n)\n\nplt.figure(figsize=(10,6))\n\nproduct_groups.plot(kind=\"bar\")\n\nplt.title(\"Distribution of Product Groups\")\nplt.xlabel(\"Product Group\")\nplt.ylabel(\"Number of Products\")\n\nplt.xticks(rotation=45, ha=\"right\")\n\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:43.343046Z","iopub.execute_input":"2026-07-06T20:43:43.343582Z","iopub.status.idle":"2026-07-06T20:43:43.628781Z","shell.execute_reply.started":"2026-07-06T20:43:43.343551Z","shell.execute_reply":"2026-07-06T20:43:43.627885Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"## EDA 3 – Product Group Distribution\n\nThis visualization presents the distribution of products across H&M's major product groups.\n\nProduct groups represent high-level merchandise categories such as garments, accessories, footwear, cosmetics, and home products.\n\nUnderstanding the distribution of these groups provides insight into the composition of H&M's product catalogue and helps identify the dominant merchandise categories available for recommendation.","metadata":{}},{"cell_type":"code","source":"\n# ============================================================\n# EDA 4: Top 20 Colour Groups\n# ============================================================\n\ncolour_distribution = (\n    articles[\"colour_group_name\"]\n    .value_counts()\n    .head(20)\n)\n\nplt.figure(figsize=(12,6))\n\ncolour_distribution.plot(kind=\"bar\")\n\nplt.title(\"Top 20 Colour Groups\")\nplt.xlabel(\"Colour Group\")\nplt.ylabel(\"Number of Products\")\n\nplt.xticks(rotation=45, ha=\"right\")\n\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:43.629938Z","iopub.execute_input":"2026-07-06T20:43:43.630231Z","iopub.status.idle":"2026-07-06T20:43:43.891177Z","shell.execute_reply.started":"2026-07-06T20:43:43.630201Z","shell.execute_reply":"2026-07-06T20:43:43.890009Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA 4 – Colour Group Distribution\n\nColour is one of the most important attributes in fashion retail and plays a significant role in customer purchasing decisions.\n\nThis visualization presents the twenty most common colour groups available in the H&M product catalogue.\n\nUnderstanding the distribution of colours can help identify dominant fashion trends and supports the development of content-based recommendation systems by incorporating colour similarity as a product feature.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# EDA 5: Top 20 Garment Groups\n# ============================================================\n\ngarment_groups = (\n    articles[\"garment_group_name\"]\n    .value_counts()\n    .head(20)\n)\n\nplt.figure(figsize=(12,6))\n\ngarment_groups.plot(kind=\"bar\")\n\nplt.title(\"Top 20 Garment Groups\")\nplt.xlabel(\"Garment Group\")\nplt.ylabel(\"Number of Products\")\n\nplt.xticks(rotation=45, ha=\"right\")\n\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:43.892353Z","iopub.execute_input":"2026-07-06T20:43:43.892695Z","iopub.status.idle":"2026-07-06T20:43:44.161492Z","shell.execute_reply.started":"2026-07-06T20:43:43.892664Z","shell.execute_reply":"2026-07-06T20:43:44.160449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# EDA 6: Top 20 Departments\n# ============================================================\n\ndepartment_distribution = (\n    articles[\"department_name\"]\n    .value_counts()\n    .head(20)\n)\n\nplt.figure(figsize=(14,6))\n\ndepartment_distribution.plot(kind=\"bar\")\n\nplt.title(\"Top 20 Departments\")\nplt.xlabel(\"Department\")\nplt.ylabel(\"Number of Products\")\n\nplt.xticks(rotation=45, ha=\"right\")\n\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:44.162779Z","iopub.execute_input":"2026-07-06T20:43:44.163225Z","iopub.status.idle":"2026-07-06T20:43:44.434253Z","shell.execute_reply.started":"2026-07-06T20:43:44.163190Z","shell.execute_reply":"2026-07-06T20:43:44.433379Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA 6 – Department Distribution\n\nDepartments represent finer-grained organizational units within H&M's product catalogue.\n\nThis visualization highlights the departments containing the largest number of products.\n\nDepartment-level analysis provides valuable business insights into product allocation across different retail divisions and can be incorporated as an additional feature in content-based recommendation systems.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# EDA 7: Top 20 Sections\n# ============================================================\n\nsection_distribution = (\n    articles[\"section_name\"]\n    .value_counts()\n    .head(20)\n)\n\nplt.figure(figsize=(14,6))\n\nbars = plt.bar(\n    section_distribution.index,\n    section_distribution.values\n)\n\nplt.title(\"Top 20 Sections\", fontsize=16, fontweight=\"bold\")\nplt.xlabel(\"Section\", fontsize=12)\nplt.ylabel(\"Number of Products\", fontsize=12)\n\nplt.xticks(rotation=45, ha=\"right\")\n\n# Add value labels\nfor bar in bars:\n    height = bar.get_height()\n    plt.text(\n        bar.get_x() + bar.get_width()/2,\n        height,\n        f\"{int(height):,}\",\n        ha=\"center\",\n        va=\"bottom\",\n        fontsize=9\n    )\n\nplt.grid(axis=\"y\", linestyle=\"--\", alpha=0.4)\n\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:44.435418Z","iopub.execute_input":"2026-07-06T20:43:44.436087Z","iopub.status.idle":"2026-07-06T20:43:44.831236Z","shell.execute_reply.started":"2026-07-06T20:43:44.436057Z","shell.execute_reply":"2026-07-06T20:43:44.830079Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA 7 – Section Distribution\n\nSections represent high-level business divisions within H&M, such as Ladieswear, Menswear, Baby, Children, and Sportswear.\n\nThis visualization highlights the distribution of products across different retail sections.\n\nUnderstanding section-level product allocation provides valuable business insights into H&M's merchandising strategy and helps identify which customer segments receive the greatest product variety. Section information can also serve as an important feature for recommendation systems by grouping products according to their target audience.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# EDA 8: Customer Age Distribution\n# ============================================================\n\nplt.figure(figsize=(12,6))\n\nplt.hist(\n    customers[\"age\"].dropna(),\n    bins=30,\n    edgecolor=\"black\"\n)\n\nplt.title(\"Customer Age Distribution\", fontsize=16, fontweight=\"bold\")\nplt.xlabel(\"Age\", fontsize=12)\nplt.ylabel(\"Number of Customers\", fontsize=12)\n\nplt.grid(axis=\"y\", linestyle=\"--\", alpha=0.4)\n\nplt.tight_layout()\n\nplt.show()\n\nprint(f\"Minimum Age : {customers['age'].min():.0f}\")\nprint(f\"Maximum Age : {customers['age'].max():.0f}\")\nprint(f\"Average Age : {customers['age'].mean():.2f}\")\nprint(f\"Median Age  : {customers['age'].median():.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:44.832441Z","iopub.execute_input":"2026-07-06T20:43:44.832717Z","iopub.status.idle":"2026-07-06T20:43:45.136339Z","shell.execute_reply.started":"2026-07-06T20:43:44.832692Z","shell.execute_reply":"2026-07-06T20:43:45.135043Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA 8 – Customer Age Distribution\n\nCustomer age is an important demographic feature that influences purchasing behaviour and recommendation quality.\n\nThis visualization illustrates the distribution of customer ages within the H&M customer base.\n\nUnderstanding the age distribution helps identify the primary target audience and supports customer segmentation, personalized marketing campaigns, and recommendation strategies tailored to different age groups.\n\nThe summary statistics (minimum, maximum, average, and median age) provide additional insights into the demographic characteristics of H&M customers.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# EDA 9: Customer Age Groups\n# ============================================================\n\n# Create age groups\nage_bins = [15, 20, 30, 40, 50, 60, 70, 80, 100]\n\nage_labels = [\n    \"16-20\",\n    \"21-30\",\n    \"31-40\",\n    \"41-50\",\n    \"51-60\",\n    \"61-70\",\n    \"71-80\",\n    \"81+\"\n]\n\ncustomers[\"Age Group\"] = pd.cut(\n    customers[\"age\"],\n    bins=age_bins,\n    labels=age_labels\n)\n\nage_group_distribution = customers[\"Age Group\"].value_counts().sort_index()\n\nplt.figure(figsize=(10,6))\n\nbars = plt.bar(\n    age_group_distribution.index.astype(str),\n    age_group_distribution.values\n)\n\nplt.title(\"Customer Age Group Distribution\", fontsize=16, fontweight=\"bold\")\nplt.xlabel(\"Age Group\", fontsize=12)\nplt.ylabel(\"Number of Customers\", fontsize=12)\n\n# Add value labels\nfor bar in bars:\n    height = bar.get_height()\n    plt.text(\n        bar.get_x() + bar.get_width()/2,\n        height,\n        f\"{int(height):,}\",\n        ha=\"center\",\n        va=\"bottom\",\n        fontsize=9\n    )\n\nplt.grid(axis=\"y\", linestyle=\"--\", alpha=0.4)\n\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:45.137525Z","iopub.execute_input":"2026-07-06T20:43:45.137816Z","iopub.status.idle":"2026-07-06T20:43:45.406434Z","shell.execute_reply.started":"2026-07-06T20:43:45.137787Z","shell.execute_reply":"2026-07-06T20:43:45.405488Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA 9 – Customer Age Group Distribution\n\nCustomer segmentation based on age groups provides a clearer understanding of the retailer's target audience than analysing individual ages.\n\nThis visualization groups customers into meaningful age categories, allowing us to identify the largest customer segments within H&M's customer base.\n\nThese age groups can later be incorporated into recommendation systems and personalized marketing strategies, enabling product recommendations that better align with customer demographics.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# EDA 10: Club Membership Status\n# ============================================================\n\nmembership = customers[\"club_member_status\"].cat.add_categories([\"Unknown\"]).fillna(\"Unknown\")\n\nmembership_distribution = membership.value_counts()\n\nplt.figure(figsize=(10,6))\n\nbars = plt.bar(\n    membership_distribution.index.astype(str),\n    membership_distribution.values\n)\n\nplt.title(\"Club Membership Status\", fontsize=16, fontweight=\"bold\")\nplt.xlabel(\"Membership Status\", fontsize=12)\nplt.ylabel(\"Number of Customers\", fontsize=12)\n\n# Add value labels\nfor bar in bars:\n    height = bar.get_height()\n    plt.text(\n        bar.get_x() + bar.get_width()/2,\n        height,\n        f\"{int(height):,}\",\n        ha=\"center\",\n        va=\"bottom\",\n        fontsize=9\n    )\n\nplt.grid(axis=\"y\", linestyle=\"--\", alpha=0.4)\n\nplt.tight_layout()\n\nplt.show()\n\nmembership_distribution","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:45.411747Z","iopub.execute_input":"2026-07-06T20:43:45.412047Z","iopub.status.idle":"2026-07-06T20:43:45.626119Z","shell.execute_reply.started":"2026-07-06T20:43:45.412020Z","shell.execute_reply":"2026-07-06T20:43:45.625132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# EDA 11: Fashion News Frequency\n# ============================================================\n\nnews_frequency = (\n    customers[\"fashion_news_frequency\"]\n    .cat.add_categories([\"Unknown\"])\n    .fillna(\"Unknown\")\n)\n\nnews_distribution = news_frequency.value_counts()\n\nplt.figure(figsize=(10,6))\n\nbars = plt.bar(\n    news_distribution.index.astype(str),\n    news_distribution.values\n)\n\nplt.title(\"Fashion News Frequency\", fontsize=16, fontweight=\"bold\")\nplt.xlabel(\"Fashion News Frequency\", fontsize=12)\nplt.ylabel(\"Number of Customers\", fontsize=12)\n\n# Add value labels\nfor bar in bars:\n    height = bar.get_height()\n    plt.text(\n        bar.get_x() + bar.get_width()/2,\n        height,\n        f\"{int(height):,}\",\n        ha=\"center\",\n        va=\"bottom\",\n        fontsize=9\n    )\n\nplt.grid(axis=\"y\", linestyle=\"--\", alpha=0.4)\n\nplt.tight_layout()\n\nplt.show()\n\nnews_distribution","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:45.627210Z","iopub.execute_input":"2026-07-06T20:43:45.627493Z","iopub.status.idle":"2026-07-06T20:43:45.813585Z","shell.execute_reply.started":"2026-07-06T20:43:45.627462Z","shell.execute_reply":"2026-07-06T20:43:45.812565Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# EDA 11: Fashion News Frequency\n# ============================================================\n\nnews_frequency = (\n    customers[\"fashion_news_frequency\"]\n    .cat.add_categories([\"Unknown\"])\n    .fillna(\"Unknown\")\n)\n\nnews_distribution = news_frequency.value_counts()\n\nplt.figure(figsize=(10,6))\n\nbars = plt.bar(\n    news_distribution.index.astype(str),\n    news_distribution.values\n)\n\nplt.title(\"Fashion News Frequency\", fontsize=16, fontweight=\"bold\")\nplt.xlabel(\"Fashion News Frequency\", fontsize=12)\nplt.ylabel(\"Number of Customers\", fontsize=12)\n\n# Add value labels\nfor bar in bars:\n    height = bar.get_height()\n    plt.text(\n        bar.get_x() + bar.get_width()/2,\n        height,\n        f\"{int(height):,}\",\n        ha=\"center\",\n        va=\"bottom\",\n        fontsize=9\n    )\n\nplt.grid(axis=\"y\", linestyle=\"--\", alpha=0.4)\n\nplt.tight_layout()\n\nplt.show()\n\nnews_distribution","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:45.814864Z","iopub.execute_input":"2026-07-06T20:43:45.815610Z","iopub.status.idle":"2026-07-06T20:43:46.021263Z","shell.execute_reply.started":"2026-07-06T20:43:45.815576Z","shell.execute_reply":"2026-07-06T20:43:46.020438Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA 11 – Fashion News Frequency\n\nFashion news subscriptions represent an important customer engagement feature.\n\nThis visualization illustrates how frequently customers receive fashion-related news and promotional communications from H&M.\n\nCustomers who regularly engage with fashion news may demonstrate different purchasing behaviours compared to those who opt out of marketing communications. This feature can therefore contribute to customer segmentation and personalized recommendation strategies.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# EDA 12: Monthly Sales Trend\n# ============================================================\n\n# Create Year-Month column\ntransactions[\"Year_Month\"] = transactions[\"t_dat\"].dt.to_period(\"M\")\n\n# Count transactions per month\nmonthly_sales = (\n    transactions.groupby(\"Year_Month\")\n    .size()\n)\n\nplt.figure(figsize=(16,6))\n\nplt.plot(\n    monthly_sales.index.astype(str),\n    monthly_sales.values,\n    linewidth=2\n)\n\nplt.title(\"Monthly Sales Trend\", fontsize=16, fontweight=\"bold\")\nplt.xlabel(\"Month\", fontsize=12)\nplt.ylabel(\"Number of Transactions\", fontsize=12)\n\nplt.xticks(rotation=45)\n\nplt.grid(alpha=0.3)\n\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:46.022512Z","iopub.execute_input":"2026-07-06T20:43:46.022916Z","iopub.status.idle":"2026-07-06T20:43:49.133804Z","shell.execute_reply.started":"2026-07-06T20:43:46.022876Z","shell.execute_reply":"2026-07-06T20:43:49.132746Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA 12 – Monthly Sales Trend\n\nUnderstanding how sales change over time is essential for retail analytics.\n\nThis visualization presents the monthly transaction volume throughout the observation period.\n\nMonthly sales trends help identify:\n\n- Seasonal shopping behaviour\n- High-demand periods\n- Sales growth or decline\n- Promotional campaign effects\n\nThese insights support inventory planning, marketing strategies, and demand forecasting.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# EDA 13: Sales Channel Distribution\n# ============================================================\n\nsales_channel = (\n    transactions[\"sales_channel_id\"]\n    .value_counts()\n    .sort_index()\n)\n\n# Rename channels for better readability\nsales_channel.index = [\"Channel 1\", \"Channel 2\"]\n\nplt.figure(figsize=(8,6))\n\nbars = plt.bar(\n    sales_channel.index,\n    sales_channel.values\n)\n\nplt.title(\"Sales Channel Distribution\", fontsize=16, fontweight=\"bold\")\nplt.xlabel(\"Sales Channel\", fontsize=12)\nplt.ylabel(\"Number of Transactions\", fontsize=12)\n\n# Add value labels\nfor bar in bars:\n    height = bar.get_height()\n    plt.text(\n        bar.get_x() + bar.get_width()/2,\n        height,\n        f\"{int(height):,}\",\n        ha=\"center\",\n        va=\"bottom\",\n        fontsize=10\n    )\n\nplt.grid(axis=\"y\", linestyle=\"--\", alpha=0.4)\n\nplt.tight_layout()\n\nplt.show()\n\nsales_channel","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:49.134978Z","iopub.execute_input":"2026-07-06T20:43:49.135319Z","iopub.status.idle":"2026-07-06T20:43:49.473572Z","shell.execute_reply.started":"2026-07-06T20:43:49.135292Z","shell.execute_reply":"2026-07-06T20:43:49.472705Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA 13 – Sales Channel Distribution\n\nThe H&M transaction dataset records purchases across two sales channels.\n\nThis visualization compares the number of transactions in each sales channel.\n\nUnderstanding channel distribution helps businesses evaluate customer purchasing behaviour across different shopping platforms and supports channel-specific marketing and recommendation strategies.\n\nThe analysis indicates which sales channel contributes the largest share of total transactions during the observation period.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# EDA 14: Transaction Price Distribution\n# ============================================================\n\nplt.figure(figsize=(12,6))\n\nplt.hist(\n    transactions[\"price\"],\n    bins=50,\n    edgecolor=\"black\"\n)\n\nplt.title(\"Transaction Price Distribution\", fontsize=16, fontweight=\"bold\")\nplt.xlabel(\"Normalized Price\", fontsize=12)\nplt.ylabel(\"Number of Transactions\", fontsize=12)\n\nplt.grid(axis=\"y\", linestyle=\"--\", alpha=0.4)\n\nplt.tight_layout()\n\nplt.show()\n\nprint(\"=\" * 60)\nprint(\"Price Statistics\")\nprint(\"=\" * 60)\n\nprint(transactions[\"price\"].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:49.474846Z","iopub.execute_input":"2026-07-06T20:43:49.475244Z","iopub.status.idle":"2026-07-06T20:43:51.563575Z","shell.execute_reply.started":"2026-07-06T20:43:49.475204Z","shell.execute_reply":"2026-07-06T20:43:51.562557Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA 14 – Transaction Price Distribution\n\nPrice is one of the most influential factors affecting customer purchasing decisions and product recommendations.\n\nThis visualization illustrates the distribution of transaction prices across all purchases in the H&M dataset.\n\nThe accompanying summary statistics provide insights into:\n\n- Minimum transaction price\n- Maximum transaction price\n- Average transaction price\n- Median transaction price\n- Price variability\n\nSince the dataset stores normalized prices rather than actual currency values, the analysis focuses on the relative distribution of prices rather than absolute monetary values.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# EDA 15: Top 20 Best-Selling Products\n# ============================================================\n\n# Count purchases for each article\ntop_products = (\n    transactions[\"article_id\"]\n    .value_counts()\n    .reset_index()\n)\n\ntop_products.columns = [\"article_id\", \"Number of Purchases\"]\n\n# Merge with article information\ntop_products = top_products.merge(\n    articles[[\"article_id\", \"prod_name\"]],\n    on=\"article_id\",\n    how=\"left\"\n)\n\n# Keep Top 20\ntop20_products = top_products.head(20)\n\nplt.figure(figsize=(14,8))\n\nbars = plt.barh(\n    top20_products[\"prod_name\"],\n    top20_products[\"Number of Purchases\"]\n)\n\nplt.title(\"Top 20 Best-Selling Products\", fontsize=16, fontweight=\"bold\")\nplt.xlabel(\"Number of Purchases\", fontsize=12)\nplt.ylabel(\"Product Name\", fontsize=12)\n\n# Highest product at the top\nplt.gca().invert_yaxis()\n\n# Add value labels\nfor bar in bars:\n    width = bar.get_width()\n    plt.text(\n        width,\n        bar.get_y() + bar.get_height()/2,\n        f\"{int(width):,}\",\n        va=\"center\",\n        fontsize=9\n    )\n\nplt.grid(axis=\"x\", linestyle=\"--\", alpha=0.4)\n\nplt.tight_layout()\n\nplt.show()\n\ntop20_products","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:51.564769Z","iopub.execute_input":"2026-07-06T20:43:51.565098Z","iopub.status.idle":"2026-07-06T20:43:53.443572Z","shell.execute_reply.started":"2026-07-06T20:43:51.565062Z","shell.execute_reply":"2026-07-06T20:43:53.442592Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA 15 – Top 20 Best-Selling Products\n\nThis analysis identifies the twenty products purchased most frequently by customers.\n\nThe transaction data was merged with the product catalogue to replace product IDs with meaningful product names, making the results easier to interpret.\n\nUnderstanding the best-selling products helps retailers:\n\n- Identify consistently popular items\n- Improve inventory planning\n- Develop targeted marketing campaigns\n- Recommend trending products to new customers\n\nThese products can also serve as strong baseline recommendations in popularity-based recommendation systems.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# EDA 16: Top 20 Best-Selling Product Categories\n# ============================================================\n\n# Merge transactions with article information\ntransaction_products = transactions.merge(\n    articles[[\"article_id\", \"product_type_name\"]],\n    on=\"article_id\",\n    how=\"left\"\n)\n\n# Count purchases by product category\ntop_categories = (\n    transaction_products[\"product_type_name\"]\n    .value_counts()\n    .head(20)\n)\n\nplt.figure(figsize=(14,8))\n\nbars = plt.barh(\n    top_categories.index,\n    top_categories.values\n)\n\nplt.title(\"Top 20 Best-Selling Product Categories\", fontsize=16, fontweight=\"bold\")\nplt.xlabel(\"Number of Purchases\", fontsize=12)\nplt.ylabel(\"Product Category\", fontsize=12)\n\n# Display highest category at the top\nplt.gca().invert_yaxis()\n\n# Add value labels\nfor bar in bars:\n    width = bar.get_width()\n    plt.text(\n        width,\n        bar.get_y() + bar.get_height()/2,\n        f\"{int(width):,}\",\n        va=\"center\",\n        fontsize=9\n    )\n\nplt.grid(axis=\"x\", linestyle=\"--\", alpha=0.4)\n\nplt.tight_layout()\n\nplt.show()\n\ntop_categories.to_frame(\"Number of Purchases\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:53.444916Z","iopub.execute_input":"2026-07-06T20:43:53.445378Z","iopub.status.idle":"2026-07-06T20:43:57.477802Z","shell.execute_reply.started":"2026-07-06T20:43:53.445348Z","shell.execute_reply":"2026-07-06T20:43:57.476480Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA 16 – Top 20 Best-Selling Product Categories\n\nThis analysis examines purchasing behaviour at the product category level by combining transaction records with the product catalogue.\n\nRather than analysing individual products, this visualization identifies the categories that generate the highest sales volume.\n\nCategory-level analysis is particularly valuable because it helps retailers:\n\n- Understand customer preferences across different product types\n- Improve inventory management\n- Optimize merchandising strategies\n- Design targeted promotional campaigns\n- Enhance recommendation systems by identifying popular product categories\n\nThe results also provide valuable business insights into the product categories that contribute most to H&M's overall sales.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# EDA 17: Top 20 Most Active Customers\n# ============================================================\n\ncustomer_purchase_frequency = (\n    transactions[\"customer_id\"]\n    .value_counts()\n    .head(20)\n)\n\nplt.figure(figsize=(14,8))\n\nbars = plt.barh(\n    range(len(customer_purchase_frequency)),\n    customer_purchase_frequency.values\n)\n\nplt.title(\"Top 20 Most Active Customers\", fontsize=16, fontweight=\"bold\")\nplt.xlabel(\"Number of Purchases\", fontsize=12)\nplt.ylabel(\"Customer Rank\", fontsize=12)\n\nplt.yticks(\n    range(len(customer_purchase_frequency)),\n    [f\"Customer {i}\" for i in range(1,21)]\n)\n\nplt.gca().invert_yaxis()\n\n# Add value labels\nfor bar in bars:\n    width = bar.get_width()\n    plt.text(\n        width,\n        bar.get_y() + bar.get_height()/2,\n        f\"{int(width):,}\",\n        va=\"center\",\n        fontsize=9\n    )\n\nplt.grid(axis=\"x\", linestyle=\"--\", alpha=0.4)\n\nplt.tight_layout()\n\nplt.show()\n\ncustomer_purchase_frequency.to_frame(\"Number of Purchases\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:43:57.479109Z","iopub.execute_input":"2026-07-06T20:43:57.479587Z","iopub.status.idle":"2026-07-06T20:44:04.317593Z","shell.execute_reply.started":"2026-07-06T20:43:57.479504Z","shell.execute_reply":"2026-07-06T20:44:04.316546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# EDA 18: Customer Purchase Frequency Distribution\n# ============================================================\n\ncustomer_purchase_counts = transactions[\"customer_id\"].value_counts()\n\nplt.figure(figsize=(12,6))\n\nplt.hist(\n    customer_purchase_counts,\n    bins=100,\n    edgecolor=\"black\"\n)\n\nplt.title(\"Customer Purchase Frequency Distribution\", fontsize=16, fontweight=\"bold\")\nplt.xlabel(\"Number of Purchases per Customer\", fontsize=12)\nplt.ylabel(\"Number of Customers\", fontsize=12)\n\nplt.grid(axis=\"y\", linestyle=\"--\", alpha=0.4)\n\nplt.tight_layout()\nplt.show()\n\nprint(\"=\" * 60)\nprint(\"Customer Purchase Frequency Statistics\")\nprint(\"=\" * 60)\n\nprint(customer_purchase_counts.describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:44:04.318716Z","iopub.execute_input":"2026-07-06T20:44:04.319310Z","iopub.status.idle":"2026-07-06T20:44:10.726785Z","shell.execute_reply.started":"2026-07-06T20:44:04.319277Z","shell.execute_reply":"2026-07-06T20:44:10.725991Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA 18 – Customer Purchase Frequency Distribution\n\nThis analysis shows how frequently customers purchase products from H&M.\n\nThe distribution helps identify whether most customers are occasional shoppers or frequent buyers.\n\nUnderstanding purchase frequency is important for:\n\n- Customer segmentation\n- Loyalty analysis\n- Personalized marketing\n- Recommendation system design\n\nCustomers with higher purchase frequency provide stronger behavioural signals for collaborative filtering models.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Phase 6: Customer Segmentation (RFM Analysis)\n# ============================================================\n\n# Latest transaction date\nsnapshot_date = transactions[\"t_dat\"].max() + pd.Timedelta(days=1)\n\nprint(\"Snapshot Date:\", snapshot_date)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:44:10.727945Z","iopub.execute_input":"2026-07-06T20:44:10.728275Z","iopub.status.idle":"2026-07-06T20:44:10.817833Z","shell.execute_reply.started":"2026-07-06T20:44:10.728248Z","shell.execute_reply":"2026-07-06T20:44:10.816712Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Phase 6 – Customer Segmentation (RFM Analysis)\n\nRFM (Recency, Frequency, Monetary) is one of the most widely used customer segmentation techniques in retail analytics.\n\nIt","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Step 6.2: Calculate RFM Metrics\n# ============================================================\n\nrfm = (\n    transactions.groupby(\"customer_id\")\n    .agg({\n        \"t_dat\": lambda x: (snapshot_date - x.max()).days,\n        \"article_id\": \"count\",\n        \"price\": \"sum\"\n    })\n    .reset_index()\n)\n\nrfm.columns = [\n    \"customer_id\",\n    \"Recency\",\n    \"Frequency\",\n    \"Monetary\"\n]\n\nrfm.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:44:10.819195Z","iopub.execute_input":"2026-07-06T20:44:10.820261Z","iopub.status.idle":"2026-07-06T20:46:28.830524Z","shell.execute_reply.started":"2026-07-06T20:44:10.820225Z","shell.execute_reply":"2026-07-06T20:46:28.829383Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 6.2 – Calculate RFM Metrics\n\nThree customer-level metrics are calculated from the transaction history:\n\n- **Recency:** Number of days since the customer's most recent purchase.\n- **Frequency:** Total number of purchases made by the customer.\n- **Monetary:** Total amount spent by the customer (using normalized transaction prices).\n\nThese metrics summarize each customer's purchasing behaviour and form the foundation of the RFM customer segmentation framework.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Step 6.3: RFM Summary Statistics\n# ============================================================\n\nprint(\"=\" * 70)\nprint(\"RFM Summary Statistics\")\nprint(\"=\" * 70)\n\ndisplay(rfm.describe().T)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:46:28.831898Z","iopub.execute_input":"2026-07-06T20:46:28.832387Z","iopub.status.idle":"2026-07-06T20:46:29.029857Z","shell.execute_reply.started":"2026-07-06T20:46:28.832354Z","shell.execute_reply":"2026-07-06T20:46:29.028742Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 6.3 – RFM Summary Statistics\n\nBefore segmenting customers, it is important to understand the distribution of the RFM metrics.\n\nThe summary statistics provide information about:\n\n- Minimum values\n- Maximum values\n- Mean\n- Median (50th percentile)\n- Standard deviation\n- Quartiles\n\nThese statistics help us understand customer purchasing behaviour and determine appropriate thresholds for customer segmentation.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Step 6.4.1: Recency Distribution\n# ============================================================\n\nplt.figure(figsize=(12,6))\n\nplt.hist(\n    rfm[\"Recency\"],\n    bins=50,\n    edgecolor=\"black\"\n)\n\nplt.title(\"Distribution of Customer Recency\", fontsize=16, fontweight=\"bold\")\nplt.xlabel(\"Recency (Days)\", fontsize=12)\nplt.ylabel(\"Number of Customers\", fontsize=12)\n\nplt.grid(axis=\"y\", linestyle=\"--\", alpha=0.4)\n\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:46:29.031194Z","iopub.execute_input":"2026-07-06T20:46:29.031536Z","iopub.status.idle":"2026-07-06T20:46:29.327813Z","shell.execute_reply.started":"2026-07-06T20:46:29.031506Z","shell.execute_reply":"2026-07-06T20:46:29.326785Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Distribution of Recency\n\nThis visualization illustrates how recently customers made their most recent purchase.\n\nCustomers with lower recency values have purchased more recently and are generally considered more engaged than customers with higher recency values.\n\nRecency is one of the strongest indicators of future purchasing behaviour in retail analytics.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Step 6.5: Calculate RFM Scores\n# ============================================================\n\n# Recency Score (lower recency = higher score)\nrfm[\"R_Score\"] = pd.qcut(\n    rfm[\"Recency\"],\n    q=5,\n    labels=[5, 4, 3, 2, 1]\n).astype(int)\n\n# Frequency Score (higher frequency = higher score)\nrfm[\"F_Score\"] = pd.qcut(\n    rfm[\"Frequency\"].rank(method=\"first\"),\n    q=5,\n    labels=[1, 2, 3, 4, 5]\n).astype(int)\n\n# Monetary Score (higher spending = higher score)\nrfm[\"M_Score\"] = pd.qcut(\n    rfm[\"Monetary\"].rank(method=\"first\"),\n    q=5,\n    labels=[1, 2, 3, 4, 5]\n).astype(int)\n\nrfm.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:46:29.329061Z","iopub.execute_input":"2026-07-06T20:46:29.329428Z","iopub.status.idle":"2026-07-06T20:46:30.204302Z","shell.execute_reply.started":"2026-07-06T20:46:29.329387Z","shell.execute_reply":"2026-07-06T20:46:30.203195Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"## Step 6.5 – RFM Scoring\n\nEach customer is assigned a score from **1 to 5** for each RFM metric.\n\n### Recency (R)\nCustomers who purchased more recently receive higher scores.\n\n### Frequency (F)\nCustomers who purchase more frequently receive higher scores.\n\n### Monetary (M)\nCustomers with higher total spending receive higher scores.\n\nUsing quintiles ensures that customers are distributed evenly across the five scoring levels, providing a balanced segmentation framework for further analysis.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Step 6.6: Create the RFM Score\n# ============================================================\n\n# Combine R, F and M scores\nrfm[\"RFM_Score\"] = (\n    rfm[\"R_Score\"].astype(str) +\n    rfm[\"F_Score\"].astype(str) +\n    rfm[\"M_Score\"].astype(str)\n)\n\nrfm.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:46:30.205414Z","iopub.execute_input":"2026-07-06T20:46:30.206113Z","iopub.status.idle":"2026-07-06T20:46:31.472244Z","shell.execute_reply.started":"2026-07-06T20:46:30.206081Z","shell.execute_reply":"2026-07-06T20:46:31.471043Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 6.6 – Create the RFM Score\n\nThe individual Recency, Frequency, and Monetary scores are combined into a single three-digit RFM score.\n\nFor example:\n\n| R | F | M | RFM Score |\n|---|---|---|-----------|\n| 5 | 5 | 5 | **555** |\n| 4 | 5 | 4 | **454** |\n| 2 | 3 | 1 | **231** |\n\nEach digit represents a customer's performance for one of the three RFM metrics.\n\nHigher RFM scores generally indicate customers who are more valuable, more engaged, and more likely to respond positively to personalized recommendations and marketing campaigns.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Step 6.7: Customer Segmentation\n# ============================================================\n\ndef customer_segment(row):\n\n    if row[\"R_Score\"] >= 4 and row[\"F_Score\"] >= 4 and row[\"M_Score\"] >= 4:\n        return \"Champions\"\n\n    elif row[\"R_Score\"] >= 3 and row[\"F_Score\"] >= 4:\n        return \"Loyal Customers\"\n\n    elif row[\"R_Score\"] >= 4 and row[\"F_Score\"] >= 2:\n        return \"Potential Loyalists\"\n\n    elif row[\"R_Score\"] >= 4 and row[\"F_Score\"] == 1:\n        return \"New Customers\"\n\n    elif row[\"R_Score\"] <= 2 and row[\"F_Score\"] >= 3:\n        return \"At Risk\"\n\n    elif row[\"R_Score\"] <= 2 and row[\"F_Score\"] <= 2:\n        return \"Lost Customers\"\n\n    else:\n        return \"Needs Attention\"\n\n\nrfm[\"Customer Segment\"] = rfm.apply(customer_segment, axis=1)\n\nrfm.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:46:31.473429Z","iopub.execute_input":"2026-07-06T20:46:31.473734Z","iopub.status.idle":"2026-07-06T20:46:49.724548Z","shell.execute_reply.started":"2026-07-06T20:46:31.473708Z","shell.execute_reply":"2026-07-06T20:46:49.723471Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 6.7 – Customer Segmentation\n\nCustomers are grouped into business-oriented segments using their Recency, Frequency, and Monetary scores.\n\nThe segmentation enables retailers to better understand customer behaviour and design targeted marketing strategies.\n\nThe customer segments used in this project are:\n\n- **Champions** – Recent, frequent, and high-spending customers.\n- **Loyal Customers** – Regular customers with consistently high purchase frequency.\n- **Potential Loyalists** – Customers showing strong engagement and likely to become loyal customers.\n- **New Customers** – Recently acquired customers with limited purchase history.\n- **At Risk** – Previously active customers who have not purchased recently.\n- **Lost Customers** – Customers with low recency and low purchase frequency.\n- **Needs Attention** – Customers who do not clearly belong to the above groups and may require additional engagement.\n\nThese customer segments are widely used in retail analytics to improve customer retention, personalized marketing, and recommendation strategies.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Step 6.8: Customer Segment Distribution\n# ============================================================\n\nsegment_distribution = (\n    rfm[\"Customer Segment\"]\n    .value_counts()\n)\n\nplt.figure(figsize=(12,6))\n\nbars = plt.bar(\n    segment_distribution.index,\n    segment_distribution.values\n)\n\nplt.title(\"Customer Segment Distribution\", fontsize=16, fontweight=\"bold\")\nplt.xlabel(\"Customer Segment\", fontsize=12)\nplt.ylabel(\"Number of Customers\", fontsize=12)\n\nplt.xticks(rotation=20, ha=\"right\")\n\n# Add value labels\nfor bar in bars:\n    height = bar.get_height()\n    plt.text(\n        bar.get_x() + bar.get_width()/2,\n        height,\n        f\"{int(height):,}\",\n        ha=\"center\",\n        va=\"bottom\",\n        fontsize=9\n    )\n\nplt.grid(axis=\"y\", linestyle=\"--\", alpha=0.4)\n\nplt.tight_layout()\n\nplt.show()\n\nsegment_distribution.to_frame(\"Number of Customers\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:46:49.725629Z","iopub.execute_input":"2026-07-06T20:46:49.725921Z","iopub.status.idle":"2026-07-06T20:46:50.037790Z","shell.execute_reply.started":"2026-07-06T20:46:49.725895Z","shell.execute_reply":"2026-07-06T20:46:50.036716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# EDA 19: Monthly Revenue Trend\n# ============================================================\n\nmonthly_revenue = (\n    transactions\n    .groupby(\"Year_Month\")[\"price\"]\n    .sum()\n)\n\nplt.figure(figsize=(15,6))\n\nplt.plot(\n    monthly_revenue.index.astype(str),\n    monthly_revenue.values,\n    linewidth=2\n)\n\nplt.title(\"Monthly Revenue Trend\", fontsize=16, fontweight=\"bold\")\nplt.xlabel(\"Month\", fontsize=12)\nplt.ylabel(\"Normalized Revenue\", fontsize=12)\n\nplt.xticks(rotation=45)\n\nplt.grid(alpha=0.3)\n\nplt.tight_layout()\n\nplt.show()\n\nmonthly_revenue.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:46:50.038951Z","iopub.execute_input":"2026-07-06T20:46:50.039305Z","iopub.status.idle":"2026-07-06T20:46:51.181600Z","shell.execute_reply.started":"2026-07-06T20:46:50.039277Z","shell.execute_reply":"2026-07-06T20:46:51.180669Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA 19 – Monthly Revenue Trend\n\nThis visualization presents the total monthly revenue generated during the observation period.\n\nUnlike transaction counts, revenue trends provide a better understanding of the financial performance of the business.\n\nMonthly revenue analysis helps retailers:\n\n- Monitor business growth\n- Identify seasonal demand\n- Evaluate promotional campaigns\n- Support inventory planning\n- Improve sales forecasting\n\nRevenue trends are frequently monitored by retail analysts and business managers to support strategic decision-making.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# EDA 20: Top 20 Revenue-Generating Product Categories\n# ============================================================\n\n# Merge transaction and article datasets\ncategory_revenue = (\n    transactions.merge(\n        articles[[\"article_id\", \"product_type_name\"]],\n        on=\"article_id\",\n        how=\"left\"\n    )\n)\n\n# Calculate total revenue by product category\ncategory_revenue = (\n    category_revenue\n    .groupby(\"product_type_name\")[\"price\"]\n    .sum()\n    .sort_values(ascending=False)\n    .head(20)\n)\n\nplt.figure(figsize=(14,8))\n\nbars = plt.barh(\n    category_revenue.index,\n    category_revenue.values\n)\n\nplt.title(\n    \"Top 20 Revenue-Generating Product Categories\",\n    fontsize=16,\n    fontweight=\"bold\"\n)\n\nplt.xlabel(\"Total Revenue (Normalized)\", fontsize=12)\nplt.ylabel(\"Product Category\", fontsize=12)\n\nplt.gca().invert_yaxis()\n\n# Add value labels\nfor bar in bars:\n    width = bar.get_width()\n    plt.text(\n        width,\n        bar.get_y() + bar.get_height()/2,\n        f\"{width:.1f}\",\n        va=\"center\",\n        fontsize=9\n    )\n\nplt.grid(axis=\"x\", linestyle=\"--\", alpha=0.4)\n\nplt.tight_layout()\n\nplt.show()\n\ncategory_revenue.to_frame(\"Total Revenue\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:46:51.182707Z","iopub.execute_input":"2026-07-06T20:46:51.183013Z","iopub.status.idle":"2026-07-06T20:46:55.382947Z","shell.execute_reply.started":"2026-07-06T20:46:51.182986Z","shell.execute_reply":"2026-07-06T20:46:55.382040Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA 20 – Top 20 Revenue-Generating Product Categories\n\nThis analysis identifies the product categories that generate the highest total revenue during the observation period.\n\nUnlike purchase frequency analysis, revenue analysis highlights the financial contribution of each product category.\n\n### Business Importance\n\nRevenue-based analysis enables retailers to:\n\n- Identify the most profitable product categories.\n- Prioritize inventory for high-revenue items.\n- Develop pricing and promotional strategies.\n- Allocate marketing resources more effectively.\n- Improve recommendation systems by balancing popularity with business value.\n\nWhile some categories may be purchased frequently, they may not necessarily generate the highest revenue. Therefore, combining purchase frequency and revenue analysis provides a more comprehensive understanding of product performance.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# EDA 21: Top 20 Revenue-Generating Products\n# ============================================================\n\n# Merge transaction and article datasets\nproduct_revenue = (\n    transactions.merge(\n        articles[[\"article_id\", \"prod_name\"]],\n        on=\"article_id\",\n        how=\"left\"\n    )\n)\n\n# Calculate total revenue by product\nproduct_revenue = (\n    product_revenue\n    .groupby([\"article_id\", \"prod_name\"])[\"price\"]\n    .sum()\n    .sort_values(ascending=False)\n    .reset_index()\n)\n\n# Select the top 20 products\ntop20_revenue_products = product_revenue.head(20)\n\nplt.figure(figsize=(14,8))\n\nbars = plt.barh(\n    top20_revenue_products[\"prod_name\"],\n    top20_revenue_products[\"price\"]\n)\n\nplt.title(\n    \"Top 20 Revenue-Generating Products\",\n    fontsize=16,\n    fontweight=\"bold\"\n)\n\nplt.xlabel(\"Total Revenue (Normalized)\", fontsize=12)\nplt.ylabel(\"Product Name\", fontsize=12)\n\n# Highest revenue at the top\nplt.gca().invert_yaxis()\n\n# Add value labels\nfor bar in bars:\n    width = bar.get_width()\n    plt.text(\n        width,\n        bar.get_y() + bar.get_height()/2,\n        f\"{width:.1f}\",\n        va=\"center\",\n        fontsize=9\n    )\n\nplt.grid(axis=\"x\", linestyle=\"--\", alpha=0.4)\n\nplt.tight_layout()\n\nplt.show()\n\ntop20_revenue_products","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:46:55.384253Z","iopub.execute_input":"2026-07-06T20:46:55.384665Z","iopub.status.idle":"2026-07-06T20:47:06.105628Z","shell.execute_reply.started":"2026-07-06T20:46:55.384625Z","shell.execute_reply":"2026-07-06T20:47:06.104675Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## EDA 21 – Top 20 Revenue-Generating Products\n\nThis analysis identifies the individual products that contribute the highest total revenue during the observation period.\n\nUnlike purchase frequency analysis, which highlights the most commonly purchased items, revenue analysis focuses on the financial contribution of each product.\n\n### Business Applications\n\nRevenue-generating products can be used to:\n\n- Prioritize inventory management\n- Optimize merchandising strategies\n- Improve pricing decisions\n- Identify premium products\n- Support profit-aware recommendation systems\n\nUnderstanding both product popularity and revenue contribution enables retailers to balance customer preferences with business objectives when designing recommendation systems.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Phase 7: Recommendation System Development\n# ============================================================\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.metrics.pairwise import cosine_similarity\nfrom scipy.sparse import csr_matrix\nfrom sklearn.decomposition import TruncatedSVD","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:06.106899Z","iopub.execute_input":"2026-07-06T20:47:06.107318Z","iopub.status.idle":"2026-07-06T20:47:06.112672Z","shell.execute_reply.started":"2026-07-06T20:47:06.107278Z","shell.execute_reply":"2026-07-06T20:47:06.111550Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Use recent transactions only to reduce memory usage\nrecent_transactions = transactions[\n    transactions[\"t_dat\"] >= transactions[\"t_dat\"].max() - pd.Timedelta(days=90)\n].copy()\n\nprint(\"Recent transactions shape:\", recent_transactions.shape)\nprint(\"Unique customers:\", recent_transactions[\"customer_id\"].nunique())\nprint(\"Unique articles:\", recent_transactions[\"article_id\"].nunique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:06.113915Z","iopub.execute_input":"2026-07-06T20:47:06.114442Z","iopub.status.idle":"2026-07-06T20:47:07.685922Z","shell.execute_reply.started":"2026-07-06T20:47:06.114401Z","shell.execute_reply":"2026-07-06T20:47:07.684711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Model 1: Popularity-Based Recommendation System\n# ============================================================\n\npopular_products = (\n    recent_transactions[\"article_id\"]\n    .value_counts()\n    .reset_index()\n)\n\npopular_products.columns = [\"article_id\", \"purchase_count\"]\n\npopular_products = popular_products.merge(\n    articles[[\"article_id\", \"prod_name\", \"product_type_name\", \"product_group_name\", \"colour_group_name\"]],\n    on=\"article_id\",\n    how=\"left\"\n)\n\ndef recommend_popular_products(top_n=10):\n    return popular_products.head(top_n)\n\nrecommend_popular_products(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:07.687002Z","iopub.execute_input":"2026-07-06T20:47:07.687342Z","iopub.status.idle":"2026-07-06T20:47:07.773723Z","shell.execute_reply.started":"2026-07-06T20:47:07.687316Z","shell.execute_reply":"2026-07-06T20:47:07.772523Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Model 1: Popularity-Based Recommendation System\n# ============================================================\n\npopular_products = (\n    recent_transactions[\"article_id\"]\n    .value_counts()\n    .reset_index()\n)\n\npopular_products.columns = [\"article_id\", \"purchase_count\"]\n\npopular_products = popular_products.merge(\n    articles[[\"article_id\", \"prod_name\", \"product_type_name\", \"product_group_name\", \"colour_group_name\"]],\n    on=\"article_id\",\n    how=\"left\"\n)\n\ndef recommend_popular_products(top_n=10):\n    return popular_products.head(top_n)\n\nrecommend_popular_products(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:07.774925Z","iopub.execute_input":"2026-07-06T20:47:07.775295Z","iopub.status.idle":"2026-07-06T20:47:07.859520Z","shell.execute_reply.started":"2026-07-06T20:47:07.775258Z","shell.execute_reply":"2026-07-06T20:47:07.858529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Model 2: Content-Based Recommendation System\n# ============================================================\n\narticles_content = articles.copy()\n\narticles_content[\"content_features\"] = (\n    articles_content[\"prod_name\"].astype(str) + \" \" +\n    articles_content[\"product_type_name\"].astype(str) + \" \" +\n    articles_content[\"product_group_name\"].astype(str) + \" \" +\n    articles_content[\"colour_group_name\"].astype(str) + \" \" +\n    articles_content[\"department_name\"].astype(str) + \" \" +\n    articles_content[\"garment_group_name\"].astype(str) + \" \" +\n    articles_content[\"detail_desc\"].fillna(\"\").astype(str)\n)\n\n# Use only products bought recently to keep cosine similarity manageable\nrecent_articles = recent_transactions[\"article_id\"].unique()\n\narticles_content_sample = articles_content[\n    articles_content[\"article_id\"].isin(recent_articles)\n].drop_duplicates(\"article_id\").reset_index(drop=True)\n\nprint(\"Content articles shape:\", articles_content_sample.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:07.860853Z","iopub.execute_input":"2026-07-06T20:47:07.861251Z","iopub.status.idle":"2026-07-06T20:47:08.194026Z","shell.execute_reply.started":"2026-07-06T20:47:07.861209Z","shell.execute_reply":"2026-07-06T20:47:08.193189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tfidf = TfidfVectorizer(\n    stop_words=\"english\",\n    max_features=5000\n)\n\ntfidf_matrix = tfidf.fit_transform(articles_content_sample[\"content_features\"])\n\nprint(\"TF-IDF matrix shape:\", tfidf_matrix.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:08.195073Z","iopub.execute_input":"2026-07-06T20:47:08.195523Z","iopub.status.idle":"2026-07-06T20:47:09.412681Z","shell.execute_reply.started":"2026-07-06T20:47:08.195494Z","shell.execute_reply":"2026-07-06T20:47:09.411735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"article_id_to_index = pd.Series(\n    articles_content_sample.index,\n    index=articles_content_sample[\"article_id\"]\n).drop_duplicates()\n\ndef recommend_similar_products(article_id, top_n=10):\n    \n    if article_id not in article_id_to_index:\n        return \"Article ID not found in content sample.\"\n    \n    idx = article_id_to_index[article_id]\n    \n    cosine_scores = cosine_similarity(\n        tfidf_matrix[idx],\n        tfidf_matrix\n    ).flatten()\n    \n    similar_indices = cosine_scores.argsort()[::-1][1:top_n+1]\n    \n    recommendations = articles_content_sample.iloc[similar_indices][\n        [\"article_id\", \"prod_name\", \"product_type_name\", \"product_group_name\", \"colour_group_name\"]\n    ].copy()\n    \n    recommendations[\"similarity_score\"] = cosine_scores[similar_indices]\n    \n    return recommendations\n\n# Example\nsample_article = articles_content_sample[\"article_id\"].iloc[0]\nrecommend_similar_products(sample_article, 10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:09.413858Z","iopub.execute_input":"2026-07-06T20:47:09.414220Z","iopub.status.idle":"2026-07-06T20:47:09.460848Z","shell.execute_reply.started":"2026-07-06T20:47:09.414183Z","shell.execute_reply":"2026-07-06T20:47:09.459993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Model 3: Collaborative Filtering with SVD\n# ============================================================\n\n# Select active customers and popular products to reduce matrix size\ntop_customers = recent_transactions[\"customer_id\"].value_counts().head(10000).index\ntop_articles = recent_transactions[\"article_id\"].value_counts().head(5000).index\n\ncf_data = recent_transactions[\n    (recent_transactions[\"customer_id\"].isin(top_customers)) &\n    (recent_transactions[\"article_id\"].isin(top_articles))\n].copy()\n\nprint(\"Collaborative filtering data shape:\", cf_data.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:09.462034Z","iopub.execute_input":"2026-07-06T20:47:09.463082Z","iopub.status.idle":"2026-07-06T20:47:10.851908Z","shell.execute_reply.started":"2026-07-06T20:47:09.463049Z","shell.execute_reply":"2026-07-06T20:47:10.850711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create customer-item interaction counts\ninteraction_data = (\n    cf_data.groupby([\"customer_id\", \"article_id\"])\n    .size()\n    .reset_index(name=\"interaction\")\n)\n\ncustomer_ids = interaction_data[\"customer_id\"].unique()\narticle_ids = interaction_data[\"article_id\"].unique()\n\ncustomer_id_to_index = {customer_id: idx for idx, customer_id in enumerate(customer_ids)}\narticle_id_to_cf_index = {article_id: idx for idx, article_id in enumerate(article_ids)}\n\ninteraction_data[\"customer_index\"] = interaction_data[\"customer_id\"].map(customer_id_to_index)\ninteraction_data[\"article_index\"] = interaction_data[\"article_id\"].map(article_id_to_cf_index)\n\nuser_item_matrix = csr_matrix(\n    (\n        interaction_data[\"interaction\"],\n        (\n            interaction_data[\"customer_index\"],\n            interaction_data[\"article_index\"]\n        )\n    ),\n    shape=(len(customer_ids), len(article_ids))\n)\n\nprint(\"User-item matrix shape:\", user_item_matrix.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:10.853498Z","iopub.execute_input":"2026-07-06T20:47:10.854614Z","iopub.status.idle":"2026-07-06T20:47:11.122776Z","shell.execute_reply.started":"2026-07-06T20:47:10.854572Z","shell.execute_reply":"2026-07-06T20:47:11.121718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"svd = TruncatedSVD(\n    n_components=50,\n    random_state=42\n)\n\ncustomer_factors = svd.fit_transform(user_item_matrix)\narticle_factors = svd.components_.T\n\nprint(\"Customer factors shape:\", customer_factors.shape)\nprint(\"Article factors shape:\", article_factors.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:11.123877Z","iopub.execute_input":"2026-07-06T20:47:11.124233Z","iopub.status.idle":"2026-07-06T20:47:11.580638Z","shell.execute_reply.started":"2026-07-06T20:47:11.124196Z","shell.execute_reply":"2026-07-06T20:47:11.577905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def recommend_svd_products(customer_id, top_n=10):\n    \n    if customer_id not in customer_id_to_index:\n        return \"Customer ID not found in collaborative filtering sample.\"\n    \n    customer_index = customer_id_to_index[customer_id]\n    \n    customer_vector = customer_factors[customer_index]\n    \n    scores = np.dot(article_factors, customer_vector)\n    \n    purchased_articles = set(\n        cf_data[cf_data[\"customer_id\"] == customer_id][\"article_id\"].unique()\n    )\n    \n    recommendations = []\n    \n    reverse_article_map = {idx: article_id for article_id, idx in article_id_to_cf_index.items()}\n    \n    for article_index in scores.argsort()[::-1]:\n        article_id = reverse_article_map[article_index]\n        \n        if article_id not in purchased_articles:\n            recommendations.append(article_id)\n        \n        if len(recommendations) == top_n:\n            break\n    \n    result = pd.DataFrame({\"article_id\": recommendations})\n    \n    result = result.merge(\n        articles[[\"article_id\", \"prod_name\", \"product_type_name\", \"product_group_name\", \"colour_group_name\"]],\n        on=\"article_id\",\n        how=\"left\"\n    )\n    \n    return result\n\n# Example\nsample_customer = cf_data[\"customer_id\"].iloc[0]\nrecommend_svd_products(sample_customer, 10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:11.581752Z","iopub.execute_input":"2026-07-06T20:47:11.585418Z","iopub.status.idle":"2026-07-06T20:47:11.677106Z","shell.execute_reply.started":"2026-07-06T20:47:11.585373Z","shell.execute_reply":"2026-07-06T20:47:11.676221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def recommend_svd_products(customer_id, top_n=10):\n    \n    if customer_id not in customer_id_to_index:\n        return \"Customer ID not found in collaborative filtering sample.\"\n    \n    customer_index = customer_id_to_index[customer_id]\n    \n    customer_vector = customer_factors[customer_index]\n    \n    scores = np.dot(article_factors, customer_vector)\n    \n    purchased_articles = set(\n        cf_data[cf_data[\"customer_id\"] == customer_id][\"article_id\"].unique()\n    )\n    \n    recommendations = []\n    \n    reverse_article_map = {idx: article_id for article_id, idx in article_id_to_cf_index.items()}\n    \n    for article_index in scores.argsort()[::-1]:\n        article_id = reverse_article_map[article_index]\n        \n        if article_id not in purchased_articles:\n            recommendations.append(article_id)\n        \n        if len(recommendations) == top_n:\n            break\n    \n    result = pd.DataFrame({\"article_id\": recommendations})\n    \n    result = result.merge(\n        articles[[\"article_id\", \"prod_name\", \"product_type_name\", \"product_group_name\", \"colour_group_name\"]],\n        on=\"article_id\",\n        how=\"left\"\n    )\n    \n    return result\n\n# Example\nsample_customer = cf_data[\"customer_id\"].iloc[0]\nrecommend_svd_products(sample_customer, 10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:11.678268Z","iopub.execute_input":"2026-07-06T20:47:11.678539Z","iopub.status.idle":"2026-07-06T20:47:11.748806Z","shell.execute_reply.started":"2026-07-06T20:47:11.678515Z","shell.execute_reply":"2026-07-06T20:47:11.747938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================================\n# Model 4: Hybrid Recommendation System\n# ============================================================\n\ndef recommend_hybrid(customer_id, top_n=10):\n    \n    hybrid_recommendations = []\n    \n    # 1. Collaborative filtering recommendations\n    svd_recs = recommend_svd_products(customer_id, top_n=top_n)\n    \n    if isinstance(svd_recs, pd.DataFrame):\n        hybrid_recommendations.extend(svd_recs[\"article_id\"].tolist())\n    \n    # 2. Add popular products if not enough recommendations\n    popular_recs = popular_products[\"article_id\"].head(50).tolist()\n    \n    for article_id in popular_recs:\n        if article_id not in hybrid_recommendations:\n            hybrid_recommendations.append(article_id)\n        \n        if len(hybrid_recommendations) == top_n:\n            break\n    \n    result = pd.DataFrame({\"article_id\": hybrid_recommendations[:top_n]})\n    \n    result = result.merge(\n        articles[[\"article_id\", \"prod_name\", \"product_type_name\", \"product_group_name\", \"colour_group_name\"]],\n        on=\"article_id\",\n        how=\"left\"\n    )\n    \n    return result\n\nrecommend_hybrid(sample_customer, 10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:11.750168Z","iopub.execute_input":"2026-07-06T20:47:11.751036Z","iopub.status.idle":"2026-07-06T20:47:11.837110Z","shell.execute_reply.started":"2026-07-06T20:47:11.751001Z","shell.execute_reply":"2026-07-06T20:47:11.836051Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Phase 7 – Recommendation System Development\n\nFour recommendation approaches were developed:\n\n1. **Popularity-Based Recommender**  \n   Recommends the most frequently purchased products.\n\n2. **Content-Based Recommender**  \n   Uses product metadata and TF-IDF cosine similarity to recommend similar products.\n\n3. **Collaborative Filtering Recommender**  \n   Uses customer-product interaction patterns and matrix factorization with SVD.\n\n4. **Hybrid Recommender**  \n   Combines collaborative filtering with popularity-based recommendations to improve coverage and handle cold-start cases.\n\nThis progression follows a real-world recommendation system development workflow, starting with a simple baseline and gradually moving toward more personalized recommendation models.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Phase 8: Recommendation System Evaluation\n# ============================================================\n\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:11.838325Z","iopub.execute_input":"2026-07-06T20:47:11.838666Z","iopub.status.idle":"2026-07-06T20:47:11.843979Z","shell.execute_reply.started":"2026-07-06T20:47:11.838638Z","shell.execute_reply":"2026-07-06T20:47:11.842772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Use recent transaction data for evaluation\neval_data = recent_transactions.copy()\n\n# Sort by customer and date\neval_data = eval_data.sort_values([\"customer_id\", \"t_dat\"])\n\n# Use the last purchased item of each customer as test data\ntest_data = eval_data.groupby(\"customer_id\").tail(1)\n\n# Use all previous purchases as train data\ntrain_data = eval_data.drop(test_data.index)\n\nprint(\"Train shape:\", train_data.shape)\nprint(\"Test shape :\", test_data.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:11.845170Z","iopub.execute_input":"2026-07-06T20:47:11.845514Z","iopub.status.idle":"2026-07-06T20:47:16.313446Z","shell.execute_reply.started":"2026-07-06T20:47:11.845478Z","shell.execute_reply":"2026-07-06T20:47:16.312538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ground_truth = (\n    test_data.groupby(\"customer_id\")[\"article_id\"]\n    .apply(list)\n    .to_dict()\n)\n\nprint(\"Number of test customers:\", len(ground_truth))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:16.314607Z","iopub.execute_input":"2026-07-06T20:47:16.314895Z","iopub.status.idle":"2026-07-06T20:47:26.533047Z","shell.execute_reply.started":"2026-07-06T20:47:16.314862Z","shell.execute_reply":"2026-07-06T20:47:26.531977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ground_truth = (\n    test_data.groupby(\"customer_id\")[\"article_id\"]\n    .apply(list)\n    .to_dict()\n)\n\nprint(\"Number of test customers:\", len(ground_truth))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:26.534197Z","iopub.execute_input":"2026-07-06T20:47:26.534531Z","iopub.status.idle":"2026-07-06T20:47:36.896709Z","shell.execute_reply.started":"2026-07-06T20:47:26.534502Z","shell.execute_reply":"2026-07-06T20:47:36.895747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def precision_at_k(recommended, actual, k=10):\n    recommended = recommended[:k]\n    actual = set(actual)\n\n    if len(recommended) == 0:\n        return 0\n\n    hits = len(set(recommended) & actual)\n    return hits / k\n\n\ndef recall_at_k(recommended, actual, k=10):\n    recommended = recommended[:k]\n    actual = set(actual)\n\n    if len(actual) == 0:\n        return 0\n\n    hits = len(set(recommended) & actual)\n    return hits / len(actual)\n\n\ndef hit_rate_at_k(recommended, actual, k=10):\n    recommended = recommended[:k]\n    actual = set(actual)\n\n    return int(len(set(recommended) & actual) > 0)\n\n\ndef average_precision_at_k(recommended, actual, k=10):\n    recommended = recommended[:k]\n    actual = set(actual)\n\n    score = 0\n    hits = 0\n\n    for i, item in enumerate(recommended):\n        if item in actual:\n            hits += 1\n            score += hits / (i + 1)\n\n    if hits == 0:\n        return 0\n\n    return score / min(len(actual), k)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:36.897753Z","iopub.execute_input":"2026-07-06T20:47:36.898132Z","iopub.status.idle":"2026-07-06T20:47:36.914889Z","shell.execute_reply.started":"2026-07-06T20:47:36.898106Z","shell.execute_reply":"2026-07-06T20:47:36.913819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_popular_recommendations(top_n=10):\n    return popular_products[\"article_id\"].head(top_n).tolist()\n\n\nsample_customers = list(ground_truth.keys())[:5000]\n\npop_precision = []\npop_recall = []\npop_hit_rate = []\npop_map = []\n\npopular_recs = get_popular_recommendations(10)\n\nfor customer_id in sample_customers:\n    actual = ground_truth[customer_id]\n\n    pop_precision.append(precision_at_k(popular_recs, actual, 10))\n    pop_recall.append(recall_at_k(popular_recs, actual, 10))\n    pop_hit_rate.append(hit_rate_at_k(popular_recs, actual, 10))\n    pop_map.append(average_precision_at_k(popular_recs, actual, 10))\n\npopularity_results = {\n    \"Model\": \"Popularity-Based\",\n    \"Precision@10\": np.mean(pop_precision),\n    \"Recall@10\": np.mean(pop_recall),\n    \"Hit Rate@10\": np.mean(pop_hit_rate),\n    \"MAP@10\": np.mean(pop_map)\n}\n\npopularity_results","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:36.916093Z","iopub.execute_input":"2026-07-06T20:47:36.916475Z","iopub.status.idle":"2026-07-06T20:47:36.982621Z","shell.execute_reply.started":"2026-07-06T20:47:36.916440Z","shell.execute_reply":"2026-07-06T20:47:36.981722Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Phase 8 – Model Evaluation\n\nThe recommendation systems are evaluated using ranking-based metrics commonly used in recommender systems.\n\nThe evaluation uses a temporal holdout strategy:\n\n- The last purchase of each customer is used as the test item.\n- Previous purchases are used as historical data.\n- The model is evaluated based on whether it can recommend the held-out product.\n\nThe metrics used are:\n\n- **Precision@10:** Fraction of recommended products that are relevant.\n- **Recall@10:** Fraction of relevant products successfully recommended.\n- **Hit Rate@10:** Whether at least one correct item appears in the top 10.\n- **MAP@10:** Mean Average Precision at 10, which rewards correct recommendations appearing earlier in the ranking.","metadata":{}},{"cell_type":"code","source":"# ============================================================\n# Step 8.5: Evaluate Collaborative Filtering (SVD)\n# ============================================================\n\nsvd_precision = []\nsvd_recall = []\nsvd_hit_rate = []\nsvd_map = []\n\n# Evaluate only customers present in the SVD model\nsvd_customers = [\n    customer for customer in sample_customers\n    if customer in customer_id_to_index\n]\n\nfor customer in svd_customers:\n\n    actual = ground_truth[customer]\n\n    recommendations = recommend_svd_products(customer, top_n=10)\n\n    if isinstance(recommendations, str):\n        continue\n\n    recommended = recommendations[\"article_id\"].tolist()\n\n    svd_precision.append(\n        precision_at_k(recommended, actual, 10)\n    )\n\n    svd_recall.append(\n        recall_at_k(recommended, actual, 10)\n    )\n\n    svd_hit_rate.append(\n        hit_rate_at_k(recommended, actual, 10)\n    )\n\n    svd_map.append(\n        average_precision_at_k(recommended, actual, 10)\n    )\n\nsvd_results = {\n    \"Model\": \"Collaborative Filtering (SVD)\",\n    \"Precision@10\": np.mean(svd_precision),\n    \"Recall@10\": np.mean(svd_recall),\n    \"Hit Rate@10\": np.mean(svd_hit_rate),\n    \"MAP@10\": np.mean(svd_map)\n}\n\nsvd_results","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:36.983670Z","iopub.execute_input":"2026-07-06T20:47:36.984050Z","iopub.status.idle":"2026-07-06T20:47:42.677534Z","shell.execute_reply.started":"2026-07-06T20:47:36.984022Z","shell.execute_reply":"2026-07-06T20:47:42.676458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"comparison = pd.DataFrame([\n    popularity_results,\n    svd_results\n])\n\ncomparison","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-06T20:47:42.678719Z","iopub.execute_input":"2026-07-06T20:47:42.679023Z","iopub.status.idle":"2026-07-06T20:47:42.691746Z","shell.execute_reply.started":"2026-07-06T20:47:42.678996Z","shell.execute_reply":"2026-07-06T20:47:42.690729Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Recommendation Model Evaluation\n\nThe popularity-based recommendation model serves as a baseline and achieved a small but non-zero recommendation accuracy.\n\nThe simplified collaborative filtering implementation produced very low evaluation scores on this large-scale dataset. This is expected because:\n\n- The evaluation uses a temporal hold-out strategy.\n- The interaction matrix was sampled to reduce computational requirements.\n- Only a subset of customers and products was used for matrix factorization due to memory constraints.\n\nIn production environments, collaborative filtering models are typically trained using specialized recommendation frameworks such as LightFM, Implicit, or Surprise on dedicated hardware.\n\nDespite these computational constraints, the implementation demonstrates the complete workflow of building and evaluating multiple recommendation approaches for a large-scale retail dataset.## Recommendation Model Evaluation\n\nThe popularity-based recommendation model serves as a baseline and achieved a small but non-zero recommendation accuracy.\n\nThe simplified collaborative filtering implementation produced very low evaluation scores on this large-scale dataset. This is expected because:\n\n- The evaluation uses a temporal hold-out strategy.\n- The interaction matrix was sampled to reduce computational requirements.\n- Only a subset of customers and products was used for matrix factorization due to memory constraints.\n\nIn production environments, collaborative filtering models are typically trained using specialized recommendation frameworks such as LightFM, Implicit, or Surprise on dedicated hardware.\n\nDespite these computational constraints, the implementation demonstrates the complete workflow of building and evaluating multiple recommendation approaches for a large-scale retail dataset.","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}