{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\nfrom tqdm.notebook import tqdm\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T16:59:14.964422Z","iopub.execute_input":"2025-03-04T16:59:14.964773Z","iopub.status.idle":"2025-03-04T16:59:16.220541Z","shell.execute_reply.started":"2025-03-04T16:59:14.964741Z","shell.execute_reply":"2025-03-04T16:59:16.219509Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Reading Data from H&M Personalized Fashion Recommendations Dataset","metadata":{}},{"cell_type":"code","source":"articles = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ncustomers = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/customers.csv\")\ntransactions = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T15:01:41.708158Z","iopub.execute_input":"2025-03-04T15:01:41.708677Z","iopub.status.idle":"2025-03-04T15:03:02.219256Z","shell.execute_reply.started":"2025-03-04T15:01:41.708641Z","shell.execute_reply":"2025-03-04T15:03:02.218381Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Displaying the first few rows of the articles dataframe","metadata":{}},{"cell_type":"code","source":"articles.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T15:03:02.255684Z","iopub.execute_input":"2025-03-04T15:03:02.256098Z","iopub.status.idle":"2025-03-04T15:03:02.295434Z","shell.execute_reply.started":"2025-03-04T15:03:02.256062Z","shell.execute_reply":"2025-03-04T15:03:02.294497Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Printing the shapes of the dataframes","metadata":{}},{"cell_type":"code","source":"print(\"Articles:\", articles.shape)\nprint(\"Customers:\", customers.shape)\nprint(\"Transactions:\", transactions.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T15:03:02.297494Z","iopub.execute_input":"2025-03-04T15:03:02.297753Z","iopub.status.idle":"2025-03-04T15:03:02.303656Z","shell.execute_reply.started":"2025-03-04T15:03:02.297730Z","shell.execute_reply":"2025-03-04T15:03:02.302715Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Checking for missing values in the transactions dataframe","metadata":{}},{"cell_type":"code","source":"# Check for missing values\nprint(\"Missing Values:\\n\", transactions.isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T15:03:02.307452Z","iopub.execute_input":"2025-03-04T15:03:02.307737Z","iopub.status.idle":"2025-03-04T15:03:05.549154Z","shell.execute_reply.started":"2025-03-04T15:03:02.307715Z","shell.execute_reply":"2025-03-04T15:03:05.548231Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Cleaning and Initial Exploration for H&M Personalized Fashion Recommendations Dataset","metadata":{}},{"cell_type":"code","source":"# Display first few rows\nprint(articles.head())\nprint(customers.head())\nprint(transactions.head())\n\n# Identify NaN or infinite values\nprint(\"Articles NaN values:\\n\", articles.isna().sum())\nprint(\"Customers NaN values:\\n\", customers.isna().sum())\nprint(\"Transactions NaN values:\\n\", transactions.isna().sum())\n\n# Handle NaN values (e.g., fill NaNs with 0 or drop rows)\narticles = articles.fillna(0)\ncustomers = customers.fillna(0)\ntransactions = transactions.fillna(0)\n\n# Remove infinite values (if any)\narticles = articles.replace([np.inf, -np.inf], np.nan).dropna()\ncustomers = customers.replace([np.inf, -np.inf], np.nan).dropna()\ntransactions = transactions.replace([np.inf, -np.inf], np.nan).dropna()\n\n# Display first few rows again after cleaning\nprint(articles.head())\nprint(customers.head())\nprint(transactions.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T15:03:05.550178Z","iopub.execute_input":"2025-03-04T15:03:05.550536Z","iopub.status.idle":"2025-03-04T15:03:48.635323Z","shell.execute_reply.started":"2025-03-04T15:03:05.550505Z","shell.execute_reply":"2025-03-04T15:03:48.633862Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n# Filling missing values in customers dataframe","metadata":{}},{"cell_type":"code","source":"transactions['t_dat'] = pd.to_datetime(transactions['t_dat']) # Convert date\ncustomers.fillna(0, inplace=True) # Fill missing values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T15:20:11.660720Z","iopub.execute_input":"2025-03-04T15:20:11.661014Z","iopub.status.idle":"2025-03-04T15:20:12.543187Z","shell.execute_reply.started":"2025-03-04T15:20:11.660980Z","shell.execute_reply":"2025-03-04T15:20:12.542329Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Analyzing and Visualizing Monthly Purchase Trends\n","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Extract month-year from date\ntransactions['month'] = transactions['t_dat'].dt.to_period('M')\n\n# Aggregate sales by month\nmonthly_sales = transactions.groupby('month').size()\n\n# Plot sales trend\nplt.figure(figsize=(12,5))\nmonthly_sales.plot(kind='line', marker='o', color='blue')\nplt.xlabel(\"Month\")\nplt.ylabel(\"Number of Purchases\")\nplt.title(\"Purchase Trend Over Time\")\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T15:03:52.634439Z","iopub.execute_input":"2025-03-04T15:03:52.634702Z","iopub.status.idle":"2025-03-04T15:03:55.262122Z","shell.execute_reply.started":"2025-03-04T15:03:52.634680Z","shell.execute_reply":"2025-03-04T15:03:55.260915Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Findings:\n\nSeasonal Peaks and Troughs: Specific months show consistently high or low sales, suggesting seasonal buying patterns.\n\nOverall Growth Trend: The trend line indicates whether sales are generally increasing, decreasing, or stable over time.\n\nMonthly Variations: Noticeable month-to-month fluctuations in sales.\n\nInsights:\n\nMarketing Timing: Schedule key campaigns during peak sales months to maximize impact.\n\nResource Allocation: Optimize inventory and staff resources based on identified sales trends.\n\nCustomer Behavior: Tailor marketing strategies and product offerings based on observed purchasing patterns.\n\nPerformance Measurement: Use trends to set benchmarks and goals, evaluating the effectiveness of business strategies.","metadata":{}},{"cell_type":"code","source":"# import seaborn as sns\n# import matplotlib.pyplot as plt\n\n# # Check and print column names\n# print(customers.columns)\n\n# # Verify and strip column names\n# customers.columns = customers.columns.str.strip().str.lower()\n\n# # Plot for 'active' column\n# if 'active' in customers.columns:\n#     sns.countplot(x=customers['active'])\n#     plt.title(\"Active Customers Distribution\")\n#     plt.show()\n# else:\n#     print(\"Column 'active' does not exist in the DataFrame.\")\n\n# # Plot for 'FN' column\n# if 'fn' in customers.columns:\n#     sns.countplot(x=customers['fn'])\n#     plt.title(\"Fashion News Subscription\")\n#     plt.show()\n# else:\n#     print(\"Column 'FN' does not exist in the DataFrame.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T15:03:55.263400Z","iopub.execute_input":"2025-03-04T15:03:55.263787Z","iopub.status.idle":"2025-03-04T15:03:55.747410Z","shell.execute_reply.started":"2025-03-04T15:03:55.263749Z","shell.execute_reply":"2025-03-04T15:03:55.746473Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Comparing Online vs Store Sales Proportions\n","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Analyze sales_channel_id (1 = Store, 2 = Online)\nsales_channel = transactions['sales_channel_id'].value_counts()\nsales_channel.plot(kind='pie', autopct='%1.1f%%', title=\"Online vs Store Sales\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T15:03:55.748484Z","iopub.execute_input":"2025-03-04T15:03:55.748805Z","iopub.status.idle":"2025-03-04T15:03:56.038499Z","shell.execute_reply.started":"2025-03-04T15:03:55.748780Z","shell.execute_reply":"2025-03-04T15:03:56.037618Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Findings:\n\nSales Channel Distribution: Proportion of sales via Store vs. Online.\n\nInsights:\n\nChannel Preference: Highlight whether customers prefer online or in-store shopping.\n\nStrategic Focus: Allocate resources based on dominant sales channels.","metadata":{}},{"cell_type":"markdown","source":"# Statistical Summary and Sales Channel Distribution Analysis\n","metadata":{}},{"cell_type":"code","source":"# Summary of numerical columns\nprint(\"\\nSummary Statistics:\\n\")\nprint(transactions.describe())\n\n# Unique values in categorical columns\nprint(\"\\nUnique Categories in 'Articles':\\n\", articles.nunique())\nprint(\"\\nUnique Categories in 'Customers':\\n\", customers.nunique())\n\n# Distribution of sales channels\nplt.figure(figsize=(6,4))\nsns.countplot(x=\"sales_channel_id\", data=transactions, palette=\"coolwarm\")\nplt.title(\"Sales Channel Distribution (1 = Store, 2 = Online)\")\nplt.xlabel(\"Sales Channel\")\nplt.ylabel(\"Count\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T15:03:56.039388Z","iopub.execute_input":"2025-03-04T15:03:56.039629Z","iopub.status.idle":"2025-03-04T15:04:04.338285Z","shell.execute_reply.started":"2025-03-04T15:03:56.039608Z","shell.execute_reply":"2025-03-04T15:04:04.337335Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Findings:\n\nSummary Statistics:\n\nNumerical Columns: The summary statistics provide an overview of the central tendencies, dispersion, and shape of the distribution of numerical data.\n\nUnique Categories:\n\nArticles: The number of unique article categories.\n\nCustomers: The number of unique customer categories.\n\nSales Channel Distribution:\n\nThe distribution of transactions across sales channels (Store vs. Online).\n\nInsights:\n\nNumerical Data Analysis:\n\nUnderstanding the mean, median, standard deviation, and range of numerical columns helps in identifying data patterns and anomalies.\n\nCategory Diversity:\n\nThe count of unique articles and customers indicates the diversity in product offerings and customer base.\n\nSales Channel Preference:\n\nThe count plot of sales channel distribution reveals customer preference for either store or online shopping, which can guide resource allocation and marketing strategies.","metadata":{}},{"cell_type":"markdown","source":"# Detailed Customer Data Analysis and Subscription Behavior Insights\n","metadata":{}},{"cell_type":"code","source":"# Check column names\nprint(customers.columns)\n\n# Age distribution of customers\nplt.figure(figsize=(8,5))\nsns.histplot(customers[\"age\"].dropna(), bins=30, kde=True, color=\"blue\")\nplt.title(\"Customer Age Distribution\")\nplt.xlabel(\"Age\")\nplt.ylabel(\"Count\")\nplt.show()\n\n# Impact of newsletter subscription (fn) and Active status\n# Replace 'fn' with the correct column name if it's different\nplt.figure(figsize=(10,5))\nsns.countplot(x=\"fn\", data=customers, palette=\"Set1\")\nplt.title(\"Customers Receiving Fashion News\")\nplt.xlabel(\"Fashion News Subscription (fn)\")\nplt.ylabel(\"Count\")\nplt.show()\n\n# Replace 'active' with the correct column name if it's different\nplt.figure(figsize=(10,5))\nsns.countplot(x=\"active\", data=customers, palette=\"Set2\")\nplt.title(\"Active Customers\")\nplt.xlabel(\"Active Status\")\nplt.ylabel(\"Count\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T15:04:04.341825Z","iopub.execute_input":"2025-03-04T15:04:04.342138Z","iopub.status.idle":"2025-03-04T15:04:11.243791Z","shell.execute_reply.started":"2025-03-04T15:04:04.342112Z","shell.execute_reply":"2025-03-04T15:04:11.242855Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Findings:\n\nColumn Names: Identify the column names in the customers dataset.\n\nAge Distribution: Histogram showing the distribution of customer ages.\n\nNewsletter Subscription: Count plot displaying the number of customers subscribed to fashion news.\n\nActive Status: Count plot showing the number of active customers.\n\nInsights:\n\nCustomer Demographics: The age distribution provides insights into the age groups of your customer base, helping tailor marketing strategies to different demographics.\n\nNewsletter Engagement: Understanding the proportion of customers subscribed to fashion news can inform the effectiveness of your newsletter campaigns and opportunities for engagement.\n\nCustomer Activity: The distribution of active customers highlights the level of customer engagement and retention, guiding efforts to maintain and increase active participation.","metadata":{}},{"cell_type":"markdown","source":"# Analyzing Top-Selling Product Categories and Colors\n","metadata":{}},{"cell_type":"code","source":"# Identify top-selling product categories and colors.\n\n# Most purchased products\ntop_products = transactions[\"article_id\"].value_counts().head(10).sort_values(ascending=False)\ntop_products.plot(kind='bar', title='Top 10 Most Purchased Products')\nplt.xlabel(\"Article ID\")\nplt.ylabel(\"Number of Purchases\")\nplt.xticks(rotation=90)\nplt.show()\n\n# Top garment groups\ntop_garments = articles[\"product_group_name\"].value_counts().head(10).sort_values(ascending=False)\ntop_garments.plot(kind='bar', title='Top 10 Product Groups')\nplt.xlabel(\"Product Group\")\nplt.ylabel(\"Count\")\nplt.xticks(rotation=90)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T15:56:46.326349Z","iopub.execute_input":"2025-03-04T15:56:46.326815Z","iopub.status.idle":"2025-03-04T15:56:48.003918Z","shell.execute_reply.started":"2025-03-04T15:56:46.326780Z","shell.execute_reply":"2025-03-04T15:56:48.002975Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Findings:\n\nTop-Selling Products: The bar chart highlights the top 10 most purchased products based on Article IDs.\n\nPopular Product Groups: The bar chart shows the top 10 product groups with the highest sales counts.\n\nInsights:\n\nBest Sellers: Recognizing the most purchased products helps optimize inventory and focus marketing efforts on high-demand items.\n\nPopular Categories: Identifying the top product groups allows for strategic planning in product development and promotions, enhancing customer satisfaction and driving sales.","metadata":{}},{"cell_type":"markdown","source":"# Temporal Analysis of Daily Purchase Trends\n","metadata":{}},{"cell_type":"code","source":"#Temporal Analysis\n#Analyze purchase trends over time.\n\n# Convert date column to datetime format\ntransactions[\"t_dat\"] = pd.to_datetime(transactions[\"t_dat\"])\n\n# Daily sales trends\ndaily_sales = transactions.groupby(\"t_dat\").size()\n\nplt.figure(figsize=(12,5))\nsns.lineplot(x=daily_sales.index, y=daily_sales.values, color=\"green\")\nplt.title(\"Daily Purchase Trends\")\nplt.xlabel(\"Date\")\nplt.ylabel(\"Number of Purchases\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T15:04:13.747901Z","iopub.execute_input":"2025-03-04T15:04:13.748218Z","iopub.status.idle":"2025-03-04T15:04:15.445829Z","shell.execute_reply.started":"2025-03-04T15:04:13.748183Z","shell.execute_reply":"2025-03-04T15:04:15.444491Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Findings:\n\nDaily Purchase Trends: The line plot shows the number of purchases made each day, highlighting fluctuations and trends over time.\n\nInsights:\n\nPeak Days: Identify days with the highest number of purchases, which can inform decisions about promotional timing and resource allocation.\n\nTrend Analysis: Observe overall trends, such as increasing or decreasing purchase volumes, to understand customer behavior and adjust strategies accordingly.\n\nAnomalies: Detect any unusual spikes or drops in daily purchases, which may require further investigation to understand underlying causes.","metadata":{}},{"cell_type":"markdown","source":"# Loading and Displaying the Transactions Dataset","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport datetime as dt\n\n# Load dataset\ntransactions = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")\n\n# Display first few rows\nprint(transactions.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T15:04:15.446931Z","iopub.execute_input":"2025-03-04T15:04:15.447389Z","iopub.status.idle":"2025-03-04T15:05:15.465326Z","shell.execute_reply.started":"2025-03-04T15:04:15.447359Z","shell.execute_reply":"2025-03-04T15:05:15.463574Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loading Dataset and Displaying Column Names","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport datetime as dt\n\n# Load dataset\ntransactions = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")\n\n# Display column names\nprint(transactions.columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T15:05:15.466666Z","iopub.execute_input":"2025-03-04T15:05:15.467063Z","iopub.status.idle":"2025-03-04T15:05:58.000645Z","shell.execute_reply.started":"2025-03-04T15:05:15.467028Z","shell.execute_reply":"2025-03-04T15:05:57.999118Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Customer Segmentation Using RFM Analysis and Visualization","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport datetime as dt\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Load datasets\narticles = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ncustomers = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/customers.csv\")\ntransactions = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")\n\n# Convert 't_dat' to datetime (assuming 't_dat' is the correct column name for transaction date)\ntransactions['t_dat'] = pd.to_datetime(transactions['t_dat'])\n\n# Handle missing values if any\ntransactions.dropna(inplace=True)\n\n# Set reference date (e.g., one day after the last transaction date)\nreference_date = transactions['t_dat'].max() + dt.timedelta(days=1)\n\n# Calculate Recency, Frequency, and Monetary values\nrfm = transactions.groupby('customer_id').agg({\n    't_dat': lambda x: (reference_date - x.max()).days,  # Recency\n    'customer_id': 'count',                             # Frequency\n    'price': 'sum'                                      # Monetary\n}).rename(columns={'t_dat': 'Recency', 'customer_id': 'Frequency', 'price': 'Monetary'})\n\n# Define RFM score function\ndef rfm_score(x, quantiles, metric):\n    if x <= quantiles[metric][0.25]:\n        return 1\n    elif x <= quantiles[metric][0.50]:\n        return 2\n    elif x <= quantiles[metric][0.75]:\n        return 3\n    else:\n        return 4\n\n# Calculate quantiles\nquantiles = rfm.quantile(q=[0.25, 0.50, 0.75]).to_dict()\n\n# Assign RFM scores\nrfm['R_Score'] = rfm['Recency'].apply(rfm_score, args=(quantiles, 'Recency'))\nrfm['F_Score'] = rfm['Frequency'].apply(rfm_score, args=(quantiles, 'Frequency'))\nrfm['M_Score'] = rfm['Monetary'].apply(rfm_score, args=(quantiles, 'Monetary'))\n\n# Combine RFM scores into a single score\nrfm['RFM_Score'] = rfm['R_Score'].map(str) + rfm['F_Score'].map(str) + rfm['M_Score'].map(str)\n\n# Define customer segments\nrfm['Segment'] = 'Other'\nrfm.loc[rfm['RFM_Score'].str.startswith('1'), 'Segment'] = 'Best Customers'\nrfm.loc[rfm['RFM_Score'].str.startswith('4'), 'Segment'] = 'Lost Customers'\n\n# Display RFM scores and segments\nprint(rfm.head())\n\n# Calculate average RFM values per segment\nsegment_analysis = rfm.groupby('Segment').agg({\n    'Recency': 'mean',\n    'Frequency': 'mean',\n    'Monetary': ['mean', 'count']\n}).round(1)\n\n# Display segment analysis\nprint(segment_analysis)\n\n# Plot segment distribution\nplt.figure(figsize=(10, 6))\nsns.countplot(data=rfm, x='Segment', order=rfm['Segment'].value_counts().index, palette='viridis')\nplt.title('Customer Segments Distribution')\nplt.xlabel('Segment')\nplt.ylabel('Number of Customers')\nplt.xticks(rotation=45)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T15:09:30.423634Z","iopub.execute_input":"2025-03-04T15:09:30.424034Z","iopub.status.idle":"2025-03-04T15:12:46.418709Z","shell.execute_reply.started":"2025-03-04T15:09:30.424003Z","shell.execute_reply":"2025-03-04T15:12:46.417691Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Findings:\n\nRFM Metrics:\n\nCalculated Recency, Frequency, and Monetary values for each customer.\n\nCustomer Segments:\n\nBest Customers: Customers with the most recent purchases, high frequency, and high monetary value.\n\nLost Customers: Customers with the least recent purchases.\n\nInsights:\n\nCustomer Loyalty:\n\nBest Customers: Focus on retaining these valuable customers through loyalty programs, personalized offers, and excellent customer service.\n\nLost Customers: Implement targeted re-engagement campaigns to win back these customers, such as special discounts or personalized emails.\n\nStrategic Resource Allocation:\n\nAllocate marketing and sales efforts efficiently by focusing on high-value segments and addressing the needs of different customer groups.","metadata":{}},{"cell_type":"markdown","source":"# Calculating RFM Metrics for Customer Segmentation\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport datetime as dt\n\n# Load Data\ntransactions = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")\n\n# Convert date column to datetime\ntransactions['t_dat'] = pd.to_datetime(transactions['t_dat'])\n\n# Data Preparation\nnow = dt.datetime(2022, 1, 1)  # assuming the current date for analysis\nrfm = transactions.groupby('customer_id').agg({\n    't_dat': lambda x: (now - x.max()).days,  # Recency\n    'customer_id': 'count',  # Frequency\n    'price': 'sum'  # Monetary\n}).rename(columns={'t_dat': 'Recency', 'customer_id': 'Frequency', 'price': 'Monetary'})\n\n# Normalize Scores\nrfm['R_Score'] = pd.qcut(rfm['Recency'], 5, labels=[5, 4, 3, 2, 1])\nrfm['F_Score'] = pd.qcut(rfm['Frequency'], 5, labels=[1, 2, 3, 4, 5])\nrfm['M_Score'] = pd.qcut(rfm['Monetary'], 5, labels=[1, 2, 3, 4, 5])\n\n# Calculate RFM Score\nrfm['RFM_Score'] = rfm.R_Score.astype(str) + rfm.F_Score.astype(str) + rfm.M_Score.astype(str)\n\n# Segment customers based on RFM score\nrfm['Segment'] = rfm.apply(lambda row: 'Champion' if row['RFM_Score'] == '555' else 'Others', axis=1)\n\n# Show the segmented customers\nprint(rfm.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T15:15:59.534897Z","iopub.execute_input":"2025-03-04T15:15:59.535287Z","iopub.status.idle":"2025-03-04T15:19:19.313383Z","shell.execute_reply.started":"2025-03-04T15:15:59.535235Z","shell.execute_reply":"2025-03-04T15:19:19.312326Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Findings:\n\nRFM Metrics:\n\nCalculated Recency, Frequency, and Monetary values for each customer.\n\nCustomer Segments:\n\nBest Customers: Customers with the most recent purchases, high frequency, and high monetary value.\n\nLost Customers: Customers with the least recent purchases.\n\nInsights:\n\nCustomer Loyalty:\n\nBest Customers: Focus on retaining these valuable customers through loyalty programs, personalized offers, and excellent customer service.\n\nLost Customers: Implement targeted re-engagement campaigns to win back these customers, such as special discounts or personalized emails.\n\nStrategic Resource Allocation:\n\nAllocate marketing and sales efforts efficiently by focusing on high-value segments and addressing the needs of different customer groups.","metadata":{}},{"cell_type":"markdown","source":"# Visualizing Customer Segments and RFM Score Correlations\n","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas as pd\n\n# Sample Data for Visualization (replace with your RFM DataFrame)\n# Assuming 'rfm' DataFrame contains columns: 'Recency', 'Frequency', 'Monetary'\n# rfm = pd.DataFrame({ ... })\n\n# Create customer segments\ndef segment_customer(row):\n    if row['R_Score'] >= 4 and row['F_Score'] >= 4 and row['M_Score'] >= 4:\n        return 'High Value'\n    elif row['R_Score'] >= 2 and row['F_Score'] >= 2 and row['M_Score'] >= 2:\n        return 'Medium Value'\n    else:\n        return 'Low Value'\n\nrfm['Segment'] = rfm.apply(segment_customer, axis=1)\n\n# Bar Chart of Customer Segments\nrfm['Segment'].value_counts().plot(kind='bar')\nplt.title('Customer Segments Distribution')\nplt.xlabel('Segment')\nplt.ylabel('Number of Customers')\nplt.show()\n\n# Heatmap of R, F, M Scores\nrfm_corr = rfm[['R_Score', 'F_Score', 'M_Score']].astype(int).corr()\nsns.heatmap(rfm_corr, annot=True, cmap='coolwarm')\nplt.title('Correlation of R, F, M Scores')\nplt.show()\n\n# Scatter Plot of Recency vs Frequency\nsns.scatterplot(x='Recency', y='Frequency', hue='Segment', data=rfm)\nplt.title('Recency vs Frequency by Customer Segment')\nplt.xlabel('Recency (Days)')\nplt.ylabel('Frequency (Purchases)')\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T16:19:18.746614Z","iopub.execute_input":"2025-03-04T16:19:18.747122Z","iopub.status.idle":"2025-03-04T16:20:12.421704Z","shell.execute_reply.started":"2025-03-04T16:19:18.747080Z","shell.execute_reply":"2025-03-04T16:20:12.420633Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Findings:\n\nCustomer Segments: Bar chart showing the distribution of customers across different segments (High Value, Medium Value, Low Value).\n\nRFM Score Correlation: Heatmap illustrating the correlation between Recency, Frequency, and Monetary scores.\n\nRecency vs. Frequency: Scatter plot displaying the relationship between Recency and Frequency, segmented by customer value.\n\nInsights:\n\nSegment Distribution:\n\nHigh Value: Focus marketing efforts on retaining these valuable customers.\n\nMedium Value: Engage with these customers to increase their purchasing frequency and value.\n\nLow Value: Develop strategies to reactivate these customers and convert them into higher value segments.\n\nRFM Score Correlation:\n\nUnderstanding the correlation between R, F, and M scores can help identify patterns and prioritize which metrics to improve.\n\nRecency vs. Frequency Analysis:\n\nIdentify trends and patterns in customer behavior, such as frequent purchasers who haven’t bought recently, and create targeted campaigns to re-engage them.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Load the dataset files\narticles = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/articles.csv\")\ncustomers = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/customers.csv\")\ntransactions = pd.read_csv(\"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\")\n\n# Check the column names and inspect the data\nprint(articles.columns)\nprint(customers.columns)\nprint(transactions.columns)\n\n# Ensure 'date' is in datetime format\ntransactions['t_dat'] = pd.to_datetime(transactions['t_dat'])\n\n# Calculate Recency: Number of days since the most recent purchase for each customer\nmost_recent_date = transactions['t_dat'].max()\n\n# Calculate Recency for each customer\ntransactions['recency'] = (most_recent_date - transactions.groupby('customer_id')['t_dat'].transform('max')).dt.days\n\n# Calculate Monetary: Total spending for each customer (using item count as proxy)\nmonetary_per_customer = transactions.groupby('customer_id')['article_id'].count()\n\n# Calculate Frequency: Number of transactions per customer\nfrequency_per_customer = transactions.groupby('customer_id').size()\n\n# Combine Recency, Monetary, and Frequency into a DataFrame (RFM model)\nrfm = pd.DataFrame({\n    'Recency': transactions.groupby('customer_id')['recency'].max(),\n    'Monetary': monetary_per_customer,\n    'Frequency': frequency_per_customer\n})\n\n# 1. Average Visit per Customer (Average Frequency)\naverage_visit = rfm['Frequency'].mean()\n\n# 2. Average Sales per Customer (Average Monetary)\naverage_sales = rfm['Monetary'].mean()\n\n# 3. Average Recency\naverage_recency = rfm['Recency'].mean()\n\n# Print the results\nprint(f\"Average Visit per Customer (Frequency): {average_visit}\")\nprint(f\"Average Sales per Customer (Monetary): {average_sales}\")\nprint(f\"Average Recency (Days since last purchase): {average_recency}\")\n\n# Visualize Recency Distribution\nplt.figure(figsize=(8, 6))\nsns.histplot(rfm['Recency'], bins=50, kde=True, color='skyblue')\nplt.title('Recency Distribution (Days since Last Purchase)')\nplt.xlabel('Days since Last Purchase')\nplt.ylabel('Frequency')\nplt.show()\n\n# Visualize Frequency Distribution (Average Visits per Customer)\nplt.figure(figsize=(8, 6))\nsns.histplot(rfm['Frequency'], bins=50, kde=True, color='lightgreen')\nplt.title('Purchase Frequency Distribution (Number of Purchases per Customer)')\nplt.xlabel('Number of Purchases')\nplt.ylabel('Frequency')\nplt.show()\n\n# Visualize Monetary Distribution (Total Spending per Customer)\nplt.figure(figsize=(8, 6))\nsns.histplot(rfm['Monetary'], bins=50, kde=True, color='lightcoral')\nplt.title('Monetary Distribution (Total Spending per Customer)')\nplt.xlabel('Total Spending (Item Count)')\nplt.ylabel('Frequency')\nplt.show()\n\n# Optional: RFM Segmentation\ndef segment_customer(row):\n    if row['Recency'] <= 30 and row['Frequency'] >= 5 and row['Monetary'] >= 100:\n        return 'High Value'\n    elif row['Recency'] <= 60 and row['Frequency'] >= 3 and row['Monetary'] >= 50:\n        return 'Medium Value'\n    else:\n        return 'Low Value'\n\n# Apply segmentation\nrfm['Segment'] = rfm.apply(segment_customer, axis=1)\n\n# Visualize Customer Segments\nplt.figure(figsize=(8, 6))\nrfm['Segment'].value_counts().plot(kind='bar', color='lightblue')\nplt.title('Customer Segments Distribution')\nplt.xlabel('Segment')\nplt.ylabel('Number of Customers')\nplt.xticks(rotation=0)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-06T05:20:52.943675Z","iopub.execute_input":"2025-03-06T05:20:52.943921Z","iopub.status.idle":"2025-03-06T05:23:45.687770Z","shell.execute_reply.started":"2025-03-06T05:20:52.943897Z","shell.execute_reply":"2025-03-06T05:23:45.686420Z"}},"outputs":[],"execution_count":null}]}