{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\n\n# List all files under the input directory, but ignore image files\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if filename.endswith('.csv'):\n            print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:19:46.690568Z","iopub.execute_input":"2025-02-16T04:19:46.690850Z","iopub.status.idle":"2025-02-16T04:23:44.134267Z","shell.execute_reply.started":"2025-02-16T04:19:46.690825Z","shell.execute_reply":"2025-02-16T04:23:44.133050Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Initial Data Exploration & Preparation\n## Objective: Understand dataset structure and filter relevant transactions\n\nFirst, we identify available data files to understand what we're working with\n\nLoad three key tables:\n\n* articles.csv: Product attributes (colors, categories)\n\n* customers.csv: Demographic/user data\n\n* transactions_train.csv: Historical purchase records\n\nInitial shape checks reveal scale: 31M+ transactions, 1.3M+ users","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom pathlib import Path\n\n# Define the dataset path\nbase_path = Path('/kaggle/input/h-and-m-personalized-fashion-recommendations') \n\n# Load the files\narticles = pd.read_csv(base_path / 'articles.csv')\ncustomers = pd.read_csv(base_path / 'customers.csv')\ntransactions = pd.read_csv(base_path / 'transactions_train.csv')\n\n# Display basic information\nprint(articles.head())\nprint(customers.head())\nprint(transactions.head())\n\nprint(f\"Number of rows in articles.csv: {articles.shape[0]}\") \nprint(f\"Number of rows in customers.csv: {customers.shape[0]}\") \nprint(f\"Number of rows in transactions_train.csv: {transactions.shape[0]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:23:44.135392Z","iopub.execute_input":"2025-02-16T04:23:44.136009Z","iopub.status.idle":"2025-02-16T04:25:17.098316Z","shell.execute_reply.started":"2025-02-16T04:23:44.135966Z","shell.execute_reply":"2025-02-16T04:25:17.097562Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Temporal Filtering Strategy\n## Objective: Focus on recent customer behavior\n\nFashion trends change seasonally → Recent data (6 months) better reflects current preferences\n\nReduces dataset from 31M to 8M transactions while maintaining relevance\n\nMitigates cold-start problem by focusing on active users/items","metadata":{}},{"cell_type":"code","source":"from datetime import datetime, timedelta\n\n# Convert t_dat to datetime\ntransactions['t_dat'] = pd.to_datetime(transactions['t_dat'])\n\n# Filter for the last 6 months (adjust as needed for size)\nmax_date = transactions['t_dat'].max()\nmin_date = max_date - timedelta(days=180)  # 6 months\n\nrecent_transactions = transactions[transactions['t_dat'] >= min_date]\n\nprint(f\"Filtered to recent transactions: {recent_transactions.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:25:17.099145Z","iopub.execute_input":"2025-02-16T04:25:17.099376Z","iopub.status.idle":"2025-02-16T04:25:21.741381Z","shell.execute_reply.started":"2025-02-16T04:25:17.099357Z","shell.execute_reply":"2025-02-16T04:25:21.740175Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Core Market Identification\n## Objective: Identify key users and products for matrix factorization\n\n* Pareto principle: 20% of users/items generate 80% of activity\n\n* Reduces dimensionality (19974 users × 2000 items vs original 1.3M×100K)\n\n* Ensures sufficient interaction density for meaningful recommendations","metadata":{}},{"cell_type":"code","source":"# Step 1: Identify top customers and articles based on activity\ntop_articles = recent_transactions['article_id'].value_counts().nlargest(2000).index\ntop_customers = recent_transactions['customer_id'].value_counts().nlargest(20000).index\n\n# Step 2: Filter transactions that involve only the top customers and articles\nfiltered_transactions = recent_transactions[\n    (recent_transactions['article_id'].isin(top_articles)) & \n    (recent_transactions['customer_id'].isin(top_customers))\n]\n# Step 4: Further refine by ensuring connected components\nconnected_articles = filtered_transactions['article_id'].unique()\nconnected_customers = filtered_transactions['customer_id'].unique()\n\nfinal_filtered_transactions = recent_transactions[\n    (recent_transactions['article_id'].isin(connected_articles)) & \n    (recent_transactions['customer_id'].isin(connected_customers))\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:25:21.743982Z","iopub.execute_input":"2025-02-16T04:25:21.744236Z","iopub.status.idle":"2025-02-16T04:25:26.468847Z","shell.execute_reply.started":"2025-02-16T04:25:21.744215Z","shell.execute_reply":"2025-02-16T04:25:26.467891Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Exploratory Data Analysis (EDA)\n## Objective: Understand interaction patterns\n\nKey Insights:\n\n* Most users have 20-35 transactions (median=26)\n\n* Power users with 200+ transactions exist\n\n* Item popularity follows heavy-tailed distribution\n\n","metadata":{}},{"cell_type":"code","source":"# Step 5: Display final dataset statistics\nretention_rate = len(final_filtered_transactions) / len(recent_transactions)\nprint(f\"Retention Rate: {retention_rate * 100:.2f}%\")\nprint(f\"Final transactions shape: {final_filtered_transactions.shape}\")\n\n# Save the filtered transactions for model training\nfinal_filtered_transactions.to_csv('filtered_transactions.csv', index=False)\n# Check missing values\nprint(articles.isnull().sum())\nprint(customers.isnull().sum())\nprint(transactions.isnull().sum())\n\n# Check for duplicates in transactions\nprint(transactions.duplicated().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:25:26.470829Z","iopub.execute_input":"2025-02-16T04:25:26.471185Z","iopub.status.idle":"2025-02-16T04:25:57.001608Z","shell.execute_reply.started":"2025-02-16T04:25:26.471149Z","shell.execute_reply":"2025-02-16T04:25:57.000271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Number of transactions per customer\ncustomer_counts = final_filtered_transactions['customer_id'].value_counts()\nprint(customer_counts.describe())\n\n# Plot the distribution\nplt.figure(figsize=(10, 6))\ncustomer_counts.hist(bins=50)\nplt.title('Distribution of Transactions per Customer')\nplt.xlabel('Number of Transactions')\nplt.ylabel('Number of Customers')\nplt.show()\n\n# Number of purchases per article\narticle_counts = final_filtered_transactions['article_id'].value_counts()\nprint(article_counts.describe())\n\n# Plot the distribution\nplt.figure(figsize=(10, 6))\narticle_counts.hist(bins=50)\nplt.title('Distribution of Purchases per Article')\nplt.xlabel('Number of Purchases')\nplt.ylabel('Number of Articles')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:25:57.003330Z","iopub.execute_input":"2025-02-16T04:25:57.003743Z","iopub.status.idle":"2025-02-16T04:25:57.673642Z","shell.execute_reply.started":"2025-02-16T04:25:57.003705Z","shell.execute_reply":"2025-02-16T04:25:57.672626Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. User-Item Matrix Construction\n## Objective: Create interaction matrix for collaborative filtering\n\nKey Insights:\n\n* Matrix sparsity = 98.94% → Typical for recommender systems\n\n* Encodes purchase frequency as implicit feedback\n\n* Enables neighborhood-based and matrix factorization approaches","metadata":{}},{"cell_type":"code","source":"import numpy as np\n\n# Create user-item matrix\nuser_item_matrix = final_filtered_transactions.pivot_table(index='customer_id', columns='article_id', aggfunc='size', fill_value=0)\n\n# Calculate sparsity\nnon_zero_entries = np.count_nonzero(user_item_matrix)\ntotal_entries = user_item_matrix.size\nsparsity = 1 - (non_zero_entries / total_entries)\n\nprint(f\"Sparsity of the dataset: {sparsity:.4f}\")\nprint(f\"Unique customers: {final_filtered_transactions['customer_id'].nunique()}\")\nprint(f\"Unique articles: {final_filtered_transactions['article_id'].nunique()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:25:57.674813Z","iopub.execute_input":"2025-02-16T04:25:57.675152Z","iopub.status.idle":"2025-02-16T04:25:59.294332Z","shell.execute_reply.started":"2025-02-16T04:25:57.675121Z","shell.execute_reply":"2025-02-16T04:25:59.293344Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. Temporal Validation Split\n## Objective: Realistic evaluation via time-based holdout\n\nKey Insights:\n\n* Simulates real-world scenario where model predicts future purchases\n\n* Avoids data leakage from random splitting\n\n* Test period (1 week) matches H&M's fast fashion cycle","metadata":{}},{"cell_type":"code","source":"from datetime import timedelta\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error\nimport torch\nfrom torch.utils.data import DataLoader, Dataset\nimport torch.nn as nn\nimport torch.optim as optim\n\n# Load data (replace with actual paths)\nfinal_filtered_transactions = pd.read_csv(\"/kaggle/working/filtered_transactions.csv\")\n\n# Convert t_dat to datetime\nfinal_filtered_transactions['t_dat'] = pd.to_datetime(final_filtered_transactions['t_dat'])\n\n# Time-based split: Last 7 days as the test set\nsplit_date = final_filtered_transactions['t_dat'].max() - timedelta(days=7)\ntrain_data = final_filtered_transactions[final_filtered_transactions['t_dat'] <= split_date]\ntest_data = final_filtered_transactions[final_filtered_transactions['t_dat'] > split_date]\n\n# Random split for validation\ntrain_data, val_data = train_test_split(train_data, test_size=0.1, random_state=42)\n\nprint(f\"Training data shape: {train_data.shape}\")\nprint(f\"Validation data shape: {val_data.shape}\")\nprint(f\"Test data shape: {test_data.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:25:59.295282Z","iopub.execute_input":"2025-02-16T04:25:59.295598Z","iopub.status.idle":"2025-02-16T04:26:05.104157Z","shell.execute_reply.started":"2025-02-16T04:25:59.295556Z","shell.execute_reply":"2025-02-16T04:26:05.103194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Collaborative Filtering Dataset class\nclass CollabDataset(Dataset):\n    def __init__(self, data):\n        self.customers = data['customer_id'].astype('category').cat.codes.values\n        self.articles = data['article_id'].astype('category').cat.codes.values\n        self.targets = data['price'].values  # Replace 'price' with the actual target column if different\n\n    def __len__(self):\n        return len(self.targets)\n\n    def __getitem__(self, idx):\n        return (\n            torch.tensor(self.customers[idx], dtype=torch.long),\n            torch.tensor(self.articles[idx], dtype=torch.long),\n            torch.tensor(self.targets[idx], dtype=torch.float),\n        )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:26:05.105139Z","iopub.execute_input":"2025-02-16T04:26:05.105396Z","iopub.status.idle":"2025-02-16T04:26:05.112276Z","shell.execute_reply.started":"2025-02-16T04:26:05.105373Z","shell.execute_reply":"2025-02-16T04:26:05.111207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Data Preparation\ntrain_dataset = CollabDataset(train_data)\nval_dataset = CollabDataset(val_data)\ntest_dataset = CollabDataset(test_data)\n\nbatch_size = 512\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False)\ntest_loader = DataLoader(test_dataset, batch_size=batch_size, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:26:05.113308Z","iopub.execute_input":"2025-02-16T04:26:05.113688Z","iopub.status.idle":"2025-02-16T04:26:05.369888Z","shell.execute_reply.started":"2025-02-16T04:26:05.113610Z","shell.execute_reply":"2025-02-16T04:26:05.368751Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7. Neural Collaborative Filtering Architecture\n## Objective: Learn user/item embeddings from interactions\n\nKey Insights:\n\n* 50D embeddings capture latent product/user features\n\n* Element-wise product models user-item affinity\n\n* Dropout (30%) prevents overfitting on sparse data\n\n","metadata":{}},{"cell_type":"code","source":"# Collaborative Filtering Model\nclass CollabModel(nn.Module):\n    def __init__(self, num_customers, num_articles, embedding_size):\n        super(CollabModel, self).__init__()\n        self.customer_embedding = nn.Embedding(num_customers, embedding_size)\n        self.article_embedding = nn.Embedding(num_articles, embedding_size)\n        self.dropout = nn.Dropout(p=0.3)\n        self.fc = nn.Linear(embedding_size, 1)\n\n    def forward(self, customers, articles):\n        customer_emb = self.dropout(self.customer_embedding(customers))\n        article_emb = self.dropout(self.article_embedding(articles))\n        interaction = customer_emb * article_emb\n        return self.fc(interaction).squeeze()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:26:05.371021Z","iopub.execute_input":"2025-02-16T04:26:05.371387Z","iopub.status.idle":"2025-02-16T04:26:05.377957Z","shell.execute_reply.started":"2025-02-16T04:26:05.371350Z","shell.execute_reply":"2025-02-16T04:26:05.376892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize Model\nnum_customers = len(train_data['customer_id'].astype('category').cat.categories)\nnum_articles = len(train_data['article_id'].astype('category').cat.categories)\nembedding_size = 50\nmodel = CollabModel(num_customers, num_articles, embedding_size)\n\n# Training Setup\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nmodel = model.to(device)\ncriterion = nn.MSELoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=0.01, weight_decay=1e-9)\n\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\nscheduler = ReduceLROnPlateau(optimizer, mode='min', patience=3, factor=0.7)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:26:05.379170Z","iopub.execute_input":"2025-02-16T04:26:05.379553Z","iopub.status.idle":"2025-02-16T04:26:06.965914Z","shell.execute_reply.started":"2025-02-16T04:26:05.379513Z","shell.execute_reply":"2025-02-16T04:26:06.965107Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 8. Model Training Dynamics\n## Objective: Optimize embedding weights\n\nKey Insights:\n\n* Adam optimizer with LR=0.01 balances speed/stability\n\n* Learning rate scheduling reduces LR on plateau\n\n* Early stopping prevents overfitting (not shown)","metadata":{}},{"cell_type":"code","source":"# Training Loop\nepochs = 10\nfor epoch in range(epochs):\n    model.train()\n    total_loss = 0\n    for customers, articles, targets in train_loader:\n        customers, articles, targets = customers.to(device), articles.to(device), targets.to(device)\n        predictions = model(customers, articles)\n        loss = criterion(predictions, targets)\n\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        total_loss += loss.item()\n\n    model.eval()\n    val_loss = 0\n    with torch.no_grad():\n        for customers, articles, targets in val_loader:\n            customers, articles, targets = customers.to(device), articles.to(device), targets.to(device)\n            predictions = model(customers, articles)\n            val_loss += criterion(predictions, targets).item() / len(targets)\n    scheduler.step(val_loss)\n\n    print(f\"Epoch {epoch+1}/{epochs}, Training Loss: {total_loss/len(train_loader):.4f}, Validation Loss: {val_loss:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:26:06.969128Z","iopub.execute_input":"2025-02-16T04:26:06.969537Z","iopub.status.idle":"2025-02-16T04:29:12.049249Z","shell.execute_reply.started":"2025-02-16T04:26:06.969505Z","shell.execute_reply":"2025-02-16T04:29:12.048105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Learning Rate Finder\ndef lr_finder(model, optimizer, criterion, dataloader, start_lr=1e-7, end_lr=1, num_iter=50):\n    model.train()\n    lrs = []\n    losses = []\n    lr = start_lr\n    optimizer.param_groups[0]['lr'] = lr\n    gamma = (end_lr / start_lr) ** (1 / num_iter)  # Multiplicative factor for learning rate\n\n    for i, (customers, articles, targets) in enumerate(dataloader):\n        if i >= num_iter:\n            break\n        customers, articles, targets = customers.to(device), articles.to(device), targets.to(device)\n        optimizer.zero_grad()\n        outputs = model(customers, articles)\n        loss = criterion(outputs, targets)\n        loss.backward()\n        optimizer.step()\n\n        lrs.append(lr)\n        losses.append(loss.item())\n        lr *= gamma\n        optimizer.param_groups[0]['lr'] = lr\n\n    return lrs, losses\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:29:12.052074Z","iopub.execute_input":"2025-02-16T04:29:12.052716Z","iopub.status.idle":"2025-02-16T04:29:12.060019Z","shell.execute_reply.started":"2025-02-16T04:29:12.052674Z","shell.execute_reply":"2025-02-16T04:29:12.058669Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 9. Model Evaluation Metrics\n## Objective: Quantify prediction accuracy\n\nKey Insights:\n\n* RMSE 0.0082 → Average error ~€0.008 per prediction\n\n* R²=0.56 → Model explains 56% of price variance\n\n* Residual analysis shows 82.55% predictions within €0.01 error","metadata":{}},{"cell_type":"code","source":"# Evaluation\nmodel.eval()\nval_loss = 0\nall_predictions = []\nall_targets = []\nwith torch.no_grad():\n    for customers, articles, targets in val_loader:\n        customers, articles, targets = customers.to(device), articles.to(device), targets.to(device)\n        predictions = model(customers, articles).cpu().numpy()\n        all_predictions.extend(predictions)\n        all_targets.extend(targets.cpu().numpy())\n        batch_loss = criterion(torch.tensor(predictions), targets.cpu()).item() / len(targets)\n        val_loss += batch_loss\n\nprint(f\"Validation Loss: {val_loss:.4f}\")\nrmse = mean_squared_error(all_targets, all_predictions, squared=False)\nmae = mean_absolute_error(all_targets, all_predictions)\nprint(f\"RMSE: {rmse:.4f}, MAE: {mae:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:29:12.061388Z","iopub.execute_input":"2025-02-16T04:29:12.061752Z","iopub.status.idle":"2025-02-16T04:29:13.403190Z","shell.execute_reply.started":"2025-02-16T04:29:12.061719Z","shell.execute_reply":"2025-02-16T04:29:13.402013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Average test loss\nval_loss /= len(val_loader)\n\n# Calculate RMSE and MAE\nrmse = mean_squared_error(all_targets, all_predictions, squared=False)\nmae = mean_absolute_error(all_targets, all_predictions)\n\nprint(f\"Test Loss: {val_loss:.4f}\")\nprint(f\"RMSE: {rmse:.4f}\")\nprint(f\"MAE: {mae:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:29:13.404395Z","iopub.execute_input":"2025-02-16T04:29:13.404862Z","iopub.status.idle":"2025-02-16T04:29:13.428149Z","shell.execute_reply.started":"2025-02-16T04:29:13.404821Z","shell.execute_reply":"2025-02-16T04:29:13.427092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Scatter plot: Predictions vs. True Values\nplt.figure(figsize=(8, 6))\nsns.scatterplot(x=all_targets, y=all_predictions, alpha=0.7)\nplt.plot([min(all_targets), max(all_targets)], [min(all_targets), max(all_targets)], color='red', linestyle='--')\nplt.title(\"Predictions vs True Values\")\nplt.xlabel(\"True Values\")\nplt.ylabel(\"Predicted Values\")\nplt.grid()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:29:13.429367Z","iopub.execute_input":"2025-02-16T04:29:13.429797Z","iopub.status.idle":"2025-02-16T04:29:14.289843Z","shell.execute_reply.started":"2025-02-16T04:29:13.429724Z","shell.execute_reply":"2025-02-16T04:29:14.288732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Residuals calculation\nresiduals = np.array(all_targets) - np.array(all_predictions)\n\n# Plot residuals\nplt.figure(figsize=(8, 6))\nsns.histplot(residuals, kde=True, bins=50, color=\"blue\")\nplt.title(\"Distribution of Residuals\")\nplt.xlabel(\"Residuals (True - Predicted)\")\nplt.ylabel(\"Frequency\")\nplt.grid()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:29:14.291039Z","iopub.execute_input":"2025-02-16T04:29:14.291716Z","iopub.status.idle":"2025-02-16T04:29:14.930334Z","shell.execute_reply.started":"2025-02-16T04:29:14.291663Z","shell.execute_reply":"2025-02-16T04:29:14.929278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# MAPE calculation\nmape = np.mean(np.abs((np.array(all_targets) - np.array(all_predictions)) / np.array(all_targets))) * 100\nprint(f\"Mean Absolute Percentage Error (MAPE): {mape:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:29:14.931372Z","iopub.execute_input":"2025-02-16T04:29:14.931737Z","iopub.status.idle":"2025-02-16T04:29:14.946779Z","shell.execute_reply.started":"2025-02-16T04:29:14.931699Z","shell.execute_reply":"2025-02-16T04:29:14.945638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import r2_score\n\n# R² calculation\nr2 = r2_score(all_targets, all_predictions)\nprint(f\"R² (Coefficient of Determination): {r2:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:29:14.947571Z","iopub.execute_input":"2025-02-16T04:29:14.947899Z","iopub.status.idle":"2025-02-16T04:29:14.969296Z","shell.execute_reply.started":"2025-02-16T04:29:14.947862Z","shell.execute_reply":"2025-02-16T04:29:14.968314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cumulative distribution function of absolute errors\nabsolute_errors = np.abs(np.array(all_targets) - np.array(all_predictions))\n\n# Plot CDF\nplt.figure(figsize=(8, 6))\nsns.ecdfplot(absolute_errors, color=\"green\")\nplt.title(\"CDF of Absolute Errors\")\nplt.xlabel(\"Absolute Error\")\nplt.ylabel(\"Cumulative Probability\")\nplt.grid()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:29:14.970376Z","iopub.execute_input":"2025-02-16T04:29:14.970717Z","iopub.status.idle":"2025-02-16T04:29:15.228097Z","shell.execute_reply.started":"2025-02-16T04:29:14.970684Z","shell.execute_reply":"2025-02-16T04:29:15.227010Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Normalize RMSE and MAE relative to the range of the target variable\n\n# Convert the list to a NumPy array\nall_targets_array = np.array(all_targets)\nall_predictions_array =  np.array(all_predictions)\n# Now you can calculate the target range\ntarget_range = all_targets_array.max() - all_targets_array.min()\nnormalized_rmse = rmse / target_range\nnormalized_mae = mae / target_range\n\n\nprint(f\"Normalized RMSE: {normalized_rmse:.4f} (Relative to Target Range)\")\nprint(f\"Normalized MAE: {normalized_mae:.4f} (Relative to Target Range)\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:29:15.229222Z","iopub.execute_input":"2025-02-16T04:29:15.229499Z","iopub.status.idle":"2025-02-16T04:29:15.241493Z","shell.execute_reply.started":"2025-02-16T04:29:15.229466Z","shell.execute_reply":"2025-02-16T04:29:15.240566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compute test metrics\ntest_mse = mean_squared_error(all_targets, all_predictions)\ntest_mae = mean_absolute_error(all_targets, all_predictions)\ntest_rmse = np.sqrt(test_mse)\ntest_r2 = r2_score(all_targets, all_predictions)\n\nprint(f\"Test Metrics:\")\nprint(f\"MAE: {test_mae:.4f}, RMSE: {test_rmse:.4f}, R²: {test_r2:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:29:15.242538Z","iopub.execute_input":"2025-02-16T04:29:15.242919Z","iopub.status.idle":"2025-02-16T04:29:15.278545Z","shell.execute_reply.started":"2025-02-16T04:29:15.242894Z","shell.execute_reply":"2025-02-16T04:29:15.277601Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate residuals\nresiduals = all_targets_array - all_predictions_array\nabsolute_residuals = np.abs(residuals)\n\n# Threshold for outliers (e.g., top 5% of errors)\nthreshold = np.percentile(absolute_residuals, 95)\noutliers = np.where(absolute_residuals > threshold)[0]\n\n# Print details about outliers\nprint(f\"Number of outliers: {len(outliers)}\")\nprint(f\"Threshold for outliers: {threshold:.4f}\")\nprint(f\"Outlier True Values: {all_targets_array[outliers]}\")\nprint(f\"Outlier Predictions: {all_predictions_array[outliers]}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:29:15.279351Z","iopub.execute_input":"2025-02-16T04:29:15.279570Z","iopub.status.idle":"2025-02-16T04:29:15.294922Z","shell.execute_reply.started":"2025-02-16T04:29:15.279550Z","shell.execute_reply":"2025-02-16T04:29:15.294012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nplt.scatter(all_targets, residuals, alpha=0.5)\nplt.axhline(0, color='red', linestyle='--', label='Zero Error')\nplt.xlabel(\"True Values\")\nplt.ylabel(\"Residuals\")\nplt.title(\"Residuals vs True Values\")\nplt.legend()\nplt.grid()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:29:15.295933Z","iopub.execute_input":"2025-02-16T04:29:15.296172Z","iopub.status.idle":"2025-02-16T04:29:16.226487Z","shell.execute_reply.started":"2025-02-16T04:29:15.296150Z","shell.execute_reply":"2025-02-16T04:29:16.225546Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Example threshold (e.g., acceptable MAE is 0.01)\nthreshold = 0.01\nwithin_threshold = np.sum(absolute_residuals <= threshold) / len(absolute_residuals) * 100\nprint(f\"Percentage of predictions within the threshold ({threshold}): {within_threshold:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:29:16.227659Z","iopub.execute_input":"2025-02-16T04:29:16.228058Z","iopub.status.idle":"2025-02-16T04:29:16.233714Z","shell.execute_reply.started":"2025-02-16T04:29:16.228030Z","shell.execute_reply":"2025-02-16T04:29:16.232742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sort errors by magnitude\nsorted_indices = np.argsort(absolute_residuals)[::-1]\ntop_errors = sorted_indices[:10]\n\n# Print top-error cases\nfor idx in top_errors:\n    print(f\"True Value: {all_targets[idx]:.4f}, Predicted: {all_predictions[idx]:.4f}, Error: {absolute_residuals[idx]:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:29:16.234590Z","iopub.execute_input":"2025-02-16T04:29:16.234868Z","iopub.status.idle":"2025-02-16T04:29:16.253604Z","shell.execute_reply.started":"2025-02-16T04:29:16.234837Z","shell.execute_reply":"2025-02-16T04:29:16.252915Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 10. Recommendation Generation\n## Objective: Serve personalized product suggestions\n\nKey Insights:\n\n* Scores all items for target user\n\n* Selects top-N highest predicted affinity items\n\n* Can be augmented with business rules (stock availability, margins)","metadata":{}},{"cell_type":"code","source":"\n# Recommendations\ndef recommend_products(model, user_id, num_articles, top_n=5):\n    model.eval()\n    user_id_tensor = torch.tensor([user_id] * num_articles, device=device)\n    article_ids_tensor = torch.arange(num_articles, device=device)\n\n    with torch.no_grad():\n        scores = model(user_id_tensor, article_ids_tensor).cpu().numpy()\n\n    top_articles = np.argsort(scores)[-top_n:][::-1]\n    return top_articles.tolist()\n\n# Example Recommendations\nuser_id = 0\nrecommended_articles = recommend_products(model, user_id, num_articles, top_n=5)\nprint(f\"Recommended articles for user {user_id}: {recommended_articles}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-16T04:29:16.254453Z","iopub.execute_input":"2025-02-16T04:29:16.254812Z","iopub.status.idle":"2025-02-16T04:29:16.264110Z","shell.execute_reply.started":"2025-02-16T04:29:16.254779Z","shell.execute_reply":"2025-02-16T04:29:16.263330Z"}},"outputs":[],"execution_count":null}]}