{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"dockerImageVersionId":31259,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This is all you need - fast and direct!\nimport numpy as np\nimport pandas as pd\n\n# Load your H&M datasets\narticles = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/articles.csv')\ncustomers = pd.read_csv('/kaggle/input/h-and-m-personalized-fashion-recommendations/customers.csv')\n\n# For transactions, start with a sample (it's huge - 31M rows!)\ntransactions_sample = pd.read_csv(\n    '/kaggle/input/h-and-m-personalized-fashion-recommendations/transactions_train.csv',\n    nrows=500000  # Just 100k rows for initial testing\n)\n\nprint(\"✅ Data loaded successfully!\")\nprint(f\"📦 Articles: {articles.shape} rows\")\nprint(f\"👥 Customers: {customers.shape} rows\")\nprint(f\"🛒 Transactions sample: {transactions_sample.shape} rows\")\nprint(\"\\nFirst few articles:\")\nprint(articles.head(3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:09.7781Z","iopub.execute_input":"2026-01-28T19:36:09.779315Z","iopub.status.idle":"2026-01-28T19:36:15.636281Z","shell.execute_reply.started":"2026-01-28T19:36:09.779276Z","shell.execute_reply":"2026-01-28T19:36:15.635253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles.shape,customers.shape,transactions_sample.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:15.638263Z","iopub.execute_input":"2026-01-28T19:36:15.638731Z","iopub.status.idle":"2026-01-28T19:36:15.646607Z","shell.execute_reply.started":"2026-01-28T19:36:15.638698Z","shell.execute_reply":"2026-01-28T19:36:15.645366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transactions_sample.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:15.6481Z","iopub.execute_input":"2026-01-28T19:36:15.648851Z","iopub.status.idle":"2026-01-28T19:36:15.731459Z","shell.execute_reply.started":"2026-01-28T19:36:15.648819Z","shell.execute_reply":"2026-01-28T19:36:15.73041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"customers.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:15.733768Z","iopub.execute_input":"2026-01-28T19:36:15.734142Z","iopub.status.idle":"2026-01-28T19:36:16.08638Z","shell.execute_reply.started":"2026-01-28T19:36:15.734101Z","shell.execute_reply":"2026-01-28T19:36:16.085392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles.drop(columns=[\"detail_desc\"],inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.087571Z","iopub.execute_input":"2026-01-28T19:36:16.088037Z","iopub.status.idle":"2026-01-28T19:36:16.123039Z","shell.execute_reply.started":"2026-01-28T19:36:16.088008Z","shell.execute_reply":"2026-01-28T19:36:16.121804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.124439Z","iopub.execute_input":"2026-01-28T19:36:16.124877Z","iopub.status.idle":"2026-01-28T19:36:16.211197Z","shell.execute_reply.started":"2026-01-28T19:36:16.124847Z","shell.execute_reply":"2026-01-28T19:36:16.210138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transactions_sample.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.213035Z","iopub.execute_input":"2026-01-28T19:36:16.213473Z","iopub.status.idle":"2026-01-28T19:36:16.232426Z","shell.execute_reply.started":"2026-01-28T19:36:16.213429Z","shell.execute_reply":"2026-01-28T19:36:16.231382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.233976Z","iopub.execute_input":"2026-01-28T19:36:16.234392Z","iopub.status.idle":"2026-01-28T19:36:16.26783Z","shell.execute_reply.started":"2026-01-28T19:36:16.23435Z","shell.execute_reply":"2026-01-28T19:36:16.266501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.269225Z","iopub.execute_input":"2026-01-28T19:36:16.269665Z","iopub.status.idle":"2026-01-28T19:36:16.300753Z","shell.execute_reply.started":"2026-01-28T19:36:16.269601Z","shell.execute_reply":"2026-01-28T19:36:16.299716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with pd.option_context(\"display.max_rows\", None):\n    print(articles[\"department_name\"].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.305137Z","iopub.execute_input":"2026-01-28T19:36:16.306076Z","iopub.status.idle":"2026-01-28T19:36:16.330896Z","shell.execute_reply.started":"2026-01-28T19:36:16.306032Z","shell.execute_reply":"2026-01-28T19:36:16.32944Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\narticles[\"department_name\"].value_counts()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.33241Z","iopub.execute_input":"2026-01-28T19:36:16.332872Z","iopub.status.idle":"2026-01-28T19:36:16.364101Z","shell.execute_reply.started":"2026-01-28T19:36:16.33284Z","shell.execute_reply":"2026-01-28T19:36:16.363063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles[\"colour_group_name\"].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.365411Z","iopub.execute_input":"2026-01-28T19:36:16.36658Z","iopub.status.idle":"2026-01-28T19:36:16.395304Z","shell.execute_reply.started":"2026-01-28T19:36:16.366548Z","shell.execute_reply":"2026-01-28T19:36:16.394148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles[\"product_group_name\"].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.396715Z","iopub.execute_input":"2026-01-28T19:36:16.397125Z","iopub.status.idle":"2026-01-28T19:36:16.421871Z","shell.execute_reply.started":"2026-01-28T19:36:16.397092Z","shell.execute_reply":"2026-01-28T19:36:16.420449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles[\"product_type_name\"].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.423503Z","iopub.execute_input":"2026-01-28T19:36:16.423914Z","iopub.status.idle":"2026-01-28T19:36:16.445313Z","shell.execute_reply.started":"2026-01-28T19:36:16.423883Z","shell.execute_reply":"2026-01-28T19:36:16.444022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles[\"index_name\"].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.446853Z","iopub.execute_input":"2026-01-28T19:36:16.447152Z","iopub.status.idle":"2026-01-28T19:36:16.469703Z","shell.execute_reply.started":"2026-01-28T19:36:16.447126Z","shell.execute_reply":"2026-01-28T19:36:16.468768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles[articles[\"product_type_no\"]==-1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T20:14:25.462182Z","iopub.execute_input":"2026-01-28T20:14:25.463364Z","iopub.status.idle":"2026-01-28T20:14:25.501704Z","shell.execute_reply.started":"2026-01-28T20:14:25.46331Z","shell.execute_reply":"2026-01-28T20:14:25.500623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles.describe().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.471317Z","iopub.execute_input":"2026-01-28T19:36:16.471767Z","iopub.status.idle":"2026-01-28T19:36:16.553052Z","shell.execute_reply.started":"2026-01-28T19:36:16.471724Z","shell.execute_reply":"2026-01-28T19:36:16.551663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.55442Z","iopub.execute_input":"2026-01-28T19:36:16.554861Z","iopub.status.idle":"2026-01-28T19:36:16.643822Z","shell.execute_reply.started":"2026-01-28T19:36:16.554816Z","shell.execute_reply":"2026-01-28T19:36:16.642748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def detect_unknown_patterns(df, column):\n    \"\"\"Find all possible unknown/placeholder values in a column\"\"\"\n    \n    print(f\"\\n=== ANALYZING: {column} ===\")\n    print(f\"Data type: {df[column].dtype}\")\n    \n    # For NUMERIC columns\n    if pd.api.types.is_numeric_dtype(df[column]):\n        # Get value counts of \"suspicious\" low/negative/high values\n        value_counts = df[column].value_counts().sort_index()\n        \n        # Check for common placeholder patterns\n        suspicious_values = []\n        \n        # Negative values (often -1, -999)\n        neg_values = value_counts[value_counts.index < 0]\n        if len(neg_values) > 0:\n            print(f\"Negative values found: {list(neg_values.index[:5])}\")\n            suspicious_values.extend(list(neg_values.index))\n        \n        # Zero values (sometimes 0 means unknown)\n        if 0 in value_counts.index:\n            print(f\"Zero count: {value_counts[0]}\")\n            suspicious_values.append(0)\n            \n        # Very high values (999, 9999 often placeholders)\n        max_val = value_counts.index.max()\n        if max_val > 500 and value_counts[max_val] < 100:  # Very high but rare\n            print(f\"Max value {max_val} appears {value_counts[max_val]} times (possible placeholder)\")\n            suspicious_values.append(max_val)\n    \n    # For STRING columns  \n    else:\n        # Look for \"unknown\", \"not specified\", etc.\n        unknown_patterns = ['unknown', 'not', 'unspecified', 'none', 'other', 'na', 'n/a']\n        value_counts = df[column].value_counts().head(20)  # Top 20 values\n        \n        for val in value_counts.index:\n            val_str = str(val).lower()\n            for pattern in unknown_patterns:\n                if pattern in val_str:\n                    print(f\"Possible unknown: '{val}' appears {value_counts[val]} times\")\n                    break\n    \n    # Show distribution summary\n    print(f\"Unique values: {df[column].nunique()}\")\n    print(f\"Min: {df[column].min()}, Max: {df[column].max()}\")\n    \n    return suspicious_values\n\n# Analyze key columns\ncolumns_to_investigate = [\n    'product_type_no', 'colour_group_code', 'graphical_appearance_no',\n    'perceived_colour_value_id', 'perceived_colour_master_id',\n    'department_no', 'garment_group_no', 'index_group_no', 'section_no'\n]\n\nunknown_mappings = {}\nfor col in columns_to_investigate:\n    unknown_vals = detect_unknown_patterns(articles, col)\n    if unknown_vals:\n        unknown_mappings[col] = unknown_vals","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.64513Z","iopub.execute_input":"2026-01-28T19:36:16.645699Z","iopub.status.idle":"2026-01-28T19:36:16.690506Z","shell.execute_reply.started":"2026-01-28T19:36:16.645621Z","shell.execute_reply":"2026-01-28T19:36:16.689522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles['department_no'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.691626Z","iopub.execute_input":"2026-01-28T19:36:16.692052Z","iopub.status.idle":"2026-01-28T19:36:16.703814Z","shell.execute_reply.started":"2026-01-28T19:36:16.692021Z","shell.execute_reply":"2026-01-28T19:36:16.702753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles[articles['product_type_no'] == -1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.705058Z","iopub.execute_input":"2026-01-28T19:36:16.705703Z","iopub.status.idle":"2026-01-28T19:36:16.745475Z","shell.execute_reply.started":"2026-01-28T19:36:16.705625Z","shell.execute_reply":"2026-01-28T19:36:16.744354Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with pd.option_context(\"display.max_rows\", None):\n    print(articles[\"department_no\"].value_counts().sort_values())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.747247Z","iopub.execute_input":"2026-01-28T19:36:16.747617Z","iopub.status.idle":"2026-01-28T19:36:16.758546Z","shell.execute_reply.started":"2026-01-28T19:36:16.747586Z","shell.execute_reply":"2026-01-28T19:36:16.757252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check which columns have -1 values\nnegative_value_cols = []\nfor col in ['product_type_no', 'colour_group_code', 'graphical_appearance_no']:\n    neg_count = (articles[col] == -1).sum()\n    if neg_count > 0:\n        print(f\"{col}: {neg_count} negative values ({neg_count/len(articles)*100:.2f}%)\")\n        negative_value_cols.append(col)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.760799Z","iopub.execute_input":"2026-01-28T19:36:16.761198Z","iopub.status.idle":"2026-01-28T19:36:16.788951Z","shell.execute_reply.started":"2026-01-28T19:36:16.761158Z","shell.execute_reply":"2026-01-28T19:36:16.787591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# A. PRIMARY FEATURES (will use these for clustering)\nprimary_features = {\n    'product_type': 'product_type_no',  # Replace -1 with imputed values\n    'department': 'department_no',       # Good, no -1 values\n    'color_group': 'colour_group_code',  # Replace -1 with 0\n    'garment_group': 'garment_group_no', # Good, no -1 values\n    'index_group': 'index_group_no'      # Good, no -1 values\n}\n\n# B. SECONDARY FEATURES (for post-processing/filtering)\nsecondary_features = {\n    'section': 'section_no',\n    'graphical_appearance': 'graphical_appearance_no'  # Replace -1\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.79085Z","iopub.execute_input":"2026-01-28T19:36:16.791319Z","iopub.status.idle":"2026-01-28T19:36:16.808174Z","shell.execute_reply.started":"2026-01-28T19:36:16.791271Z","shell.execute_reply":"2026-01-28T19:36:16.806723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Create clean dataframe\nclean_articles = articles.copy()\n\n# 2. Fix -1 values (simplest method)\nclean_articles['product_type_no'] = clean_articles['product_type_no'].replace(-1, 9999)\nclean_articles['colour_group_code'] = clean_articles['colour_group_code'].replace(-1, 0)\nclean_articles['graphical_appearance_no'] = clean_articles['graphical_appearance_no'].replace(-1, 0)\n\n# 3. Select features for clustering\nclustering_features = clean_articles[[\n    'product_type_no', \n    'department_no', \n    'colour_group_code',\n    'garment_group_no',\n    'index_group_no'\n]]\n\nprint(f\"Features shape: {clustering_features.shape}\")\nprint(\"No missing values ✓\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.810046Z","iopub.execute_input":"2026-01-28T19:36:16.810514Z","iopub.status.idle":"2026-01-28T19:36:16.865808Z","shell.execute_reply.started":"2026-01-28T19:36:16.810469Z","shell.execute_reply":"2026-01-28T19:36:16.864471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\n# Scale the features\nscaler = StandardScaler()\nscaled_features = scaler.fit_transform(clustering_features)\n\nprint(\"Feature means after scaling:\", scaled_features.mean(axis=0))\nprint(\"Feature stds after scaling:\", scaled_features.std(axis=0))\n# Should be ~0 mean and ~1 std for all features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.867211Z","iopub.execute_input":"2026-01-28T19:36:16.867541Z","iopub.status.idle":"2026-01-28T19:36:16.899362Z","shell.execute_reply.started":"2026-01-28T19:36:16.867514Z","shell.execute_reply":"2026-01-28T19:36:16.898129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create final quality report\ndef data_quality_report(df, feature_cols):\n    report = {}\n    for col in feature_cols:\n        report[col] = {\n            'min': df[col].min(),\n            'max': df[col].max(),\n            'unique': df[col].nunique(),\n            'has_negative': (df[col] < 0).any(),\n            'missing': df[col].isnull().sum()\n        }\n    return pd.DataFrame(report).T\n\n# Generate report\nfeature_cols = ['product_type_no', 'department_no', 'colour_group_code', \n                'garment_group_no', 'index_group_no']\nquality_report = data_quality_report(clean_articles, feature_cols)\nprint(quality_report)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.901302Z","iopub.execute_input":"2026-01-28T19:36:16.902535Z","iopub.status.idle":"2026-01-28T19:36:16.929755Z","shell.execute_reply.started":"2026-01-28T19:36:16.902498Z","shell.execute_reply":"2026-01-28T19:36:16.928092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\n\n# Set style\nplt.style.use('seaborn-v0_8-darkgrid')\nfig, axes = plt.subplots(3, 2, figsize=(15, 12))\nfig.suptitle('Distribution of Key Features', fontsize=16, y=1.02)\n\nfeatures_to_plot = [\n    ('product_type_no', 'Product Type'),\n    ('department_no', 'Department'),\n    ('colour_group_code', 'Color Group'),\n    ('garment_group_no', 'Garment Group'),\n    ('index_group_no', 'Index Group')\n]\n\nfor idx, (col, title) in enumerate(features_to_plot):\n    ax = axes[idx//2, idx%2]\n    \n    # Plot histogram\n    ax.hist(articles[col], bins=50, alpha=0.7, color='skyblue', edgecolor='black')\n    \n    # Add vertical lines for stats\n    mean_val = articles[col].mean()\n    median_val = articles[col].median()\n    ax.axvline(mean_val, color='red', linestyle='--', linewidth=2, label=f'Mean: {mean_val:.1f}')\n    ax.axvline(median_val, color='green', linestyle='-', linewidth=2, label=f'Median: {median_val:.1f}')\n    \n    ax.set_title(f'{title} Distribution')\n    ax.set_xlabel('Value')\n    ax.set_ylabel('Frequency')\n    ax.legend()\n    ax.grid(True, alpha=0.3)\n\n# Hide empty subplot\naxes[2, 1].axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:16.931271Z","iopub.execute_input":"2026-01-28T19:36:16.931747Z","iopub.status.idle":"2026-01-28T19:36:18.259794Z","shell.execute_reply.started":"2026-01-28T19:36:16.931712Z","shell.execute_reply":"2026-01-28T19:36:18.258485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 3, figsize=(16, 10))\nfig.suptitle('Box Plots for Outlier Detection', fontsize=16, y=1.02)\n\nfeatures = ['product_type_no', 'department_no', 'colour_group_code', \n            'garment_group_no', 'index_group_no']\n\nfor idx, col in enumerate(features):\n    ax = axes[idx//3, idx%3]\n    \n    # Create box plot\n    box = ax.boxplot(articles[col].dropna(), vert=True, patch_artist=True,\n                     boxprops=dict(facecolor='lightblue', color='darkblue'),\n                     medianprops=dict(color='red', linewidth=2))\n    \n    # Calculate outlier thresholds\n    Q1 = articles[col].quantile(0.25)\n    Q3 = articles[col].quantile(0.75)\n    IQR = Q3 - Q1\n    lower_bound = Q1 - 1.5 * IQR\n    upper_bound = Q3 + 1.5 * IQR\n    \n    # Count outliers\n    outliers = articles[(articles[col] < lower_bound) | (articles[col] > upper_bound)]\n    outlier_count = len(outliers)\n    \n    ax.set_title(f'{col}\\nOutliers: {outlier_count} ({outlier_count/len(articles)*100:.1f}%)')\n    ax.set_ylabel('Value')\n    ax.grid(True, alpha=0.3)\n\n# Hide empty subplot\naxes[1, 2].axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:18.264135Z","iopub.execute_input":"2026-01-28T19:36:18.265093Z","iopub.status.idle":"2026-01-28T19:36:18.998896Z","shell.execute_reply.started":"2026-01-28T19:36:18.265044Z","shell.execute_reply":"2026-01-28T19:36:18.996474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate correlations between numeric features\nnumeric_features = articles[features]\ncorrelation_matrix = numeric_features.corr()\n\nplt.figure(figsize=(10, 8))\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm', center=0,\n            square=True, linewidths=1, cbar_kws={\"shrink\": 0.8})\nplt.title('Feature Correlation Heatmap', fontsize=16, pad=20)\nplt.tight_layout()\nplt.show()\n\nprint(\"💡 Correlation Insights:\")\nprint(\"1. High positive correlation: Features that increase together\")\nprint(\"2. High negative correlation: Features that move opposite\")\nprint(\"3. Near zero: Features are independent\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:19.001127Z","iopub.execute_input":"2026-01-28T19:36:19.001883Z","iopub.status.idle":"2026-01-28T19:36:19.341355Z","shell.execute_reply.started":"2026-01-28T19:36:19.001841Z","shell.execute_reply":"2026-01-28T19:36:19.340067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# REAL SIMPLE VERSION - NO COMPLEXITY\nimport pandas as pd\n\n# 1. Just fix the garment_group_no = 1001\narticles_clean = articles.copy()\n\n# Find the most common garment group (excluding 1001)\nmost_common_garment = articles_clean[articles_clean['garment_group_no'] != 1001]['garment_group_no'].mode()[0]\n\n# Replace 1001 with the mode\narticles_clean.loc[articles_clean['garment_group_no'] == 1001, 'garment_group_no'] = most_common_garment\n\nprint(f\"Replaced {len(articles) - len(articles_clean)} garment group unknowns\")\n\n# 2. For product_type_no = -1, just remove (small %)\narticles_clean = articles_clean[articles_clean['product_type_no'] != -1]\n\n# 3. That's literally it\nprint(f\"✅ Done. Working with {len(articles_clean)} articles\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:19.343281Z","iopub.execute_input":"2026-01-28T19:36:19.344104Z","iopub.status.idle":"2026-01-28T19:36:19.424545Z","shell.execute_reply.started":"2026-01-28T19:36:19.344067Z","shell.execute_reply":"2026-01-28T19:36:19.423487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# STEP 2: BALANCED FEATURES (3 features)\nfeatures = articles_clean[['product_type_no', 'department_no', 'colour_group_code']].values\nprint(f\"📊 Balanced features: {features.shape}\")\nprint(\"  • product_type_no (what it is)\")\nprint(\"  • department_no (where it's sold)\")  \nprint(\"  • colour_group_code (visual similarity)\")\n\n# STEP 3: SCALE\nfrom sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nscaled_features = scaler.fit_transform(features)\n\n# STEP 4: CLUSTER WITH OPTIMAL K\n# For 105K items with 3 features, try 150 clusters\nfrom sklearn.cluster import KMeans\nkmeans = KMeans(n_clusters=150, random_state=42, n_init=10, max_iter=300)\nclusters = kmeans.fit_predict(scaled_features)\n\narticles_clean['cluster'] = clusters\n\nprint(f\"🎯 Created {len(set(clusters))} clusters\")\nprint(f\"📈 Average: ~{len(articles_clean)//150:,} items per cluster\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:19.425981Z","iopub.execute_input":"2026-01-28T19:36:19.426449Z","iopub.status.idle":"2026-01-28T19:36:34.190428Z","shell.execute_reply.started":"2026-01-28T19:36:19.426415Z","shell.execute_reply":"2026-01-28T19:36:34.189347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Quick quality metrics\nfrom sklearn.metrics import silhouette_score, davies_bouldin_score\n\n# Calculate scores (sample for speed)\nsample_size = 10000\nif len(articles_clean) > sample_size:\n    sample_idx = np.random.choice(len(articles_clean), sample_size, replace=False)\n    sample_features = scaled_features[sample_idx]\n    sample_clusters = clusters[sample_idx]\nelse:\n    sample_features = scaled_features\n    sample_clusters = clusters\n\n# Silhouette score (-1 to 1, higher better)\nsil_score = silhouette_score(sample_features, sample_clusters)\nprint(f\"\\n📊 Cluster Quality Scores:\")\nprint(f\"  Silhouette Score: {sil_score:.3f}\")\nif sil_score > 0.5:\n    print(\"    ✅ Strong clustering structure\")\nelif sil_score > 0.25:\n    print(\"    👍 Reasonable structure\")  \nelse:\n    print(\"    ⚠️ Weak structure - might need more clusters\")\n\n# Davies-Bouldin score (lower better)\ndb_score = davies_bouldin_score(sample_features, sample_clusters)\nprint(f\"  Davies-Bouldin Score: {db_score:.3f}\")\nprint(\"    Lower is better (<1 is good)\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:34.191792Z","iopub.execute_input":"2026-01-28T19:36:34.192196Z","iopub.status.idle":"2026-01-28T19:36:35.911691Z","shell.execute_reply.started":"2026-01-28T19:36:34.192167Z","shell.execute_reply":"2026-01-28T19:36:35.910464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transactions_sample.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:35.913179Z","iopub.execute_input":"2026-01-28T19:36:35.913848Z","iopub.status.idle":"2026-01-28T19:36:35.926754Z","shell.execute_reply.started":"2026-01-28T19:36:35.913804Z","shell.execute_reply":"2026-01-28T19:36:35.925434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 0: ANALYZE TRANSACTIONS FIRST\nprint(\"📊 ANALYZING TRANSACTIONS DATA...\")\n\n# Group items bought together in same purchase\ntransactions_sample['basket_id'] = transactions_sample['customer_id'] + '_' + transactions_sample['t_dat']\n\n# Create item pairs that are frequently bought together\nfrom itertools import combinations\nfrom collections import defaultdict\n\nitem_pairs = defaultdict(int)\n\n# Process baskets (sample for speed)\nbasket_sample = transactions_sample.groupby('basket_id')['article_id'].apply(list)\nbasket_sample = basket_sample[basket_sample.apply(len) > 1]  # Only baskets with multiple items\n\nprint(f\"Analyzing {len(basket_sample)} multi-item baskets...\")\n\nfor items in basket_sample.head(10000):  # Start with 10K baskets\n    for item1, item2 in combinations(sorted(items), 2):\n        item_pairs[(item1, item2)] += 1\n\nprint(f\"Found {len(item_pairs)} unique item pairs\")\n\n# Convert to DataFrame for easy merging\npair_df = pd.DataFrame([\n    {'article_id_1': pair[0], 'article_id_2': pair[1], 'pair_count': count}\n    for pair, count in item_pairs.items() if count >= 2  # Minimum support\n])\n\nprint(f\"Pairs with count ≥ 2: {len(pair_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:35.928296Z","iopub.execute_input":"2026-01-28T19:36:35.928704Z","iopub.status.idle":"2026-01-28T19:36:39.581503Z","shell.execute_reply.started":"2026-01-28T19:36:35.928625Z","shell.execute_reply":"2026-01-28T19:36:39.580018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# COMPLETE RECOMMENDATION SYSTEM PIPELINE\nprint(\"🚀 COMPLETE RECOMMENDATION SYSTEM PIPELINE\")\n\n# 1. Clean articles\narticles_clean = articles[articles['product_type_no'] != -1].copy()\n\n# 2. Cluster articles (content-based)\nfeatures = articles_clean[['product_type_no', 'department_no', 'colour_group_code']].values\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.cluster import KMeans\n\nscaler = StandardScaler()\nscaled = scaler.fit_transform(features)\nkmeans = KMeans(n_clusters=150, random_state=42, n_init=10)\nclusters = kmeans.fit_predict(scaled)\narticles_clean['cluster'] = clusters\n\nprint(f\"✅ Created {len(set(clusters))} clusters\")\n\n# 3. Build recommender that combines clustering + transactions\nclass FinalRecommender:\n    def __init__(self, articles_df, transactions_df):\n        self.articles = articles_df\n        self.transactions = transactions_df.copy()  # Make a copy\n        self.cluster_model = kmeans\n        self.scaler = scaler\n        \n        # Build purchase graph\n        self.purchase_graph = self._build_purchase_graph()\n        print(f\"📊 Purchase graph built: {len(self.purchase_graph)} items with purchase data\")\n    \n    def _build_purchase_graph(self):\n        \"\"\"Build graph of items bought together\"\"\"\n        # Get item popularity from transactions\n        if len(self.transactions) > 0:\n            return self.transactions['article_id'].value_counts().to_dict()\n        return {}\n    \n    def recommend_from_basket(self, basket_article_ids, n=5):\n        \"\"\"Recommend items for a shopping basket\"\"\"\n        all_recs = []\n        \n        for article_id in basket_article_ids:\n            if article_id in self.articles['article_id'].values:\n                # Get cluster-based recommendations\n                cluster_id = self.articles.loc[\n                    self.articles['article_id'] == article_id, 'cluster'\n                ].values[0]\n                \n                cluster_recs = self.articles[\n                    (self.articles['cluster'] == cluster_id) & \n                    (~self.articles['article_id'].isin(basket_article_ids))\n                ]\n                \n                all_recs.append(cluster_recs)\n        \n        # Combine and rank recommendations\n        if all_recs:\n            combined = pd.concat(all_recs).drop_duplicates('article_id')\n            \n            # Rank by purchase popularity (if available)\n            if self.purchase_graph:\n                combined['popularity'] = combined['article_id'].map(\n                    lambda x: self.purchase_graph.get(x, 0)\n                )\n                combined = combined.sort_values('popularity', ascending=False)\n            else:\n                # Just return first N if no purchase data\n                combined = combined.head(n)\n            \n            return combined.head(n)\n        \n        return pd.DataFrame()\n\n# 4. Test basket recommendations\nprint(\"\\n🧺 TESTING BASKET RECOMMENDATIONS:\")\n\n# Make sure you have transactions_sample loaded\n# If not, let me check what you have:\nprint(\"Available variables:\", [var for var in dir() if 'transact' in var.lower()])\n\n# Sample basket from your data\nsample_basket = articles_clean['article_id'].head(3).tolist()\nprint(f\"Sample basket article IDs: {sample_basket}\")\n\n# Show what these items are\nprint(\"\\n📦 Items in basket:\")\nbasket_items = articles_clean[articles_clean['article_id'].isin(sample_basket)]\nfor _, item in basket_items.iterrows():\n    print(f\"  • {item['article_id']}: '{item['prod_name'][:30]}...' ({item['product_type_name']})\")\n\n# Initialize recommender (check if transactions_sample exists)\nif 'transactions_sample' in locals() or 'transactions_sample' in globals():\n    final_rec = FinalRecommender(articles_clean, transactions_sample)\n    \n    # Get recommendations\n    basket_recs = final_rec.recommend_from_basket(sample_basket, n=5)\n    \n    print(f\"\\n✅ Found {len(basket_recs)} recommendations for this basket\")\n    \n    if len(basket_recs) > 0:\n        print(\"\\nRecommended items:\")\n        for _, rec in basket_recs.iterrows():\n            print(f\"  • {rec['article_id']}: '{rec['prod_name'][:40]}...'\")\n            print(f\"    Type: {rec['product_type_name']}, Color: {rec['colour_group_name']}\")\n            if 'popularity' in rec:\n                print(f\"    Purchase count: {rec['popularity']}\")\n            print()\n    else:\n        print(\"No recommendations found\")\nelse:\n    print(\"\\n⚠️ transactions_sample not found. Using clustering-only recommendations.\")\n    \n    # Simple clustering-only version\n    print(\"\\n📊 Clustering-only recommendations:\")\n    for article_id in sample_basket:\n        cluster_id = articles_clean.loc[articles_clean['article_id'] == article_id, 'cluster'].values[0]\n        similar = articles_clean[\n            (articles_clean['cluster'] == cluster_id) & \n            (articles_clean['article_id'] != article_id)\n        ].head(2)\n        \n        print(f\"\\nFor article {article_id} (cluster {cluster_id}):\")\n        for _, item in similar.iterrows():\n            print(f\"  • {item['prod_name'][:40]}...\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:39.583086Z","iopub.execute_input":"2026-01-28T19:36:39.583425Z","iopub.status.idle":"2026-01-28T19:36:55.020458Z","shell.execute_reply.started":"2026-01-28T19:36:39.583393Z","shell.execute_reply":"2026-01-28T19:36:55.019461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# First, let's make sure we have clustered data\nprint(\"🔍 CHECKING IF WE HAVE CLUSTERED DATA...\")\nimages_base_path=\"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/\"\nif 'cluster' not in articles_clean.columns:\n    print(\"Running clustering first...\")\n    \n    # Clean data\n    articles_clean = articles[articles['product_type_no'] != -1].copy()\n    \n    # Balanced features clustering\n    features = articles_clean[['product_type_no', 'department_no', 'colour_group_code']].values\n    \n    from sklearn.preprocessing import StandardScaler\n    from sklearn.cluster import KMeans\n    \n    scaler = StandardScaler()\n    scaled_features = scaler.fit_transform(features)\n    \n    kmeans = KMeans(n_clusters=150, random_state=42, n_init=10)\n    clusters = kmeans.fit_predict(scaled_features)\n    \n    articles_clean['cluster'] = clusters\n    \n    print(f\"✅ Created {len(set(clusters))} clusters\")\n    print(f\"📊 Working with {len(articles_clean)} articles\")\nelse:\n    print(f\"✅ Already have {articles_clean['cluster'].nunique()} clusters\")\n    print(f\"📊 Working with {len(articles_clean)} articles\")\n\n# Now create the visual recommendation functions\nprint(\"\\n\" + \"=\"*60)\nprint(\"CREATING VISUAL RECOMMENDATION SYSTEM\")\nprint(\"=\"*60)\n\ndef get_recommendations_with_images(article_id, n_recommendations=6):\n    \"\"\"\n    Get recommendations for an article and display them with images\n    Returns: recommendations DataFrame and displays images\n    \"\"\"\n    \n    print(f\"\\n🎯 GETTING RECOMMENDATIONS FOR ARTICLE {article_id}\")\n    print(\"-\" * 50)\n    \n    # Check if article exists\n    if article_id not in articles_clean['article_id'].values:\n        print(f\"❌ Article {article_id} not found in data\")\n        return None\n    \n    # Get article info\n    article_info = articles_clean[articles_clean['article_id'] == article_id].iloc[0]\n    \n    print(f\"📋 INPUT PRODUCT:\")\n    print(f\"   Name: {article_info['prod_name']}\")\n    print(f\"   Type: {article_info['product_type_name']}\")\n    print(f\"   Color: {article_info['colour_group_name']}\")\n    print(f\"   Department: {article_info['department_name']}\")\n    print(f\"   Cluster: {article_info['cluster']}\")\n    \n    # Get recommendations from same cluster\n    cluster_id = article_info['cluster']\n    recommendations = articles_clean[\n        (articles_clean['cluster'] == cluster_id) & \n        (articles_clean['article_id'] != article_id)\n    ].head(n_recommendations)\n    \n    print(f\"\\n🔍 FOUND {len(recommendations)} SIMILAR ITEMS IN CLUSTER {cluster_id}\")\n    \n    if len(recommendations) == 0:\n        print(\"⚠️ No recommendations found in this cluster\")\n        return None\n    \n    # Display images\n    all_article_ids = [article_id] + recommendations['article_id'].tolist()\n    all_titles = [\"INPUT\"] + [f\"SIMILAR {i+1}\" for i in range(len(recommendations))]\n    \n    print(\"\\n🖼️ DISPLAYING VISUAL RECOMMENDATIONS...\")\n    display_images_simple(all_article_ids, all_titles, max_cols=4)\n    \n    # Print recommendation details\n    print(f\"\\n📊 RECOMMENDATION DETAILS:\")\n    print(\"-\" * 40)\n    \n    for idx, (_, rec) in enumerate(recommendations.iterrows()):\n        similarity_score = 0\n        \n        # Calculate similarity\n        if rec['product_type_name'] == article_info['product_type_name']:\n            similarity_score += 1\n        if rec['colour_group_name'] == article_info['colour_group_name']:\n            similarity_score += 1\n        if rec['department_name'] == article_info['department_name']:\n            similarity_score += 1\n        \n        similarity_stars = \"★\" * similarity_score + \"☆\" * (3 - similarity_score)\n        \n        print(f\"\\nRecommendation {idx+1} [{similarity_stars}]:\")\n        print(f\"  ID: {rec['article_id']}\")\n        print(f\"  Name: {rec['prod_name']}\")\n        print(f\"  Type: {rec['product_type_name']}\")\n        print(f\"  Color: {rec['colour_group_name']}\")\n        print(f\"  Department: {rec['department_name']}\")\n    \n    return recommendations\n\ndef evaluate_recommendation_quality(article_id, recommendations):\n    \"\"\"\n    Evaluate how good the recommendations are\n    \"\"\"\n    if recommendations is None or len(recommendations) == 0:\n        return\n    \n    article_info = articles_clean[articles_clean['article_id'] == article_id].iloc[0]\n    \n    print(f\"\\n📈 EVALUATION FOR ARTICLE {article_id}:\")\n    print(\"-\" * 40)\n    \n    # Calculate metrics\n    same_type = sum(1 for _, rec in recommendations.iterrows() \n                   if rec['product_type_name'] == article_info['product_type_name'])\n    \n    same_color = sum(1 for _, rec in recommendations.iterrows() \n                    if rec['colour_group_name'] == article_info['colour_group_name'])\n    \n    same_dept = sum(1 for _, rec in recommendations.iterrows() \n                   if rec['department_name'] == article_info['department_name'])\n    \n    total = len(recommendations)\n    \n    print(f\"✅ Same product type: {same_type}/{total} ({same_type/total*100:.0f}%)\")\n    print(f\"✅ Same color group: {same_color}/{total} ({same_color/total*100:.0f}%)\")\n    print(f\"✅ Same department: {same_dept}/{total} ({same_dept/total*100:.0f}%)\")\n    \n    overall_score = (same_type * 0.5 + same_color * 0.3 + same_dept * 0.2) / total\n    \n    print(f\"\\n🎯 OVERALL SIMILARITY SCORE: {overall_score:.2f}/1.0\")\n    \n    if overall_score > 0.7:\n        print(\"🌟 EXCELLENT: Recommendations are very similar!\")\n    elif overall_score > 0.5:\n        print(\"👍 GOOD: Recommendations are reasonably similar\")\n    elif overall_score > 0.3:\n        print(\"⚠️ FAIR: Some similarity, room for improvement\")\n    else:\n        print(\"❌ POOR: Recommendations not very similar\")\n    \n    return overall_score\n\n# TEST WITH YOUR BASKET ITEMS\nprint(\"\\n\" + \"=\"*60)\nprint(\"TESTING VISUAL RECOMMENDATION SYSTEM\")\nprint(\"=\"*60)\n\n# Your basket items\nsample_basket = [108775015, 108775044, 108775051]\n\nprint(f\"\\n📦 TESTING WITH BASKET OF {len(sample_basket)} ITEMS:\")\nfor idx, article_id in enumerate(sample_basket):\n    print(f\"  {idx+1}. Article {article_id}\")\n\n# Test each item in the basket\nscores = []\nfor article_id in sample_basket:\n    print(f\"\\n{'='*60}\")\n    \n    # Get recommendations\n    recs = get_recommendations_with_images(article_id, n_recommendations=4)\n    \n    # Evaluate\n    if recs is not None:\n        score = evaluate_recommendation_quality(article_id, recs)\n        scores.append(score)\n\n# Overall evaluation\nif scores:\n    print(f\"\\n{'='*60}\")\n    print(\"📊 OVERALL SYSTEM PERFORMANCE\")\n    print(f\"{'='*60}\")\n    \n    avg_score = np.mean(scores)\n    print(f\"Average similarity score: {avg_score:.2f}/1.0\")\n    \n    if avg_score > 0.7:\n        print(\"🎉 EXCELLENT SYSTEM: 'More Like This' works very well!\")\n        print(\"   Jeans → More jeans, Dresses → More dresses ✓\")\n    elif avg_score > 0.5:\n        print(\"👍 GOOD SYSTEM: Recommendations are reasonable\")\n        print(\"   Most recommendations are similar\")\n    else:\n        print(\"⚠️ SYSTEM NEEDS IMPROVEMENT\")\n        print(\"   Consider adjusting features or cluster count\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:38:34.880711Z","iopub.execute_input":"2026-01-28T19:38:34.881856Z","iopub.status.idle":"2026-01-28T19:38:39.036105Z","shell.execute_reply.started":"2026-01-28T19:38:34.88181Z","shell.execute_reply":"2026-01-28T19:38:39.035035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def simple_recommend(article_id, n=5):\n    \"\"\"Simple 'More Like This' recommendation\"\"\"\n    if article_id not in articles_clean['article_id'].values:\n        return None\n    \n    article_info = articles_clean[articles_clean['article_id'] == article_id].iloc[0]\n    cluster_id = article_info['cluster']\n    \n    # Get items from same cluster\n    recommendations = articles_clean[\n        (articles_clean['cluster'] == cluster_id) & \n        (articles_clean['article_id'] != article_id)\n    ].head(n)\n    \n    return recommendations\n\nprint(\"\\n\" + \"=\"*60)\nprint(\"FINAL SIMPLE RECOMMENDATION SYSTEM\")\nprint(\"=\"*60)\n\n# Test it\ntest_article = 108775015\nprint(f\"\\n🧪 Testing recommendation for article {test_article}\")\n\nrecs = simple_recommend(test_article, n=4)\nif recs is not None and len(recs) > 0:\n    print(f\"Found {len(recs)} recommendations\")\n    \n    # Check quality\n    article_type = articles_clean[articles_clean['article_id'] == test_article].iloc[0]['product_type_name']\n    same_type = sum(1 for _, rec in recs.iterrows() \n                   if rec['product_type_name'] == article_type)\n    \n    print(f\"\\n📊 Quality check:\")\n    print(f\"  Input type: {article_type}\")\n    print(f\"  Same type in recommendations: {same_type}/{len(recs)}\")\n    \n    if same_type == len(recs):\n        print(\"  🎉 PERFECT! System works: jeans → more jeans ✓\")\n    elif same_type >= len(recs) / 2:\n        print(\"  👍 GOOD: System mostly works\")\n    else:\n        print(\"  ⚠️ Needs adjustment\")\n    \n    # Show recommendations\n    print(f\"\\n📋 Recommendations:\")\n    for idx, (_, rec) in enumerate(recs.iterrows()):\n        print(f\"  {idx+1}. {rec['article_id']}: {rec['prod_name'][:30]}...\")\n        print(f\"     Type: {rec['product_type_name']}, Color: {rec['colour_group_name']}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:38:46.4082Z","iopub.execute_input":"2026-01-28T19:38:46.409343Z","iopub.status.idle":"2026-01-28T19:38:46.437893Z","shell.execute_reply.started":"2026-01-28T19:38:46.409293Z","shell.execute_reply":"2026-01-28T19:38:46.436707Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles.describe().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:55.881536Z","iopub.status.idle":"2026-01-28T19:36:55.882148Z","shell.execute_reply.started":"2026-01-28T19:36:55.881862Z","shell.execute_reply":"2026-01-28T19:36:55.881895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"articles.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:55.883857Z","iopub.status.idle":"2026-01-28T19:36:55.884334Z","shell.execute_reply.started":"2026-01-28T19:36:55.884103Z","shell.execute_reply":"2026-01-28T19:36:55.884129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -----------------------------\n# 1️⃣ Imports\n# -----------------------------\nimport pandas as pd\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.cluster import KMeans\nfrom sklearn.metrics.pairwise import cosine_similarity\nimport numpy as np\n\n# -----------------------------\n# 2️⃣ Load your data\n# -----------------------------\n# Replace this with your CSV / dataframe\n# Example: articles = pd.read_csv('articles.csv')\narticles = articles_clean.copy()\n\n# We'll only keep relevant columns\nfeatures = articles[['article_id', 'product_type_no', 'garment_group_no', 'colour_group_code']]\n\n# -----------------------------\n# 3️⃣ Sparse One-Hot Encoding\n# -----------------------------\nencoder = OneHotEncoder(sparse_output=True, handle_unknown='ignore')\n\n# Fit and transform the features (exclude article_id)\nX_encoded = encoder.fit_transform(features[['product_type_no', 'garment_group_no', 'colour_group_code']])\n\nprint(\"Encoded shape:\", X_encoded.shape)  # e.g., (105542, 1500+)\n\n# -----------------------------\n# 4️⃣ KMeans Clustering\n# -----------------------------\nk = 15  # number of clusters\nkmeans = KMeans(n_clusters=k, random_state=42)\nkmeans.fit(X_encoded)\n\n# Add cluster labels to original dataframe\narticles['cluster'] = kmeans.labels_\n\nprint(\"Sample clusters:\")\nprint(articles[['article_id', 'product_type_no', 'garment_group_no', 'colour_group_code', 'cluster']].head())\n\n# -----------------------------\n# 5️⃣ Basket-based recommendation\n# -----------------------------\ndef recommend_for_basket(basket_article_ids, top_n=10):\n    \"\"\"\n    basket_article_ids : list of article_ids in user's basket\n    top_n : number of recommendations\n    \"\"\"\n    # Filter encoded vectors for basket items\n    basket_mask = features['article_id'].isin(basket_article_ids)\n    basket_vectors = X_encoded[basket_mask.values]\n    \n    # Compute cosine similarity between basket items and all items\n    sims = cosine_similarity(basket_vectors, X_encoded)  # shape: (len(basket), n_articles)\n    \n    # Take mean similarity across basket items\n    mean_sims = sims.mean(axis=0)\n    \n    # Sort indices by similarity\n    sorted_idx = np.argsort(-mean_sims)\n    \n    # Filter out items already in basket\n    recommended_idx = [i for i in sorted_idx if features.iloc[i]['article_id'] not in basket_article_ids]\n    \n    # Return top_n recommended articles\n    return articles.iloc[recommended_idx[:top_n]]\n\n# -----------------------------\n# 6️⃣ Example usage\n# -----------------------------\nbasket = [108775015, 108775044]  # example article_ids\nrecommendations = recommend_for_basket(basket, top_n=10)\n\nprint(\"Recommended items for basket:\")\nprint(recommendations[['article_id', 'product_type_no', 'garment_group_no', 'colour_group_code', 'cluster']])\n   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:36:55.886609Z","iopub.status.idle":"2026-01-28T19:36:55.887163Z","shell.execute_reply.started":"2026-01-28T19:36:55.886915Z","shell.execute_reply":"2026-01-28T19:36:55.886943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\nfrom sklearn.cluster import KMeans\nfrom sklearn.metrics.pairwise import cosine_similarity\nimport numpy as np\nimport pandas as pd\n\n# -----------------------------\n# 1️⃣ Clean your data\n# -----------------------------\narticles_clean = articles[articles['product_type_no'] != -1].copy()\n\n# Select the categorical features for clustering\ncat_features = ['product_type_no', 'garment_group_no', 'colour_group_code']\n\n# -----------------------------\n# 2️⃣ One-Hot Encoding (sparse)\n# -----------------------------\nencoder = OneHotEncoder(sparse_output=True, handle_unknown='ignore')\nX_encoded = encoder.fit_transform(articles_clean[cat_features])\n\nprint(\"🔹 Shape of One-Hot encoded matrix:\", X_encoded.shape)\n\n# -----------------------------\n# 3️⃣ KMeans Clustering\n# -----------------------------\nk = 150  # number of clusters\nkmeans = KMeans(n_clusters=k, random_state=42, n_init=10)\nclusters = kmeans.fit_predict(X_encoded)\n\n# Assign cluster labels\narticles_clean['cluster'] = clusters\nprint(f\"✅ Created {len(set(clusters))} clusters for {len(articles_clean)} articles\")\n\n# -----------------------------\n# 4️⃣ Visual Recommendation function (same as before)\n# -----------------------------\nimages_base_path=\"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/\"\n\ndef display_images_simple(article_ids, titles=None, max_cols=4):\n    import matplotlib.pyplot as plt\n    from PIL import Image\n    n = len(article_ids)\n    ncols = min(n, max_cols)\n    nrows = (n + ncols - 1) // ncols\n    plt.figure(figsize=(4*ncols, 4*nrows))\n    for i, aid in enumerate(article_ids):\n        img_path = f\"{images_base_path}{str(aid).zfill(12)}.jpg\"\n        try:\n            img = Image.open(img_path)\n        except:\n            continue\n        plt.subplot(nrows, ncols, i+1)\n        plt.imshow(img)\n        plt.axis('off')\n        if titles:\n            plt.title(titles[i], fontsize=12)\n    plt.show()\n\ndef get_recommendations_with_images(article_id, n_recommendations=6):\n    if article_id not in articles_clean['article_id'].values:\n        print(f\"❌ Article {article_id} not found\")\n        return None\n    \n    article_info = articles_clean[articles_clean['article_id'] == article_id].iloc[0]\n    cluster_id = article_info['cluster']\n    \n    # Get recommendations from same cluster\n    recommendations = articles_clean[\n        (articles_clean['cluster'] == cluster_id) &\n        (articles_clean['article_id'] != article_id)\n    ].head(n_recommendations)\n    \n    all_article_ids = [article_id] + recommendations['article_id'].tolist()\n    all_titles = [\"INPUT\"] + [f\"SIMILAR {i+1}\" for i in range(len(recommendations))]\n    \n    # Display images\n    display_images_simple(all_article_ids, all_titles)\n    \n    return recommendations\n\n# -----------------------------\n# 5️⃣ Example usage\n# -----------------------------\nsample_article = 108775015\nrecs = get_recommendations_with_images(sample_article, n_recommendations=4)\nprint(recs[['article_id', 'prod_name', 'product_type_name', 'colour_group_name', 'cluster']])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:40:57.656892Z","iopub.execute_input":"2026-01-28T19:40:57.657589Z","iopub.status.idle":"2026-01-28T19:41:19.367452Z","shell.execute_reply.started":"2026-01-28T19:40:57.657557Z","shell.execute_reply":"2026-01-28T19:41:19.366419Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# First, let's make sure we have clustered data WITH ONE-HOT ENCODING\nprint(\"🔍 CHECKING IF WE HAVE CLUSTERED DATA...\")\nimages_base_path=\"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/\"\n\nif 'cluster_onehot' not in articles_clean.columns:\n    print(\"Running clustering with ONE-HOT ENCODING...\")\n    \n    # Clean data\n    articles_clean = articles[articles['product_type_no'] != -1].copy()\n    \n    # ONE-HOT ENCODING for product type (MOST IMPORTANT)\n    from sklearn.preprocessing import OneHotEncoder\n    \n    print(\"📊 Creating one-hot features...\")\n    \n    # 1. One-hot encode product_type (categorical - most important)\n    encoder_type = OneHotEncoder(sparse_output=False, handle_unknown='ignore')\n    product_type_ohe = encoder_type.fit_transform(articles_clean[['product_type_no']])\n    \n    # 2. One-hot encode department (categorical - important)\n    encoder_dept = OneHotEncoder(sparse_output=False, handle_unknown='ignore')\n    department_ohe = encoder_dept.fit_transform(articles_clean[['department_no']])\n    \n    # 3. Keep color as numeric but scaled (less important)\n    from sklearn.preprocessing import StandardScaler\n    scaler = StandardScaler()\n    color_scaled = scaler.fit_transform(articles_clean[['colour_group_code']])\n    \n    # 4. Combine with WEIGHTS: product_type(80%), department(15%), color(5%)\n    print(\"⚖️ Weighting features: Product Type 80%, Department 15%, Color 5%\")\n    \n    # Calculate weights based on number of features\n    n_product_features = product_type_ohe.shape[1]\n    n_dept_features = department_ohe.shape[1]\n    \n    # Weight each one-hot column\n    weighted_product = product_type_ohe * 0.8\n    weighted_dept = department_ohe * 0.15\n    weighted_color = color_scaled * 0.05\n    \n    # Combine all features\n    features_onehot = np.hstack([weighted_product, weighted_dept, weighted_color])\n    \n    print(f\"✅ One-hot features shape: {features_onehot.shape}\")\n    print(f\"   - Product type: {n_product_features} features (80% weight)\")\n    print(f\"   - Department: {n_dept_features} features (15% weight)\")\n    print(f\"   - Color: 1 feature (5% weight)\")\n    \n    # 5. Cluster with ONE-HOT features\n    from sklearn.cluster import KMeans\n    \n    # More clusters since one-hot gives better separation\n    kmeans = KMeans(n_clusters=200, random_state=42, n_init=10)\n    clusters_onehot = kmeans.fit_predict(features_onehot)\n    \n    articles_clean['cluster_onehot'] = clusters_onehot\n    \n    print(f\"✅ Created {len(set(clusters_onehot))} clusters with one-hot encoding\")\n    print(f\"📊 Working with {len(articles_clean)} articles\")\n    \n    # Save the encoders for later use if needed\n    import joblib\n    joblib.dump(encoder_type, 'product_type_encoder.pkl')\n    joblib.dump(encoder_dept, 'department_encoder.pkl')\n    joblib.dump(scaler, 'color_scaler.pkl')\n    \nelse:\n    print(f\"✅ Already have {articles_clean['cluster_onehot'].nunique()} one-hot clusters\")\n    print(f\"📊 Working with {len(articles_clean)} articles\")\n\n# Now create the visual recommendation functions USING ONE-HOT CLUSTERS\nprint(\"\\n\" + \"=\"*60)\nprint(\"CREATING VISUAL RECOMMENDATION SYSTEM WITH ONE-HOT\")\nprint(\"=\"*60)\n\ndef get_recommendations_with_images_onehot(article_id, n_recommendations=6):\n    \"\"\"\n    Get recommendations using ONE-HOT clustering\n    \"\"\"\n    \n    print(f\"\\n🎯 GETTING RECOMMENDATIONS FOR ARTICLE {article_id}\")\n    print(\"-\" * 50)\n    \n    # Check if article exists\n    if article_id not in articles_clean['article_id'].values:\n        print(f\"❌ Article {article_id} not found in data\")\n        return None\n    \n    # Get article info\n    article_info = articles_clean[articles_clean['article_id'] == article_id].iloc[0]\n    \n    print(f\"📋 INPUT PRODUCT:\")\n    print(f\"   Name: {article_info['prod_name']}\")\n    print(f\"   Type: {article_info['product_type_name']}\")\n    print(f\"   Color: {article_info['colour_group_name']}\")\n    print(f\"   Department: {article_info['department_name']}\")\n    \n    # Use ONE-HOT cluster\n    if 'cluster_onehot' in article_info:\n        cluster_id = article_info['cluster_onehot']\n        print(f\"   One-Hot Cluster: {cluster_id}\")\n    else:\n        print(f\"   ❌ No one-hot cluster found, using regular cluster\")\n        cluster_id = article_info['cluster']\n    \n    # Get recommendations from same ONE-HOT cluster\n    recommendations = articles_clean[\n        (articles_clean['cluster_onehot'] == cluster_id) & \n        (articles_clean['article_id'] != article_id)\n    ].head(n_recommendations)\n    \n    print(f\"\\n🔍 FOUND {len(recommendations)} SIMILAR ITEMS IN CLUSTER {cluster_id}\")\n    \n    if len(recommendations) == 0:\n        print(\"⚠️ No recommendations found in this cluster\")\n        return None\n    \n    # Display images\n    all_article_ids = [article_id] + recommendations['article_id'].tolist()\n    all_titles = [\"INPUT\"] + [f\"SIMILAR {i+1}\" for i in range(len(recommendations))]\n    \n    print(\"\\n🖼️ DISPLAYING VISUAL RECOMMENDATIONS...\")\n    display_images_simple(all_article_ids, all_titles, max_cols=4)\n    \n    # Print recommendation details\n    print(f\"\\n📊 RECOMMENDATION DETAILS:\")\n    print(\"-\" * 40)\n    \n    for idx, (_, rec) in enumerate(recommendations.iterrows()):\n        similarity_score = 0\n        \n        # Calculate similarity\n        if rec['product_type_name'] == article_info['product_type_name']:\n            similarity_score += 1\n        if rec['colour_group_name'] == article_info['colour_group_name']:\n            similarity_score += 1\n        if rec['department_name'] == article_info['department_name']:\n            similarity_score += 1\n        \n        similarity_stars = \"★\" * similarity_score + \"☆\" * (3 - similarity_score)\n        \n        print(f\"\\nRecommendation {idx+1} [{similarity_stars}]:\")\n        print(f\"  ID: {rec['article_id']}\")\n        print(f\"  Name: {rec['prod_name']}\")\n        print(f\"  Type: {rec['product_type_name']}\")\n        print(f\"  Color: {rec['colour_group_name']}\")\n        print(f\"  Department: {rec['department_name']}\")\n    \n    return recommendations\n\ndef evaluate_onehot_clusters():\n    \"\"\"\n    Evaluate how well one-hot clustering separates product types\n    \"\"\"\n    print(\"\\n📊 EVALUATING ONE-HOT CLUSTER QUALITY\")\n    print(\"-\" * 40)\n    \n    # Check cluster purity\n    cluster_purities = []\n    \n    for cluster_id in articles_clean['cluster_onehot'].unique():\n        cluster_items = articles_clean[articles_clean['cluster_onehot'] == cluster_id]\n        \n        if len(cluster_items) > 0:\n            # Most common product type in cluster\n            main_type = cluster_items['product_type_name'].mode()[0]\n            main_type_count = (cluster_items['product_type_name'] == main_type).sum()\n            \n            purity = main_type_count / len(cluster_items)\n            cluster_purities.append(purity)\n            \n            if purity < 0.5:  # Mixed cluster\n                print(f\"  Cluster {cluster_id}: {len(cluster_items)} items, purity: {purity:.1%}\")\n                print(f\"    Mixed types: {cluster_items['product_type_name'].value_counts().head(3).to_dict()}\")\n    \n    avg_purity = np.mean(cluster_purities)\n    print(f\"\\n📈 AVERAGE CLUSTER PURITY: {avg_purity:.1%}\")\n    \n    if avg_purity > 0.8:\n        print(\"🎉 EXCELLENT: One-hot clustering separates product types well!\")\n    elif avg_purity > 0.6:\n        print(\"👍 GOOD: Most clusters have dominant product type\")\n    else:\n        print(\"⚠️ FAIR: Clusters are somewhat mixed\")\n\n# TEST WITH ONE-HOT CLUSTERING\nprint(\"\\n\" + \"=\"*60)\nprint(\"TESTING ONE-HOT VISUAL RECOMMENDATION SYSTEM\")\nprint(\"=\"*60)\n\n# Evaluate cluster quality first\nevaluate_onehot_clusters()\n\n# Your basket items\nsample_basket = [108775015, 108775044, 108775051]\n\nprint(f\"\\n📦 TESTING WITH BASKET OF {len(sample_basket)} ITEMS:\")\nfor idx, article_id in enumerate(sample_basket):\n    print(f\"  {idx+1}. Article {article_id}\")\n\n# Test each item with ONE-HOT clustering\nscores = []\nfor article_id in sample_basket:\n    print(f\"\\n{'='*60}\")\n    \n    # Get recommendations using ONE-HOT\n    recs = get_recommendations_with_images_onehot(article_id, n_recommendations=4)\n    \n    # Evaluate\n    if recs is not None and len(recs) > 0:\n        article_info = articles_clean[articles_clean['article_id'] == article_id].iloc[0]\n        \n        same_type = sum(1 for _, rec in recs.iterrows() \n                       if rec['product_type_name'] == article_info['product_type_name'])\n        \n        score = same_type / len(recs)\n        scores.append(score)\n        \n        print(f\"\\n📈 ONE-HOT RESULT: {same_type}/{len(recs)} same type ({score:.0%})\")\n        \n        if score == 1.0:\n            print(\"🎉 PERFECT! All recommendations same product type\")\n        elif score >= 0.75:\n            print(\"👍 EXCELLENT: Most recommendations same type\")\n        elif score >= 0.5:\n            print(\"⚠️ GOOD: Half or more same type\")\n        else:\n            print(\"❌ NEEDS IMPROVEMENT: Too many different types\")\n\n# Overall evaluation\nif scores:\n    print(f\"\\n{'='*60}\")\n    print(\"📊 ONE-HOT SYSTEM PERFORMANCE\")\n    print(f\"{'='*60}\")\n    \n    avg_score = np.mean(scores)\n    print(f\"Average same-type score: {avg_score:.1%}\")\n    \n    if avg_score > 0.8:\n        print(\"🎉 EXCELLENT SYSTEM: One-hot encoding works perfectly!\")\n        print(\"   Jeans → More jeans, Dresses → More dresses ✓\")\n    elif avg_score > 0.6:\n        print(\"👍 GOOD SYSTEM: One-hot encoding works well\")\n        print(\"   Most recommendations are same type\")\n    else:\n        print(\"⚠️ SYSTEM NEEDS ADJUSTMENT\")\n        print(\"   Try increasing product type weight or using more clusters\")\n\n# Quick test: Show what's in a Vest top cluster\nprint(\"\\n\" + \"=\"*60)\nprint(\"🔍 ANALYZING VEST TOP CLUSTERS\")\nprint(\"=\"*60)\n\nvest_top_items = articles_clean[articles_clean['product_type_name'] == 'Vest top']\n\nif len(vest_top_items) > 0:\n    # Check how many different clusters vest tops are in\n    unique_clusters = vest_top_items['cluster_onehot'].nunique()\n    print(f\"Vest tops are in {unique_clusters} different one-hot clusters\")\n    \n    # Show the largest vest top cluster\n    largest_cluster = vest_top_items['cluster_onehot'].mode()[0]\n    vest_top_cluster = articles_clean[articles_clean['cluster_onehot'] == largest_cluster]\n    \n    print(f\"\\nLargest Vest top cluster ({largest_cluster}):\")\n    print(f\"  Contains {len(vest_top_cluster)} items\")\n    \n    # Check what types are in this cluster\n    type_distribution = vest_top_cluster['product_type_name'].value_counts()\n    print(f\"  Product types in this cluster:\")\n    for type_name, count in type_distribution.head().items():\n        percentage = count / len(vest_top_cluster) * 100\n        print(f\"    • {type_name}: {count} items ({percentage:.0f}%)\")\n    \n    # Should be mostly Vest tops!\n    vest_top_percentage = type_distribution.get('Vest top', 0) / len(vest_top_cluster) * 100\n    print(f\"\\n  Vest top purity: {vest_top_percentage:.0f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:41:24.270917Z","iopub.execute_input":"2026-01-28T19:41:24.271741Z","iopub.status.idle":"2026-01-28T19:44:09.867377Z","shell.execute_reply.started":"2026-01-28T19:41:24.271696Z","shell.execute_reply":"2026-01-28T19:44:09.866351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# First, let's make sure we have clustered data WITH ONE-HOT ENCODING\nprint(\"🔍 CHECKING IF WE HAVE CLUSTERED DATA...\")\nimages_base_path=\"/kaggle/input/h-and-m-personalized-fashion-recommendations/images/\"\n\n# RESTORE THE IMAGE DISPLAY FUNCTIONS\nimport matplotlib.pyplot as plt\nimport os\nfrom PIL import Image\nimport numpy as np\n\ndef get_image_path(article_id):\n    \"\"\"Get the correct image path from article_id\"\"\"\n    article_str = str(article_id).zfill(10)  # Make 10 digits with leading zeros\n    folder = article_str[:3]\n    filename = article_str + '.jpg'\n    return os.path.join(images_base_path, folder, filename)\n\ndef display_images_simple(article_ids, titles=None, max_cols=4):\n    \"\"\"Simple, reliable image display\"\"\"\n    if titles is None:\n        titles = [f\"Item {i+1}\" for i in range(len(article_ids))]\n    \n    n_items = len(article_ids)\n    n_cols = min(n_items, max_cols)\n    n_rows = (n_items + n_cols - 1) // n_cols\n    \n    fig, axes = plt.subplots(n_rows, n_cols, figsize=(15, 4 * n_rows))\n    \n    # Handle single axes case\n    if n_rows == 1 and n_cols == 1:\n        axes = [axes]\n    elif n_rows == 1:\n        axes = axes\n    else:\n        axes = axes.flatten()\n    \n    for idx, (article_id, title) in enumerate(zip(article_ids, titles)):\n        if idx >= len(axes):\n            break\n            \n        ax = axes[idx]\n        img_path = get_image_path(article_id)\n        \n        if os.path.exists(img_path):\n            try:\n                img = Image.open(img_path)\n                ax.imshow(img)\n                ax.axis('off')\n                ax.set_title(f\"{title}\\nID: {article_id}\", fontsize=10)\n            except Exception as e:\n                ax.text(0.5, 0.5, f\"Error loading\\n{article_id}\", \n                       ha='center', va='center', fontsize=8)\n                ax.axis('off')\n        else:\n            ax.text(0.5, 0.5, f\"No image found\\n{article_id}\", \n                   ha='center', va='center', fontsize=8)\n            ax.axis('off')\n    \n    # Hide empty axes\n    for idx in range(len(article_ids), len(axes)):\n        axes[idx].axis('off')\n    \n    plt.tight_layout()\n    plt.show()\n\n# NOW PROCEED WITH ONE-HOT CLUSTERING\nif 'cluster_onehot' not in articles_clean.columns:\n    print(\"Running clustering with ONE-HOT ENCODING...\")\n    \n    # Clean data\n    articles_clean = articles[articles['product_type_no'] != -1].copy()\n    \n    # ONE-HOT ENCODING for product type (MOST IMPORTANT)\n    from sklearn.preprocessing import OneHotEncoder\n    \n    print(\"📊 Creating one-hot features...\")\n    \n    # 1. One-hot encode product_type (categorical - most important)\n    encoder_type = OneHotEncoder(sparse_output=False, handle_unknown='ignore')\n    product_type_ohe = encoder_type.fit_transform(articles_clean[['product_type_no']])\n    \n    # 2. One-hot encode department (categorical - important)\n    encoder_dept = OneHotEncoder(sparse_output=False, handle_unknown='ignore')\n    department_ohe = encoder_dept.fit_transform(articles_clean[['department_no']])\n    \n    # 3. Keep color as numeric but scaled (less important)\n    from sklearn.preprocessing import StandardScaler\n    scaler = StandardScaler()\n    color_scaled = scaler.fit_transform(articles_clean[['colour_group_code']])\n    \n    # 4. Combine with WEIGHTS: product_type(80%), department(15%), color(5%)\n    print(\"⚖️ Weighting features: Product Type 80%, Department 15%, Color 5%\")\n    \n    # Calculate weights based on number of features\n    n_product_features = product_type_ohe.shape[1]\n    n_dept_features = department_ohe.shape[1]\n    \n    # Weight each one-hot column\n    weighted_product = product_type_ohe * 0.8\n    weighted_dept = department_ohe * 0.15\n    weighted_color = color_scaled * 0.05\n    \n    # Combine all features\n    features_onehot = np.hstack([weighted_product, weighted_dept, weighted_color])\n    \n    print(f\"✅ One-hot features shape: {features_onehot.shape}\")\n    print(f\"   - Product type: {n_product_features} features (80% weight)\")\n    print(f\"   - Department: {n_dept_features} features (15% weight)\")\n    print(f\"   - Color: 1 feature (5% weight)\")\n    \n    # 5. Cluster with ONE-HOT features\n    from sklearn.cluster import KMeans\n    \n    # More clusters since one-hot gives better separation\n    kmeans = KMeans(n_clusters=200, random_state=42, n_init=10)\n    clusters_onehot = kmeans.fit_predict(features_onehot)\n    \n    articles_clean['cluster_onehot'] = clusters_onehot\n    \n    print(f\"✅ Created {len(set(clusters_onehot))} clusters with one-hot encoding\")\n    print(f\"📊 Working with {len(articles_clean)} articles\")\n    \nelse:\n    print(f\"✅ Already have {articles_clean['cluster_onehot'].nunique()} one-hot clusters\")\n    print(f\"📊 Working with {len(articles_clean)} articles\")\n\n# Now create the visual recommendation functions USING ONE-HOT CLUSTERS\nprint(\"\\n\" + \"=\"*60)\nprint(\"CREATING VISUAL RECOMMENDATION SYSTEM WITH ONE-HOT\")\nprint(\"=\"*60)\n\ndef get_recommendations_with_images_onehot(article_id, n_recommendations=6):\n    \"\"\"\n    Get recommendations using ONE-HOT clustering\n    \"\"\"\n    \n    print(f\"\\n🎯 GETTING RECOMMENDATIONS FOR ARTICLE {article_id}\")\n    print(\"-\" * 50)\n    \n    # Check if article exists\n    if article_id not in articles_clean['article_id'].values:\n        print(f\"❌ Article {article_id} not found in data\")\n        return None\n    \n    # Get article info\n    article_info = articles_clean[articles_clean['article_id'] == article_id].iloc[0]\n    \n    print(f\"📋 INPUT PRODUCT:\")\n    print(f\"   Name: {article_info['prod_name']}\")\n    print(f\"   Type: {article_info['product_type_name']}\")\n    print(f\"   Color: {article_info['colour_group_name']}\")\n    print(f\"   Department: {article_info['department_name']}\")\n    \n    # Use ONE-HOT cluster\n    if 'cluster_onehot' in article_info:\n        cluster_id = article_info['cluster_onehot']\n        print(f\"   One-Hot Cluster: {cluster_id}\")\n    else:\n        print(f\"   ❌ No one-hot cluster found\")\n        return None\n    \n    # Get recommendations from same ONE-HOT cluster\n    recommendations = articles_clean[\n        (articles_clean['cluster_onehot'] == cluster_id) & \n        (articles_clean['article_id'] != article_id)\n    ].head(n_recommendations)\n    \n    print(f\"\\n🔍 FOUND {len(recommendations)} SIMILAR ITEMS IN CLUSTER {cluster_id}\")\n    \n    if len(recommendations) == 0:\n        print(\"⚠️ No recommendations found in this cluster\")\n        return None\n    \n    # DISPLAY IMAGES - THIS IS THE KEY VISUAL PART!\n    all_article_ids = [article_id] + recommendations['article_id'].tolist()\n    all_titles = [\"INPUT\"] + [f\"SIMILAR {i+1}\" for i in range(len(recommendations))]\n    \n    print(\"\\n🖼️ DISPLAYING VISUAL RECOMMENDATIONS...\")\n    display_images_simple(all_article_ids, all_titles, max_cols=4)\n    \n    # Print recommendation details\n    print(f\"\\n📊 RECOMMENDATION DETAILS:\")\n    print(\"-\" * 40)\n    \n    for idx, (_, rec) in enumerate(recommendations.iterrows()):\n        similarity_score = 0\n        \n        # Calculate similarity\n        if rec['product_type_name'] == article_info['product_type_name']:\n            similarity_score += 1\n        if rec['colour_group_name'] == article_info['colour_group_name']:\n            similarity_score += 1\n        if rec['department_name'] == article_info['department_name']:\n            similarity_score += 1\n        \n        similarity_stars = \"★\" * similarity_score + \"☆\" * (3 - similarity_score)\n        \n        print(f\"\\nRecommendation {idx+1} [{similarity_stars}]:\")\n        print(f\"  ID: {rec['article_id']}\")\n        print(f\"  Name: {rec['prod_name']}\")\n        print(f\"  Type: {rec['product_type_name']}\")\n        print(f\"  Color: {rec['colour_group_name']}\")\n        print(f\"  Department: {rec['department_name']}\")\n    \n    return recommendations\n\n# TEST WITH ONE-HOT CLUSTERING AND VISUALIZATION\nprint(\"\\n\" + \"=\"*60)\nprint(\"TESTING ONE-HOT VISUAL RECOMMENDATION SYSTEM\")\nprint(\"=\"*60)\n\n# Your basket items\nsample_basket = [108775015, 108775044, 108775051]\n\nprint(f\"\\n📦 TESTING WITH BASKET OF {len(sample_basket)} ITEMS:\")\nfor idx, article_id in enumerate(sample_basket):\n    article_name = articles_clean[articles_clean['article_id'] == article_id]['prod_name'].iloc[0] if article_id in articles_clean['article_id'].values else \"Unknown\"\n    print(f\"  {idx+1}. Article {article_id}: '{article_name[:30]}...'\")\n\n# Test each item with ONE-HOT clustering AND IMAGES\nprint(\"\\n\" + \"=\"*60)\nprint(\"🧪 VISUAL TEST 1: Vest top (should show similar vest tops)\")\nprint(\"=\"*60)\n\nfirst_article = sample_basket[0]\nrecs = get_recommendations_with_images_onehot(first_article, n_recommendations=6)\n\nif recs is not None and len(recs) > 0:\n    # Check if recommendations are same type\n    article_type = articles_clean[articles_clean['article_id'] == first_article].iloc[0]['product_type_name']\n    same_type_count = sum(1 for _, rec in recs.iterrows() if rec['product_type_name'] == article_type)\n    \n    print(f\"\\n📈 RESULTS FOR VEST TOP TEST:\")\n    print(f\"  Input type: {article_type}\")\n    print(f\"  Same type in recommendations: {same_type_count}/{len(recs)} ({same_type_count/len(recs)*100:.0f}%)\")\n    \n    if same_type_count == len(recs):\n        print(\"  🎉 PERFECT! One-hot clustering works: All recommendations are Vest tops!\")\n    elif same_type_count >= len(recs) / 2:\n        print(\"  👍 GOOD: Most recommendations are same type\")\n    else:\n        print(\"  ⚠️ NEEDS IMPROVEMENT: Too many different types\")\n\n# Test second item\nprint(\"\\n\" + \"=\"*60)\nprint(\"🧪 VISUAL TEST 2: Another Vest top\")\nprint(\"=\"*60)\n\nsecond_article = sample_basket[1]\nrecs2 = get_recommendations_with_images_onehot(second_article, n_recommendations=6)\n\n# Test third item  \nprint(\"\\n\" + \"=\"*60)\nprint(\"🧪 VISUAL TEST 3: Striped Vest top\")\nprint(\"=\"*60)\n\nthird_article = sample_basket[2]\nrecs3 = get_recommendations_with_images_onehot(third_article, n_recommendations=6)\n\n# COMPARE CLUSTERS VISUALLY\nprint(\"\\n\" + \"=\"*60)\nprint(\"🔍 COMPARING BASKET ITEMS CLUSTERS\")\nprint(\"=\"*60)\n\nprint(\"\\n📊 YOUR BASKET ITEMS AND THEIR ONE-HOT CLUSTERS:\")\ncluster_summary = {}\n\nfor article_id in sample_basket:\n    if article_id in articles_clean['article_id'].values:\n        info = articles_clean[articles_clean['article_id'] == article_id].iloc[0]\n        cluster_id = info['cluster_onehot']\n        \n        print(f\"\\n  Article {article_id}:\")\n        print(f\"    Name: '{info['prod_name'][:30]}...'\")\n        print(f\"    Type: {info['product_type_name']}\")\n        print(f\"    Color: {info['colour_group_name']}\")\n        print(f\"    One-Hot Cluster: {cluster_id}\")\n        \n        # Track which cluster has which items\n        if cluster_id not in cluster_summary:\n            cluster_summary[cluster_id] = []\n        cluster_summary[cluster_id].append(article_id)\n\nprint(f\"\\n📈 CLUSTER ANALYSIS:\")\nprint(f\"  Your {len(sample_basket)} basket items are in {len(cluster_summary)} different one-hot clusters\")\n\nfor cluster_id, items in cluster_summary.items():\n    print(f\"\\n  Cluster {cluster_id} contains {len(items)} basket items:\")\n    for article_id in items:\n        item_info = articles_clean[articles_clean['article_id'] == article_id].iloc[0]\n        print(f\"    • {article_id}: {item_info['prod_name'][:25]}... ({item_info['product_type_name']})\")\n\n# SHOW WHAT'S IN EACH CLUSTER VISUALLY\nprint(\"\\n\" + \"=\"*60)\nprint(\"🖼️ VISUAL CLUSTER COMPARISON\")\nprint(\"=\"*60)\n\nfor cluster_id, items in cluster_summary.items():\n    print(f\"\\n📦 CLUSTER {cluster_id} VISUALIZATION:\")\n    \n    # Get all items in this cluster (limit to 8 for display)\n    cluster_items = articles_clean[articles_clean['cluster_onehot'] == cluster_id]\n    \n    print(f\"  Total items in cluster: {len(cluster_items)}\")\n    \n    # Show product type distribution\n    top_types = cluster_items['product_type_name'].value_counts().head(3)\n    print(f\"  Main product types:\")\n    for type_name, count in top_types.items():\n        percentage = count / len(cluster_items) * 100\n        print(f\"    • {type_name}: {count} items ({percentage:.0f}%)\")\n    \n    # Display sample items from this cluster\n    sample_size = min(6, len(cluster_items))\n    sample_items = cluster_items.head(sample_size)\n    \n    print(f\"\\n  🖼️ Sample of {sample_size} items from this cluster:\")\n    display_images_simple(\n        sample_items['article_id'].tolist(),\n        [f\"Item {i+1}\" for i in range(sample_size)],\n        max_cols=3\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:47:49.987591Z","iopub.execute_input":"2026-01-28T19:47:49.988861Z","iopub.status.idle":"2026-01-28T19:47:56.141876Z","shell.execute_reply.started":"2026-01-28T19:47:49.988827Z","shell.execute_reply":"2026-01-28T19:47:56.140861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_smart_basket_recommendations(basket_article_ids, n_recommendations=10):\n    \"\"\"\n    Advanced: Use cosine similarity on one-hot encoded features\n    to find items similar to the ENTIRE basket\n    \"\"\"\n    \n    print(f\"\\n🧠 SMART BASKET RECOMMENDATIONS (Cosine Similarity)\")\n    print(\"=\"*60)\n    \n    # Recreate one-hot features (same as before)\n    from sklearn.preprocessing import OneHotEncoder, StandardScaler\n    \n    # One-hot encode product_type\n    encoder_type = OneHotEncoder(sparse_output=False, handle_unknown='ignore')\n    product_type_ohe = encoder_type.fit_transform(articles_clean[['product_type_no']])\n    \n    # One-hot encode department  \n    encoder_dept = OneHotEncoder(sparse_output=False, handle_unknown='ignore')\n    department_ohe = encoder_dept.fit_transform(articles_clean[['department_no']])\n    \n    # Scale color\n    scaler = StandardScaler()\n    color_scaled = scaler.fit_transform(articles_clean[['colour_group_code']])\n    \n    # Weighted combination\n    features_matrix = np.hstack([\n        product_type_ohe * 0.8,\n        department_ohe * 0.15,\n        color_scaled * 0.05\n    ])\n    \n    print(f\"📊 Feature matrix: {features_matrix.shape}\")\n    \n    # Get indices of basket items\n    basket_indices = []\n    for article_id in basket_article_ids:\n        if article_id in articles_clean['article_id'].values:\n            idx = articles_clean[articles_clean['article_id'] == article_id].index[0]\n            basket_indices.append(idx)\n    \n    if not basket_indices:\n        return None\n    \n    print(f\"✅ Found {len(basket_indices)} basket items\")\n    \n    # Calculate basket CENTROID (average feature vector)\n    basket_vectors = features_matrix[basket_indices]\n    basket_centroid = basket_vectors.mean(axis=0)\n    \n    print(f\"📈 Created basket centroid (average of {len(basket_indices)} items)\")\n    \n    # Calculate cosine similarity between centroid and ALL items\n    from sklearn.metrics.pairwise import cosine_similarity\n    \n    # Reshape for cosine_similarity\n    basket_centroid_reshaped = basket_centroid.reshape(1, -1)\n    \n    # Calculate similarities\n    similarities = cosine_similarity(basket_centroid_reshaped, features_matrix)[0]\n    \n    # Sort by similarity (highest first)\n    sorted_indices = np.argsort(-similarities)\n    \n    # Filter out basket items\n    recommendations = []\n    article_ids_set = set(basket_article_ids)\n    \n    for idx in sorted_indices:\n        article_id = articles_clean.iloc[idx]['article_id']\n        if article_id not in article_ids_set:\n            recommendations.append(articles_clean.iloc[idx])\n            if len(recommendations) >= n_recommendations:\n                break\n    \n    final_recs = pd.DataFrame(recommendations)\n    \n    # Display results\n    print(f\"\\n🎯 SMART RECOMMENDATIONS:\")\n    print(f\"  Found {len(final_recs)} recommendations\")\n    \n    # Show basket items\n    print(f\"\\n🖼️ YOUR BASKET:\")\n    display_images_simple(\n        basket_article_ids,\n        [f\"Item {i+1}\" for i in range(len(basket_article_ids))],\n        max_cols=min(4, len(basket_article_ids))\n    )\n    \n    # Show recommendations\n    print(f\"\\n🖼️ SMART RECOMMENDATIONS:\")\n    display_images_simple(\n        final_recs['article_id'].tolist(),\n        [f\"Smart {i+1}\" for i in range(len(final_recs))],\n        max_cols=4\n    )\n    \n    # Analyze what the basket \"wants\"\n    print(f\"\\n📊 BASKET ANALYSIS:\")\n    \n    # What product types are in basket?\n    basket_items = articles_clean[articles_clean['article_id'].isin(basket_article_ids)]\n    basket_types = basket_items['product_type_name'].value_counts()\n    \n    print(f\"  Basket contains:\")\n    for product_type, count in basket_types.items():\n        print(f\"    • {product_type}: {count} items\")\n    \n    # What product types are recommended?\n    rec_types = final_recs['product_type_name'].value_counts()\n    \n    print(f\"\\n  Recommendations contain:\")\n    for product_type, count in rec_types.head(5).items():\n        percentage = count / len(final_recs) * 100\n        print(f\"    • {product_type}: {count} items ({percentage:.0f}%)\")\n    \n    # Check if recommendations match basket\n    matching = sum(1 for _, rec in final_recs.iterrows() \n                  if rec['product_type_name'] in basket_types.index)\n    \n    print(f\"\\n📈 MATCHING SCORE: {matching}/{len(final_recs)} ({matching/len(final_recs)*100:.0f}%)\")\n    \n    return final_recs\n\n# TEST SMART BASKET RECOMMENDATIONS\nprint(\"\\n\" + \"=\"*60)\nprint(\"🧠 TESTING SMART BASKET RECOMMENDATIONS\")\nprint(\"=\"*60)\n\nsmart_basket_recs = get_smart_basket_recommendations(sample_basket, n_recommendations=8)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:50:14.267394Z","iopub.execute_input":"2026-01-28T19:50:14.268276Z","iopub.status.idle":"2026-01-28T19:50:17.940741Z","shell.execute_reply.started":"2026-01-28T19:50:14.268237Z","shell.execute_reply":"2026-01-28T19:50:17.939557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n\" + \"=\"*60)\nprint(\"🧺 BASKET-LEVEL RECOMMENDATION SYSTEM\")\nprint(\"=\"*60)\n\ndef get_basket_recommendations(basket_article_ids, n_recommendations=10):\n    \"\"\"\n    Get recommendations for an ENTIRE basket of items\n    Combines similarities from all items in the basket\n    \"\"\"\n    \n    print(f\"🧺 PROCESSING BASKET WITH {len(basket_article_ids)} ITEMS\")\n    print(\"-\" * 50)\n    \n    # Check which basket items exist in our data\n    valid_basket_ids = []\n    basket_items_info = []\n    \n    for article_id in basket_article_ids:\n        if article_id in articles_clean['article_id'].values:\n            valid_basket_ids.append(article_id)\n            item_info = articles_clean[articles_clean['article_id'] == article_id].iloc[0]\n            basket_items_info.append(item_info)\n    \n    if not valid_basket_ids:\n        print(\"❌ No valid basket items found in data\")\n        return None\n    \n    print(f\"✅ Found {len(valid_basket_ids)} valid items in basket:\")\n    for item in basket_items_info:\n        print(f\"  • {item['article_id']}: '{item['prod_name'][:30]}...' ({item['product_type_name']})\")\n    \n    # STRATEGY 1: UNION of clusters (items from any basket item's cluster)\n    print(f\"\\n📊 STRATEGY 1: Union of all basket clusters\")\n    \n    # Get all unique clusters from basket items\n    basket_clusters = set()\n    for item in basket_items_info:\n        if 'cluster_onehot' in item:\n            basket_clusters.add(item['cluster_onehot'])\n    \n    print(f\"  Basket items are in {len(basket_clusters)} different clusters\")\n    \n    # Get items from ALL these clusters (excluding basket items)\n    recommendations_union = articles_clean[\n        articles_clean['cluster_onehot'].isin(basket_clusters) & \n        (~articles_clean['article_id'].isin(valid_basket_ids))\n    ]\n    \n    print(f\"  Found {len(recommendations_union)} potential recommendations\")\n    \n    # STRATEGY 2: Weighted by basket composition\n    print(f\"\\n📊 STRATEGY 2: Weighted by basket frequency\")\n    \n    # Count how many basket items are in each product type\n    basket_product_types = {}\n    for item in basket_items_info:\n        product_type = item['product_type_name']\n        basket_product_types[product_type] = basket_product_types.get(product_type, 0) + 1\n    \n    print(f\"  Basket contains these product types:\")\n    for product_type, count in basket_product_types.items():\n        percentage = count / len(basket_items_info) * 100\n        print(f\"    • {product_type}: {count} items ({percentage:.0f}%)\")\n    \n    # Get recommendations weighted by basket composition\n    all_recommendations = []\n    \n    for product_type, count in basket_product_types.items():\n        # Weight = percentage of basket with this type\n        weight = count / len(basket_items_info)\n        \n        # Get items of this product type (excluding basket items)\n        type_recommendations = articles_clean[\n            (articles_clean['product_type_name'] == product_type) & \n            (~articles_clean['article_id'].isin(valid_basket_ids))\n        ]\n        \n        # Take proportion based on weight\n        n_to_take = int(n_recommendations * weight)\n        if len(type_recommendations) > 0:\n            sampled = type_recommendations.sample(min(n_to_take, len(type_recommendations)), random_state=42)\n            all_recommendations.append(sampled)\n    \n    recommendations_weighted = pd.concat(all_recommendations).head(n_recommendations)\n    \n    print(f\"\\n🎯 FINAL RECOMMENDATIONS FOR BASKET\")\n    print(\"-\" * 40)\n    \n    # Choose which strategy to use (or combine)\n    # Let's use weighted strategy as it's more sophisticated\n    final_recommendations = recommendations_weighted\n    \n    if len(final_recommendations) == 0:\n        # Fallback to union strategy\n        final_recommendations = recommendations_union.head(n_recommendations)\n    \n    print(f\"📦 Showing {len(final_recommendations)} recommendations for the basket\")\n    \n    # Display basket items\n    print(f\"\\n🖼️ YOUR BASKET ITEMS:\")\n    display_images_simple(\n        valid_basket_ids,\n        [f\"Basket {i+1}\" for i in range(len(valid_basket_ids))],\n        max_cols=min(4, len(valid_basket_ids))\n    )\n    \n    # Display recommendations\n    print(f\"\\n🖼️ RECOMMENDATIONS FOR THIS BASKET:\")\n    display_images_simple(\n        final_recommendations['article_id'].tolist(),\n        [f\"Rec {i+1}\" for i in range(len(final_recommendations))],\n        max_cols=4\n    )\n    \n    # Show recommendation details\n    print(f\"\\n📋 RECOMMENDATION DETAILS:\")\n    print(\"-\" * 40)\n    \n    basket_types = list(basket_product_types.keys())\n    \n    for idx, (_, rec) in enumerate(final_recommendations.iterrows()):\n        # Check which basket product type this matches\n        matches_basket_type = rec['product_type_name'] in basket_types\n        \n        print(f\"\\nRecommendation {idx+1}:\")\n        print(f\"  ID: {rec['article_id']}\")\n        print(f\"  Name: '{rec['prod_name'][:40]}...'\")\n        print(f\"  Type: {rec['product_type_name']} {'✅ (Matches basket)' if matches_basket_type else ''}\")\n        print(f\"  Color: {rec['colour_group_name']}\")\n        print(f\"  Department: {rec['department_name']}\")\n    \n    # Evaluation\n    print(f\"\\n📊 BASKET RECOMMENDATION EVALUATION:\")\n    print(\"-\" * 40)\n    \n    matching_types = sum(1 for _, rec in final_recommendations.iterrows() \n                        if rec['product_type_name'] in basket_types)\n    \n    print(f\"  Recommendations matching basket types: {matching_types}/{len(final_recommendations)} ({matching_types/len(final_recommendations)*100:.0f}%)\")\n    \n    if matching_types == len(final_recommendations):\n        print(\"  🎉 PERFECT! All recommendations match basket types\")\n    elif matching_types >= len(final_recommendations) / 2:\n        print(\"  👍 GOOD: Most recommendations match basket\")\n    else:\n        print(\"  ⚠️ Could be better: Many recommendations don't match basket\")\n    \n    return final_recommendations\n\n# TEST WITH YOUR BASKET\nprint(\"\\n\" + \"=\"*60)\nprint(\"🧪 TESTING BASKET-LEVEL RECOMMENDATIONS\")\nprint(\"=\"*60)\n\n# Your basket\nsample_basket = [108775015, 108775044, 108775051]\n\nprint(f\"Testing with basket: {sample_basket}\")\nprint(\"(All items are Vest tops - should recommend more Vest tops)\")\n\nbasket_recs = get_basket_recommendations(sample_basket, n_recommendations=8)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:51:28.539354Z","iopub.execute_input":"2026-01-28T19:51:28.540001Z","iopub.status.idle":"2026-01-28T19:51:31.011218Z","shell.execute_reply.started":"2026-01-28T19:51:28.53997Z","shell.execute_reply":"2026-01-28T19:51:31.010191Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def test_mixed_basket():\n    \"\"\"\n    Test with a realistic mixed basket (jeans + dress + top)\n    \"\"\"\n    print(\"\\n\" + \"=\"*60)\n    print(\"👗 TESTING MIXED BASKET RECOMMENDATIONS\")\n    print(\"=\"*60)\n    \n    # Find sample items of different types\n    jeans_items = articles_clean[articles_clean['product_type_name'].str.contains('jean', case=False, na=False)]\n    dress_items = articles_clean[articles_clean['product_type_name'].str.contains('dress', case=False, na=False)]\n    shirt_items = articles_clean[articles_clean['product_type_name'].str.contains('shirt|top', case=False, na=False)]\n    \n    # Create a mixed basket (if we have examples)\n    mixed_basket = []\n    \n    if len(jeans_items) > 0:\n        mixed_basket.append(jeans_items.iloc[0]['article_id'])\n        print(f\"Added jeans: {jeans_items.iloc[0]['prod_name'][:30]}...\")\n    \n    if len(dress_items) > 0:\n        mixed_basket.append(dress_items.iloc[0]['article_id'])\n        print(f\"Added dress: {dress_items.iloc[0]['prod_name'][:30]}...\")\n    \n    if len(shirt_items) > 0:\n        mixed_basket.append(shirt_items.iloc[0]['article_id'])\n        print(f\"Added top: {shirt_items.iloc[0]['prod_name'][:30]}...\")\n    \n    if len(mixed_basket) >= 2:\n        print(f\"\\n🧺 MIXED BASKET: {mixed_basket}\")\n        recommendations = get_basket_recommendations(mixed_basket, n_recommendations=8)\n        \n        # The system should recommend a mix of jeans, dresses, and tops!\n        if recommendations is not None:\n            print(f\"\\n📊 EXPECTED: Recommendations should include jeans, dresses, AND tops\")\n            \n            rec_types = recommendations['product_type_name'].value_counts()\n            print(f\"\\n📈 ACTUAL RECOMMENDATION DISTRIBUTION:\")\n            for product_type, count in rec_types.items():\n                print(f\"  • {product_type}: {count} items\")\n    else:\n        print(\"⚠️ Not enough diverse items for mixed basket test\")\n\n# Run mixed basket test\ntest_mixed_basket()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T19:51:57.585795Z","iopub.execute_input":"2026-01-28T19:51:57.586844Z","iopub.status.idle":"2026-01-28T19:52:00.344734Z","shell.execute_reply.started":"2026-01-28T19:51:57.586808Z","shell.execute_reply":"2026-01-28T19:52:00.343572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"\\n\" + \"=\"*60)\nprint(\"🛒 TESTING ON REAL BASKETS FROM TRANSACTIONS\")\nprint(\"=\"*60)\n\n# First, extract baskets from transactions\nprint(\"1️⃣ EXTRACTING REAL BASKETS FROM TRANSACTIONS...\")\n\n# Create basket ID: customer + date (same shopping trip)\ntransactions_sample['basket_id'] = (\n    transactions_sample['customer_id'].astype(str) + '_' + \n    pd.to_datetime(transactions_sample['t_dat']).dt.date.astype(str)\n)\n\n# Group items into baskets\nbaskets = transactions_sample.groupby('basket_id')['article_id'].apply(list).reset_index()\nbaskets['basket_size'] = baskets['article_id'].apply(len)\n\nprint(f\"✅ Extracted {len(baskets):,} shopping baskets\")\nprint(f\"📊 Basket size distribution:\")\nprint(baskets['basket_size'].value_counts().sort_index().head(10))\n\n# Filter for usable baskets (2-6 items)\nusable_baskets = baskets[(baskets['basket_size'] >= 2) & (baskets['basket_size'] <= 6)].copy()\nprint(f\"\\n🎯 Using {len(usable_baskets):,} baskets with 2-6 items\")\n\n# Test your algorithm on REAL baskets\nprint(\"\\n2️⃣ TESTING YOUR ALGORITHM ON REAL BASKETS\")\nprint(\"=\"*60)\n\ndef test_real_basket(basket_article_ids, basket_number=1):\n    \"\"\"\n    Test your get_basket_recommendations function on a real basket\n    \"\"\"\n    print(f\"\\n🧪 TEST {basket_number}: REAL CUSTOMER BASKET\")\n    print(\"=\"*50)\n    \n    # Get basket details\n    print(f\"🛒 ORIGINAL CUSTOMER BASKET ({len(basket_article_ids)} items):\")\n    \n    # Show basket items with IMAGES\n    basket_details = articles_clean[articles_clean['article_id'].isin(basket_article_ids)]\n    \n    if len(basket_details) == 0:\n        print(\"❌ No basket items found in articles_clean\")\n        return None\n    \n    # Display basket items with images\n    print(\"\\n🖼️ CUSTOMER'S ORIGINAL BASKET:\")\n    display_images_simple(\n        basket_details['article_id'].tolist(),\n        [f\"Basket Item {i+1}\" for i in range(len(basket_details))],\n        max_cols=min(4, len(basket_details))\n    )\n    \n    # Print basket details\n    print(\"\\n📋 BASKET CONTENTS:\")\n    for idx, (_, item) in enumerate(basket_details.iterrows()):\n        print(f\"  {idx+1}. {item['article_id']}: {item['prod_name']}\")\n        print(f\"     Type: {item['product_type_name']}, Color: {item['colour_group_name']}\")\n    \n    # Run your basket recommendation algorithm\n    print(f\"\\n🎯 RUNNING YOUR RECOMMENDATION ALGORITHM...\")\n    recommendations = get_basket_recommendations(basket_article_ids, n_recommendations=8)\n    \n    if recommendations is not None and len(recommendations) > 0:\n        # Analyze results\n        print(f\"\\n📊 ANALYSIS OF RECOMMENDATIONS:\")\n        \n        # Get basket product types\n        basket_types = basket_details['product_type_name'].unique()\n        print(f\"  Basket contains these product types: {', '.join(basket_types)}\")\n        \n        # Check recommendations\n        matching_count = 0\n        recommendation_types = []\n        \n        for _, rec in recommendations.iterrows():\n            if rec['product_type_name'] in basket_types:\n                matching_count += 1\n                recommendation_types.append(f\"✅ {rec['product_type_name']}\")\n            else:\n                recommendation_types.append(f\"❌ {rec['product_type_name']}\")\n        \n        print(f\"  Recommendations matching basket types: {matching_count}/{len(recommendations)}\")\n        print(f\"  Recommendation types: {', '.join(recommendation_types)}\")\n        \n        if matching_count == len(recommendations):\n            print(\"  🎉 PERFECT! All recommendations match basket types\")\n        elif matching_count >= len(recommendations) / 2:\n            print(\"  👍 GOOD: Most recommendations are relevant\")\n        else:\n            print(\"  ⚠️ ROOM FOR IMPROVEMENT: Many recommendations don't match\")\n    \n    return recommendations\n\n# Test on first 3 real baskets\nprint(\"\\n\" + \"=\"*60)\nprint(\"🧪 RUNNING TESTS ON REAL CUSTOMER BASKETS\")\nprint(\"=\"*60)\n\n# Find baskets that have items in our articles_clean data\ntest_baskets = []\n\nfor idx, basket in usable_baskets.iterrows():\n    basket_items = basket['article_id']\n    \n    # Check how many items exist in articles_clean\n    existing_items = [item for item in basket_items if item in articles_clean['article_id'].values]\n    \n    if len(existing_items) >= 2:  # At least 2 items must exist\n        test_baskets.append(existing_items)\n    \n    if len(test_baskets) >= 3:  # Test 3 baskets\n        break\n\nif len(test_baskets) == 0:\n    print(\"⚠️ No baskets found with items in articles_clean\")\n    # Try with your original basket as fallback\n    test_baskets = [sample_basket]\n    print(\"Using your sample basket instead\")\n\n# Run tests\nall_results = []\nfor i, basket_items in enumerate(test_baskets):\n    result = test_real_basket(basket_items, basket_number=i+1)\n    if result is not None:\n        all_results.append(result)\n\n# Special test: Mixed basket (if available)\nprint(\"\\n\" + \"=\"*60)\nprint(\"👗 TESTING MIXED PRODUCT TYPE BASKET\")\nprint(\"=\"*60)\n\n# Try to find a basket with mixed product types\nmixed_basket_found = False\n\nfor idx, basket in usable_baskets.iterrows():\n    basket_items = basket['article_id']\n    \n    # Get product types for this basket\n    basket_details = articles_clean[articles_clean['article_id'].isin(basket_items)]\n    \n    if len(basket_details) >= 3:\n        unique_types = basket_details['product_type_name'].nunique()\n        \n        if unique_types >= 2:  # Basket has mixed types\n            print(f\"🎯 Found mixed basket with {unique_types} different product types\")\n            test_real_basket(basket_items, basket_number=\"MIXED\")\n            mixed_basket_found = True\n            break\n\nif not mixed_basket_found:\n    print(\"⚠️ No mixed basket found in sample\")\n    print(\"Creating an artificial mixed basket...\")\n    \n    # Find items of different types\n    jeans_items = articles_clean[articles_clean['product_type_name'].str.contains('jean', case=False, na=False)]\n    dress_items = articles_clean[articles_clean['product_type_name'].str.contains('dress', case=False, na=False)]\n    top_items = articles_clean[articles_clean['product_type_name'].str.contains('top|shirt', case=False, na=False)]\n    \n    mixed_basket = []\n    if len(jeans_items) > 0:\n        mixed_basket.append(jeans_items.iloc[0]['article_id'])\n    if len(dress_items) > 0:\n        mixed_basket.append(dress_items.iloc[0]['article_id'])\n    if len(top_items) > 0:\n        mixed_basket.append(top_items.iloc[0]['article_id'])\n    \n    if len(mixed_basket) >= 2:\n        test_real_basket(mixed_basket, basket_number=\"ARTIFICIAL MIXED\")\n\n# Summary statistics\nprint(\"\\n\" + \"=\"*60)\nprint(\"📊 TEST RESULTS SUMMARY\")\nprint(\"=\"*60)\n\nif all_results:\n    total_recommendations = sum(len(r) for r in all_results)\n    print(f\"Total tests run: {len(all_results)}\")\n    print(f\"Total recommendations generated: {total_recommendations}\")\n    \n    # Check what types of baskets we tested\n    print(f\"\\n🧺 BASKETS TESTED:\")\n    for i, basket_items in enumerate(test_baskets):\n        basket_details = articles_clean[articles_clean['article_id'].isin(basket_items)]\n        types = basket_details['product_type_name'].unique() if len(basket_details) > 0 else [\"Unknown\"]\n        print(f\"  Basket {i+1}: {len(basket_items)} items, Types: {', '.join(types[:3])}\")\n    \n    print(f\"\\n🎯 GOAL: 'More Like This' for each basket\")\n    print(f\"   • Jeans basket → more jeans\")\n    print(f\"   • Dress basket → more dresses\")\n    print(f\"   • Mixed basket → mix of relevant items\")\nelse:\n    print(\"⚠️ No test results available\")\n\n# Bonus: Show a large basket\nprint(\"\\n\" + \"=\"*60)\nprint(\"📦 BONUS: TESTING LARGER BASKET (4+ items)\")\nprint(\"=\"*60)\n\n# Find a larger basket\nlarge_baskets = baskets[baskets['basket_size'] >= 4]\n\nfor idx, basket in large_baskets.iterrows():\n    basket_items = basket['article_id']\n    \n    # Check how many items exist\n    existing_items = [item for item in basket_items if item in articles_clean['article_id'].values]\n    \n    if len(existing_items) >= 4:\n        print(f\"\\n🧺 Testing large basket with {len(existing_items)} items\")\n        test_real_basket(existing_items, basket_number=\"LARGE\")\n        break\n\nprint(\"\\n✅ TESTING COMPLETE!\")\nprint(\"You can now see if your algorithm works on REAL customer shopping baskets!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-28T20:00:26.069078Z","iopub.execute_input":"2026-01-28T20:00:26.069898Z","iopub.status.idle":"2026-01-28T20:00:46.200915Z","shell.execute_reply.started":"2026-01-28T20:00:26.069866Z","shell.execute_reply":"2026-01-28T20:00:46.199863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}