{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ==================================================================================\n# H&M GRAPH ANALYTICS - NETWORKX VERSION \n# ==================================================================================\n\nprint(\"=\"*100)\nprint(\" H&M PERSONALIZED FASHION - GRAPH ANALYTICS WITH NETWORKX\")\nprint(\"=\"*100)\n\n# ==================================================================================\n# STEP 1: INITIALIZE SPARK & IMPORTS\n# ==================================================================================\nprint(\"\\n[STEP 1] INITIALIZING ENVIRONMENT\")\nprint(\"-\"*100)\n\nfrom pyspark.sql import SparkSession\nfrom pyspark.sql import functions as F\nfrom datetime import datetime, timedelta\nimport pandas as pd\nimport numpy as np\nimport networkx as nx\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Initialize Spark (simple, no GraphFrames)\nspark = SparkSession.builder \\\n    .appName(\"HM-GraphAnalytics-NetworkX\") \\\n    .config(\"spark.driver.memory\", \"10g\") \\\n    .getOrCreate()\n\nprint(f\"✓ Spark Session initialized (Version: {spark.version})\")\nprint(f\"✓ NetworkX version: {nx.__version__}\")\n\n# ==================================================================================\n# STEP 2: LOAD & FILTER DATA\n# ==================================================================================\nprint(\"\\n[STEP 2] LOADING TRANSACTION DATA\")\nprint(\"-\"*100)\n\nBASE_PATH = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/\"\nTRANSACTIONS_PATH = BASE_PATH + \"transactions_train.csv\"\n\ntransactions_raw = spark.read.csv(TRANSACTIONS_PATH, header=True, inferSchema=True)\ntotal_rows = transactions_raw.count()\nprint(f\"✓ Raw transactions loaded: {total_rows:,} rows\")\n\n# Filter last 6 months\nlast_date = transactions_raw.agg(F.max(\"t_dat\")).collect()[0][0]\nsix_months_ago = last_date - timedelta(days=180)\n\ntransactions_filtered = transactions_raw.filter(F.col(\"t_dat\") >= F.lit(six_months_ago))\nprint(f\"✓ Filtered to last 6 months: {transactions_filtered.count():,} rows\")\n\n# Sample top 3000 customers\nprint(\"  → Sampling top 3000 active customers for graph analysis...\")\ntop_customers = (\n    transactions_filtered\n    .groupBy(\"customer_id\")\n    .agg(F.count(\"*\").alias(\"txn_count\"))\n    .orderBy(F.desc(\"txn_count\"))\n    .limit(3000)\n    .select(\"customer_id\")\n)\n\nsampled_transactions = transactions_filtered.join(top_customers, on=\"customer_id\", how=\"inner\")\nprint(f\"✓ Sampled dataset: {sampled_transactions.count():,} transactions\")\n\nprint(\"\\n✓ Data preparation complete!\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-02-05T12:02:06.894389Z","iopub.execute_input":"2026-02-05T12:02:06.894987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==================================================================================\n# STEP 3: BUILD BIPARTITE GRAPH WITH NETWORKX\n# ==================================================================================\nprint(\"\\n[STEP 3] BUILDING BIPARTITE GRAPH (CUSTOMERS ↔ PRODUCTS)\")\nprint(\"-\"*100)\n\n# Aggregate edges\nprint(\"  → Aggregating customer-product relationships...\")\nedges_df = (\n    sampled_transactions\n    .groupBy(\"customer_id\", \"article_id\")\n    .agg(\n        F.count(\"*\").alias(\"purchase_count\"),\n        F.sum(\"price\").alias(\"total_amount\"),\n        F.max(\"t_dat\").alias(\"last_purchase_date\")\n    )\n)\n\n# Convert to Pandas (limit for memory efficiency)\nprint(\"  → Converting to Pandas (limiting to 100K edges for performance)...\")\nedges_pd = edges_df.limit(100000).toPandas()\nprint(f\"✓ Loaded {len(edges_pd):,} edges into memory\")\n\n# Build NetworkX Bipartite Graph\nprint(\"  → Building NetworkX bipartite graph...\")\nG = nx.Graph()\n\n# Add edges (customer -> product)\nfor _, row in edges_pd.iterrows():\n    customer = row['customer_id']\n    product = str(row['article_id'])\n    \n    # Add nodes with type attribute\n    G.add_node(customer, node_type='customer')\n    G.add_node(product, node_type='product')\n    \n    # Add edge with attributes\n    G.add_edge(customer, product,\n               purchase_count=row['purchase_count'],\n               total_amount=row['total_amount'],\n               last_purchase_date=row['last_purchase_date'])\n\n# Get node counts\ncustomers_set = {n for n, d in G.nodes(data=True) if d.get('node_type') == 'customer'}\nproducts_set = {n for n, d in G.nodes(data=True) if d.get('node_type') == 'product'}\n\nnum_customers = len(customers_set)\nnum_products = len(products_set)\nnum_vertices = G.number_of_nodes()\nnum_edges = G.number_of_edges()\n\nprint(f\"\\n✓ Bipartite Graph successfully built!\")\nprint(f\"  ┌─ Total Nodes: {num_vertices:,}\")\nprint(f\"  ├─ Customers: {num_customers:,}\")\nprint(f\"  ├─ Products: {num_products:,}\")\nprint(f\"  ├─ Total Edges: {num_edges:,}\")\nprint(f\"  └─ Graph Density: {(num_edges / (num_customers * num_products)):.6f}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==================================================================================\n# STEP 4: BASIC GRAPH QUERIES & METRICS\n# ==================================================================================\nprint(\"\\n[STEP 4] BASIC GRAPH QUERIES\")\nprint(\"-\"*100)\n\n# Query 1: Degree Analysis\nprint(\"\\n📊 QUERY 1: NODE DEGREE ANALYSIS\")\nprint(\"-\"*50)\n\n# Calculate degrees\ndegree_dict = dict(G.degree())\n\n# Top customers by degree\ncustomer_degrees = {n: d for n, d in degree_dict.items() if n in customers_set}\ntop_customers_deg = sorted(customer_degrees.items(), key=lambda x: x[1], reverse=True)[:10]\n\nprint(\"Top 10 Customers (Most Products Purchased):\")\nfor i, (customer, degree) in enumerate(top_customers_deg, 1):\n    print(f\"  {i:2d}. {customer[:20]:20s} → {degree:3d} products\")\n\n# Top products by degree\nproduct_degrees = {n: d for n, d in degree_dict.items() if n in products_set}\ntop_products_deg = sorted(product_degrees.items(), key=lambda x: x[1], reverse=True)[:10]\n\nprint(\"\\nTop 10 Products (Most Customers):\")\nfor i, (product, degree) in enumerate(top_products_deg, 1):\n    print(f\"  {i:2d}. Article {product:10s} → {degree:3d} customers\")\n\n# Query 2: Graph Statistics\nprint(\"\\n📊 QUERY 2: GRAPH STATISTICS\")\nprint(\"-\"*50)\n\navg_degree = sum(degree_dict.values()) / len(degree_dict)\nmax_degree = max(degree_dict.values())\nmin_degree = min(degree_dict.values())\n\ntotal_purchases = sum(G[u][v]['purchase_count'] for u, v in G.edges())\ntotal_revenue = sum(G[u][v]['total_amount'] for u, v in G.edges())\n\nprint(f\"  Average Degree        : {avg_degree:.2f}\")\nprint(f\"  Max Degree            : {max_degree}\")\nprint(f\"  Min Degree            : {min_degree}\")\nprint(f\"  Total Purchases       : {total_purchases:,}\")\nprint(f\"  Total Revenue         : ${total_revenue:,.2f}\")\nprint(f\"  Avg Revenue per Edge  : ${total_revenue/num_edges:,.2f}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==================================================================================\n# STEP 5: GRAPH PATTERN MATCHING (MOTIF FINDING)\n# ==================================================================================\nprint(\"\\n[STEP 5] GRAPH PATTERN MATCHING - SIMULATED GRAPH QUERIES\")\nprint(\"-\"*100)\n\n# Pattern 1: Co-Purchase Analysis (Customer1 → Product ← Customer2)\nprint(\"\\n🔍 PATTERN 1: CO-PURCHASE ANALYSIS\")\nprint(\"   Query Pattern: (Customer1)-[:PURCHASED]->(Product)<-[:PURCHASED]-(Customer2)\")\nprint(\"-\"*50)\n\nprint(\"  → Finding co-purchase patterns...\")\ncopurchase_patterns = []\n\n# Sample products to analyze (top 50 for speed)\nsample_products = [p for p, _ in top_products_deg[:50]]\n\nfor product in sample_products:\n    # Get all customers who bought this product\n    customers_bought = [n for n in G.neighbors(product) if n in customers_set]\n    \n    # Find all pairs of customers who bought the same product\n    if len(customers_bought) >= 2:\n        for i, c1 in enumerate(customers_bought):\n            for c2 in customers_bought[i+1:]:\n                copurchase_patterns.append({\n                    'customer1': c1,\n                    'customer2': c2,\n                    'product': product,\n                    'c1_purchases': G[c1][product]['purchase_count'],\n                    'c2_purchases': G[c2][product]['purchase_count'],\n                    'total_score': G[c1][product]['purchase_count'] + G[c2][product]['purchase_count']\n                })\n\ncopurchase_df = pd.DataFrame(copurchase_patterns).sort_values('total_score', ascending=False)\n\nprint(f\"✓ Found {len(copurchase_df):,} co-purchase patterns\")\nprint(\"\\nTop 10 Co-Purchase Patterns:\")\nprint(copurchase_df.head(10).to_string(index=False))\n\n# Pattern 2: Product Association (Frequently Bought Together)\nprint(\"\\n\\n🔍 PATTERN 2: PRODUCT ASSOCIATIONS (Frequently Bought Together)\")\nprint(\"   Query Pattern: (Customer)-[:PURCHASED]->(Product1), (Customer)-[:PURCHASED]->(Product2)\")\nprint(\"-\"*50)\n\nprint(\"  → Finding product associations...\")\nproduct_associations = {}\n\n# Sample customers (top 500 for speed)\nsample_customers = [c for c, _ in top_customers_deg[:500]]\n\nfor customer in sample_customers:\n    # Get all products bought by this customer\n    products_bought = [n for n in G.neighbors(customer) if n in products_set]\n    \n    # Find all pairs of products bought by same customer\n    if len(products_bought) >= 2:\n        for i, p1 in enumerate(products_bought):\n            for p2 in products_bought[i+1:]:\n                pair = tuple(sorted([p1, p2]))\n                if pair not in product_associations:\n                    product_associations[pair] = 0\n                product_associations[pair] += 1\n\n# Sort by frequency\nsorted_associations = sorted(product_associations.items(), key=lambda x: x[1], reverse=True)[:10]\n\nprint(f\"✓ Found {len(product_associations):,} unique product pairs\")\nprint(\"\\nTop 10 Product Pairs (Frequently Bought Together):\")\nfor i, ((p1, p2), count) in enumerate(sorted_associations, 1):\n    print(f\"  {i:2d}. {p1:10s} + {p2:10s} → {count:3d} co-occurrences\")\n\n# Pattern 3: Customer Similarity\nprint(\"\\n\\n🔍 PATTERN 3: CUSTOMER SIMILARITY (Shared Product Preferences)\")\nprint(\"   Finding customers with overlapping purchase patterns\")\nprint(\"-\"*50)\n\nprint(\"  → Calculating customer similarity...\")\ncustomer_similarity = []\n\n# Sample pairs from copurchase patterns\nunique_pairs = copurchase_df[['customer1', 'customer2']].drop_duplicates()\n\nfor _, row in unique_pairs.head(1000).iterrows():\n    c1, c2 = row['customer1'], row['customer2']\n    \n    # Get products bought by each\n    c1_products = set(n for n in G.neighbors(c1) if n in products_set)\n    c2_products = set(n for n in G.neighbors(c2) if n in products_set)\n    \n    # Calculate shared products\n    shared = c1_products & c2_products\n    \n    if len(shared) > 0:\n        customer_similarity.append({\n            'customer1': c1,\n            'customer2': c2,\n            'shared_products': len(shared),\n            'similarity_score': len(shared) / (len(c1_products) + len(c2_products) - len(shared))  # Jaccard\n        })\n\nsimilarity_df = pd.DataFrame(customer_similarity).sort_values('shared_products', ascending=False)\n\nprint(f\"✓ Found {len(similarity_df):,} customer pairs with shared products\")\nprint(\"\\nTop 10 Most Similar Customer Pairs:\")\nprint(similarity_df.head(10).to_string(index=False))\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==================================================================================\n# STEP 6: ADVANCED GRAPH ALGORITHMS\n# ==================================================================================\nprint(\"\\n[STEP 6] ADVANCED GRAPH ALGORITHMS\")\nprint(\"-\"*100)\n\n# Degree Centrality\nprint(\"\\n📊 DEGREE CENTRALITY\")\nprint(\"-\"*50)\ndegree_centrality = nx.degree_centrality(G)\ntop_central = sorted(degree_centrality.items(), key=lambda x: x[1], reverse=True)[:5]\n\nprint(\"Top 5 Most Central Nodes:\")\nfor node, centrality in top_central:\n    node_type = \"Customer\" if node in customers_set else \"Product\"\n    print(f\"  {node_type:8s} {str(node)[:20]:20s} → Centrality: {centrality:.4f}\")\n\n# Connected Components\nprint(\"\\n📊 CONNECTED COMPONENTS ANALYSIS\")\nprint(\"-\"*50)\ncomponents = list(nx.connected_components(G))\nnum_components = len(components)\nlargest_component_size = len(max(components, key=len))\n\nprint(f\"  Number of Connected Components : {num_components}\")\nprint(f\"  Largest Component Size        : {largest_component_size:,} nodes ({largest_component_size/num_vertices*100:.1f}%)\")\n\n# Component size distribution\ncomponent_sizes = sorted([len(c) for c in components], reverse=True)[:5]\nprint(f\"  Top 5 Component Sizes         : {component_sizes}\")\n\n# Clustering Coefficient (on sample for speed)\nprint(\"\\n📊 CLUSTERING COEFFICIENT (Sample)\")\nprint(\"-\"*50)\nsample_nodes = list(customers_set)[:100]\nclustering_dict = nx.clustering(G, nodes=sample_nodes)\navg_clustering = sum(clustering_dict.values()) / len(clustering_dict)\n\nprint(f\"  Average Clustering Coefficient: {avg_clustering:.6f}\")\nprint(f\"  (Calculated on {len(sample_nodes)} sample nodes for performance)\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==================================================================================\n# STEP 7: EXPORT RESULTS\n# ==================================================================================\nprint(\"\\n[STEP 7] EXPORTING RESULTS\")\nprint(\"-\"*100)\n\n# Export graph structure\nprint(\"  → Exporting graph nodes...\")\nnodes_export = pd.DataFrame([\n    {'id': node, 'type': 'customer' if node in customers_set else 'product', 'degree': G.degree(node)}\n    for node in G.nodes()\n])\nnodes_export.to_csv('/kaggle/working/graph_nodes.csv', index=False)\nprint(f\"✓ Exported {len(nodes_export):,} nodes\")\n\n# Export edges\nprint(\"  → Exporting graph edges...\")\nedges_export = pd.DataFrame([\n    {'source': u, 'target': v, \n     'purchase_count': G[u][v]['purchase_count'],\n     'total_amount': G[u][v]['total_amount']}\n    for u, v in list(G.edges())[:5000]  # Limit for file size\n])\nedges_export.to_csv('/kaggle/working/graph_edges.csv', index=False)\nprint(f\"✓ Exported {len(edges_export):,} edges\")\n\n# Export patterns\ncopurchase_df.head(1000).to_csv('/kaggle/working/copurchase_patterns.csv', index=False)\nprint(\"✓ Exported copurchase_patterns.csv\")\n\npd.DataFrame(sorted_associations, columns=['product_pair', 'co_occurrence']).to_csv(\n    '/kaggle/working/product_associations.csv', index=False)\nprint(\"✓ Exported product_associations.csv\")\n\nprint(\"\\n✅ All exports completed!\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==================================================================================\n# STEP 8: GRAPH VISUALIZATIONS\n# ==================================================================================\nprint(\"\\n[STEP 8] GENERATING GRAPH VISUALIZATIONS\")\nprint(\"=\"*100)\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas as pd\nimport numpy as np\n\n# Set style\nsns.set_style(\"whitegrid\")\nplt.rcParams['figure.facecolor'] = 'white'\nplt.rcParams['font.size'] = 10\n\n# ==================================================================================\n# VISUALIZATION 1: DEGREE DISTRIBUTION\n# ==================================================================================\nprint(\"\\n📊 Visualization 1: Degree Distribution...\")\n\nfig, axes = plt.subplots(1, 2, figsize=(16, 6))\n\n# Customer degree distribution\ncustomer_degrees_list = [degree_dict[c] for c in customers_set]\naxes[0].hist(customer_degrees_list, bins=30, color='#3498db', edgecolor='black', alpha=0.7)\naxes[0].set_xlabel('Number of Products Purchased', fontsize=12, fontweight='bold')\naxes[0].set_ylabel('Number of Customers', fontsize=12, fontweight='bold')\naxes[0].set_title('Customer Degree Distribution\\n(Purchase Diversity)', fontsize=14, fontweight='bold')\naxes[0].grid(alpha=0.3)\naxes[0].text(0.98, 0.97, \n             f'Mean: {np.mean(customer_degrees_list):.1f}\\nStd: {np.std(customer_degrees_list):.1f}\\nMax: {max(customer_degrees_list)}', \n             transform=axes[0].transAxes, fontsize=11, verticalalignment='top', horizontalalignment='right',\n             bbox=dict(boxstyle='round', facecolor='wheat', alpha=0.8))\n\n# Product degree distribution\nproduct_degrees_list = [degree_dict[p] for p in products_set]\naxes[1].hist(product_degrees_list, bins=30, color='#e74c3c', edgecolor='black', alpha=0.7)\naxes[1].set_xlabel('Number of Customers', fontsize=12, fontweight='bold')\naxes[1].set_ylabel('Number of Products', fontsize=12, fontweight='bold')\naxes[1].set_title('Product Degree Distribution\\n(Product Popularity)', fontsize=14, fontweight='bold')\naxes[1].grid(alpha=0.3)\naxes[1].text(0.98, 0.97, \n             f'Mean: {np.mean(product_degrees_list):.1f}\\nStd: {np.std(product_degrees_list):.1f}\\nMax: {max(product_degrees_list)}', \n             transform=axes[1].transAxes, fontsize=11, verticalalignment='top', horizontalalignment='right',\n             bbox=dict(boxstyle='round', facecolor='lightblue', alpha=0.8))\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/degree_distribution.png', dpi=300, bbox_inches='tight')\nplt.show()\nprint(\"✓ Saved: degree_distribution.png\")\n\n# ==================================================================================\n# VISUALIZATION 2: TOP CUSTOMERS & PRODUCTS (BAR CHARTS)\n# ==================================================================================\nprint(\"\\n📊 Visualization 2: Top Nodes Analysis...\")\n\nfig, axes = plt.subplots(1, 2, figsize=(18, 7))\n\n# Top 15 customers\ntop15_customers = top_customers_deg[:15]\ncustomer_labels = [f\"C{i+1}\" for i in range(len(top15_customers))]\ncustomer_degrees_top = [deg for _, deg in top15_customers]\n\naxes[0].barh(range(len(top15_customers)), customer_degrees_top, \n             color='#3498db', edgecolor='black', linewidth=1.5)\naxes[0].set_yticks(range(len(top15_customers)))\naxes[0].set_yticklabels(customer_labels, fontsize=10)\naxes[0].set_xlabel('Number of Products Purchased', fontsize=12, fontweight='bold')\naxes[0].set_title('Top 15 Customers by Degree\\n(Most Diverse Shoppers)', fontsize=14, fontweight='bold')\naxes[0].invert_yaxis()\naxes[0].grid(axis='x', alpha=0.3)\nfor i, v in enumerate(customer_degrees_top):\n    axes[0].text(v + 0.5, i, str(v), va='center', fontsize=9, fontweight='bold')\n\n# Top 15 products\ntop15_products = top_products_deg[:15]\nproduct_labels = [f\"P{i+1}\" for i in range(len(top15_products))]\nproduct_degrees_top = [deg for _, deg in top15_products]\n\naxes[1].barh(range(len(top15_products)), product_degrees_top, \n             color='#e74c3c', edgecolor='black', linewidth=1.5)\naxes[1].set_yticks(range(len(top15_products)))\naxes[1].set_yticklabels(product_labels, fontsize=10)\naxes[1].set_xlabel('Number of Customers', fontsize=12, fontweight='bold')\naxes[1].set_title('Top 15 Products by Degree\\n(Most Popular Items)', fontsize=14, fontweight='bold')\naxes[1].invert_yaxis()\naxes[1].grid(axis='x', alpha=0.3)\nfor i, v in enumerate(product_degrees_top):\n    axes[1].text(v + 0.5, i, str(v), va='center', fontsize=9, fontweight='bold')\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/top_nodes_analysis.png', dpi=300, bbox_inches='tight')\nplt.show()\nprint(\"✓ Saved: top_nodes_analysis.png\")\n\n# ==================================================================================\n# VISUALIZATION 3: CO-PURCHASE NETWORK SAMPLE\n# ==================================================================================\nprint(\"\\n📊 Visualization 3: Co-Purchase Network Sample...\")\n\nfig, ax = plt.subplots(figsize=(14, 10))\n\n# Build sample graph from top co-purchase patterns\nG_copurchase = nx.Graph()\ncopurchase_sample = copurchase_df.head(100)\n\nfor _, row in copurchase_sample.iterrows():\n    c1, c2, p = row['customer1'], row['customer2'], row['product']\n    G_copurchase.add_node(c1, node_type='customer')\n    G_copurchase.add_node(c2, node_type='customer')\n    G_copurchase.add_node(p, node_type='product')\n    G_copurchase.add_edge(c1, p, weight=row['c1_purchases'])\n    G_copurchase.add_edge(c2, p, weight=row['c2_purchases'])\n\n# Separate nodes\ncustomer_nodes_viz = [n for n, d in G_copurchase.nodes(data=True) if d.get('node_type') == 'customer']\nproduct_nodes_viz = [n for n, d in G_copurchase.nodes(data=True) if d.get('node_type') == 'product']\n\n# Layout\npos = nx.spring_layout(G_copurchase, k=0.5, iterations=50, seed=42)\n\n# Draw\nnx.draw_networkx_nodes(G_copurchase, pos, nodelist=customer_nodes_viz, \n                       node_color='#3498db', node_size=150, alpha=0.8, label='Customers', ax=ax)\nnx.draw_networkx_nodes(G_copurchase, pos, nodelist=product_nodes_viz, \n                       node_color='#e74c3c', node_size=250, alpha=0.9, label='Products', ax=ax)\nnx.draw_networkx_edges(G_copurchase, pos, width=0.5, alpha=0.3, ax=ax)\n\nax.set_title('Co-Purchase Network Sample\\n(Customers sharing product preferences)', \n             fontsize=14, fontweight='bold', pad=20)\nax.legend(loc='upper right', fontsize=12, framealpha=0.9)\nax.axis('off')\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/copurchase_network.png', dpi=300, bbox_inches='tight')\nplt.show()\nprint(\"✓ Saved: copurchase_network.png\")\n\n# ==================================================================================\n# VISUALIZATION 4: PRODUCT ASSOCIATIONS HEATMAP\n# ==================================================================================\nprint(\"\\n📊 Visualization 4: Product Association Heatmap...\")\n\n# Get top 15 product pairs for heatmap\ntop_associations_data = sorted_associations[:15]\nproducts_in_assoc = list(set([p for pair, _ in top_associations_data for p in pair]))[:12]  # Limit to 12 for readability\n\n# Build association matrix\nn_prod = len(products_in_assoc)\nassoc_matrix = np.zeros((n_prod, n_prod))\nprod_to_idx = {p: i for i, p in enumerate(products_in_assoc)}\n\nfor (p1, p2), count in top_associations_data:\n    if p1 in prod_to_idx and p2 in prod_to_idx:\n        i, j = prod_to_idx[p1], prod_to_idx[p2]\n        assoc_matrix[i, j] = count\n        assoc_matrix[j, i] = count\n\n# Plot\nfig, ax = plt.subplots(figsize=(12, 10))\nproduct_labels_hm = [f\"P{i+1}\" for i in range(n_prod)]\n\nsns.heatmap(assoc_matrix, annot=True, fmt='.0f', cmap='YlOrRd', \n            xticklabels=product_labels_hm, yticklabels=product_labels_hm,\n            linewidths=0.5, cbar_kws={'label': 'Co-occurrence Count'}, ax=ax)\n\nax.set_title('Product Association Heatmap\\n(Frequently Bought Together - Top Products)', \n             fontsize=14, fontweight='bold', pad=20)\nax.set_xlabel('Product', fontsize=12, fontweight='bold')\nax.set_ylabel('Product', fontsize=12, fontweight='bold')\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/product_association_heatmap.png', dpi=300, bbox_inches='tight')\nplt.show()\nprint(\"✓ Saved: product_association_heatmap.png\")\n\n# ==================================================================================\n# VISUALIZATION 5: GRAPH METRICS SUMMARY (4-PANEL DASHBOARD)\n# ==================================================================================\nprint(\"\\n📊 Visualization 5: Graph Metrics Summary Dashboard...\")\n\nfig = plt.figure(figsize=(16, 12))\ngs = fig.add_gridspec(2, 2, hspace=0.3, wspace=0.3)\n\n# Panel 1: Node Type Distribution (Pie Chart)\nax1 = fig.add_subplot(gs[0, 0])\nnode_type_counts = [num_customers, num_products]\nnode_type_labels = [f'Customers\\n({num_customers:,})', f'Products\\n({num_products:,})']\ncolors_pie = ['#3498db', '#e74c3c']\n\nax1.pie(node_type_counts, labels=node_type_labels, autopct='%1.1f%%', \n        colors=colors_pie, startangle=90, textprops={'fontsize': 11, 'fontweight': 'bold'})\nax1.set_title('Node Type Distribution', fontsize=14, fontweight='bold', pad=20)\n\n# Panel 2: Edge Statistics (Bar Chart)\nax2 = fig.add_subplot(gs[0, 1])\nedge_metrics_names = ['Unique\\nPairs', 'Total\\nPurchases', 'Total\\nRevenue\\n(x$1000)']\nedge_metrics_values = [num_edges, total_purchases, total_revenue/1000]\n\nbars2 = ax2.bar(edge_metrics_names, edge_metrics_values, \n                color=['#16a085', '#f39c12', '#9b59b6'], edgecolor='black', linewidth=1.5)\nax2.set_ylabel('Count / Amount', fontsize=11, fontweight='bold')\nax2.set_title('Transaction Metrics', fontsize=14, fontweight='bold', pad=20)\nax2.grid(axis='y', alpha=0.3)\n\nfor bar, val in zip(bars2, edge_metrics_values):\n    height = bar.get_height()\n    ax2.text(bar.get_x() + bar.get_width()/2., height + max(edge_metrics_values)*0.02,\n             f'{val:,.0f}', ha='center', va='bottom', fontsize=10, fontweight='bold')\n\n# Panel 3: Degree Distribution Boxplot\nax3 = fig.add_subplot(gs[1, 0])\ndegree_data = [customer_degrees_list, product_degrees_list]\nbp = ax3.boxplot(degree_data, labels=['Customers', 'Products'], patch_artist=True,\n                 boxprops=dict(facecolor='#3498db', alpha=0.7),\n                 medianprops=dict(color='red', linewidth=2),\n                 whiskerprops=dict(linewidth=1.5),\n                 capprops=dict(linewidth=1.5))\n\n# Color boxes differently\nbp['boxes'][0].set_facecolor('#3498db')\nbp['boxes'][1].set_facecolor('#e74c3c')\n\nax3.set_ylabel('Degree', fontsize=11, fontweight='bold')\nax3.set_title('Degree Distribution Comparison', fontsize=14, fontweight='bold', pad=20)\nax3.grid(axis='y', alpha=0.3)\n\n# Panel 4: Pattern Counts (Bar Chart)\nax4 = fig.add_subplot(gs[1, 1])\npattern_names = ['Co-Purchase\\nPatterns', 'Product\\nAssociations', 'Customer\\nSimilarities']\npattern_counts_list = [len(copurchase_df), len(product_associations), len(similarity_df)]\n\nbars4 = ax4.bar(pattern_names, pattern_counts_list, \n                color=['#e67e22', '#27ae60', '#8e44ad'], edgecolor='black', linewidth=1.5)\nax4.set_ylabel('Count', fontsize=11, fontweight='bold')\nax4.set_title('Graph Pattern Matching Results', fontsize=14, fontweight='bold', pad=20)\nax4.grid(axis='y', alpha=0.3)\n\nfor bar in bars4:\n    height = bar.get_height()\n    ax4.text(bar.get_x() + bar.get_width()/2., height + max(pattern_counts_list)*0.02,\n             f'{int(height):,}', ha='center', va='bottom', fontsize=10, fontweight='bold')\n\nplt.savefig('/kaggle/working/graph_metrics_summary.png', dpi=300, bbox_inches='tight')\nplt.show()\nprint(\"✓ Saved: graph_metrics_summary.png\")\n\n# ==================================================================================\n# VISUALIZATION 6: BIPARTITE NETWORK (TOP CUSTOMERS ↔ PRODUCTS)\n# ==================================================================================\nprint(\"\\n📊 Visualization 6: Bipartite Network Sample...\")\n\nfig, ax = plt.subplots(figsize=(16, 10))\n\n# Build bipartite graph for top 10 customers\ntop10_customers_nodes = [c for c, _ in top_customers_deg[:10]]\nG_bipartite = nx.Graph()\n\nfor customer in top10_customers_nodes:\n    products_bought = [n for n in G.neighbors(customer) if n in products_set]\n    for product in products_bought[:10]:  # Max 10 products per customer for clarity\n        G_bipartite.add_node(customer, bipartite=0)\n        G_bipartite.add_node(product, bipartite=1)\n        G_bipartite.add_edge(customer, product)\n\n# Get node sets\ncustomers_bp = {n for n, d in G_bipartite.nodes(data=True) if d.get('bipartite') == 0}\nproducts_bp = {n for n, d in G_bipartite.nodes(data=True) if d.get('bipartite') == 1}\n\n# Bipartite layout\npos_bp = {}\ncustomers_bp_list = list(customers_bp)\nproducts_bp_list = list(products_bp)\n\n# Position customers on left (x=0)\ny_spacing_c = 10 / max(1, len(customers_bp_list) - 1) if len(customers_bp_list) > 1 else 1\nfor i, node in enumerate(customers_bp_list):\n    pos_bp[node] = (0, i * y_spacing_c)\n\n# Position products on right (x=3)\ny_spacing_p = 10 / max(1, len(products_bp_list) - 1) if len(products_bp_list) > 1 else 1\nfor i, node in enumerate(products_bp_list):\n    pos_bp[node] = (3, i * y_spacing_p)\n\n# Draw\nnx.draw_networkx_nodes(G_bipartite, pos_bp, nodelist=customers_bp_list, \n                       node_color='#3498db', node_size=500, alpha=0.8, label='Customers (Top 10)', ax=ax)\nnx.draw_networkx_nodes(G_bipartite, pos_bp, nodelist=products_bp_list, \n                       node_color='#e74c3c', node_size=300, alpha=0.8, label='Products', ax=ax)\nnx.draw_networkx_edges(G_bipartite, pos_bp, width=1, alpha=0.2, ax=ax)\n\n# Add labels for top 3 customers\nlabels_bp = {node: f\"C{i+1}\" for i, node in enumerate(customers_bp_list[:3])}\nnx.draw_networkx_labels(G_bipartite, pos_bp, labels=labels_bp, font_size=10, font_weight='bold', ax=ax)\n\nax.set_title('Bipartite Network: Top 10 Customers ↔ Products\\n(Customer-Product Purchase Relationships)', \n             fontsize=14, fontweight='bold', pad=20)\nax.legend(loc='upper right', fontsize=12, framealpha=0.9, markerscale=1.5)\nax.axis('off')\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/bipartite_network.png', dpi=300, bbox_inches='tight')\nplt.show()\nprint(\"✓ Saved: bipartite_network.png\")\n\n# ==================================================================================\n# FINAL SUMMARY\n# ==================================================================================\nprint(\"\\n\" + \"=\"*100)\nprint(\"✅ ALL 6 VISUALIZATIONS GENERATED SUCCESSFULLY!\")\nprint(\"=\"*100)\n\nprint(f\"\"\"\n📊 EXPORTED VISUALIZATIONS:\n   1. degree_distribution.png          - Customer & product degree histograms\n   2. top_nodes_analysis.png           - Top 15 customers & products bar charts\n   3. copurchase_network.png           - Co-purchase network graph (spring layout)\n   4. product_association_heatmap.png  - Frequently bought together matrix\n   5. graph_metrics_summary.png        - 4-panel metrics dashboard\n   6. bipartite_network.png            - Customer-product bipartite graph\n\n📁 EXPORTED DATA FILES:\n   ✓ graph_nodes.csv          ({len(nodes_export):,} nodes)\n   ✓ graph_edges.csv          ({len(edges_export):,} edges)\n   ✓ copurchase_patterns.csv  (Top 1,000 co-purchase patterns)\n   ✓ product_associations.csv (Product pair frequencies)\n\n💾 All files saved to: /kaggle/working/\n   → Download from Output panel (right sidebar)\n\n🎯 READY FOR REPORT!\n\"\"\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==================================================================================\n# STEP 8: NEO4J-STYLE GRAPH VISUALIZATIONS (WITH LABELS & RELATIONSHIPS)\n# ==================================================================================\nprint(\"\\n[STEP 8] GENERATING NEO4J-STYLE GRAPH VISUALIZATIONS\")\nprint(\"=\"*100)\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas as pd\nimport numpy as np\nimport networkx as nx\nfrom matplotlib.patches import FancyBboxPatch, FancyArrowPatch\nimport matplotlib.patches as mpatches\n\nsns.set_style(\"white\")\nplt.rcParams['figure.facecolor'] = 'white'\n\n# ==================================================================================\n# VISUALIZATION 1: LABELED PURCHASE GRAPH (Neo4j Style)\n# ==================================================================================\nprint(\"\\n📊 Visualization 1: Labeled Purchase Graph (Neo4j Style)...\")\n\nfig, ax = plt.subplots(figsize=(16, 12))\n\n# Sample: 1 customer, their top 5 products\nsample_customer = top_customers_deg[0][0]  # Top customer\nsample_products = [n for n in G.neighbors(sample_customer) if n in products_set][:5]\n\n# Build small graph\nG_sample = nx.DiGraph()  # Directed graph for clear arrows\nG_sample.add_node(sample_customer, node_type='customer', label=f\"Customer_{sample_customer[:8]}\")\n\nfor i, product in enumerate(sample_products):\n    product_label = f\"Product_{product}\"\n    G_sample.add_node(product, node_type='product', label=product_label)\n    # Add directed edge with relationship\n    G_sample.add_edge(sample_customer, product, \n                     relationship='PURCHASED',\n                     count=G[sample_customer][product]['purchase_count'])\n\n# Manual layout for clarity (star pattern)\npos_manual = {}\npos_manual[sample_customer] = (0, 0)  # Center\n\n# Arrange products in circle\nangle_step = 2 * np.pi / len(sample_products)\nradius = 3\nfor i, product in enumerate(sample_products):\n    angle = i * angle_step\n    x = radius * np.cos(angle)\n    y = radius * np.sin(angle)\n    pos_manual[product] = (x, y)\n\n# Draw nodes with LABELS\ncustomer_nodes = [n for n, d in G_sample.nodes(data=True) if d.get('node_type') == 'customer']\nproduct_nodes = [n for n, d in G_sample.nodes(data=True) if d.get('node_type') == 'product']\n\n# Customer node (larger, blue)\nnx.draw_networkx_nodes(G_sample, pos_manual, nodelist=customer_nodes, \n                       node_color='#3498db', node_size=3500, alpha=0.9, ax=ax)\n\n# Product nodes (smaller, red)\nnx.draw_networkx_nodes(G_sample, pos_manual, nodelist=product_nodes, \n                       node_color='#e74c3c', node_size=2500, alpha=0.9, ax=ax)\n\n# Draw edges with arrows\nfor u, v in G_sample.edges():\n    x1, y1 = pos_manual[u]\n    x2, y2 = pos_manual[v]\n    \n    # Arrow\n    arrow = FancyArrowPatch((x1, y1), (x2, y2),\n                           arrowstyle='->', mutation_scale=30, \n                           linewidth=2.5, color='#34495e', alpha=0.7,\n                           connectionstyle=\"arc3,rad=0.1\")\n    ax.add_patch(arrow)\n    \n    # Relationship label on edge\n    mid_x, mid_y = (x1 + x2) / 2, (y1 + y2) / 2\n    count = G_sample[u][v]['count']\n    ax.text(mid_x, mid_y + 0.3, f\"PURCHASED\\n({count}x)\", \n           fontsize=9, ha='center', va='center',\n           bbox=dict(boxstyle='round,pad=0.4', facecolor='white', \n                    edgecolor='gray', linewidth=1.5, alpha=0.9),\n           fontweight='bold', style='italic')\n\n# Node labels (inside nodes)\nnode_labels = {}\nfor node, data in G_sample.nodes(data=True):\n    if data.get('node_type') == 'customer':\n        node_labels[node] = f\"Customer\\n{node[:8]}\"\n    else:\n        node_labels[node] = f\"Product\\n{node[:8]}\"\n\nnx.draw_networkx_labels(G_sample, pos_manual, labels=node_labels, \n                        font_size=10, font_weight='bold', font_color='white', ax=ax)\n\n# Legend\ncustomer_patch = mpatches.Patch(color='#3498db', label='Customer Node')\nproduct_patch = mpatches.Patch(color='#e74c3c', label='Product Node')\nax.legend(handles=[customer_patch, product_patch], loc='upper right', \n         fontsize=12, framealpha=0.95, edgecolor='black')\n\nax.set_title('Purchase Relationship Graph\\nPattern: (Customer)-[PURCHASED]->(Product)', \n            fontsize=16, fontweight='bold', pad=30)\nax.axis('off')\nax.set_xlim(-4.5, 4.5)\nax.set_ylim(-4.5, 4.5)\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/neo4j_style_purchase_graph.png', dpi=300, bbox_inches='tight')\nplt.show()\nprint(\"✓ Saved: neo4j_style_purchase_graph.png\")\n\n# ==================================================================================\n# VISUALIZATION 2: CO-PURCHASE PATTERN (Neo4j Style)\n# ==================================================================================\nprint(\"\\n📊 Visualization 2: Co-Purchase Pattern (Neo4j Style)...\")\n\nfig, ax = plt.subplots(figsize=(16, 12))\n\n# Sample: 2 customers sharing 1 product\ncopurchase_sample = copurchase_df.iloc[0]\nc1 = copurchase_sample['customer1']\nc2 = copurchase_sample['customer2']\nshared_product = copurchase_sample['product']\n\n# Build graph\nG_copurchase_demo = nx.DiGraph()\nG_copurchase_demo.add_node(c1, node_type='customer', label=f\"Customer1_{c1[:6]}\")\nG_copurchase_demo.add_node(c2, node_type='customer', label=f\"Customer2_{c2[:6]}\")\nG_copurchase_demo.add_node(shared_product, node_type='product', label=f\"Product_{shared_product[:8]}\")\n\nG_copurchase_demo.add_edge(c1, shared_product, relationship='PURCHASED', \n                           count=copurchase_sample['c1_purchases'])\nG_copurchase_demo.add_edge(c2, shared_product, relationship='PURCHASED', \n                           count=copurchase_sample['c2_purchases'])\n\n# Manual layout (triangle)\npos_copurchase = {\n    c1: (-2, 2),\n    c2: (2, 2),\n    shared_product: (0, -1)\n}\n\n# Draw nodes\ncustomer_nodes_cp = [c1, c2]\nproduct_nodes_cp = [shared_product]\n\nnx.draw_networkx_nodes(G_copurchase_demo, pos_copurchase, nodelist=customer_nodes_cp, \n                       node_color='#3498db', node_size=4000, alpha=0.9, ax=ax)\nnx.draw_networkx_nodes(G_copurchase_demo, pos_copurchase, nodelist=product_nodes_cp, \n                       node_color='#e74c3c', node_size=4000, alpha=0.9, ax=ax)\n\n# Draw arrows with labels\nfor u, v in G_copurchase_demo.edges():\n    x1, y1 = pos_copurchase[u]\n    x2, y2 = pos_copurchase[v]\n    \n    arrow = FancyArrowPatch((x1, y1), (x2, y2),\n                           arrowstyle='->', mutation_scale=35, \n                           linewidth=3, color='#2c3e50', alpha=0.8,\n                           connectionstyle=\"arc3,rad=0.15\")\n    ax.add_patch(arrow)\n    \n    # Relationship label\n    mid_x, mid_y = (x1 + x2) / 2, (y1 + y2) / 2\n    count = G_copurchase_demo[u][v]['count']\n    \n    # Position offset for readability\n    offset_x = 0.3 if x1 < x2 else -0.3\n    ax.text(mid_x + offset_x, mid_y + 0.2, f\":PURCHASED\\n({count}x)\", \n           fontsize=10, ha='center', va='center',\n           bbox=dict(boxstyle='round,pad=0.5', facecolor='lightyellow', \n                    edgecolor='black', linewidth=1.5, alpha=0.95),\n           fontweight='bold', style='italic')\n\n# Node labels\nlabels_copurchase = {\n    c1: f\"Customer\\n{c1[:10]}\",\n    c2: f\"Customer\\n{c2[:10]}\",\n    shared_product: f\"Product\\n{shared_product[:10]}\"\n}\nnx.draw_networkx_labels(G_copurchase_demo, pos_copurchase, labels=labels_copurchase, \n                        font_size=11, font_weight='bold', font_color='white', ax=ax)\n\n# Add pattern explanation\nax.text(0, 3.5, 'CO-PURCHASE PATTERN', fontsize=14, ha='center', \n       fontweight='bold', color='#2c3e50',\n       bbox=dict(boxstyle='round,pad=0.8', facecolor='lightgreen', alpha=0.7))\n\nax.text(0, -3, f'Both customers purchased the same product\\nSimilarity Score: {copurchase_sample[\"total_score\"]}', \n       fontsize=11, ha='center', style='italic',\n       bbox=dict(boxstyle='round,pad=0.6', facecolor='white', edgecolor='gray', alpha=0.9))\n\nax.set_title('Graph Query Pattern: Co-Purchase Analysis\\nCypher: MATCH (c1:Customer)-[:PURCHASED]->(p:Product)<-[:PURCHASED]-(c2:Customer)', \n            fontsize=14, fontweight='bold', pad=20)\nax.axis('off')\nax.set_xlim(-3.5, 3.5)\nax.set_ylim(-4, 4.5)\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/neo4j_copurchase_pattern.png', dpi=300, bbox_inches='tight')\nplt.show()\nprint(\"✓ Saved: neo4j_copurchase_pattern.png\")\n\n# ==================================================================================\n# VISUALIZATION 3: PRODUCT ASSOCIATION PATTERN\n# ==================================================================================\nprint(\"\\n📊 Visualization 3: Product Association Pattern (Neo4j Style)...\")\n\nfig, ax = plt.subplots(figsize=(16, 12))\n\n# Sample: 1 customer buying 3 products\nsample_customer_assoc = top_customers_deg[1][0]\nproducts_bought = [n for n in G.neighbors(sample_customer_assoc) if n in products_set][:3]\n\n# Build graph\nG_assoc = nx.DiGraph()\nG_assoc.add_node(sample_customer_assoc, node_type='customer')\n\nfor product in products_bought:\n    G_assoc.add_node(product, node_type='product')\n    G_assoc.add_edge(sample_customer_assoc, product, relationship='PURCHASED',\n                    amount=G[sample_customer_assoc][product]['total_amount'])\n\n# Layout (star)\npos_assoc = {sample_customer_assoc: (0, 0)}\nangle_step = 2 * np.pi / len(products_bought)\nradius = 3.5\n\nfor i, product in enumerate(products_bought):\n    angle = i * angle_step + np.pi/2\n    x = radius * np.cos(angle)\n    y = radius * np.sin(angle)\n    pos_assoc[product] = (x, y)\n\n# Draw\nnx.draw_networkx_nodes(G_assoc, pos_assoc, nodelist=[sample_customer_assoc], \n                       node_color='#3498db', node_size=4500, alpha=0.9, ax=ax)\nnx.draw_networkx_nodes(G_assoc, pos_assoc, nodelist=products_bought, \n                       node_color='#e74c3c', node_size=3500, alpha=0.9, ax=ax)\n\n# Arrows\nfor u, v in G_assoc.edges():\n    x1, y1 = pos_assoc[u]\n    x2, y2 = pos_assoc[v]\n    \n    arrow = FancyArrowPatch((x1, y1), (x2, y2),\n                           arrowstyle='->', mutation_scale=32, \n                           linewidth=3, color='#34495e', alpha=0.75)\n    ax.add_patch(arrow)\n    \n    mid_x, mid_y = (x1 + x2) / 2, (y1 + y2) / 2\n    amount = G_assoc[u][v]['amount']\n    ax.text(mid_x, mid_y, f\":PURCHASED\\n${amount:.2f}\", \n           fontsize=9, ha='center', va='center',\n           bbox=dict(boxstyle='round,pad=0.4', facecolor='white', \n                    edgecolor='gray', linewidth=1.5, alpha=0.95),\n           fontweight='bold')\n\n# Labels\nlabels_assoc = {sample_customer_assoc: f\"Customer\\n{sample_customer_assoc[:10]}\"}\nfor i, product in enumerate(products_bought, 1):\n    labels_assoc[product] = f\"Product {i}\\n{product[:8]}\"\n\nnx.draw_networkx_labels(G_assoc, pos_assoc, labels=labels_assoc, \n                        font_size=11, font_weight='bold', font_color='white', ax=ax)\n\nax.set_title('Product Association Pattern\\nCypher: MATCH (c:Customer)-[:PURCHASED]->(p1:Product), (c)-[:PURCHASED]->(p2:Product)', \n            fontsize=14, fontweight='bold', pad=20)\nax.text(0, -4.5, 'These products are frequently bought together by the same customer', \n       fontsize=11, ha='center', style='italic',\n       bbox=dict(boxstyle='round,pad=0.6', facecolor='lightyellow', alpha=0.8))\n\nax.axis('off')\nax.set_xlim(-5, 5)\nax.set_ylim(-5.5, 5)\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/neo4j_product_association.png', dpi=300, bbox_inches='tight')\nplt.show()\nprint(\"✓ Saved: neo4j_product_association.png\")\n\n# ==================================================================================\n# VISUALIZATION 4: MULTI-CUSTOMER NETWORK (Labeled)\n# ==================================================================================\nprint(\"\\n📊 Visualization 4: Multi-Customer Network (Labeled)...\")\n\nfig, ax = plt.subplots(figsize=(18, 14))\n\n# Sample: Top 4 customers + 1 shared product\ntop4_customers = [c for c, _ in top_customers_deg[:4]]\n# Find a product all 4 bought\nproduct_buyers = {}\nfor c in top4_customers:\n    for p in [n for n in G.neighbors(c) if n in products_set]:\n        if p not in product_buyers:\n            product_buyers[p] = []\n        product_buyers[p].append(c)\n\n# Find product with most buyers from top 4\nshared_product_multi = max(product_buyers.items(), key=lambda x: len(x[1]))[0]\ncustomers_bought_it = product_buyers[shared_product_multi][:4]\n\n# Build graph\nG_multi = nx.DiGraph()\nG_multi.add_node(shared_product_multi, node_type='product')\n\nfor customer in customers_bought_it:\n    G_multi.add_node(customer, node_type='customer')\n    G_multi.add_edge(customer, shared_product_multi, relationship='PURCHASED',\n                    count=G[customer][shared_product_multi]['purchase_count'])\n\n# Layout (star with product in center)\npos_multi = {shared_product_multi: (0, 0)}\nangle_step = 2 * np.pi / len(customers_bought_it)\nradius = 4\n\nfor i, customer in enumerate(customers_bought_it):\n    angle = i * angle_step\n    x = radius * np.cos(angle)\n    y = radius * np.sin(angle)\n    pos_multi[customer] = (x, y)\n\n# Draw\nnx.draw_networkx_nodes(G_multi, pos_multi, nodelist=[shared_product_multi], \n                       node_color='#e74c3c', node_size=5000, alpha=0.95, ax=ax,\n                       edgecolors='black', linewidths=3)\nnx.draw_networkx_nodes(G_multi, pos_multi, nodelist=customers_bought_it, \n                       node_color='#3498db', node_size=3500, alpha=0.9, ax=ax,\n                       edgecolors='black', linewidths=2)\n\n# Arrows\nfor u, v in G_multi.edges():\n    x1, y1 = pos_multi[u]\n    x2, y2 = pos_multi[v]\n    \n    arrow = FancyArrowPatch((x1, y1), (x2, y2),\n                           arrowstyle='->', mutation_scale=35, \n                           linewidth=3.5, color='#2c3e50', alpha=0.8)\n    ax.add_patch(arrow)\n    \n    mid_x, mid_y = (x1 + x2) / 2, (y1 + y2) / 2\n    count = G_multi[u][v]['count']\n    \n    # Offset for readability\n    offset_angle = np.arctan2(y2 - y1, x2 - x1) + np.pi/2\n    offset_x = 0.4 * np.cos(offset_angle)\n    offset_y = 0.4 * np.sin(offset_angle)\n    \n    ax.text(mid_x + offset_x, mid_y + offset_y, f\":PURCHASED\\n({count})\", \n           fontsize=9, ha='center', va='center',\n           bbox=dict(boxstyle='round,pad=0.4', facecolor='lightyellow', \n                    edgecolor='black', linewidth=1.3, alpha=0.95),\n           fontweight='bold', rotation=0)\n\n# Labels\nlabels_multi = {shared_product_multi: f\"Product\\n{shared_product_multi[:10]}\"}\nfor i, customer in enumerate(customers_bought_it, 1):\n    labels_multi[customer] = f\"Customer {i}\\n{customer[:8]}\"\n\nnx.draw_networkx_labels(G_multi, pos_multi, labels=labels_multi, \n                        font_size=12, font_weight='bold', font_color='white', ax=ax)\n\nax.set_title('Popular Product Network\\nMultiple Customers Purchasing Same Product', \n            fontsize=16, fontweight='bold', pad=30)\n\nax.text(0, -5.5, f'Product {shared_product_multi[:15]} has {len(customers_bought_it)} buyers from top customers', \n       fontsize=12, ha='center', style='italic',\n       bbox=dict(boxstyle='round,pad=0.7', facecolor='lightgreen', alpha=0.8))\n\nax.axis('off')\nax.set_xlim(-6, 6)\nax.set_ylim(-6.5, 6)\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/neo4j_multi_customer_network.png', dpi=300, bbox_inches='tight')\nplt.show()\nprint(\"✓ Saved: neo4j_multi_customer_network.png\")\n\n# ==================================================================================\n# VISUALIZATION 5: GRAPH SCHEMA DIAGRAM\n# ==================================================================================\nprint(\"\\n📊 Visualization 5: Graph Schema Diagram...\")\n\nfig, ax = plt.subplots(figsize=(14, 8))\n\n# Define schema nodes\nschema_nodes = {\n    'Customer': (-3, 0),\n    'Product': (3, 0)\n}\n\n# Draw schema nodes\nfor node, pos in schema_nodes.items():\n    circle = plt.Circle(pos, 1.2, color='#3498db' if node == 'Customer' else '#e74c3c', \n                       alpha=0.9, zorder=2, edgecolor='black', linewidth=3)\n    ax.add_patch(circle)\n    ax.text(pos[0], pos[1], node, fontsize=16, ha='center', va='center', \n           fontweight='bold', color='white', zorder=3)\n\n# Draw relationship arrow\narrow_schema = FancyArrowPatch((-1.8, 0), (1.8, 0),\n                              arrowstyle='->', mutation_scale=50, \n                              linewidth=4, color='#2c3e50', alpha=0.9, zorder=1)\nax.add_patch(arrow_schema)\n\n# Relationship label\nax.text(0, 0.5, 'PURCHASED', fontsize=14, ha='center', va='center',\n       bbox=dict(boxstyle='round,pad=0.6', facecolor='white', \n                edgecolor='black', linewidth=2, alpha=1),\n       fontweight='bold', zorder=4)\n\n# Properties\ncustomer_props = \"Properties:\\n• customer_id\\n• age\\n• club_status\"\nproduct_props = \"Properties:\\n• article_id\\n• price\\n• category\"\n\nax.text(-3, -2.5, customer_props, fontsize=11, ha='center', va='top',\n       bbox=dict(boxstyle='round,pad=0.5', facecolor='lightblue', alpha=0.8))\nax.text(3, -2.5, product_props, fontsize=11, ha='center', va='top',\n       bbox=dict(boxstyle='round,pad=0.5', facecolor='#ffcccb', alpha=0.8))\n\n# Relationship properties\nrel_props = \"count: Integer\\namount: Float\\ndate: Date\"\nax.text(0, -1.2, rel_props, fontsize=10, ha='center', va='top',\n       bbox=dict(boxstyle='round,pad=0.4', facecolor='lightyellow', alpha=0.9),\n       style='italic')\n\nax.set_title('H&M Graph Database Schema\\n(Property Graph Model)', \n            fontsize=16, fontweight='bold', pad=30)\nax.axis('off')\nax.set_xlim(-5.5, 5.5)\nax.set_ylim(-3.5, 2)\n\nplt.tight_layout()\nplt.savefig('/kaggle/working/graph_schema_diagram.png', dpi=300, bbox_inches='tight')\nplt.show()\nprint(\"✓ Saved: graph_schema_diagram.png\")\n\n# ==================================================================================\n# FINAL SUMMARY\n# ==================================================================================\nprint(\"\\n\" + \"=\"*100)\nprint(\"✅ NEO4J-STYLE VISUALIZATIONS COMPLETE!\")\nprint(\"=\"*100)\n\nprint(f\"\"\"\n📊 GENERATED 5 NEO4J-STYLE VISUALIZATIONS:\n   1. neo4j_style_purchase_graph.png     - Single customer → multiple products (labeled)\n   2. neo4j_copurchase_pattern.png       - Co-purchase pattern with relationship labels\n   3. neo4j_product_association.png      - Product association with transaction amounts\n   4. neo4j_multi_customer_network.png   - Multiple customers → shared product\n   5. graph_schema_diagram.png           - Property graph schema\n\n💡 GRAPH QUERY PATTERNS DEMONSTRATED:\n   Pattern 1: (Customer)-[:PURCHASED]->(Product)\n   Pattern 2: (Customer1)-[:PURCHASED]->(Product)<-[:PURCHASED]-(Customer2)\n   Pattern 3: (Customer)-[:PURCHASED]->(Product1), (Customer)-[:PURCHASED]->(Product2)\n\"\"\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}