{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n        # pass\n        # print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-06T14:57:44.337715Z","iopub.execute_input":"2025-11-06T14:57:44.337945Z","iopub.status.idle":"2025-11-06T14:57:44.592541Z","shell.execute_reply.started":"2025-11-06T14:57:44.337928Z","shell.execute_reply":"2025-11-06T14:57:44.592002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip -q install open-clip-torch==2.24.0 faiss-cpu==1.8.0.post1 torchmetrics==1.4.0.post0 umap-learn==0.5.6 networkx==3.2.1 rich==13.7.1 --no-input","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T14:57:48.109468Z","iopub.execute_input":"2025-11-06T14:57:48.110331Z","iopub.status.idle":"2025-11-06T14:59:21.161726Z","shell.execute_reply.started":"2025-11-06T14:57:48.110303Z","shell.execute_reply":"2025-11-06T14:59:21.161005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install faiss-cpu","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T14:59:21.163355Z","iopub.execute_input":"2025-11-06T14:59:21.163652Z","iopub.status.idle":"2025-11-06T14:59:24.269739Z","shell.execute_reply.started":"2025-11-06T14:59:21.163619Z","shell.execute_reply":"2025-11-06T14:59:24.269016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install open-clip-torch==2.24","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T14:59:24.271014Z","iopub.execute_input":"2025-11-06T14:59:24.271327Z","iopub.status.idle":"2025-11-06T14:59:27.451432Z","shell.execute_reply.started":"2025-11-06T14:59:24.271291Z","shell.execute_reply":"2025-11-06T14:59:27.450674Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Libraries","metadata":{}},{"cell_type":"code","source":"\n\nimport os, gc, math, time, random, json, pathlib, collections, warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\n# Viz\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport networkx as nx\n\n# kNN\nimport faiss\n\n# CLIP\nimport open_clip\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T14:59:41.497022Z","iopub.execute_input":"2025-11-06T14:59:41.497332Z","iopub.status.idle":"2025-11-06T14:59:41.502481Z","shell.execute_reply.started":"2025-11-06T14:59:41.497305Z","shell.execute_reply":"2025-11-06T14:59:41.501822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\nSEED = 42\nrandom.seed(SEED); np.random.seed(SEED); torch.manual_seed(SEED)\ntorch.backends.cudnn.benchmark = True\nDEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nprint(\"Device:\", DEVICE)\n\nclass CFG:\n    DATA_DIR = \"/kaggle/input/h-and-m-personalized-fashion-recommendations\"\n    IMG_DIR  = \"/kaggle/input/h-and-m-personalized-fashion-recommendations/images\"\n    WORK_DIR = \"/kaggle/working\"\n\n    # subset for speed (tune up later)\n    MAX_USERS = 20000\n    MAX_ITEMS = 80000\n    MIN_USER_INTERACTIONS = 5\n    MIN_ITEM_INTERACTIONS = 5\n\n    USE_TIME_SPLIT = True  # 7-day holdout\n\n    # CLIP\n    CLIP_MODEL = \"ViT-B-32\"\n    CLIP_PRETRAIN = \"laion2b_s34b_b79k\"\n    CLIP_BATCH = 128\n    IMG_SIZE = 224\n\n    # II graph fusion\n    TOPK_II = 20\n    ALPHA = 0.5  # image\n    BETA  = 0.4  # text\n    GAMMA = 0.1  # co-occur\n\n    # LightGCN\n    EMBED_DIM = 64\n    LAYERS = 3\n    LR = 1e-3\n    WEIGHT_DECAY = 0.0\n    BATCH_SIZE = 4096\n    EPOCHS = 5\n\n    # losses\n    LAMBDA_G = 0.1\n    TAU = 0.4\n\n    # eval\n    K_EVAL = 20\n\ncfg = CFG()\nprint(cfg.__dict__)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T14:59:50.311398Z","iopub.execute_input":"2025-11-06T14:59:50.311696Z","iopub.status.idle":"2025-11-06T14:59:50.400843Z","shell.execute_reply.started":"2025-11-06T14:59:50.311673Z","shell.execute_reply":"2025-11-06T14:59:50.400015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ARTICLES_CSV = os.path.join(cfg.DATA_DIR, \"articles.csv\")\nCUSTOMERS_CSV = os.path.join(cfg.DATA_DIR, \"customers.csv\")\nTRANS_CSV    = os.path.join(cfg.DATA_DIR, \"transactions_train.csv\")\n\narticles = pd.read_csv(ARTICLES_CSV)\ncustomers = pd.read_csv(CUSTOMERS_CSV)\ntransactions = pd.read_csv(TRANS_CSV, parse_dates=[\"t_dat\"])\n\nprint(articles.shape, customers.shape, transactions.shape)\ndisplay(articles.head(2))\ndisplay(customers.head(2))\ndisplay(transactions.head(2))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T14:59:53.581658Z","iopub.execute_input":"2025-11-06T14:59:53.581938Z","iopub.status.idle":"2025-11-06T15:01:18.783015Z","shell.execute_reply.started":"2025-11-06T14:59:53.581916Z","shell.execute_reply":"2025-11-06T15:01:18.782404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"daily = transactions.groupby(transactions[\"t_dat\"].dt.date).size()\nplt.figure(figsize=(10,3)); daily.plot(); plt.title(\"Transactions per day\"); plt.tight_layout(); plt.show()\n\nif \"product_type_name\" in articles.columns:\n    top_types = articles[\"product_type_name\"].value_counts().head(15)\n    plt.figure(figsize=(8,4)); sns.barplot(x=top_types.values, y=top_types.index)\n    plt.title(\"Top product types in catalog\"); plt.tight_layout(); plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T15:01:45.356787Z","iopub.execute_input":"2025-11-06T15:01:45.357127Z","iopub.status.idle":"2025-11-06T15:01:54.983217Z","shell.execute_reply.started":"2025-11-06T15:01:45.357103Z","shell.execute_reply":"2025-11-06T15:01:54.982412Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4) Filter cold users/items, map IDs, time split","metadata":{}},{"cell_type":"code","source":"user_cnt = transactions[\"customer_id\"].value_counts()\nitem_cnt = transactions[\"article_id\"].value_counts()\n\nkeep_users = user_cnt[user_cnt >= cfg.MIN_USER_INTERACTIONS].index\nkeep_items = item_cnt[item_cnt >= cfg.MIN_ITEM_INTERACTIONS].index\n\ndf = transactions[transactions[\"customer_id\"].isin(keep_users) & transactions[\"article_id\"].isin(keep_items)].copy()\n\nif cfg.MAX_USERS:\n    sel_users = set(df[\"customer_id\"].drop_duplicates().sample(min(cfg.MAX_USERS, df[\"customer_id\"].nunique()), random_state=SEED))\n    df = df[df[\"customer_id\"].isin(sel_users)]\nif cfg.MAX_ITEMS:\n    sel_items = set(df[\"article_id\"].drop_duplicates().sample(min(cfg.MAX_ITEMS, df[\"article_id\"].nunique()), random_state=SEED))\n    df = df[df[\"article_id\"].isin(sel_items)]\n\nuser_le = LabelEncoder().fit(df[\"customer_id\"])\nitem_le = LabelEncoder().fit(df[\"article_id\"])\ndf[\"uid\"] = user_le.transform(df[\"customer_id\"])\ndf[\"iid\"] = item_le.transform(df[\"article_id\"])\n\nn_users = df[\"uid\"].nunique()\nn_items = df[\"iid\"].nunique()\nprint(\"Users:\", n_users, \"Items:\", n_items, \"Interactions:\", len(df))\n\ntmax = df[\"t_dat\"].max()\nif cfg.USE_TIME_SPLIT:\n    cutoff = tmax - pd.Timedelta(days=7)\n    train_df = df[df[\"t_dat\"] <= cutoff]\n    test_df  = df[df[\"t_dat\"] > cutoff]\n    cutoff_val = cutoff - pd.Timedelta(days=3)\n    val_df = train_df[train_df[\"t_dat\"] > cutoff_val]\n    train_df = train_df[train_df[\"t_dat\"] <= cutoff_val]\nelse:\n    train_df, tail = train_test_split(df, test_size=0.2, random_state=SEED)\n    val_df, test_df = train_test_split(tail, test_size=0.5, random_state=SEED)\n\nprint(\"Train/Val/Test:\", len(train_df), len(val_df), len(test_df))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T15:01:59.041446Z","iopub.execute_input":"2025-11-06T15:01:59.041771Z","iopub.status.idle":"2025-11-06T15:02:26.985046Z","shell.execute_reply.started":"2025-11-06T15:01:59.041735Z","shell.execute_reply":"2025-11-06T15:02:26.984353Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5) Build per-user positives & visualize distribution","metadata":{}},{"cell_type":"code","source":"from collections import defaultdict\n\ndef build_user_pos(df_part):\n    pos = defaultdict(set)\n    for u, i in zip(df_part[\"uid\"].values, df_part[\"iid\"].values):\n        pos[int(u)].add(int(i))\n    return pos\n\nuser_pos_train = build_user_pos(train_df)\nuser_pos_val   = build_user_pos(val_df)\nuser_pos_test  = build_user_pos(test_df)\n\nlens = [len(v) for v in user_pos_train.values()]\nplt.figure(figsize=(6,3)); sns.histplot(lens, bins=50); plt.title(\"Train positives per user\"); plt.tight_layout(); plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T15:02:43.813521Z","iopub.execute_input":"2025-11-06T15:02:43.81382Z","iopub.status.idle":"2025-11-06T15:02:44.606999Z","shell.execute_reply.started":"2025-11-06T15:02:43.813799Z","shell.execute_reply":"2025-11-06T15:02:44.606178Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6) Prepare CLIP inputs (texts & image paths)","metadata":{}},{"cell_type":"code","source":"# Align articles to filtered items\nvalid_ids = item_le.inverse_transform(np.arange(n_items))\nart = articles[articles[\"article_id\"].isin(valid_ids)].copy()\nart[\"iid\"] = item_le.transform(art[\"article_id\"])\nart = art.set_index(\"iid\").sort_index()\n\ndef build_text(row):\n    parts = []\n    if isinstance(row.get(\"detail_desc\"), str) and len(row[\"detail_desc\"]) > 0:\n        parts.append(row[\"detail_desc\"][:200])\n    if isinstance(row.get(\"product_type_name\"), str):\n        parts.append(row[\"product_type_name\"])\n    if isinstance(row.get(\"index_name\"), str):\n        parts.append(row[\"index_name\"])\n    return \". \".join(parts) if parts else \"fashion item\"\n\ntexts = [build_text(row._asdict()) for row in art.itertuples()]\n\ndef article_id_to_path(article_id):\n    s = str(article_id).zfill(10)\n    return os.path.join(cfg.IMG_DIR, s[:3], s + \".jpg\")\n\ninverse_ids = item_le.inverse_transform(np.arange(n_items))\nimg_paths = [article_id_to_path(int(aid)) for aid in inverse_ids]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T15:03:00.585016Z","iopub.execute_input":"2025-11-06T15:03:00.585298Z","iopub.status.idle":"2025-11-06T15:03:01.053592Z","shell.execute_reply.started":"2025-11-06T15:03:00.585279Z","shell.execute_reply":"2025-11-06T15:03:01.053005Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7) CLIP: encode text & images, compute grounding score g","metadata":{}},{"cell_type":"code","source":"clip_model, _, clip_preprocess = open_clip.create_model_and_transforms(cfg.CLIP_MODEL, pretrained=cfg.CLIP_PRETRAIN, device=DEVICE)\nclip_tokenizer = open_clip.get_tokenizer(cfg.CLIP_MODEL)\n\n\n\ndef encode_texts(texts, batch=512):\n    all_feats = []\n    for i in tqdm(range(0, len(texts), batch), desc=\"CLIP Text\"):\n        tok = clip_tokenizer(texts[i:i+batch]).to(DEVICE)\n        with torch.no_grad(), torch.cuda.amp.autocast(enabled=(DEVICE=='cuda')):\n            feats = clip_model.encode_text(tok)\n        feats = F.normalize(feats, dim=1)\n        all_feats.append(feats.detach().cpu())\n    return torch.cat(all_feats, dim=0)\n\nfrom PIL import Image\nfrom torch.utils.data import Dataset, DataLoader\n\n# ADD THIS CLASS\nclass ImageDataset(Dataset):\n    def __init__(self, paths, preprocess_fn):\n        self.paths = paths\n        self.preprocess = preprocess_fn\n\n    def __len__(self):\n        return len(self.paths)\n\n    def __getitem__(self, idx):\n        p = self.paths[idx]\n        try:\n            with Image.open(p) as im:\n                im = im.convert(\"RGB\")\n            return self.preprocess(im)\n        except:\n            # Return a blank tensor if image is corrupt\n            return torch.zeros(3, cfg.IMG_SIZE, cfg.IMG_SIZE)\n\n# ADD THIS FUNCTION\ndef encode_images_fast(paths, batch=cfg.CLIP_BATCH):\n    dataset = ImageDataset(paths, clip_preprocess)\n    # Use num_workers=4 (or 2) to load in parallel\n    loader = DataLoader(dataset, batch_size=batch, shuffle=False, num_workers=4, pin_memory=True) \n\n    all_feats = []\n    with torch.no_grad(), torch.cuda.amp.autocast(enabled=(DEVICE=='cuda')):\n        for imgs in tqdm(loader, desc=\"CLIP Image (Fast)\"):\n            imgs = imgs.to(DEVICE)\n            feats = clip_model.encode_image(imgs)\n            feats = F.normalize(feats, dim=1)\n            all_feats.append(feats.detach().cpu())\n\n    return torch.cat(all_feats, dim=0)\n\n\nZ_txt = encode_texts(texts).numpy().astype(\"float32\")\nZ_img = encode_images_fast(img_paths).numpy().astype(\"float32\")\n\ndef rowwise_cos(a, b): return (a*b).sum(-1)\ng = rowwise_cos(Z_img, Z_txt)\ng = np.clip(g, 0.0, 1.0).astype(\"float32\")\n# print(\"Z_img:\", Z_img.shape, \"Z_txt:\", Z_txt.shape, \"g mean:\", g.mean())\n\nplt.figure(figsize=(6,3)); sns.histplot(g, bins=50); plt.title(\"Grounding score g\"); plt.tight_layout(); plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T15:03:03.562627Z","iopub.execute_input":"2025-11-06T15:03:03.562906Z","iopub.status.idle":"2025-11-06T15:19:58.285988Z","shell.execute_reply.started":"2025-11-06T15:03:03.562889Z","shell.execute_reply":"2025-11-06T15:19:58.285012Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 8) Peak Items with g","metadata":{}},{"cell_type":"code","source":"from PIL import Image\ndef show_examples(idx_list, cols=5):\n    rows = math.ceil(len(idx_list)/cols)\n    plt.figure(figsize=(3*cols, 3*rows))\n    for k, idx in enumerate(idx_list):\n        p = img_paths[idx]\n        try:\n            with Image.open(p) as im: im = im.convert(\"RGB\")\n        except: \n            im = Image.new(\"RGB\", (224,224), (200,200,200))\n        plt.subplot(rows, cols, k+1)\n        plt.imshow(im); plt.axis(\"off\")\n        title = (texts[idx][:35] + \"...\") if len(texts[idx])>35 else texts[idx]\n        plt.title(f\"iid={idx}  g={g[idx]:.2f}\\n{title}\", fontsize=9)\n    plt.tight_layout(); plt.show()\n\nshow_examples(random.sample(range(n_items), k=min(10, n_items)))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T15:20:01.981307Z","iopub.execute_input":"2025-11-06T15:20:01.981608Z","iopub.status.idle":"2025-11-06T15:20:04.76641Z","shell.execute_reply.started":"2025-11-06T15:20:01.981576Z","shell.execute_reply":"2025-11-06T15:20:04.765619Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 9) Build item–item kNN (image, text), co-occurrence, and fuse with grounding","metadata":{}},{"cell_type":"code","source":"def faiss_knn(x, k):\n    index = faiss.IndexFlatIP(x.shape[1])\n    faiss.normalize_L2(x)\n    index.add(x)\n    sims, ids = index.search(x, k+1)  # includes self\n    return sims[:,1:], ids[:,1:]\n\nprint(\"kNN img...\")\nimg_sims, img_ids = faiss_knn(Z_img.copy(), cfg.TOPK_II)\nprint(\"kNN txt...\")\ntxt_sims, txt_ids = faiss_knn(Z_txt.copy(), cfg.TOPK_II)\n\n# PASTE THIS NEW BLOCK IN\nfrom scipy.sparse import coo_matrix\n\nprint(\"Building user-item sparse matrix...\")\n# Build the User-Item interaction matrix A\nuu = train_df[\"uid\"].values.astype(\"int64\")\nii = train_df[\"iid\"].values.astype(\"int64\")\nv  = np.ones_like(uu, dtype=\"float32\")\nA_ui_sparse = coo_matrix((v, (uu, ii)), shape=(n_users, n_items), dtype=np.float32).tocsr()\n\nprint(\"Calculating Co-occurrence matrix (A.T @ A)...\")\n# C = A.T @ A\n# This is the co-occurrence matrix. C[i, j] = # of users who bought both i and j.\nC_cooccur = A_ui_sparse.T @ A_ui_sparse\n\nprint(\"Calculating Jaccard from co-occurrence...\")\n# Get item popularity (purchases per item)\nitem_pops = A_ui_sparse.sum(axis=0).A1 # .A1 converts to flat numpy array\nco_ids = np.zeros_like(img_ids)\nco_sims = np.zeros_like(img_sims)\n\nfor i in tqdm(range(n_items), desc=\"Co-occur kNN (Fast)\"):\n    # Get the i-th row from the sparse matrix\n    row = C_cooccur.getrow(i)\n    if row.nnz == 0:\n        co_ids[i] = img_ids[i]; co_sims[i] = 0.0; continue\n\n    j_indices = row.indices\n    cij_vals = row.data # This is (A intersect B)\n\n    # Calculate union\n    pop_i = item_pops[i]\n    pop_j = item_pops[j_indices]\n    union = pop_i + pop_j - cij_vals\n\n    # Calculate Jaccard\n    jaccard = cij_vals / (union + 1e-8) # Add epsilon to avoid zero division\n\n    # Get top-k\n    order = np.argsort(-jaccard)[:cfg.TOPK_II]\n    sel_idx = j_indices[order]\n    sel_sim = jaccard[order]\n\n    if len(sel_idx) < cfg.TOPK_II:\n        need = cfg.TOPK_II - len(sel_idx)\n        sel_idx = np.concatenate([sel_idx, img_ids[i,:need]])\n        sel_sim = np.concatenate([sel_sim, np.zeros(need)])\n\n    co_ids[i] = sel_idx\n    co_sims[i] = sel_sim\n\n\n\ndef fuse_and_gate(img_ids, img_sims, txt_ids, txt_sims, co_ids, co_sims, g, alpha, beta, gamma):\n    rows, cols, vals = [], [], []\n    for i in range(n_items):\n        nbrs = {}\n        for ids, sims, w in [(img_ids, img_sims, alpha), (txt_ids, txt_sims, beta), (co_ids, co_sims, gamma)]:\n            for j, s in zip(ids[i], sims[i]):\n                nbrs[j] = nbrs.get(j, 0.0) + w * float(s)\n        for j, s in nbrs.items():\n            w = s * float(0.5*(g[i] + g[j]))\n            if w>0:\n                rows.append(i); cols.append(j); vals.append(w)\n    return np.array(rows), np.array(cols), np.array(vals, dtype=\"float32\")\n\nrows, cols, vals = fuse_and_gate(img_ids, img_sims, txt_ids, txt_sims, co_ids, co_sims, g, cfg.ALPHA, cfg.BETA, cfg.GAMMA)\nprint(\"II edges:\", len(vals))\n\ndef build_norm_adj(n, rows, cols, vals):\n    all_r = np.concatenate([rows, cols])\n    all_c = np.concatenate([cols, rows])\n    all_v = np.concatenate([vals, vals])\n    deg = np.bincount(all_r, weights=all_v, minlength=n) + 1e-8\n    deg2 = np.sqrt(deg)\n    norm_v = all_v / (deg2[all_r] * deg2[all_c])\n    idx = np.vstack([all_r, all_c])\n    A = torch.sparse_coo_tensor(indices=torch.tensor(idx, dtype=torch.long),\n                                values=torch.tensor(norm_v, dtype=torch.float32),\n                                size=(n, n))\n    return A.coalesce()\n\nA_II = build_norm_adj(n_items, rows, cols, vals)\nA_II\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T15:20:30.361564Z","iopub.execute_input":"2025-11-06T15:20:30.36183Z","iopub.status.idle":"2025-11-06T16:25:33.235536Z","shell.execute_reply.started":"2025-11-06T15:20:30.361811Z","shell.execute_reply":"2025-11-06T16:25:33.234896Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 10) Build User–Item adjacency; lift II to total graph","metadata":{}},{"cell_type":"code","source":"n_total = n_users + n_items\n\ndef build_ui_norm_adj(train_df, n_users, n_items):\n    uu = train_df[\"uid\"].values.astype(\"int64\")\n    ii = train_df[\"iid\"].values.astype(\"int64\") + n_users\n    v  = np.ones_like(uu, dtype=\"float32\")\n    rows = np.concatenate([uu, ii])\n    cols = np.concatenate([ii, uu])\n    vals = np.concatenate([v, v])\n\n    deg = np.bincount(rows, weights=vals, minlength=n_users+n_items) + 1e-8\n    deg2 = np.sqrt(deg)\n    norm_v = vals / (deg2[rows] * deg2[cols])\n    idx = np.vstack([rows, cols])\n    A = torch.sparse_coo_tensor(indices=torch.tensor(idx, dtype=torch.long),\n                                values=torch.tensor(norm_v, dtype=torch.float32),\n                                size=(n_users+n_items, n_users+n_items))\n    return A.coalesce()\n\nA_UI = build_ui_norm_adj(train_df, n_users, n_items)\n\ndef lift_item_adj_to_total(A_II, n_users, n_items):\n    idx = A_II.indices()\n    val = A_II.values()\n    idx_lift = torch.vstack([idx[0]+n_users, idx[1]+n_users])\n    A = torch.sparse_coo_tensor(idx_lift, val, (n_users+n_items, n_users+n_items)).coalesce()\n    return A\n\nA_II_total = lift_item_adj_to_total(A_II, n_users, n_items)\nA_UI, A_II_total = A_UI.to(DEVICE), A_II_total.to(DEVICE)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T16:35:06.137773Z","iopub.execute_input":"2025-11-06T16:35:06.138431Z","iopub.status.idle":"2025-11-06T16:35:07.045623Z","shell.execute_reply.started":"2025-11-06T16:35:06.138407Z","shell.execute_reply":"2025-11-06T16:35:07.045002Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 11) Visualize a small neighborhood (graph picture)","metadata":{}},{"cell_type":"code","source":"def visualize_user_neighborhood(u, max_items=30):\n    items = list(user_pos_train.get(u, []))[:max_items]\n    G = nx.Graph()\n    G.add_node(f\"U{u}\", bipartite=0)\n    for it in items:\n        G.add_node(f\"I{it}\", bipartite=1)\n        G.add_edge(f\"U{u}\", f\"I{it}\", color=\"tab:blue\")\n        try:\n            nbrs = img_ids[it][:3]  # quick visual neighbors from image-kNN\n        except:\n            nbrs = []\n        for j in nbrs:\n            G.add_node(f\"I{j}\", bipartite=1)\n            G.add_edge(f\"I{it}\", f\"I{j}\", color=\"tab:orange\")\n    colors = [G[u][v]['color'] for u,v in G.edges()]\n    pos = nx.spring_layout(G, seed=SEED, k=0.7)\n    plt.figure(figsize=(8,6))\n    nx.draw(G, pos, with_labels=False, node_size=80, edge_color=colors)\n    plt.title(f\"Neighborhood around user {u}\")\n    plt.show()\n\nvisualize_user_neighborhood(u=random.randint(0, n_users-1))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T16:35:11.035826Z","iopub.execute_input":"2025-11-06T16:35:11.036515Z","iopub.status.idle":"2025-11-06T16:35:11.186229Z","shell.execute_reply.started":"2025-11-06T16:35:11.036491Z","shell.execute_reply":"2025-11-06T16:35:11.185395Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 12) LightGCN with BPR + grounding loss","metadata":{}},{"cell_type":"code","source":"class LightGCN(nn.Module):\n    def __init__(self, n_users, n_items, d=64, layers=3):\n        super().__init__()\n        self.emb = nn.Embedding(n_users+n_items, d)\n        nn.init.xavier_uniform_(self.emb.weight)\n        self.layers = layers\n        self.n_users = n_users\n        self.n_items = n_items\n\n    def propagate(self, A_ui, A_ii):\n        E0 = self.emb.weight\n        acc = [E0]\n        E = E0\n        for _ in range(self.layers):\n            E = 0.8*torch.sparse.mm(A_ui, E) + 0.2*torch.sparse.mm(A_ii, E)\n            acc.append(E)\n        return torch.stack(acc, dim=0).mean(0)\n\n    def forward(self, A_ui, A_ii):\n        E = self.propagate(A_ui, A_ii)\n        U = E[:self.n_users]\n        I = E[self.n_users:]\n        return U, I\n\ndef bpr_loss(u_e, i_pos_e, i_neg_e):\n    pos = (u_e*i_pos_e).sum(-1)\n    neg = (u_e*i_neg_e).sum(-1)\n    return -F.logsigmoid(pos - neg).mean()\n\nuser_pos_array = {u: np.array(list(items)) for u, items in user_pos_train.items()}\nall_items = np.arange(n_items)\n\ndef sample_batch(batch_size=4096):\n    users = np.random.choice(list(user_pos_array.keys()), size=batch_size, replace=True)\n    pos_items = np.array([np.random.choice(user_pos_array[u]) for u in users])\n    neg_items = []\n    for u in users:\n        while True:\n            j = np.random.randint(0, n_items)\n            if j not in user_pos_array[u]:\n                neg_items.append(j); break\n    return users, pos_items, np.array(neg_items)\n\ng_t = torch.tensor(g, dtype=torch.float32, device=DEVICE)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T16:35:17.92723Z","iopub.execute_input":"2025-11-06T16:35:17.927543Z","iopub.status.idle":"2025-11-06T16:35:18.045365Z","shell.execute_reply.started":"2025-11-06T16:35:17.92751Z","shell.execute_reply":"2025-11-06T16:35:18.044746Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 13) Evaluation utilities (Recall/NDCG, Coverage, Grounding)","metadata":{}},{"cell_type":"code","source":"@torch.no_grad()\ndef compute_embeddings(model):\n    U, I = model(A_UI, A_II_total)\n    return U, I\n\ndef evaluate_split(model, user_pos_true, K=20):\n    model.eval()\n    U, I = compute_embeddings(model)\n    recalls, ndcgs = [], []\n    \n    # Store per-user recalls for the histogram\n    user_recalls_list = [] \n    \n    users_list = list(user_pos_true.keys())\n    for start in range(0, len(users_list), 512):\n        batch_users = users_list[start:start+512]\n        u_emb = U[batch_users]\n        scores = torch.matmul(u_emb, I.T)\n        \n        for bi, u in enumerate(batch_users):\n            train_pos = list(user_pos_train.get(u, []))\n            if train_pos:\n                scores[bi, torch.tensor(train_pos, device=DEVICE)] = -1e9\n                \n        topk = torch.topk(scores, k=K, dim=1).indices.cpu().numpy()\n        \n        for bi, u in enumerate(batch_users):\n            truth = set(user_pos_true.get(u, []))\n            if not truth: \n                user_recalls_list.append(0.0) # Add 0.0 for users with no test items\n                continue\n                \n            preds = list(topk[bi])\n            hit = len(set(preds) & truth)\n            \n            current_recall = hit / min(K, len(truth))\n            recalls.append(current_recall)\n            user_recalls_list.append(current_recall) # Store this user's recall\n            \n            dcg = 0.0\n            for rank, it in enumerate(preds, start=1):\n                if it in truth:\n                    dcg += 1.0 / math.log2(rank+1)\n            idcg = sum(1.0 / math.log2(r+1) for r in range(1, min(K, len(truth))+1))\n            ndcgs.append(dcg / idcg if idcg>0 else 0.0)\n            \n    R = float(np.mean(recalls)) if recalls else 0.0\n    N = float(np.mean(ndcgs)) if ndcgs else 0.0\n    \n    # Return the mean metrics AND the per-user recall list\n    return R, N, user_recalls_list\n\ndef coverage_at_k(model, K=20, sample_users=2000):\n    model.eval()\n    U, I = compute_embeddings(model)\n    users_list = list(user_pos_test.keys())[:sample_users]\n    seen = set()\n    for start in range(0, len(users_list), 512):\n        batch_users = users_list[start:start+512]\n        u_emb = U[batch_users]\n        scores = torch.matmul(u_emb, I.T)\n        for bi, u in enumerate(batch_users):\n            train_pos = list(user_pos_train.get(u, []))\n            if train_pos:\n                scores[bi, torch.tensor(train_pos, device=DEVICE)] = -1e9\n        topk = torch.topk(scores, k=K, dim=1).indices.cpu().numpy()\n        for row in topk: seen.update(row.tolist())\n        \n    # Return the metric AND the set of all recommended items\n    return len(seen) / n_items, seen\n\ndef grounding_at_k(model, K=20, sample_users=2000):\n    model.eval()\n    U, I = compute_embeddings(model)\n    users_list = list(user_pos_test.keys())[:sample_users]\n    \n    # Store per-user grounding scores\n    grounding_vals_list = [] \n    \n    for start in range(0, len(users_list), 512):\n        batch_users = users_list[start:start+512]\n        u_emb = U[batch_users]\n        scores = torch.matmul(u_emb, I.T)\n        for bi, u in enumerate(batch_users):\n            train_pos = list(user_pos_train.get(u, []))\n            if train_pos:\n                scores[bi, torch.tensor(train_pos, device=DEVICE)] = -1e9\n        topk = torch.topk(scores, k=K, dim=1).indices\n        \n        # Calculate mean grounding for this batch of users' top-k\n        batch_grounding = g_t[topk].mean(dim=1).cpu().numpy()\n        grounding_vals_list.extend(batch_grounding)\n        \n    G = float(np.mean(grounding_vals_list)) if grounding_vals_list else 0.0\n    \n    # Return the mean metric AND the per-user grounding list\n    return G, grounding_vals_list","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T16:35:29.127694Z","iopub.execute_input":"2025-11-06T16:35:29.128025Z","iopub.status.idle":"2025-11-06T16:35:29.142542Z","shell.execute_reply.started":"2025-11-06T16:35:29.128003Z","shell.execute_reply":"2025-11-06T16:35:29.141857Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 14) Train loop (BPR + grounding)","metadata":{}},{"cell_type":"code","source":"# 14) Train loop (BPR + grounding)\nmodel = LightGCN(n_users, n_items, d=cfg.EMBED_DIM, layers=cfg.LAYERS).to(DEVICE)\nopt = torch.optim.Adam(model.parameters(), lr=cfg.LR, weight_decay=cfg.WEIGHT_DECAY)\n\ndef train(epochs=cfg.EPOCHS, log_every=200):\n    # List to store all performance data\n    performance_data = []\n    \n    for ep in range(1, epochs+1):\n        model.train()\n        losses = []\n        steps = max(1, len(train_df)//cfg.BATCH_SIZE)\n        for it in range(steps):\n            users, pos_items, neg_items = sample_batch(cfg.BATCH_SIZE)\n            users_t = torch.tensor(users, dtype=torch.long, device=DEVICE)\n            pos_t   = torch.tensor(pos_items, dtype=torch.long, device=DEVICE)\n            neg_t   = torch.tensor(neg_items, dtype=torch.long, device=DEVICE)\n\n            # THIS IS THE NEW, CORRECTED CODE\n            Ue, Ie = model(A_UI, A_II_total)\n            u_e = Ue[users_t]; i_pos_e = Ie[pos_t]; i_neg_e = Ie[neg_t]\n            \n            # 1. Get the original ranking scores\n            pos_score = (u_e * i_pos_e).sum(-1)\n            neg_score = (u_e * i_neg_e).sum(-1)\n            \n            # 2. Get the \"grounding bonus\" for each item\n            # LAMBDA_G controls how important coherence is\n            pos_bonus = cfg.LAMBDA_G * g_t[pos_t]\n            neg_bonus = cfg.LAMBDA_G * g_t[neg_t]\n            \n            # 3. Calculate the new loss\n            # We want (pos_score + pos_bonus) to be greater than (neg_score + neg_bonus)\n            loss = -F.logsigmoid( (pos_score + pos_bonus) - (neg_score + neg_bonus) ).mean()\n            \n            opt.zero_grad(set_to_none=True)\n            # ... rest of loop ...\n            loss.backward()\n            opt.step()\n\n            losses.append(loss.item())\n            if (it+1)%log_every==0:\n                print(f\"ep{ep} step{it+1}/{steps} loss={np.mean(losses):.4f}\")\n\n        # --- Epoch Eval ---\n        # Get metrics AND raw data from the new eval functions\n        Rv, Nv, _ = evaluate_split(model, user_pos_val, K=cfg.K_EVAL)\n        Rt, Nt, user_recalls = evaluate_split(model, user_pos_test, K=cfg.K_EVAL)\n        Cov, rec_items = coverage_at_k(model, K=cfg.K_EVAL)\n        Grd, grounding_vals = grounding_at_k(model, K=cfg.K_EVAL)\n        \n        # Print logs as before\n        print(f\"[E{ep}] Val R@{cfg.K_EVAL}={Rv:.4f} N@{cfg.K_EVAL}={Nv:.4f} | \"\n              f\"Test R@{cfg.K_EVAL}={Rt:.4f} N@{cfg.K_EVAL}={Nt:.4f} | \"\n              f\"Cov@{cfg.K_EVAL}={Cov:.4f} | Ground@{cfg.K_EVAL}={Grd:.3f}\")\n        \n        # Store all data in a dictionary\n        epoch_data = {\n            \"Epoch\": ep,\n            \"Val_Recall@20\": Rv,\n            \"Val_NDCG@20\": Nv,\n            \"Test_Recall@20\": Rt,\n            \"Test_NDCG@20\": Nt,\n            \"Coverage@20\": Cov,\n            \"Grounding@20\": Grd,\n            \"user_recalls\": user_recalls,      # Full list of recalls for all test users\n            \"grounding_vals\": grounding_vals,  # Full list of grounding scores for all test users\n            \"rec_items\": rec_items             # Set of all recommended items\n        }\n        performance_data.append(epoch_data)\n\n    # Return the full history\n    return performance_data\n\n# NOTE: The 'train()' call that was at the bottom of your image \n# is now moved to the new 'RESULT SECTION' cell.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T16:35:38.690567Z","iopub.execute_input":"2025-11-06T16:35:38.691304Z","iopub.status.idle":"2025-11-06T16:35:38.772285Z","shell.execute_reply.started":"2025-11-06T16:35:38.69128Z","shell.execute_reply":"2025-11-06T16:35:38.771678Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 15) Inference & visualization for one user","metadata":{}},{"cell_type":"code","source":"from PIL import Image\n\n@torch.no_grad()\ndef recommend_for_user(u, K=10, show=True):\n    model.eval()\n    U, I = compute_embeddings(model)\n    scores = torch.matmul(U[u:u+1], I.T).squeeze(0)\n    train_pos = list(user_pos_train.get(u, []))\n    if train_pos:\n        scores[torch.tensor(train_pos, device=DEVICE)] = -1e9\n    topk = torch.topk(scores, k=K).indices.cpu().numpy().tolist()\n    if show:\n        cols = 5; rows = math.ceil(K/cols)\n        plt.figure(figsize=(3*cols,3*rows))\n        for idx, it in enumerate(topk):\n            p = img_paths[it]\n            try:\n                with Image.open(p) as im: im = im.convert(\"RGB\")\n            except:\n                im = Image.new(\"RGB\", (224,224), (200,200,200))\n            plt.subplot(rows, cols, idx+1)\n            plt.imshow(im); plt.axis(\"off\")\n            title = (texts[it][:30]+\"...\") if len(texts[it])>30 else texts[it]\n            plt.title(f\"iid={it}  g={g[it]:.2f}\\n{title}\", fontsize=9)\n        plt.tight_layout(); plt.show()\n    return topk\n\nu_example = 3762\nprint(\"User:\", u_example)\n_ = recommend_for_user(u_example, K=10, show=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T16:38:20.522207Z","iopub.execute_input":"2025-11-06T16:38:20.52294Z","iopub.status.idle":"2025-11-06T16:38:23.393831Z","shell.execute_reply.started":"2025-11-06T16:38:20.522915Z","shell.execute_reply":"2025-11-06T16:38:23.393022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from matplotlib.offsetbox import OffsetImage, AnnotationBbox\nfrom PIL import Image\nimport networkx as nx\nimport matplotlib.pyplot as plt\nimport random\n\ndef add_image_to_node(ax, pos, node, path, zoom=0.15):\n    try:\n        im = Image.open(path).convert(\"RGB\")\n        img = OffsetImage(im, zoom=zoom)\n        ab = AnnotationBbox(img, pos[node], frameon=False)\n        ax.add_artist(ab)\n    except Exception as e:\n        print(f\"Error adding image for {node}: {e}\")\n\ndef visualize_user_neighborhood_with_images(u, max_items=6):\n    items = list(user_pos_train.get(u, []))[:max_items]\n    G = nx.Graph()\n    G.add_node(f\"User_{u}\", type='user')\n    for it in items:\n        G.add_node(f\"Item_{it}\", type='item')\n        G.add_edge(f\"User_{u}\", f\"Item_{it}\")\n        for j in img_ids[it][:2]:\n            G.add_node(f\"Sim_{j}\", type='similar')\n            G.add_edge(f\"Item_{it}\", f\"Sim_{j}\")\n    \n    pos = nx.spring_layout(G, seed=42)\n    fig, ax = plt.subplots(figsize=(12,8))\n    nx.draw(G, pos, with_labels=False, node_color=\"white\", edge_color=\"gray\", node_size=100, ax=ax)\n    \n    add_image_to_node(ax, pos, f\"User_{u}\", random.choice(img_paths))\n    \n    for n in G.nodes():\n        if n.startswith(\"Item_\"):\n            iid = int(n.split(\"_\")[1])\n            add_image_to_node(ax, pos, n, img_paths[iid])\n    \n    plt.title(f\"Visual Neighborhood of User {u}\", fontsize=14)\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T16:38:45.057458Z","iopub.execute_input":"2025-11-06T16:38:45.057754Z","iopub.status.idle":"2025-11-06T16:38:45.066111Z","shell.execute_reply.started":"2025-11-06T16:38:45.057733Z","shell.execute_reply":"2025-11-06T16:38:45.065291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@torch.no_grad()\ndef export_embeddings(model):\n    U, I = compute_embeddings(model)\n    return U.cpu().numpy(), I.cpu().numpy()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T17:59:37.173237Z","iopub.execute_input":"2025-11-06T17:59:37.173709Z","iopub.status.idle":"2025-11-06T17:59:37.177525Z","shell.execute_reply.started":"2025-11-06T17:59:37.173687Z","shell.execute_reply":"2025-11-06T17:59:37.176657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 📊 RESULT SECTION\n# This cell will now train the model AND generate the plots.\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nimport random\nfrom sklearn.manifold import TSNE\nimport pickle\n# 1️⃣ Initialize Model and Optimizer\nmodel = LightGCN(n_users, n_items, d=cfg.EMBED_DIM, layers=cfg.LAYERS).to(DEVICE)\nopt = torch.optim.Adam(model.parameters(), lr=cfg.LR, weight_decay=cfg.WEIGHT_DECAY)\n\n# 2️⃣ Run Training and get performance data\n# Set epochs to a reasonable number, e.g., 25+\ncfg.EPOCHS = 20\nperformance_data = train(epochs=cfg.EPOCHS)\n\nckpt_dir = os.path.join(cfg.WORK_DIR, \"checkpoints\")\nos.makedirs(ckpt_dir, exist_ok=True)\n\ntorch.save({\"state_dict\": model.state_dict(), \"cfg\": cfg.__dict__},\n           os.path.join(ckpt_dir, \"lightgcn.pt\"))\n\nwith open(os.path.join(ckpt_dir, \"user_le.pkl\"), \"wb\") as f:\n    pickle.dump(user_le, f)\nwith open(os.path.join(ckpt_dir, \"item_le.pkl\"), \"wb\") as f:\n    pickle.dump(item_le, f)\n\n# Optional artifacts for fast reuse\nU_emb, I_emb = export_embeddings(model)  # from the code you have\nnp.save(os.path.join(ckpt_dir, \"U_emb.npy\"), U_emb)\nnp.save(os.path.join(ckpt_dir, \"I_emb.npy\"), I_emb)\nnp.save(os.path.join(ckpt_dir, \"Z_img.npy\"), Z_img)\nnp.save(os.path.join(ckpt_dir, \"Z_txt.npy\"), Z_txt)\nnp.save(os.path.join(ckpt_dir, \"g.npy\"), g)\n\n\n# 3️⃣ Process the data from the final epoch for plotting\nfinal_epoch_data = performance_data[-1]\nperformance_table = pd.DataFrame(performance_data) # For the learning curve\n\n# Get data from the *final* epoch\nuser_recalls = final_epoch_data[\"user_recalls\"]\ngrounding_vals = final_epoch_data[\"grounding_vals\"]\nrec_items = list(final_epoch_data[\"rec_items\"]) # Convert set to list\n\n# Ensure recall and grounding lists are of same length for the scatter plot\nmin_len = min(len(user_recalls), len(grounding_vals))\nuser_recalls_sync = user_recalls[:min_len]\ngrounding_vals_sync = grounding_vals[:min_len]\n\n\n# 4️⃣ Generate all plots\n\nprint(\"\\n--- 📈 Generating Final Plots ---\")\n\n# Plot 1: Learning Curves (Recall, NDCG, Grounding)\nplt.figure(figsize=(7,4))\nplt.plot(performance_table[\"Epoch\"], performance_table[\"Test_Recall@20\"], '-o', label='Recall@20')\nplt.plot(performance_table[\"Epoch\"], performance_table[\"Test_NDCG@20\"], '-s', label='NDCG@20')\nplt.plot(performance_table[\"Epoch\"], performance_table[\"Grounding@20\"], '-^', label='Grounding')\nplt.xlabel(\"Epoch\")\nplt.ylabel(\"Score\")\nplt.title(\"Learning Trend of LightGCN (BPR + Grounding)\")\nplt.legend()\nplt.grid(True, alpha=0.3)\nplt.tight_layout()\nplt.show()\n\n# Plot 2: Final Metrics Comparison (Bar Plot)\nfinal_metrics = {\n    \"Recall@20\": final_epoch_data[\"Test_Recall@20\"],\n    \"NDCG@20\": final_epoch_data[\"Test_NDCG@20\"],\n    \"Coverage@20\": final_epoch_data[\"Coverage@20\"],\n    \"Grounding@20\": final_epoch_data[\"Grounding@20\"]\n}\nplt.figure(figsize=(6, 4))\nsns.barplot(x=list(final_metrics.keys()), y=list(final_metrics.values()), palette=\"viridis\")\nplt.title(\"Final Model Evaluation Metrics (Epoch \" + str(final_epoch_data[\"Epoch\"]) + \")\")\nplt.ylabel(\"Score\")\nplt.tight_layout()\nplt.show()\n\n# Plot 3: Recall Distribution Across Users\nplt.figure(figsize=(7,4))\nsns.histplot(user_recalls, bins=30, kde=True, color='skyblue')\nplt.title(\"Distribution of Recall@20 Across Users\")\nplt.xlabel(\"Recall@20\")\nplt.ylabel(\"User Count\")\nplt.tight_layout()\nplt.show()\n\n# Plot 4: t-SNE of Item Embeddings (using the pre-computed Z_img)\nsubset = np.random.choice(len(Z_img), size=min(2000, len(Z_img)), replace=False)\ntsne = TSNE(n_components=2, perplexity=30, random_state=42, n_init=1, learning_rate='auto')\nproj = tsne.fit_transform(Z_img[subset])\n\nplt.figure(figsize=(8,6))\nsns.scatterplot(x=proj[:,0], y=proj[:,1],\n                hue=art.iloc[subset][\"product_type_name\"],\n                legend=False, s=40, alpha=0.8)\nplt.title(\"t-SNE of Item Embeddings (CLIP-Based)\")\nplt.xlabel(\"Dim 1\")\nplt.ylabel(\"Dim 2\")\nplt.tight_layout()\nplt.show()\n\n# Plot 5: Grounding vs Recall (Correlation Plot)\nplt.figure(figsize=(7,5))\nsns.scatterplot(x=grounding_vals_sync, y=user_recalls_sync, alpha=0.7)\nplt.xlabel(\"Avg Grounding Score of Recommendations\")\nplt.ylabel(\"User Recall@20\")\nplt.title(\"Grounding vs Recommendation Accuracy\")\nplt.grid(True, alpha=0.3)\nplt.tight_layout()\nplt.show()\n\n# Plot 6: Top Recommended Categories\nif rec_items:\n    types = art.loc[rec_items][\"product_type_name\"].value_counts().head(10)\n    plt.figure(figsize=(7, 5))\n    sns.barplot(y=types.index, x=types.values, palette=\"crest\")\n    plt.title(\"Top Recommended Product Categories\")\n    plt.xlabel(\"Recommendation Frequency\")\n    plt.tight_layout()\n    plt.show()\nelse:\n    print(\"No items recommended, skipping 'Top Categories' plot.\")\n\n# Plot 7: User–Item Graph Visualization (Example from original code)\ntry:\n    visualize_user_neighborhood(u=random.randint(0, n_users-1))\nexcept NameError:\n    print(\"Skipping `visualize_user_neighborhood` (function not found).\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T17:59:39.577477Z","iopub.execute_input":"2025-11-06T17:59:39.577764Z","iopub.status.idle":"2025-11-06T18:18:49.608774Z","shell.execute_reply.started":"2025-11-06T17:59:39.577743Z","shell.execute_reply":"2025-11-06T18:18:49.607841Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\n\n# Path where files were saved\nckpt_dir = os.path.join(cfg.WORK_DIR, \"checkpoints\")\n\n# Create a zip archive\nshutil.make_archive(\"/kaggle/working/lightgcn_checkpoints\", 'zip', ckpt_dir)\nprint(\"✅ Zipped all model files as lightgcn_checkpoints.zip\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T18:21:45.979652Z","iopub.execute_input":"2025-11-06T18:21:45.97994Z","iopub.status.idle":"2025-11-06T18:22:09.843949Z","shell.execute_reply.started":"2025-11-06T18:21:45.979924Z","shell.execute_reply":"2025-11-06T18:22:09.843266Z"}},"outputs":[],"execution_count":null}]}