{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":31254,"databundleVersionId":3103714}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 1. Install dependencies","metadata":{}},{"cell_type":"code","source":"!pip install -q open_clip_torch>=2.20.0 transformers torch torchvision pillow tqdm pandas numpy ftfy\nimport importlib.metadata\noc_ver = importlib.metadata.version(\"open_clip_torch\")\nprint(f\"open_clip_torch installed: {oc_ver}\")\n\nfrom packaging.version import Version\nassert Version(oc_ver) >= Version(\"2.20.0\"), \\\n    f\"open_clip_torch {oc_ver} is too old — restart kernel and re-run Cell 1\"\nprint(\"✔ Version check passed\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T10:48:00.114757Z","iopub.execute_input":"2026-06-06T10:48:00.115190Z","iopub.status.idle":"2026-06-06T10:48:05.413867Z","shell.execute_reply.started":"2026-06-06T10:48:00.115161Z","shell.execute_reply":"2026-06-06T10:48:05.413101Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Import and config","metadata":{}},{"cell_type":"code","source":"import os\nimport re\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom tqdm import tqdm\n\nimport torch\nimport open_clip\nfrom PIL import Image\n\n# Paths\nINPUT_DIR   = Path(\"/kaggle/input/competitions/h-and-m-personalized-fashion-recommendations\")\nIMAGES_DIR  = INPUT_DIR / \"images\"\nOUTPUT_DIR  = Path(\"/kaggle/working/data/hm\")\nOUTPUT_DIR.mkdir(parents=True, exist_ok=True)\n\n# Temporal Window\nANCHOR_DATE     = pd.Timestamp(\"2020-09-22\")\nWINDOW_START    = pd.Timestamp(\"2020-07-28\")   # trailing 8 weeks\nVALID_START     = pd.Timestamp(\"2020-09-09\")   # week 7\nTEST_START      = pd.Timestamp(\"2020-09-16\")   # week 8\n\n# K-Core\nK_CORE = 5\n\n# Feature Model\nMARQO_MODEL   = \"hf-hub:Marqo/marqo-fashionSigLIP\"\nEMBED_DIM     = 768\nBATCH_SIZE    = 32\nSUPPORTED_EXT = {\".jpg\", \".jpeg\", \".png\", \".webp\"}\n\nDEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nprint(f\"Device      : {DEVICE}\")\nprint(f\"MARQO_MODEL : {MARQO_MODEL}\")\nprint(f\"Output dir  : {OUTPUT_DIR}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T10:48:05.415300Z","iopub.execute_input":"2026-06-06T10:48:05.416009Z","iopub.status.idle":"2026-06-06T10:48:31.244425Z","shell.execute_reply.started":"2026-06-06T10:48:05.415931Z","shell.execute_reply":"2026-06-06T10:48:31.243706Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Temporal Windowing","metadata":{}},{"cell_type":"code","source":"trans_raw = pd.read_csv(\n    INPUT_DIR / \"transactions_train.csv\",\n    dtype={\"article_id\": str, \"customer_id\": str},\n    parse_dates=[\"t_dat\"],\n)\nprint(f\"  Raw transactions          : {trans_raw.shape}\")\n\ndef assign_split(date):\n    if date >= TEST_START:   return 2.0\n    elif date >= VALID_START: return 1.0\n    else:                     return 0.0\n\ntrans = trans_raw[\n    (trans_raw[\"t_dat\"] >= WINDOW_START) &\n    (trans_raw[\"t_dat\"] <= ANCHOR_DATE)\n].copy()\ntrans[\"x_label\"] = trans[\"t_dat\"].apply(assign_split)\n\nprint(f\"  Windowed transactions     : {len(trans):,}\")\nprint(f\"  Date range                : {trans['t_dat'].min().date()} → {trans['t_dat'].max().date()}\")\nprint(f\"  Split counts:\\n{trans['x_label'].value_counts().sort_index()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T10:48:31.245294Z","iopub.execute_input":"2026-06-06T10:48:31.245796Z","iopub.status.idle":"2026-06-06T10:49:30.259049Z","shell.execute_reply.started":"2026-06-06T10:48:31.245770Z","shell.execute_reply":"2026-06-06T10:49:30.257943Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Image scan","metadata":{}},{"cell_type":"code","source":"print(\"  Scanning image directory …\")\nvalid_image_paths: dict[str, Path] = {}\n\nfor img_path in IMAGES_DIR.rglob(\"*\"):\n    if img_path.suffix.lower() not in SUPPORTED_EXT:\n        continue\n    stem = img_path.stem\n    if not re.fullmatch(r\"\\d{10}\", stem):\n        continue\n    if stem not in valid_image_paths:\n        valid_image_paths[stem] = img_path\n\nvalid_image_ids: set[str] = set(valid_image_paths.keys())\nprint(f\"  Images found on disk      : {len(valid_image_ids):,}\")\n\nfrom collections import Counter\next_counts = Counter(p.suffix.lower() for p in valid_image_paths.values())\nfor ext, cnt in sorted(ext_counts.items()):\n    print(f\"    {ext:<8}: {cnt:,}\")\n\n# Normalise article_id in transactions (pad only, NO drop)\ntrans[\"article_id\"] = trans[\"article_id\"].astype(str).str.zfill(10)\n\ncoverage = trans[\"article_id\"].isin(valid_image_ids).mean() * 100\nprint(f\"\\n  Transaction image coverage: {coverage:.1f}% of rows have a physical image\")\nprint(f\"  ⚠ Items without images will be IMPUTED in Cell 6 (not dropped)\")\n\n# Load full articles metadata (no image filter)\narticles = pd.read_csv(\n    INPUT_DIR / \"articles.csv\",\n    dtype={\"article_id\": str},\n)\narticles[\"article_id\"] = articles[\"article_id\"].astype(str).str.zfill(10)\nprint(f\"  Articles loaded           : {len(articles):,}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T10:49:30.261090Z","iopub.execute_input":"2026-06-06T10:49:30.261411Z","iopub.status.idle":"2026-06-06T10:53:01.408433Z","shell.execute_reply.started":"2026-06-06T10:49:30.261385Z","shell.execute_reply":"2026-06-06T10:53:01.407624Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. K-Core Filtering & Re-indexing","metadata":{}},{"cell_type":"code","source":"# 3a. Deduplication\nbefore_dedup = len(trans)\ntrans = trans.drop_duplicates(\n    subset=[\"customer_id\", \"article_id\", \"t_dat\"], keep=\"first\"\n).copy()\nprint(f\"  Rows before dedup         : {before_dedup:,}\")\nprint(f\"  Rows after  dedup         : {len(trans):,}\")\n\n# 3b. Iterative K-Core\nprint(f\"\\n  Applying iterative {K_CORE}-core filtering …\")\niteration = 0\nwhile True:\n    iteration += 1\n    user_counts = trans[\"customer_id\"].value_counts()\n    item_counts = trans[\"article_id\"].value_counts()\n    valid_users = user_counts[user_counts >= K_CORE].index\n    valid_items = item_counts[item_counts >= K_CORE].index\n\n    filtered = trans[\n        trans[\"customer_id\"].isin(valid_users) &\n        trans[\"article_id\"].isin(valid_items)\n    ]\n    if len(filtered) == len(trans):\n        print(f\"  Converged after {iteration} iteration(s).\")\n        break\n    trans = filtered.copy()\n    print(f\"  Iter {iteration:3d}: {trans['customer_id'].nunique():,} users | \"\n          f\"{trans['article_id'].nunique():,} items | {len(trans):,} rows\")\n\ntrans = trans.reset_index(drop=True)\n\n# 3c. ID Re-indexing\nunique_users = sorted(trans[\"customer_id\"].unique())\nunique_items = sorted(trans[\"article_id\"].unique())\n\nuser2id = {uid: idx for idx, uid in enumerate(unique_users)}\nitem2id = {aid: idx for idx, aid in enumerate(unique_items)}\nid2article = {v: k for k, v in item2id.items()}    # itemID → article_id\n\ntrans[\"userID\"] = trans[\"customer_id\"].map(user2id)\ntrans[\"itemID\"] = trans[\"article_id\"].map(item2id)\n\nN_USERS = len(unique_users)\nN_ITEMS = len(unique_items)\n\n# 3d. Compute has_image mask\nhas_image = np.array(\n    [id2article[i] in valid_image_ids for i in range(N_ITEMS)],\n    dtype=bool,\n)\nmissing_mask = ~has_image\nN_VALID   = has_image.sum()\nN_MISSING = missing_mask.sum()\n\nprint(f\"\\n  ── Final counts after K-core ──────────────────\")\nprint(f\"  Users  (U)                : {N_USERS:,}\")\nprint(f\"  Items  (I)                : {N_ITEMS:,}\")\nprint(f\"    ├── with image on disk  : {N_VALID:,}  ({N_VALID/N_ITEMS*100:.1f}%)\")\nprint(f\"    └── missing image       : {N_MISSING:,}  ({N_MISSING/N_ITEMS*100:.1f}%)\")\nprint(f\"  Interactions              : {len(trans):,}\")\nprint(f\"  Train / Val / Test        : \"\n      f\"{int((trans.x_label==0).sum()):,} / \"\n      f\"{int((trans.x_label==1).sum()):,} / \"\n      f\"{int((trans.x_label==2).sum()):,}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T10:53:01.409328Z","iopub.execute_input":"2026-06-06T10:53:01.409541Z","iopub.status.idle":"2026-06-06T10:53:12.900026Z","shell.execute_reply.started":"2026-06-06T10:53:01.409518Z","shell.execute_reply":"2026-06-06T10:53:12.899127Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. Visual Feature Extraction (FashionCLIP)","metadata":{}},{"cell_type":"code","source":"# 1. Load model\nprint(f\"  Loading model: {MARQO_MODEL}\")\n_model, _, _preprocess = open_clip.create_model_and_transforms(MARQO_MODEL)\n_model = _model.to(DEVICE).eval()\n_tokenizer = open_clip.get_tokenizer(MARQO_MODEL)\nprint(f\"  ✔ Model loaded on {DEVICE}\")\n\ndef load_image(path: Path) -> Image.Image:\n    return Image.open(path).convert(\"RGB\")\n\n# 2. Extract embeddings for items WITH image\nvalid_indices = np.where(has_image)[0].tolist()\nimage_feat    = np.zeros((N_ITEMS, EMBED_DIM), dtype=np.float32)\n\nprint(f\"  Encoding {N_VALID:,} images (batch_size={BATCH_SIZE}) …\")\nwith torch.no_grad():\n    for batch_start in tqdm(\n        range(0, len(valid_indices), BATCH_SIZE), desc=\"  Vision batches\"\n    ):\n        batch_ids  = valid_indices[batch_start : batch_start + BATCH_SIZE]\n        batch_imgs = [\n            load_image(valid_image_paths[id2article[i]]) for i in batch_ids\n        ]\n        pixel_values = torch.stack(\n            [_preprocess(img) for img in batch_imgs]\n        ).to(DEVICE)\n\n        feats = _model.encode_image(pixel_values, normalize=True)  # (B, 768)\n        image_feat[batch_ids] = feats.cpu().float().numpy()\n\n# 3. Category Mean Imputation for missing items\nif N_MISSING > 0:\n    print(f\"\\n  Imputing {N_MISSING:,} missing image vectors …\")\n\n    items_df = pd.DataFrame({\n        \"article_id\": [id2article[i] for i in range(N_ITEMS)],\n        \"itemID\"    : list(range(N_ITEMS)),\n    }).merge(articles[[\"article_id\", \"product_type_name\"]], on=\"article_id\", how=\"left\")\n    items_df[\"product_type_name\"] = items_df[\"product_type_name\"].fillna(\"__unknown__\")\n\n    global_mean = image_feat[has_image].mean(axis=0)\n    global_mean /= np.linalg.norm(global_mean) + 1e-10\n\n    valid_df = items_df[items_df[\"itemID\"].isin(valid_indices)]\n    category_mean: dict[str, np.ndarray] = {}\n    for cat, group in valid_df.groupby(\"product_type_name\"):\n        vecs = image_feat[group[\"itemID\"].values]\n        mean = vecs.mean(axis=0)\n        mean /= np.linalg.norm(mean) + 1e-10\n        category_mean[cat] = mean\n\n    missing_df = items_df[missing_mask[items_df[\"itemID\"].values]]\n    for _, row in missing_df.iterrows():\n        iid = int(row[\"itemID\"])\n        image_feat[iid] = category_mean.get(row[\"product_type_name\"], global_mean)\n\n    zero_mask = (image_feat == 0).all(axis=1)\n    if zero_mask.any():\n        print(f\"  ⚠ {zero_mask.sum()} zero rows → applying global mean.\")\n        image_feat[zero_mask] = global_mean\n\nprint(f\"\\n  image_feat shape          : {image_feat.shape}\")\nprint(f\"  dtype                     : {image_feat.dtype}\")\nprint(f\"  Sample L2-norm (row 0)    : {np.linalg.norm(image_feat[0]):.6f}\")\nprint(f\"  Sample L2-norm (row -1)   : {np.linalg.norm(image_feat[-1]):.6f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T10:53:12.901756Z","iopub.execute_input":"2026-06-06T10:53:12.902103Z","iopub.status.idle":"2026-06-06T11:11:21.802206Z","shell.execute_reply.started":"2026-06-06T10:53:12.902077Z","shell.execute_reply":"2026-06-06T11:11:21.801206Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. Textual Feature Extraction","metadata":{}},{"cell_type":"code","source":"TEXT_COLS = [\n    \"prod_name\",\n    \"colour_group_name\",\n    \"product_type_name\",\n    \"detail_desc\",\n]\n\n# Build article_id → text string lookup\narticles_indexed = (\n    articles[[\"article_id\"] + TEXT_COLS]\n    .drop_duplicates(\"article_id\")\n    .set_index(\"article_id\")\n)\n\ndef build_text(article_id: str) -> str:\n    try:\n        row = articles_indexed.loc[article_id]\n    except KeyError:\n        return \"fashion item\"\n    parts = [\n        str(row[col]).strip()\n        for col in TEXT_COLS\n        if pd.notna(row.get(col)) and str(row.get(col, \"\")).strip()\n    ]\n    return \" . \".join(parts) if parts else \"fashion item\"\n\n# Build in strict itemID order (0 … I-1) for row alignment\ntext_strings = [build_text(id2article[i]) for i in range(N_ITEMS)]\nprint(f\"  Example (itemID=0)  : {text_strings[0][:100]}\")\nprint(f\"  Example (itemID=1)  : {text_strings[1][:100]}\")\n\n# Tokenise & encode in batches\ntext_feat = np.zeros((N_ITEMS, EMBED_DIM), dtype=np.float32)\n\nprint(f\"\\n  Encoding {N_ITEMS:,} texts (batch_size={BATCH_SIZE}) …\")\nwith torch.no_grad():\n    for start in tqdm(range(0, N_ITEMS, BATCH_SIZE), desc=\"  Text batches\"):\n        end         = min(start + BATCH_SIZE, N_ITEMS)\n        batch_texts = text_strings[start:end]\n\n        # open_clip tokenizer: handles truncation to model context length,\n        # returns (B, context_length) int64 tensor\n        tokens = _tokenizer(batch_texts).to(DEVICE)\n\n        # encode_text with normalize=True → (B, 768) unit-norm float32\n        feats  = _model.encode_text(tokens, normalize=True)\n        text_feat[start:end] = feats.cpu().float().numpy()\n\nprint(f\"\\n  text_feat shape     : {text_feat.shape}\")\nprint(f\"  dtype               : {text_feat.dtype}\")\nprint(f\"  Sample L2-norm (0)  : {np.linalg.norm(text_feat[0]):.6f}\")\n\n# Free GPU memory\ndel _model, _tokenizer, _preprocess\ntorch.cuda.empty_cache()\nprint(\"  GPU memory released.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T11:11:21.803422Z","iopub.execute_input":"2026-06-06T11:11:21.803824Z","iopub.status.idle":"2026-06-06T11:12:53.355726Z","shell.execute_reply.started":"2026-06-06T11:11:21.803797Z","shell.execute_reply":"2026-06-06T11:12:53.355056Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 8 . Save output","metadata":{}},{"cell_type":"code","source":"# 1. hm.inter\ninter_path = OUTPUT_DIR / \"hm.inter\"\ninter_df   = trans[[\"userID\", \"itemID\", \"x_label\"]].copy()\n\nwith open(inter_path, \"w\", encoding=\"utf-8\") as f:\n    f.write(\"userID:token\\titemID:token\\tx_label:float\\n\")\n    for row in inter_df.itertuples(index=False):\n        f.write(f\"{row.userID}\\t{row.itemID}\\t{row.x_label}\\n\")\n\nwith open(inter_path) as f:\n    lines = f.readlines()\nprint(f\"  ✔ hm.inter                : {len(lines)-1:,} rows\")\nprint(f\"    Header                  : {lines[0].strip()}\")\nprint(f\"    First row               : {lines[1].strip()}\")\n\n# 2. image_feat.npy\nnp.save(OUTPUT_DIR / \"image_feat.npy\", image_feat)\nprint(f\"\\n  ✔ image_feat.npy          : {image_feat.shape}  {image_feat.dtype}\")\n\n# 3. text_feat.npy\nnp.save(OUTPUT_DIR / \"text_feat.npy\", text_feat)\nprint(f\"  ✔ text_feat.npy           : {text_feat.shape}  {text_feat.dtype}\")\n\n# 4. u_id_mapping.csv\nu_map_df = (\n    pd.DataFrame(list(user2id.items()), columns=[\"customer_id\", \"userID\"])\n    .sort_values(\"userID\")\n    .reset_index(drop=True)\n)\nu_map_df.to_csv(OUTPUT_DIR / \"u_id_mapping.csv\", index=False)\nprint(f\"  ✔ u_id_mapping.csv        : {len(u_map_df):,} rows\")\n\n# 5. i_id_mapping.csv\ni_map_df = (\n    pd.DataFrame(list(item2id.items()), columns=[\"article_id\", \"itemID\"])\n    .sort_values(\"itemID\")\n    .reset_index(drop=True)\n)\ni_map_df.to_csv(OUTPUT_DIR / \"i_id_mapping.csv\", index=False)\nprint(f\"  ✔ i_id_mapping.csv        : {len(i_map_df):,} rows\")\n\n# Alignment assertion\nassert image_feat.shape == text_feat.shape == (N_ITEMS, EMBED_DIM), \\\n    \"Shape mismatch between feature matrices!\"\nprint(f\"\\n  ✔ Alignment check passed  : both feature arrays are ({N_ITEMS}, {EMBED_DIM})\")\n\n# Final summary\nprint(\"\\n\" + \"=\" * 60)\nprint(\"PIPELINE COMPLETE — FINAL SUMMARY\")\nprint(\"=\" * 60)\nprint(f\"  Users  (U)        : {N_USERS:,}\")\nprint(f\"  Items  (I)        : {N_ITEMS:,}\")\nprint(f\"    with image      : {N_VALID:,}  ({N_VALID/N_ITEMS*100:.1f}%)\")\nprint(f\"    imputed         : {N_MISSING:,}  ({N_MISSING/N_ITEMS*100:.1f}%)\")\nprint(f\"  Interactions      : {len(inter_df):,}\")\nprint(f\"  image_feat        : {image_feat.shape}  float32\")\nprint(f\"  text_feat         : {text_feat.shape}  float32\")\nprint(f\"\\n  Output files:\")\nfor p in sorted(OUTPUT_DIR.iterdir()):\n    print(f\"    {p.name:<25} {p.stat().st_size/1_048_576:8.2f} MB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T11:12:53.356767Z","iopub.execute_input":"2026-06-06T11:12:53.357160Z","iopub.status.idle":"2026-06-06T11:12:55.734066Z","shell.execute_reply.started":"2026-06-06T11:12:53.357135Z","shell.execute_reply":"2026-06-06T11:12:55.733124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!zip -r hm_data.zip /kaggle/working/data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-06T11:12:55.735118Z","iopub.execute_input":"2026-06-06T11:12:55.735432Z","iopub.status.idle":"2026-06-06T11:13:04.273452Z","shell.execute_reply.started":"2026-06-06T11:12:55.735397Z","shell.execute_reply":"2026-06-06T11:13:04.272661Z"}},"outputs":[],"execution_count":null}]}