{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":126777,"databundleVersionId":15314950}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"96326190-421c-4cde-999e-4099c05751b8","cell_type":"markdown","source":"# Experiment 14 — Seed Stability Analysis\n## DINOv2-ViT-L/14 + Optimised ArcFace Head, 7 Random Seeds\n\n**Builds on:** Experiment 13 — ArcFace Hyperparameter Sweep on DINOv2  \n**Research Question:** Is the best configuration (val mAP = 0.8631) a statistically robust\nresult, or could it be a lucky random initialisation?\n\n## Motivation\nRunning the same configuration with 5–10 different seeds increases the significance of the result\nand reduces the chance of reporting a lucky run. If the standard deviation across seeds is small\n(e.g. < 0.01), the improvement over the Experiment 4 baseline (0.8447) is reliable.\nIf it is large, the result is unstable and the strength of the claim must be qualified.\n\n## Best Configuration (from Experiment 13)\n| Parameter    | Value  | Source                        |\n|--------------|--------|-------------------------------|\n| Backbone     | DINOv2-ViT-L/14 (frozen) | Experiment 4 best backbone |\n| margin       | 0.6    | Stage 1 best                  |\n| scale        | 48     | Stage 1 best                  |\n| emb_dim      | 512    | Stage 2 best                  |\n| hidden_dim   | 1024   | Stage 3 best                  |\n| dropout      | 0.3    | Stage 3 best                  |\n\n**Seeds tested:** 42, 0, 1, 7, 123, 777, 2024\n\n**Val split fixed:** seed=42 always (same 379 val images for every run).\nOnly model initialisation, DataLoader shuffle, and dropout randomness vary.\n\n","metadata":{}},{"id":"4ddd626a-dae1-48a0-973f-200f0a5e5ffb","cell_type":"markdown","source":"## 1. Setup and Imports","metadata":{}},{"id":"0b94f3c6-1834-43a2-ab1b-0b8793183a2a","cell_type":"code","source":"import os, math, random, time\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nimport timm\nfrom torchvision import transforms\nfrom PIL import Image\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom collections import defaultdict\nfrom tqdm.notebook import tqdm\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nimport wandb\nfrom kaggle_secrets import UserSecretsClient\n\nuser_secrets = UserSecretsClient()\nos.environ[\"HF_TOKEN\"]      = user_secrets.get_secret(\"hf_api\")\nos.environ[\"WANDB_API_KEY\"] = user_secrets.get_secret(\"wandb_api\")\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Device:  {device}\")\nprint(f\"PyTorch: {torch.__version__}  |  timm: {timm.__version__}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:44:22.000460Z","iopub.execute_input":"2026-03-14T15:44:22.000746Z","iopub.status.idle":"2026-03-14T15:44:34.820550Z","shell.execute_reply.started":"2026-03-14T15:44:22.000719Z","shell.execute_reply":"2026-03-14T15:44:34.819855Z"}},"outputs":[],"execution_count":null},{"id":"b81dbc76-d0ba-46fd-8d58-c66d5e8b8edf","cell_type":"markdown","source":"## 2. Configuration","metadata":{}},{"id":"c14b03a7-7706-487d-a57e-d15b7485d911","cell_type":"code","source":"# Fixed config\nBEST_CONFIG = {\n    # Backbone\n    \"backbone_model_id\": \"vit_large_patch14_dinov2.lvd142m\",\n    \"backbone_name\":     \"DINOv2-ViT-L/14\",\n    \"input_size\":        518,\n    \"backbone_dim\":      1024,          # DINOv2-ViT-L output dim\n\n    # ArcFace head — Experiment 13 best\n    \"arcface_margin\":   0.6,\n    \"arcface_scale\":    48.0,\n    \"embedding_dim\":    512,\n    \"hidden_dim\":       1024,\n    \"dropout\":          0.3,\n\n    # Training\n    \"learning_rate\":    1e-4,\n    \"weight_decay\":     1e-4,\n    \"num_epochs\":       50,\n    \"batch_size\":       32,\n    \"scheduler_factor\": 0.5,\n    \"scheduler_patience\": 5,\n}\n\n# Val split seed fixed — same 379 val images every run\nVAL_SPLIT_SEED = 42\nVAL_SPLIT      = 0.2\n\n# ── Seeds to test\nSEEDS = [42, 0, 1, 7, 123, 777, 2024]\n\n# Paths\nDATA_DIR       = Path(\"/kaggle/input/competitions/jaguar-re-id\")\nCHECKPOINT_DIR = Path(\"/kaggle/working/checkpoints\")\nCACHE_DIR      = Path(\"/kaggle/working/embeddings\")\nCHECKPOINT_DIR.mkdir(parents=True, exist_ok=True)\nCACHE_DIR.mkdir(parents=True, exist_ok=True)\n\n# Experiment 4 baseline for reference\nEXP4_BASELINE_MAP = 0.8447\nEXP13_BEST_MAP    = 0.8631   # seed=42, Stage 3 best\n\nprint(\"Seed Stability Configuration:\")\nprint(f\"  Backbone:      {BEST_CONFIG['backbone_name']}\")\nprint(f\"  margin:        {BEST_CONFIG['arcface_margin']}\")\nprint(f\"  scale:         {BEST_CONFIG['arcface_scale']}\")\nprint(f\"  emb_dim:       {BEST_CONFIG['embedding_dim']}\")\nprint(f\"  hidden_dim:    {BEST_CONFIG['hidden_dim']}\")\nprint(f\"  dropout:       {BEST_CONFIG['dropout']}\")\nprint(f\"  Val split:     {VAL_SPLIT} (fixed seed={VAL_SPLIT_SEED})\")\nprint(f\"  Seeds:         {SEEDS}  (n={len(SEEDS)})\")\nprint(f\"  Exp4 baseline: {EXP4_BASELINE_MAP}\")\nprint(f\"  Exp13 seed=42: {EXP13_BEST_MAP}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:44:34.822168Z","iopub.execute_input":"2026-03-14T15:44:34.822654Z","iopub.status.idle":"2026-03-14T15:44:34.831527Z","shell.execute_reply.started":"2026-03-14T15:44:34.822627Z","shell.execute_reply":"2026-03-14T15:44:34.830770Z"}},"outputs":[],"execution_count":null},{"id":"3f7110fc-cef8-4e03-8eb1-8b8dd0820598","cell_type":"markdown","source":"## 3. W&B Initialisation","metadata":{}},{"id":"0257ed74-46e3-4628-83a9-e0a0884eae84","cell_type":"code","source":"wandb.login()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:44:34.832626Z","iopub.execute_input":"2026-03-14T15:44:34.832872Z","iopub.status.idle":"2026-03-14T15:44:42.101802Z","shell.execute_reply.started":"2026-03-14T15:44:34.832851Z","shell.execute_reply":"2026-03-14T15:44:42.101156Z"}},"outputs":[],"execution_count":null},{"id":"7b2ea408-0271-4046-8d87-d70258910ec5","cell_type":"code","source":"run = wandb.init(\n    project=os.getenv(\"WANDB_PROJECT\", \"jaguar-reid-iota\"),\n    name=\"seed-stability-dinov2\",\n    config={\n        **BEST_CONFIG,\n        \"seeds\":          SEEDS,\n        \"val_split_seed\": VAL_SPLIT_SEED,\n        \"val_split\":      VAL_SPLIT,\n        \"exp4_baseline\":  EXP4_BASELINE_MAP,\n        \"exp13_seed42\":   EXP13_BEST_MAP,\n        \"experiment\":     \"seed-stability-dinov2\",\n    }\n)\nprint(\"W&B run initialised: seed-stability-dinov2\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:44:42.102680Z","iopub.execute_input":"2026-03-14T15:44:42.103182Z","iopub.status.idle":"2026-03-14T15:44:49.911076Z","shell.execute_reply.started":"2026-03-14T15:44:42.103149Z","shell.execute_reply":"2026-03-14T15:44:49.910409Z"}},"outputs":[],"execution_count":null},{"id":"296f1679-c53a-444b-b7dc-3baccd72cd04","cell_type":"markdown","source":"## 4. Load Data and Fixed Val Split","metadata":{}},{"id":"f3669c86-9bfd-41fb-8a57-567d449b0e84","cell_type":"code","source":"train_df = pd.read_csv(DATA_DIR / \"train.csv\")\nlabel_encoder = LabelEncoder()\ntrain_df[\"label_encoded\"] = label_encoder.fit_transform(train_df[\"ground_truth\"])\nnum_classes = len(label_encoder.classes_)\n\n# Val split is ALWAYS fixed at seed=42\n# This ensures we measure the same 379 val images across all seed runs.\n# Only model initialisation + DataLoader shuffle change per seed.\ntrain_data, val_data = train_test_split(\n    train_df,\n    test_size    = VAL_SPLIT,\n    random_state = VAL_SPLIT_SEED,\n    stratify     = train_df[\"ground_truth\"],\n)\n\nval_labels  = val_data[\"ground_truth\"].values\nval_encoded = val_data[\"label_encoded\"].values\n\ntrain_image_paths = [DATA_DIR / \"train\" / \"train\" / fn\n                     for fn in train_data[\"filename\"].astype(str)]\nval_image_paths   = [DATA_DIR / \"train\" / \"train\" / fn\n                     for fn in val_data[\"filename\"].astype(str)]\n\nprint(f\"Dataset: {len(train_df)} total | {len(train_data)} train | {len(val_data)} val\")\nprint(f\"Identities: {num_classes}  |  Val split seed: {VAL_SPLIT_SEED} (fixed)\")\nassert set(train_data[\"ground_truth\"].unique()) == set(val_data[\"ground_truth\"].unique()),     \"Identity leak!\"\nprint(\"All identities present in both sets ✓\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:44:49.912750Z","iopub.execute_input":"2026-03-14T15:44:49.913425Z","iopub.status.idle":"2026-03-14T15:44:49.961227Z","shell.execute_reply.started":"2026-03-14T15:44:49.913388Z","shell.execute_reply":"2026-03-14T15:44:49.960483Z"}},"outputs":[],"execution_count":null},{"id":"99bb4d4b-6f38-4206-94ea-d3af9c6afa02","cell_type":"markdown","source":"## 5. Load DINOv2 Backbone and Cache Embeddings\n\nThe backbone is loaded once and frozen. Embeddings are extracted and cached to disk.\nAll 7 seed runs reuse the same cached embeddings — only the projection head changes per seed.\n","metadata":{}},{"id":"e1635a09-8de7-4cae-a099-bdcaa22b29e3","cell_type":"code","source":"transform = transforms.Compose([\n    transforms.Resize((BEST_CONFIG[\"input_size\"], BEST_CONFIG[\"input_size\"])),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406],\n                         std=[0.229, 0.224, 0.225]),\n])\n\n@torch.no_grad()\ndef extract_embeddings(backbone, image_paths, desc=\"Extracting\"):\n    backbone.eval()\n    all_embs = []\n    for path in tqdm(image_paths, desc=desc):\n        try:\n            img = Image.open(path).convert(\"RGB\")\n        except Exception as e:\n            print(f\"  [WARN] {path}: {e}\")\n            img = Image.new(\"RGB\", (BEST_CONFIG[\"input_size\"], BEST_CONFIG[\"input_size\"]))\n        t = transform(img).unsqueeze(0).to(device)\n        emb = backbone(t).cpu().numpy()\n        all_embs.append(emb)\n    return np.vstack(all_embs)\n\n# Cache paths\ntrain_cache = CACHE_DIR / \"dinov2_train_embeddings.npz\"\nval_cache   = CACHE_DIR / \"dinov2_val_embeddings.npz\"\n\nif train_cache.exists() and val_cache.exists():\n    print(\"Loading cached DINOv2 embeddings...\")\n    train_embeddings = np.load(train_cache)[\"embeddings\"]\n    val_embeddings   = np.load(val_cache)[\"embeddings\"]\n    print(f\"  Train: {train_embeddings.shape}  |  Val: {val_embeddings.shape}\")\nelse:\n    print(f\"Loading DINOv2-ViT-L/14 backbone...\")\n    backbone = timm.create_model(\n        BEST_CONFIG[\"backbone_model_id\"],\n        pretrained=True, num_classes=0,\n        img_size=BEST_CONFIG[\"input_size\"],\n    )\n    backbone.eval()\n    for p in backbone.parameters():\n        p.requires_grad = False\n    backbone.to(device)\n    backbone_dim_check = backbone(\n        torch.randn(1, 3, BEST_CONFIG[\"input_size\"],\n                    BEST_CONFIG[\"input_size\"]).to(device)\n    ).shape[1]\n    print(f\"  Backbone output dim: {backbone_dim_check}\")\n    assert backbone_dim_check == BEST_CONFIG[\"backbone_dim\"],         f\"Dim mismatch: {backbone_dim_check} vs {BEST_CONFIG['backbone_dim']}\"\n\n    train_embeddings = extract_embeddings(backbone, train_image_paths, \"DINOv2 train\")\n    val_embeddings   = extract_embeddings(backbone, val_image_paths,   \"DINOv2 val\")\n\n    np.savez_compressed(train_cache, embeddings=train_embeddings)\n    np.savez_compressed(val_cache,   embeddings=val_embeddings)\n    print(f\"  Train: {train_embeddings.shape}  |  Val: {val_embeddings.shape}\")\n    print(\"  Embeddings cached ✓\")\n    del backbone\n    torch.cuda.empty_cache() if torch.cuda.is_available() else None\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T15:44:49.962185Z","iopub.execute_input":"2026-03-14T15:44:49.962417Z","iopub.status.idle":"2026-03-14T16:06:45.863104Z","shell.execute_reply.started":"2026-03-14T15:44:49.962396Z","shell.execute_reply":"2026-03-14T16:06:45.862449Z"}},"outputs":[],"execution_count":null},{"id":"1d7906fd-f3dd-45d4-b29c-5822c9b4feda","cell_type":"markdown","source":"## 6. Model Architecture","metadata":{}},{"id":"211e9af2-c9bc-4656-a7e0-59ed99d7e9ef","cell_type":"code","source":"class EmbeddingProjection(nn.Module):\n    def __init__(self, input_dim, hidden_dim=1024, output_dim=512, dropout=0.3):\n        super().__init__()\n        self.net = nn.Sequential(\n            nn.Linear(input_dim, hidden_dim),\n            nn.BatchNorm1d(hidden_dim),\n            nn.ReLU(inplace=True),\n            nn.Dropout(dropout),\n            nn.Linear(hidden_dim, output_dim),\n            nn.BatchNorm1d(output_dim),\n        )\n    def forward(self, x): return self.net(x)\n\n\nclass ArcFaceLayer(nn.Module):\n    def __init__(self, embedding_dim, num_classes, margin=0.6, scale=48.0):\n        super().__init__()\n        self.scale = scale\n        self.cos_m = math.cos(margin)\n        self.sin_m = math.sin(margin)\n        self.th    = math.cos(math.pi - margin)\n        self.mm    = math.sin(math.pi - margin) * margin\n        self.weight = nn.Parameter(torch.FloatTensor(num_classes, embedding_dim))\n        nn.init.xavier_uniform_(self.weight)\n\n    def forward(self, emb, labels):\n        emb_n = F.normalize(emb, p=2, dim=1)\n        w_n   = F.normalize(self.weight, p=2, dim=1)\n        cos   = torch.clamp(F.linear(emb_n, w_n), -1.0, 1.0)\n        sin   = torch.sqrt(1.0 - cos ** 2)\n        phi   = cos * self.cos_m - sin * self.sin_m\n        phi   = torch.where(cos > self.th, phi, cos - self.mm)\n        oh    = torch.zeros_like(cos).scatter_(1, labels.view(-1, 1).long(), 1.0)\n        return (oh * phi + (1 - oh) * cos) * self.scale\n\n\nclass ArcFaceModel(nn.Module):\n    def __init__(self, input_dim, num_classes,\n                 embedding_dim=512, hidden_dim=1024,\n                 margin=0.6, scale=48.0, dropout=0.3):\n        super().__init__()\n        self.projection = EmbeddingProjection(input_dim, hidden_dim, embedding_dim, dropout)\n        self.arcface    = ArcFaceLayer(embedding_dim, num_classes, margin, scale)\n\n    def forward(self, x, labels):\n        emb    = self.projection(x)\n        logits = self.arcface(emb, labels)\n        return logits, emb\n\n    def get_embeddings(self, x):\n        return F.normalize(self.projection(x), p=2, dim=1)\n\n\nprint(\"ArcFaceModel defined.\")\nprint(f\"  Architecture: Linear({BEST_CONFIG['backbone_dim']} → {BEST_CONFIG['hidden_dim']} → {BEST_CONFIG['embedding_dim']})\")\nprint(f\"  ArcFace:      margin={BEST_CONFIG['arcface_margin']}, scale={BEST_CONFIG['arcface_scale']}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T16:06:45.864259Z","iopub.execute_input":"2026-03-14T16:06:45.864685Z","iopub.status.idle":"2026-03-14T16:06:45.878669Z","shell.execute_reply.started":"2026-03-14T16:06:45.864649Z","shell.execute_reply":"2026-03-14T16:06:45.878013Z"}},"outputs":[],"execution_count":null},{"id":"f39b636c-505d-4df9-b465-0dbec3ebe7c3","cell_type":"markdown","source":"## 7. Dataset and Training Utilities","metadata":{}},{"id":"ce3b2c3e-71d6-4546-ab06-0e7ae81f80eb","cell_type":"code","source":"class EmbeddingDataset(Dataset):\n    def __init__(self, embeddings, labels):\n        self.embeddings = torch.FloatTensor(embeddings)\n        self.labels     = torch.LongTensor(labels)\n    def __len__(self):   return len(self.labels)\n    def __getitem__(self, i): return self.embeddings[i], self.labels[i]\n\n\ndef set_seed(seed: int):\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark     = False\n\n\ndef compute_val_map(model, val_embeddings_np, val_labels_str):\n    \"\"\"Identity-balanced mAP (same function used across all experiments).\"\"\"\n    model.eval()\n    with torch.no_grad():\n        t    = torch.FloatTensor(val_embeddings_np).to(device)\n        embs = model.get_embeddings(t).cpu().numpy()\n    sim = embs @ embs.T\n    np.fill_diagonal(sim, -1)\n    id_aps = defaultdict(list)\n    for q in range(len(val_labels_str)):\n        ql       = val_labels_str[q]\n        is_match = (val_labels_str == ql).astype(int); is_match[q] = 0\n        n_pos    = is_match.sum()\n        if n_pos == 0: continue\n        order = np.argsort(-sim[q]); sm = is_match[order]\n        cum   = np.cumsum(sm)\n        prec  = cum / np.arange(1, len(sm) + 1)\n        id_aps[ql].append(float(np.sum(prec * sm) / n_pos))\n    return float(np.mean([np.mean(v) for v in id_aps.values()]))\n\n\ndef train_one_seed(seed, train_embeddings, val_embeddings,\n                   train_labels_enc, val_labels_str):\n    \"\"\"Full training run for one seed. Returns best val mAP.\"\"\"\n    set_seed(seed)\n\n    model = ArcFaceModel(\n        input_dim     = BEST_CONFIG[\"backbone_dim\"],\n        num_classes   = num_classes,\n        embedding_dim = BEST_CONFIG[\"embedding_dim\"],\n        hidden_dim    = BEST_CONFIG[\"hidden_dim\"],\n        margin        = BEST_CONFIG[\"arcface_margin\"],\n        scale         = BEST_CONFIG[\"arcface_scale\"],\n        dropout       = BEST_CONFIG[\"dropout\"],\n    ).to(device)\n\n    train_ds = EmbeddingDataset(train_embeddings, train_labels_enc)\n    val_ds   = EmbeddingDataset(val_embeddings,   val_data[\"label_encoded\"].values)\n\n    # DataLoader shuffle is affected by the seed set above\n    train_loader = DataLoader(train_ds, batch_size=BEST_CONFIG[\"batch_size\"],\n                              shuffle=True,  num_workers=0)\n    val_loader   = DataLoader(val_ds,   batch_size=BEST_CONFIG[\"batch_size\"],\n                              shuffle=False, num_workers=0)\n\n    criterion = nn.CrossEntropyLoss()\n    optimizer = torch.optim.AdamW(model.parameters(),\n                                   lr=BEST_CONFIG[\"learning_rate\"],\n                                   weight_decay=BEST_CONFIG[\"weight_decay\"])\n    scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(\n        optimizer, mode=\"min\",\n        factor=BEST_CONFIG[\"scheduler_factor\"],\n        patience=BEST_CONFIG[\"scheduler_patience\"],\n    )\n\n    best_map        = 0.0\n    best_val_loss   = float(\"inf\")\n    best_epoch      = 0\n    ckpt_path       = CHECKPOINT_DIR / f\"seed_stability_seed{seed}_best.pth\"\n    t0              = time.time()\n\n    for epoch in range(BEST_CONFIG[\"num_epochs\"]):\n        # Train\n        model.train()\n        tr_loss = 0.0\n        for embs, labs in train_loader:\n            embs, labs = embs.to(device), labs.to(device)\n            logits, _  = model(embs, labs)\n            loss       = criterion(logits, labs)\n            optimizer.zero_grad(); loss.backward(); optimizer.step()\n            tr_loss   += loss.item()\n        tr_loss /= len(train_loader)\n\n        # Validate\n        model.eval()\n        vl_loss = 0.0\n        with torch.no_grad():\n            for embs, labs in val_loader:\n                embs, labs = embs.to(device), labs.to(device)\n                logits, _  = model(embs, labs)\n                vl_loss   += criterion(logits, labs).item()\n        vl_loss /= len(val_loader)\n\n        val_map = compute_val_map(model, val_embeddings, val_labels_str)\n\n        scheduler.step(vl_loss)\n        lr = optimizer.param_groups[0][\"lr\"]\n\n        # Log per-epoch to W&B (prefixed by seed)\n        wandb.log({\n            f\"seed{seed}/epoch\":    epoch + 1,\n            f\"seed{seed}/tr_loss\":  tr_loss,\n            f\"seed{seed}/vl_loss\":  vl_loss,\n            f\"seed{seed}/val_map\":  val_map,\n            f\"seed{seed}/lr\":       lr,\n        })\n\n        if val_map > best_map:\n            best_map      = val_map\n            best_val_loss = vl_loss\n            best_epoch    = epoch + 1\n            torch.save({\n                \"seed\": seed, \"epoch\": epoch + 1,\n                \"model_state_dict\": model.state_dict(),\n                \"val_map\": val_map, \"val_loss\": vl_loss,\n            }, str(ckpt_path))\n\n        if (epoch + 1) % 10 == 0:\n            elapsed = time.time() - t0\n            print(f\"    ep {epoch+1:2d}/{BEST_CONFIG['num_epochs']} | \"\n                  f\"tr={tr_loss:.4f} vl={vl_loss:.4f} \"\n                  f\"mAP={val_map:.4f} (best={best_map:.4f}) | {elapsed/60:.1f}min\")\n\n    elapsed = time.time() - t0\n    print(f\"  Seed {seed:4d}: best mAP={best_map:.4f}  ep={best_epoch}  \"\n          f\"({elapsed/60:.1f} min)\")\n    return best_map, best_epoch\n\n\nprint(\"Training utilities defined ✓\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T16:06:45.879529Z","iopub.execute_input":"2026-03-14T16:06:45.879757Z","iopub.status.idle":"2026-03-14T16:06:45.901454Z","shell.execute_reply.started":"2026-03-14T16:06:45.879735Z","shell.execute_reply":"2026-03-14T16:06:45.900896Z"}},"outputs":[],"execution_count":null},{"id":"4f039627-96c9-48f0-84d5-dfafc2c4d301","cell_type":"markdown","source":"## 8. Run All Seeds","metadata":{}},{"id":"2f6b2b8b-c663-4aea-b0c6-e90608622a54","cell_type":"code","source":"seed_results = []   # list of (seed, best_map, best_epoch)\n\nprint(f\"Running {len(SEEDS)} seeds with best DINOv2 + ArcFace config...\")\nprint(f\"Config: margin={BEST_CONFIG['arcface_margin']}, scale={BEST_CONFIG['arcface_scale']}, \"\n      f\"emb_dim={BEST_CONFIG['embedding_dim']}, hidden={BEST_CONFIG['hidden_dim']}, \"\n      f\"dropout={BEST_CONFIG['dropout']}\")\nprint()\n\nfor i, seed in enumerate(SEEDS, 1):\n    print(f\"[{i}/{len(SEEDS)}] Seed {seed}  ─────────────────────────────────────────\")\n    best_map, best_epoch = train_one_seed(\n        seed,\n        train_embeddings,\n        val_embeddings,\n        train_data[\"label_encoded\"].values,\n        val_labels,\n    )\n    seed_results.append({\"seed\": seed, \"val_map\": best_map, \"best_epoch\": best_epoch})\n    wandb.log({\n        f\"seed{seed}/best_map\":   best_map,\n        f\"seed{seed}/best_epoch\": best_epoch,\n    })\n\nmaps = np.array([r[\"val_map\"] for r in seed_results])\nmean_map = float(maps.mean())\nstd_map  = float(maps.std())\nmin_map  = float(maps.min())\nmax_map  = float(maps.max())\n\nprint()\nprint(\"=\" * 60)\nprint(f\"SEED STABILITY RESULTS  (n={len(SEEDS)} seeds)\")\nprint(\"=\" * 60)\nprint(f\"{'Seed':>6}  {'Val mAP':>8}  {'Best Ep':>7}\")\nprint(\"-\" * 30)\nfor r in seed_results:\n    marker = \" ← seed=42 (Exp13)\" if r[\"seed\"] == 42 else \"\"\n    print(f\"{r['seed']:>6}  {r['val_map']:.4f}  {r['best_epoch']:>7}{marker}\")\nprint(\"=\" * 60)\nprint(f\"Mean mAP: {mean_map:.4f}\")\nprint(f\"Std  mAP: {std_map:.4f}\")\nprint(f\"Min  mAP: {min_map:.4f}\")\nprint(f\"Max  mAP: {max_map:.4f}\")\nprint(f\"Range:    {max_map - min_map:.4f}\")\nprint()\nprint(f\"Exp4 baseline:   {EXP4_BASELINE_MAP:.4f}\")\nprint(f\"Mean - Exp4:     {mean_map - EXP4_BASELINE_MAP:+.4f}\")\nprint(f\"Min  - Exp4:     {min_map  - EXP4_BASELINE_MAP:+.4f}\")\nprint(\"=\" * 60)\n\nwandb.log({\n    \"summary/mean_val_map\":  mean_map,\n    \"summary/std_val_map\":   std_map,\n    \"summary/min_val_map\":   min_map,\n    \"summary/max_val_map\":   max_map,\n    \"summary/range_val_map\": max_map - min_map,\n    \"summary/exp4_baseline\": EXP4_BASELINE_MAP,\n    \"summary/mean_minus_baseline\": mean_map - EXP4_BASELINE_MAP,\n    \"summary/n_seeds\": len(SEEDS),\n})\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T16:06:45.902230Z","iopub.execute_input":"2026-03-14T16:06:45.902465Z","iopub.status.idle":"2026-03-14T16:08:11.826627Z","shell.execute_reply.started":"2026-03-14T16:06:45.902445Z","shell.execute_reply":"2026-03-14T16:08:11.825977Z"}},"outputs":[],"execution_count":null},{"id":"adbe389d-fcfc-4590-ab87-8b1bfd295e80","cell_type":"markdown","source":"## 9. Visualisation","metadata":{}},{"id":"8b75f04b-7238-4cf5-853d-ce15f7dfd8ce","cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize=(14, 5))\n\n# Left: bar chart per seed\nax = axes[0]\nseed_labels = [str(r[\"seed\"]) for r in seed_results]\nseed_maps   = [r[\"val_map\"]   for r in seed_results]\nbar_colors  = [\"steelblue\" if s != 42 else \"darkorange\" for s in SEEDS]\n\nbars = ax.bar(seed_labels, seed_maps, color=bar_colors, edgecolor=\"white\", width=0.6)\nax.axhline(y=mean_map,           color=\"crimson\",    linestyle=\"-\",  linewidth=2,\n           label=f\"Mean {mean_map:.4f}\")\nax.axhline(y=EXP4_BASELINE_MAP,  color=\"gray\",       linestyle=\"--\", linewidth=1.5,\n           label=f\"Exp4 baseline {EXP4_BASELINE_MAP:.4f}\")\nax.fill_between(range(-1, len(SEEDS) + 1),\n                mean_map - std_map, mean_map + std_map,\n                color=\"crimson\", alpha=0.12,\n                label=f\"±1 std ({std_map:.4f})\")\n\nfor bar, v in zip(bars, seed_maps):\n    ax.text(bar.get_x() + bar.get_width() / 2, bar.get_height() + 0.001,\n            f\"{v:.4f}\", ha=\"center\", va=\"bottom\", fontsize=8)\n\nax.set_xlim(-0.5, len(SEEDS) - 0.5)\nax.set_ylim(max(0, min(seed_maps) - 0.02), max(seed_maps) + 0.03)\nax.set_xlabel(\"Random Seed\")\nax.set_ylabel(\"Val mAP (Identity-Balanced)\")\nax.set_title(\n    \"Seed Stability: Val mAP per Random Seed\\n\"\n    \"orange = seed=42 from Exp13; blue = new seeds\"\n)\nax.legend(fontsize=9)\nax.grid(axis=\"y\", alpha=0.3)\n\n# Right: distribution (box + strip)\nax2 = axes[1]\nax2.boxplot(\n    seed_maps,\n    vert=True,\n    widths=0.4,\n    patch_artist=True,\n    boxprops=dict(facecolor=\"lightsteelblue\", color=\"steelblue\"),\n    medianprops=dict(color=\"crimson\", linewidth=2),\n    whiskerprops=dict(color=\"steelblue\"),\n    capprops=dict(color=\"steelblue\"),\n    flierprops=dict(marker=\"o\", color=\"steelblue\"),\n)\n\nnp.random.seed(0)\njitter = np.random.uniform(-0.1, 0.1, len(seed_maps))\nax2.scatter(\n    [1 + j for j in jitter],\n    seed_maps,\n    color=\"darkorange\",\n    zorder=5,\n    s=60,\n    label=\"Individual seeds\",\n)\nax2.axhline(\n    y=EXP4_BASELINE_MAP,\n    color=\"gray\",\n    linestyle=\"--\",\n    linewidth=1.5,\n    label=f\"Exp4 baseline {EXP4_BASELINE_MAP:.4f}\",\n)\nax2.axhline(\n    y=mean_map,\n    color=\"crimson\",\n    linestyle=\"-\",\n    linewidth=2,\n    label=f\"Mean {mean_map:.4f}\",\n)\n\nax2.set_xticks([1])\nax2.set_xticklabels([\"Best DINOv2 Config 7 seeds\"])\nax2.set_ylabel(\"Val mAP (Identity-Balanced)\")\nax2.set_title(\n    f\"Distribution of Val mAP\\n\"\n    f\"mean={mean_map:.4f}, std={std_map:.4f}, n={len(SEEDS)}\"\n)\n\nplt.suptitle(\n    \"Experiment 14 — Seed Stability Analysis\\n\"\n    \"DINOv2-ViT-L/14 + ArcFace (margin=0.6, scale=48, dim=512, hidden=1024)\",\n    fontsize=12,\n    fontweight=\"bold\",\n    y=1.02,\n)\nplt.tight_layout()\nwandb.log({\"summary/seed_stability_chart\": wandb.Image(fig)})\nplt.savefig(CHECKPOINT_DIR / \"seed_stability.png\", dpi=150, bbox_inches=\"tight\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T16:08:11.827971Z","iopub.execute_input":"2026-03-14T16:08:11.829664Z","iopub.status.idle":"2026-03-14T16:08:12.991680Z","shell.execute_reply.started":"2026-03-14T16:08:11.829636Z","shell.execute_reply":"2026-03-14T16:08:12.991023Z"}},"outputs":[],"execution_count":null},{"id":"ef8f6dca-ba03-4c3c-a277-a6840bc10f90","cell_type":"markdown","source":"## 10. Results Table and Interpretation","metadata":{}},{"id":"b15f5cc2-c8ad-4a12-a82b-ee1b428236e5","cell_type":"code","source":"results_df = pd.DataFrame(seed_results)\nresults_df[\"vs_baseline\"] = results_df[\"val_map\"] - EXP4_BASELINE_MAP\n\nprint(\"Full Results Table:\")\nprint(results_df.to_string(index=False))\nprint()\nprint(f\"Summary Stats (n={len(SEEDS)} seeds):\")\nprint(f\"  Mean ± Std:  {mean_map:.4f} ± {std_map:.4f}\")\nprint(f\"  95% CI (approx ±2σ):  [{mean_map - 2*std_map:.4f}, {mean_map + 2*std_map:.4f}]\")\nprint(f\"  Min / Max:   {min_map:.4f} / {max_map:.4f}\")\nprint(f\"  Range:       {max_map - min_map:.4f}\")\nprint()\n\nall_beat_baseline = all(v > EXP4_BASELINE_MAP for v in seed_maps)\nif all_beat_baseline:\n    print(f\"ALL {len(SEEDS)} seeds beat the Exp4 baseline ({EXP4_BASELINE_MAP:.4f})\")\nelse:\n    n_beat = sum(v > EXP4_BASELINE_MAP for v in seed_maps)\n    print(f\"{n_beat}/{len(SEEDS)} seeds beat the Exp4 baseline ({EXP4_BASELINE_MAP:.4f})\")\n\nstability_label = \"STABLE\" if std_map < 0.010 else (\"MODERATE\" if std_map < 0.020 else \"UNSTABLE\")\nprint(f\"Stability assessment: {stability_label}  (std={std_map:.4f}, threshold 0.010)\")\nprint()\nprint(\"Interpretation:\")\nif std_map < 0.010:\n    print(f\"  The standard deviation ({std_map:.4f}) is small — the result is robust.\")\n    print(f\"  The improvement over Exp4 (mean +{mean_map - EXP4_BASELINE_MAP:.4f}) is \")\n    print(f\"  not due to a lucky initialisation.\")\nelif std_map < 0.020:\n    print(f\"  The standard deviation ({std_map:.4f}) is moderate — the result is\")\n    print(f\"  reasonably stable but some seed sensitivity exists.\")\nelse:\n    print(f\"  The standard deviation ({std_map:.4f}) is large — the result is sensitive\")\n    print(f\"  to initialisation. Interpret Exp13 mAP with caution.\")\n\n# Log table to W&B\nwandb.log({\n    \"summary/results_table\": wandb.Table(dataframe=results_df),\n    \"summary/all_beat_baseline\": all_beat_baseline,\n    \"summary/stability_label\": stability_label,\n    \"summary/ci_lower\": mean_map - 2 * std_map,\n    \"summary/ci_upper\": mean_map + 2 * std_map,\n})\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T16:08:12.992721Z","iopub.execute_input":"2026-03-14T16:08:12.993054Z","iopub.status.idle":"2026-03-14T16:08:13.944661Z","shell.execute_reply.started":"2026-03-14T16:08:12.993031Z","shell.execute_reply":"2026-03-14T16:08:13.944079Z"}},"outputs":[],"execution_count":null},{"id":"20a0f318-15e3-44aa-b996-e8ebdf3a8c43","cell_type":"markdown","source":"## 11. Generate Kaggle Submission (Best Seed)","metadata":{}},{"id":"0e1dd1f2-cd2e-491e-a5a0-d916ee8fd128","cell_type":"code","source":"# Use the seed that achieved the highest mAP for submission\nbest_result = max(seed_results, key=lambda r: r[\"val_map\"])\nbest_seed   = best_result[\"seed\"]\nbest_ckpt   = CHECKPOINT_DIR / f\"seed_stability_seed{best_seed}_best.pth\"\n\nprint(f\"Generating submission from best seed: {best_seed}  (mAP={best_result['val_map']:.4f})\")\n\n# Reload best checkpoint\nckpt = torch.load(best_ckpt, map_location=device, weights_only=False)\nbest_model = ArcFaceModel(\n    input_dim     = BEST_CONFIG[\"backbone_dim\"],\n    num_classes   = num_classes,\n    embedding_dim = BEST_CONFIG[\"embedding_dim\"],\n    hidden_dim    = BEST_CONFIG[\"hidden_dim\"],\n    margin        = BEST_CONFIG[\"arcface_margin\"],\n    scale         = BEST_CONFIG[\"arcface_scale\"],\n    dropout       = BEST_CONFIG[\"dropout\"],\n).to(device)\nbest_model.load_state_dict(ckpt[\"model_state_dict\"])\nbest_model.eval()\nprint(f\"  Checkpoint loaded: epoch={ckpt['epoch']}, val_map={ckpt['val_map']:.4f}\")\n\n# Extract test embeddings\ntest_pairs_df = pd.read_csv(DATA_DIR / \"test.csv\")\ntest_images   = sorted(set(test_pairs_df[\"query_image\"].unique()) |\n                        set(test_pairs_df[\"gallery_image\"].unique()))\n\nbackbone = timm.create_model(\n    BEST_CONFIG[\"backbone_model_id\"], pretrained=True,\n    num_classes=0, img_size=BEST_CONFIG[\"input_size\"],\n)\nbackbone.eval()\nfor p in backbone.parameters(): p.requires_grad = False\nbackbone.to(device)\n\nprint(f\"Extracting embeddings for {len(test_images)} test images...\")\ntest_emb_dict = {}\nwith torch.no_grad():\n    for img_name in tqdm(test_images, desc=\"Test embeddings\"):\n        img_path = DATA_DIR / \"test\" / \"test\" / img_name\n        try:\n            img = Image.open(img_path).convert(\"RGB\")\n        except Exception:\n            img = Image.new(\"RGB\", (BEST_CONFIG[\"input_size\"], BEST_CONFIG[\"input_size\"]))\n        t     = transform(img).unsqueeze(0).to(device)\n        bb_e  = backbone(t)\n        emb   = best_model.get_embeddings(bb_e).cpu().numpy().flatten()\n        test_emb_dict[img_name] = emb\n\n# Compute similarities\nprint(\"Computing pair similarities...\")\nsims = []\nfor _, row in tqdm(test_pairs_df.iterrows(), total=len(test_pairs_df)):\n    q = test_emb_dict.get(row[\"query_image\"],   np.zeros(BEST_CONFIG[\"embedding_dim\"]))\n    g = test_emb_dict.get(row[\"gallery_image\"], np.zeros(BEST_CONFIG[\"embedding_dim\"]))\n    qn = q / (np.linalg.norm(q) + 1e-12)\n    gn = g / (np.linalg.norm(g) + 1e-12)\n    sims.append(float(qn @ gn))\n\nsims_clipped = [max(0.0, min(1.0, s)) for s in sims]\nsubmission_df = pd.DataFrame({\"row_id\": test_pairs_df[\"row_id\"], \"similarity\": sims_clipped})\nsub_path = CHECKPOINT_DIR / f\"submission_seed{best_seed}.csv\"\nsubmission_df.to_csv(sub_path, index=False)\nsubmission_df.to_csv('/kaggle/working/submission.csv', index=False)\nprint(f\"Submission saved: {sub_path}  ({len(submission_df)} rows)\")\nprint(f\"Score distribution: min={min(sims):.4f}  mean={np.mean(sims):.4f}  max={max(sims):.4f}\")\n\ndel backbone\ntorch.cuda.empty_cache() if torch.cuda.is_available() else None\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T16:16:53.496673Z","iopub.execute_input":"2026-03-14T16:16:53.497393Z","iopub.status.idle":"2026-03-14T16:21:01.835508Z","shell.execute_reply.started":"2026-03-14T16:16:53.497351Z","shell.execute_reply":"2026-03-14T16:21:01.834885Z"}},"outputs":[],"execution_count":null},{"id":"cb139c9d-44d1-4ee0-9db1-49c2472b2bc3","cell_type":"markdown","source":"## 12. Log Artifacts and Finish","metadata":{}},{"id":"de707e96-5654-4249-b6b1-4e3b22b3ee67","cell_type":"code","source":"# Log submission as W&B artifact\nsub_art = wandb.Artifact(\n    \"submission-seed-stability\",\n    type=\"submission\",\n    description=f\"Seed stability best: seed={best_seed}, mAP={best_result['val_map']:.4f}\",\n    metadata={\n        \"best_seed\": best_seed,\n        \"best_map\": best_result[\"val_map\"],\n        \"mean_map\": mean_map,\n        \"std_map\":  std_map,\n        \"seeds\":    SEEDS,\n    }\n)\nsub_art.add_file(str(sub_path))\nwandb.log_artifact(sub_art)\n\n# Log best model checkpoint\nmodel_art = wandb.Artifact(\n    \"seed-stability-best-model\",\n    type=\"model\",\n    description=f\"Best seed={best_seed} checkpoint, mAP={best_result['val_map']:.4f}\",\n)\nmodel_art.add_file(str(best_ckpt))\nwandb.log_artifact(model_art)\n\nwandb.finish()\nprint(\"W&B run completed ✓\")\nprint()\nprint(\"=\" * 60)\nprint(\"EXPERIMENT 14 COMPLETE — SEED STABILITY SUMMARY\")\nprint(\"=\" * 60)\nprint(f\"  Config: DINOv2-ViT-L/14 + ArcFace\")\nprint(f\"  margin={BEST_CONFIG['arcface_margin']}, scale={BEST_CONFIG['arcface_scale']}, \"\n      f\"emb_dim={BEST_CONFIG['embedding_dim']}, hidden={BEST_CONFIG['hidden_dim']}\")\nprint(f\"  Seeds tested: {SEEDS}\")\nprint(f\"  Mean mAP: {mean_map:.4f} ± {std_map:.4f}\")\nprint(f\"  Exp4 baseline: {EXP4_BASELINE_MAP:.4f}\")\nprint(f\"  Net improvement: {mean_map - EXP4_BASELINE_MAP:+.4f} (mean)\")\nprint(\"=\" * 60)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-14T16:12:31.704032Z","iopub.execute_input":"2026-03-14T16:12:31.704316Z","iopub.status.idle":"2026-03-14T16:12:33.455507Z","shell.execute_reply.started":"2026-03-14T16:12:31.704283Z","shell.execute_reply":"2026-03-14T16:12:33.454821Z"}},"outputs":[],"execution_count":null}]}