{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":126777,"databundleVersionId":15314950,"isSourceIdPinned":false}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"md_0","cell_type":"markdown","source":"# 🐆 Jaguar Re-Identification","metadata":{}},{"id":"code_1","cell_type":"code","source":"import os, gc, math, random\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image, ImageFilter\nfrom tqdm.auto import tqdm\nfrom collections import Counter\nfrom pathlib import Path\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader, WeightedRandomSampler\nimport torchvision.transforms as T\n\nimport timm\nfrom sklearn.model_selection import StratifiedKFold\n\n# -- Configuration -------------------------------------------------------------\nclass CFG:\n    seed           = 42\n    # EVA02-Large uses GeM pooling natively and excels at fine-grained retrieval.\n    model_name     = \"eva02_large_patch14_448.mim_m38m_ft_in22k_in1k\"\n    img_size       = 448 # EVA02's native resolution\n    embed_dim      = 1024 # backbone output dim\n\n    # Training\n    batch_size     = 4\n    accum_steps    = 4\n    epochs_stage1  = 15\n    epochs_stage2  = 5\n    lr             = 3e-4\n    lr_stage2      = 1e-5 # conservative for fine-tuning\n    min_lr         = 1e-6\n    weight_decay   = 1e-4\n\n    # SubCenterArcFace - higher margin (0.5) enforces tighter clusters\n    arc_scale      = 30.0\n    arc_margin     = 0.50\n    arc_k          = 1 # subcenter count (1 = standard ArcFace)\n\n    # Pseudo-labeling\n    pl_threshold   = 0.90 # confidence cutoff\n    pl_max_add     = 500 # safety cap\n\n    # Paths\n    train_dir  = \"/kaggle/input/competitions/jaguar-re-id/train/train\"\n    test_dir   = \"/kaggle/input/competitions/jaguar-re-id/test/test\"\n    train_csv  = \"/kaggle/input/competitions/jaguar-re-id/train.csv\"\n    test_csv   = \"/kaggle/input/competitions/jaguar-re-id/test.csv\"\n\n    device     = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    device_str = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n\nseed_everything(CFG.seed)\nprint(f\"Device: {CFG.device}  |  GPUs: {torch.cuda.device_count()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-28T13:41:41.203991Z","iopub.execute_input":"2026-02-28T13:41:41.204267Z","iopub.status.idle":"2026-02-28T13:41:54.991617Z","shell.execute_reply.started":"2026-02-28T13:41:41.204234Z","shell.execute_reply":"2026-02-28T13:41:54.990848Z"}},"outputs":[],"execution_count":null},{"id":"md_2","cell_type":"markdown","source":"### Deduplication & Stratified Validation","metadata":{}},{"id":"code_3","cell_type":"code","source":"import imagehash\nfrom joblib import Parallel, delayed\nimport multiprocessing\n\ndf = pd.read_csv(CFG.train_csv)\ndf[\"image_path\"] = df[\"filename\"].apply(lambda x: os.path.join(CFG.train_dir, x))\n\nsample_path = df[\"image_path\"].iloc[0]\nif not os.path.exists(sample_path):\n    raise FileNotFoundError(f\"Image not found: {sample_path}\")\nprint(f\"Original training images: {len(df)}\")\n\ndef compute_phash(path):\n    with Image.open(path) as img:\n        return str(imagehash.phash(img.convert(\"RGB\"))) # RGB-only, lighter\n\nnum_cores = multiprocessing.cpu_count()\nprint(f\"Calculating pHashes across {num_cores} cores...\")\nhashes = Parallel(n_jobs=-1)(\n    delayed(compute_phash)(p) for p in tqdm(df[\"image_path\"], desc=\"pHash\")\n)\ndf[\"phash\"] = hashes\ndf = df.drop_duplicates(subset=[\"ground_truth\", \"phash\"]).reset_index(drop=True)\ndf.drop(columns=[\"phash\"], inplace=True)\ngc.collect()\nprint(f\"Images after deduplication: {len(df)}\")\n\n# Encode labels\nunique_ids      = sorted(df[\"ground_truth\"].unique())\nid2label        = {n: i for i, n in enumerate(unique_ids)}\ndf[\"label\"]     = df[\"ground_truth\"].map(id2label)\nCFG.num_classes = len(unique_ids)\n\nskf = StratifiedKFold(n_splits=10, shuffle=True, random_state=CFG.seed)\nfor fold, (_, val_idx) in enumerate(skf.split(df, df[\"label\"])):\n    df.loc[val_idx, \"fold\"] = fold\n\ntrain_df = df[df[\"fold\"] != 0].reset_index(drop=True)\nval_df   = df[df[\"fold\"] == 0].reset_index(drop=True)\nprint(f\"Train: {len(train_df)} | Val: {len(val_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-28T13:41:54.993331Z","iopub.execute_input":"2026-02-28T13:41:54.994100Z","iopub.status.idle":"2026-02-28T13:46:19.988516Z","shell.execute_reply.started":"2026-02-28T13:41:54.994064Z","shell.execute_reply":"2026-02-28T13:46:19.987716Z"}},"outputs":[],"execution_count":null},{"id":"md_4","cell_type":"markdown","source":"### Dataset & Augmentations","metadata":{}},{"id":"code_5","cell_type":"code","source":"class SharpenTransform:\n    \"\"\"Randomly apply PIL sharpen filter to improve blur robustness.\"\"\"\n    def __init__(self, p=0.3):\n        self.p = p\n    def __call__(self, img):\n        if random.random() < self.p:\n            return img.filter(ImageFilter.SHARPEN)\n        return img\n\n# EVA02 stats\nMEAN = [0.481, 0.457, 0.408]\nSTD  = [0.268, 0.261, 0.275]\n\ntrain_transforms = T.Compose([\n    T.Resize((CFG.img_size, CFG.img_size)),\n    SharpenTransform(p=0.3),\n    T.RandomHorizontalFlip(p=0.5), # enabled - empirically beneficial\n    T.RandomAffine(degrees=15, translate=(0.1, 0.1), scale=(0.9, 1.1)),\n    T.ColorJitter(brightness=0.2, contrast=0.2),\n    T.ToTensor(),\n    T.Normalize(mean=MEAN, std=STD),\n    T.RandomErasing(p=0.25), # masks random patches -> regularization\n])\n\nval_transforms = T.Compose([\n    T.Resize((CFG.img_size, CFG.img_size)),\n    T.ToTensor(),\n    T.Normalize(mean=MEAN, std=STD),\n])\n\n\nclass JaguarDataset(Dataset):\n    def __init__(self, df, transform=None, is_test=False, img_dir_col=\"image_path\"):\n        self.df         = df.reset_index(drop=True)\n        self.transform  = transform\n        self.is_test    = is_test\n        self.dir_col    = img_dir_col\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row  = self.df.iloc[idx]\n        path = row[self.dir_col]\n        try:\n            img = Image.open(path).convert(\"RGB\") # RGB only\n        except Exception:\n            img = Image.new(\"RGB\", (CFG.img_size, CFG.img_size))\n        if self.transform:\n            img = self.transform(img)\n        if self.is_test:\n            return img, row[\"filename\"]\n        return img, torch.tensor(row[\"label\"], dtype=torch.long)\n\n\n# Weighted sampler to handle class imbalance\nclass_counts   = train_df[\"label\"].value_counts().sort_index().values\nclass_weights  = 1.0 / class_counts\nsample_weights = [class_weights[l] for l in train_df[\"label\"]]\nsampler        = WeightedRandomSampler(sample_weights, len(train_df), replacement=True)\n\ntrain_dataset = JaguarDataset(train_df, transform=train_transforms)\nval_dataset   = JaguarDataset(val_df,   transform=val_transforms)\n\n# NOTE: 2 workers instead of 4 to halve DataLoader RAM overhead\ntrain_loader = DataLoader(train_dataset, batch_size=CFG.batch_size,\n                          sampler=sampler, num_workers=2, pin_memory=True, drop_last=True)\nval_loader   = DataLoader(val_dataset,   batch_size=CFG.batch_size,\n                          shuffle=False,  num_workers=2, pin_memory=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-28T13:47:32.531204Z","iopub.execute_input":"2026-02-28T13:47:32.531519Z","iopub.status.idle":"2026-02-28T13:47:32.545190Z","shell.execute_reply.started":"2026-02-28T13:47:32.531495Z","shell.execute_reply":"2026-02-28T13:47:32.544590Z"}},"outputs":[],"execution_count":null},{"id":"md_6","cell_type":"markdown","source":"### Architecture: EVA02 + GeM + SubCenterArcFace","metadata":{}},{"id":"code_7","cell_type":"code","source":"class GeM(nn.Module):\n    \"\"\"Generalised Mean Pooling - learnable pooling exponent p.\"\"\"\n    def __init__(self, p=3, eps=1e-6):\n        super().__init__()\n        self.p   = nn.Parameter(torch.ones(1) * p)\n        self.eps = eps\n\n    def forward(self, x):\n        # x: (B, C, H, W)\n        return F.avg_pool2d(\n            x.clamp(min=self.eps).pow(self.p),\n            (x.size(-2), x.size(-1))\n        ).pow(1.0 / self.p)\n\n\nclass SubCenterArcFace(nn.Module):\n    \"\"\"ArcFace with optional k sub-centers per class.\"\"\"\n    def __init__(self, in_features, out_features, s=30.0, m=0.5, k=1):\n        super().__init__()\n        self.s    = s\n        self.m    = m\n        self.k    = k\n        self.out  = out_features\n        self.weight = nn.Parameter(\n            torch.FloatTensor(out_features * k, in_features)\n        )\n        nn.init.xavier_uniform_(self.weight)\n\n    def forward(self, embeddings, labels=None):\n        cosine = F.linear(F.normalize(embeddings), F.normalize(self.weight))\n        if self.k > 1:\n            cosine = cosine.view(-1, self.out, self.k)\n            cosine, _ = cosine.max(dim=2) # take best sub-center\n        if labels is None:\n            return cosine\n        # Standard ArcFace margin\n        phi = cosine - self.m\n        one_hot = torch.zeros_like(cosine)\n        one_hot.scatter_(1, labels.view(-1, 1), 1)\n        output = (one_hot * phi) + ((1.0 - one_hot) * cosine)\n        return output * self.s\n\n\nclass JaguarEVA(nn.Module):\n    def __init__(self, num_classes):\n        super().__init__()\n        self.backbone = timm.create_model(CFG.model_name, pretrained=True, num_classes=0)\n        self.backbone.set_grad_checkpointing(True) # VRAM saver\n        self.feat_dim = self.backbone.num_features # 1024 for EVA02-L\n\n        self.gem      = GeM(p=3)\n        self.bn       = nn.BatchNorm1d(self.feat_dim)\n        self.arcface  = SubCenterArcFace(\n            self.feat_dim, num_classes,\n            s=CFG.arc_scale, m=CFG.arc_margin, k=CFG.arc_k\n        )\n\n    def _pool(self, x):\n        \"\"\"Extract and pool features from the backbone.\"\"\"\n        feat = self.backbone.forward_features(x) # (B, N, C) for ViT\n        if feat.dim() == 3:\n            B, N, C = feat.shape\n            H = W = int(math.sqrt(N))\n            if H * W != N:\n                feat = feat[:, -H * W:, :] # strip class token if present\n            feat = feat.permute(0, 2, 1).reshape(B, C, H, W)\n        emb = self.gem(feat).flatten(1) # (B, C)\n        emb = self.bn(emb)\n        return emb\n\n    def forward(self, x, labels=None):\n        emb = self._pool(x)\n        if labels is not None:\n            return self.arcface(emb, labels)\n        return F.normalize(emb, dim=1) # L2-normalised embedding\n\n    @torch.no_grad()\n    def extract(self, x):\n        self.eval()\n        return self.forward(x)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-28T13:47:32.546494Z","iopub.execute_input":"2026-02-28T13:47:32.546756Z","iopub.status.idle":"2026-02-28T13:47:32.565127Z","shell.execute_reply.started":"2026-02-28T13:47:32.546735Z","shell.execute_reply":"2026-02-28T13:47:32.564501Z"}},"outputs":[],"execution_count":null},{"id":"md_8","cell_type":"markdown","source":"### Stage 1 - Full Training","metadata":{}},{"id":"code_9","cell_type":"code","source":"model = JaguarEVA(CFG.num_classes)\nif torch.cuda.device_count() > 1:\n    print(f\"Using {torch.cuda.device_count()} GPUs (DataParallel)\")\n    model = nn.DataParallel(model)\nmodel = model.to(CFG.device)\n\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.AdamW(model.parameters(), lr=CFG.lr, weight_decay=CFG.weight_decay)\nscheduler = torch.optim.lr_scheduler.CosineAnnealingLR(\n    optimizer, T_max=CFG.epochs_stage1, eta_min=CFG.min_lr\n)\nscaler = torch.amp.GradScaler(CFG.device_str)\n\nbest_val_loss = float(\"inf\")\n\nfor epoch in range(CFG.epochs_stage1):\n    # -- Train --\n    model.train()\n    train_loss = 0.0\n    optimizer.zero_grad()\n    for i, (imgs, labels) in enumerate(\n            tqdm(train_loader, desc=f\"Epoch {epoch+1}/{CFG.epochs_stage1} [Train]\")):\n        imgs, labels = imgs.to(CFG.device), labels.to(CFG.device)\n        with torch.amp.autocast(CFG.device_str):\n            logits = model(imgs, labels)\n            loss   = criterion(logits, labels) / CFG.accum_steps\n        scaler.scale(loss).backward()\n        if (i + 1) % CFG.accum_steps == 0 or (i + 1) == len(train_loader):\n            scaler.step(optimizer)\n            scaler.update()\n            optimizer.zero_grad()\n        train_loss += loss.item() * CFG.accum_steps\n    scheduler.step()\n\n    # -- Validate --\n    model.eval()\n    val_loss = 0.0\n    with torch.no_grad():\n        for imgs, labels in val_loader:\n            imgs, labels = imgs.to(CFG.device), labels.to(CFG.device)\n            with torch.amp.autocast(CFG.device_str):\n                logits  = model(imgs, labels)\n                val_loss += criterion(logits, labels).item()\n\n    avg_train = train_loss / len(train_loader)\n    avg_val   = val_loss   / len(val_loader)\n    print(f\"Epoch {epoch+1:02d} | Train {avg_train:.4f} | Val {avg_val:.4f}\", end=\"\")\n\n    if val_loss < best_val_loss:\n        best_val_loss = val_loss\n        base = model.module if isinstance(model, nn.DataParallel) else model\n        torch.save(base.state_dict(), \"best_stage1_model.pth\")\n        print(\"  ✓ saved\")\n    else:\n        print()\n\n# -- Free training RAM ----------------------------------------------------------\ndel optimizer, scheduler, scaler\ntorch.cuda.empty_cache()\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-28T13:47:32.566143Z","iopub.execute_input":"2026-02-28T13:47:32.566449Z","iopub.status.idle":"2026-02-28T15:34:20.453486Z","shell.execute_reply.started":"2026-02-28T13:47:32.566420Z","shell.execute_reply":"2026-02-28T15:34:20.452825Z"}},"outputs":[],"execution_count":null},{"id":"md_10","cell_type":"markdown","source":"### Confidence-Based Pseudo-Labeling","metadata":{}},{"id":"code_11","cell_type":"code","source":"# -- Load best Stage-1 weights -------------------------------------------------\nbase_model = JaguarEVA(CFG.num_classes).to(CFG.device)\nbase_model.load_state_dict(torch.load(\"best_stage1_model.pth\", map_location=CFG.device))\nif torch.cuda.device_count() > 1:\n    model = nn.DataParallel(base_model)\nelse:\n    model = base_model\nmodel.eval()\n\n# -- Build test DataLoader -----------------------------------------------------\ntest_pairs_df = pd.read_csv(CFG.test_csv)\nunique_test   = sorted(set(test_pairs_df[\"query_image\"]) | set(test_pairs_df[\"gallery_image\"]))\ntest_df       = pd.DataFrame({\"filename\": unique_test})\ntest_df[\"image_path\"] = test_df[\"filename\"].apply(lambda x: os.path.join(CFG.test_dir, x))\nprint(f\"Unique test images: {len(test_df)}\")\n\ntest_dataset = JaguarDataset(test_df, transform=val_transforms, is_test=True)\ntest_loader  = DataLoader(test_dataset, batch_size=CFG.batch_size,\n                          shuffle=False, num_workers=2, pin_memory=False)\n\n# -- Extract test embeddings (float16 to save RAM) -----------------------------\ntest_embeds, test_names = [], []\nwith torch.no_grad():\n    for imgs, names in tqdm(test_loader, desc=\"Test embeddings\"):\n        imgs = imgs.to(CFG.device)\n        with torch.amp.autocast(CFG.device_str):\n            emb = model(imgs)\n        test_embeds.append(emb.cpu().half())\n        test_names.extend(names)\ntest_embeds = torch.cat(test_embeds, dim=0)\n\n# -- Extract centroids from ArcFace head --------------------------------------\nhead = base_model.arcface\ncentroids = F.normalize(head.weight.data.cpu().float(), dim=1) # (C, D)\nif CFG.arc_k > 1:\n    centroids = centroids.view(CFG.num_classes, CFG.arc_k, -1).max(dim=1).values\n\n# -- Cosine similarity -> softmax probability -----------------------------------\nemb_float = test_embeds.float()\nlogits    = emb_float @ centroids.T # (N_test, C)\nprobs     = torch.softmax(logits * CFG.arc_scale, dim=1)\nmax_probs, preds = probs.max(dim=1)\n\n# -- Filter high-confidence samples -------------------------------------------\npl_rows = []\nfor i, (prob, pred) in enumerate(zip(max_probs.tolist(), preds.tolist())):\n    if prob >= CFG.pl_threshold:\n        pl_rows.append({\n            \"filename\":   test_names[i],\n            \"image_path\": os.path.join(CFG.test_dir, test_names[i]),\n            \"label\":      pred,\n            \"conf\":       prob,\n        })\n        if len(pl_rows) >= CFG.pl_max_add:\n            break\n\npl_df = pd.DataFrame(pl_rows)\nprint(f\"Pseudo-labels selected: {len(pl_df)}/{len(test_df)} \"\n      f\"(threshold={CFG.pl_threshold})\")\n\n# Free large tensors\ndel logits, probs, max_probs, preds, emb_float, centroids\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-28T15:34:20.454976Z","iopub.execute_input":"2026-02-28T15:34:20.455235Z","iopub.status.idle":"2026-02-28T15:35:43.998050Z","shell.execute_reply.started":"2026-02-28T15:34:20.455210Z","shell.execute_reply":"2026-02-28T15:35:43.997027Z"}},"outputs":[],"execution_count":null},{"id":"md_12","cell_type":"markdown","source":"### Stage 2 - Fine-Tuning with Pseudo-Labels","metadata":{}},{"id":"code_13","cell_type":"code","source":"if len(pl_df) > 0:\n    # Combine real train + pseudo-labeled test\n    combined_df = pd.concat([\n        train_df[[\"filename\", \"image_path\", \"label\"]],\n        pl_df[   [\"filename\", \"image_path\", \"label\"]],\n    ], ignore_index=True)\n    print(f\"Combined dataset: {len(combined_df)} images \"\n          f\"({len(train_df)} train + {len(pl_df)} pseudo)\")\n\n    combined_dataset = JaguarDataset(combined_df, transform=train_transforms)\n    combined_loader  = DataLoader(\n        combined_dataset, batch_size=CFG.batch_size,\n        shuffle=True, num_workers=2, pin_memory=True, drop_last=True\n    )\n\n    # Lower LR for conservative fine-tuning\n    optimizer2 = torch.optim.AdamW(model.parameters(),\n                                   lr=CFG.lr_stage2, weight_decay=CFG.weight_decay)\n    scheduler2 = torch.optim.lr_scheduler.CosineAnnealingLR(\n        optimizer2, T_max=CFG.epochs_stage2, eta_min=CFG.min_lr\n    )\n    scaler2 = torch.amp.GradScaler(CFG.device_str)\n    criterion2 = nn.CrossEntropyLoss()\n\n    model.train()\n    for epoch in range(CFG.epochs_stage2):\n        ep_loss = 0.0\n        optimizer2.zero_grad()\n        for i, (imgs, labels) in enumerate(\n                tqdm(combined_loader, desc=f\"Stage2 Epoch {epoch+1}/{CFG.epochs_stage2}\")):\n            imgs, labels = imgs.to(CFG.device), labels.to(CFG.device)\n            with torch.amp.autocast(CFG.device_str):\n                logits = model(imgs, labels)\n                loss   = criterion2(logits, labels) / CFG.accum_steps\n            scaler2.scale(loss).backward()\n            if (i + 1) % CFG.accum_steps == 0 or (i + 1) == len(combined_loader):\n                scaler2.step(optimizer2)\n                scaler2.update()\n                optimizer2.zero_grad()\n            ep_loss += loss.item() * CFG.accum_steps\n        scheduler2.step()\n        print(f\"Stage2 Epoch {epoch+1} | Loss: {ep_loss/len(combined_loader):.4f}\")\n\n    del optimizer2, scheduler2, scaler2\n    torch.cuda.empty_cache()\n    gc.collect()\nelse:\n    print(\"No pseudo-labels found - using Stage-1 model for inference.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-28T15:35:44.003411Z","iopub.execute_input":"2026-02-28T15:35:44.003814Z","iopub.status.idle":"2026-02-28T15:35:44.021238Z","shell.execute_reply.started":"2026-02-28T15:35:44.003775Z","shell.execute_reply":"2026-02-28T15:35:44.020487Z"}},"outputs":[],"execution_count":null},{"id":"md_14","cell_type":"markdown","source":"### Final Inference & Submission","metadata":{}},{"id":"code_15","cell_type":"code","source":"# -- Extract final test embeddings --------------------------------------------\nmodel.eval()\nfinal_embeds, final_names = [], []\nwith torch.no_grad():\n    for imgs, names in tqdm(test_loader, desc=\"Final test embeddings\"):\n        imgs = imgs.to(CFG.device)\n        with torch.amp.autocast(CFG.device_str):\n            emb = model(imgs)\n        final_embeds.append(emb.cpu().half())\n        final_names.extend(names)\n\nfinal_embeds = torch.cat(final_embeds, dim=0).float().numpy()\n\nsim_matrix   = final_embeds @ final_embeds.T\n\nfname_to_idx = {fname: i for i, fname in enumerate(final_names)}\n\n# -- Build submission -----------------------------------------------------------\nq_idx = test_pairs_df[\"query_image\"].map(fname_to_idx).values\ng_idx = test_pairs_df[\"gallery_image\"].map(fname_to_idx).values\nscores = np.clip(sim_matrix[q_idx, g_idx], 0.0, 1.0)\n\nsubmission = pd.DataFrame({\"row_id\": test_pairs_df[\"row_id\"], \"similarity\": scores})\nsubmission.to_csv(\"submission.csv\", index=False)\nprint(f\"✅ submission.csv saved - {len(submission)} rows\")\nprint(submission.head())\n\n# -- Sanity check: score distribution -----------------------------------------\nprint(f\"\\nSimilarity stats:\\n{submission['similarity'].describe()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-28T15:35:44.022732Z","iopub.execute_input":"2026-02-28T15:35:44.023482Z","iopub.status.idle":"2026-02-28T15:36:48.869367Z","shell.execute_reply.started":"2026-02-28T15:35:44.023448Z","shell.execute_reply":"2026-02-28T15:36:48.868630Z"}},"outputs":[],"execution_count":null}]}