{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":126777,"databundleVersionId":15314950,"isSourceIdPinned":false}],"dockerImageVersionId":31287,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n**Metric**: Identity-balanced Mean Average Precision (mAP)\n\n**Key Challenges**:\n- Intra-class variation (same jaguar looks different under different conditions)\n- Inter-class similarity (different jaguars can have similar patterns)\n- Significant class imbalance (13 to 183 images per individual)\n- Spurious correlations (backgrounds, riverbanks)","metadata":{}},{"cell_type":"code","source":"# Core libraries\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pathlib import Path\nimport os\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Image processing\nfrom PIL import Image\nimport cv2\n\n# Deep Learning\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nimport torchvision.transforms as T\nfrom torchvision import models\n\n# Sklearn\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics.pairwise import cosine_similarity\n\n# Progress bar\nfrom tqdm.notebook import tqdm\n\n# Set seeds for reproducibility\ndef seed_everything(seed=42):\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\nseed_everything(42)\n\n# Device configuration\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f\"Using device: {device}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T04:57:59.689463Z","iopub.execute_input":"2026-03-11T04:57:59.690368Z","iopub.status.idle":"2026-03-11T04:57:59.699451Z","shell.execute_reply.started":"2026-03-11T04:57:59.690335Z","shell.execute_reply":"2026-03-11T04:57:59.698616Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Configuration","metadata":{}},{"cell_type":"code","source":"class CFG:\n    # Paths\n    data_dir = Path('/kaggle/input/jaguar-re-id')\n    train_dir = data_dir / 'train'\n    test_dir = data_dir / 'test'\n    output_dir = Path('/kaggle/working')\n    \n    # Model\n    model_name = 'efficientnet_b3'  # Options: resnet50, efficientnet_b3, convnext_base\n    embedding_dim = 512\n    pretrained = True\n    \n    # Training\n    epochs = 30\n    batch_size = 32\n    lr = 1e-4\n    weight_decay = 1e-4\n    num_workers = 4\n    n_folds = 5\n    \n    # Image\n    img_size = 384\n    \n    # ArcFace\n    arcface_s = 30.0\n    arcface_m = 0.5\n    \n    # Augmentation\n    use_augmentation = True\n    \n    # Misc\n    seed = 42\n    debug = False  # Set to True for quick testing","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T04:57:59.700657Z","iopub.execute_input":"2026-03-11T04:57:59.700969Z","iopub.status.idle":"2026-03-11T04:57:59.714521Z","shell.execute_reply.started":"2026-03-11T04:57:59.700938Z","shell.execute_reply":"2026-03-11T04:57:59.713561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n    data_dir = Path('/kaggle/input/competitions/jaguar-re-id')\n    train_dir = data_dir / 'train' / 'train'  # исправлено\n    test_dir = data_dir / 'test' / 'test'     # исправлено\n    output_dir = Path('/kaggle/working')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:03:04.485579Z","iopub.execute_input":"2026-03-11T05:03:04.486241Z","iopub.status.idle":"2026-03-11T05:03:04.490151Z","shell.execute_reply.started":"2026-03-11T05:03:04.486212Z","shell.execute_reply":"2026-03-11T05:03:04.489482Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Data Loading & EDA","metadata":{}},{"cell_type":"code","source":"# Load data\ntrain_df = pd.read_csv(CFG.data_dir / 'train.csv')\ntest_df = pd.read_csv(CFG.data_dir / 'test.csv')\nsample_submission = pd.read_csv(CFG.data_dir / 'sample_submission.csv')\n\nprint(f\"Train samples: {len(train_df)}\")\nprint(f\"Test pairs: {len(test_df)}\")\nprint(f\"Unique jaguars in train: {train_df['ground_truth'].nunique()}\")\n\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:03:07.089301Z","iopub.execute_input":"2026-03-11T05:03:07.089948Z","iopub.status.idle":"2026-03-11T05:03:07.213456Z","shell.execute_reply.started":"2026-03-11T05:03:07.089918Z","shell.execute_reply":"2026-03-11T05:03:07.212905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Class distribution\nclass_counts = train_df['ground_truth'].value_counts()\n\nfig, axes = plt.subplots(1, 2, figsize=(16, 5))\n\n# Bar plot\naxes[0].barh(class_counts.index, class_counts.values, color='steelblue')\naxes[0].set_xlabel('Number of Images')\naxes[0].set_ylabel('Jaguar ID')\naxes[0].set_title('Images per Jaguar (Training Set)')\naxes[0].invert_yaxis()\n\n# Box plot of distribution\naxes[1].boxplot(class_counts.values, vert=False)\naxes[1].set_xlabel('Number of Images')\naxes[1].set_title('Distribution of Images per Jaguar')\n\nplt.tight_layout()\nplt.show()\n\nprint(f\"\\nClass imbalance stats:\")\nprint(f\"  Min images: {class_counts.min()} ({class_counts.idxmin()})\")\nprint(f\"  Max images: {class_counts.max()} ({class_counts.idxmax()})\")\nprint(f\"  Mean: {class_counts.mean():.1f}\")\nprint(f\"  Median: {class_counts.median():.1f}\")\nprint(f\"  Imbalance ratio: {class_counts.max() / class_counts.min():.1f}x\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:03:11.723607Z","iopub.execute_input":"2026-03-11T05:03:11.723883Z","iopub.status.idle":"2026-03-11T05:03:12.071153Z","shell.execute_reply.started":"2026-03-11T05:03:11.723862Z","shell.execute_reply":"2026-03-11T05:03:12.07037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Посмотрим что внутри\nbase = '/kaggle/input/competitions/jaguar-re-id'\nprint(\"Contents of data_dir:\")\nprint(os.listdir(base))\n\nprint(\"\\nContents of train folder:\")\ntrain_path = os.path.join(base, 'train')\nif os.path.exists(train_path):\n    print(os.listdir(train_path)[:10])  # первые 10 файлов\nelse:\n    print(\"train folder not found!\")\n    # может images внутри?\n    for item in os.listdir(base):\n        full_path = os.path.join(base, item)\n        if os.path.isdir(full_path):\n            print(f\"\\n{item}/:\")\n            print(os.listdir(full_path)[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:03:17.180994Z","iopub.execute_input":"2026-03-11T05:03:17.181652Z","iopub.status.idle":"2026-03-11T05:03:17.189586Z","shell.execute_reply.started":"2026-03-11T05:03:17.181622Z","shell.execute_reply":"2026-03-11T05:03:17.188881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(os.listdir('/kaggle/input/competitions/jaguar-re-id/train/train')[:10])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:03:22.708621Z","iopub.execute_input":"2026-03-11T05:03:22.709692Z","iopub.status.idle":"2026-03-11T05:03:22.715718Z","shell.execute_reply.started":"2026-03-11T05:03:22.709661Z","shell.execute_reply":"2026-03-11T05:03:22.714777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize sample images\ndef show_jaguar_samples(jaguar_name, n_samples=5):\n    samples = train_df[train_df['ground_truth'] == jaguar_name].head(n_samples)\n    \n    fig, axes = plt.subplots(1, n_samples, figsize=(4*n_samples, 4))\n    if n_samples == 1:\n        axes = [axes]\n    \n    for ax, (_, row) in zip(axes, samples.iterrows()):\n        img_path = CFG.train_dir / row['filename']\n        img = Image.open(img_path)\n        ax.imshow(img)\n        ax.set_title(f\"{jaguar_name}\\n{row['filename']}\")\n        ax.axis('off')\n    \n    plt.suptitle(f'Sample images of {jaguar_name}', fontsize=14)\n    plt.tight_layout()\n    plt.show()\n\n# Show samples from different jaguars\nfor jaguar in ['Marcela', 'Ousado', 'Bernard']:\n    show_jaguar_samples(jaguar, n_samples=4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:03:24.487477Z","iopub.execute_input":"2026-03-11T05:03:24.488174Z","iopub.status.idle":"2026-03-11T05:03:30.527658Z","shell.execute_reply.started":"2026-03-11T05:03:24.488146Z","shell.execute_reply":"2026-03-11T05:03:30.526869Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Dataset & Augmentations","metadata":{}},{"cell_type":"code","source":"# Transforms\ndef get_transforms(mode='train'):\n    if mode == 'train':\n        return T.Compose([\n            T.Resize((CFG.img_size, CFG.img_size)),\n            T.RandomHorizontalFlip(p=0.5),\n            T.RandomRotation(degrees=15),\n            T.ColorJitter(brightness=0.2, contrast=0.2, saturation=0.2, hue=0.1),\n            T.RandomAffine(degrees=0, translate=(0.1, 0.1), scale=(0.9, 1.1)),\n            T.ToTensor(),\n            T.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n            T.RandomErasing(p=0.3, scale=(0.02, 0.2)),\n        ])\n    else:\n        return T.Compose([\n            T.Resize((CFG.img_size, CFG.img_size)),\n            T.ToTensor(),\n            T.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n        ])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:03:41.313713Z","iopub.execute_input":"2026-03-11T05:03:41.314132Z","iopub.status.idle":"2026-03-11T05:03:41.320097Z","shell.execute_reply.started":"2026-03-11T05:03:41.314105Z","shell.execute_reply":"2026-03-11T05:03:41.319302Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class JaguarDataset(Dataset):\n    def __init__(self, df, img_dir, transform=None, mode='train'):\n        self.df = df.reset_index(drop=True)\n        self.img_dir = img_dir\n        self.transform = transform\n        self.mode = mode\n        \n        if mode == 'train':\n            self.label_encoder = LabelEncoder()\n            self.labels = self.label_encoder.fit_transform(df['ground_truth'])\n    \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img_path = self.img_dir / row['filename']\n        \n        # Load image with alpha channel handling\n        img = Image.open(img_path).convert('RGBA')\n        \n        # Create white background and paste image\n        background = Image.new('RGB', img.size, (255, 255, 255))\n        background.paste(img, mask=img.split()[3])  # Use alpha as mask\n        img = background\n        \n        if self.transform:\n            img = self.transform(img)\n        \n        if self.mode == 'train':\n            label = self.labels[idx]\n            return img, label\n        else:\n            return img, row['filename']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:03:45.456328Z","iopub.execute_input":"2026-03-11T05:03:45.456958Z","iopub.status.idle":"2026-03-11T05:03:45.463649Z","shell.execute_reply.started":"2026-03-11T05:03:45.456931Z","shell.execute_reply":"2026-03-11T05:03:45.462793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class JaguarTestDataset(Dataset):\n    \"\"\"Dataset for test images (for embedding extraction)\"\"\"\n    def __init__(self, image_files, img_dir, transform=None):\n        self.image_files = image_files\n        self.img_dir = img_dir\n        self.transform = transform\n    \n    def __len__(self):\n        return len(self.image_files)\n    \n    def __getitem__(self, idx):\n        img_file = self.image_files[idx]\n        img_path = self.img_dir / img_file\n        \n        img = Image.open(img_path).convert('RGBA')\n        background = Image.new('RGB', img.size, (255, 255, 255))\n        background.paste(img, mask=img.split()[3])\n        img = background\n        \n        if self.transform:\n            img = self.transform(img)\n        \n        return img, img_file","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:03:48.992327Z","iopub.execute_input":"2026-03-11T05:03:48.993007Z","iopub.status.idle":"2026-03-11T05:03:48.9984Z","shell.execute_reply.started":"2026-03-11T05:03:48.992978Z","shell.execute_reply":"2026-03-11T05:03:48.997618Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Model Architecture","metadata":{}},{"cell_type":"code","source":"class ArcFaceHead(nn.Module):\n    \"\"\"ArcFace loss for metric learning\"\"\"\n    def __init__(self, in_features, out_features, s=30.0, m=0.50):\n        super().__init__()\n        self.in_features = in_features\n        self.out_features = out_features\n        self.s = s\n        self.m = m\n        \n        self.weight = nn.Parameter(torch.FloatTensor(out_features, in_features))\n        nn.init.xavier_uniform_(self.weight)\n        \n        self.cos_m = np.cos(m)\n        self.sin_m = np.sin(m)\n        self.th = np.cos(np.pi - m)\n        self.mm = np.sin(np.pi - m) * m\n    \n    def forward(self, input, label=None):\n        # Normalize weight and input\n        cosine = F.linear(F.normalize(input), F.normalize(self.weight))\n        \n        if label is None:\n            return cosine * self.s\n        \n        sine = torch.sqrt(1.0 - torch.pow(cosine, 2))\n        phi = cosine * self.cos_m - sine * self.sin_m\n        phi = torch.where(cosine > self.th, phi, cosine - self.mm)\n        \n        one_hot = torch.zeros(cosine.size(), device=input.device)\n        one_hot.scatter_(1, label.view(-1, 1).long(), 1)\n        \n        output = (one_hot * phi) + ((1.0 - one_hot) * cosine)\n        output *= self.s\n        \n        return output","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:03:52.84793Z","iopub.execute_input":"2026-03-11T05:03:52.848473Z","iopub.status.idle":"2026-03-11T05:03:52.855441Z","shell.execute_reply.started":"2026-03-11T05:03:52.848447Z","shell.execute_reply":"2026-03-11T05:03:52.854909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class JaguarReIDModel(nn.Module):\n    def __init__(self, model_name, embedding_dim, num_classes, pretrained=True):\n        super().__init__()\n        \n        # Backbone\n        if 'efficientnet' in model_name:\n            from torchvision.models import efficientnet_b3, EfficientNet_B3_Weights\n            weights = EfficientNet_B3_Weights.IMAGENET1K_V1 if pretrained else None\n            self.backbone = efficientnet_b3(weights=weights)\n            backbone_out = self.backbone.classifier[1].in_features\n            self.backbone.classifier = nn.Identity()\n            \n        elif 'resnet50' in model_name:\n            from torchvision.models import resnet50, ResNet50_Weights\n            weights = ResNet50_Weights.IMAGENET1K_V2 if pretrained else None\n            self.backbone = resnet50(weights=weights)\n            backbone_out = self.backbone.fc.in_features\n            self.backbone.fc = nn.Identity()\n            \n        elif 'convnext' in model_name:\n            from torchvision.models import convnext_base, ConvNeXt_Base_Weights\n            weights = ConvNeXt_Base_Weights.IMAGENET1K_V1 if pretrained else None\n            self.backbone = convnext_base(weights=weights)\n            backbone_out = self.backbone.classifier[2].in_features\n            self.backbone.classifier = nn.Identity()\n        else:\n            raise ValueError(f\"Unknown model: {model_name}\")\n        \n        # Embedding head\n        self.bn1 = nn.BatchNorm1d(backbone_out)\n        self.dropout = nn.Dropout(p=0.3)\n        self.fc = nn.Linear(backbone_out, embedding_dim)\n        self.bn2 = nn.BatchNorm1d(embedding_dim)\n        \n        # ArcFace head\n        self.arcface = ArcFaceHead(\n            in_features=embedding_dim,\n            out_features=num_classes,\n            s=CFG.arcface_s,\n            m=CFG.arcface_m\n        )\n    \n    def extract_features(self, x):\n        \"\"\"Extract embeddings for inference\"\"\"\n        x = self.backbone(x)\n        x = self.bn1(x)\n        x = self.dropout(x)\n        x = self.fc(x)\n        x = self.bn2(x)\n        return F.normalize(x, p=2, dim=1)\n    \n    def forward(self, x, labels=None):\n        embeddings = self.extract_features(x)\n        \n        if labels is not None:\n            logits = self.arcface(embeddings, labels)\n            return logits, embeddings\n        \n        return embeddings","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:03:56.927859Z","iopub.execute_input":"2026-03-11T05:03:56.928382Z","iopub.status.idle":"2026-03-11T05:03:56.937096Z","shell.execute_reply.started":"2026-03-11T05:03:56.928357Z","shell.execute_reply":"2026-03-11T05:03:56.936396Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. Training Utilities","metadata":{}},{"cell_type":"code","source":"class FocalLoss(nn.Module):\n    \"\"\"Focal Loss for handling class imbalance\"\"\"\n    def __init__(self, gamma=2.0, alpha=None):\n        super().__init__()\n        self.gamma = gamma\n        self.alpha = alpha\n    \n    def forward(self, inputs, targets):\n        ce_loss = F.cross_entropy(inputs, targets, reduction='none')\n        pt = torch.exp(-ce_loss)\n        focal_loss = ((1 - pt) ** self.gamma) * ce_loss\n        \n        if self.alpha is not None:\n            alpha_t = self.alpha[targets]\n            focal_loss = alpha_t * focal_loss\n        \n        return focal_loss.mean()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:04:04.192956Z","iopub.execute_input":"2026-03-11T05:04:04.193606Z","iopub.status.idle":"2026-03-11T05:04:04.198648Z","shell.execute_reply.started":"2026-03-11T05:04:04.193577Z","shell.execute_reply":"2026-03-11T05:04:04.198007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def compute_map(embeddings, labels):\n    \"\"\"Compute identity-balanced mAP\"\"\"\n    # Compute similarity matrix\n    sim_matrix = cosine_similarity(embeddings)\n    \n    unique_labels = np.unique(labels)\n    identity_aps = []\n    \n    for identity in unique_labels:\n        # Get indices for this identity\n        identity_mask = labels == identity\n        identity_indices = np.where(identity_mask)[0]\n        \n        query_aps = []\n        for query_idx in identity_indices:\n            # Get similarities for this query (excluding self)\n            sims = sim_matrix[query_idx].copy()\n            sims[query_idx] = -np.inf  # Exclude self\n            \n            # Get ground truth (same identity)\n            gt = identity_mask.copy()\n            gt[query_idx] = False\n            \n            # Rank by similarity\n            ranked_indices = np.argsort(sims)[::-1]\n            ranked_gt = gt[ranked_indices]\n            \n            # Compute AP\n            n_relevant = ranked_gt.sum()\n            if n_relevant == 0:\n                continue\n            \n            precisions = np.cumsum(ranked_gt) / (np.arange(len(ranked_gt)) + 1)\n            ap = (precisions * ranked_gt).sum() / n_relevant\n            query_aps.append(ap)\n        \n        if query_aps:\n            identity_aps.append(np.mean(query_aps))\n    \n    return np.mean(identity_aps) if identity_aps else 0.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:04:07.610377Z","iopub.execute_input":"2026-03-11T05:04:07.610938Z","iopub.status.idle":"2026-03-11T05:04:07.617Z","shell.execute_reply.started":"2026-03-11T05:04:07.610912Z","shell.execute_reply":"2026-03-11T05:04:07.616272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_epoch(model, train_loader, optimizer, criterion, scheduler, device):\n    model.train()\n    total_loss = 0\n    correct = 0\n    total = 0\n    \n    pbar = tqdm(train_loader, desc='Training')\n    for images, labels in pbar:\n        images = images.to(device)\n        labels = labels.to(device)\n        \n        optimizer.zero_grad()\n        \n        logits, embeddings = model(images, labels)\n        loss = criterion(logits, labels)\n        \n        loss.backward()\n        optimizer.step()\n        \n        total_loss += loss.item()\n        _, predicted = logits.max(1)\n        total += labels.size(0)\n        correct += predicted.eq(labels).sum().item()\n        \n        pbar.set_postfix({'loss': f'{loss.item():.4f}', 'acc': f'{100.*correct/total:.2f}%'})\n    \n    if scheduler is not None:\n        scheduler.step()\n    \n    return total_loss / len(train_loader), 100. * correct / total","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:04:11.801919Z","iopub.execute_input":"2026-03-11T05:04:11.802563Z","iopub.status.idle":"2026-03-11T05:04:11.808531Z","shell.execute_reply.started":"2026-03-11T05:04:11.802536Z","shell.execute_reply":"2026-03-11T05:04:11.807617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@torch.no_grad()\ndef validate(model, val_loader, criterion, device):\n    model.eval()\n    total_loss = 0\n    all_embeddings = []\n    all_labels = []\n    \n    for images, labels in tqdm(val_loader, desc='Validating'):\n        images = images.to(device)\n        labels_tensor = labels.to(device)\n        \n        logits, embeddings = model(images, labels_tensor)\n        loss = criterion(logits, labels_tensor)\n        \n        total_loss += loss.item()\n        all_embeddings.append(embeddings.cpu().numpy())\n        all_labels.append(labels.numpy())\n    \n    all_embeddings = np.vstack(all_embeddings)\n    all_labels = np.concatenate(all_labels)\n    \n    # Compute mAP\n    val_map = compute_map(all_embeddings, all_labels)\n    \n    return total_loss / len(val_loader), val_map","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:04:13.97685Z","iopub.execute_input":"2026-03-11T05:04:13.977258Z","iopub.status.idle":"2026-03-11T05:04:13.982816Z","shell.execute_reply.started":"2026-03-11T05:04:13.977234Z","shell.execute_reply":"2026-03-11T05:04:13.98198Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. Training Loop","metadata":{}},{"cell_type":"code","source":"def train_fold(fold, train_df, train_idx, val_idx):\n    print(f\"\\n{'='*50}\")\n    print(f\"Fold {fold + 1}\")\n    print(f\"{'='*50}\")\n    \n    # Split data\n    train_data = train_df.iloc[train_idx].reset_index(drop=True)\n    val_data = train_df.iloc[val_idx].reset_index(drop=True)\n    \n    print(f\"Train samples: {len(train_data)}, Val samples: {len(val_data)}\")\n    \n    # Datasets\n    train_dataset = JaguarDataset(\n        train_data, CFG.train_dir,\n        transform=get_transforms('train'),\n        mode='train'\n    )\n    val_dataset = JaguarDataset(\n        val_data, CFG.train_dir,\n        transform=get_transforms('val'),\n        mode='train'\n    )\n    \n    # Use same label encoder\n    val_dataset.label_encoder = train_dataset.label_encoder\n    val_dataset.labels = train_dataset.label_encoder.transform(val_data['ground_truth'])\n    \n    # DataLoaders\n    train_loader = DataLoader(\n        train_dataset, batch_size=CFG.batch_size,\n        shuffle=True, num_workers=CFG.num_workers,\n        pin_memory=True, drop_last=True\n    )\n    val_loader = DataLoader(\n        val_dataset, batch_size=CFG.batch_size * 2,\n        shuffle=False, num_workers=CFG.num_workers,\n        pin_memory=True\n    )\n    \n    # Model\n    num_classes = len(train_dataset.label_encoder.classes_)\n    model = JaguarReIDModel(\n        CFG.model_name,\n        CFG.embedding_dim,\n        num_classes,\n        pretrained=CFG.pretrained\n    ).to(device)\n    \n    # Loss with class weights for imbalance\n    class_counts = np.bincount(train_dataset.labels)\n    class_weights = 1.0 / np.sqrt(class_counts)\n    class_weights = torch.FloatTensor(class_weights / class_weights.sum()).to(device)\n    criterion = FocalLoss(gamma=2.0, alpha=class_weights)\n    \n    # Optimizer & Scheduler\n    optimizer = torch.optim.AdamW(\n        model.parameters(),\n        lr=CFG.lr,\n        weight_decay=CFG.weight_decay\n    )\n    scheduler = torch.optim.lr_scheduler.CosineAnnealingLR(\n        optimizer, T_max=CFG.epochs, eta_min=1e-6\n    )\n    \n    # Training\n    best_map = 0\n    best_epoch = 0\n    history = {'train_loss': [], 'val_loss': [], 'val_map': []}\n    \n    for epoch in range(CFG.epochs):\n        print(f\"\\nEpoch {epoch + 1}/{CFG.epochs}\")\n        \n        train_loss, train_acc = train_epoch(\n            model, train_loader, optimizer, criterion, scheduler, device\n        )\n        val_loss, val_map = validate(model, val_loader, criterion, device)\n        \n        history['train_loss'].append(train_loss)\n        history['val_loss'].append(val_loss)\n        history['val_map'].append(val_map)\n        \n        print(f\"Train Loss: {train_loss:.4f}, Train Acc: {train_acc:.2f}%\")\n        print(f\"Val Loss: {val_loss:.4f}, Val mAP: {val_map:.4f}\")\n        \n        if val_map > best_map:\n            best_map = val_map\n            best_epoch = epoch\n            torch.save({\n                'model_state_dict': model.state_dict(),\n                'label_encoder': train_dataset.label_encoder,\n                'fold': fold,\n                'best_map': best_map\n            }, CFG.output_dir / f'model_fold{fold}.pth')\n            print(f\"  -> New best mAP! Saved model.\")\n        \n        # Early stopping\n        if epoch - best_epoch > 10:\n            print(f\"Early stopping at epoch {epoch + 1}\")\n            break\n    \n    print(f\"\\nFold {fold + 1} Best mAP: {best_map:.4f} at epoch {best_epoch + 1}\")\n    \n    return best_map, history","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:04:21.105429Z","iopub.execute_input":"2026-03-11T05:04:21.105952Z","iopub.status.idle":"2026-03-11T05:04:21.116533Z","shell.execute_reply.started":"2026-03-11T05:04:21.105926Z","shell.execute_reply":"2026-03-11T05:04:21.115761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n    # Paths — исправленные\n    data_dir = Path('/kaggle/input/competitions/jaguar-re-id')\n    train_dir = data_dir / 'train' / 'train'\n    test_dir = data_dir / 'test' / 'test'\n    output_dir = Path('/kaggle/working')\n    \n    # Model\n    model_name = 'efficientnet_b3'\n    embedding_dim = 512\n    pretrained = True\n    \n    # Training\n    epochs = 30\n    batch_size = 16  # для T4\n    lr = 1e-4\n    weight_decay = 1e-4\n    num_workers = 4\n    n_folds = 5\n    \n    # Image\n    img_size = 384\n    \n    # ArcFace\n    arcface_s = 30.0\n    arcface_m = 0.5\n    \n    # Augmentation\n    use_augmentation = True\n    \n    # Misc\n    seed = 42\n    debug = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:09:38.716975Z","iopub.execute_input":"2026-03-11T05:09:38.717658Z","iopub.status.idle":"2026-03-11T05:09:38.722921Z","shell.execute_reply.started":"2026-03-11T05:09:38.717629Z","shell.execute_reply":"2026-03-11T05:09:38.722136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@torch.no_grad()\ndef extract_embeddings(model, dataloader, device)\n    model.eval()\n    embeddings = []\n    filenames = []\n    \n    for images, names in tqdm(dataloader, desc='Extracting embeddings'):\n        images = images.to(device)\n        emb = model.extract_features(images)\n        embeddings.append(emb.cpu().numpy())\n        filenames.extend(names)\n    \n    return np.vstack(embeddings), filenames","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T04:57:59.885438Z","iopub.status.idle":"2026-03-11T04:57:59.88574Z","shell.execute_reply.started":"2026-03-11T04:57:59.885586Z","shell.execute_reply":"2026-03-11T04:57:59.885599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# K-Fold Cross Validation\nskf = StratifiedKFold(n_splits=CFG.n_folds, shuffle=True, random_state=CFG.seed)\n\nfold_results = []\nall_histories = []\n\nfor fold, (train_idx, val_idx) in enumerate(skf.split(train_df, train_df['ground_truth'])):\n    if CFG.debug and fold > 0:\n        break\n    \n    best_map, history = train_fold(fold, train_df, train_idx, val_idx)\n    fold_results.append(best_map)\n    all_histories.append(history)\n\nprint(f\"\\n{'='*50}\")\nprint(f\"Cross-Validation Results\")\nprint(f\"{'='*50}\")\nfor i, score in enumerate(fold_results):\n    print(f\"Fold {i+1}: {score:.4f}\")\nprint(f\"Mean mAP: {np.mean(fold_results):.4f} ± {np.std(fold_results):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T05:09:43.014958Z","iopub.execute_input":"2026-03-11T05:09:43.015644Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 8. Inference","metadata":{}},{"cell_type":"code","source":"def tta_embeddings(model, image_path, transforms_list, device):\n    \"\"\"Test-Time Augmentation for more robust embeddings\"\"\"\n    model.eval()\n    embeddings = []\n    \n    img = Image.open(image_path).convert('RGBA')\n    background = Image.new('RGB', img.size, (255, 255, 255))\n    background.paste(img, mask=img.split()[3])\n    img = background\n    \n    # Original\n    base_transform = get_transforms('val')\n    img_tensor = base_transform(img).unsqueeze(0).to(device)\n    emb = model.extract_features(img_tensor)\n    embeddings.append(emb.cpu().numpy())\n    \n    # Horizontal flip\n    img_flip = img.transpose(Image.FLIP_LEFT_RIGHT)\n    img_tensor = base_transform(img_flip).unsqueeze(0).to(device)\n    emb = model.extract_features(img_tensor)\n    embeddings.append(emb.cpu().numpy())\n    \n    # Average embeddings\n    avg_emb = np.mean(embeddings, axis=0)\n    avg_emb = avg_emb / np.linalg.norm(avg_emb)\n    \n    return avg_emb","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T08:39:01.486225Z","iopub.execute_input":"2026-03-11T08:39:01.486957Z","iopub.status.idle":"2026-03-11T08:39:01.492989Z","shell.execute_reply.started":"2026-03-11T08:39:01.486923Z","shell.execute_reply":"2026-03-11T08:39:01.492247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get unique test images\ntest_images = sorted(set(test_df['query_image'].tolist() + test_df['gallery_image'].tolist()))\nprint(f\"Unique test images: {len(test_images)}\")\n\n# Create test dataset\ntest_dataset = JaguarTestDataset(\n    test_images, CFG.test_dir,\n    transform=get_transforms('val')\n)\n\ntest_loader = DataLoader(\n    test_dataset, batch_size=CFG.batch_size,\n    shuffle=False, num_workers=CFG.num_workers,\n    pin_memory=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T08:39:06.452541Z","iopub.execute_input":"2026-03-11T08:39:06.453238Z","iopub.status.idle":"2026-03-11T08:39:06.467683Z","shell.execute_reply.started":"2026-03-11T08:39:06.453209Z","shell.execute_reply":"2026-03-11T08:39:06.466925Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@torch.no_grad()\ndef extract_embeddings(model, dataloader, device):\n    \"\"\"Extract embeddings for all images\"\"\"\n    model.eval()\n    embeddings = []\n    filenames = []\n    \n    for images, names in tqdm(dataloader, desc='Extracting embeddings'):\n        images = images.to(device)\n        emb = model.extract_features(images)\n        embeddings.append(emb.cpu().numpy())\n        filenames.extend(names)\n    \n    return np.vstack(embeddings), filenames","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T08:40:59.262404Z","iopub.execute_input":"2026-03-11T08:40:59.26303Z","iopub.status.idle":"2026-03-11T08:40:59.267918Z","shell.execute_reply.started":"2026-03-11T08:40:59.262998Z","shell.execute_reply":"2026-03-11T08:40:59.26711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ensemble predictions - только обученные фолды\nall_fold_embeddings = []\n\n# Только fold 0 (который ты обучил)\nfor fold in range(1):  # было range(CFG.n_folds)\n    print(f\"\\nLoading fold {fold + 1} model...\")\n    \n    checkpoint = torch.load(CFG.output_dir / f'model_fold{fold}.pth', weights_only=False)\n    num_classes = len(checkpoint['label_encoder'].classes_)\n    \n    model = JaguarReIDModel(\n        CFG.model_name,\n        CFG.embedding_dim,\n        num_classes,\n        pretrained=False\n    ).to(device)\n    model.load_state_dict(checkpoint['model_state_dict'])\n    \n    embeddings, filenames = extract_embeddings(model, test_loader, device)\n    all_fold_embeddings.append(embeddings)\n    \n    print(f\"Fold {fold + 1} embeddings shape: {embeddings.shape}\")\n\nensemble_embeddings = np.mean(all_fold_embeddings, axis=0)\nensemble_embeddings = ensemble_embeddings / np.linalg.norm(ensemble_embeddings, axis=1, keepdims=True)\nprint(f\"\\nEnsemble embeddings shape: {ensemble_embeddings.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T08:43:58.342448Z","iopub.execute_input":"2026-03-11T08:43:58.343167Z","iopub.status.idle":"2026-03-11T08:44:56.705271Z","shell.execute_reply.started":"2026-03-11T08:43:58.343134Z","shell.execute_reply":"2026-03-11T08:44:56.704587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create filename to embedding mapping\nfilename_to_idx = {name: idx for idx, name in enumerate(filenames)}\n\n# Compute similarities for test pairs\nprint(\"Computing similarities for test pairs...\")\n\nsimilarities = []\nfor _, row in tqdm(test_df.iterrows(), total=len(test_df)):\n    query_idx = filename_to_idx[row['query_image']]\n    gallery_idx = filename_to_idx[row['gallery_image']]\n    \n    query_emb = ensemble_embeddings[query_idx]\n    gallery_emb = ensemble_embeddings[gallery_idx]\n    \n    # Cosine similarity (already normalized, so dot product)\n    sim = np.dot(query_emb, gallery_emb)\n    \n    # Map from [-1, 1] to [0, 1]\n    sim = (sim + 1) / 2\n    similarities.append(sim)\n\nprint(f\"Generated {len(similarities)} similarity scores\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T08:45:05.703797Z","iopub.execute_input":"2026-03-11T08:45:05.704255Z","iopub.status.idle":"2026-03-11T08:45:14.137023Z","shell.execute_reply.started":"2026-03-11T08:45:05.704224Z","shell.execute_reply":"2026-03-11T08:45:14.133297Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 9. Create Submission","metadata":{}},{"cell_type":"code","source":"# Create submission\nsubmission = pd.DataFrame({\n    'row_id': test_df['row_id'],\n    'similarity': similarities\n})\n\n# Validate\nprint(\"Validating submission...\")\nassert len(submission) == 137270, f\"Wrong number of rows: {len(submission)}\"\nassert (submission['similarity'] >= 0).all(), \"Found values < 0\"\nassert (submission['similarity'] <= 1).all(), \"Found values > 1\"\nassert submission['row_id'].tolist() == list(range(137270)), \"row_id mismatch\"\n\nprint(\"Validation passed!\")\nprint(f\"\\nSubmission statistics:\")\nprint(submission['similarity'].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T08:45:28.49408Z","iopub.execute_input":"2026-03-11T08:45:28.494725Z","iopub.status.idle":"2026-03-11T08:45:28.561029Z","shell.execute_reply.started":"2026-03-11T08:45:28.494695Z","shell.execute_reply":"2026-03-11T08:45:28.560298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save submission\nsubmission.to_csv('submission.csv', index=False)\nprint(f\"\\nSubmission saved: submission.csv\")\nprint(f\"File size: {Path('submission.csv').stat().st_size / 1024 / 1024:.2f} MB\")\n\n# Preview\nsubmission.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T08:45:33.20429Z","iopub.execute_input":"2026-03-11T08:45:33.204953Z","iopub.status.idle":"2026-03-11T08:45:33.401695Z","shell.execute_reply.started":"2026-03-11T08:45:33.204924Z","shell.execute_reply":"2026-03-11T08:45:33.400839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize similarity distribution\nfig, axes = plt.subplots(1, 2, figsize=(14, 5))\n\naxes[0].hist(submission['similarity'], bins=50, edgecolor='black', alpha=0.7)\naxes[0].set_xlabel('Similarity Score')\naxes[0].set_ylabel('Count')\naxes[0].set_title('Distribution of Predicted Similarities')\n\n# Log scale for better visibility\naxes[1].hist(submission['similarity'], bins=50, edgecolor='black', alpha=0.7, log=True)\naxes[1].set_xlabel('Similarity Score')\naxes[1].set_ylabel('Count (log scale)')\naxes[1].set_title('Distribution of Predicted Similarities (Log Scale)')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T08:45:36.876362Z","iopub.execute_input":"2026-03-11T08:45:36.876748Z","iopub.status.idle":"2026-03-11T08:45:37.460095Z","shell.execute_reply.started":"2026-03-11T08:45:36.876724Z","shell.execute_reply":"2026-03-11T08:45:37.459432Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 10. Additional Analysis (Optional)","metadata":{}},{"cell_type":"code","source":"# Visualize embedding space using t-SNE (on training data)\nfrom sklearn.manifold import TSNE\n\n# Extract embeddings for training data\ntrain_dataset_full = JaguarDataset(\n    train_df, CFG.train_dir,\n    transform=get_transforms('val'),\n    mode='train'\n)\ntrain_loader_full = DataLoader(\n    train_dataset_full, batch_size=CFG.batch_size,\n    shuffle=False, num_workers=CFG.num_workers\n)\n\n# Use best fold model\nbest_fold = np.argmax(fold_results)\ncheckpoint = torch.load(CFG.output_dir / f'model_fold{best_fold}.pth', weights_only=False)\nmodel = JaguarReIDModel(\n    CFG.model_name, CFG.embedding_dim,\n    len(checkpoint['label_encoder'].classes_),\n    pretrained=False\n).to(device)\nmodel.load_state_dict(checkpoint['model_state_dict'])\n\ntrain_embeddings = []\ntrain_labels = []\n\nmodel.eval()\nwith torch.no_grad():\n    for images, labels in tqdm(train_loader_full, desc='Extracting train embeddings'):\n        images = images.to(device)\n        emb = model.extract_features(images)\n        train_embeddings.append(emb.cpu().numpy())\n        train_labels.extend(labels.numpy())\n\ntrain_embeddings = np.vstack(train_embeddings)\ntrain_labels = np.array(train_labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T08:45:45.82241Z","iopub.execute_input":"2026-03-11T08:45:45.823142Z","iopub.status.idle":"2026-03-11T08:50:38.875073Z","shell.execute_reply.started":"2026-03-11T08:45:45.823114Z","shell.execute_reply":"2026-03-11T08:50:38.874346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# t-SNE visualization\nprint(\"Running t-SNE (this may take a few minutes)...\")\ntsne = TSNE(n_components=2, random_state=42, perplexity=30, n_iter=1000)\nembeddings_2d = tsne.fit_transform(train_embeddings)\n\n# Plot\nplt.figure(figsize=(14, 10))\n\nunique_labels = np.unique(train_labels)\ncolors = plt.cm.tab20(np.linspace(0, 1, len(unique_labels)))\n\nfor label, color in zip(unique_labels, colors):\n    mask = train_labels == label\n    jaguar_name = train_dataset_full.label_encoder.inverse_transform([label])[0]\n    plt.scatter(\n        embeddings_2d[mask, 0],\n        embeddings_2d[mask, 1],\n        c=[color],\n        label=jaguar_name,\n        alpha=0.6,\n        s=20\n    )\n\nplt.legend(bbox_to_anchor=(1.05, 1), loc='upper left', fontsize=8, ncol=2)\nplt.title('t-SNE Visualization of Jaguar Embeddings')\nplt.xlabel('t-SNE 1')\nplt.ylabel('t-SNE 2')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-11T09:56:12.085356Z","iopub.execute_input":"2026-03-11T09:56:12.085822Z","iopub.status.idle":"2026-03-11T09:56:18.781561Z","shell.execute_reply.started":"2026-03-11T09:56:12.085769Z","shell.execute_reply":"2026-03-11T09:56:18.780703Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 11. Summary & Next Steps\n\n### What we implemented:\n1. **EfficientNet-B3 backbone** with ArcFace loss for metric learning\n2. **Focal Loss** with class weights to handle imbalance\n3. **5-Fold Cross Validation** with ensemble predictions\n4. **Data augmentation**: flips, rotations, color jitter, random erasing\n5. **Proper alpha channel handling** for transparent PNGs\n\n### Potential improvements:\n- [ ] Use **MegaDescriptor** or **DINOv2** as backbone\n- [ ] Add **Test-Time Augmentation (TTA)** for inference\n- [ ] Try **CosFace** or **Sub-Center ArcFace**\n- [ ] Implement **triplet loss** or **contrastive loss**\n- [ ] Use **larger image size** (448, 512)\n- [ ] Apply **mixup/cutmix** augmentation\n- [ ] Train with **gradient accumulation** for larger effective batch size\n- [ ] Use **label smoothing**\n- [ ] Try **re-ranking** with k-reciprocal encoding","metadata":{}}]}