{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-08-21T12:25:20.423525Z","iopub.execute_input":"2026-08-21T12:25:20.424390Z","iopub.status.idle":"2026-08-21T12:26:39.269297Z","shell.execute_reply.started":"2026-08-21T12:25:20.424358Z","shell.execute_reply":"2026-08-21T12:26:39.267114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# CELL 1 — Setup, config, determinism\n# =============================================================================\n# Everything the run is configured with lives in the CONFIG dict below. It is\n# dumped to config_used.yaml next to the checkpoint so the run can be reproduced.\n\nimport hashlib\nimport io\nimport json\nimport math\nimport os\nimport random\nimport time\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom PIL import Image, ImageDraw\nfrom scipy import ndimage\n\nimport matplotlib\nmatplotlib.use(\"Agg\")\nimport matplotlib.pyplot as plt\n\n\nCONFIG = {\n    \"seed\": 42,\n\n    # --- data -----------------------------------------------------------------\n    \"data_root\": \"/kaggle/input/competitions/imaterialist-fashion-2020-fgvc7\",\n    \"images_dirname\": \"train\",\n    \"annotations_csv\": \"train.csv\",\n    \"categories_json\": \"label_descriptions.json\",\n    \"image_size\": 384,        # not 256: at 256 the armhole seam is 1-2px and is\n                              # partly destroyed by mask resizing before training starts\n    \"rle_order\": \"auto\",      # auto | F | C  -- detected by mask compactness, then logged\n    \"max_images\": 600,\n    \"min_body_pixel_frac.02,tion\": 0\n    \"require_sleeve\": True,\n\n    # --- classes --------------------------------------------------------------\n    # Resolved BY NAME at runtime against label_descriptions.json, never by\n    # hardcoded integer id: the id ordering differs between the Fashionpedia COCO\n    # release and this Kaggle release, and a wrong id still trains to a\n    # believable-looking loss curve.\n    \"class_names\": [\"background\", \"body\", \"sleeve\"],\n    \"body_categories\": [\n        \"shirt, blouse\", \"top, t-shirt, sweatshirt\", \"sweater\", \"cardigan\",\n        \"jacket\", \"vest\", \"coat\", \"dress\", \"jumpsuit\", \"cape\",\n    ],\n    \"sleeve_categories\": [\"sleeve\"],\n    # Garment parts that are NOT the placeable torso panel. Subtracted from body,\n    # and used to supervise the boundary head.\n    \"trim_categories\": [\"collar\", \"lapel\", \"hood\", \"epaulette\", \"neckline\"],\n\n    # --- splits ---------------------------------------------------------------\n    # Manifest-first. The brief supplies exact train/val image ids (\"use these and\n    # only these\"); if data/splits/train.txt + val.txt exist they are used verbatim\n    # and hashed. The hash-based fallback exists only so this runs before they arrive.\n    \"manifest_dir\": \"data/splits\",\n    \"strict_manifest\": False,\n    \"fallback_val_fraction\": 0.2,\n    \"dev_fraction\": 0.15,      # carved out of TRAIN. Every threshold is picked here,\n                               # never on val, so val stays a fair proxy for their test split.\n    \"hash_salt\": \"fabrics-v1\",\n\n    # --- leakage audit --------------------------------------------------------\n    \"leakage_hash_size\": 8,        # 8x8 dHash -> 64 bits\n    \"leakage_max_hamming\": 6,\n\n    # --- augmentation ---------------------------------------------------------\n    \"aug_rotation_degrees\": 10,\n    \"aug_translate_fraction\": 0.10,\n    \"aug_scale_jitter\": 0.10,\n    \"aug_brightness\": 0.15,\n    \"aug_contrast\": 0.15,\n    \"aug_hflip_probability\": 0.5,\n    \"aug_jpeg_quality\": [70, 100],\n\n    # --- model ----------------------------------------------------------------\n    \"backbone\": \"resnet34\",        # frozen; frozen params do not count toward the 2M cap\n    \"pretrained\": True,            # needs Kaggle notebook internet ON\n    \"decoder_channels\": 128,   # ~0.27M trainable. We are ~7x under the 2M cap on\n                               # purpose: with a few hundred images the binding\n                               # constraint is data, not parameters.\n    \"use_refine_head\": True,\n    \"use_boundary_head\": True,\n    \"trainable_parameter_cap\": 2_000_000,\n\n    # --- training -------------------------------------------------------------\n    \"epochs\": 40,\n    \"batch_size\": 8,\n    \"learning_rate\": 3e-3,\n    \"weight_decay\": 1e-4,\n    \"warmup_epochs\": 2,\n    \"amp\": True,\n    \"num_workers\": 2,\n    \"early_stop_patience\": 8,\n    \"dice_weight\": 1.0,\n    \"boundary_weight\": 0.5,\n    \"boundary_pos_weight\": 10.0,\n    \"train_seam_band_width\": 3,\n\n    # --- evaluation -----------------------------------------------------------\n    \"boundary_tolerance_px\": 2,\n    \"qualitative_samples\": 6,\n\n    # --- placement ------------------------------------------------------------\n    # Bands are in normalised GARMENT coordinates: u across the garment\n    # (0 = viewer-left, 1 = viewer-right), v down it (0 = neck, 1 = hem).\n    # Because both axes come from the mask's own principal axes, a band follows the\n    # garment when the photo is tilted or cropped. Image-aligned boxes do not.\n    \"band_vertical\": {\n        \"upper\":  [0.10, 0.30],\n        \"chest\":  [0.18, 0.45],\n        \"centre\": [0.25, 0.60],\n        \"hem\":    [0.70, 0.92],\n    },\n    \"band_horizontal\": {\n        \"viewer_left\":  [0.12, 0.45],\n        \"viewer_right\": [0.55, 0.88],\n        # Wide, because a \"centred\" print is a large front print spanning most of\n        # the chest -- not a small logo that happens to sit in the middle.\n        \"centre\":       [0.22, 0.78],\n    },\n    # Apparel convention: \"left chest\" means the WEARER's left, which on a\n    # front-facing photo is the RIGHT-hand side of the image. Flip this one flag if\n    # their evaluation_cases.csv turns out to use viewer-relative wording.\n    \"wearer_left_is_viewer_right\": True,\n\n    \"place_seam_band_width\": 5,\n    \"place_boundary_threshold\": 0.5,\n    \"place_safety_erosion_px\": 4,\n    # Artwork size caps, as a fraction of the garment's own width. These are\n    # apparel constants, not tuning knobs: a left-chest logo is ~3in on a ~20in\n    # chest, while a centred front print is a different product entirely -- the\n    # brief itself notes the old system \"holds up for a large centred print\".\n    \"place_max_width_fraction_chest_logo\": 0.22,\n    \"place_max_width_fraction_centred_print\": 0.55,\n    \"place_min_width_fraction\": 0.06,\n    \"place_min_axis_ratio\": 1.15,\n    # assigned_region = every artwork pixel must land inside the instructed band\n    # (strict, and what the brief asks for). safe_mask_only relaxes it to \"anywhere\n    # on the torso panel\", which is useful for debugging but not for scoring.\n    \"containment_mode\": \"assigned_region\",\n\n    # --- confidence gate ------------------------------------------------------\n    \"confidence_threshold\": 0.35,       # chosen on DEV, never on val\n    \"min_region_area_fraction\": 0.01,\n    \"axis_ratio_full_confidence\": 1.5,  # above this the garment axis is unambiguous\n    \"fallback_to_centre_chest\": True,\n\n    # --- output ---------------------------------------------------------------\n    \"output_dir\": \"outputs\",\n    \"cache_dir\": \"data/cache\",\n}\n\n# Class ids. Also the paint order when rasterising the label map: later wins.\nBACKGROUND, BODY, SLEEVE = 0, 1, 2\nNUM_CLASSES = 3\n\nIMAGENET_MEAN = (0.485, 0.456, 0.406)\nIMAGENET_STD = (0.229, 0.224, 0.225)\n\n# Refusal reason codes, so refusals are countable in production rather than one\n# undifferentiated \"failed\" bucket.\nPLACED = \"PLACED\"\nNO_GARMENT_DETECTED = \"NO_GARMENT_DETECTED\"\nREGION_TOO_SMALL = \"REGION_TOO_SMALL\"\nARTWORK_DOES_NOT_FIT = \"ARTWORK_DOES_NOT_FIT\"\nMASK_FRAGMENTED = \"MASK_FRAGMENTED\"\nAXIS_UNSTABLE = \"AXIS_UNSTABLE\"\nLOW_CONFIDENCE = \"LOW_CONFIDENCE\"\n\n\ndef seed_everything(seed):\n    \"\"\"Seed every source of randomness this notebook touches.\n\n    torch.manual_seed alone is not enough: the DataLoader workers each need\n    `random` and `numpy` reseeded too (done in worker_init below), otherwise the\n    augmentation stream silently differs between runs.\n    \"\"\"\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    os.environ.setdefault(\"CUBLAS_WORKSPACE_CONFIG\", \":4096:8\")\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    torch.use_deterministic_algorithms(True, warn_only=True)\n\n\ndef worker_init(worker_id):\n    \"\"\"Reseed `random` and `numpy` inside each DataLoader worker.\"\"\"\n    worker_seed = torch.initial_seed() % 2**32\n    random.seed(worker_seed)\n    np.random.seed(worker_seed)\n\n\ndef stable_unit_hash(value, salt):\n    \"\"\"Map a string to a stable float in [0, 1).\n\n    Used to derive split membership from an image id, so the split is reproducible\n    from CONFIG alone with no extra file to keep in sync. Python's built-in hash()\n    is randomised per process and must never be used for this.\n    \"\"\"\n    digest = hashlib.sha256(f\"{salt}:{value}\".encode(\"utf-8\")).digest()\n    return int.from_bytes(digest[:8], \"big\") / float(1 << 64)\n\n\ndef config_hash(config):\n    \"\"\"Short stable hash of the config, stamped onto every artefact produced.\"\"\"\n    payload = json.dumps(config, sort_keys=True).encode(\"utf-8\")\n    return hashlib.sha256(payload).hexdigest()[:12]\n\n\n# ---------------------------------------------------------------------------\n# Mask primitives. SciPy rather than OpenCV: it is already in the Kaggle image.\n# ---------------------------------------------------------------------------\n\ndef disk(radius):\n    \"\"\"Disk-shaped structuring element.\n\n    A disk, not a square: a square dilation biases the seam band along the\n    diagonals, which shows up as corner artefacts on near-45-degree armholes.\n    \"\"\"\n    if radius < 1:\n        return np.ones((1, 1), dtype=bool)\n    span = np.arange(-radius, radius + 1)\n    yy, xx = np.meshgrid(span, span, indexing=\"ij\")\n    return (yy ** 2 + xx ** 2) <= radius ** 2\n\n\ndef dilate(mask, radius):\n    \"\"\"Grow a boolean mask by `radius` pixels.\"\"\"\n    return mask.copy() if radius < 1 else ndimage.binary_dilation(mask, disk(radius))\n\n\ndef erode(mask, radius):\n    \"\"\"Shrink a boolean mask by `radius` pixels.\"\"\"\n    return mask.copy() if radius < 1 else ndimage.binary_erosion(mask, disk(radius))\n\n\ndef largest_component(mask):\n    \"\"\"Keep only the biggest connected blob.\n\n    Predicted garment masks routinely carry stray specks on the background, and the\n    placement principal axis is sensitive to those outliers.\n    \"\"\"\n    if not mask.any():\n        return np.zeros_like(mask, dtype=bool)\n    labelled, count = ndimage.label(mask)\n    if count <= 1:\n        return mask.copy()\n    sizes = np.bincount(labelled.ravel())\n    sizes[0] = 0  # index 0 is background and must not win the argmax\n    return labelled == int(sizes.argmax())\n\n\ndef component_cohesion(mask):\n    \"\"\"Fraction of a mask's area living in its single largest component.\n\n    1.0 is one clean blob. Well below 1.0 means the prediction broke the garment\n    into pieces, which is one of the signals that vetoes a placement.\n    \"\"\"\n    total = int(mask.sum())\n    return 0.0 if total == 0 else float(largest_component(mask).sum()) / total\n\n\ndef distance_to_edge(mask):\n    \"\"\"Euclidean distance from each True pixel to the nearest False pixel.\n\n    This is what makes the placement anchor well defined: the maximum is the\n    deepest point inside the safe region, i.e. the most clearance in every\n    direction at once.\n    \"\"\"\n    return ndimage.distance_transform_edt(mask).astype(np.float32)\n\n\ndef resize_mask(mask, size):\n    \"\"\"Nearest-neighbour resize of a boolean mask to `size` x `size`.\n\n    Never interpolate a label map. Bilinear resampling invents fractional class\n    values, and it does so exactly at the boundaries this project exists to find.\n    \"\"\"\n    resized = Image.fromarray(mask.astype(np.uint8) * 255).resize(\n        (size, size), resample=Image.NEAREST\n    )\n    return np.array(resized) > 127\n\n\ndef resize_label_map(label_map, size):\n    \"\"\"Nearest-neighbour resize of a multi-class label map.\"\"\"\n    resized = Image.fromarray(label_map).resize((size, size), resample=Image.NEAREST)\n    return np.array(resized, dtype=np.uint8)\n\n\nseed_everything(CONFIG[\"seed\"])\n\nDEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nOUTPUT_DIR = Path(CONFIG[\"output_dir\"])\nOUTPUT_DIR.mkdir(parents=True, exist_ok=True)\nCONFIG_HASH = config_hash(CONFIG)\n\nprint(f\"device={DEVICE}  torch={torch.__version__}  config_hash={CONFIG_HASH}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T12:26:39.271067Z","iopub.execute_input":"2026-08-21T12:26:39.271491Z","iopub.status.idle":"2026-08-21T12:26:39.307008Z","shell.execute_reply.started":"2026-08-21T12:26:39.271454Z","shell.execute_reply":"2026-08-21T12:26:39.306080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# CELL 2 — Dataset loading, RLE decode, category resolution, label map\n# =============================================================================\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/imaterialist-fashion-2020-fgvc7'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\nDATA_ROOT = Path(CONFIG[\"data_root\"])\nIMAGES_DIR = DATA_ROOT / CONFIG[\"images_dirname\"]\n\nannotations = pd.read_csv(DATA_ROOT / CONFIG[\"annotations_csv\"])\n\n# ClassId is sometimes written \"<class>_<attribute>\"; only the leading integer is\n# the category.\nannotations[\"category_id\"] = (\n    annotations[\"ClassId\"].astype(str).str.split(\"_\").str[0].astype(int)\n)\n\nwith open(DATA_ROOT / CONFIG[\"categories_json\"]) as handle:\n    label_descriptions = json.load(handle)\n\nCATEGORY_NAMES = {int(c[\"id\"]): str(c[\"name\"]) for c in label_descriptions[\"categories\"]}\nprint(f\"{len(annotations):,} annotations over {annotations.ImageId.nunique():,} images\")\nprint(f\"{len(CATEGORY_NAMES)} categories\")\n\n\n# ---------------------------------------------------------------------------\n# Run-length decoding\n# ---------------------------------------------------------------------------\n\ndef decode_rle(encoded_pixels, height, width, order):\n    \"\"\"Decode a Kaggle run-length string into a boolean mask.\n\n    Format is space-separated `start length start length ...` with 1-indexed\n    starts over the flattened image.\n    \"\"\"\n    tokens = encoded_pixels.split()\n    assert len(tokens) % 2 == 0, f\"odd RLE token count: {len(tokens)}\"\n\n    values = np.asarray(tokens, dtype=np.int64)\n    starts = values[0::2] - 1          # the format is 1-indexed\n    ends = starts + values[1::2]\n\n    flat = np.zeros(height * width, dtype=bool)\n    assert ends.max() <= flat.size, \"RLE runs past the end of the image\"\n    for start, end in zip(starts, ends):\n        flat[start:end] = True\n\n    # Column-major ('F') and row-major ('C') both decode without error; only one is\n    # correct, and the wrong one is a striped mess that still trains. See\n    # detect_rle_order below.\n    return flat.reshape((width, height)).T if order == \"F\" else flat.reshape((height, width))\n\n\ndef mask_compactness(mask):\n    \"\"\"perimeter^2 / area. Lower means more blob-like.\n\n    A correctly decoded garment mask is one or two compact blobs. Decoded with the\n    wrong flattening order it becomes fine stripes, whose perimeter explodes while\n    the area stays the same — a difference of one to two orders of magnitude.\n    \"\"\"\n    area = int(mask.sum())\n    if area == 0:\n        return float(\"inf\")\n    padded = np.pad(mask, 1)\n    perimeter = int(\n        (padded[1:-1, 1:] != padded[1:-1, :-1]).sum()\n        + (padded[1:, 1:-1] != padded[:-1, 1:-1]).sum()\n    )\n    return (perimeter ** 2) / area\n\n\ndef detect_rle_order(annotations_df, sample_size=24, seed=0):\n    \"\"\"Decide whether this release flattens column-major or row-major.\n\n    Decodes a fixed sample both ways and keeps the order producing more compact\n    masks, rather than trusting a remembered convention.\n    \"\"\"\n    sample = annotations_df.sample(n=min(sample_size, len(annotations_df)), random_state=seed)\n    scores = {\"F\": [], \"C\": []}\n    for _, row in sample.iterrows():\n        for order in scores:\n            mask = decode_rle(row.EncodedPixels, int(row.Height), int(row.Width), order)\n            scores[order].append(mask_compactness(mask))\n    means = {order: float(np.mean(values)) for order, values in scores.items()}\n    chosen = min(means, key=means.get)\n    print(f\"RLE order = {chosen}  (mean perimeter^2/area: F={means['F']:.1f}, C={means['C']:.1f})\")\n    return chosen\n\n\nRLE_ORDER = (\n    detect_rle_order(annotations) if CONFIG[\"rle_order\"] == \"auto\" else CONFIG[\"rle_order\"]\n)\n\n\n# ---------------------------------------------------------------------------\n# Category names -> our three classes\n# ---------------------------------------------------------------------------\n\ndef normalise_name(name):\n    \"\"\"Canonicalise a category name so 'Shirt, blouse' and 'shirt,blouse' match.\"\"\"\n    return \" \".join(name.strip().lower().split()).replace(\", \", \",\")\n\n\ndef resolve_categories(configured_names, group, category_names):\n    \"\"\"Map configured category names onto this dataset's integer ids.\n\n    Fails loudly on an unmatched name. An unmatched category silently becomes\n    background and still trains to a believable loss curve, which is the most\n    expensive kind of bug in this project.\n    \"\"\"\n    lookup = {normalise_name(name): cid for cid, name in category_names.items()}\n    resolved, missing = set(), []\n    for name in configured_names:\n        cid = lookup.get(normalise_name(name))\n        (missing.append(name) if cid is None else resolved.add(cid))\n    assert not missing, (\n        f\"{group} categories not in this dataset: {missing}\\n\"\n        f\"available: {sorted(category_names.values())}\"\n    )\n    return resolved\n\n\nBODY_IDS = resolve_categories(CONFIG[\"body_categories\"], \"body\", CATEGORY_NAMES)\nSLEEVE_IDS = resolve_categories(CONFIG[\"sleeve_categories\"], \"sleeve\", CATEGORY_NAMES)\nTRIM_IDS = resolve_categories(CONFIG[\"trim_categories\"], \"trim\", CATEGORY_NAMES)\n\nprint(\"body   ids:\", sorted(BODY_IDS), [CATEGORY_NAMES[i] for i in sorted(BODY_IDS)])\nprint(\"sleeve ids:\", sorted(SLEEVE_IDS), [CATEGORY_NAMES[i] for i in sorted(SLEEVE_IDS)])\nprint(\"trim   ids:\", sorted(TRIM_IDS), [CATEGORY_NAMES[i] for i in sorted(TRIM_IDS)])\n\n\nANNOTATIONS_BY_IMAGE = annotations.groupby(\"ImageId\", sort=True)\nALL_IMAGE_IDS = list(ANNOTATIONS_BY_IMAGE.groups.keys())\n\n\ndef image_path_for(image_id):\n    \"\"\"Resolve an image id to a file path, tolerating a missing extension.\"\"\"\n    direct = IMAGES_DIR / str(image_id)\n    if direct.exists():\n        return direct\n    return IMAGES_DIR / f\"{image_id}.jpg\"\n\n\ndef build_label_map(image_id):\n    \"\"\"Collapse one image's overlapping annotations into a single label map.\n\n    Fashionpedia annotates parts as separate instances that overlap their parent\n    garment, so the torso panel is the garment mask MINUS every non-torso part.\n\n    Paint order, later wins:\n      1. background everywhere\n      2. body   -- union of the sleeved upper-body garment categories\n      3. trim   -- collar/lapel/hood/epaulette/neckline painted back to background,\n                   because a logo must never land there and we have no 4th class\n      4. sleeve -- last, so a sleeve overlapping a collar reads as sleeve\n\n    Returns (label_map uint8 HxW in {0,1,2}, trim_mask bool HxW). The trim mask is\n    returned separately because it supervises the boundary head.\n    \"\"\"\n    rows = ANNOTATIONS_BY_IMAGE.get_group(image_id)\n    height, width = int(rows.iloc[0].Height), int(rows.iloc[0].Width)\n\n    body = np.zeros((height, width), dtype=bool)\n    sleeve = np.zeros((height, width), dtype=bool)\n    trim = np.zeros((height, width), dtype=bool)\n\n    for _, row in rows.iterrows():\n        category_id = int(row.category_id)\n        if category_id not in BODY_IDS and category_id not in SLEEVE_IDS and category_id not in TRIM_IDS:\n            continue\n        mask = decode_rle(row.EncodedPixels, height, width, RLE_ORDER)\n        if category_id in BODY_IDS:\n            body |= mask\n        elif category_id in SLEEVE_IDS:\n            sleeve |= mask\n        else:\n            trim |= mask\n\n    label_map = np.full((height, width), BACKGROUND, dtype=np.uint8)\n    label_map[body] = BODY\n    label_map[trim] = BACKGROUND\n    label_map[sleeve] = SLEEVE\n    return label_map, trim","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T12:26:39.308093Z","iopub.execute_input":"2026-08-21T12:26:39.308498Z","iopub.status.idle":"2026-08-21T12:26:57.599174Z","shell.execute_reply.started":"2026-08-21T12:26:39.308463Z","shell.execute_reply":"2026-08-21T12:26:57.598469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# CELL 3 — Candidate selection, preprocessing cache, visual sanity check\n# =============================================================================\n# Decoding RLE at full resolution costs about a second per image. Doing it once\n# into a cache rather than once per epoch per worker is the difference between a\n# run that fits in a free Kaggle session and one that does not.\n\nCACHE_DIR = Path(CONFIG[\"cache_dir\"]) / f\"size{CONFIG['image_size']}\"\nCACHE_DIR.mkdir(parents=True, exist_ok=True)\n\nCLASS_OVERLAY_COLOURS = {BODY: (0, 168, 168), SLEEVE: (240, 160, 40)}\nSEAM_OVERLAY_COLOUR = (220, 40, 180)\n\n\ndef select_candidate_image_ids():\n    \"\"\"Images that carry a torso garment, and a sleeve if we require one.\n\n    An image with no sleeve annotation teaches nothing about the body/sleeve\n    boundary, which is the structure this whole system exists to find.\n\n    Ordered by a salted hash rather than alphabetically before the cap is applied,\n    because iMaterialist image ids are content-addressed and an alphabetical prefix\n    is an arbitrary but non-random subset.\n    \"\"\"\n    has_body = set(annotations.loc[annotations.category_id.isin(BODY_IDS), \"ImageId\"])\n    candidates = has_body\n    if CONFIG[\"require_sleeve\"]:\n        has_sleeve = set(annotations.loc[annotations.category_id.isin(SLEEVE_IDS), \"ImageId\"])\n        candidates = has_body & has_sleeve\n\n    ordered = sorted(candidates, key=lambda i: stable_unit_hash(str(i), CONFIG[\"hash_salt\"] + \":pick\"))\n    if CONFIG[\"max_images\"]:\n        ordered = ordered[: CONFIG[\"max_images\"]]\n    return ordered\n\n\ndef cache_path_for(image_id):\n    \"\"\"Where one preprocessed sample lives.\"\"\"\n    return CACHE_DIR / f\"{image_id}.npz\"\n\n\ndef preprocess_and_cache(image_ids):\n    \"\"\"Build the cache and return the ids that survived, plus per-image stats.\n\n    Drops images whose torso is a speck (bad crop, or a garment that is mostly\n    occluded), because they contribute almost no supervision and they distort the\n    class balance numbers.\n    \"\"\"\n    size = CONFIG[\"image_size\"]\n    kept, dropped, stats = [], [], []\n\n    for position, image_id in enumerate(image_ids):\n        if position % 50 == 0:\n            print(f\"  preprocessing {position}/{len(image_ids)}\", flush=True)\n\n        path = cache_path_for(image_id)\n        if path.exists():\n            with np.load(path) as payload:\n                label_map = payload[\"label_map\"]\n        else:\n            try:\n                full_label_map, full_trim = build_label_map(image_id)\n                with Image.open(image_path_for(image_id)) as handle:\n                    image = np.array(\n                        handle.convert(\"RGB\").resize((size, size), resample=Image.BILINEAR),\n                        dtype=np.uint8,\n                    )\n            except (OSError, ValueError, AssertionError) as error:\n                dropped.append((image_id, f\"unreadable: {error}\"))\n                continue\n\n            label_map = resize_label_map(full_label_map, size)\n            trim_mask = resize_mask(full_trim, size)\n            np.savez_compressed(\n                path, image=image, label_map=label_map, trim_mask=trim_mask.astype(np.uint8)\n            )\n\n        body_fraction = float((label_map == BODY).mean())\n        sleeve_fraction = float((label_map == SLEEVE).mean())\n\n        if body_fraction < CONFIG[\"min_body_pixel_fraction\"]:\n            dropped.append((image_id, f\"body fraction {body_fraction:.4f} below threshold\"))\n            continue\n        if CONFIG[\"require_sleeve\"] and sleeve_fraction == 0.0:\n            dropped.append((image_id, \"sleeve vanished at training resolution\"))\n            continue\n\n        kept.append(image_id)\n        stats.append({\"image_id\": image_id, \"body\": body_fraction, \"sleeve\": sleeve_fraction})\n\n    return kept, dropped, pd.DataFrame(stats)\n\n\ndef load_cached(image_id):\n    \"\"\"Read one preprocessed sample back out of the cache.\"\"\"\n    with np.load(cache_path_for(image_id)) as payload:\n        return (\n            payload[\"image\"],\n            payload[\"label_map\"],\n            payload[\"trim_mask\"].astype(bool),\n        )\n\n\ndef overlay_label_map(image, label_map, seam=None, alpha=0.45):\n    \"\"\"Tint an image by class, for eyeballing that the labels are actually right.\"\"\"\n    canvas = image.astype(np.float32).copy()\n    for class_id, colour in CLASS_OVERLAY_COLOURS.items():\n        selection = label_map == class_id\n        canvas[selection] = (1 - alpha) * canvas[selection] + alpha * np.array(colour, dtype=np.float32)\n    if seam is not None:\n        canvas[seam] = np.array(SEAM_OVERLAY_COLOUR, dtype=np.float32)\n    return canvas.astype(np.uint8)\n\n\ndef save_label_preview(image_ids, filename=\"label_sanity_check.png\", columns=4):\n    \"\"\"Save a grid of label overlays.\n\n    This is the check that the RLE order and the class mapping are right. It is\n    worth thirty seconds of looking: both failure modes produce a mask that trains\n    without complaint and evaluates to a plausible number.\n    \"\"\"\n    rows = math.ceil(len(image_ids) / columns)\n    figure, axes = plt.subplots(rows, columns, figsize=(3.2 * columns, 3.2 * rows))\n    for axis, image_id in zip(np.ravel(axes), image_ids):\n        image, label_map, _ = load_cached(image_id)\n        axis.imshow(overlay_label_map(image, label_map))\n        axis.set_title(str(image_id)[:14], fontsize=8)\n    for axis in np.ravel(axes):\n        axis.axis(\"off\")\n    figure.suptitle(\"teal = body   amber = sleeve   (verify before training)\", fontsize=10)\n    figure.tight_layout()\n    figure.savefig(OUTPUT_DIR / filename, dpi=110)\n    plt.close(figure)\n    print(f\"wrote {OUTPUT_DIR / filename}\")\n\n\ncandidate_ids = select_candidate_image_ids()\nprint(f\"{len(candidate_ids)} candidate images (torso + sleeve present)\")\n\nUSABLE_IDS, DROPPED, CLASS_STATS = preprocess_and_cache(candidate_ids)\nprint(f\"kept {len(USABLE_IDS)}, dropped {len(DROPPED)}\")\nif DROPPED:\n    print(\"  example drops:\", DROPPED[:3])\n\nprint(\n    \"mean pixel fraction  body={:.3f}  sleeve={:.3f}  background={:.3f}\".format(\n        CLASS_STATS.body.mean(),\n        CLASS_STATS.sleeve.mean(),\n        1 - CLASS_STATS.body.mean() - CLASS_STATS.sleeve.mean(),\n    )\n)\n\nsave_label_preview(USABLE_IDS[:8])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T12:26:57.602282Z","iopub.execute_input":"2026-08-21T12:26:57.602593Z","iopub.status.idle":"2026-08-21T12:26:59.077897Z","shell.execute_reply.started":"2026-08-21T12:26:57.602567Z","shell.execute_reply":"2026-08-21T12:26:59.076785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# CELL 4 — Train / val / dev split, and the cross-split leakage audit\n# =============================================================================\n# Manifest-first. The brief supplies exact train/val image ids and says \"use these\n# and only these\", so a supplied manifest is used verbatim and its SHA-256 is\n# recorded. The hash-based fallback exists only so this runs before those files\n# arrive; it is ignored the moment they are dropped into data/splits/.\n\nMANIFEST_DIR = Path(CONFIG[\"manifest_dir\"])\n\n\ndef read_id_list(path):\n    \"\"\"Read a newline-delimited id list, ignoring blanks and comments.\"\"\"\n    return [\n        line.strip()\n        for line in path.read_text().splitlines()\n        if line.strip() and not line.startswith(\"#\")\n    ]\n\n\ndef find_supplied_manifest():\n    \"\"\"Return (train_ids, val_ids, sha256) if the company's manifest is present.\"\"\"\n    train_path, val_path = MANIFEST_DIR / \"train.txt\", MANIFEST_DIR / \"val.txt\"\n    combined_path = MANIFEST_DIR / \"splits.csv\"\n\n    if train_path.exists() and val_path.exists():\n        digest = hashlib.sha256()\n        for path in (train_path, val_path):\n            digest.update(path.read_bytes())\n        return read_id_list(train_path), read_id_list(val_path), digest.hexdigest()\n\n    if combined_path.exists():\n        frame = pd.read_csv(combined_path)\n        frame.columns = [c.lower() for c in frame.columns]\n        assert {\"image_id\", \"split\"} <= set(frame.columns), (\n            f\"{combined_path} needs 'image_id' and 'split' columns\"\n        )\n        train_ids = frame.loc[frame.split.str.lower().str.startswith(\"train\"), \"image_id\"].astype(str).tolist()\n        val_ids = frame.loc[frame.split.str.lower().str.startswith(\"val\"), \"image_id\"].astype(str).tolist()\n        return train_ids, val_ids, hashlib.sha256(combined_path.read_bytes()).hexdigest()\n\n    return None\n\n\ndef carve_dev_slice(train_ids):\n    \"\"\"Split train ids into (train_without_dev, dev), deterministically by id hash.\n\n    Every threshold and every early-stopping decision is made on dev, so that the\n    validation number stays a fair proxy for the held-out test split. Hash-based\n    rather than a shuffled slice, so dev membership is recomputable from CONFIG\n    alone with no extra file to keep in sync.\n    \"\"\"\n    if CONFIG[\"dev_fraction\"] <= 0:\n        return list(train_ids), []\n    dev = [\n        i for i in train_ids\n        if stable_unit_hash(str(i), CONFIG[\"hash_salt\"] + \":dev\") < CONFIG[\"dev_fraction\"]\n    ]\n    dev_set = set(dev)\n    return [i for i in train_ids if i not in dev_set], dev\n\n\ndef resolve_split(usable_ids):\n    \"\"\"Produce the train/val/dev split and say where it came from.\"\"\"\n    supplied = find_supplied_manifest() if MANIFEST_DIR.exists() else None\n\n    if supplied is None:\n        assert not CONFIG[\"strict_manifest\"], (\n            f\"No manifest in {MANIFEST_DIR} and strict_manifest is set.\"\n        )\n        print(f\"!! No supplied manifest in {MANIFEST_DIR}; using the deterministic \"\n              f\"hash fallback (salt={CONFIG['hash_salt']}).\")\n        val = [\n            i for i in usable_ids\n            if stable_unit_hash(str(i), CONFIG[\"hash_salt\"]) < CONFIG[\"fallback_val_fraction\"]\n        ]\n        val_set = set(val)\n        train, dev = carve_dev_slice([i for i in usable_ids if i not in val_set])\n        return train, val, dev, \"fallback_hash\", None\n\n    train_ids, val_ids, manifest_sha = supplied\n    overlap = set(train_ids) & set(val_ids)\n    assert not overlap, f\"manifest lists {len(overlap)} ids in both splits\"\n\n    # Honour the manifest exactly, but say out loud which of its ids this dataset\n    # copy cannot provide. Silently shrinking a supplied split is precisely what\n    # makes our numbers and theirs disagree.\n    available = set(usable_ids)\n    kept_train = [i for i in train_ids if i in available]\n    kept_val = [i for i in val_ids if i in available]\n    unavailable = (len(train_ids) - len(kept_train)) + (len(val_ids) - len(kept_val))\n    if unavailable:\n        print(f\"!! {unavailable} of {len(train_ids) + len(val_ids)} manifest ids are not \"\n              f\"usable from this dataset copy (missing file, or filtered out in cell 3).\")\n\n    train, dev = carve_dev_slice(kept_train)\n    return train, kept_val, dev, \"supplied\", manifest_sha\n\n\nTRAIN_IDS, VAL_IDS, DEV_IDS, SPLIT_SOURCE, MANIFEST_SHA = resolve_split(USABLE_IDS)\nprint(f\"split source = {SPLIT_SOURCE}\"\n      + (f\"  sha256={MANIFEST_SHA[:12]}\" if MANIFEST_SHA else \"\"))\nprint(f\"train={len(TRAIN_IDS)}  dev={len(DEV_IDS)}  val={len(VAL_IDS)}\")\n\n\n# ---------------------------------------------------------------------------\n# Leakage audit\n# ---------------------------------------------------------------------------\n# Splitting by image id proves no image is in both splits. It does NOT prove there\n# is no leakage: Fashionpedia's images come from free-licence stock sites, where\n# the same model in the same garment is routinely uploaded as several frames with\n# different ids. Those are visually the same photograph and they inflate val IoU.\n\ndef difference_hash(path, hash_size=8):\n    \"\"\"dHash: compare each pixel to its right neighbour on a small grey thumbnail.\n\n    Invariant to resolution, mild recompression and small brightness shifts —\n    exactly the family of differences that separates two uploads of one stock photo.\n    \"\"\"\n    try:\n        with Image.open(path) as handle:\n            thumb = handle.convert(\"L\").resize((hash_size + 1, hash_size), Image.LANCZOS)\n    except (OSError, ValueError):\n        return None\n    pixels = np.asarray(thumb, dtype=np.int16)\n    return (pixels[:, 1:] > pixels[:, :-1]).ravel()\n\n\ndef hash_split(image_ids, hash_size):\n    \"\"\"Hash every image in a split; return (ids kept, stacked hashes).\"\"\"\n    kept, hashes = [], []\n    for image_id in image_ids:\n        digest = difference_hash(image_path_for(image_id), hash_size)\n        if digest is not None:\n            kept.append(image_id)\n            hashes.append(digest)\n    stacked = np.stack(hashes) if hashes else np.zeros((0, hash_size ** 2), dtype=bool)\n    return kept, stacked\n\n\ndef audit_cross_split_duplicates(train_ids, val_ids):\n    \"\"\"Find val images that are perceptual near-duplicates of training images.\n\n    We do NOT move them when the split came from a supplied manifest — that would\n    violate \"use these and only these\". Instead the contaminated val ids are\n    reported, and cell 8 scores validation twice: on the full split and with these\n    excluded. The gap between the two is an honest estimate of how optimistic the\n    headline number is.\n    \"\"\"\n    size = CONFIG[\"leakage_hash_size\"]\n    kept_train, train_hashes = hash_split(train_ids, size)\n    kept_val, val_hashes = hash_split(val_ids, size)\n\n    if not len(train_hashes) or not len(val_hashes):\n        return [], []\n\n    # XOR every val hash against every train hash in one shot.\n    distances = (val_hashes[:, None, :] ^ train_hashes[None, :, :]).sum(axis=2)\n    val_index, train_index = np.nonzero(distances <= CONFIG[\"leakage_max_hamming\"])\n\n    pairs = [\n        {\n            \"val_image_id\": kept_val[int(v)],\n            \"train_image_id\": kept_train[int(t)],\n            \"hamming\": int(distances[v, t]),\n        }\n        for v, t in zip(val_index, train_index)\n    ]\n    contaminated = sorted({pair[\"val_image_id\"] for pair in pairs})\n    return pairs, contaminated\n\n\nDUPLICATE_PAIRS, CONTAMINATED_VAL_IDS = audit_cross_split_duplicates(\n    TRAIN_IDS + DEV_IDS, VAL_IDS\n)\nprint(f\"leakage audit: {len(DUPLICATE_PAIRS)} near-duplicate pairs, \"\n      f\"{len(CONTAMINATED_VAL_IDS)}/{len(VAL_IDS)} val images contaminated \"\n      f\"({100 * len(CONTAMINATED_VAL_IDS) / max(len(VAL_IDS), 1):.1f}%)\")\n\npd.DataFrame(DUPLICATE_PAIRS).to_csv(OUTPUT_DIR / \"leakage_pairs.csv\", index=False)\nfor name, ids in ((\"train\", TRAIN_IDS), (\"val\", VAL_IDS), (\"dev\", DEV_IDS)):\n    (OUTPUT_DIR / f\"split_{name}.txt\").write_text(\"\\n\".join(map(str, ids)) + \"\\n\")\nprint(f\"wrote split_*.txt and leakage_pairs.csv to {OUTPUT_DIR}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T12:26:59.078965Z","iopub.execute_input":"2026-08-21T12:26:59.079226Z","iopub.status.idle":"2026-08-21T12:27:31.437464Z","shell.execute_reply.started":"2026-08-21T12:26:59.079201Z","shell.execute_reply":"2026-08-21T12:27:31.436479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# CELL 5 — Augmentation, seam target, Dataset and DataLoaders\n# =============================================================================\n# Augmentation chosen for garment photography specifically. What was rejected is\n# as much of the answer as what was kept:\n#\n#   KEPT\n#     rotation +/-10 deg      handheld shot angle varies between stock sources\n#     translate/scale +/-10%  framing varies\n#     brightness/contrast     studio vs ambient lighting\n#     JPEG quality 70-100     stock images arrive at wildly different compression,\n#                             and JPEG ringing sits exactly on garment edges\n#     horizontal flip         label-safe HERE ONLY because our three classes do not\n#                             distinguish left from right sleeve. If sleeve is ever\n#                             split into left/right this becomes a label-CORRUPTING\n#                             augmentation that must swap the pair. Worth knowing\n#                             too: a mirrored garment is physically implausible,\n#                             since button plackets are sided (menswear buttons\n#                             right-over-left, womenswear the reverse) -- we accept\n#                             it because we do not model plackets.\n#\n#   REJECTED\n#     heavy colour/hue jitter fabric colour is diagnostic (it separates garment\n#                             from skin and background) and the product is a\n#                             colour-accurate preview\n#     vertical flip           garments have a gravity direction, and it would break\n#                             the neck/hem orientation the placement frame depends on\n#     elastic / perspective   would teach the model to accept distorted garment\n#                             geometry, which is the very input the placement PCA\n#                             frame reads\n#     cutout on the torso     encourages hallucinating body pixels through occlusion,\n#                             so the confidence gate would fire LESS when it should\n#                             fire most (arms crossed over chest)\n\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision.transforms import functional as TF\n\n\ndef seam_band_from_labels(label_map, trim_mask, band_width):\n    \"\"\"Training target for the boundary head.\n\n    Fashionpedia has no seam annotation, but the boundary between the body mask and\n    the sleeve mask IS the armhole seam. Trim counts too: a chest print must not run\n    over a collar seam any more than an armhole seam.\n    \"\"\"\n    radius = max(1, band_width // 2)\n    body = label_map == BODY\n    adjoining = (label_map == SLEEVE) | trim_mask\n    if not body.any() or not adjoining.any():\n        return np.zeros_like(body, dtype=bool)\n    return dilate(body, radius) & dilate(adjoining, radius)\n\n\ndef augment_sample(image, packed_label):\n    \"\"\"Apply one shared geometric transform to the image and the label, plus\n    photometric jitter to the image only.\n\n    `packed_label` carries the trim mask in spare value 3 so a single affine keeps\n    image, labels and trim in registration.\n    \"\"\"\n    image_pil = Image.fromarray(image)\n    label_pil = Image.fromarray(packed_label)\n\n    if random.random() < CONFIG[\"aug_hflip_probability\"]:\n        image_pil, label_pil = TF.hflip(image_pil), TF.hflip(label_pil)\n\n    angle = random.uniform(-CONFIG[\"aug_rotation_degrees\"], CONFIG[\"aug_rotation_degrees\"])\n    max_shift = CONFIG[\"aug_translate_fraction\"] * image_pil.width\n    translate = [int(round(random.uniform(-max_shift, max_shift))) for _ in range(2)]\n    scale = 1.0 + random.uniform(-CONFIG[\"aug_scale_jitter\"], CONFIG[\"aug_scale_jitter\"])\n\n    image_pil = TF.affine(image_pil, angle=angle, translate=translate, scale=scale,\n                          shear=[0.0, 0.0], interpolation=TF.InterpolationMode.BILINEAR, fill=[0, 0, 0])\n    # Areas rotated in from outside the frame become background, which is the honest\n    # label for them: there is genuinely no garment there.\n    label_pil = TF.affine(label_pil, angle=angle, translate=translate, scale=scale,\n                          shear=[0.0, 0.0], interpolation=TF.InterpolationMode.NEAREST, fill=[0])\n\n    image_pil = TF.adjust_brightness(image_pil, 1.0 + random.uniform(-CONFIG[\"aug_brightness\"], CONFIG[\"aug_brightness\"]))\n    image_pil = TF.adjust_contrast(image_pil, 1.0 + random.uniform(-CONFIG[\"aug_contrast\"], CONFIG[\"aug_contrast\"]))\n\n    low, high = CONFIG[\"aug_jpeg_quality\"]\n    buffer = io.BytesIO()\n    image_pil.save(buffer, format=\"JPEG\", quality=random.randint(low, high))\n    buffer.seek(0)\n    with Image.open(buffer) as reopened:\n        image_pil = reopened.convert(\"RGB\")\n\n    return np.array(image_pil, dtype=np.uint8), np.array(label_pil, dtype=np.uint8)\n\n\ndef normalise_image(image):\n    \"\"\"uint8 HWC -> normalised CHW float tensor, ImageNet statistics.\"\"\"\n    tensor = torch.from_numpy(np.ascontiguousarray(image)).permute(2, 0, 1).float().div_(255.0)\n    mean = torch.tensor(IMAGENET_MEAN).view(3, 1, 1)\n    std = torch.tensor(IMAGENET_STD).view(3, 1, 1)\n    return (tensor - mean) / std\n\n\nclass GarmentDataset(Dataset):\n    \"\"\"Serves cached samples, augmenting them for the training split only.\n\n    The seam target is derived here, per item, AFTER augmentation. A band is a thin\n    structure: rotating a pre-computed one with nearest-neighbour sampling\n    perforates it.\n    \"\"\"\n\n    def __init__(self, image_ids, augment):\n        self.image_ids = list(image_ids)\n        self.augment = augment\n\n    def __len__(self):\n        return len(self.image_ids)\n\n    def __getitem__(self, index):\n        image_id = self.image_ids[index]\n        image, label_map, trim_mask = load_cached(image_id)\n\n        if self.augment:\n            packed = label_map.copy()\n            packed[trim_mask & (label_map == BACKGROUND)] = 3\n            image, packed = augment_sample(image, packed)\n            trim_mask = packed == 3\n            label_map = packed.copy()\n            label_map[trim_mask] = BACKGROUND\n\n        seam = seam_band_from_labels(label_map, trim_mask, CONFIG[\"train_seam_band_width\"])\n\n        return {\n            \"image\": normalise_image(image),\n            \"label_map\": torch.from_numpy(label_map.astype(np.int64)),\n            \"seam\": torch.from_numpy(seam.astype(np.float32)),\n        }\n\n\ndef build_loader(image_ids, augment, shuffle):\n    \"\"\"DataLoader with a seeded generator and per-worker reseeding.\"\"\"\n    generator = torch.Generator()\n    generator.manual_seed(CONFIG[\"seed\"])\n    return DataLoader(\n        GarmentDataset(image_ids, augment=augment),\n        batch_size=CONFIG[\"batch_size\"],\n        shuffle=shuffle,\n        num_workers=CONFIG[\"num_workers\"],\n        pin_memory=(DEVICE.type == \"cuda\"),\n        drop_last=False,\n        worker_init_fn=worker_init,\n        generator=generator,\n    )\n\n\ntrain_loader = build_loader(TRAIN_IDS, augment=True, shuffle=True)\ndev_loader = build_loader(DEV_IDS, augment=False, shuffle=False)\nval_loader = build_loader(VAL_IDS, augment=False, shuffle=False)\n\n# Look at one augmented batch before trusting any of it.\n_batch = next(iter(train_loader))\nprint(\"batch image\", tuple(_batch[\"image\"].shape), \"label\", tuple(_batch[\"label_map\"].shape),\n      \"seam positive fraction\", float(_batch[\"seam\"].mean()))\n\n_figure, _axes = plt.subplots(1, 4, figsize=(13, 3.4))\nfor _axis, _index in zip(_axes, range(min(4, len(_batch[\"image\"])))):\n    _image = (_batch[\"image\"][_index].permute(1, 2, 0).numpy()\n              * np.array(IMAGENET_STD) + np.array(IMAGENET_MEAN)).clip(0, 1)\n    _axis.imshow(overlay_label_map((_image * 255).astype(np.uint8),\n                                   _batch[\"label_map\"][_index].numpy(),\n                                   seam=_batch[\"seam\"][_index].numpy() > 0.5))\n    _axis.axis(\"off\")\n_figure.suptitle(\"augmented batch — teal body, amber sleeve, magenta seam target\", fontsize=10)\n_figure.tight_layout()\n_figure.savefig(OUTPUT_DIR / \"augmentation_check.png\", dpi=110)\nplt.close(_figure)\nprint(f\"wrote {OUTPUT_DIR / 'augmentation_check.png'}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T12:27:31.438750Z","iopub.execute_input":"2026-08-21T12:27:31.439113Z","iopub.status.idle":"2026-08-21T12:27:32.553105Z","shell.execute_reply.started":"2026-08-21T12:27:31.439076Z","shell.execute_reply":"2026-08-21T12:27:32.552037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# CELL 6 — Model: frozen ImageNet encoder + light trainable decoder\n# =============================================================================\n# The brief says frozen pretrained backbone layers do not count toward the 2M cap;\n# only what we actually train does. We use that deliberately and say so: a frozen\n# ResNet-34 (21.3M params, all frozen) gives much stronger features than a tiny\n# from-scratch U-Net would, and the trainable decoder comes in around 0.2M.\n#\n# One detail that matters more than it looks: requires_grad=False stops gradients\n# but does NOT stop BatchNorm running statistics from updating on every forward\n# pass in train() mode. A \"frozen\" backbone left in train() mode silently drifts\n# away from its pretrained statistics, which breaks reproducibility in a way that\n# is very hard to find. So the encoder is forced back into eval() every epoch, and\n# assert_frozen_backbone() checks it.\n\nimport torchvision\n\n\ndef separable_conv(in_channels, out_channels):\n    \"\"\"Depthwise 3x3 + pointwise 1x1, with BN/ReLU. Cheap way to buy receptive field.\"\"\"\n    return nn.Sequential(\n        nn.Conv2d(in_channels, in_channels, 3, padding=1, groups=in_channels, bias=False),\n        nn.BatchNorm2d(in_channels),\n        nn.ReLU(inplace=True),\n        nn.Conv2d(in_channels, out_channels, 1, bias=False),\n        nn.BatchNorm2d(out_channels),\n        nn.ReLU(inplace=True),\n    )\n\n\nclass FrozenResNetEncoder(nn.Module):\n    \"\"\"ImageNet ResNet trunk, frozen, tapped at strides 4, 8, 16 and 32.\"\"\"\n\n    def __init__(self, name=\"resnet34\", pretrained=True):\n        super().__init__()\n        weights = \"IMAGENET1K_V1\" if pretrained else None\n        trunk = getattr(torchvision.models, name)(weights=weights)\n\n        self.stem = nn.Sequential(trunk.conv1, trunk.bn1, trunk.relu, trunk.maxpool)\n        self.layer1, self.layer2 = trunk.layer1, trunk.layer2\n        self.layer3, self.layer4 = trunk.layer3, trunk.layer4\n        self.out_channels = (64, 128, 256, 512) if name in (\"resnet18\", \"resnet34\") else (256, 512, 1024, 2048)\n\n        for parameter in self.parameters():\n            parameter.requires_grad = False\n\n    def train(self, mode=True):\n        \"\"\"Keep the encoder in eval mode even when the model is put in train mode.\n\n        This is what actually freezes BatchNorm's running statistics.\n        \"\"\"\n        super().train(mode)\n        for module in self.modules():\n            if isinstance(module, nn.BatchNorm2d):\n                module.eval()\n        return self\n\n    def forward(self, x):\n        c1 = self.layer1(self.stem(x))   # stride 4\n        c2 = self.layer2(c1)             # stride 8\n        c3 = self.layer3(c2)             # stride 16\n        c4 = self.layer4(c3)             # stride 32\n        return c1, c2, c3, c4\n\n\nclass GarmentSegmenter(nn.Module):\n    \"\"\"Frozen encoder + FPN-lite decoder, a 3-class head and an auxiliary seam head.\n\n    The seam head is a separate 1-channel output rather than a fourth class. A thin\n    band class scores terribly under IoU and destabilises the argmax exactly at the\n    pixels we care most about; a separate head gets the boundary signal without\n    corrupting the three-class map, for a few thousand parameters.\n    \"\"\"\n\n    def __init__(self, config):\n        super().__init__()\n        channels = config[\"decoder_channels\"]\n        self.encoder = FrozenResNetEncoder(config[\"backbone\"], config[\"pretrained\"])\n        self.use_refine_head = config[\"use_refine_head\"]\n        self.use_boundary_head = config[\"use_boundary_head\"]\n\n        c1, c2, c3, c4 = self.encoder.out_channels\n        self.lateral4 = nn.Conv2d(c4, channels, 1)\n        self.lateral3 = nn.Conv2d(c3, channels, 1)\n        self.lateral2 = nn.Conv2d(c2, channels, 1)\n        self.lateral1 = nn.Conv2d(c1, channels, 1)\n        self.smooth3 = separable_conv(channels, channels)\n        self.smooth2 = separable_conv(channels, channels)\n        self.smooth1 = separable_conv(channels, channels)\n\n        if self.use_refine_head:\n            # A trainable look at the raw image at stride 2. The frozen encoder's\n            # finest tap is stride 4, which is too coarse for an armhole seam.\n            self.image_stem = nn.Sequential(\n                nn.Conv2d(3, 16, 3, stride=2, padding=1, bias=False),\n                nn.BatchNorm2d(16), nn.ReLU(inplace=True),\n            )\n            self.fuse = separable_conv(channels + 16, channels)\n\n        self.class_head = nn.Conv2d(channels, NUM_CLASSES, 1)\n        if self.use_boundary_head:\n            self.boundary_head = nn.Conv2d(channels, 1, 1)\n\n    def forward(self, image):\n        \"\"\"Return (class logits BxCxHxW, seam logits Bx1xHxW or None).\"\"\"\n        c1, c2, c3, c4 = self.encoder(image)\n\n        p4 = self.lateral4(c4)\n        p3 = self.smooth3(self.lateral3(c3) + F.interpolate(p4, size=c3.shape[-2:], mode=\"bilinear\", align_corners=False))\n        p2 = self.smooth2(self.lateral2(c2) + F.interpolate(p3, size=c2.shape[-2:], mode=\"bilinear\", align_corners=False))\n        features = self.smooth1(self.lateral1(c1) + F.interpolate(p2, size=c1.shape[-2:], mode=\"bilinear\", align_corners=False))\n\n        if self.use_refine_head:\n            half = F.interpolate(features, scale_factor=2, mode=\"bilinear\", align_corners=False)\n            features = self.fuse(torch.cat([half, self.image_stem(image)], dim=1))\n\n        size = image.shape[-2:]\n        class_logits = F.interpolate(self.class_head(features), size=size, mode=\"bilinear\", align_corners=False)\n        seam_logits = None\n        if self.use_boundary_head:\n            seam_logits = F.interpolate(self.boundary_head(features), size=size, mode=\"bilinear\", align_corners=False)\n        return class_logits, seam_logits\n\n\ndef count_parameters(model):\n    \"\"\"Return (total, trainable). This is the number the brief asks us to report.\"\"\"\n    total = sum(p.numel() for p in model.parameters())\n    trainable = sum(p.numel() for p in model.parameters() if p.requires_grad)\n    return total, trainable\n\n\ndef assert_frozen_backbone(model):\n    \"\"\"Verify the encoder is frozen in both senses: no grads, and BN in eval mode.\"\"\"\n    model.train()\n    leaked = [n for n, p in model.encoder.named_parameters() if p.requires_grad]\n    assert not leaked, f\"encoder parameters still trainable: {leaked[:5]}\"\n    training_bn = [\n        n for n, m in model.encoder.named_modules()\n        if isinstance(m, nn.BatchNorm2d) and m.training\n    ]\n    assert not training_bn, f\"encoder BatchNorm still updating running stats: {training_bn[:5]}\"\n\n\nmodel = GarmentSegmenter(CONFIG).to(DEVICE)\nassert_frozen_backbone(model)\n\nTOTAL_PARAMS, TRAINABLE_PARAMS = count_parameters(model)\nassert TRAINABLE_PARAMS < CONFIG[\"trainable_parameter_cap\"], (\n    f\"{TRAINABLE_PARAMS:,} trainable parameters exceeds the \"\n    f\"{CONFIG['trainable_parameter_cap']:,} cap\"\n)\n\nprint(f\"backbone           : {CONFIG['backbone']} (frozen, ImageNet)\")\nprint(f\"total parameters   : {TOTAL_PARAMS:,}\")\nprint(f\"TRAINABLE          : {TRAINABLE_PARAMS:,}   (cap {CONFIG['trainable_parameter_cap']:,})\")\nprint(f\"frozen (not counted): {TOTAL_PARAMS - TRAINABLE_PARAMS:,}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T12:27:32.555031Z","iopub.execute_input":"2026-08-21T12:27:32.555335Z","iopub.status.idle":"2026-08-21T12:27:32.984941Z","shell.execute_reply.started":"2026-08-21T12:27:32.555301Z","shell.execute_reply":"2026-08-21T12:27:32.984097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# CELL 7 — Loss, metrics, training loop, curves\n# =============================================================================\n\ndef dice_loss(logits, targets, ignore_background=True, epsilon=1.0):\n    \"\"\"Soft Dice over the softmax probabilities.\n\n    Paired with cross-entropy because CE optimises per-pixel likelihood while Dice\n    optimises overlap directly, which is what IoU measures. Background is excluded\n    by default: it is ~80% of pixels and its Dice term is near 1.0 from epoch one,\n    so including it just dilutes the gradient from the classes we care about.\n    \"\"\"\n    probabilities = F.softmax(logits, dim=1)\n    one_hot = F.one_hot(targets, NUM_CLASSES).permute(0, 3, 1, 2).float()\n\n    start = 1 if ignore_background else 0\n    probabilities, one_hot = probabilities[:, start:], one_hot[:, start:]\n\n    intersection = (probabilities * one_hot).sum(dim=(0, 2, 3))\n    denominator = probabilities.sum(dim=(0, 2, 3)) + one_hot.sum(dim=(0, 2, 3))\n    return 1.0 - ((2 * intersection + epsilon) / (denominator + epsilon)).mean()\n\n\ndef compute_loss(class_logits, seam_logits, label_map, seam_target):\n    \"\"\"Cross-entropy + Dice, plus BCE on the auxiliary seam head.\"\"\"\n    loss = F.cross_entropy(class_logits, label_map)\n    loss = loss + CONFIG[\"dice_weight\"] * dice_loss(class_logits, label_map)\n\n    if seam_logits is not None:\n        # The seam band is a few percent of pixels, so BCE needs a positive weight\n        # or the head learns to predict \"no seam\" everywhere and stops there.\n        pos_weight = torch.tensor(CONFIG[\"boundary_pos_weight\"], device=class_logits.device)\n        seam_loss = F.binary_cross_entropy_with_logits(\n            seam_logits.squeeze(1), seam_target, pos_weight=pos_weight\n        )\n        loss = loss + CONFIG[\"boundary_weight\"] * seam_loss\n    return loss\n\n\n# ---------------------------------------------------------------------------\n# Metrics\n# ---------------------------------------------------------------------------\n# Per-class IoU has two common definitions and they give different numbers, so we\n# report BOTH and label them, rather than quoting one and hoping it matches the\n# company's supplied metric definition:\n#\n#   dataset-level : sum intersections and unions over the whole split, then divide.\n#                   Dominated by images where the class is large.\n#   image-mean    : IoU per image, then average. Weights every photo equally and is\n#                   much harsher when a class is nearly absent from an image.\n#\n# Not COCO mAP -- this is semantic segmentation, not instance detection.\n\ndef confusion_matrix(predicted, target, num_classes=NUM_CLASSES):\n    \"\"\"Accumulate a num_classes x num_classes confusion matrix for one batch.\"\"\"\n    valid = (target >= 0) & (target < num_classes)\n    indices = num_classes * target[valid].astype(np.int64) + predicted[valid].astype(np.int64)\n    return np.bincount(indices, minlength=num_classes ** 2).reshape(num_classes, num_classes)\n\n\ndef iou_from_confusion(matrix):\n    \"\"\"Per-class IoU from a confusion matrix: TP / (TP + FP + FN).\"\"\"\n    true_positive = np.diag(matrix).astype(np.float64)\n    union = matrix.sum(axis=1) + matrix.sum(axis=0) - true_positive\n    with np.errstate(divide=\"ignore\", invalid=\"ignore\"):\n        return np.where(union > 0, true_positive / union, np.nan)\n\n\ndef per_image_iou(predicted, target, num_classes=NUM_CLASSES):\n    \"\"\"Per-class IoU for a single image; NaN where the class is absent from both.\"\"\"\n    scores = np.full(num_classes, np.nan)\n    for class_id in range(num_classes):\n        prediction_mask = predicted == class_id\n        target_mask = target == class_id\n        union = np.logical_or(prediction_mask, target_mask).sum()\n        if union:\n            scores[class_id] = np.logical_and(prediction_mask, target_mask).sum() / union\n    return scores\n\n\ndef boundary_f1(predicted_boundary, target_boundary, tolerance):\n    \"\"\"F1 between two thin boundary maps, with a pixel tolerance.\n\n    A raw pixel-wise score on a 3px band is dominated by one-pixel misalignment and\n    tells you nothing. Tolerance-based matching is the standard fix: a predicted\n    boundary pixel counts as correct if a true boundary pixel is within `tolerance`.\n    \"\"\"\n    if not predicted_boundary.any() and not target_boundary.any():\n        return np.nan\n    if not predicted_boundary.any() or not target_boundary.any():\n        return 0.0\n\n    distance_to_target = ndimage.distance_transform_edt(~target_boundary)\n    distance_to_prediction = ndimage.distance_transform_edt(~predicted_boundary)\n\n    precision = (distance_to_target[predicted_boundary] <= tolerance).mean()\n    recall = (distance_to_prediction[target_boundary] <= tolerance).mean()\n    if precision + recall == 0:\n        return 0.0\n    return float(2 * precision * recall / (precision + recall))\n\n\n@torch.no_grad()\ndef evaluate(model, loader, exclude_ids=None):\n    \"\"\"Run the model over a split and return every metric we report.\n\n    `exclude_ids` drops specific images -- used in cell 8 to re-score validation\n    with the leakage-contaminated images removed.\n    \"\"\"\n    model.eval()\n    matrix = np.zeros((NUM_CLASSES, NUM_CLASSES), dtype=np.int64)\n    image_scores, boundary_scores, losses = [], [], []\n    excluded = set(exclude_ids or [])\n    image_ids = loader.dataset.image_ids\n    position = 0\n\n    for batch in loader:\n        image = batch[\"image\"].to(DEVICE, non_blocking=True)\n        label_map = batch[\"label_map\"].to(DEVICE, non_blocking=True)\n        seam_target = batch[\"seam\"].to(DEVICE, non_blocking=True)\n\n        class_logits, seam_logits = model(image)\n        losses.append(float(compute_loss(class_logits, seam_logits, label_map, seam_target)))\n\n        predicted = class_logits.argmax(dim=1).cpu().numpy()\n        truth = label_map.cpu().numpy()\n        seam_probability = torch.sigmoid(seam_logits).cpu().numpy() if seam_logits is not None else None\n\n        for index in range(len(predicted)):\n            image_id = image_ids[position]\n            position += 1\n            if image_id in excluded:\n                continue\n\n            matrix += confusion_matrix(predicted[index], truth[index])\n            image_scores.append(per_image_iou(predicted[index], truth[index]))\n            if seam_probability is not None:\n                boundary_scores.append(\n                    boundary_f1(\n                        seam_probability[index, 0] >= CONFIG[\"place_boundary_threshold\"],\n                        batch[\"seam\"][index].numpy() > 0.5,\n                        CONFIG[\"boundary_tolerance_px\"],\n                    )\n                )\n\n    dataset_iou = iou_from_confusion(matrix)\n    stacked = np.stack(image_scores) if image_scores else np.zeros((0, NUM_CLASSES))\n    with np.errstate(invalid=\"ignore\"):\n        image_mean_iou = np.nanmean(stacked, axis=0) if len(stacked) else np.full(NUM_CLASSES, np.nan)\n\n    return {\n        \"loss\": float(np.mean(losses)) if losses else float(\"nan\"),\n        \"iou_dataset\": dataset_iou,\n        \"iou_image_mean\": image_mean_iou,\n        \"miou_dataset\": float(np.nanmean(dataset_iou)),\n        \"miou_image_mean\": float(np.nanmean(image_mean_iou)),\n        \"boundary_f1\": float(np.nanmean(boundary_scores)) if boundary_scores else float(\"nan\"),\n        \"pixel_fraction\": matrix.sum(axis=1) / max(matrix.sum(), 1),\n        \"images_scored\": len(image_scores),\n    }\n\n\n# ---------------------------------------------------------------------------\n# Training loop\n# ---------------------------------------------------------------------------\n\ndef build_optimiser(model):\n    \"\"\"AdamW over the trainable decoder only, plus cosine schedule with warmup.\"\"\"\n    trainable = [p for p in model.parameters() if p.requires_grad]\n    optimiser = torch.optim.AdamW(\n        trainable, lr=CONFIG[\"learning_rate\"], weight_decay=CONFIG[\"weight_decay\"]\n    )\n\n    def learning_rate_scale(epoch):\n        # Linear warmup then cosine decay. Warmup matters here because the decoder\n        # is randomly initialised while the features feeding it are already good.\n        if epoch < CONFIG[\"warmup_epochs\"]:\n            return (epoch + 1) / max(CONFIG[\"warmup_epochs\"], 1)\n        progress = (epoch - CONFIG[\"warmup_epochs\"]) / max(CONFIG[\"epochs\"] - CONFIG[\"warmup_epochs\"], 1)\n        return 0.5 * (1 + math.cos(math.pi * progress))\n\n    scheduler = torch.optim.lr_scheduler.LambdaLR(optimiser, learning_rate_scale)\n    return optimiser, scheduler\n\n\ndef train_one_epoch(model, loader, optimiser, scaler):\n    \"\"\"One pass over the training split; returns mean loss.\"\"\"\n    model.train()          # the encoder overrides this back to eval() -- see cell 6\n    losses = []\n    for batch in loader:\n        image = batch[\"image\"].to(DEVICE, non_blocking=True)\n        label_map = batch[\"label_map\"].to(DEVICE, non_blocking=True)\n        seam_target = batch[\"seam\"].to(DEVICE, non_blocking=True)\n\n        optimiser.zero_grad(set_to_none=True)\n        with torch.autocast(device_type=DEVICE.type, enabled=CONFIG[\"amp\"] and DEVICE.type == \"cuda\"):\n            class_logits, seam_logits = model(image)\n            loss = compute_loss(class_logits, seam_logits, label_map, seam_target)\n\n        scaler.scale(loss).backward()\n        scaler.step(optimiser)\n        scaler.update()\n        losses.append(float(loss))\n    return float(np.mean(losses))\n\n\nCHECKPOINT_PATH = OUTPUT_DIR / \"garment_segmenter.pt\"\nHISTORY_PATH = OUTPUT_DIR / \"training_log.csv\"\n\noptimiser, scheduler = build_optimiser(model)\nscaler = torch.amp.GradScaler(enabled=CONFIG[\"amp\"] and DEVICE.type == \"cuda\")\n\nhistory, best_dev_miou, epochs_without_improvement = [], -1.0, 0\ntraining_started = time.time()\n\nfor epoch in range(CONFIG[\"epochs\"]):\n    train_loss = train_one_epoch(model, train_loader, optimiser, scaler)\n    scheduler.step()\n\n    # Model selection happens on DEV, never on val. Val is scored once, in cell 8.\n    dev_metrics = evaluate(model, dev_loader)\n\n    history.append({\n        \"epoch\": epoch,\n        \"learning_rate\": optimiser.param_groups[0][\"lr\"],\n        \"train_loss\": train_loss,\n        \"dev_loss\": dev_metrics[\"loss\"],\n        \"dev_miou\": dev_metrics[\"miou_dataset\"],\n        \"dev_iou_background\": dev_metrics[\"iou_dataset\"][BACKGROUND],\n        \"dev_iou_body\": dev_metrics[\"iou_dataset\"][BODY],\n        \"dev_iou_sleeve\": dev_metrics[\"iou_dataset\"][SLEEVE],\n        \"dev_boundary_f1\": dev_metrics[\"boundary_f1\"],\n    })\n    pd.DataFrame(history).to_csv(HISTORY_PATH, index=False)\n\n    marker = \"\"\n    if dev_metrics[\"miou_dataset\"] > best_dev_miou:\n        best_dev_miou = dev_metrics[\"miou_dataset\"]\n        epochs_without_improvement = 0\n        marker = \"  <- best\"\n        torch.save({\n            \"state_dict\": model.state_dict(),\n            \"config\": CONFIG,\n            \"config_hash\": CONFIG_HASH,\n            \"epoch\": epoch,\n            \"dev_metrics\": {k: (v.tolist() if isinstance(v, np.ndarray) else v)\n                            for k, v in dev_metrics.items()},\n            \"trainable_parameters\": TRAINABLE_PARAMS,\n            \"total_parameters\": TOTAL_PARAMS,\n            \"split_source\": SPLIT_SOURCE,\n            \"manifest_sha256\": MANIFEST_SHA,\n            \"rle_order\": RLE_ORDER,\n            \"torch_version\": torch.__version__,\n        }, CHECKPOINT_PATH)\n    else:\n        epochs_without_improvement += 1\n\n    print(f\"epoch {epoch:02d}  train {train_loss:.4f}  dev {dev_metrics['loss']:.4f}  \"\n          f\"dev mIoU {dev_metrics['miou_dataset']:.4f}  \"\n          f\"body {dev_metrics['iou_dataset'][BODY]:.3f}  \"\n          f\"sleeve {dev_metrics['iou_dataset'][SLEEVE]:.3f}  \"\n          f\"bF1 {dev_metrics['boundary_f1']:.3f}{marker}\", flush=True)\n\n    if epochs_without_improvement >= CONFIG[\"early_stop_patience\"]:\n        print(f\"early stop: dev mIoU has not improved for {CONFIG['early_stop_patience']} epochs\")\n        break\n\nprint(f\"training took {(time.time() - training_started) / 60:.1f} min; \"\n      f\"best dev mIoU {best_dev_miou:.4f}; checkpoint -> {CHECKPOINT_PATH}\")\n\n\ndef plot_curves(history_frame, filename=\"training_curves.png\"):\n    \"\"\"Loss and metric curves. The train/dev gap is the thing to actually look at.\"\"\"\n    figure, axes = plt.subplots(1, 3, figsize=(15, 4))\n\n    axes[0].plot(history_frame.epoch, history_frame.train_loss, label=\"train\")\n    axes[0].plot(history_frame.epoch, history_frame.dev_loss, label=\"dev\")\n    axes[0].set_title(\"loss\"); axes[0].set_xlabel(\"epoch\"); axes[0].legend()\n\n    for column, label in ((\"dev_iou_body\", \"body\"), (\"dev_iou_sleeve\", \"sleeve\"), (\"dev_miou\", \"mean\")):\n        axes[1].plot(history_frame.epoch, history_frame[column], label=label)\n    axes[1].set_title(\"dev IoU (dataset-level)\"); axes[1].set_xlabel(\"epoch\"); axes[1].legend()\n\n    axes[2].plot(history_frame.epoch, history_frame.dev_boundary_f1, color=\"tab:purple\")\n    axes[2].set_title(f\"dev boundary F1 @{CONFIG['boundary_tolerance_px']}px\"); axes[2].set_xlabel(\"epoch\")\n\n    figure.tight_layout()\n    figure.savefig(OUTPUT_DIR / filename, dpi=120)\n    plt.close(figure)\n    print(f\"wrote {OUTPUT_DIR / filename}\")\n\n\nplot_curves(pd.DataFrame(history))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T12:27:32.986258Z","iopub.execute_input":"2026-08-21T12:27:32.986604Z","iopub.status.idle":"2026-08-21T12:34:56.105743Z","shell.execute_reply.started":"2026-08-21T12:27:32.986560Z","shell.execute_reply":"2026-08-21T12:34:56.104907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# CELL 8 — Validation, scored once, plus error analysis\n# =============================================================================\n# The validation split is evaluated here, once, using the checkpoint selected on\n# dev. Nothing above this line ever looked at val, which is what keeps it a fair\n# proxy for the company's held-out test split.\n\ncheckpoint = torch.load(CHECKPOINT_PATH, map_location=DEVICE, weights_only=False)\nmodel.load_state_dict(checkpoint[\"state_dict\"])\nmodel.eval()\nprint(f\"loaded checkpoint from epoch {checkpoint['epoch']} \"\n      f\"({checkpoint['trainable_parameters']:,} trainable parameters)\")\n\nval_metrics = evaluate(model, val_loader)\n\n# Scored a second time with the leakage-contaminated val images removed. The gap\n# between the two is an honest estimate of how optimistic the headline number is.\nval_metrics_clean = (\n    evaluate(model, val_loader, exclude_ids=CONTAMINATED_VAL_IDS)\n    if CONTAMINATED_VAL_IDS else None\n)\n\n\ndef results_table(metrics, clean_metrics=None):\n    \"\"\"Build the per-class results table, both IoU definitions side by side.\"\"\"\n    rows = []\n    for class_id, name in enumerate(CONFIG[\"class_names\"]):\n        row = {\n            \"class\": name,\n            \"pixel_fraction\": round(float(metrics[\"pixel_fraction\"][class_id]), 4),\n            \"IoU_dataset_level\": round(float(metrics[\"iou_dataset\"][class_id]), 4),\n            \"IoU_image_mean\": round(float(metrics[\"iou_image_mean\"][class_id]), 4),\n        }\n        if clean_metrics is not None:\n            row[\"IoU_dataset_level_deduped\"] = round(float(clean_metrics[\"iou_dataset\"][class_id]), 4)\n        rows.append(row)\n\n    summary = {\n        \"class\": \"MEAN\",\n        \"pixel_fraction\": 1.0,\n        \"IoU_dataset_level\": round(metrics[\"miou_dataset\"], 4),\n        \"IoU_image_mean\": round(metrics[\"miou_image_mean\"], 4),\n    }\n    if clean_metrics is not None:\n        summary[\"IoU_dataset_level_deduped\"] = round(clean_metrics[\"miou_dataset\"], 4)\n    rows.append(summary)\n    return pd.DataFrame(rows)\n\n\nVAL_RESULTS = results_table(val_metrics, val_metrics_clean)\nVAL_RESULTS.to_csv(OUTPUT_DIR / \"val_per_class_iou.csv\", index=False)\n\nprint(\"\\nVALIDATION — per-class IoU\")\nprint(\"  dataset-level = sum intersections and unions over the split, then divide\")\nprint(\"  image-mean    = IoU per image, then average (harsher when a class is small)\")\nprint(VAL_RESULTS.to_string(index=False))\nprint(f\"\\nboundary F1 @{CONFIG['boundary_tolerance_px']}px : {val_metrics['boundary_f1']:.4f}\")\nprint(f\"images scored              : {val_metrics['images_scored']}\")\nif val_metrics_clean is not None:\n    drop = val_metrics[\"miou_dataset\"] - val_metrics_clean[\"miou_dataset\"]\n    print(f\"mIoU excluding {len(CONTAMINATED_VAL_IDS)} near-duplicate val images: \"\n          f\"{val_metrics_clean['miou_dataset']:.4f}  (delta {drop:+.4f})\")\n\n\n# ---------------------------------------------------------------------------\n# Error analysis: find the worst cases rather than cherry-picking pretty ones\n# ---------------------------------------------------------------------------\n\n@torch.no_grad()\ndef predict_single(model, image_id):\n    \"\"\"Predict one cached image; returns (image, truth, predicted, seam probability).\"\"\"\n    image, label_map, trim_mask = load_cached(image_id)\n    tensor = normalise_image(image).unsqueeze(0).to(DEVICE)\n    class_logits, seam_logits = model(tensor)\n    predicted = class_logits.argmax(dim=1)[0].cpu().numpy().astype(np.uint8)\n    seam_probability = (\n        torch.sigmoid(seam_logits)[0, 0].cpu().numpy() if seam_logits is not None else None\n    )\n    return image, label_map, predicted, seam_probability\n\n\ndef rank_val_images_by_error(model, image_ids):\n    \"\"\"Rank validation images by mean IoU, worst first.\n\n    This is where the Q5 failure case comes from. Picking the worst by a number,\n    then looking at them, is a different exercise from browsing for something that\n    looks bad -- and it surfaces label noise, which browsing does not.\n    \"\"\"\n    rows = []\n    for image_id in image_ids:\n        _, truth, predicted, _ = predict_single(model, image_id)\n        scores = per_image_iou(predicted, truth)\n        rows.append({\n            \"image_id\": image_id,\n            \"mean_iou\": float(np.nanmean(scores)),\n            \"iou_body\": float(scores[BODY]),\n            \"iou_sleeve\": float(scores[SLEEVE]),\n            \"body_pixel_fraction\": float((truth == BODY).mean()),\n        })\n    return pd.DataFrame(rows).sort_values(\"mean_iou\").reset_index(drop=True)\n\n\nERROR_RANKING = rank_val_images_by_error(model, VAL_IDS)\nERROR_RANKING.to_csv(OUTPUT_DIR / \"val_error_ranking.csv\", index=False)\nprint(\"\\nworst validation images by mean IoU:\")\nprint(ERROR_RANKING.head(8).to_string(index=False))\n\n\ndef save_qualitative_panel(model, image_ids, filename, title):\n    \"\"\"Save ground-truth vs prediction pairs for a set of images.\"\"\"\n    rows = len(image_ids)\n    figure, axes = plt.subplots(rows, 3, figsize=(10.5, 3.4 * rows))\n    axes = np.atleast_2d(axes)\n\n    for row, image_id in enumerate(image_ids):\n        image, truth, predicted, seam_probability = predict_single(model, image_id)\n        seam = seam_probability >= CONFIG[\"place_boundary_threshold\"] if seam_probability is not None else None\n\n        axes[row, 0].imshow(image)\n        axes[row, 0].set_title(f\"{str(image_id)[:12]}\", fontsize=9)\n        axes[row, 1].imshow(overlay_label_map(image, truth))\n        axes[row, 1].set_title(\"ground truth\", fontsize=9)\n        axes[row, 2].imshow(overlay_label_map(image, predicted, seam=seam))\n        axes[row, 2].set_title(\n            f\"predicted  (mIoU {np.nanmean(per_image_iou(predicted, truth)):.2f})\", fontsize=9\n        )\n\n    for axis in np.ravel(axes):\n        axis.axis(\"off\")\n    figure.suptitle(title, fontsize=11)\n    figure.tight_layout()\n    figure.savefig(OUTPUT_DIR / filename, dpi=110)\n    plt.close(figure)\n    print(f\"wrote {OUTPUT_DIR / filename}\")\n\n\nsample_count = min(CONFIG[\"qualitative_samples\"], len(ERROR_RANKING))\nsave_qualitative_panel(\n    model, ERROR_RANKING.image_id.head(sample_count // 2).tolist(),\n    \"qualitative_worst.png\", \"worst validation cases — where the model breaks\",\n)\nsave_qualitative_panel(\n    model, ERROR_RANKING.image_id.tail(sample_count // 2).tolist(),\n    \"qualitative_best.png\", \"best validation cases\",\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T12:34:56.107251Z","iopub.execute_input":"2026-08-21T12:34:56.107581Z","iopub.status.idle":"2026-08-21T12:35:04.241161Z","shell.execute_reply.started":"2026-08-21T12:34:56.107549Z","shell.execute_reply":"2026-08-21T12:35:04.240051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# CELL 9 — Placement geometry: seam band, garment frame, instruction -> region\n# =============================================================================\n# This is the fix for the defect described in the brief. The existing system uses\n# fixed pixel offsets, so \"left chest\" is the same coordinates every time and drifts\n# whenever the garment is tilted, cropped or framed differently. Everything below\n# is computed in the garment's OWN coordinate system, derived from the predicted\n# mask, so a band follows the garment instead of the image frame.\n\nfrom collections import namedtuple\n\nGarmentFrame = namedtuple(\"GarmentFrame\", [\n    \"centroid_yx\", \"u_axis\", \"v_axis\",\n    \"u_min\", \"u_max\", \"v_min\", \"v_max\",\n    \"axis_ratio\", \"used_image_axes\", \"orientation_source\",\n])\n\n\ndef seam_band_from_prediction(body, sleeve, band_width, seam_probability=None, threshold=0.5):\n    \"\"\"The band that artwork must not cross.\n\n    UNION of a geometric interface (dilated body meeting dilated sleeve) and the\n    learned boundary head, not the intersection. A false seam pixel costs \"logo sits\n    3mm further in\"; a missed seam pixel costs \"print runs over a seam\", which is\n    the failure the brief names explicitly. The learned head also recovers the\n    neckline, which has no class of its own.\n    \"\"\"\n    radius = max(1, band_width // 2)\n    geometric = np.zeros_like(body, dtype=bool)\n    if body.any() and sleeve.any():\n        geometric = dilate(body, radius) & dilate(sleeve, radius)\n    if seam_probability is None:\n        return geometric\n    return geometric | (seam_probability >= threshold)\n\n\ndef build_garment_frame(body_mask, sleeve_mask):\n    \"\"\"Derive the garment's own coordinate system from its predicted torso mask.\n\n    u runs across the garment (0 at viewer-left, 1 at viewer-right), v runs down it\n    (0 at the neck, 1 at the hem). Both come from a PCA of the body pixels, which is\n    what makes a band follow a tilted garment.\n\n    Returns None when there is no usable torso mask.\n    \"\"\"\n    body = largest_component(body_mask)\n    if body.sum() < 50:\n        return None\n\n    ys, xs = np.nonzero(body)\n    points = np.stack([ys, xs]).astype(np.float64)          # 2 x N, in (y, x)\n    centroid = points.mean(axis=1)\n    centred = points - centroid[:, None]\n\n    eigenvalues, eigenvectors = np.linalg.eigh(np.cov(centred))\n    v_axis = eigenvectors[:, 1]     # largest eigenvalue -> the garment's long axis\n    u_axis = eigenvectors[:, 0]\n    axis_ratio = float(np.sqrt(max(eigenvalues[1], 1e-9) / max(eigenvalues[0], 1e-9)))\n\n    # A wide flat-lay is close to square, so its principal axis is ill-defined and\n    # would swing wildly between near-identical photos. Fall back to image axes and\n    # let the confidence score record that we did.\n    used_image_axes = axis_ratio < CONFIG[\"place_min_axis_ratio\"]\n    if used_image_axes:\n        v_axis = np.array([1.0, 0.0])   # down the image\n        u_axis = np.array([0.0, 1.0])   # right across the image\n\n    # Fix u to point toward viewer-right, so u = 1 is always the right of the frame.\n    if u_axis[1] < 0:\n        u_axis = -u_axis\n\n    # Orient v from neck to hem. Sleeves attach at the shoulders, so the sleeve\n    # centroid should project to the NECK side of the body centroid.\n    orientation_source = \"image_up\"\n    sleeve = largest_component(sleeve_mask)\n    if sleeve.any():\n        sleeve_ys, sleeve_xs = np.nonzero(sleeve)\n        sleeve_centroid = np.array([sleeve_ys.mean(), sleeve_xs.mean()])\n        if float(np.dot(sleeve_centroid - centroid, v_axis)) > 0:\n            v_axis = -v_axis\n        orientation_source = \"sleeve\"\n    elif v_axis[0] < 0:\n        # No sleeve to orient against: assume the garment hangs downward in frame.\n        v_axis = -v_axis\n\n    u_projection = u_axis @ centred\n    v_projection = v_axis @ centred\n\n    return GarmentFrame(\n        centroid_yx=(float(centroid[0]), float(centroid[1])),\n        u_axis=u_axis, v_axis=v_axis,\n        u_min=float(u_projection.min()), u_max=float(u_projection.max()),\n        v_min=float(v_projection.min()), v_max=float(v_projection.max()),\n        axis_ratio=axis_ratio,\n        used_image_axes=used_image_axes,\n        orientation_source=orientation_source,\n    )\n\n\ndef garment_coordinates(shape, frame):\n    \"\"\"Return (u, v) arrays over the whole image, each normalised to [0, 1].\n\n    Values outside [0, 1] simply mean \"beyond the garment's extent\", which is fine:\n    the band masks intersect with the body mask anyway.\n    \"\"\"\n    ys, xs = np.mgrid[0 : shape[0], 0 : shape[1]]\n    dy = ys - frame.centroid_yx[0]\n    dx = xs - frame.centroid_yx[1]\n\n    u_projection = frame.u_axis[0] * dy + frame.u_axis[1] * dx\n    v_projection = frame.v_axis[0] * dy + frame.v_axis[1] * dx\n\n    u = (u_projection - frame.u_min) / max(frame.u_max - frame.u_min, 1e-6)\n    v = (v_projection - frame.v_min) / max(frame.v_max - frame.v_min, 1e-6)\n    return u, v\n\n\ndef parse_instruction(text):\n    \"\"\"Parse \"front, left chest\" into (side, horizontal, vertical).\n\n    `horizontal` is WEARER-relative for left/right. Deliberately a small vocabulary\n    with a hard failure on anything unrecognised: silently defaulting an\n    unparseable instruction to \"centre chest\" would produce a confident, wrong\n    preview, which is worse than an error.\n    \"\"\"\n    tokens = set(text.lower().replace(\",\", \" \").replace(\"-\", \" \").split())\n\n    side = \"back\" if \"back\" in tokens else \"front\"\n\n    if tokens & {\"centre\", \"center\", \"centred\", \"centered\", \"middle\"}:\n        horizontal = \"centre\"\n    elif \"left\" in tokens:\n        horizontal = \"left\"\n    elif \"right\" in tokens:\n        horizontal = \"right\"\n    else:\n        horizontal = \"centre\"\n\n    if \"chest\" in tokens:\n        vertical = \"chest\"\n    elif tokens & {\"hem\", \"bottom\", \"waist\"}:\n        vertical = \"hem\"\n    elif tokens & {\"upper\", \"yoke\", \"shoulder\"}:\n        vertical = \"upper\"\n    else:\n        # \"back, centred\" and similar carry no vertical token; a centre-back print\n        # sits mid-torso.\n        vertical = \"centre\"\n\n    assert tokens & {\"front\", \"back\", \"left\", \"right\", \"centre\", \"center\", \"centred\",\n                     \"centered\", \"chest\", \"hem\", \"upper\", \"middle\", \"waist\", \"bottom\",\n                     \"yoke\", \"shoulder\"}, f\"unrecognised placement instruction: {text!r}\"\n\n    return side, horizontal, vertical\n\n\ndef resolve_band(side, horizontal, vertical):\n    \"\"\"Turn a parsed instruction into (u_range, v_range) in garment coordinates.\n\n    The left/right convention is the landmine here. In apparel, \"left chest\" means\n    the WEARER's left, which on a front-facing photo appears on the RIGHT of the\n    image; on a back view, wearer-left and viewer-left coincide. If the supplied\n    evaluation_cases.csv turns out to be viewer-relative, flip the single config\n    flag `wearer_left_is_viewer_right` and nothing else changes.\n    \"\"\"\n    v_range = CONFIG[\"band_vertical\"][vertical]\n\n    if horizontal == \"centre\":\n        return CONFIG[\"band_horizontal\"][\"centre\"], v_range\n\n    if not CONFIG[\"wearer_left_is_viewer_right\"]:\n        viewer_right = horizontal == \"right\"          # instruction is viewer-relative\n    elif side == \"front\":\n        viewer_right = horizontal == \"left\"           # wearer's left faces us on the right\n    else:\n        viewer_right = horizontal == \"right\"          # back view: sides coincide\n\n    u_range = CONFIG[\"band_horizontal\"][\"viewer_right\" if viewer_right else \"viewer_left\"]\n    return u_range, v_range\n\n\ndef band_mask(shape, frame, u_range, v_range):\n    \"\"\"Boolean mask of one garment-coordinate band over the whole image.\"\"\"\n    u, v = garment_coordinates(shape, frame)\n    return (u >= u_range[0]) & (u <= u_range[1]) & (v >= v_range[0]) & (v <= v_range[1])\n\n\ndef build_safe_mask(body, sleeve, seam):\n    \"\"\"Where artwork is allowed to land at all.\n\n    Torso, minus sleeve, minus the seam band, restricted to the largest blob, then\n    eroded by a safety margin. Containment is checked against THIS, not the raw body\n    mask -- which is what turns \"must not cross a seam\" from an intention into a\n    checked postcondition.\n    \"\"\"\n    safe = largest_component(body) & ~sleeve & ~seam\n    safe = largest_component(safe)\n    return erode(safe, CONFIG[\"place_safety_erosion_px\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T12:35:04.243719Z","iopub.execute_input":"2026-08-21T12:35:04.244080Z","iopub.status.idle":"2026-08-21T12:35:04.265358Z","shell.execute_reply.started":"2026-08-21T12:35:04.244048Z","shell.execute_reply":"2026-08-21T12:35:04.264332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# CELL 10 — place_artwork(): prediction -> confidence -> deterministic composite\n# =============================================================================\n# The geometry in here is pure NumPy and contains no randomness: same inputs, same\n# output, every run. Artwork is only ever scaled uniformly -- never rotated,\n# mirrored, stretched or recoloured.\n\n# A refusal names the signal that vetoed it. These are the codes a production\n# dashboard would group by.\nREFUSAL_CODE_BY_LIMITING_FACTOR = {\n    \"region_area\": REGION_TOO_SMALL,\n    \"mask_confidence\": LOW_CONFIDENCE,\n    \"mask_cohesion\": MASK_FRAGMENTED,\n    \"fit_headroom\": ARTWORK_DOES_NOT_FIT,\n    \"axis_stability\": AXIS_UNSTABLE,\n}\n\n\n@torch.no_grad()\ndef predict_regions(model, pil_image):\n    \"\"\"Predict body / sleeve / seam masks at the image's NATIVE resolution.\n\n    The network runs at CONFIG[\"image_size\"], but the class probabilities are\n    upsampled before the argmax rather than after. Upsampling an argmax map would\n    quantise the boundary to the network's stride, which is the one place we cannot\n    afford to be sloppy.\n    \"\"\"\n    model.eval()\n    width, height = pil_image.size\n\n    resized = pil_image.convert(\"RGB\").resize(\n        (CONFIG[\"image_size\"], CONFIG[\"image_size\"]), resample=Image.BILINEAR\n    )\n    tensor = normalise_image(np.array(resized, dtype=np.uint8)).unsqueeze(0).to(DEVICE)\n\n    class_logits, seam_logits = model(tensor)\n    probabilities = F.interpolate(\n        F.softmax(class_logits, dim=1), size=(height, width), mode=\"bilinear\", align_corners=False\n    )[0].cpu().numpy()\n\n    predicted = probabilities.argmax(axis=0).astype(np.uint8)\n    class_confidence = probabilities.max(axis=0)\n\n    seam_probability = None\n    if seam_logits is not None:\n        seam_probability = F.interpolate(\n            torch.sigmoid(seam_logits), size=(height, width), mode=\"bilinear\", align_corners=False\n        )[0, 0].cpu().numpy()\n\n    body = predicted == BODY\n    sleeve = predicted == SLEEVE\n    seam = seam_band_from_prediction(\n        body, sleeve, CONFIG[\"place_seam_band_width\"],\n        seam_probability, CONFIG[\"place_boundary_threshold\"],\n    )\n    return {\"body\": body, \"sleeve\": sleeve, \"seam\": seam, \"confidence\": class_confidence}\n\n\ndef resize_artwork(artwork_rgba, width, height):\n    \"\"\"Uniformly scale an RGBA artwork, resampling in PREMULTIPLIED alpha.\n\n    Resampling straight (non-premultiplied) RGBA pulls the colour of fully\n    transparent pixels -- usually black -- into the logo's edges and produces a dark\n    halo. That is a visible defect and it is a form of recolouring, which the brief\n    forbids. Premultiplying first, then dividing the alpha back out, avoids it.\n    \"\"\"\n    source = np.asarray(artwork_rgba.convert(\"RGBA\"), dtype=np.float32)\n    alpha = source[..., 3:4] / 255.0\n\n    premultiplied = np.concatenate([source[..., :3] * alpha, source[..., 3:4]], axis=-1)\n    resized = np.asarray(\n        Image.fromarray(premultiplied.astype(np.uint8), \"RGBA\").resize(\n            (width, height), resample=Image.LANCZOS\n        ),\n        dtype=np.float32,\n    )\n\n    resized_alpha = resized[..., 3:4] / 255.0\n    rgb = np.where(resized_alpha > 0, resized[..., :3] / np.maximum(resized_alpha, 1e-6), 0.0)\n    return np.clip(rgb, 0, 255), np.clip(resized_alpha[..., 0], 0, 1)\n\n\ndef alpha_composite(base_rgb, artwork_rgb, artwork_alpha, top_left):\n    \"\"\"Alpha-over the artwork onto a copy of the base image.\n\n    Where alpha == 1 the output pixel equals the artwork pixel exactly, which\n    assert_artwork_unaltered() checks.\n    \"\"\"\n    x0, y0 = top_left\n    height, width = artwork_alpha.shape\n    canvas = base_rgb.astype(np.float32).copy()\n\n    region = canvas[y0 : y0 + height, x0 : x0 + width]\n    alpha = artwork_alpha[..., None]\n    canvas[y0 : y0 + height, x0 : x0 + width] = region * (1 - alpha) + artwork_rgb * alpha\n\n    return np.clip(np.rint(canvas), 0, 255).astype(np.uint8)\n\n\ndef assert_artwork_unaltered(composite, artwork_rgb, artwork_alpha, top_left):\n    \"\"\"Verify the opaque part of the artwork survived compositing pixel for pixel.\"\"\"\n    x0, y0 = top_left\n    height, width = artwork_alpha.shape\n    opaque = artwork_alpha >= 1.0\n    if not opaque.any():\n        return\n    placed = composite[y0 : y0 + height, x0 : x0 + width][opaque]\n    expected = np.rint(artwork_rgb[opaque]).astype(np.uint8)\n    assert np.array_equal(placed, expected), \"artwork pixels were altered during compositing\"\n\n\ndef fit_artwork(artwork_rgba, anchor_yx, containment_mask, max_width, min_width, steps=24):\n    \"\"\"Largest uniform scale whose alpha footprint stays fully inside the region.\n\n    A descending scan rather than a binary search: it makes no monotonicity\n    assumption about the alpha shape, it is only ~24 small resizes, and it is\n    trivially deterministic.\n\n    Returns (top_left_xy, artwork_rgb, artwork_alpha, width) or None.\n    \"\"\"\n    aspect = artwork_rgba.height / max(artwork_rgba.width, 1)\n    image_height, image_width = containment_mask.shape\n    anchor_y, anchor_x = anchor_yx\n\n    if max_width < min_width:\n        return None\n\n    for width in np.linspace(max_width, min_width, steps).astype(int):\n        if width < 1:\n            continue\n        height = max(int(round(width * aspect)), 1)\n\n        x0 = int(round(anchor_x - width / 2))\n        y0 = int(round(anchor_y - height / 2))\n        if x0 < 0 or y0 < 0 or x0 + width > image_width or y0 + height > image_height:\n            continue\n\n        rgb, alpha = resize_artwork(artwork_rgba, width, height)\n        footprint = alpha > 0\n        # THE postcondition: no artwork pixel may land outside the assigned region.\n        if (footprint & ~containment_mask[y0 : y0 + height, x0 : x0 + width]).any():\n            continue\n\n        return (x0, y0), rgb, alpha, width\n\n    return None\n\n\ndef compute_confidence(regions, frame, allowed, fit_headroom):\n    \"\"\"Five sub-scores combined with min(), not a weighted average.\n\n    A weighted mean lets four healthy signals mask one disqualifying one: a\n    perfectly confident mask of a garment whose chest is entirely occluded still\n    cannot take a logo. Any single failure should be able to veto, and the limiting\n    factor is recorded so a production refusal says WHICH signal failed.\n    \"\"\"\n    body = largest_component(regions[\"body\"])\n    body_area = max(int(body.sum()), 1)\n\n    region_area_fraction = float(allowed.sum()) / body_area\n    mean_class_confidence = float(regions[\"confidence\"][body].mean()) if body.any() else 0.0\n\n    scores = {\n        # full marks at 3x the minimum acceptable region size\n        \"region_area\": np.clip(region_area_fraction / (3 * CONFIG[\"min_region_area_fraction\"]), 0, 1),\n        # max-softmax over 3 classes floors at 0.33; 0.95 is treated as certain\n        \"mask_confidence\": np.clip((mean_class_confidence - 0.33) / (0.95 - 0.33), 0, 1),\n        # Steeply mapped: a raw cohesion of 0.6 sounds mild but means 40% of the\n        # torso came back as a separate blob, which makes every axis derived from\n        # it fiction. Full marks only above 0.95.\n        \"mask_cohesion\": np.clip((component_cohesion(regions[\"body\"]) - 0.70) / 0.25, 0, 1),\n        # How much the artwork had to shrink below its intended size to fit. 1.0\n        # means it went in at full size; near 0 means the region only just took it,\n        # which is the honest signal that this preview is marginal. (Clearance from\n        # the seam is NOT measured here -- it is guaranteed structurally, by eroding\n        # the safe mask, so it cannot be traded away.)\n        \"fit_headroom\": np.clip(fit_headroom, 0, 1),\n        \"axis_stability\": np.clip(\n            (frame.axis_ratio - CONFIG[\"place_min_axis_ratio\"])\n            / max(CONFIG[\"axis_ratio_full_confidence\"] - CONFIG[\"place_min_axis_ratio\"], 1e-6),\n            0, 1,\n        ),\n    }\n    if frame.used_image_axes:\n        # The garment silhouette is too square for PCA to give a trustworthy axis, so\n        # we assumed the photo is upright. That is usually true for a flat-lay, so\n        # this is a fixed PENALTY, not a veto: it sits just above the refusal\n        # threshold, meaning an image-axes garment still places unless some other\n        # signal is also weak.\n        scores[\"axis_stability\"] = 0.4\n\n    limiting_factor = min(scores, key=scores.get)\n    return {\n        **{k: float(v) for k, v in scores.items()},\n        \"region_area_fraction\": region_area_fraction,\n        \"mean_class_confidence\": mean_class_confidence,\n        \"combined\": float(scores[limiting_factor]),\n        \"limiting_factor\": limiting_factor,\n    }\n\n\ndef max_width_fraction_for(horizontal):\n    \"\"\"A centred print and a left-chest logo are different products, not the same\n    logo at different positions. Size them accordingly.\"\"\"\n    key = \"centred_print\" if horizontal == \"centre\" else \"chest_logo\"\n    return CONFIG[f\"place_max_width_fraction_{key}\"]\n\n\ndef _attempt_placement(base_rgb, artwork_rgba, regions, frame, u_range, v_range, horizontal):\n    \"\"\"One placement attempt against one band. Returns a dict, placed True or False.\"\"\"\n    shape = base_rgb.shape[:2]\n    safe = build_safe_mask(regions[\"body\"], regions[\"sleeve\"], regions[\"seam\"])\n    if not safe.any():\n        return {\"placed\": False, \"reason_code\": REGION_TOO_SMALL}\n\n    allowed = band_mask(shape, frame, u_range, v_range) & safe\n    if not allowed.any():\n        return {\"placed\": False, \"reason_code\": REGION_TOO_SMALL}\n\n    containment = allowed if CONFIG[\"containment_mode\"] == \"assigned_region\" else safe\n\n    # The anchor is the deepest point inside the region: the position with the most\n    # clearance in every direction at once. argmax takes the first maximum in\n    # row-major order, so ties break deterministically.\n    distance = distance_to_edge(containment)\n    anchor_index = int(np.argmax(np.where(allowed, distance, -1.0)))\n    anchor_yx = np.unravel_index(anchor_index, shape)\n\n    garment_width = frame.u_max - frame.u_min\n    max_width = int(max_width_fraction_for(horizontal) * garment_width)\n    min_width = int(CONFIG[\"place_min_width_fraction\"] * garment_width)\n\n    fitted = fit_artwork(artwork_rgba, anchor_yx, containment, max_width, min_width)\n    if fitted is None:\n        return {\"placed\": False, \"reason_code\": ARTWORK_DOES_NOT_FIT}\n\n    top_left, artwork_rgb, artwork_alpha, width = fitted\n    fit_headroom = (width - min_width) / max(max_width - min_width, 1)\n\n    confidence = compute_confidence(regions, frame, allowed, fit_headroom)\n    if confidence[\"combined\"] < CONFIG[\"confidence_threshold\"]:\n        # Report WHICH signal vetoed it, not just \"low confidence\". In production\n        # this is the difference between an actionable alert and a mystery.\n        return {\n            \"placed\": False,\n            \"reason_code\": REFUSAL_CODE_BY_LIMITING_FACTOR[confidence[\"limiting_factor\"]],\n            \"confidence\": confidence,\n        }\n\n    composite = alpha_composite(base_rgb, artwork_rgb, artwork_alpha, top_left)\n    assert_artwork_unaltered(composite, artwork_rgb, artwork_alpha, top_left)\n\n    return {\n        \"placed\": True,\n        \"reason_code\": PLACED,\n        \"composite\": composite,\n        \"anchor_xy\": (int(anchor_yx[1]), int(anchor_yx[0])),\n        \"box_xyxy\": (top_left[0], top_left[1], top_left[0] + width, top_left[1] + artwork_alpha.shape[0]),\n        \"artwork_width_px\": int(width),\n        \"scale_factor\": width / artwork_rgba.width,\n        \"confidence\": confidence,\n        \"allowed_mask\": allowed,\n    }\n\n\ndef place_artwork_from_regions(base_rgb, artwork_rgba, instruction, regions):\n    \"\"\"The whole placement decision, given regions that are already predicted.\n\n    Split out from place_artwork() so every geometric rule can be tested against\n    hand-built masks without a checkpoint -- and so the geometry provably never\n    touches the model, the image pixels, or anything random.\n    \"\"\"\n    side, horizontal, vertical = parse_instruction(instruction)\n    parsed = {\"side\": side, \"horizontal\": horizontal, \"vertical\": vertical, \"raw\": instruction}\n\n    frame = build_garment_frame(regions[\"body\"], regions[\"sleeve\"])\n    if frame is None:\n        return {\"placed\": False, \"reason_code\": NO_GARMENT_DETECTED, \"instruction\": parsed}\n\n    if component_cohesion(regions[\"body\"]) < 0.5:\n        # The torso came back in pieces; any axis we derive from it is fiction.\n        return {\"placed\": False, \"reason_code\": MASK_FRAGMENTED, \"instruction\": parsed,\n                \"regions\": regions, \"frame\": frame}\n\n    u_range, v_range = resolve_band(side, horizontal, vertical)\n    result = _attempt_placement(base_rgb, artwork_rgba, regions, frame, u_range, v_range, horizontal)\n    result[\"fallback_applied\"] = None\n\n    # Fallback ladder: the assigned band failed, so try the safest region on any\n    # garment -- centre chest -- and record that we did. Never silently.\n    if not result[\"placed\"] and CONFIG[\"fallback_to_centre_chest\"] and (horizontal, vertical) != (\"centre\", \"chest\"):\n        fallback = _attempt_placement(\n            base_rgb, artwork_rgba, regions, frame,\n            CONFIG[\"band_horizontal\"][\"centre\"], CONFIG[\"band_vertical\"][\"chest\"], \"centre\",\n        )\n        if fallback[\"placed\"]:\n            fallback[\"fallback_applied\"] = f\"{horizontal} {vertical} -> centre chest\"\n            fallback[\"notes\"] = [f\"assigned band refused with {result['reason_code']}\"]\n            result = fallback\n\n    result[\"instruction\"] = parsed\n    result[\"regions\"] = regions\n    result[\"frame\"] = frame\n    result[\"config_hash\"] = CONFIG_HASH\n    return result\n\n\ndef place_artwork(garment_image, artwork_rgba, instruction, model):\n    \"\"\"Composite an artwork onto a garment photo at an instructed region.\n\n    Args:\n        garment_image: PIL image of the garment.\n        artwork_rgba: PIL RGBA image of the logo (alpha preserved).\n        instruction: e.g. \"front, left chest\" or \"back, centred\".\n        model: the trained segmenter.\n\n    Returns a dict that always carries `placed` and `reason_code`. On refusal there\n    is no composite: the system returns the garment untouched and flags it, rather\n    than printing a confident guess.\n    \"\"\"\n    base_rgb = np.array(garment_image.convert(\"RGB\"), dtype=np.uint8)\n    regions = predict_regions(model, garment_image)\n    return place_artwork_from_regions(base_rgb, artwork_rgba, instruction, regions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T12:35:04.266542Z","iopub.execute_input":"2026-08-21T12:35:04.266811Z","iopub.status.idle":"2026-08-21T12:35:04.301683Z","shell.execute_reply.started":"2026-08-21T12:35:04.266786Z","shell.execute_reply":"2026-08-21T12:35:04.300806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"place_artwork is defined:\", callable(place_artwork))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T12:39:09.943877Z","iopub.execute_input":"2026-08-21T12:39:09.944254Z","iopub.status.idle":"2026-08-21T12:39:09.949188Z","shell.execute_reply.started":"2026-08-21T12:39:09.944225Z","shell.execute_reply":"2026-08-21T12:39:09.948292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# CELL 11 — Run evaluation_cases.csv end to end, plus placement self-checks\n# =============================================================================\n# If the company's assets are present (artwork/ and evaluation_cases.csv) they are\n# used as-is. Otherwise placeholders are generated so the end-to-end path is\n# runnable and reviewable now; swapping in the real files changes nothing else.\n\nARTWORK_DIR = Path(\"data/artwork\")\nEVAL_CASES_PATH = Path(\"data/evaluation_cases.csv\")\nRENDER_DIR = OUTPUT_DIR / \"renders\"\nRENDER_DIR.mkdir(parents=True, exist_ok=True)\n\n\ndef make_placeholder_artwork():\n    \"\"\"Three transparent PNGs of different aspect ratios, for testing the fit logic.\"\"\"\n    ARTWORK_DIR.mkdir(parents=True, exist_ok=True)\n\n    square = Image.new(\"RGBA\", (400, 400), (0, 0, 0, 0))\n    painter = ImageDraw.Draw(square)\n    painter.ellipse([20, 20, 380, 380], fill=(20, 30, 90, 255))\n    painter.ellipse([110, 110, 290, 290], fill=(250, 210, 60, 255))\n    square.save(ARTWORK_DIR / \"roundel.png\")\n\n    wide = Image.new(\"RGBA\", (600, 180), (0, 0, 0, 0))\n    painter = ImageDraw.Draw(wide)\n    painter.rounded_rectangle([0, 0, 599, 179], radius=28, fill=(15, 15, 15, 255))\n    painter.text((40, 70), \"FABRICS\", fill=(255, 255, 255, 255))\n    wide.save(ARTWORK_DIR / \"wordmark.png\")\n\n    tall = Image.new(\"RGBA\", (200, 460), (0, 0, 0, 0))\n    painter = ImageDraw.Draw(tall)\n    painter.polygon([(100, 10), (190, 450), (10, 450)], fill=(190, 40, 60, 255))\n    tall.save(ARTWORK_DIR / \"crest.png\")\n\n    print(f\"wrote placeholder artwork to {ARTWORK_DIR}\")\n\n\ndef make_placeholder_eval_cases(image_ids, count=12):\n    \"\"\"A stand-in evaluation_cases.csv, matching the columns the brief describes.\"\"\"\n    instructions = [\n        (\"front, left chest\", \"left chest\"),\n        (\"front, right chest\", \"right chest\"),\n        (\"front, centred\", \"centre chest\"),\n        (\"back, centred\", \"centre back\"),\n        (\"front, centred hem\", \"hem\"),\n    ]\n    artworks = sorted(p.name for p in ARTWORK_DIR.glob(\"*.png\"))\n\n    rows = []\n    for index in range(min(count, len(image_ids))):\n        instruction, expected = instructions[index % len(instructions)]\n        rows.append({\n            \"case_id\": f\"case_{index:03d}\",\n            \"image_id\": image_ids[index],\n            \"artwork\": artworks[index % len(artworks)],\n            \"instruction\": instruction,\n            \"expected_region\": expected,\n        })\n\n    frame = pd.DataFrame(rows)\n    EVAL_CASES_PATH.parent.mkdir(parents=True, exist_ok=True)\n    frame.to_csv(EVAL_CASES_PATH, index=False)\n    print(f\"wrote placeholder {EVAL_CASES_PATH}\")\n    return frame\n\n\nif not ARTWORK_DIR.exists() or not any(ARTWORK_DIR.glob(\"*.png\")):\n    make_placeholder_artwork()\n\nCOLUMN_ALIASES = {\n    \"case_id\": [\"case_id\", \"id\", \"case\"],\n    \"image_id\": [\"image_id\", \"imageid\", \"garment_image\", \"garment_image_path\", \"image\", \"file_name\"],\n    \"artwork\": [\"artwork\", \"artwork_path\", \"artwork_png\", \"logo\", \"logo_path\"],\n    \"instruction\": [\"instruction\", \"placement\", \"placement_instruction\", \"prompt\"],\n    \"expected_region\": [\"expected_region\", \"expected\", \"region\", \"target_region\"],\n}\n\n\ndef normalise_eval_case_columns(frame):\n    \"\"\"Accept whatever column names the supplied evaluation_cases.csv uses.\n\n    This is the one file whose exact schema we cannot know in advance, so the\n    aliases absorb the difference here rather than scattering `.get()` calls through\n    the runner. Paths are reduced to a bare id/filename, since we resolve those\n    ourselves against the dataset and the artwork directory.\n    \"\"\"\n    lowered = {c.lower().strip(): c for c in frame.columns}\n    renamed = {}\n    for canonical, aliases in COLUMN_ALIASES.items():\n        match = next((lowered[a] for a in aliases if a in lowered), None)\n        if match is not None:\n            renamed[match] = canonical\n\n    frame = frame.rename(columns=renamed)\n    missing = {\"image_id\", \"artwork\", \"instruction\"} - set(frame.columns)\n    assert not missing, (\n        f\"{EVAL_CASES_PATH} is missing required columns {sorted(missing)}; \"\n        f\"found {list(frame.columns)}\"\n    )\n\n    if \"case_id\" not in frame.columns:\n        frame[\"case_id\"] = [f\"case_{i:03d}\" for i in range(len(frame))]\n    if \"expected_region\" not in frame.columns:\n        frame[\"expected_region\"] = \"\"\n\n    frame[\"image_id\"] = frame.image_id.astype(str).map(lambda v: Path(v).stem)\n    frame[\"artwork\"] = frame.artwork.astype(str).map(lambda v: Path(v).name)\n    return frame\n\n\nif EVAL_CASES_PATH.exists():\n    eval_cases = normalise_eval_case_columns(pd.read_csv(EVAL_CASES_PATH))\n    print(f\"using supplied {EVAL_CASES_PATH}: {len(eval_cases)} cases\")\nelse:\n    eval_cases = make_placeholder_eval_cases(VAL_IDS)\n\n\ndef expected_region_mask(expected_region, frame, shape):\n    \"\"\"Band mask for the case's `expected_region` string.\n\n    Reuses the same parser as the instruction, so \"expected region\" and \"instructed\n    region\" cannot drift apart. Returns None if the string carries no side token, in\n    which case we score containment against the safe mask instead.\n    \"\"\"\n    try:\n        side, horizontal, vertical = parse_instruction(expected_region)\n    except AssertionError:\n        return None\n    u_range, v_range = resolve_band(side, horizontal, vertical)\n    return band_mask(shape, frame, u_range, v_range)\n\n\ndef artwork_footprint_mask(result, shape):\n    \"\"\"Boolean mask of the pixels the artwork actually covers.\"\"\"\n    footprint = np.zeros(shape, dtype=bool)\n    x0, y0, x1, y1 = result[\"box_xyxy\"]\n    footprint[y0:y1, x0:x1] = True\n    return footprint\n\n\ndef draw_case_figure(image, result, path, title):\n    \"\"\"Save the composite next to the predicted regions, so a reviewer can see why.\"\"\"\n    figure, axes = plt.subplots(1, 2, figsize=(9, 4.6))\n    regions = result[\"regions\"]\n\n    label_map = np.where(regions[\"body\"], BODY, BACKGROUND).astype(np.uint8)\n    label_map[regions[\"sleeve\"]] = SLEEVE\n    axes[0].imshow(overlay_label_map(np.array(image.convert(\"RGB\")), label_map, seam=regions[\"seam\"]))\n    axes[0].set_title(\"predicted regions + seam band\", fontsize=9)\n\n    if result[\"placed\"]:\n        axes[1].imshow(result[\"composite\"])\n        axes[1].set_title(f\"placed  ({result['confidence']['combined']:.2f} conf)\", fontsize=9)\n    else:\n        axes[1].imshow(np.array(image.convert(\"RGB\")))\n        axes[1].set_title(f\"REFUSED — {result['reason_code']}\", fontsize=9, color=\"crimson\")\n\n    for axis in axes:\n        axis.axis(\"off\")\n    figure.suptitle(title, fontsize=10)\n    figure.tight_layout()\n    figure.savefig(path, dpi=110)\n    plt.close(figure)\n\n\ndef run_eval_cases(eval_cases_frame):\n    \"\"\"Run every placement case, render it, and record what happened and why.\"\"\"\n    rows = []\n\n    for _, case in eval_cases_frame.iterrows():\n        image_path = image_path_for(case.image_id)\n        garment = Image.open(image_path).convert(\"RGB\")\n        artwork = Image.open(ARTWORK_DIR / case.artwork).convert(\"RGBA\")\n\n        result = place_artwork(garment, artwork, case.instruction, model)\n\n        record = {\n            \"case_id\": case.case_id,\n            \"image_id\": case.image_id,\n            \"artwork\": case.artwork,\n            \"instruction\": case.instruction,\n            \"expected_region\": case.get(\"expected_region\", \"\"),\n            \"placed\": result[\"placed\"],\n            \"reason_code\": result[\"reason_code\"],\n            \"fallback_applied\": result.get(\"fallback_applied\"),\n            \"confidence\": round(result[\"confidence\"][\"combined\"], 4) if result.get(\"confidence\") else None,\n            \"limiting_factor\": result[\"confidence\"][\"limiting_factor\"] if result.get(\"confidence\") else None,\n            \"anchor_xy\": str(result.get(\"anchor_xy\")),\n            \"artwork_width_px\": result.get(\"artwork_width_px\"),\n            \"axis_ratio\": round(result[\"frame\"].axis_ratio, 3) if result.get(\"frame\") else None,\n            \"orientation_source\": result[\"frame\"].orientation_source if result.get(\"frame\") else None,\n            \"config_hash\": CONFIG_HASH,\n        }\n\n        if result[\"placed\"]:\n            shape = result[\"composite\"].shape[:2]\n            footprint = artwork_footprint_mask(result, shape)\n\n            # The two guarantees the brief asks for, checked rather than assumed.\n            record[\"crosses_seam\"] = bool((footprint & result[\"regions\"][\"seam\"]).any())\n            expected = expected_region_mask(record[\"expected_region\"], result[\"frame\"], shape)\n            record[\"inside_expected_region\"] = (\n                bool(not (footprint & ~expected).any()) if expected is not None else None\n            )\n\n            Image.fromarray(result[\"composite\"]).save(RENDER_DIR / f\"{case.case_id}.png\")\n        else:\n            record[\"crosses_seam\"] = None\n            record[\"inside_expected_region\"] = None\n\n        draw_case_figure(\n            garment, result, RENDER_DIR / f\"{case.case_id}_explained.png\",\n            f\"{case.case_id} — {case.instruction}\",\n        )\n        rows.append(record)\n\n    return pd.DataFrame(rows)\n\n\nEVAL_RESULTS = run_eval_cases(eval_cases)\nEVAL_RESULTS.to_csv(OUTPUT_DIR / \"evaluation_case_results.csv\", index=False)\n\nplaced = EVAL_RESULTS.placed.sum()\ninside = EVAL_RESULTS.inside_expected_region.fillna(False).sum()\ncrossing = EVAL_RESULTS.crosses_seam.fillna(False).sum()\n\nprint(f\"\\nplacement cases      : {len(EVAL_RESULTS)}\")\nprint(f\"placed               : {placed}  ({100 * placed / len(EVAL_RESULTS):.0f}%)\")\nprint(f\"refused              : {len(EVAL_RESULTS) - placed}\")\nprint(f\"inside expected region: {inside}/{placed}   <- placement accuracy\")\nprint(f\"crossing a seam       : {crossing}   <- must be 0\")\nif len(EVAL_RESULTS) - placed:\n    print(\"\\nrefusal reasons:\")\n    print(EVAL_RESULTS.loc[~EVAL_RESULTS.placed, \"reason_code\"].value_counts().to_string())\nprint(f\"\\nrenders -> {RENDER_DIR}\")\n\n\n# ---------------------------------------------------------------------------\n# Self-checks. These are the claims the brief makes us responsible for.\n# ---------------------------------------------------------------------------\n\ndef check_determinism_and_integrity():\n    \"\"\"Same inputs, same output; and no seam crossing on any placed case.\"\"\"\n    case = eval_cases.iloc[0]\n    garment = Image.open(image_path_for(case.image_id)).convert(\"RGB\")\n    artwork = Image.open(ARTWORK_DIR / case.artwork).convert(\"RGBA\")\n\n    first = place_artwork(garment, artwork, case.instruction, model)\n    second = place_artwork(garment, artwork, case.instruction, model)\n\n    assert first[\"placed\"] == second[\"placed\"], \"placement decision is not deterministic\"\n    if first[\"placed\"]:\n        assert np.array_equal(first[\"composite\"], second[\"composite\"]), \\\n            \"composite is not byte-identical across runs\"\n        assert first[\"box_xyxy\"] == second[\"box_xyxy\"], \"artwork box is not deterministic\"\n\n    assert not EVAL_RESULTS.crosses_seam.fillna(False).any(), \\\n        \"at least one placed artwork crosses a seam band\"\n\n    print(\"self-checks passed: deterministic output, no seam crossings\")\n\n\ncheck_determinism_and_integrity()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T12:41:01.022882Z","iopub.execute_input":"2026-08-21T12:41:01.023175Z","iopub.status.idle":"2026-08-21T12:41:33.862539Z","shell.execute_reply.started":"2026-08-21T12:41:01.023151Z","shell.execute_reply":"2026-08-21T12:41:33.861660Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(EVAL_RESULTS[[\"case_id\",\"placed\",\"fallback_applied\",\"inside_expected_region\",\"artwork\"]])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T12:44:22.676804Z","iopub.execute_input":"2026-08-21T12:44:22.678056Z","iopub.status.idle":"2026-08-21T12:44:22.686584Z","shell.execute_reply.started":"2026-08-21T12:44:22.678018Z","shell.execute_reply":"2026-08-21T12:44:22.685831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# CELL 12 — Package the run: config, environment, and the numbers for the note\n# =============================================================================\n\nimport subprocess\nimport sys\n\ntry:\n    import yaml\n    (OUTPUT_DIR / \"config_used.yaml\").write_text(yaml.safe_dump(CONFIG, sort_keys=False))\nexcept ImportError:\n    (OUTPUT_DIR / \"config_used.json\").write_text(json.dumps(CONFIG, indent=2))\n\n# The reproducible environment record is the one resolved in THIS session, not a\n# hand-written list.\nfrozen = subprocess.run([sys.executable, \"-m\", \"pip\", \"freeze\"], capture_output=True, text=True).stdout\n(OUTPUT_DIR / \"requirements_frozen.txt\").write_text(frozen)\n\nRUN_SUMMARY = {\n    \"config_hash\": CONFIG_HASH,\n    \"torch_version\": torch.__version__,\n    \"device\": str(DEVICE),\n    \"rle_order\": RLE_ORDER,\n\n    \"split_source\": SPLIT_SOURCE,\n    \"manifest_sha256\": MANIFEST_SHA,\n    \"n_train\": len(TRAIN_IDS), \"n_dev\": len(DEV_IDS), \"n_val\": len(VAL_IDS),\n    \"leakage_near_duplicate_pairs\": len(DUPLICATE_PAIRS),\n    \"leakage_contaminated_val_images\": len(CONTAMINATED_VAL_IDS),\n\n    \"backbone\": CONFIG[\"backbone\"],\n    \"frozen_backbone\": True,\n    \"total_parameters\": int(TOTAL_PARAMS),\n    \"trainable_parameters\": int(TRAINABLE_PARAMS),\n    \"trainable_parameter_cap\": CONFIG[\"trainable_parameter_cap\"],\n    \"best_epoch\": int(checkpoint[\"epoch\"]),\n\n    \"val_iou_dataset_level\": {\n        name: float(val_metrics[\"iou_dataset\"][i]) for i, name in enumerate(CONFIG[\"class_names\"])\n    },\n    \"val_iou_image_mean\": {\n        name: float(val_metrics[\"iou_image_mean\"][i]) for i, name in enumerate(CONFIG[\"class_names\"])\n    },\n    \"val_miou_dataset_level\": float(val_metrics[\"miou_dataset\"]),\n    \"val_miou_deduped\": float(val_metrics_clean[\"miou_dataset\"]) if val_metrics_clean else None,\n    \"val_boundary_f1\": float(val_metrics[\"boundary_f1\"]),\n\n    \"placement_cases\": int(len(EVAL_RESULTS)),\n    \"placement_placed\": int(EVAL_RESULTS.placed.sum()),\n    \"placement_inside_expected_region\": int(EVAL_RESULTS.inside_expected_region.fillna(False).sum()),\n    \"placement_seam_crossings\": int(EVAL_RESULTS.crosses_seam.fillna(False).sum()),\n}\n\n(OUTPUT_DIR / \"run_summary.json\").write_text(json.dumps(RUN_SUMMARY, indent=2))\n\nprint(json.dumps(RUN_SUMMARY, indent=2))\nprint(f\"\\nDeliverables in {OUTPUT_DIR.resolve()}:\")\nfor path in sorted(OUTPUT_DIR.iterdir()):\n    print(f\"  {path.name}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-21T12:46:13.420348Z","iopub.execute_input":"2026-08-21T12:46:13.420767Z","iopub.status.idle":"2026-08-21T12:46:19.283242Z","shell.execute_reply.started":"2026-08-21T12:46:13.420694Z","shell.execute_reply":"2026-08-21T12:46:19.282292Z"}},"outputs":[],"execution_count":null}]}