{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nfor root, dirs, files in os.walk(\"/kaggle/input\"):\n    level = root.replace(\"/kaggle/input\", \"\").count(os.sep)\n    indent = \"  \" * level\n    print(f\"{indent}{os.path.basename(root)}/\")\n    for file in files[:10]:\n        print(f\"{indent}  {file}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:08:21.968451Z","iopub.execute_input":"2026-09-19T18:08:21.968768Z","iopub.status.idle":"2026-09-19T18:10:44.923037Z","shell.execute_reply.started":"2026-09-19T18:08:21.968738Z","shell.execute_reply":"2026-09-19T18:10:44.922298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom PIL import Image\nimport os\n\nbase = \"/kaggle/input/competitions/airbus-ship-detection\"\ncsv_path = os.path.join(base, \"train_ship_segmentations_v2.csv\")\n\ndf = pd.read_csv(csv_path)\n\nprint(\"CSV shape:\", df.shape)\nprint(\"Columns:\", df.columns.tolist())\nprint(\"Unique images:\", df[\"ImageId\"].nunique())\nprint(\"Rows with ships:\", df[\"EncodedPixels\"].notna().sum())\nprint(\"Rows without ships:\", df[\"EncodedPixels\"].isna().sum())\nprint(\"Unique ship images:\", df.loc[df[\"EncodedPixels\"].notna(), \"ImageId\"].nunique())\n\nimg_path = os.path.join(base, \"train_v2\", df[\"ImageId\"].iloc[0])\nimg = Image.open(img_path)\n\nprint(\"Sample image:\", df[\"ImageId\"].iloc[0])\nprint(\"Image resolution:\", img.size)\nprint(\"Image mode:\", img.mode)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:12:44.170564Z","iopub.execute_input":"2026-09-19T18:12:44.170906Z","iopub.status.idle":"2026-09-19T18:12:45.68101Z","shell.execute_reply.started":"2026-09-19T18:12:44.170877Z","shell.execute_reply":"2026-09-19T18:12:45.680252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ship_counts = (\n    df.dropna(subset=[\"EncodedPixels\"])\n      .groupby(\"ImageId\")\n      .size()\n)\n\nprint(\"Ship images:\", len(ship_counts))\nprint(\"Total ships:\", ship_counts.sum())\nprint()\nprint(\"Ships per image:\")\nprint(ship_counts.value_counts().sort_index())\nprint()\nprint(\"Maximum ships in one image:\", ship_counts.max())\nprint(\"Average ships per ship image:\", ship_counts.mean())\nprint(\"Median ships per ship image:\", ship_counts.median())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:14:13.77386Z","iopub.execute_input":"2026-09-19T18:14:13.774506Z","iopub.status.idle":"2026-09-19T18:14:13.867911Z","shell.execute_reply.started":"2026-09-19T18:14:13.774473Z","shell.execute_reply":"2026-09-19T18:14:13.866928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nship_counts = (\n    df.dropna(subset=[\"EncodedPixels\"])\n      .groupby(\"ImageId\")\n      .size()\n      .rename(\"ship_count\")\n)\n\nship_images = ship_counts.reset_index()\n\nempty_images = (\n    df[df[\"EncodedPixels\"].isna()][[\"ImageId\"]]\n    .drop_duplicates()\n)\n\nrng = np.random.default_rng(42)\n\n# Target number of ship images\ntarget_ship = {\n    \"train\": 3500,\n    \"val\": 525,\n    \"test\": 1050\n}\n\n# Target number of empty images\ntarget_empty = {\n    \"train\": 1500,\n    \"val\": 225,\n    \"test\": 450\n}\n\n# Stratified sampling of ship images by number of ships\nselected_ship = {}\n\nfor split, target in target_ship.items():\n    parts = []\n\n    for count, group in ship_images.groupby(\"ship_count\"):\n        proportion = len(group) / len(ship_images)\n        n = round(target * proportion)\n\n        if n > 0:\n            sampled = group.sample(\n                n=min(n, len(group)),\n                random_state=42\n            )\n            parts.append(sampled)\n\n    selected = pd.concat(parts)[\"ImageId\"].tolist()\n\n    # Adjust to exact target\n    if len(selected) > target:\n        selected = rng.choice(selected, target, replace=False).tolist()\n\n    selected_ship[split] = selected\n\n# Sample empty images\nempty_array = empty_images[\"ImageId\"].to_numpy()\nrng.shuffle(empty_array)\n\nselected_empty = {\n    \"train\": empty_array[:1500],\n    \"val\": empty_array[1500:1725],\n    \"test\": empty_array[1725:2175]\n}\n\n# Create final split table\nsplit_rows = []\n\nfor split in [\"train\", \"val\", \"test\"]:\n    for image_id in selected_ship[split]:\n        split_rows.append([image_id, split, \"ship\"])\n\n    for image_id in selected_empty[split]:\n        split_rows.append([image_id, split, \"empty\"])\n\nsplits = pd.DataFrame(\n    split_rows,\n    columns=[\"ImageId\", \"split\", \"image_type\"]\n)\n\n# Shuffle\nsplits = splits.sample(frac=1, random_state=42).reset_index(drop=True)\n\nprint(\"Split counts:\")\nprint(splits[\"split\"].value_counts())\n\nprint()\nprint(\"Image type counts:\")\nprint(\n    splits.groupby([\"split\", \"image_type\"])\n          .size()\n)\n\nprint()\nprint(\"Total:\", len(splits))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:16:35.83291Z","iopub.execute_input":"2026-09-19T18:16:35.833268Z","iopub.status.idle":"2026-09-19T18:16:36.014801Z","shell.execute_reply.started":"2026-09-19T18:16:35.833236Z","shell.execute_reply":"2026-09-19T18:16:36.013903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"manifest_path = \"/kaggle/working/airbus_subset_manifest.csv\"\n\nsplits.to_csv(manifest_path, index=False)\n\nprint(\"Saved:\", manifest_path)\nprint(\"Rows:\", len(splits))\nprint()\nprint(splits.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:18:59.498459Z","iopub.execute_input":"2026-09-19T18:18:59.498825Z","iopub.status.idle":"2026-09-19T18:18:59.532248Z","shell.execute_reply.started":"2026-09-19T18:18:59.498793Z","shell.execute_reply":"2026-09-19T18:18:59.531484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"check = pd.read_csv(manifest_path)\n\nprint(\"Manifest rows:\", len(check))\nprint()\nprint(check.groupby([\"split\", \"image_type\"]).size())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:19:19.341058Z","iopub.execute_input":"2026-09-19T18:19:19.341901Z","iopub.status.idle":"2026-09-19T18:19:19.356276Z","shell.execute_reply.started":"2026-09-19T18:19:19.341867Z","shell.execute_reply":"2026-09-19T18:19:19.355358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_ids = set(splits[\"ImageId\"])\n\nsubset_annotations = df[df[\"ImageId\"].isin(selected_ids)].copy()\n\nsubset_annotations_path = \"/kaggle/working/airbus_subset_annotations.csv\"\n\nsubset_annotations.to_csv(\n    subset_annotations_path,\n    index=False\n)\n\nprint(\"Selected images:\", len(selected_ids))\nprint(\"Annotation rows:\", len(subset_annotations))\nprint(\"Images with annotations:\",\n      subset_annotations[\"ImageId\"].nunique())\nprint(\"Images without annotations:\",\n      len(selected_ids) - subset_annotations[\"ImageId\"].nunique())\nprint()\nprint(\"Saved:\", subset_annotations_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:20:27.580996Z","iopub.execute_input":"2026-09-19T18:20:27.581345Z","iopub.status.idle":"2026-09-19T18:20:27.744985Z","shell.execute_reply.started":"2026-09-19T18:20:27.581314Z","shell.execute_reply":"2026-09-19T18:20:27.744187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_ship_counts = (\n    subset_annotations.dropna(subset=[\"EncodedPixels\"])\n    .groupby(\"ImageId\")\n    .size()\n)\n\nprint(\"Selected ship images:\", len(selected_ship_counts))\nprint(\"Selected ships:\", selected_ship_counts.sum())\nprint()\nprint(\"Ships per image:\")\nprint(selected_ship_counts.value_counts().sort_index())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:21:21.372349Z","iopub.execute_input":"2026-09-19T18:21:21.372691Z","iopub.status.idle":"2026-09-19T18:21:21.387972Z","shell.execute_reply.started":"2026-09-19T18:21:21.37266Z","shell.execute_reply":"2026-09-19T18:21:21.387011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Manifest:\")\nprint(splits.groupby([\"split\", \"image_type\"]).size())\n\nprint()\nprint(\"Unique IDs in manifest:\", splits[\"ImageId\"].nunique())\n\nprint()\nprint(\"Ship IDs in manifest:\", \n      splits.loc[splits[\"image_type\"] == \"ship\", \"ImageId\"].nunique())\n\nprint()\nprint(\"Ship IDs in annotation CSV:\",\n      subset_annotations[\"ImageId\"].nunique())\n\nprint()\nprint(\"Missing ship IDs:\",\n      len(\n          set(splits.loc[splits[\"image_type\"] == \"ship\", \"ImageId\"])\n          - set(subset_annotations[\"ImageId\"])\n      ))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:22:52.656558Z","iopub.execute_input":"2026-09-19T18:22:52.65688Z","iopub.status.idle":"2026-09-19T18:22:52.678817Z","shell.execute_reply.started":"2026-09-19T18:22:52.656852Z","shell.execute_reply":"2026-09-19T18:22:52.678067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Total manifest rows:\", len(splits))\nprint(\"Unique Image IDs:\", splits[\"ImageId\"].nunique())\nprint()\nprint(splits.groupby([\"split\", \"image_type\"]).size())\nprint()\nprint(\"Duplicate Image IDs:\", splits[\"ImageId\"].duplicated().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:23:58.615172Z","iopub.execute_input":"2026-09-19T18:23:58.615519Z","iopub.status.idle":"2026-09-19T18:23:58.627475Z","shell.execute_reply.started":"2026-09-19T18:23:58.615492Z","shell.execute_reply":"2026-09-19T18:23:58.626701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nship_counts = (\n    df.dropna(subset=[\"EncodedPixels\"])\n      .groupby(\"ImageId\")\n      .size()\n      .rename(\"ship_count\")\n      .reset_index()\n)\n\nempty_images = (\n    df[df[\"EncodedPixels\"].isna()][\"ImageId\"]\n    .drop_duplicates()\n    .to_numpy()\n)\n\nrng = np.random.default_rng(42)\n\n# Shuffle all ship images once\nship_ids = ship_counts[\"ImageId\"].to_numpy()\nrng.shuffle(ship_ids)\n\n# Shuffle all empty images once\nrng.shuffle(empty_images)\n\n# Exact targets\nship_targets = {\n    \"train\": 3498,\n    \"val\": 525,\n    \"test\": 1050\n}\n\nempty_targets = {\n    \"train\": 1500,\n    \"val\": 225,\n    \"test\": 450\n}\n\n# Allocate without overlap\nselected_ship = {}\nstart = 0\n\nfor split, count in ship_targets.items():\n    selected_ship[split] = ship_ids[start:start + count].tolist()\n    start += count\n\nselected_empty = {}\nstart = 0\n\nfor split, count in empty_targets.items():\n    selected_empty[split] = empty_images[start:start + count].tolist()\n    start += count\n\n# Build manifest\nrows = []\n\nfor split in [\"train\", \"val\", \"test\"]:\n    for image_id in selected_ship[split]:\n        rows.append([image_id, split, \"ship\"])\n\n    for image_id in selected_empty[split]:\n        rows.append([image_id, split, \"empty\"])\n\nsplits = pd.DataFrame(\n    rows,\n    columns=[\"ImageId\", \"split\", \"image_type\"]\n)\n\nsplits = splits.sample(\n    frac=1,\n    random_state=42\n).reset_index(drop=True)\n\n# Save\nmanifest_path = \"/kaggle/working/airbus_subset_manifest.csv\"\nsplits.to_csv(manifest_path, index=False)\n\nprint(\"Total rows:\", len(splits))\nprint(\"Unique images:\", splits[\"ImageId\"].nunique())\nprint(\"Duplicate IDs:\", splits[\"ImageId\"].duplicated().sum())\nprint()\nprint(splits.groupby([\"split\", \"image_type\"]).size())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:24:36.470031Z","iopub.execute_input":"2026-09-19T18:24:36.470386Z","iopub.status.idle":"2026-09-19T18:24:36.69389Z","shell.execute_reply.started":"2026-09-19T18:24:36.470353Z","shell.execute_reply":"2026-09-19T18:24:36.692954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_ids = set(splits[\"ImageId\"])\n\nsubset_annotations = df[\n    df[\"ImageId\"].isin(selected_ids)\n].copy()\n\nsubset_annotations_path = \"/kaggle/working/airbus_subset_annotations.csv\"\n\nsubset_annotations.to_csv(\n    subset_annotations_path,\n    index=False\n)\n\nprint(\"Selected images:\", len(selected_ids))\nprint(\"Annotation rows:\", len(subset_annotations))\nprint(\"Unique images in annotations:\", subset_annotations[\"ImageId\"].nunique())\nprint(\"Images with ships:\",\n      subset_annotations[\"EncodedPixels\"].notna().groupby(\n          subset_annotations[\"ImageId\"]\n      ).any().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:25:24.257569Z","iopub.execute_input":"2026-09-19T18:25:24.25792Z","iopub.status.idle":"2026-09-19T18:25:24.480177Z","shell.execute_reply.started":"2026-09-19T18:25:24.257889Z","shell.execute_reply":"2026-09-19T18:25:24.479294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\ndef rle_to_bbox(rle, height=768, width=768):\n    values = list(map(int, rle.split()))\n    starts = np.array(values[0::2]) - 1\n    lengths = np.array(values[1::2])\n\n    mask = np.zeros(height * width, dtype=np.uint8)\n\n    for start, length in zip(starts, lengths):\n        mask[start:start + length] = 1\n\n    mask = mask.reshape((height, width), order=\"F\")\n\n    ys, xs = np.where(mask)\n\n    x_min = xs.min()\n    y_min = ys.min()\n    x_max = xs.max()\n    y_max = ys.max()\n\n    return x_min, y_min, x_max, y_max, mask.sum()\n\nsample_row = subset_annotations[\n    subset_annotations[\"EncodedPixels\"].notna()\n].iloc[0]\n\nbbox = rle_to_bbox(sample_row[\"EncodedPixels\"])\n\nprint(\"Image:\", sample_row[\"ImageId\"])\nprint(\"Bounding box:\", bbox[:4])\nprint(\"Mask pixels:\", bbox[4])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:26:14.417394Z","iopub.execute_input":"2026-09-19T18:26:14.417707Z","iopub.status.idle":"2026-09-19T18:26:14.433879Z","shell.execute_reply.started":"2026-09-19T18:26:14.417676Z","shell.execute_reply":"2026-09-19T18:26:14.433089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rle_to_yolo(rle, height=768, width=768):\n    x_min, y_min, x_max, y_max, _ = rle_to_bbox(\n        rle, height, width\n    )\n\n    x_center = ((x_min + x_max) / 2) / width\n    y_center = ((y_min + y_max) / 2) / height\n\n    box_width = (x_max - x_min + 1) / width\n    box_height = (y_max - y_min + 1) / height\n\n    return 0, x_center, y_center, box_width, box_height\n\n\nship_annotations = subset_annotations[\n    subset_annotations[\"EncodedPixels\"].notna()\n].copy()\n\nyolo_rows = []\n\nfor _, row in ship_annotations.iterrows():\n    values = rle_to_yolo(row[\"EncodedPixels\"])\n\n    yolo_rows.append([\n        row[\"ImageId\"],\n        *values\n    ])\n\nyolo_annotations = pd.DataFrame(\n    yolo_rows,\n    columns=[\n        \"ImageId\",\n        \"class_id\",\n        \"x_center\",\n        \"y_center\",\n        \"width\",\n        \"height\"\n    ]\n)\n\nprint(\"YOLO annotation rows:\", len(yolo_annotations))\nprint(\"Images with YOLO annotations:\",\n      yolo_annotations[\"ImageId\"].nunique())\n\nprint()\nprint(yolo_annotations.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:26:52.574035Z","iopub.execute_input":"2026-09-19T18:26:52.574785Z","iopub.status.idle":"2026-09-19T18:27:24.087471Z","shell.execute_reply.started":"2026-09-19T18:26:52.574752Z","shell.execute_reply":"2026-09-19T18:27:24.086438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Invalid class IDs:\",\n      (~yolo_annotations[\"class_id\"].isin([0])).sum())\n\nprint(\"Invalid x_center:\",\n      ((yolo_annotations[\"x_center\"] < 0) |\n       (yolo_annotations[\"x_center\"] > 1)).sum())\n\nprint(\"Invalid y_center:\",\n      ((yolo_annotations[\"y_center\"] < 0) |\n       (yolo_annotations[\"y_center\"] > 1)).sum())\n\nprint(\"Invalid width:\",\n      ((yolo_annotations[\"width\"] <= 0) |\n       (yolo_annotations[\"width\"] > 1)).sum())\n\nprint(\"Invalid height:\",\n      ((yolo_annotations[\"height\"] <= 0) |\n       (yolo_annotations[\"height\"] > 1)).sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:28:27.461011Z","iopub.execute_input":"2026-09-19T18:28:27.461433Z","iopub.status.idle":"2026-09-19T18:28:27.470904Z","shell.execute_reply.started":"2026-09-19T18:28:27.461402Z","shell.execute_reply":"2026-09-19T18:28:27.470089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport shutil\n\noutput_dir = \"/kaggle/working/airbus_yolo\"\n\nfor split in [\"train\", \"val\", \"test\"]:\n    os.makedirs(f\"{output_dir}/images/{split}\", exist_ok=True)\n    os.makedirs(f\"{output_dir}/labels/{split}\", exist_ok=True)\n\n# Create label files\nfor image_id, group in yolo_annotations.groupby(\"ImageId\"):\n    split = splits.loc[\n        splits[\"ImageId\"] == image_id, \"split\"\n    ].iloc[0]\n\n    label_path = f\"{output_dir}/labels/{split}/{image_id[:-4]}.txt\"\n\n    with open(label_path, \"w\") as f:\n        for _, row in group.iterrows():\n            f.write(\n                f\"{int(row['class_id'])} \"\n                f\"{row['x_center']:.6f} \"\n                f\"{row['y_center']:.6f} \"\n                f\"{row['width']:.6f} \"\n                f\"{row['height']:.6f}\\n\"\n            )\n\nprint(\"Label files created.\")\n\n# Copy images\nfor _, row in splits.iterrows():\n    image_id = row[\"ImageId\"]\n    split = row[\"split\"]\n\n    source = os.path.join(\n        base,\n        \"train_v2\",\n        image_id\n    )\n\n    destination = os.path.join(\n        output_dir,\n        \"images\",\n        split,\n        image_id\n    )\n\n    shutil.copy2(source, destination)\n\nprint(\"Images copied.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:29:08.451586Z","iopub.execute_input":"2026-09-19T18:29:08.452522Z","iopub.status.idle":"2026-09-19T18:30:08.95629Z","shell.execute_reply.started":"2026-09-19T18:29:08.452485Z","shell.execute_reply":"2026-09-19T18:30:08.955506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\noutput_dir = \"/kaggle/working/airbus_yolo\"\n\nfor split in [\"train\", \"val\", \"test\"]:\n    image_dir = os.path.join(output_dir, \"images\", split)\n    label_dir = os.path.join(output_dir, \"labels\", split)\n\n    images = [\n        f for f in os.listdir(image_dir)\n        if f.lower().endswith(\".jpg\")\n    ]\n\n    labels = [\n        f for f in os.listdir(label_dir)\n        if f.lower().endswith(\".txt\")\n    ]\n\n    print(f\"{split}:\")\n    print(\"  Images:\", len(images))\n    print(\"  Labels:\", len(labels))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:34:14.401133Z","iopub.execute_input":"2026-09-19T18:34:14.401499Z","iopub.status.idle":"2026-09-19T18:34:14.41895Z","shell.execute_reply.started":"2026-09-19T18:34:14.401466Z","shell.execute_reply":"2026-09-19T18:34:14.417975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nfor split in [\"train\", \"val\", \"test\"]:\n    label_dir = os.path.join(\n        output_dir, \"labels\", split\n    )\n\n    empty_labels = []\n    malformed_labels = []\n\n    for filename in os.listdir(label_dir):\n        if not filename.endswith(\".txt\"):\n            continue\n\n        path = os.path.join(label_dir, filename)\n\n        with open(path, \"r\") as f:\n            lines = [line.strip() for line in f if line.strip()]\n\n        if not lines:\n            empty_labels.append(filename)\n            continue\n\n        for line in lines:\n            values = line.split()\n\n            if len(values) != 5:\n                malformed_labels.append(filename)\n                break\n\n    print(split)\n    print(\"  Empty label files:\", len(empty_labels))\n    print(\"  Malformed label files:\", len(malformed_labels))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-09-19T18:35:18.00126Z","iopub.execute_input":"2026-09-19T18:35:18.001638Z","iopub.status.idle":"2026-09-19T18:35:18.156324Z","shell.execute_reply.started":"2026-09-19T18:35:18.001558Z","shell.execute_reply":"2026-09-19T18:35:18.155349Z"}},"outputs":[],"execution_count":null}]}