{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":99003,"databundleVersionId":11811197,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport random\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport xml.etree.ElementTree as ET\nfrom PIL import Image\n\nrandom.seed(42)\nnp.random.seed(42)\n\nTRAIN_IMAGES_DIR = './valdis_split_dataset/train/images'\nTRAIN_LABELS_DIR = './valdis_split_dataset/train/labels'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Helper functions for EDA","metadata":{}},{"cell_type":"code","source":"def count_valdis_annotations(labels_dir):\n    with_valdis = 0\n    total = 0\n    for xml_file in os.listdir(labels_dir):\n        if not xml_file.endswith('.xml'):\n            continue\n        total += 1\n        xml_path = os.path.join(labels_dir, xml_file)\n        tree = ET.parse(xml_path)\n        root = tree.getroot()\n        # Check if any object has name \"valdis\" (case-insensitive)\n        if any(obj.find('name').text.lower() == 'valdis' for obj in root.findall('object')):\n            with_valdis += 1\n    return with_valdis, total - with_valdis, total\n\ndef compute_bbox_stats(labels_dir):\n    widths, heights, areas = [], [], []\n    \n    for xml_file in os.listdir(labels_dir):\n        if not xml_file.endswith('.xml'):\n            continue\n        xml_path = os.path.join(labels_dir, xml_file)\n        tree = ET.parse(xml_path)\n        root = tree.getroot()\n        for obj in root.findall('object'):\n            if obj.find('name').text.lower() != 'valdis':\n                continue\n            bndbox = obj.find('bndbox')\n            xmin = float(bndbox.find('xmin').text)\n            ymin = float(bndbox.find('ymin').text)\n            xmax = float(bndbox.find('xmax').text)\n            ymax = float(bndbox.find('ymax').text)\n            width = xmax - xmin\n            height = ymax - ymin\n            widths.append(width)\n            heights.append(height)\n            areas.append(width * height)\n    \n    return np.mean(widths), np.mean(heights), np.mean(areas)\n\ndef plot_annotation_for_frame(image_filename, images_dir, labels_dir):\n    img_path = os.path.join(images_dir, image_filename)\n    image = np.array(Image.open(img_path))\n        \n    xml_filename = image_filename.replace('.PNG', '.xml').replace('.jpg', '.xml').replace('.jpeg', '.xml')\n    xml_path = os.path.join(labels_dir, xml_filename)\n    bboxes = []\n    if os.path.exists(xml_path):\n        tree = ET.parse(xml_path)\n        root = tree.getroot()\n        for obj in root.findall('object'):\n            if obj.find('name').text.lower() == 'valdis':\n                bndbox = obj.find('bndbox')\n                xmin = float(bndbox.find('xmin').text)\n                ymin = float(bndbox.find('ymin').text)\n                xmax = float(bndbox.find('xmax').text)\n                ymax = float(bndbox.find('ymax').text)\n                bboxes.append((xmin, ymin, xmax, ymax))\n    \n    plt.figure(figsize=(8, 6))\n    plt.imshow(image)\n    for (xmin, ymin, xmax, ymax) in bboxes:\n        rect = plt.Rectangle((xmin, ymin), xmax - xmin, ymax - ymin,\n                             edgecolor='red', facecolor='none', linewidth=2)\n        plt.gca().add_patch(rect)\n    plt.title(image_filename)\n    plt.axis('off')\n    plt.show()\n\navg_width, avg_height, avg_area = compute_bbox_stats(TRAIN_LABELS_DIR)\nframes_with, frames_without, total_frames = count_valdis_annotations(TRAIN_LABELS_DIR)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Total annotation files: {total_frames}\")\nprint(f\"Frames with Valdis: {frames_with}\")\nprint(f\"Frames without Valdis: {frames_without}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Average Width: {avg_width:.2f}\")\nprint(f\"Average Height: {avg_height:.2f}\")\nprint(f\"Average Area: {avg_area:.2f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_annotation_for_frame(\"frame_001948.PNG\", TRAIN_IMAGES_DIR, TRAIN_LABELS_DIR)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_rows = [{\n    'image_id': 'frame_000026.PNG',\n    'xmin': 479.41,\n    'ymin': 431.03,\n    'xmax': 1216.91,\n    'ymax': 527.04\n}]\n\nsubmission_df = pd.DataFrame(submission_rows)\nsubmission_csv_path = 'submission.csv'\nsubmission_df.to_csv(submission_csv_path, index=False)\nprint(f\"Submission CSV saved to: {submission_csv_path}\")\n\nsubmission_df.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!kaggle competitions submit -c where-is-valdis -f submission.csv -m \"Sample submission test\"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}