{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84302,"databundleVersionId":9430771,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" # Load Annotations from Subfolders","metadata":{"execution":{"iopub.status.busy":"2024-08-26T06:27:31.820059Z","iopub.execute_input":"2024-08-26T06:27:31.821252Z","iopub.status.idle":"2024-08-26T06:27:32.209989Z","shell.execute_reply.started":"2024-08-26T06:27:31.821205Z","shell.execute_reply":"2024-08-26T06:27:32.208965Z"}}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport xml.etree.ElementTree as ET\n\n# Set the main directory path\nmain_directory = '/kaggle/input/pmu-cai-competition2024'\n\n# Paths to folders\nannotation_folder = os.path.join(\"/kaggle/input/pmu-cai-competition2024/Dataset/Dataset/annotation\")\nval_folder = os.path.join(main_directory, 'val')\n\ndef parse_xml(xml_path):\n    tree = ET.parse(xml_path)\n    root = tree.getroot()\n    objects = []\n    for obj in root.findall('.//object'):\n        class_name = obj.find('name').text\n        bbox = obj.find('bndbox')\n        xmin = int(bbox.find('xmin').text)\n        ymin = int(bbox.find('ymin').text)\n        xmax = int(bbox.find('xmax').text)\n        ymax = int(bbox.find('ymax').text)\n        objects.append({'class': class_name, 'bbox': (xmin, ymin, xmax, ymax)})\n    return objects\n\ndef load_annotations_from_subfolders(annotation_folder):\n    data = []\n    for subfolder in os.listdir(annotation_folder):\n        subfolder_path = os.path.join(annotation_folder, subfolder)\n        if os.path.isdir(subfolder_path):\n            for xml_file in os.listdir(subfolder_path):\n                if xml_file.endswith('.xml'):\n                    file_path = os.path.join(subfolder_path, xml_file)\n                    objects = parse_xml(file_path)\n                    for obj in objects:\n                        data.append({\n                            'file': xml_file,\n                            'class': obj['class'],\n                            'xmin': obj['bbox'][0],\n                            'ymin': obj['bbox'][1],\n                            'xmax': obj['bbox'][2],\n                            'ymax': obj['bbox'][3]\n                        })\n    return pd.DataFrame(data)\n\n# Load and inspect data\nannotations_df = load_annotations_from_subfolders(annotation_folder)\nprint(annotations_df.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-08-26T07:25:42.238942Z","iopub.execute_input":"2024-08-26T07:25:42.239846Z","iopub.status.idle":"2024-08-26T07:27:01.939241Z","shell.execute_reply.started":"2024-08-26T07:25:42.239800Z","shell.execute_reply":"2024-08-26T07:27:01.938140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Validetion","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport os\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\n\n# Set the main directory path\nmain_directory = '/kaggle/input/pmu-cai-competition2024'\nannotation_folder = os.path.join(main_directory, 'val')\ntxt_path = os.path.join(annotation_folder, 'val_annotation.txt')\ncsv_path = os.path.join(annotation_folder, 'val_annotation.csv')\n\n# Load annotations from val_annotation.txt\ntry:\n    annotations_txt = pd.read_csv(txt_path, delimiter='\\s+', header=None, names=['image_id', 'class', 'xmin', 'ymin', 'xmax', 'ymax'], on_bad_lines='skip')\n    print(\"TXT Annotations Loaded Successfully\")\nexcept Exception as e:\n    print(f\"Error loading TXT annotations: {e}\")\n\n# Load annotations from val_annotation.csv\ntry:\n    annotations_csv = pd.read_csv(csv_path, on_bad_lines='skip')\n    print(\"CSV Annotations Loaded Successfully\")\nexcept Exception as e:\n    print(f\"Error loading CSV annotations: {e}\")\n\n# Display a sample of the loaded annotation data\nprint(\"\\nTXT Annotations:\")\nprint(annotations_txt.head())\nprint(\"\\nCSV Annotations:\")\nprint(annotations_csv.head())\n\n# Function to display image with bounding boxes\ndef display_image_with_boxes(image_path, annotations):\n    image = Image.open(image_path)\n    plt.figure(figsize=(10, 10))\n    plt.imshow(image)\n    ax = plt.gca()\n    for _, row in annotations.iterrows():\n        rect = patches.Rectangle((row['xmin'], row['ymin']), row['xmax'] - row['xmin'], row['ymax'] - row['ymin'],\n                                 linewidth=2, edgecolor='r', facecolor='none')\n        ax.add_patch(rect)\n        plt.text(row['xmin'], row['ymin'], row['class'], color='red', fontsize=12, bbox=dict(facecolor='white', alpha=0.5))\n    plt.show()\n\n# Function to show all images with annotations\ndef show_all_images_with_annotations(annotations):\n    image_filenames = [f for f in os.listdir(os.path.join(annotation_folder, 'val')) if f.endswith('.JPEG')]\n    for image_filename in image_filenames:\n        image_path = os.path.join(annotation_folder, 'val', image_filename)\n        if os.path.exists(image_path):\n            display_image_with_boxes(image_path, annotations[annotations['image_id'] == image_filename])\n        else:\n            print(f\"Image file not found: {image_filename}\")\n\n# Show all images with annotations\nshow_all_images_with_annotations(annotations_txt)  # Replace with annotations_csv if needed\n","metadata":{"execution":{"iopub.status.busy":"2024-08-26T07:30:46.593433Z","iopub.execute_input":"2024-08-26T07:30:46.593875Z","iopub.status.idle":"2024-08-26T07:30:49.133766Z","shell.execute_reply.started":"2024-08-26T07:30:46.593834Z","shell.execute_reply":"2024-08-26T07:30:49.132043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# trying EDA","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom PIL import Image\nimport matplotlib.patches as patches\nimport xml.etree.ElementTree as ET\n\n# Set the main directory path\nmain_directory = '/kaggle/input/pmu-cai-competition2024'\nannotation_folder = os.path.join(main_directory, 'Dataset/Dataset/annotation')\nval_folder = os.path.join(main_directory, 'val')\ntxt_path = os.path.join(val_folder, 'val_annotation.txt')\ncsv_path = os.path.join(val_folder, 'val_annotation.csv')\n\n# Function to parse XML\ndef parse_xml(xml_path):\n    tree = ET.parse(xml_path)\n    root = tree.getroot()\n    objects = []\n    for obj in root.findall('.//object'):\n        class_name = obj.find('name').text\n        bbox = obj.find('bndbox')\n        xmin = int(bbox.find('xmin').text)\n        ymin = int(bbox.find('ymin').text)\n        xmax = int(bbox.find('xmax').text)\n        ymax = int(bbox.find('ymax').text)\n        objects.append({'class': class_name, 'bbox': (xmin, ymin, xmax, ymax)})\n    return objects\n\n# Load XML annotations\ndef load_annotations_from_subfolders(annotation_folder):\n    data = []\n    for subfolder in os.listdir(annotation_folder):\n        subfolder_path = os.path.join(annotation_folder, subfolder)\n        if os.path.isdir(subfolder_path):\n            for xml_file in os.listdir(subfolder_path):\n                if xml_file.endswith('.xml'):\n                    file_path = os.path.join(subfolder_path, xml_file)\n                    objects = parse_xml(file_path)\n                    for obj in objects:\n                        data.append({\n                            'file': xml_file,\n                            'class': obj['class'],\n                            'xmin': obj['bbox'][0],\n                            'ymin': obj['bbox'][1],\n                            'xmax': obj['bbox'][2],\n                            'ymax': obj['bbox'][3]\n                        })\n    return pd.DataFrame(data)\n\n# Load XML annotations data\nannotations_df = load_annotations_from_subfolders(annotation_folder)\n\n# Load TXT annotations\ntry:\n    annotations_txt = pd.read_csv(txt_path, delimiter='\\s+', header=None, names=['image_id', 'class', 'xmin', 'ymin', 'xmax', 'ymax'], on_bad_lines='skip')\n    # Remove trailing characters and convert columns to numeric\n    annotations_txt['xmin'] = pd.to_numeric(annotations_txt['xmin'], errors='coerce')\n    annotations_txt['ymin'] = pd.to_numeric(annotations_txt['ymin'], errors='coerce')\n    annotations_txt['xmax'] = pd.to_numeric(annotations_txt['xmax'], errors='coerce')\n    annotations_txt['ymax'] = pd.to_numeric(annotations_txt['ymax'], errors='coerce')\n    annotations_txt['class'] = annotations_txt['class'].astype(str)  # Ensure class is treated as string\n    print(\"TXT Annotations Loaded Successfully\")\nexcept Exception as e:\n    print(f\"Error loading TXT annotations: {e}\")\n\n# Load CSV annotations\ntry:\n    # Check the first few rows to understand the structure\n    sample_csv = pd.read_csv(csv_path, nrows=5)\n    print(\"Sample of CSV Annotations:\")\n    print(sample_csv.head())\n    \n    # Adjust column names based on inspection\n    annotations_csv = pd.read_csv(csv_path, header=None, names=['image_id', 'class', 'xmin', 'ymin', 'xmax', 'ymax'], on_bad_lines='skip')\n    # Remove trailing characters and convert columns to numeric\n    annotations_csv['xmin'] = pd.to_numeric(annotations_csv['xmin'], errors='coerce')\n    annotations_csv['ymin'] = pd.to_numeric(annotations_csv['ymin'], errors='coerce')\n    annotations_csv['xmax'] = pd.to_numeric(annotations_csv['xmax'], errors='coerce')\n    annotations_csv['ymax'] = pd.to_numeric(annotations_csv['ymax'], errors='coerce')\n    annotations_csv['class'] = annotations_csv['class'].astype(str)  # Ensure class is treated as string\n    print(\"CSV Annotations Loaded Successfully\")\nexcept Exception as e:\n    print(f\"Error loading CSV annotations: {e}\")\n\n# Display a sample of the loaded annotation data\nprint(\"\\nTXT Annotations:\")\nprint(annotations_txt.head())\nprint(\"\\nCSV Annotations:\")\nprint(annotations_csv.head())\nprint(\"\\nXML Annotations:\")\nprint(annotations_df.head())\n\n# Check for missing values\nprint(\"\\nTXT Annotations missing values:\")\nprint(annotations_txt.isnull().sum())\nprint(\"\\nCSV Annotations missing values:\")\nprint(annotations_csv.isnull().sum())\nprint(\"\\nXML Annotations missing values:\")\nprint(annotations_df.isnull().sum())\n\n# Verify column types\nprint(\"\\nTXT Annotations column types:\")\nprint(annotations_txt.dtypes)\nprint(\"\\nCSV Annotations column types:\")\nprint(annotations_csv.dtypes)\nprint(\"\\nXML Annotations column types:\")\nprint(annotations_df.dtypes)\n\n# Statistical summary\nprint(\"\\nTXT Annotations statistics:\")\nprint(annotations_txt.describe())\nprint(\"\\nCSV Annotations statistics:\")\nprint(annotations_csv.describe())\nprint(\"\\nXML Annotations statistics:\")\nprint(annotations_df.describe())\n\n# Class distribution\nprint(\"\\nTXT Annotations class distribution:\")\nprint(annotations_txt['class'].value_counts())\nprint(\"\\nCSV Annotations class distribution:\")\nprint(annotations_csv['class'].value_counts())\nprint(\"\\nXML Annotations class distribution:\")\nprint(annotations_df['class'].value_counts())\n\n# Function to plot bounding box distribution\ndef plot_bbox_distribution(df, title):\n    plt.figure(figsize=(12, 8))\n    # Flatten and filter out NaNs\n    bbox_values = df[['xmin', 'ymin', 'xmax', 'ymax']].values.flatten()\n    bbox_values = bbox_values[~pd.isna(bbox_values)]\n    sns.histplot(bbox_values, bins=50)\n    plt.title(title)\n    plt.xlabel('Pixel Value')\n    plt.ylabel('Frequency')\n    plt.show()\n\nplot_bbox_distribution(annotations_txt, 'TXT Annotations Bounding Box Distribution')\nplot_bbox_distribution(annotations_csv, 'CSV Annotations Bounding Box Distribution')\nplot_bbox_distribution(annotations_df, 'XML Annotations Bounding Box Distribution')\n\n# Function to display image with bounding boxes\ndef display_image_with_boxes(image_path, annotations):\n    image = Image.open(image_path)\n    plt.figure(figsize=(10, 10))\n    plt.imshow(image)\n    ax = plt.gca()\n    for _, row in annotations.iterrows():\n        rect = patches.Rectangle((row['xmin'], row['ymin']), row['xmax'] - row['xmin'], row['ymax'] - row['ymin'],\n                                 linewidth=2, edgecolor='r', facecolor='none')\n        ax.add_patch(rect)\n        plt.text(row['xmin'], row['ymin'], row['class'], color='red', fontsize=12, bbox=dict(facecolor='white', alpha=0.5))\n    plt.show()\n\n# Function to show all images with annotations\ndef show_all_images_with_annotations(annotations):\n    image_filenames = [f for f in os.listdir(val_folder) if f.endswith('.JPEG')]\n    for image_filename in image_filenames:\n        image_path = os.path.join(val_folder, image_filename)\n        if os.path.exists(image_path):\n            display_image_with_boxes(image_path, annotations[annotations['image_id'] == image_filename])\n        else:\n            print(f\"Image file not found: {image_filename}\")\n\n# Show all images with annotations\nshow_all_images_with_annotations(annotations_txt)  # Replace with annotations_csv if needed\n\n# Check for duplicate annotations\nprint(\"\\nTXT Annotations duplicates:\", annotations_txt.duplicated().sum())\nprint(\"\\nCSV Annotations duplicates:\", annotations_csv.duplicated().sum())\nprint(\"\\nXML Annotations duplicates:\", annotations_df.duplicated().sum())\n","metadata":{"execution":{"iopub.status.busy":"2024-08-26T07:40:50.919439Z","iopub.execute_input":"2024-08-26T07:40:50.919873Z","iopub.status.idle":"2024-08-26T07:41:12.903136Z","shell.execute_reply.started":"2024-08-26T07:40:50.919830Z","shell.execute_reply":"2024-08-26T07:41:12.901854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}