{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":103103,"databundleVersionId":13042974,"sourceType":"competition"},{"sourceId":13513626,"sourceType":"datasetVersion","datasetId":8579938},{"sourceId":13528317,"sourceType":"datasetVersion","datasetId":8589949},{"sourceId":271198659,"sourceType":"kernelVersion"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 1. Initial setup","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport sys\nimport time\nimport glob\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as mpatches\nfrom matplotlib.patches import Patch\nfrom matplotlib.collections import PatchCollection\nimport seaborn as sns\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nimport yaml\nimport random\nfrom PIL import Image\nimport warnings\nwarnings.filterwarnings('ignore')\nimport json\n\nimport torch\nimport cv2\nimport albumentations as A   # Image augmentation","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:07:15.163209Z","iopub.execute_input":"2025-10-31T02:07:15.163805Z","iopub.status.idle":"2025-10-31T02:07:30.921901Z","shell.execute_reply.started":"2025-10-31T02:07:15.163768Z","shell.execute_reply":"2025-10-31T02:07:30.920635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set random seeds for reproducibility\nrandom.seed(42)\nnp.random.seed(42)\n\n# Set deterministic behavior for PyTorch\ntorch.manual_seed(42)\nif torch.cuda.is_available():\n    torch.cuda.manual_seed(42)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\nprint(f'\\nPyTorch Version: {torch.__version__}')\nprint(f'CUDA Available: {torch.cuda.is_available()}')\nif torch.cuda.is_available():\n    print(f'CUDA Device: {torch.cuda.get_device_name(0)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:07:30.924310Z","iopub.execute_input":"2025-10-31T02:07:30.925362Z","iopub.status.idle":"2025-10-31T02:07:30.948241Z","shell.execute_reply.started":"2025-10-31T02:07:30.925328Z","shell.execute_reply":"2025-10-31T02:07:30.946696Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Load data and configuration settings","metadata":{}},{"cell_type":"code","source":"# Define class information\nCLASS_INFO = {\n    0: {'name': 'Abrasion', 'description': 'Teeth with mechanical wear of hard tissues'},\n    1: {'name': 'Filling', 'description': 'Dental fillings of various types'},\n    2: {'name': 'Crown', 'description': 'Dental crown (restoration)'},\n    3: {'name': 'Caries Class 1', 'description': 'Caries in fissures and pits'},\n    4: {'name': 'Caries Class 2', 'description': 'Caries on proximal surfaces of molars/premolars'},\n    5: {'name': 'Caries Class 3', 'description': 'Caries on proximal surfaces of incisors/canines without incisal edge'},\n    6: {'name': 'Caries Class 4', 'description': 'Caries on proximal surfaces of incisors/canines with incisal edge'},\n    7: {'name': 'Caries Class 5', 'description': 'Cervical caries (buccal/lingual surfaces)'},\n    8: {'name': 'Caries Class 6', 'description': 'Caries on incisal edges or cusps'}\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:07:30.949208Z","iopub.execute_input":"2025-10-31T02:07:30.949473Z","iopub.status.idle":"2025-10-31T02:07:30.957635Z","shell.execute_reply.started":"2025-10-31T02:07:30.949452Z","shell.execute_reply":"2025-10-31T02:07:30.956467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define paths\nBASE_PATH = '/kaggle/input/alpha-dent/AlphaDent'\nTRAIN_IMAGES_PATH = f'{BASE_PATH}/images/train'\nVALID_IMAGES_PATH = f'{BASE_PATH}/images/valid'\nTEST_IMAGES_PATH = f'{BASE_PATH}/images/test'\nTRAIN_LABELS_PATH = f'{BASE_PATH}/labels/train'\nVALID_LABELS_PATH = f'{BASE_PATH}/labels/valid'\n\n# Output paths\nOUTPUT_DIR = '/kaggle/working/'\nWEIGHTS_DIR = f'{OUTPUT_DIR}/weights'\nos.makedirs(WEIGHTS_DIR, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:07:30.958998Z","iopub.execute_input":"2025-10-31T02:07:30.959365Z","iopub.status.idle":"2025-10-31T02:07:30.986518Z","shell.execute_reply.started":"2025-10-31T02:07:30.959330Z","shell.execute_reply":"2025-10-31T02:07:30.984873Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 경로 확인\npaths_info = {\n    'Train Images': TRAIN_IMAGES_PATH,\n    'Valid Images': VALID_IMAGES_PATH,\n    'Test Images': TEST_IMAGES_PATH,\n    'Train Labels': TRAIN_LABELS_PATH,\n    'Valid Labels': VALID_LABELS_PATH\n}\n\nprint(\"\\nPath verification:\")\nfor name, path in paths_info.items():\n    exists = os.path.exists(path)\n    status = \"OK\" if exists else \"NOT FOUND\"\n    print(f\"  {name}: {status}\")\n    if not exists:\n        raise FileNotFoundError(f\"{name} not found: {path}\")\n\n# 파일 개수 확인\ntrain_count = len([f for f in os.listdir(TRAIN_IMAGES_PATH) if f.endswith('.jpg')])\nvalid_count = len([f for f in os.listdir(VALID_IMAGES_PATH) if f.endswith('.jpg')])\ntest_count = len([f for f in os.listdir(TEST_IMAGES_PATH) if f.endswith('.jpg')])\n\nprint(f\"\\nDataset Statistics:\")\nprint(f\"  Training images: {train_count}\")\nprint(f\"  Validation images: {valid_count}\")\nprint(f\"  Test images: {test_count}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:07:30.987451Z","iopub.execute_input":"2025-10-31T02:07:30.987827Z","iopub.status.idle":"2025-10-31T02:07:31.119386Z","shell.execute_reply.started":"2025-10-31T02:07:30.987796Z","shell.execute_reply":"2025-10-31T02:07:31.117911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def yolo_to_coco_annotations(images_dir, labels_dir, output_json_path):\n    \"\"\"\n    YOLO 형식을 COCO 형식으로 변환\n\n    Args:\n        images_dir: 이미지 디렉토리 경로\n        labels_dir: 라벨 디렉토리 경로\n        output_json_path: 출력 COCO JSON 경로\n\n    Returns:\n        출력 JSON 파일 경로\n    \"\"\"\n    print(f\"\\nConverting YOLO to COCO: {os.path.basename(images_dir)}\")\n\n    images = []\n    annotations = []\n    annotation_id = 1\n\n    # 모든 이미지 파일 가져오기\n    image_files = sorted([f for f in os.listdir(images_dir) if f.endswith('.jpg')])\n\n    for image_id, img_file in enumerate(tqdm(image_files, desc=\"Converting\"), start=1):\n        img_path = os.path.join(images_dir, img_file)\n\n        # 이미지 크기 가져오기\n        with Image.open(img_path) as img:\n            width, height = img.size\n\n        # 이미지 정보 추가\n        images.append({\n            'id': image_id,\n            'file_name': img_file,\n            'width': width,\n            'height': height\n        })\n\n        # 해당 라벨 파일 읽기\n        label_file = img_file.replace('.jpg', '.txt')\n        label_path = os.path.join(labels_dir, label_file)\n\n        if not os.path.exists(label_path):\n            continue\n\n        with open(label_path, 'r') as f:\n            lines = f.readlines()\n\n        for line in lines:\n            line = line.strip()\n            if not line:\n                continue\n\n            parts = line.split()\n            if len(parts) < 5:\n                continue\n\n            # YOLO 형식 파싱\n            class_id = int(parts[0])\n            coords = [float(x) for x in parts[1:]]\n\n            if len(coords) < 6:  # 최소 3개 포인트 필요 (6 좌표)\n                continue\n\n            # 정규화된 좌표를 절대 좌표로 변환\n            absolute_coords = []\n            for i in range(0, len(coords), 2):\n                if i + 1 < len(coords):\n                    x = coords[i] * width\n                    y = coords[i + 1] * height\n                    absolute_coords.extend([x, y])\n\n            if len(absolute_coords) < 6:\n                continue\n\n            # Bounding box 계산\n            x_coords = [absolute_coords[i] for i in range(0, len(absolute_coords), 2)]\n            y_coords = [absolute_coords[i] for i in range(1, len(absolute_coords), 2)]\n\n            x_min, x_max = min(x_coords), max(x_coords)\n            y_min, y_max = min(y_coords), max(y_coords)\n            bbox_width = x_max - x_min\n            bbox_height = y_max - y_min\n\n            # 면적 계산 (bounding box 기준)\n            area = bbox_width * bbox_height\n\n            # Annotation 추가\n            annotations.append({\n                'id': annotation_id,\n                'image_id': image_id,\n                'category_id': class_id + 1,  # COCO는 1부터 시작\n                'segmentation': [absolute_coords],\n                'bbox': [x_min, y_min, bbox_width, bbox_height],\n                'area': area,\n                'iscrowd': 0\n            })\n            annotation_id += 1\n\n    # 카테고리 생성\n    categories = [\n        {'id': i + 1, 'name': CLASS_INFO[i]['name']}\n        for i in range(9)\n    ]\n\n    # COCO 형식 딕셔너리 생성\n    coco_dict = {\n        'images': images,\n        'annotations': annotations,\n        'categories': categories\n    }\n\n    # JSON으로 저장\n    with open(output_json_path, 'w') as f:\n        json.dump(coco_dict, f)\n\n    print(f\"  Images: {len(images)}\")\n    print(f\"  Annotations: {len(annotations)}\")\n    print(f\"  Saved to: {output_json_path}\")\n\n    return output_json_path\n\n\ndef polygon_to_mask(polygon, img_shape):\n    \"\"\"\n    Polygon 좌표를 binary mask로 변환\n\n    Args:\n        polygon: (N, 2) array of polygon coordinates\n        img_shape: (height, width) tuple\n\n    Returns:\n        Binary mask as numpy array\n    \"\"\"\n    mask = np.zeros(img_shape, dtype=np.uint8)\n    if len(polygon) > 0:\n        polygon_int = polygon.astype(np.int32)\n        cv2.fillPoly(mask, [polygon_int], 1)\n    return mask\n\nprint(\"Utility functions defined\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:07:31.120641Z","iopub.execute_input":"2025-10-31T02:07:31.121086Z","iopub.status.idle":"2025-10-31T02:07:31.144289Z","shell.execute_reply.started":"2025-10-31T02:07:31.121052Z","shell.execute_reply":"2025-10-31T02:07:31.140158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train과 validation 데이터를 COCO 형식으로 변환\nprint(\"\\n\" + \"=\"*60)\nprint(\"Converting YOLO to COCO Format\")\nprint(\"=\"*60)\n\nCOCO_TRAIN_PATH = os.path.join(OUTPUT_DIR, 'coco_train.json')\nCOCO_VAL_PATH = os.path.join(OUTPUT_DIR, 'coco_val.json')\n# COCO_TEST_PATH = os.path.join(OUTPUT_DIR, 'coco_test.json')\n\nif not os.path.exists(COCO_TRAIN_PATH):\n    yolo_to_coco_annotations(TRAIN_IMAGES_PATH, TRAIN_LABELS_PATH, COCO_TRAIN_PATH)\nelse:\n    print(f\"\\nTrain COCO file already exists: {COCO_TRAIN_PATH}\")\n\nif not os.path.exists(COCO_VAL_PATH):\n    yolo_to_coco_annotations(VALID_IMAGES_PATH, VALID_LABELS_PATH, COCO_VAL_PATH)\nelse:\n    print(f\"\\nVal COCO file already exists: {COCO_VAL_PATH}\")\n\nprint(\"\\nCOCO format conversion completed\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:07:31.148616Z","iopub.execute_input":"2025-10-31T02:07:31.148971Z","iopub.status.idle":"2025-10-31T02:08:36.706842Z","shell.execute_reply.started":"2025-10-31T02:07:31.148947Z","shell.execute_reply":"2025-10-31T02:08:36.705721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n# Load COCO json\nwith open(COCO_TRAIN_PATH, \"r\") as f:\n    coco_train = json.load(f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:08:36.707890Z","iopub.execute_input":"2025-10-31T02:08:36.708235Z","iopub.status.idle":"2025-10-31T02:08:41.278388Z","shell.execute_reply.started":"2025-10-31T02:08:36.708196Z","shell.execute_reply":"2025-10-31T02:08:41.277577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"images_train = pd.DataFrame(coco_train[\"images\"])\nannotations_train = pd.DataFrame(coco_train[\"annotations\"])\ncategories_train = pd.DataFrame(coco_train[\"categories\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:08:41.279630Z","iopub.execute_input":"2025-10-31T02:08:41.280068Z","iopub.status.idle":"2025-10-31T02:08:41.330838Z","shell.execute_reply.started":"2025-10-31T02:08:41.280038Z","shell.execute_reply":"2025-10-31T02:08:41.329907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Train Images:\", images_train.shape)\nprint(\"Train Annotations:\", annotations_train.shape)\nprint(\"Train Categories:\", categories_train.shape, \"\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:08:41.331621Z","iopub.execute_input":"2025-10-31T02:08:41.331902Z","iopub.status.idle":"2025-10-31T02:08:41.337958Z","shell.execute_reply.started":"2025-10-31T02:08:41.331882Z","shell.execute_reply":"2025-10-31T02:08:41.336842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load COCO json\nwith open(COCO_VAL_PATH, \"r\") as f:\n    coco_val = json.load(f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:08:41.339003Z","iopub.execute_input":"2025-10-31T02:08:41.339386Z","iopub.status.idle":"2025-10-31T02:08:41.679713Z","shell.execute_reply.started":"2025-10-31T02:08:41.339363Z","shell.execute_reply":"2025-10-31T02:08:41.678556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"images_val = pd.DataFrame(coco_val[\"images\"])\nannotations_val = pd.DataFrame(coco_val[\"annotations\"])\ncategories_val = pd.DataFrame(coco_val[\"categories\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:08:41.680731Z","iopub.execute_input":"2025-10-31T02:08:41.681064Z","iopub.status.idle":"2025-10-31T02:08:41.694024Z","shell.execute_reply.started":"2025-10-31T02:08:41.681034Z","shell.execute_reply":"2025-10-31T02:08:41.692753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Val Images:\", images_val.shape)\nprint(\"Val Annotations:\", annotations_val.shape)\nprint(\"Val Categories:\", categories_val.shape, \"\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:08:41.695513Z","iopub.execute_input":"2025-10-31T02:08:41.695882Z","iopub.status.idle":"2025-10-31T02:08:41.715276Z","shell.execute_reply.started":"2025-10-31T02:08:41.695853Z","shell.execute_reply":"2025-10-31T02:08:41.714074Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. EDA","metadata":{}},{"cell_type":"markdown","source":"## 3-1. Distribution of Train / Validation / Test dataset","metadata":{}},{"cell_type":"code","source":"train_images = images_train['file_name'].tolist()\nvalid_images = images_val['file_name'].tolist()\n# test_images = images_test['file_name'].tolist() #if 'images_test' in locals() else []\n\nprint(f\"\\n=== Dataset Statistics ===\")\nprint(f\"Training images: {len(train_images)}\")\nprint(f\"Validation images: {len(valid_images)}\")\nprint(f\"Test images: {test_count}\")\n\ndataset_sizes = {\n    'train': len(train_images),\n    'valid': len(valid_images),\n    'test': test_count\n}\ndf_sizes = pd.DataFrame(list(dataset_sizes.items()), columns=['Dataset', 'Image Count'])\ntotal_images = df_sizes['Image Count'].sum()\ndf_sizes['Ratio (%)'] = (df_sizes['Image Count'] / total_images * 100).round(1)\n\nplt.figure(figsize=(5,4))\nbar_plot = sns.barplot(data=df_sizes, x='Dataset', y='Image Count', palette='coolwarm')\nplt.title('Dataset Split Distribution', fontsize=14)\nplt.xlabel('Dataset')\nplt.ylabel('Number of Images')\n\nfor idx, row in df_sizes.iterrows():\n    bar_plot.text(\n        idx, \n        row['Image Count'] + max(df_sizes['Image Count'])*0.001,\n        f\"{row['Ratio (%)']:.1f}%\", \n        ha='center', va='bottom', fontsize=10, fontweight='bold'\n    )\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:08:41.716422Z","iopub.execute_input":"2025-10-31T02:08:41.716778Z","iopub.status.idle":"2025-10-31T02:08:42.028330Z","shell.execute_reply.started":"2025-10-31T02:08:41.716752Z","shell.execute_reply":"2025-10-31T02:08:42.027147Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3-2. Statistics by patients per dataset","metadata":{}},{"cell_type":"code","source":"import os\nimport re\nimport pandas as pd\nfrom glob import glob\n\ndef parse_alphadent_filenames(image_dir, split_name):\n    \"\"\"\n    Parse AlphaDent-style filenames into structured metadata.\n    Example filename: p002_F_52_001.jpg\n    \"\"\"\n    records = []\n    img_files = sorted(glob(f\"{image_dir}/*.jpg\"))\n\n    pattern = re.compile(r\"p(\\d+)_([FM])_(\\d+)_([0-9]+)\\.jpg\", re.IGNORECASE)\n    for img_path in img_files:\n        fname = os.path.basename(img_path)\n        match = pattern.match(fname)\n        if match:\n            patient_id = int(match.group(1))\n            gender = match.group(2).upper()\n            value = int(match.group(3))\n            instance = int(match.group(4))\n            records.append({\n                \"split\": split_name,\n                \"file_name\": fname,\n                \"patient_id\": patient_id,\n                \"gender\": gender,\n                \"value\": value,\n                \"instance_id\": instance,\n                \"image_path\": img_path\n            })\n        else:\n            # 만약 패턴 안 맞는 파일이 있다면 로그로 남기기\n            print(f\"⚠️ Skipped (no match): {fname}\")\n\n    df = pd.DataFrame(records)\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:08:42.029624Z","iopub.execute_input":"2025-10-31T02:08:42.030062Z","iopub.status.idle":"2025-10-31T02:08:42.038318Z","shell.execute_reply.started":"2025-10-31T02:08:42.030029Z","shell.execute_reply":"2025-10-31T02:08:42.037055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_DIR = \"/kaggle/input/alpha-dent/AlphaDent/images/train\"\nVALID_DIR = \"/kaggle/input/alpha-dent/AlphaDent/images/valid\"\n\ndf_train_meta = parse_alphadent_filenames(TRAIN_DIR, \"train\")\ndf_valid_meta = parse_alphadent_filenames(VALID_DIR, \"valid\")\n\ndf_meta = pd.concat([df_train_meta, df_valid_meta], ignore_index=True)\nprint(f\"Total images parsed: {len(df_meta)}\")\ndf_meta.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:08:42.039572Z","iopub.execute_input":"2025-10-31T02:08:42.040032Z","iopub.status.idle":"2025-10-31T02:08:42.105936Z","shell.execute_reply.started":"2025-10-31T02:08:42.040003Z","shell.execute_reply":"2025-10-31T02:08:42.104808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# groupby by patient\npatient_summary = (\n    df_meta\n    .groupby([\"split\", \"patient_id\", \"gender\"])\n    .agg(num_images=(\"file_name\", \"count\"))   # 환자당 이미지 개수\n    .reset_index()\n)\n\n# Split별 통계 요약\nsplit_summary = (\n    patient_summary\n    .groupby(\"split\")\n    .agg(\n        num_patients=(\"patient_id\", \"nunique\"),        \n        num_females=(\"gender\", lambda x: (x == \"F\").sum()),\n        num_males=(\"gender\", lambda x: (x == \"M\").sum()),\n        avg_images_per_patient=(\"num_images\", \"mean\"), \n        # std_images_per_patient=(\"num_images\", \"std\"),  \n        min_images_per_patient=(\"num_images\", \"min\"),\n        max_images_per_patient=(\"num_images\", \"max\"),\n        total_images=(\"num_images\", \"sum\")\n    )\n    .reset_index()\n)\n\nsplit_summary[\"female_ratio (%)\"] = (split_summary[\"num_females\"] / split_summary[\"num_patients\"] * 100).round(1)\nsplit_summary[\"male_ratio (%)\"] = (split_summary[\"num_males\"] / split_summary[\"num_patients\"] * 100).round(1)\n\ndisplay(split_summary)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:08:42.107323Z","iopub.execute_input":"2025-10-31T02:08:42.107714Z","iopub.status.idle":"2025-10-31T02:08:42.164622Z","shell.execute_reply.started":"2025-10-31T02:08:42.107678Z","shell.execute_reply":"2025-10-31T02:08:42.163791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"unqiue_ids_train = set(df_train_meta['patient_id'].values)\nprint(f\"Unique patient IDs in train data: {unqiue_ids_train}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:08:42.165799Z","iopub.execute_input":"2025-10-31T02:08:42.166143Z","iopub.status.idle":"2025-10-31T02:08:42.172510Z","shell.execute_reply.started":"2025-10-31T02:08:42.166110Z","shell.execute_reply":"2025-10-31T02:08:42.171426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"unqiue_ids_valid = set(df_valid_meta['patient_id'].values)\nprint(f\"Unique patient IDs in validation data: {unqiue_ids_valid}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:08:42.173553Z","iopub.execute_input":"2025-10-31T02:08:42.173892Z","iopub.status.idle":"2025-10-31T02:08:42.193858Z","shell.execute_reply.started":"2025-10-31T02:08:42.173867Z","shell.execute_reply":"2025-10-31T02:08:42.192881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train data and validation data does not share patient\n# Note that the distribution of train might not match that of validation.\nunqiue_ids_train & unqiue_ids_valid","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:08:42.195399Z","iopub.execute_input":"2025-10-31T02:08:42.195871Z","iopub.status.idle":"2025-10-31T02:08:42.222580Z","shell.execute_reply.started":"2025-10-31T02:08:42.195845Z","shell.execute_reply":"2025-10-31T02:08:42.221618Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3-3. Distribution of image resolutions","metadata":{}},{"cell_type":"code","source":"def get_image_sizes(df_images):\n    widths, heights = df_images['width'].tolist(), df_images['height'].tolist()\n    return widths, heights\n\ntrain_w, train_h = get_image_sizes(images_train)\ndf_res = pd.DataFrame({'width': train_w, 'height': train_h})\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 5))\nsns.kdeplot(df_res['width'], fill=True, ax=axes[0], color='royalblue', alpha=0.6)\naxes[0].set_title('Distribution of Image Width (Train Set)', fontsize=13)\nsns.kdeplot(df_res['height'], fill=True, ax=axes[1], color='darkorange', alpha=0.6)\naxes[1].set_title('Distribution of Image Height (Train Set)', fontsize=13)\nplt.tight_layout()\nplt.show()\n\nsns.jointplot(data=df_res, x='width', y='height', kind='hex', color='teal')\nplt.suptitle('Image Width vs Height', y=1.05)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:08:42.223599Z","iopub.execute_input":"2025-10-31T02:08:42.223996Z","iopub.status.idle":"2025-10-31T02:08:43.411111Z","shell.execute_reply.started":"2025-10-31T02:08:42.223972Z","shell.execute_reply":"2025-10-31T02:08:43.410026Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3-4. Number of instances per image","metadata":{}},{"cell_type":"code","source":"train_counts = annotations_train.groupby('image_id')['id'].count().values\nval_counts = annotations_val.groupby('image_id')['id'].count().values\n\ndf_instance = pd.DataFrame({\n    'split': ['train'] * len(train_counts) + ['valid'] * len(val_counts),\n    'instances': list(train_counts) + list(val_counts)\n})\n\nplt.figure(figsize=(7,5))\nsns.boxplot(data=df_instance, x='split', y='instances', palette='Set2')\nplt.title('Number of Instances per Image')\nplt.xlabel('Dataset Split')\nplt.ylabel('Instance Count per Image')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:08:43.415691Z","iopub.execute_input":"2025-10-31T02:08:43.416097Z","iopub.status.idle":"2025-10-31T02:08:43.611854Z","shell.execute_reply.started":"2025-10-31T02:08:43.416071Z","shell.execute_reply":"2025-10-31T02:08:43.610695Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3-5. Visualize images & masks","metadata":{}},{"cell_type":"code","source":"import os, cv2, random\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as mpatches\nfrom PIL import Image\n\ndef visualize_random_samples_coco(images_df, annotations_df, n_rows=1, n_cols=5, seed=42):\n    random.seed(seed)\n    sample_imgs = random.sample(images_df['id'].tolist(), n_rows * n_cols)\n\n    COLORS = {\n        0: (31, 119, 180),\n        1: (255, 127, 14),\n        2: (214, 39, 40),\n        3: (44, 160, 44),\n        4: (148, 103, 189),\n        5: (140, 86, 75),\n        6: (227, 119, 194),\n        7: (127, 127, 127),\n        8: (188, 189, 34)\n    }\n\n    fig, axes = plt.subplots(n_rows, n_cols * 2, figsize=(n_cols * 5, n_rows * 3.5),\n                             gridspec_kw={'wspace': 0.25, 'hspace': 0.05})\n    fig.suptitle(\"Original (Left) vs Segmentation Mask (Right)\", fontsize=15, y=0.98)\n\n    for i, img_id in enumerate(sample_imgs):\n        img_info = images_df[images_df['id'] == img_id].iloc[0]\n        file_name = img_info['file_name']\n\n        # ✅ 1️⃣ split 감지 (train/valid/test)\n        if 'valid' in file_name or 'val' in file_name:\n            base_dir = \"/kaggle/input/alpha-dent/AlphaDent/images/valid\"\n        elif 'test' in file_name:\n            base_dir = \"/kaggle/input/alpha-dent/AlphaDent/images/test\"\n        else:\n            base_dir = \"/kaggle/input/alpha-dent/AlphaDent/images/train\"\n\n        img_path = os.path.join(base_dir, file_name.split('/')[-1])\n        if not os.path.exists(img_path):\n            print(f\"⚠️ Missing image: {img_path}\")\n            continue\n\n        # ✅ 2️⃣ 이미지 읽기 (OpenCV → PIL fallback)\n        img = cv2.imread(img_path)\n        if img is None:\n            try:\n                img = np.array(Image.open(img_path).convert(\"RGB\"))\n            except Exception as e:\n                print(f\"❌ Failed to open {img_path}: {e}\")\n                continue\n        else:\n            img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n        overlay = img.copy()\n\n        # ✅ 3️⃣ segmentation mask 그리기\n        annots = annotations_df[annotations_df['image_id'] == img_id]\n        for _, row in annots.iterrows():\n            seg = np.array(row['segmentation'][0]).reshape(-1, 2).astype(int)\n            cls = row['category_id'] - 1\n            color = COLORS.get(cls, (255,255,255))\n            cv2.polylines(overlay, [seg], True, color, 2)\n            cv2.fillPoly(overlay, [seg], color)\n            cv2.putText(overlay, CLASS_INFO[cls]['name'],\n                        (seg[0][0], seg[0][1]-5),\n                        cv2.FONT_HERSHEY_SIMPLEX, 0.5, (255,255,255), 1)\n\n        r, c = divmod(i, n_cols)\n        axes[r, c*2].imshow(img)\n        axes[r, c*2].axis('off')\n        axes[r, c*2+1].imshow(overlay)\n        axes[r, c*2+1].axis('off')\n\n    legend_patches = [mpatches.Patch(color=np.array(rgb)/255.0, label=CLASS_INFO[i]['name'])\n                      for i, rgb in COLORS.items()]\n    fig.legend(handles=legend_patches, loc='upper right', bbox_to_anchor=(1.12, 0.96),\n               title=\"Class Legend\", fontsize=9, title_fontsize=10)\n    plt.subplots_adjust(right=0.88, top=0.92)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:08:43.613006Z","iopub.execute_input":"2025-10-31T02:08:43.613265Z","iopub.status.idle":"2025-10-31T02:08:43.632463Z","shell.execute_reply.started":"2025-10-31T02:08:43.613246Z","shell.execute_reply":"2025-10-31T02:08:43.630957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"visualize_random_samples_coco(images_train, annotations_train, n_rows=3, n_cols=2, seed=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:08:43.634064Z","iopub.execute_input":"2025-10-31T02:08:43.634494Z","iopub.status.idle":"2025-10-31T02:09:02.072222Z","shell.execute_reply.started":"2025-10-31T02:08:43.634460Z","shell.execute_reply":"2025-10-31T02:09:02.071136Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3-6. Distribution of instances (class imbalance)","metadata":{}},{"cell_type":"code","source":"df_class = annotations_train.copy()\ndf_class['class_id'] = df_class['category_id'] - 1\ndf_class['class_name'] = df_class['class_id'].map(lambda i: CLASS_INFO[i]['name'])\nclass_counts = df_class['class_name'].value_counts().reset_index()\nclass_counts.columns = ['class_name', 'count']\nclass_counts['ratio'] = (class_counts['count'] / class_counts['count'].sum()) * 100\n\nplt.figure(figsize=(9,5))\nbar_plot = sns.barplot(data=class_counts, x='class_name', y='count', palette='viridis')\nplt.xticks(rotation=45, ha='right')\nplt.title('Training Class Distribution (with % Ratio)')\nplt.ylabel('Annotation Count')\nplt.xlabel('Class')\n\nfor idx, row in class_counts.iterrows():\n    bar_plot.text(idx, row['count'] + max(class_counts['count'])*0.01,\n                  f\"{row['ratio']:.1f}%\", color='black', ha='center', va='bottom',\n                  fontsize=9, fontweight='bold')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:09:02.073386Z","iopub.execute_input":"2025-10-31T02:09:02.073972Z","iopub.status.idle":"2025-10-31T02:09:02.371987Z","shell.execute_reply.started":"2025-10-31T02:09:02.073945Z","shell.execute_reply":"2025-10-31T02:09:02.370966Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3-7. Instance-level feature extraction","metadata":{}},{"cell_type":"code","source":"# WARNING: This code block takes approximately 17 minutes.\n# To save the hassle, set this NEWRUN = False\n# Otherwise, set NEWRUN = True.\n\nNEWRUN = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:09:02.373177Z","iopub.execute_input":"2025-10-31T02:09:02.373533Z","iopub.status.idle":"2025-10-31T02:09:02.378735Z","shell.execute_reply.started":"2025-10-31T02:09:02.373509Z","shell.execute_reply":"2025-10-31T02:09:02.377554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_instance_features_coco(images_df, annotations_df, class_info):\n    records = []\n    for _, ann in tqdm(annotations_df.iterrows(), total=len(annotations_df)):\n        img_info = images_df[images_df['id'] == ann['image_id']].iloc[0]\n        img_path = f\"/kaggle/input/alpha-dent/AlphaDent/images/valid/{img_info['file_name']}\"\n        img = np.array(Image.open(img_path).convert('RGB'))\n        h, w = img.shape[:2]\n\n        pts = np.array(ann['segmentation'][0]).reshape(-1, 2).astype(np.int32)\n        area = cv2.contourArea(pts)\n        if area <= 0:\n            continue\n        x, y, bw, bh = cv2.boundingRect(pts)\n        aspect_ratio = bw / (bh + 1e-6)\n\n        M = cv2.moments(pts)\n        cx = int(M[\"m10\"]/M[\"m00\"]) if M[\"m00\"] else x + bw//2\n        cy = int(M[\"m01\"]/M[\"m00\"]) if M[\"m00\"] else y + bh//2\n\n        mask = np.zeros((h, w), dtype=np.uint8)\n        cv2.fillPoly(mask, [pts], 255)\n        masked_pixels = img[mask == 255]\n        mean_r, mean_g, mean_b = masked_pixels.mean(axis=0)\n        std_r, std_g, std_b = masked_pixels.std(axis=0)\n        cls = ann['category_id'] - 1\n\n        records.append({\n            'filename': img_info['file_name'],\n            'class_id': cls,\n            'class_name': class_info[cls]['name'],\n            'area_px': area,\n            'aspect_ratio': aspect_ratio,\n            'centroid_x': cx,\n            'centroid_y': cy,\n            'mean_R': mean_r, 'mean_G': mean_g, 'mean_B': mean_b,\n            'std_R': std_r, 'std_G': std_g, 'std_B': std_b\n        })\n    return pd.DataFrame(records)\nif NEWRUN:\n    df_instances = extract_instance_features_coco(images_val, annotations_val, CLASS_INFO)\nelse:\n    df_instances = pd.read_csv('/kaggle/input/alphadentreport2/instance_report.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:11:36.519796Z","iopub.execute_input":"2025-10-31T02:11:36.520090Z","iopub.status.idle":"2025-10-31T02:11:36.586901Z","shell.execute_reply.started":"2025-10-31T02:11:36.520070Z","shell.execute_reply":"2025-10-31T02:11:36.584734Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3.8. Average polygon area & ratio distribution per class","metadata":{}},{"cell_type":"code","source":"# Polygon area\nplt.figure(figsize=(8,4))\nsns.barplot(data=df_instances, x='class_name', y='area_px', palette='viridis', estimator=np.mean)\nplt.title('Average Polygon Area per Class')\nplt.xticks(rotation=45, ha='right')\nplt.ylabel('Mean Area (pixels²)')\nplt.show()\n\nplt.figure(figsize=(8,4))\nsns.boxplot(data=df_instances, x='class_name', y='aspect_ratio', palette='coolwarm')\nplt.title('Aspect Ratio Distribution per Class')\nplt.xticks(rotation=45, ha='right')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:11:38.004986Z","iopub.execute_input":"2025-10-31T02:11:38.005356Z","iopub.status.idle":"2025-10-31T02:11:38.812387Z","shell.execute_reply.started":"2025-10-31T02:11:38.005333Z","shell.execute_reply":"2025-10-31T02:11:38.810847Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3.9. Average color (RGB) distribution per class","metadata":{}},{"cell_type":"code","source":"# Brightness distribution of each class\ndf_instances['brightness'] = (df_instances['mean_R'] + df_instances['mean_G'] + df_instances['mean_B']) / 3\n\nplt.figure(figsize=(8,4))\nsns.boxplot(data=df_instances, x='class_name', y='brightness', palette='magma')\nplt.title('Brightness Distribution per Class')\nplt.xticks(rotation=45, ha='right')\nplt.ylabel('Mean Brightness')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:11:41.922154Z","iopub.execute_input":"2025-10-31T02:11:41.922473Z","iopub.status.idle":"2025-10-31T02:11:42.195072Z","shell.execute_reply.started":"2025-10-31T02:11:41.922454Z","shell.execute_reply":"2025-10-31T02:11:42.193971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Average RGB per class\ndf_rgb = df_instances.groupby('class_name')[['mean_R','mean_G','mean_B']].mean().reset_index()\ndf_rgb_melted = df_rgb.melt(id_vars='class_name', var_name='Channel', value_name='Mean Value')\n\nplt.figure(figsize=(8,4))\nsns.barplot(data=df_rgb_melted, x='class_name', y='Mean Value', hue='Channel', palette='Set2')\nplt.title('Average RGB Channel per Class')\nplt.xticks(rotation=45, ha='right')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:13:52.350771Z","iopub.execute_input":"2025-10-31T02:13:52.351176Z","iopub.status.idle":"2025-10-31T02:13:52.723621Z","shell.execute_reply.started":"2025-10-31T02:13:52.351149Z","shell.execute_reply":"2025-10-31T02:13:52.722075Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3.10. Average centroid distribution per class","metadata":{}},{"cell_type":"code","source":"# Centroid distance map across classes\nplt.figure(figsize=(6,5))\nsns.kdeplot(\n    x=df_instances['centroid_x'], \n    y=df_instances['centroid_y'], \n    fill=True, cmap='viridis', bw_adjust=0.5\n)\nplt.gca().invert_yaxis()  # 이미지 좌표 기준으로 뒤집기\nplt.title('Instance Centroid Density Map')\nplt.xlabel('X')\nplt.ylabel('Y')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:13:55.061798Z","iopub.execute_input":"2025-10-31T02:13:55.062112Z","iopub.status.idle":"2025-10-31T02:14:05.967339Z","shell.execute_reply.started":"2025-10-31T02:13:55.062092Z","shell.execute_reply":"2025-10-31T02:14:05.966082Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the mean centroid of each class\ncentroid_means = (\n    df_instances.groupby('class_name')[['centroid_x', 'centroid_y']]\n    .mean()\n    .reset_index()\n)\n\nplt.figure(figsize=(6, 6))\nplt.scatter(\n    centroid_means['centroid_x'],\n    centroid_means['centroid_y'],\n    s=120,\n    c='red',\n    edgecolors='black'\n)\nplt.gca().invert_yaxis()\nfor i, row in centroid_means.iterrows():\n    plt.text(\n        row['centroid_x'] + 5,\n        row['centroid_y'],\n        row['class_name'],\n        fontsize=9,\n        va='center'\n    )\n\nplt.title('Average Centroid Position per Class')\nplt.xlabel('X')\nplt.ylabel('Y')\nplt.grid(alpha=0.3)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:14:05.969148Z","iopub.execute_input":"2025-10-31T02:14:05.969487Z","iopub.status.idle":"2025-10-31T02:14:06.232221Z","shell.execute_reply.started":"2025-10-31T02:14:05.969464Z","shell.execute_reply":"2025-10-31T02:14:06.231046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import math\n\nunique_classes = df_instances['class_name'].unique()\nn_classes = len(unique_classes)\n\nn_cols = 3\nn_rows = math.ceil(n_classes / n_cols)\n\nfig, axes = plt.subplots(n_rows, n_cols, figsize=(12, n_rows * 3))\naxes = axes.flatten()\n\nfor i, cls in enumerate(unique_classes):\n    ax = axes[i]\n    subset = df_instances[df_instances['class_name'] == cls]\n    sns.kdeplot(\n        x=subset['centroid_x'],\n        y=subset['centroid_y'],\n        fill=True,\n        cmap='viridis',\n        ax=ax,\n        bw_adjust=0.5\n    )\n    ax.set_title(cls)\n    ax.invert_yaxis()\n    ax.set_xlabel('X')\n    ax.set_ylabel('Y')\n\n# 남는 subplot 숨기기\nfor j in range(i + 1, len(axes)):\n    axes[j].axis('off')\n\nplt.suptitle('Centroid Density per Class', fontsize=15, y=1.02)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:14:06.233614Z","iopub.execute_input":"2025-10-31T02:14:06.234078Z","iopub.status.idle":"2025-10-31T02:14:17.586774Z","shell.execute_reply.started":"2025-10-31T02:14:06.234051Z","shell.execute_reply":"2025-10-31T02:14:17.585409Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3.11. Distribution of object size category per class","metadata":{}},{"cell_type":"code","source":"NEWRUN = False\n\nif NEWRUN: \n    df_instances = extract_instance_features(TRAIN_IMAGES_PATH, TRAIN_LABELS_PATH, CLASS_INFO)\n    print(f\"Extracted {len(df_instances)} instances\")\n    df_instances.to_csv('/kaggle/working/instance_report.csv', index = False)\n\nelse: \n    df_instances = pd.read_csv('/kaggle/input/alphadentreport2/instance_report.csv')\n\ndf_instances.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:14:17.588686Z","iopub.execute_input":"2025-10-31T02:14:17.589009Z","iopub.status.idle":"2025-10-31T02:14:17.644741Z","shell.execute_reply.started":"2025-10-31T02:14:17.588986Z","shell.execute_reply":"2025-10-31T02:14:17.643580Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_instances['rel_area'] = df_instances['area_px'] / (df_instances['img_width'] * df_instances['img_height'])\ndf_instances['size_category'] = pd.cut(\n    df_instances['rel_area'],\n    bins=[0, 0.001, 0.01, 0.1, 1],\n    labels=['tiny', 'small', 'medium', 'large']\n)\nsns.countplot(data=df_instances, x='size_category', hue='class_name', palette='tab20')\nplt.title('Object Size Category Distribution')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:14:17.645811Z","iopub.execute_input":"2025-10-31T02:14:17.646092Z","iopub.status.idle":"2025-10-31T02:14:18.011327Z","shell.execute_reply.started":"2025-10-31T02:14:17.646070Z","shell.execute_reply":"2025-10-31T02:14:18.009815Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3.12. Identify filenames of minor classes","metadata":{}},{"cell_type":"code","source":"# Unique filenames of Carries Class 4\nclass_4_names = list(set(df_instances[df_instances['class_name'] == 'Caries Class 4']['filename'].values))\nprint(len(class_4_names))\nclass_4_names[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T03:28:31.837209Z","iopub.execute_input":"2025-10-31T03:28:31.837567Z","iopub.status.idle":"2025-10-31T03:28:31.851563Z","shell.execute_reply.started":"2025-10-31T03:28:31.837539Z","shell.execute_reply":"2025-10-31T03:28:31.850376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Unique filenames of Carries Class 6\nclass_6_names = list(set(df_instances[df_instances['class_name'] == 'Caries Class 6']['filename'].values))\nprint(len(class_6_names))\nclass_6_names[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T03:28:23.148570Z","iopub.execute_input":"2025-10-31T03:28:23.148945Z","iopub.status.idle":"2025-10-31T03:28:23.160882Z","shell.execute_reply.started":"2025-10-31T03:28:23.148922Z","shell.execute_reply":"2025-10-31T03:28:23.159556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Unique filenames of Carries Class 3\nclass_3_names = list(set(df_instances[df_instances['class_name'] == 'Caries Class 3']['filename'].values))\nprint(len(class_3_names))\nclass_3_names[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T03:28:27.144423Z","iopub.execute_input":"2025-10-31T03:28:27.144800Z","iopub.status.idle":"2025-10-31T03:28:27.157349Z","shell.execute_reply.started":"2025-10-31T03:28:27.144775Z","shell.execute_reply":"2025-10-31T03:28:27.156246Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Unique filenames of Carries Crown\ncrown_names = list(set(df_instances[df_instances['class_name'] == 'Crown']['filename'].values))\nprint(len(crown_names))\ncrown_names[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T03:28:13.917989Z","iopub.execute_input":"2025-10-31T03:28:13.918334Z","iopub.status.idle":"2025-10-31T03:28:13.931713Z","shell.execute_reply.started":"2025-10-31T03:28:13.918309Z","shell.execute_reply":"2025-10-31T03:28:13.930716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# id (images_train) ↔ image_id (annotations_train) 기준으로 조인\nmerged_train = annotations_train.merge(\n    images_train,\n    left_on=\"image_id\",\n    right_on=\"id\",\n    suffixes=(\"_ann\", \"_img\")\n)\nmerged_train.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T03:04:34.412144Z","iopub.execute_input":"2025-10-31T03:04:34.412526Z","iopub.status.idle":"2025-10-31T03:04:34.450936Z","shell.execute_reply.started":"2025-10-31T03:04:34.412500Z","shell.execute_reply":"2025-10-31T03:04:34.448321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_masked_image(\n    df,\n    img_filename,\n    base_path=\"/kaggle/input/alpha-dent/AlphaDent/images/train\",\n    class_info=None  \n):\n    image_path = os.path.join(base_path, img_filename)\n    image = cv2.imread(image_path)\n    if image is None:\n        raise FileNotFoundError(f\"⚠️ 이미지 파일을 찾을 수 없습니다: {image_path}\")\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n    subset = df[df[\"file_name\"] == img_filename]\n    if len(subset) == 0:\n        print(f\"⚠️ {img_filename} 에 해당하는 annotation이 없습니다.\")\n        return\n\n    overlay = image.copy()\n    legend_handles = []\n    color_map = {}\n\n    for _, row in subset.iterrows():\n        cat_id = int(row[\"category_id\"])\n\n        # ✅ 1-based → 0-based 변환\n        adjusted_id = cat_id - 1\n\n        # ✅ CLASS_INFO 매핑\n        if class_info is not None and adjusted_id in class_info:\n            class_name = class_info[adjusted_id]['name']\n        else:\n            class_name = f\"Class {cat_id}\"\n\n        # 색상 생성\n        if class_name not in color_map:\n            color_map[class_name] = tuple(np.random.randint(0, 255, 3).tolist())\n        color = color_map[class_name]\n\n        for seg in row[\"segmentation\"]:\n            pts = np.array(seg, np.int32).reshape((-1, 1, 2))\n            cv2.fillPoly(overlay, [pts], color)\n            cv2.polylines(overlay, [pts], isClosed=True, color=(0,0,0), thickness=2)\n\n        legend_handles.append(Patch(facecolor=np.array(color)/255.0, label=class_name))\n\n    # 시각화\n    fig, axes = plt.subplots(1, 2, figsize=(14, 7))\n    fig.suptitle(f\"Visualization for {img_filename}\", fontsize=16)\n\n    axes[0].imshow(image)\n    axes[0].set_title(\"① Original Image\")\n    axes[0].axis(\"off\")\n\n    axes[1].imshow(cv2.addWeighted(image, 0.5, overlay, 0.5, 0))\n    axes[1].set_title(\"② Masked Overlay + Class Legend\")\n    axes[1].axis(\"off\")\n\n    unique_handles = {h.get_label(): h for h in legend_handles}.values()\n    plt.legend(handles=unique_handles, bbox_to_anchor=(1.05, 1), loc=\"upper left\")\n\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:56:11.474427Z","iopub.execute_input":"2025-10-31T02:56:11.475058Z","iopub.status.idle":"2025-10-31T02:56:11.497966Z","shell.execute_reply.started":"2025-10-31T02:56:11.475020Z","shell.execute_reply":"2025-10-31T02:56:11.496847Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 3.12.1. Visualize Caries Class 6","metadata":{}},{"cell_type":"code","source":"visualize_masked_image(\n    df=merged_train,\n    img_filename=class_6_names[0],\n    base_path=\"/kaggle/input/alpha-dent/AlphaDent/images/train\",\n    class_info=CLASS_INFO  # ✅ CLASS_INFO 전달\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:56:35.616003Z","iopub.execute_input":"2025-10-31T02:56:35.616390Z","iopub.status.idle":"2025-10-31T02:56:41.195436Z","shell.execute_reply.started":"2025-10-31T02:56:35.616360Z","shell.execute_reply":"2025-10-31T02:56:41.194378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"visualize_masked_image(\n    df=merged_train,\n    img_filename=class_6_names[1],\n    base_path=\"/kaggle/input/alpha-dent/AlphaDent/images/train\",\n    class_info=CLASS_INFO  # ✅ CLASS_INFO 전달\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:56:52.470490Z","iopub.execute_input":"2025-10-31T02:56:52.470841Z","iopub.status.idle":"2025-10-31T02:56:57.707846Z","shell.execute_reply.started":"2025-10-31T02:56:52.470816Z","shell.execute_reply":"2025-10-31T02:56:57.706075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"visualize_masked_image(\n    df=merged_train,\n    img_filename=class_6_names[2],\n    base_path=\"/kaggle/input/alpha-dent/AlphaDent/images/train\",\n    class_info=CLASS_INFO\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:57:12.478340Z","iopub.execute_input":"2025-10-31T02:57:12.478725Z","iopub.status.idle":"2025-10-31T02:57:17.579408Z","shell.execute_reply.started":"2025-10-31T02:57:12.478693Z","shell.execute_reply":"2025-10-31T02:57:17.578050Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"visualize_masked_image(\n    df=merged_train,\n    img_filename=class_6_names[3],\n    base_path=\"/kaggle/input/alpha-dent/AlphaDent/images/train\",\n    class_info=CLASS_INFO\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T02:58:10.686918Z","iopub.execute_input":"2025-10-31T02:58:10.687380Z","iopub.status.idle":"2025-10-31T02:58:16.255754Z","shell.execute_reply.started":"2025-10-31T02:58:10.687344Z","shell.execute_reply":"2025-10-31T02:58:16.254017Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 3.12.1. Visualize Caries Class 4","metadata":{}},{"cell_type":"code","source":"visualize_masked_image(\n    df=merged_train,\n    img_filename=class_4_names[0],\n    base_path=\"/kaggle/input/alpha-dent/AlphaDent/images/train\",\n    class_info=CLASS_INFO \n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T03:01:32.822546Z","iopub.execute_input":"2025-10-31T03:01:32.823557Z","iopub.status.idle":"2025-10-31T03:01:38.130489Z","shell.execute_reply.started":"2025-10-31T03:01:32.823519Z","shell.execute_reply":"2025-10-31T03:01:38.129123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"visualize_masked_image(\n    df=merged_train,\n    img_filename=class_4_names[1],\n    base_path=\"/kaggle/input/alpha-dent/AlphaDent/images/train\",\n    class_info=CLASS_INFO \n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T03:01:52.647122Z","iopub.execute_input":"2025-10-31T03:01:52.647466Z","iopub.status.idle":"2025-10-31T03:01:58.249078Z","shell.execute_reply.started":"2025-10-31T03:01:52.647436Z","shell.execute_reply":"2025-10-31T03:01:58.247860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"visualize_masked_image(\n    df=merged_train,\n    img_filename=class_4_names[2],\n    base_path=\"/kaggle/input/alpha-dent/AlphaDent/images/train\",\n    class_info=CLASS_INFO \n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T03:03:46.172605Z","iopub.execute_input":"2025-10-31T03:03:46.173295Z","iopub.status.idle":"2025-10-31T03:03:51.344921Z","shell.execute_reply.started":"2025-10-31T03:03:46.173263Z","shell.execute_reply":"2025-10-31T03:03:51.343710Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 3.12.3. Visualize Caries Class 3","metadata":{}},{"cell_type":"code","source":"visualize_masked_image(\n    df=merged_train,\n    img_filename=class_3_names[0],\n    base_path=\"/kaggle/input/alpha-dent/AlphaDent/images/train\",\n    class_info=CLASS_INFO \n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T03:28:47.170343Z","iopub.execute_input":"2025-10-31T03:28:47.170711Z","iopub.status.idle":"2025-10-31T03:28:52.455680Z","shell.execute_reply.started":"2025-10-31T03:28:47.170641Z","shell.execute_reply":"2025-10-31T03:28:52.454618Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"visualize_masked_image(\n    df=merged_train,\n    img_filename=class_3_names[1],\n    base_path=\"/kaggle/input/alpha-dent/AlphaDent/images/train\",\n    class_info=CLASS_INFO \n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T03:29:15.001485Z","iopub.execute_input":"2025-10-31T03:29:15.001845Z","iopub.status.idle":"2025-10-31T03:29:20.636757Z","shell.execute_reply.started":"2025-10-31T03:29:15.001819Z","shell.execute_reply":"2025-10-31T03:29:20.635613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"visualize_masked_image(\n    df=merged_train,\n    img_filename=class_3_names[2],\n    base_path=\"/kaggle/input/alpha-dent/AlphaDent/images/train\",\n    class_info=CLASS_INFO \n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T03:30:23.919383Z","iopub.execute_input":"2025-10-31T03:30:23.919808Z","iopub.status.idle":"2025-10-31T03:30:29.334151Z","shell.execute_reply.started":"2025-10-31T03:30:23.919783Z","shell.execute_reply":"2025-10-31T03:30:29.332988Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 3.12.3. Visualize Crown","metadata":{}},{"cell_type":"code","source":"visualize_masked_image(\n    df=merged_train,\n    img_filename=crown_names[0],\n    base_path=\"/kaggle/input/alpha-dent/AlphaDent/images/train\",\n    class_info=CLASS_INFO \n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T03:31:08.268801Z","iopub.execute_input":"2025-10-31T03:31:08.269122Z","iopub.status.idle":"2025-10-31T03:31:13.655416Z","shell.execute_reply.started":"2025-10-31T03:31:08.269100Z","shell.execute_reply":"2025-10-31T03:31:13.654340Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"visualize_masked_image(\n    df=merged_train,\n    img_filename=crown_names[1],\n    base_path=\"/kaggle/input/alpha-dent/AlphaDent/images/train\",\n    class_info=CLASS_INFO \n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T03:31:30.236153Z","iopub.execute_input":"2025-10-31T03:31:30.236514Z","iopub.status.idle":"2025-10-31T03:31:35.252027Z","shell.execute_reply.started":"2025-10-31T03:31:30.236489Z","shell.execute_reply":"2025-10-31T03:31:35.250366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"visualize_masked_image(\n    df=merged_train,\n    img_filename=crown_names[2],\n    base_path=\"/kaggle/input/alpha-dent/AlphaDent/images/train\",\n    class_info=CLASS_INFO \n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-31T03:31:51.429267Z","iopub.execute_input":"2025-10-31T03:31:51.429581Z","iopub.status.idle":"2025-10-31T03:31:56.959510Z","shell.execute_reply.started":"2025-10-31T03:31:51.429560Z","shell.execute_reply":"2025-10-31T03:31:56.958079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}