{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":103103,"databundleVersionId":13042974,"isSourceIdPinned":false},{"sourceType":"modelInstanceVersion","sourceId":755725,"databundleVersionId":15751140,"modelInstanceId":577133,"modelId":589458,"isSourceIdPinned":false}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div align=\"center\">\n\n# 🦷 Dental Segmentation Using SAM2.1 Fine-Tuning\n\n<img src=\"https://github.com/facebookresearch/sam2/raw/main/assets/sa_v_dataset.jpg?raw=true\" width=\"650\"/>\n\n</div>\n\n---\n\n## Project Overview\n\nIn this project, the **SAM2.1 model** has been fine-tuned for dental image segmentation tasks.  \nThe model outputs are presented as reference results for researchers and practitioners who aim to explore similar fine-tuning approaches.\n\nThe fine-tuning process emphasizes the **testing pipeline**, while the training pipeline is also provided for completeness.  \nIt is important to note that the actual training process was conducted on a more powerful hardware environment.\n\n---\n\n\n","metadata":{}},{"cell_type":"markdown","source":"\n## Data Analysis\n\n- The model was fine-tuned on a dental segmentation dataset.\n- The training process was conducted for **40 epochs**.\n- Training performance was monitored and recorded using log files.\n- Both training and testing codes are included in the project.\n","metadata":{}},{"cell_type":"code","source":"import os\nimport random\nimport cv2\nimport matplotlib.pyplot as plt\nimport numpy as np","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:14:36.414470Z","iopub.execute_input":"2026-02-19T10:14:36.415188Z","iopub.status.idle":"2026-02-19T10:14:36.418890Z","shell.execute_reply.started":"2026-02-19T10:14:36.415140Z","shell.execute_reply":"2026-02-19T10:14:36.418285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMAGE_DIR = \"/kaggle/input/competitions/alpha-dent/AlphaDent/images/train\"  # kendi path'ine göre güncelle\nIMG_SIZE = 64\nROWS, COLS = 5, 5\nNUM_IMAGES = ROWS * COLS\n\nimage_files = [\n    os.path.join(IMAGE_DIR, f)\n    for f in os.listdir(IMAGE_DIR)\n    if f.lower().endswith((\".jpg\", \".jpeg\", \".png\"))\n]\n\nselected_images = random.sample(image_files, NUM_IMAGES)\n\nfig, axes = plt.subplots(ROWS, COLS, figsize=(8, 8))\n\nfor ax, img_path in zip(axes.flatten(), selected_images):\n    img = cv2.imread(img_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    img = cv2.resize(img, (IMG_SIZE, IMG_SIZE))\n    \n    ax.imshow(img)\n    ax.axis(\"off\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:01:46.885585Z","iopub.execute_input":"2026-02-19T10:01:46.886591Z","iopub.status.idle":"2026-02-19T10:01:51.718502Z","shell.execute_reply.started":"2026-02-19T10:01:46.886555Z","shell.execute_reply":"2026-02-19T10:01:51.717582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMAGE_DIR = \"/kaggle/input/competitions/alpha-dent/AlphaDent/images/train\"\nLABEL_DIR = \"/kaggle/input/competitions/alpha-dent/AlphaDent/labels/train\"\n\nIMG_SIZE = 128\nROWS, COLS = 5, 5\nNUM_IMAGES = ROWS * COLS\nALPHA = 0.5 \n\nimage_files = [\n    f for f in os.listdir(IMAGE_DIR)\n    if f.lower().endswith((\".jpg\", \".jpeg\", \".png\"))\n]\n\nselected_images = random.sample(image_files, NUM_IMAGES)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:14:15.985602Z","iopub.execute_input":"2026-02-19T10:14:15.985991Z","iopub.status.idle":"2026-02-19T10:14:16.010463Z","shell.execute_reply.started":"2026-02-19T10:14:15.985959Z","shell.execute_reply":"2026-02-19T10:14:16.009806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def seg_to_mask_and_classes(label_path, img_shape):\n    h, w = img_shape[:2]\n    mask = np.zeros((h, w, 3), dtype=np.uint8)\n    classes = set()\n\n    if not os.path.exists(label_path):\n        return mask, classes\n\n    with open(label_path, \"r\") as f:\n        lines = f.readlines()\n\n    for line in lines:\n        data = list(map(float, line.strip().split()))\n        class_id = int(data[0])\n        classes.add(class_id)\n\n        points = np.array(data[1:]).reshape(-1, 2)\n        points[:, 0] *= w\n        points[:, 1] *= h\n        points = points.astype(np.int32)\n\n        # random color per object\n        color = np.random.randint(0, 255, size=3).tolist()\n        cv2.fillPoly(mask, [points], color)\n\n    return mask, classes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:14:18.197014Z","iopub.execute_input":"2026-02-19T10:14:18.197335Z","iopub.status.idle":"2026-02-19T10:14:18.204533Z","shell.execute_reply.started":"2026-02-19T10:14:18.197308Z","shell.execute_reply":"2026-02-19T10:14:18.203764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(ROWS, COLS, figsize=(8, 8))\n\nfor ax, img_name in zip(axes.flatten(), selected_images):\n    img_path = os.path.join(IMAGE_DIR, img_name)\n    label_path = os.path.join(LABEL_DIR, img_name.rsplit(\".\", 1)[0] + \".txt\")\n\n    img = cv2.imread(img_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    overlay = img.copy()\n    mask, classes = seg_to_mask_and_classes(label_path, img.shape)\n\n    overlay = cv2.addWeighted(overlay, 1, mask, ALPHA, 0)\n    overlay = cv2.resize(overlay, (IMG_SIZE, IMG_SIZE))\n\n    # Class ID text\n    if classes:\n        class_text = \", \".join([f\"Class {c}\" for c in sorted(classes)])\n    else:\n        class_text = \"No Label\"\n\n    ax.imshow(overlay)\n    ax.set_title(class_text, fontsize=8)\n    ax.axis(\"off\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:15:26.711075Z","iopub.execute_input":"2026-02-19T10:15:26.711397Z","iopub.status.idle":"2026-02-19T10:15:31.602360Z","shell.execute_reply.started":"2026-02-19T10:15:26.711370Z","shell.execute_reply":"2026-02-19T10:15:31.601536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import Counter\nLABEL_DIR = \"/kaggle/input/competitions/alpha-dent/AlphaDent/labels/train\" \n\nclass_counter = Counter()\n\nlabel_files = [\n    f for f in os.listdir(LABEL_DIR)\n    if f.endswith(\".txt\")\n]\n\nfor label_file in label_files:\n    label_path = os.path.join(LABEL_DIR, label_file)\n\n    with open(label_path, \"r\") as f:\n        for line in f:\n            if line.strip() == \"\":\n                continue\n            class_id = int(line.split()[0])\n            class_counter[class_id] += 1\n\nclass_ids = sorted(class_counter.keys())\ncounts = [class_counter[c] for c in class_ids]\nlabels = [f\"Class {c}\" for c in class_ids]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:26:28.567909Z","iopub.execute_input":"2026-02-19T10:26:28.568522Z","iopub.status.idle":"2026-02-19T10:26:44.634407Z","shell.execute_reply.started":"2026-02-19T10:26:28.568489Z","shell.execute_reply":"2026-02-19T10:26:44.633800Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_counter","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:26:44.635625Z","iopub.execute_input":"2026-02-19T10:26:44.635917Z","iopub.status.idle":"2026-02-19T10:26:44.640942Z","shell.execute_reply.started":"2026-02-19T10:26:44.635893Z","shell.execute_reply":"2026-02-19T10:26:44.640275Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div align=\"center\">\n\n##  SAM2.1 Model Overview\n\n\n<img src=\"https://raw.githubusercontent.com/facebookresearch/sam2/refs/heads/main/assets/model_diagram.png\" width=\"700\"/>\n\n</div>\n\n---\n\nThe figure above illustrates the **SAM2.1 architecture**:\n\n1. **Image Encoder** – extracts high-level visual features from input images.  \n2. **Memory Attention** – allows the model to reference past frames or masks for consistent segmentation.  \n3. **Prompt Encoder** – encodes user inputs such as points, boxes, or masks to guide segmentation.  \n4. **Mask Decoder** – combines encoded image features and prompts to produce accurate masks.  \n5. **Memory Encoder & Memory Bank** – stores and manages representations from previous frames for temporal consistency.\n\nThis pipeline enables SAM2.1 to generate high-quality masks efficiently with minimal user input.\n\n</div>\n","metadata":{}},{"cell_type":"code","source":"!git clone https://github.com/facebookresearch/sam2.git","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:58:48.971889Z","iopub.execute_input":"2026-02-19T08:58:48.972215Z","iopub.status.idle":"2026-02-19T08:58:56.527895Z","shell.execute_reply.started":"2026-02-19T08:58:48.972188Z","shell.execute_reply":"2026-02-19T08:58:56.527006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!wget -O /kaggle/working/sam2/sam2/configs/train.yaml 'https://drive.usercontent.google.com/download?id=11cmbxPPsYqFyWq87tmLgBAQ6OZgEhPG3'\n\n# Downloading the train configuration file.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:58:56.529766Z","iopub.execute_input":"2026-02-19T08:58:56.530006Z","iopub.status.idle":"2026-02-19T08:58:57.476791Z","shell.execute_reply.started":"2026-02-19T08:58:56.529979Z","shell.execute_reply":"2026-02-19T08:58:57.475989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%cd /kaggle/working/sam2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:58:58.400449Z","iopub.execute_input":"2026-02-19T08:58:58.401088Z","iopub.status.idle":"2026-02-19T08:58:58.408091Z","shell.execute_reply.started":"2026-02-19T08:58:58.401049Z","shell.execute_reply":"2026-02-19T08:58:58.407354Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -e .[dev] -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T08:58:58.856945Z","iopub.execute_input":"2026-02-19T08:58:58.857296Z","iopub.status.idle":"2026-02-19T09:03:49.327732Z","shell.execute_reply.started":"2026-02-19T08:58:58.857270Z","shell.execute_reply":"2026-02-19T09:03:49.326697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!cd ./checkpoints && ./download_ckpts.sh\n\n# Downloading the model weights.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T09:03:49.329919Z","iopub.execute_input":"2026-02-19T09:03:49.330184Z","iopub.status.idle":"2026-02-19T09:04:22.211738Z","shell.execute_reply.started":"2026-02-19T09:03:49.330154Z","shell.execute_reply":"2026-02-19T09:04:22.210988Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training SAM2.1\n\nIn the training section, the downloaded configuration `.yaml` file needs some adjustments depending on the environment we are using. For example:\n\n```yaml\nscratch:\n  resolution: 1024\n  train_batch_size: 1\n  num_train_workers: 10\n  num_frames: 1\n  max_num_objects: 3\n  base_lr: 5.0e-6\n  vision_lr: 3.0e-06\n  phases_per_epoch: 1\n  num_epochs: 40\n\ndataset:\n  img_folder: /content/data/train\n  gt_folder: /content/data/train\n  multiplier: 2\n","metadata":{}},{"cell_type":"markdown","source":"These fields should be updated according to the location of our dataset and the GPU capacity.\n\nI performed the training in the Colab environment using an A100 GPU. With a batch size of 8 and resolution 640, the training used around 20 GB of GPU memory and completed successfully.\n\nThe trained model for testing is publicly available at:\n/kaggle/input/models/berkutayasan/sam-2-1/pytorch/default/1\n\nTherefore, no additional training was performed in the Kaggle environment.","metadata":{}},{"cell_type":"code","source":"!python training/train.py -c 'configs/train.yaml' --use-cluster 0 --num-gpus 1","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Results & Analysis\n\n- Model outputs and qualitative results are demonstrated.\n- Comparative insights regarding model size and performance are discussed.","metadata":{}},{"cell_type":"code","source":"import os\nimport shutil\n\nSOURCE_LOG = \"/kaggle/input/models/berkutayasan/sam-2-1/pytorch/default/1/tensorboard\"\n\nDEST_LOG = \"/kaggle/working/tb_logs\"\n\nos.makedirs(DEST_LOG, exist_ok=True)\n\nfor file in os.listdir(SOURCE_LOG):\n    shutil.copy(os.path.join(SOURCE_LOG, file), DEST_LOG)\n\nprint(\"Log dosyaları kopyalandı:\", DEST_LOG)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:26:55.314693Z","iopub.execute_input":"2026-02-19T10:26:55.315296Z","iopub.status.idle":"2026-02-19T10:26:55.325513Z","shell.execute_reply.started":"2026-02-19T10:26:55.315267Z","shell.execute_reply":"2026-02-19T10:26:55.324812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport matplotlib.pyplot as plt\nfrom tensorboard.backend.event_processing import event_accumulator","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:27:38.795044Z","iopub.execute_input":"2026-02-19T10:27:38.795906Z","iopub.status.idle":"2026-02-19T10:27:38.799438Z","shell.execute_reply.started":"2026-02-19T10:27:38.795872Z","shell.execute_reply":"2026-02-19T10:27:38.798731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"EVENT_PATH = \"/kaggle/working/tb_logs/events.out.tfevents.1771431011.d57f3371ce46.3848.0f5e8e2a1-b049-4f75-ac9f-2f3e5e540baa\"\n\nea = event_accumulator.EventAccumulator(EVENT_PATH)\nea.Reload()\n\nall_tags = ea.Tags()[\"tensors\"]\n\nimportant_tags = [t for t in all_tags if t.startswith(\"Losses/\")]\n\nlr_tag = [t for t in all_tags if \"Optim/0_/lr\" in t]\nif lr_tag:\n    important_tags.append(lr_tag[0])\n\nimportant_tags = important_tags[:8]\n\nfig, axes = plt.subplots(2, 4, figsize=(16, 8))\naxes = axes.flatten()\n\nfor i in range(8):\n    if i < len(important_tags):\n        tag = important_tags[i]\n        events = ea.Tensors(tag)\n\n        steps = []\n        values = []\n\n        for e in events:\n            steps.append(e.step)\n            val = tf.make_ndarray(e.tensor_proto)\n            values.append(val.item() if val.size == 1 else val.mean())\n\n        axes[i].plot(steps, values)\n        axes[i].set_title(tag.split(\"/\")[-1], fontsize=10)\n        axes[i].tick_params(axis='both', labelsize=8)\n        axes[i].grid(True)\n    else:\n        axes[i].axis(\"off\")\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T09:48:09.583159Z","iopub.execute_input":"2026-02-19T09:48:09.584244Z","iopub.status.idle":"2026-02-19T09:48:10.677750Z","shell.execute_reply.started":"2026-02-19T09:48:09.584210Z","shell.execute_reply":"2026-02-19T09:48:10.676852Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training Log Analysis (SAM 2.1, 40 Epochs)\n\n- **Total Loss:** Decreased from ~1.66 to ~1.33, showing stable convergence.  \n- **Mask & Dice Loss:** Gradual improvement, indicating better segmentation.  \n- **IoU Loss:** Dropped from ~0.44 to ~0.34, reflecting increased accuracy.  \n- **Class Loss:** Minimal contribution, stable around ~1e-7.  \n- **Learning Rate:** Gradually decayed from ~5e-6 to ~5e-7, supporting stable training.\n\n**Summary:** The model converged smoothly with consistent improvements in segmentation metrics.\n","metadata":{}},{"cell_type":"markdown","source":"## Test Set Outputs","metadata":{}},{"cell_type":"code","source":"!pip install supervision -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T09:04:59.708686Z","iopub.execute_input":"2026-02-19T09:04:59.709000Z","iopub.status.idle":"2026-02-19T09:05:04.013143Z","shell.execute_reply.started":"2026-02-19T09:04:59.708970Z","shell.execute_reply":"2026-02-19T09:05:04.012005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport os\nimport cv2\nimport numpy as np\nimport random\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport supervision as sv\n\nfrom sam2.build_sam import build_sam2\nfrom sam2.automatic_mask_generator import SAM2AutomaticMaskGenerator","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:34:06.608510Z","iopub.execute_input":"2026-02-19T10:34:06.609226Z","iopub.status.idle":"2026-02-19T10:34:06.613323Z","shell.execute_reply.started":"2026-02-19T10:34:06.609194Z","shell.execute_reply":"2026-02-19T10:34:06.612526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DEVICE = \"cuda\"\n\ntorch.backends.cuda.matmul.allow_tf32 = True\ntorch.backends.cudnn.allow_tf32 = True\n\ncheckpoint = \"/kaggle/input/models/berkutayasan/sam-2-1/pytorch/default/1/checkpoints/checkpoint.pt\"  # fine-tuned model\nmodel_cfg = \"configs/sam2.1/sam2.1_hiera_b+.yaml\"\nsam2 = build_sam2(model_cfg, checkpoint, device=DEVICE)\n\nmask_generator = SAM2AutomaticMaskGenerator(\n    sam2,\n    points_per_side=16,\n    crop_n_layers=0\n)\n\nTEST_PATH = \"/kaggle/input/competitions/alpha-dent/AlphaDent/images/test\"\ntest_images = [f for f in os.listdir(TEST_PATH) if f.endswith(\".jpg\")]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:34:13.507409Z","iopub.execute_input":"2026-02-19T10:34:13.508006Z","iopub.status.idle":"2026-02-19T10:34:15.049105Z","shell.execute_reply.started":"2026-02-19T10:34:13.507973Z","shell.execute_reply":"2026-02-19T10:34:15.048494Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_images = random.sample(test_images, 25)\n\nfig, axes = plt.subplots(5, 5, figsize=(20, 20))\naxes = axes.flatten()\n\nfor ax, image_name in zip(axes, selected_images):\n    image_path = os.path.join(TEST_PATH, image_name)\n    image = np.array(Image.open(image_path).convert(\"RGB\"))\n    image_resized = cv2.resize(image, (640, 640))\n\n    with torch.inference_mode(), torch.autocast(\"cuda\", dtype=torch.bfloat16):\n        result = mask_generator.generate(image_resized)\n\n    detections = sv.Detections.from_sam(sam_result=result)\n\n    mask_annotator = sv.MaskAnnotator(color_lookup=sv.ColorLookup.INDEX)\n    annotated_image = mask_annotator.annotate(\n        scene=image_resized.copy(),\n        detections=detections\n    )\n\n    ax.imshow(annotated_image)\n    ax.axis(\"off\")\n    ax.set_title(image_name, fontsize=8)\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:34:28.539673Z","iopub.execute_input":"2026-02-19T10:34:28.540363Z","iopub.status.idle":"2026-02-19T10:35:21.845224Z","shell.execute_reply.started":"2026-02-19T10:34:28.540333Z","shell.execute_reply":"2026-02-19T10:35:21.843490Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Validation Set Output Comparison","metadata":{}},{"cell_type":"code","source":"import torch\nimport os\nimport cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport supervision as sv\nimport random\n\nfrom sam2.build_sam import build_sam2\nfrom sam2.automatic_mask_generator import SAM2AutomaticMaskGenerator","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:38:29.703591Z","iopub.execute_input":"2026-02-19T10:38:29.704417Z","iopub.status.idle":"2026-02-19T10:38:29.708527Z","shell.execute_reply.started":"2026-02-19T10:38:29.704381Z","shell.execute_reply":"2026-02-19T10:38:29.707871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"VALID_IMAGE_PATH = \"/kaggle/input/competitions/alpha-dent/AlphaDent/images/valid\"\nVALID_MASK_PATH = \"/kaggle/input/competitions/alpha-dent/AlphaDent/labels/valid\"\n\nCHECKPOINT = \"/kaggle/input/models/berkutayasan/sam-2-1/pytorch/default/1/checkpoints/checkpoint.pt\"\nMODEL_CFG = \"configs/sam2.1/sam2.1_hiera_b+.yaml\"\n\nDEVICE = \"cuda\"\n\ntorch.backends.cuda.matmul.allow_tf32 = True\ntorch.backends.cudnn.allow_tf32 = True\n\nsam2 = build_sam2(MODEL_CFG, CHECKPOINT, device=DEVICE)\n\nmask_generator = SAM2AutomaticMaskGenerator(\n    sam2,\n    points_per_side=16,\n    crop_n_layers=0\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:38:32.019036Z","iopub.execute_input":"2026-02-19T10:38:32.019348Z","iopub.status.idle":"2026-02-19T10:38:33.520352Z","shell.execute_reply.started":"2026-02-19T10:38:32.019323Z","shell.execute_reply":"2026-02-19T10:38:33.519463Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def txt_to_colored_mask(txt_path, img_shape):\n    h, w = img_shape[:2]\n    colored_mask = np.zeros((h, w, 3), dtype=np.uint8)\n\n    if not os.path.exists(txt_path):\n        return colored_mask\n\n    with open(txt_path, \"r\") as f:\n        lines = f.readlines()\n\n    for line in lines:\n        parts = line.strip().split()\n        class_id = int(parts[0])\n        coords = np.array(parts[1:], dtype=float).reshape(-1, 2)\n\n        coords[:, 0] *= w\n        coords[:, 1] *= h\n        coords = coords.astype(np.int32)\n\n        random.seed(class_id)\n        color = [random.randint(60,255) for _ in range(3)]\n\n        cv2.fillPoly(colored_mask, [coords], color)\n\n    return colored_mask\n\nimage_files = sorted([f for f in os.listdir(VALID_IMAGE_PATH) if f.endswith(\".jpg\")])[:25]\n\nsam_outputs = []\ngt_outputs = []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:38:34.755088Z","iopub.execute_input":"2026-02-19T10:38:34.755392Z","iopub.status.idle":"2026-02-19T10:38:34.763340Z","shell.execute_reply.started":"2026-02-19T10:38:34.755366Z","shell.execute_reply":"2026-02-19T10:38:34.762744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for image_name in image_files:\n\n    image_path = os.path.join(VALID_IMAGE_PATH, image_name)\n    mask_path = os.path.join(VALID_MASK_PATH, image_name.replace(\".jpg\", \".txt\"))\n\n    image = np.array(Image.open(image_path).convert(\"RGB\"))\n    image = cv2.resize(image, (256, 256))\n\n    # ---------- SAM ----------\n    with torch.inference_mode(), torch.autocast(\"cuda\", dtype=torch.bfloat16):\n        result = mask_generator.generate(image)\n\n    detections = sv.Detections.from_sam(result)\n    mask_annotator = sv.MaskAnnotator(color_lookup=sv.ColorLookup.INDEX)\n\n    pred_image = mask_annotator.annotate(\n        scene=image.copy(),\n        detections=detections\n    )\n\n    sam_outputs.append(pred_image)\n\n    # ---------- GT ----------\n    gt_colored = txt_to_colored_mask(mask_path, image.shape)\n    gt_overlay = cv2.addWeighted(image, 0.6, gt_colored, 0.4, 0)\n\n    gt_outputs.append(gt_overlay)\n\nfig, axes = plt.subplots(5, 10, figsize=(30, 15))\n\nfor i in range(25):\n    row = i // 5\n    col = i % 5\n\n    # SAM solda\n    axes[row, col].imshow(sam_outputs[i])\n    axes[row, col].axis(\"off\")\n\n    # GT sağda\n    axes[row, col + 5].imshow(gt_outputs[i])\n    axes[row, col + 5].axis(\"off\")\n\nplt.suptitle(\" SAM2 Predictions  |  Ground Truth \", fontsize=20)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T10:38:37.118080Z","iopub.execute_input":"2026-02-19T10:38:37.118641Z","iopub.status.idle":"2026-02-19T10:39:28.664379Z","shell.execute_reply.started":"2026-02-19T10:38:37.118612Z","shell.execute_reply":"2026-02-19T10:39:28.662812Z"}},"outputs":[],"execution_count":null}]}