{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":6799,"databundleVersionId":4225553},{"sourceType":"datasetVersion","sourceId":9697599,"datasetId":5929630,"databundleVersionId":9930126}],"dockerImageVersionId":31154,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!git clone https://github.com/nelson1425/EfficientAD.git","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T10:16:24.602369Z","iopub.execute_input":"2026-02-24T10:16:24.603113Z","iopub.status.idle":"2026-02-24T10:16:26.951196Z","shell.execute_reply.started":"2026-02-24T10:16:24.603085Z","shell.execute_reply":"2026-02-24T10:16:26.950427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install tifffile==2021.7.30 tqdm==4.56.0 scikit-learn==1.2.2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T10:16:26.952794Z","iopub.execute_input":"2026-02-24T10:16:26.953039Z","iopub.status.idle":"2026-02-24T10:16:33.82934Z","shell.execute_reply.started":"2026-02-24T10:16:26.953019Z","shell.execute_reply":"2026-02-24T10:16:33.828573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install pycocoevalcap nltk","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T10:54:03.081037Z","iopub.execute_input":"2026-02-24T10:54:03.081756Z","iopub.status.idle":"2026-02-24T10:54:06.541643Z","shell.execute_reply.started":"2026-02-24T10:54:03.081717Z","shell.execute_reply":"2026-02-24T10:54:06.540761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import nltk\nnltk.download(\"wordnet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T10:54:25.785586Z","iopub.execute_input":"2026-02-24T10:54:25.786327Z","iopub.status.idle":"2026-02-24T10:54:27.207339Z","shell.execute_reply.started":"2026-02-24T10:54:25.786301Z","shell.execute_reply":"2026-02-24T10:54:27.206461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport csv\nimport shutil\nfrom tqdm import tqdm\n\ndef restructure_visa(source_dir: str, target_dir: str, use_symlink: bool = False) -> None:\n    \"\"\"Restructure the VISA dataset into a MVTec-style directory layout.\n\n    Parameters\n    ----------\n    source_dir : str\n        Path to the directory that contains the original VISA dataset\n        (must contain split_csv/1cls.csv).\n    target_dir : str\n        Path where the restructured dataset will be created.\n    use_symlink : bool, optional\n        If True create symbolic links. If False copy the files instead.\n    \"\"\"\n    visa_src = source_dir\n    visa_dst = target_dir\n\n    os.makedirs(visa_dst, exist_ok=True)\n\n    csv_path = \"/kaggle/input/visa-anomaly-detection/split_csv/1cls.csv\"\n    print(\"Using split file:\", csv_path)\n    with open(csv_path, newline=\"\") as file:\n        reader = csv.reader(file)\n        _ = next(reader)  # skip header line\n\n        for _, row in tqdm(enumerate(reader), desc=\"Restructuring\"):\n            class_, split_, label, img_rel_path, mask_rel_path = row\n\n            # File names\n            img_name = os.path.basename(img_rel_path)\n            mask_name = os.path.basename(mask_rel_path) if mask_rel_path else \"\"\n\n            # Map VISA label → folder label\n            label_dir = \"good\" if label == \"normal\" else \"bad\"\n\n            # ---- image ----\n            img_dst_dir = os.path.join(visa_dst, class_, split_, label_dir)\n            os.makedirs(img_dst_dir, exist_ok=True)\n\n            img_src_abs = os.path.join(visa_src, img_rel_path)\n            img_dst_abs = os.path.join(img_dst_dir, img_name)\n\n            _link_or_copy(img_src_abs, img_dst_abs, use_symlink)\n\n            # ---- mask (only for anomalous test samples) ----\n            if not mask_rel_path:\n                continue\n\n            assert split_ == \"test\" and label_dir == \"bad\", (\n                \"Mask present on non-anomalous or non-test sample: \"\n                f\"{class_}/{split_}/{label}\"\n            )\n\n            mask_dst_dir = os.path.join(visa_dst, class_, \"ground_truth\", label_dir)\n            os.makedirs(mask_dst_dir, exist_ok=True)\n\n            mask_src_abs = os.path.join(visa_src, mask_rel_path)\n            mask_dst_abs = os.path.join(mask_dst_dir, mask_name)\n            # Ensure masks end with \"_mask.png\" (MVTec style)\n            if mask_dst_abs.lower().endswith(\".png\"):\n                mask_dst_abs = mask_dst_abs[:-4] + \"_mask.png\"\n\n            _link_or_copy(mask_src_abs, mask_dst_abs, use_symlink)\n\n\ndef _link_or_copy(src: str, dst: str, use_symlink: bool) -> None:\n    \"\"\"Create *dst* pointing to *src* via symlink or copy.\n\n    If *dst* already exists, it is silently overwritten when copying, and left\n    untouched when linking to avoid FileExistsError.\n    \"\"\"\n    if use_symlink:\n        try:\n            os.symlink(src, dst)\n        except FileExistsError:\n            # Keep the existing link to allow re-running the script\n            pass\n    else:\n        # Overwrite if exists\n        shutil.copy2(src, dst)\n\n\n# ======== RUN ON KAGGLE ========\n\nsource_dir = \"/kaggle/input/visa-anomaly-detection\"   # your VisA dataset root\ntarget_dir = \"/kaggle/working/visa_mvtec\"            # where MVTec-style VisA will go\n\nrestructure_visa(source_dir, target_dir, use_symlink=False)\n\nprint(\"Done. Converted VisA is at:\", target_dir)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T10:16:33.830347Z","iopub.execute_input":"2026-02-24T10:16:33.830578Z","iopub.status.idle":"2026-02-24T10:19:49.273329Z","shell.execute_reply.started":"2026-02-24T10:16:33.830556Z","shell.execute_reply":"2026-02-24T10:19:49.27262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install torch transformers accelerate qwen-vl-utils pycocoevalcap Pillow tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T10:19:49.274945Z","iopub.execute_input":"2026-02-24T10:19:49.275167Z","iopub.status.idle":"2026-02-24T10:21:21.54523Z","shell.execute_reply.started":"2026-02-24T10:19:49.275146Z","shell.execute_reply":"2026-02-24T10:21:21.544425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================================\n# Convert mask -> bounding boxes\n# + Visualization (Before / After)\n# =========================================\n\nimport os\nimport numpy as np\nimport cv2\nimport matplotlib.pyplot as plt\nfrom matplotlib.patches import Rectangle\n\n# -------------------------------------------------\n# 1. Function: Extract bounding boxes from mask\n# -------------------------------------------------\ndef mask_to_bboxes(mask):\n    \"\"\"\n    Convert a binary mask into bounding boxes.\n    If mask has multiple disconnected regions,\n    returns multiple bounding boxes.\n    \"\"\"\n    # Ensure binary\n    mask = (mask > 0).astype(np.uint8)\n    \n    # Find connected components\n    num_labels, labels = cv2.connectedComponents(mask)\n    \n    bboxes = []\n    \n    for label_id in range(1, num_labels):  # skip background (0)\n        ys, xs = np.where(labels == label_id)\n        \n        if len(xs) == 0 or len(ys) == 0:\n            continue\n        \n        xmin, xmax = xs.min(), xs.max()\n        ymin, ymax = ys.min(), ys.max()\n        \n        bboxes.append([xmin, ymin, xmax, ymax])\n    \n    return bboxes\n\n\n# -------------------------------------------------\n# 2. Example real image + mask paths\n# -------------------------------------------------\n\n# Create dummy image\nimage_path = \"/kaggle/working/visa_mvtec/candle/test/bad/004.JPG\"\nmask_path = \"/kaggle/working/visa_mvtec/candle/ground_truth/bad/004_mask.png\"\n\nimage = cv2.imread(image_path)\nimage = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\nmask = cv2.imread(mask_path, 0)\n\n# -------------------------------------------------\n# 3. Extract bounding boxes\n# -------------------------------------------------\nbboxes = mask_to_bboxes(mask)\n\nprint(\"Detected bounding boxes:\")\nfor bbox in bboxes:\n    print(\"xmin, ymin, xmax, ymax =\", bbox)\n\n\n# -------------------------------------------------\n# 4. Visualization BEFORE (Image + Mask)\n#    Single plot\n# -------------------------------------------------\nplt.figure()\nplt.imshow(image)\nplt.imshow(mask, alpha=0.5)\nplt.title(\"Before: Image with Mask Overlay\")\nplt.axis(\"off\")\nplt.show()\n\n\n# -------------------------------------------------\n# 5. Visualization AFTER (Image + BBoxes)\n#    Single plot\n# -------------------------------------------------\nplt.figure()\nplt.imshow(image)\n\nfor bbox in bboxes:\n    xmin, ymin, xmax, ymax = bbox\n    width = xmax - xmin\n    height = ymax - ymin\n    rect = Rectangle((xmin, ymin), width, height, fill=False)\n    plt.gca().add_patch(rect)\n\nplt.title(\"After: Image with Bounding Boxes\")\nplt.axis(\"off\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T10:21:21.546304Z","iopub.execute_input":"2026-02-24T10:21:21.546623Z","iopub.status.idle":"2026-02-24T10:21:22.982228Z","shell.execute_reply.started":"2026-02-24T10:21:21.546597Z","shell.execute_reply":"2026-02-24T10:21:22.981299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport json\nimport cv2\nimport numpy as np\nfrom tqdm import tqdm\n\n# ========================\n# CONFIG\n# ========================\nROOT_DIR = \"/kaggle/working/visa_mvtec\"\nOUTPUT_JSON = \"/kaggle/working/visa_mvtec_annotations.json\"\n\nassert os.path.exists(ROOT_DIR), f\"{ROOT_DIR} not found\"\n\n# ========================\n# Helper: mask -> bboxes\n# ========================\ndef mask_to_bboxes(mask):\n    mask = (mask > 0).astype(np.uint8)\n    num_labels, labels = cv2.connectedComponents(mask)\n    bboxes = []\n    for label_id in range(1, num_labels):\n        ys, xs = np.where(labels == label_id)\n        if len(xs) == 0:\n            continue\n        xmin, xmax = xs.min(), xs.max()\n        ymin, ymax = ys.min(), ys.max()\n        width = xmax - xmin\n        height = ymax - ymin\n        area = width * height\n        bboxes.append([xmin, ymin, width, height, area])\n    return bboxes\n\n# ========================\n# Build COCO\n# ========================\ncoco = {\n    \"images\": [],\n    \"annotations\": [],\n    \"categories\": []\n}\n\nimage_id = 0\nannotation_id = 0\ncategory_id_map = {}\n\n# Only keep valid class folders\nclasses = [\n    d for d in os.listdir(ROOT_DIR)\n    if os.path.isdir(os.path.join(ROOT_DIR, d))\n]\n\nprint(\"Found classes:\", classes)\n\n# Multi-class = object class\nfor idx, class_name in enumerate(sorted(classes)):\n    category_id_map[class_name] = idx + 1\n    coco[\"categories\"].append({\n        \"id\": idx + 1,\n        \"name\": class_name,\n        \"supercategory\": \"object\"\n    })\n\nprint(\"Category mapping:\", category_id_map)\n\n# ========================\n# Iterate dataset\n# ========================\nfor class_name in tqdm(sorted(classes), desc=\"Processing classes\"):\n\n    test_bad_dir = os.path.join(ROOT_DIR, class_name, \"test\", \"bad\")\n    gt_dir = os.path.join(ROOT_DIR, class_name, \"ground_truth\", \"bad\")\n\n    if not os.path.exists(test_bad_dir):\n        print(f\"Skip {class_name}: no test/bad folder\")\n        continue\n\n    if not os.path.exists(gt_dir):\n        print(f\"Skip {class_name}: no ground_truth/bad folder\")\n        continue\n\n    image_files = [\n        f for f in os.listdir(test_bad_dir)\n        if f.lower().endswith((\".png\", \".jpg\", \".jpeg\", \".bmp\"))\n    ]\n\n    print(f\"{class_name}: {len(image_files)} anomaly images\")\n\n    for img_name in image_files:\n\n        img_path = os.path.join(test_bad_dir, img_name)\n\n        # Find corresponding mask automatically\n        base_name = os.path.splitext(img_name)[0]\n        possible_masks = [\n            f for f in os.listdir(gt_dir)\n            if f.startswith(base_name)\n        ]\n\n        if len(possible_masks) == 0:\n            continue\n\n        mask_path = os.path.join(gt_dir, possible_masks[0])\n\n        img = cv2.imread(img_path)\n        if img is None:\n            continue\n\n        height, width = img.shape[:2]\n        mask = cv2.imread(mask_path, 0)\n\n        bboxes = mask_to_bboxes(mask)\n\n        if len(bboxes) == 0:\n            continue\n\n        # Add image only if has bbox\n        coco[\"images\"].append({\n            \"id\": image_id,\n            \"file_name\": os.path.join(class_name, \"test\", \"bad\", img_name),\n            \"width\": width,\n            \"height\": height\n        })\n\n        for bbox in bboxes:\n            xmin, ymin, w, h, area = bbox\n            coco[\"annotations\"].append({\n                \"id\": int(annotation_id),\n                \"image_id\": int(image_id),\n                \"category_id\": int(category_id_map[class_name]),\n                \"bbox\": [int(xmin), int(ymin), int(w), int(h)],\n                \"area\": float(area),\n                \"iscrowd\": 0\n            })\n            annotation_id += 1\n\n        image_id += 1\n\n# ========================\n# Save JSON\n# ========================\nwith open(OUTPUT_JSON, \"w\") as f:\n    json.dump(coco, f)\n\nprint(\"Done.\")\nprint(\"Total images:\", len(coco[\"images\"]))\nprint(\"Total annotations:\", len(coco[\"annotations\"]))\nprint(\"Saved to:\", OUTPUT_JSON)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T10:21:22.983124Z","iopub.execute_input":"2026-02-24T10:21:22.983345Z","iopub.status.idle":"2026-02-24T10:22:06.710269Z","shell.execute_reply.started":"2026-02-24T10:21:22.983326Z","shell.execute_reply":"2026-02-24T10:22:06.709413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =========================================\n# Visualize 1 sample from COCO annotation\n# =========================================\n\nimport json\nimport os\nimport cv2\nimport matplotlib.pyplot as plt\nfrom matplotlib.patches import Rectangle\nimport random\n\nROOT_DIR = \"/kaggle/working/visa_mvtec\"\nANNOTATION_PATH = \"/kaggle/working/visa_mvtec_annotations.json\"\n\n# Load COCO json\nwith open(ANNOTATION_PATH, \"r\") as f:\n    coco = json.load(f)\n\n# Pick random image\nsample_image = random.choice(coco[\"images\"])\nimage_id = sample_image[\"id\"]\nfile_name = sample_image[\"file_name\"]\n\n# Get annotations for this image\nsample_annotations = [\n    ann for ann in coco[\"annotations\"]\n    if ann[\"image_id\"] == image_id\n]\n\n# Load image\nimg_path = os.path.join(ROOT_DIR, file_name)\nimage = cv2.imread(img_path)\nimage = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n# Plot image\nplt.figure()\nplt.imshow(image)\n\n# Draw bounding boxes\nfor ann in sample_annotations:\n    xmin, ymin, w, h = ann[\"bbox\"]\n    rect = Rectangle((xmin, ymin), w, h, fill=False)\n    plt.gca().add_patch(rect)\n\nplt.title(f\"Sample Image ID: {image_id}\")\nplt.axis(\"off\")\nplt.show()\n\nprint(\"Visualized image:\", img_path)\nprint(\"Number of bounding boxes:\", len(sample_annotations))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T10:22:06.711028Z","iopub.execute_input":"2026-02-24T10:22:06.711254Z","iopub.status.idle":"2026-02-24T10:22:07.045361Z","shell.execute_reply.started":"2026-02-24T10:22:06.711238Z","shell.execute_reply":"2026-02-24T10:22:07.044687Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"BASELINE A: Qwen-VL Auto Prompt","metadata":{}},{"cell_type":"code","source":"!pip install bitsandbytes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T10:22:07.046297Z","iopub.execute_input":"2026-02-24T10:22:07.046569Z","iopub.status.idle":"2026-02-24T10:22:12.599981Z","shell.execute_reply.started":"2026-02-24T10:22:07.046545Z","shell.execute_reply":"2026-02-24T10:22:12.599193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nfrom transformers import AutoProcessor, AutoModelForVision2Seq, BitsAndBytesConfig\nfrom qwen_vl_utils import process_vision_info\n\n# 1. Chống phân mảnh VRAM trên Kaggle\nos.environ[\"PYTORCH_CUDA_ALLOC_CONF\"] = \"expandable_segments:True\"\n\n\nmodel_path = \"Qwen/Qwen2.5-VL-7B-Instruct\"\n\n# 2. Cấu hình 4-bit (bắt buộc cho T4 x2)\nquantization_config = BitsAndBytesConfig(\n    load_in_4bit=True,\n    bnb_4bit_compute_dtype=torch.float16,\n)\n\nprint(\"Đang tải Processor và Model...\")\n# Dùng AutoProcessor và AutoModelForVision2Seq\nprocessor = AutoProcessor.from_pretrained(model_path)\nmodel = AutoModelForVision2Seq.from_pretrained(\n    model_path, \n    device_map=\"auto\", \n    quantization_config=quantization_config\n).eval()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T10:26:55.081245Z","iopub.execute_input":"2026-02-24T10:26:55.081571Z","iopub.status.idle":"2026-02-24T10:27:56.245999Z","shell.execute_reply.started":"2026-02-24T10:26:55.081549Z","shell.execute_reply":"2026-02-24T10:27:56.245118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport random\nfrom pathlib import Path\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport torch\n\n# --- PHẦN 1: CHỌN ẢNH NGẪU NHIÊN VÀ HIỂN THỊ ---\nmodel.generation_config.temperature = None\ndataset_root = Path(\"/kaggle/working/visa_mvtec\")\n\nall_bad_images = []\n\nfor ext in ('*.png', '*.jpg', '*.JPG'):\n    for img_path in dataset_root.rglob(ext):\n        parts = img_path.parts\n        \n        # Chỉ lấy ảnh thuộc folder test và label bad/anomaly\n        if \"test\" in parts and (\"bad\" in parts or \"anomaly\" in parts):\n            all_bad_images.append(img_path)\n\nif not all_bad_images:\n    raise ValueError(\"Không tìm thấy ảnh BAD nào!\")\n\nrandom_image_file = random.choice(all_bad_images)\nimage_path = str(random_image_file)\n\nrelative_path = random_image_file.relative_to(dataset_root)\nclass_name = relative_path.parts[0]\nlabel_name = relative_path.parts[2]\n\nprint(f\"Ảnh được chọn: {image_path}\")\nprint(f\"Class: {class_name} | Label: {label_name}\")\n# 4. Hiển thị ảnh trực tiếp trên Notebook\nimg = Image.open(image_path)\nplt.figure(figsize=(6, 6))\nplt.imshow(img)\nplt.title(f\"Class: {class_name.upper()} | Label: {label_name.upper()}\")\nplt.axis(\"off\")\nplt.show()\n\n\n# --- PHẦN 2: CHẠY QWEN-VL SINH PROMPT TỰ ĐỘNG ---\n\nprompt_text = f\"\"\"\nAnalyze this {class_name}.\nBriefly describe:\nThe object\nThe defect type\nThe exact defect location\n\nBe concise. Maximum 3 sentences.\n\"\"\"\n\nmessages = [\n    {\n        \"role\": \"user\",\n        \"content\": [\n            {\"type\": \"image\", \"image\": image_path},\n            {\"type\": \"text\", \"text\": prompt_text},\n        ],\n    }\n]\n\ntext = processor.apply_chat_template(\n    messages,\n    tokenize=False,\n    add_generation_prompt=True\n)\n\nimage_inputs, video_inputs = process_vision_info(messages)\n\ninputs = processor(\n    text=[text],\n    images=image_inputs,\n    videos=video_inputs,\n    padding=True,\n    return_tensors=\"pt\",\n).to(model.device)\n\nprint(\"\\nPrompt is generating...\")\n\nwith torch.no_grad():\n    generated_ids = model.generate(\n        **inputs,\n        max_new_tokens=80,\n        do_sample=False,\n        pad_token_id=model.config.eos_token_id\n    )\ngenerated_ids_trimmed = [\n    out_ids[len(in_ids):]\n    for in_ids, out_ids in zip(inputs.input_ids, generated_ids)\n]\n\npred = processor.batch_decode(\n    generated_ids_trimmed,\n    skip_special_tokens=True\n)[0].strip()\n\nprint(\"\\n--- Result: ---\")\nprint(pred)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-24T11:00:08.771006Z","iopub.execute_input":"2026-02-24T11:00:08.771316Z","iopub.status.idle":"2026-02-24T11:00:17.282406Z","shell.execute_reply.started":"2026-02-24T11:00:08.771295Z","shell.execute_reply":"2026-02-24T11:00:17.281728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}