{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":30201,"databundleVersionId":2750748,"sourceType":"competition"},{"sourceId":11977273,"sourceType":"datasetVersion","datasetId":7510849},{"sourceId":11987188,"sourceType":"datasetVersion","datasetId":7510607}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install --no-index --find-links /kaggle/input/cellpose-whl --no-deps cellpose fastremap fill_voids roifile","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-31T00:17:40.878862Z","iopub.execute_input":"2025-05-31T00:17:40.879396Z","iopub.status.idle":"2025-05-31T00:17:43.565646Z","shell.execute_reply.started":"2025-05-31T00:17:40.879377Z","shell.execute_reply":"2025-05-31T00:17:43.564934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from cellpose import models, io\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nimport cv2\n\nmodel = models.CellposeModel(gpu=True, pretrained_model='/kaggle/input/my-cellpose-models/best_model_0.30468207597732544.pth')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-31T00:17:43.566634Z","iopub.execute_input":"2025-05-31T00:17:43.566926Z","iopub.status.idle":"2025-05-31T00:18:20.433888Z","shell.execute_reply.started":"2025-05-31T00:17:43.566898Z","shell.execute_reply":"2025-05-31T00:18:20.433346Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rle_encode(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels = img.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\ndef rle_decode(mask_rle, shape=(520, 704)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape)  # Needed to align to RLE direction","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-31T00:18:20.435704Z","iopub.execute_input":"2025-05-31T00:18:20.436095Z","iopub.status.idle":"2025-05-31T00:18:20.441895Z","shell.execute_reply.started":"2025-05-31T00:18:20.436076Z","shell.execute_reply":"2025-05-31T00:18:20.441184Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def compute_iou(mask1, mask2):\n    \"\"\"\n    計算兩個二值 mask 的 IoU（Intersection over Union）。\n    \"\"\"\n    intersection = np.logical_and(mask1, mask2).sum()\n    union = np.logical_or(mask1, mask2).sum()\n    return intersection / union if union > 0 else 0.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-31T00:18:20.442609Z","iopub.execute_input":"2025-05-31T00:18:20.442865Z","iopub.status.idle":"2025-05-31T00:18:20.458295Z","shell.execute_reply.started":"2025-05-31T00:18:20.442843Z","shell.execute_reply":"2025-05-31T00:18:20.457599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def tta_predict_probability_and_flows(model, image, image_id):\n    \"\"\"\n    使用測試時間增強 (TTA) 來預測機率圖和流向量。\n    包括原始圖像、水平翻轉、垂直翻轉、水平垂直翻轉，\n    以及 90 度、180 度、270 度旋轉。\n\n    Args:\n        model: Cellpose 模型實例。\n        image (np.ndarray): 輸入的原始圖像。\n        image_id (str): 圖像的 ID (用於傳遞，本函數中未使用)。\n\n    Returns:\n        tuple: (averaged_prob_map, averaged_dp_map)\n               - averaged_prob_map (np.ndarray): 平均後的細胞機率圖。\n               - averaged_dp_map (np.ndarray): 平均後的流向量圖 (2, H, W)。\n    \"\"\"\n    \n    # 圖像增強列表\n    images = [\n        image,                                # 原始圖像\n        np.flip(image, axis=1),               # 水平翻轉\n        np.flip(image, axis=0),               # 垂直翻轉\n        np.flip(np.flip(image, axis=0), axis=1), # 水平垂直翻轉\n        np.rot90(image, k=1),                 # 90 度逆時針旋轉\n        np.rot90(image, k=2),                 # 180 度旋轉\n        np.rot90(image, k=3),                 # 270 度逆時針旋轉\n    ]\n\n    # 機率圖的逆變換操作列表 (與圖像變換相反)\n    inverse_ops_prob = [\n        lambda x: x,\n        lambda x: np.flip(x, axis=1),\n        lambda x: np.flip(x, axis=0),\n        lambda x: np.flip(np.flip(x, axis=0), axis=1),\n        lambda x: np.rot90(x, k=-1), # 90 度逆時針的逆 (即 270 度順時針)\n        lambda x: np.rot90(x, k=-2), # 180 度的逆 (即 180 度)\n        lambda x: np.rot90(x, k=-3), # 270 度逆時針的逆 (即 90 度順時針)\n    ]\n\n    # 流向量圖 (dP) 的逆變換操作列表 (處理 (2, H, W) dP)\n    # 這裡的逆變換需要同時考慮翻轉/旋轉和流向量方向的變化\n    inverse_ops_dp = [\n        lambda dp: dp, # 0. Original\n        # 1. 水平翻轉: x 分量翻轉，y 分量變號\n        lambda dp: np.stack([np.flip(dp[0], axis=1), -np.flip(dp[1], axis=1)], axis=0),\n        # 2. 垂直翻轉: y 分量翻轉，x 分量變號\n        lambda dp: np.stack([-np.flip(dp[0], axis=0), np.flip(dp[1], axis=0)], axis=0),\n        # 3. 水平垂直翻轉: x 和 y 分量都翻轉，且都變號\n        lambda dp: np.stack([-np.flip(np.flip(dp[0], axis=0), axis=1), -np.flip(np.flip(dp[1], axis=0), axis=1)], axis=0),\n        # 4. 90 度逆時針旋轉 (k=1):\n        # 圖像 (x, y) -> (-y, x)。\n        # 流場 (dp_x, dp_y) 在旋轉後會變成 (-dp_y, dp_x) 相對於新的座標軸。\n        # 逆變換：先將流場圖逆旋轉回來，然後調整分量。\n        # 例如，原始 dp_x 在逆時針旋轉 90 度後，會成為新的 dp_y (但在新的座標系下為負值)。\n        # 所以逆回來時，新 dp_x 應為原 dp_y (逆旋轉後) 的負值，新 dp_y 應為原 dp_x (逆旋轉後)。\n        lambda dp: np.stack([np.rot90(dp[1], k=-1), -np.rot90(dp[0], k=-1)], axis=0),\n        # 5. 180 度旋轉 (k=2):\n        # 圖像 (x, y) -> (-x, -y)。\n        # 流場 (dp_x, dp_y) -> (-dp_x, -dp_y) 相對於新的座標軸。\n        # 逆變換：將流場圖逆旋轉回來，然後將兩個分量都變號。\n        lambda dp: np.stack([-np.rot90(dp[0], k=-2), -np.rot90(dp[1], k=-2)], axis=0),\n        # 6. 270 度逆時針旋轉 (k=3):\n        # 圖像 (x, y) -> (y, -x)。\n        # 流場 (dp_x, dp_y) -> (dp_y, -dp_x) 相對於新的座標軸。\n        # 逆變換：將流場圖逆旋轉回來，然後調整分量。\n        # 類似於 90 度旋轉，但分量互換和變號的方式不同。\n        lambda dp: np.stack([-np.rot90(dp[1], k=-3), np.rot90(dp[0], k=-3)], axis=0),\n    ]\n\n    all_prob_maps = []\n    all_dp_maps = []\n    \n    # 獲取圖像的原始 H, W (對於 np.rot90 會改變 H, W 的順序，但我們處理的是原始圖像尺寸)\n    # Cellpose 的 eval_flows 應該會根據圖像的實際尺寸來處理\n    # 不過為了保險，可以在 `inverse_ops_dp` 裡面對 `np.rot90` 的結果進行尺寸調整\n    # 但 Cellpose 內部通常會處理輸入圖像的尺寸變化，並輸出正確尺寸的流場，\n    # 所以這裡主要確保逆變換的邏輯正確。\n    H, W = image.shape[:2] \n\n    for idx, aug_img in enumerate(images):\n        _, eval_flows, _ = model.eval(aug_img, compute_masks=False)\n        \n        pred_dp_map_raw = eval_flows[1].astype(np.float32)\n        pred_prob_map_raw = eval_flows[2].astype(np.float32)\n\n        # 執行機率圖的逆變換\n        pred_prob_map = inverse_ops_prob[idx](pred_prob_map_raw)\n        \n        # 執行流向量圖的逆變換\n        pred_dp_map_transformed = inverse_ops_dp[idx](pred_dp_map_raw)\n        \n        # 由於 np.rot90 可能改變圖像的 H, W 順序，Cellpose 返回的 eval_flows 也會是旋轉後的尺寸。\n        # 在這裡，我們需要確保所有逆變換後的圖都回到原始的 (H, W) 尺寸。\n        # 使用 cv2.resize 或 skimage.transform.resize 來統一尺寸。\n        # 但 Cellpose 通常在內部處理好這些尺寸，如果 `inverse_ops_prob` 和 `inverse_ops_dp`\n        # 已經包含 `np.rot90` 的逆變換，那麼最終結果的尺寸應該是正確的。\n        # 這裡假設 `eval_flows` 返回的 `pred_prob_map_raw` 和 `pred_dp_map_raw` 已經是處理過旋轉的尺寸。\n        # 而 `inverse_ops_prob` 和 `inverse_ops_dp` 負責將其轉回原始的 (H,W) 方向和分量。\n\n        # 驗證尺寸是否一致，並在不一致時進行調整 (可選，但推薦用於除錯)\n        if pred_prob_map.shape[:2] != (H, W):\n            # print(f\"警告: 機率圖尺寸不符，正在調整 {pred_prob_map.shape[:2]} -> {(H, W)}\")\n            pred_prob_map = cv2.resize(pred_prob_map, (W, H), interpolation=cv2.INTER_LINEAR)\n        if pred_dp_map_transformed.shape[1:] != (H, W):\n            # print(f\"警告: 流向量圖尺寸不符，正在調整 {pred_dp_map_transformed.shape[1:]} -> {(H, W)}\")\n            # 對於 dp map，需要對兩個通道分別 resize\n            pred_dp_map_transformed_x = cv2.resize(pred_dp_map_transformed[0], (W, H), interpolation=cv2.INTER_LINEAR)\n            pred_dp_map_transformed_y = cv2.resize(pred_dp_map_transformed[1], (W, H), interpolation=cv2.INTER_LINEAR)\n            pred_dp_map_transformed = np.stack([pred_dp_map_transformed_x, pred_dp_map_transformed_y], axis=0)\n\n\n        all_prob_maps.append(pred_prob_map)\n        all_dp_maps.append(pred_dp_map_transformed)\n\n    # 對所有增強圖像的預測結果進行平均\n    averaged_prob_map = np.mean(all_prob_maps, axis=0)\n    averaged_dp_map = np.mean(all_dp_maps, axis=0)\n    \n    return averaged_prob_map, averaged_dp_map\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-31T00:18:20.459480Z","iopub.execute_input":"2025-05-31T00:18:20.459725Z","iopub.status.idle":"2025-05-31T00:18:20.479252Z","shell.execute_reply.started":"2025-05-31T00:18:20.459709Z","shell.execute_reply":"2025-05-31T00:18:20.478530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_ids, final_masks = [], []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-31T00:18:20.479947Z","iopub.execute_input":"2025-05-31T00:18:20.480151Z","iopub.status.idle":"2025-05-31T00:18:20.497244Z","shell.execute_reply.started":"2025-05-31T00:18:20.480136Z","shell.execute_reply":"2025-05-31T00:18:20.496706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from collections import defaultdict\nfrom tqdm import tqdm\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nimport skimage.io as io\nfrom cellpose import dynamics\n\ntest_dir = Path('/kaggle/input/sartorius-cell-instance-segmentation/test')\ntest_files = sorted([f for f in test_dir.iterdir() if f.suffix == '.png']) \n\nsubmission_data = []\n\nfor img_path in tqdm(test_files, desc=\"🚀 推論中\"):\n    img_id = img_path.stem\n    img = io.imread(img_path)\n    \n    # 獲取圖像的原始尺寸，用於 dynamics.resize_and_compute_masks\n    original_shape = img.shape[:2] # 假設是 (H, W)\n\n    # 使用 TTA 預測並得到平均機率圖和平均流向量\n    averaged_prob_map, averaged_dp_map = tta_predict_probability_and_flows(model, img, img_id)\n    \n    # 使用 Cellpose 的動力學模組從平均機率圖和流向量中計算 Mask\n    # 這裡的參數應根據你的需求調整\n    # 例如：flow_threshold, cellprob_threshold, min_size 等\n    combined_mask = dynamics.resize_and_compute_masks(\n        averaged_dp_map, \n        averaged_prob_map,\n        cellprob_threshold=0.0, # 預設值，可以調整\n        flow_threshold=0.4,     # 預設值，可以調整\n        min_size=20,            # 預設值，可以調整\n        resize=original_shape,  # 確保 Mask 返回到原始圖像尺寸\n        device=model.device     # 傳遞模型所在的設備\n    )\n\n    if combined_mask.max() > 0: # 確保有偵測到實例\n        instance_ids = np.unique(combined_mask)\n        instance_ids = instance_ids[instance_ids != 0] # 排除背景 0\n        for inst_id in instance_ids:\n            binary_mask = (combined_mask == inst_id).astype(np.uint8)\n            rle = rle_encode(binary_mask) # 使用你提供的 rle_encode\n            submission_data.append({'id': img_id, 'predicted': rle})\n    else: # 如果沒有偵測到任何實例，則提交空字串\n        submission_data.append({'id': img_id, 'predicted': ''})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-31T00:18:20.498001Z","iopub.execute_input":"2025-05-31T00:18:20.498301Z","iopub.status.idle":"2025-05-31T00:19:00.667506Z","shell.execute_reply.started":"2025-05-31T00:18:20.498281Z","shell.execute_reply":"2025-05-31T00:19:00.666727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ 建立 DataFrame 並補齊漏掉圖片\nsubmission_df = pd.DataFrame(submission_data)\nall_test_ids = [f.stem for f in test_files]\nmissing_ids = set(all_test_ids) - set(submission_df['id'].unique())\nfor img_id in missing_ids:\n    submission_df = pd.concat([submission_df, pd.DataFrame([{'id': img_id, 'predicted': ''}])], ignore_index=True)\nsubmission_df = submission_df.sort_values(by='id').reset_index(drop=True)\n\n# ✅ 檢查是否重複 id（應該沒有）\nduplicate_ids = submission_df.duplicated(subset=['id'], keep=False)\nif duplicate_ids.any():\n    print(\"⚠️ 存在重複 id，請檢查合併邏輯！\")\n    \n# ✅ 像素重疊檢查\nrle_dict = defaultdict(list)\nfor img_id, rle in zip(submission_df['id'], submission_df['predicted']):\n    if isinstance(rle, str) and rle.strip() != '':\n        rle_dict[img_id].append(rle)\n\noverlap_count = 0\nfor img_id, rles in tqdm(rle_dict.items(), desc=\"🔍 檢查像素重疊\"):\n    mask_sum = np.zeros((520, 704), dtype=np.uint8)\n    for rle in rles:\n        mask = rle_decode(rle, (520, 704))\n        mask_sum += mask\n    if (mask_sum > 1).any():\n        print(f\"⚠️ 圖像 {img_id} 有重疊的 pixel！\")\n        overlap_count += 1\n\nif overlap_count == 0:\n    print(\"✅ 所有圖像都沒有重疊 pixel\")\nelse:\n    print(f\"⚠️ 共 {overlap_count} 張圖像有重疊問題，建議檢查合併策略\")\n\n# ✅ 輸出 CSV\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-31T00:19:00.668373Z","iopub.execute_input":"2025-05-31T00:19:00.668621Z","iopub.status.idle":"2025-05-31T00:19:00.721635Z","shell.execute_reply.started":"2025-05-31T00:19:00.668602Z","shell.execute_reply":"2025-05-31T00:19:00.721026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\nimport numpy as np\n\ndef random_color(seed=None):\n    if seed is not None:\n        random.seed(seed)\n    return [random.randint(0, 255) for _ in range(3)]\n\ndef mask_to_color(mask):\n    \"\"\"\n    將多個 instance mask 合成彩色圖像。\n    mask: numpy array, shape = (N, H, W)\n    return: RGB 彩色 mask，shape = (H, W, 3)\n    \"\"\"\n    if mask.ndim == 2:\n        mask = mask[np.newaxis, ...]  # 單一 mask 也包成 (1, H, W)\n\n    h, w = mask.shape[1:]\n    color_mask = np.zeros((h, w, 3), dtype=np.uint8)\n    for i in range(mask.shape[0]):\n        color = random_color(seed=i)\n        color_mask[mask[i] > 0] = color\n    return color_mask","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-31T00:19:00.723268Z","iopub.execute_input":"2025-05-31T00:19:00.723479Z","iopub.status.idle":"2025-05-31T00:19:00.728772Z","shell.execute_reply.started":"2025-05-31T00:19:00.723463Z","shell.execute_reply":"2025-05-31T00:19:00.728020Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport cv2\nimport numpy as np\nfrom pathlib import Path\n\n# ✅ 讀 submission.csv\nsubmission_df = pd.read_csv(\"submission.csv\")\n\n# ✅ 測試資料夾位置\ntest_dir = Path('/kaggle/input/sartorius-cell-instance-segmentation/test')\n\n# ✅ 你想要可視化幾張圖\nN = 3\nvisualized_ids = submission_df['id'].unique()[:N]\n\nfor img_id in visualized_ids:\n    img_path = test_dir / f\"{img_id}.png\"\n    image = cv2.imread(str(img_path))\n\n    masks_rle = submission_df[submission_df['id'] == img_id]['predicted'].tolist()\n    decoded_masks = []\n\n    for rle in masks_rle:\n        if isinstance(rle, str) and rle.strip() != '':\n            decoded = rle_decode(rle, shape=(520, 704))\n            decoded_masks.append(decoded)\n\n    if len(decoded_masks) == 0:\n        print(f\"⚠️ {img_id} 沒有預測到 mask\")\n        continue\n\n    instance_masks = np.array(decoded_masks)\n    colored_mask = mask_to_color(instance_masks)\n\n    plt.figure(figsize=(15, 15))\n    plt.imshow(image[..., ::-1])\n    plt.imshow(colored_mask, alpha=0.5)\n    plt.axis(\"off\")\n    plt.title(f\"Image {img_id} — Loaded from submission.csv\", fontsize=20)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-31T00:19:00.729424Z","iopub.execute_input":"2025-05-31T00:19:00.729619Z","iopub.status.idle":"2025-05-31T00:19:02.961806Z","shell.execute_reply.started":"2025-05-31T00:19:00.729597Z","shell.execute_reply":"2025-05-31T00:19:02.960910Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ 終端輸出資訊\nprint(submission_df.head(20))\nprint(\"欄位名稱:\", submission_df.columns)\nprint(\"是否有重複 id:\", submission_df['id'].duplicated().any())\nprint(\"是否有 null:\", submission_df.isnull().sum())\nprint(\"是否有空字串以外的空值:\", (submission_df['predicted'].astype(str).str.strip() == '').sum())\nprint(\"id 總數:\", submission_df['id'].nunique(), \"submission 行數:\", len(submission_df))\nprint(\"predicted 欄型別:\", submission_df['predicted'].apply(type).value_counts())\nprint(\"✅ Submission file 'submission.csv' created successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-31T00:19:02.962587Z","iopub.execute_input":"2025-05-31T00:19:02.962791Z","iopub.status.idle":"2025-05-31T00:19:02.977259Z","shell.execute_reply.started":"2025-05-31T00:19:02.962776Z","shell.execute_reply":"2025-05-31T00:19:02.976520Z"}},"outputs":[],"execution_count":null}]}