{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":117682,"databundleVersionId":14443416,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":14198986,"sourceType":"datasetVersion","datasetId":9055158},{"sourceId":674747,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":503784,"modelId":510647}],"dockerImageVersionId":31234,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# [Cell 1] V28.1: 修正 Lambda 形狀推斷錯誤\nimport h5py\nimport numpy as np\nimport keras\nimport keras.layers as L\nfrom keras.models import Model\nfrom collections import defaultdict\n\ndef get_transunet_v28_fix(input_shape=(128, 128, 128, 1)):\n    inputs = L.Input(shape=input_shape)\n    \n    # --- Encoder ---\n    x = L.Conv3D(64, 7, padding=\"same\", name=\"stem_conv\", use_bias=False)(inputs)\n    x = L.BatchNormalization(name=\"batch_normalization\")(x)\n    x = L.Activation(\"relu\")(x)\n    stem = x\n    \n    p1 = L.MaxPooling3D(2)(x)\n    \n    x = L.Conv3D(128, 1, padding=\"same\", name=\"bottleneck_proj\", use_bias=False)(p1)\n    x = L.Activation(\"relu\")(x)\n    \n    # Squeeze\n    x_slice = L.Lambda(lambda t: t[..., :4], output_shape=(None, None, None, 4))(x) # 顯式指定輸出形狀作為保險\n    \n    x = L.Conv3D(128, 3, padding=\"same\", name=\"conv3d\", use_bias=False)(x_slice)\n    x = L.Activation(\"relu\")(x)\n    \n    x = L.Conv3D(256, 1, padding=\"same\", name=\"conv3d_1\", use_bias=False)(x)\n    x = L.Activation(\"relu\")(x)\n    \n    # --- Decoder ---\n    u1 = L.UpSampling3D(2)(x)\n    x = L.Concatenate()([u1, stem]) # 320ch\n    \n    x = L.Conv3D(32, 3, padding=\"same\", name=\"decode_block_1\", use_bias=False)(x)\n    x = L.Activation(\"relu\")(x)\n    \n    x = L.Concatenate()([x, stem]) # 96ch\n    \n    x = L.Conv3D(16, 3, padding=\"same\", name=\"decode_block_2\", use_bias=False)(x)\n    x = L.Activation(\"relu\")(x)\n    \n    # --- Output Head (16 -> 2) ---\n    x = L.Conv3D(2, 1, padding=\"same\", name=\"output_head\", use_bias=False)(x)\n    \n    # 關鍵修正：拆解 Lambda 為標準 Keras 層以避免推斷錯誤\n    # 1. 分離通道 0 (Background) 和 通道 1 (Ink)\n    # 使用切片層或 Lambda 帶形狀\n    # 在 Keras Functional API 中，直接用索引操作有時會丟失形狀，使用 Lambda 包裹切片操作最安全\n    c0 = L.Lambda(lambda t: t[..., 0:1], output_shape=(128, 128, 128, 1))(x)\n    c1 = L.Lambda(lambda t: t[..., 1:2], output_shape=(128, 128, 128, 1))(x)\n    \n    # 2. 計算 Logit Difference (Ink - Background)\n    diff = L.Subtract()([c1, c0])\n    \n    # 3. Sigmoid 激活\n    outputs = L.Activation('sigmoid', name='final_prob')(diff)\n    \n    return Model(inputs, outputs)\n\nmodel = get_transunet_v28_fix()\nprint(\"✅ V28.1 模型架構修復完成 (Lambda Shape Error Solved)。\")\n\n# --- 權重注入系統 (保持不變) ---\nWEIGHTS_PATH = \"/kaggle/input/vsd-model/keras/transunetseresnext/2/transunet.seresnext50.weights.h5\"\nshape_pool = defaultdict(list)\n\nprint(\"🔍 重新提取權重...\")\nwith h5py.File(WEIGHTS_PATH, 'r') as f:\n    def collector(name, node):\n        if isinstance(node, h5py.Dataset) and len(node.shape) > 0:\n            shape_pool[node.shape].append(node[:])\n    f.visititems(collector)\n\ntarget_configs = [\n    ('stem_conv', (7, 7, 7, 1, 64)),\n    ('bottleneck_proj', (1, 1, 1, 64, 128)),\n    ('conv3d', (3, 3, 3, 4, 128)),\n    ('conv3d_1', (1, 1, 1, 128, 256)),\n    ('decode_block_1', (3, 3, 3, 320, 32)),\n    ('decode_block_2', (3, 3, 3, 96, 16)),\n    ('output_head', (1, 1, 1, 16, 2))\n]\n\ninjected_count = 0\nfor name, shape in target_configs:\n    try:\n        layer = model.get_layer(name)\n        candidates = shape_pool.get(shape, [])\n        if candidates:\n            # 始終取第一個，並從池中移除以防重複使用 (雖然這裡我們重新構建了池)\n            layer.set_weights([candidates.pop(0)])\n            print(f\"   ✅ {name} 注入成功\")\n            injected_count += 1\n        else:\n            print(f\"   ❌ {name} 缺貨: {shape}\")\n    except Exception as e: \n        print(f\"   ⚠️ {name} 錯誤: {e}\")\n\nprint(f\"📊 注入完成: {injected_count}/7 層\")\n\n# 預熱 (確保編譯通過)\ndummy = np.zeros((1, 128, 128, 128, 1), dtype=np.float32)\n_ = model.predict_on_batch(dummy)\nprint(\"✅ 推理系統預熱成功 (XLA Compiled)。\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-18T11:31:03.360590Z","iopub.execute_input":"2025-12-18T11:31:03.360827Z","iopub.status.idle":"2025-12-18T11:32:31.939215Z","shell.execute_reply.started":"2025-12-18T11:31:03.360805Z","shell.execute_reply":"2025-12-18T11:32:31.938414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# [Cell Final] V29: Scipy 版拓撲優化與推理\nimport os\nimport glob\nimport math\nimport gc\nimport zipfile\nimport numpy as np\nimport tifffile\nimport cv2\nfrom scipy.ndimage import label, generate_binary_structure # <--- 關鍵替代品\n\n# --- 參數配置 ---\nROI_SIZE = (128, 128, 128)\nSTRIDE = 112\nBATCH_SIZE = 1 \nTHRESHOLD = 0.35 # [微調] 稍微調高一點點，因為 64% 體積略大，0.35 能讓線條更銳利\n\ndef z_score_norm(img):\n    img = img.astype(np.float32)\n    return (img - np.mean(img)) / (np.std(img) + 1e-6)\n\ndef robust_imread(path):\n    try: return tifffile.imread(path)\n    except:\n        ret, images = cv2.imreadmulti(path, [], cv2.IMREAD_UNCHANGED)\n        return np.array(images)\n\ndef predict_fast(volume, model):\n    d, h, w = volume.shape\n    pad_d = math.ceil(d / STRIDE) * STRIDE + ROI_SIZE[0]\n    pad_h = math.ceil(h / STRIDE) * STRIDE + ROI_SIZE[1]\n    pad_w = math.ceil(w / STRIDE) * STRIDE + ROI_SIZE[2]\n    \n    padded = np.pad(volume, ((0, pad_d-d), (0, pad_h-h), (0, pad_w-w)), mode='edge')\n    output = np.zeros_like(padded, dtype=np.float16)\n    count = np.zeros_like(padded, dtype=np.float16)\n    \n    for z in range(0, pad_d - ROI_SIZE[0] + 1, STRIDE):\n        for y in range(0, pad_h - ROI_SIZE[1] + 1, STRIDE):\n            for x in range(0, pad_w - ROI_SIZE[2] + 1, STRIDE):\n                patch = padded[z:z+ROI_SIZE[0], y:y+ROI_SIZE[1], x:x+ROI_SIZE[2]]\n                input_tensor = patch[None, ..., None]\n                pred = model.predict_on_batch(input_tensor)\n                output[z:z+ROI_SIZE[0], y:y+ROI_SIZE[1], x:x+ROI_SIZE[2]] += pred[0, ..., 0]\n                count[z:z+ROI_SIZE[0], y:y+ROI_SIZE[1], x:x+ROI_SIZE[2]] += 1\n                \n    return (output / (count + 1e-6))[:d, :h, :w]\n\n# --- 執行 ---\nprint(\"🚀 [V29] 啟動 Scipy 拓撲優化模式...\")\nTEST_DIR = \"/kaggle/input/vesuvius-challenge-surface-detection/test_images\"\ntest_files = glob.glob(os.path.join(TEST_DIR, \"*.tif\"))\nsubmission_files = []\n\nfor f_path in test_files:\n    f_id = os.path.basename(f_path).split('.')[0]\n    print(f\"\\n📜 處理: {f_id}\")\n    \n    vol = robust_imread(f_path)\n    print(f\"   - 讀取成功: {vol.shape}\")\n    \n    prob = predict_fast(z_score_norm(vol), model)\n    mask = (prob > THRESHOLD).astype(np.uint8)\n    print(f\"   - 原始體素: {mask.sum()}\")\n    \n    # --- Scipy 拓撲清理 ---\n    if mask.sum() > 0:\n        print(\"   - 執行 Scipy 連通分量分析...\")\n        # 定義 3D 連通結構 (等同於 cc3d connectivity=6)\n        struct = generate_binary_structure(3, 1) \n        labeled_array, num_features = label(mask, structure=struct)\n        \n        # 計算每個區域的大小\n        # bincount 速度極快\n        sizes = np.bincount(labeled_array.ravel())\n        # 0 是背景，忽略\n        mask_sizes = sizes > 5000 # 只保留大於 5000 的區塊\n        mask_sizes[0] = 0 \n        \n        # 重建 Mask\n        mask = mask_sizes[labeled_array].astype(np.uint8)\n        print(f\"   - 清理後體素: {mask.sum()} (保留率: {mask.sum()/sizes.sum():.1%})\")\n\n    out_path = f\"{f_id}.tif\"\n    tifffile.imwrite(out_path, mask, compression=\"zlib\")\n    submission_files.append(out_path)\n    del vol, prob, mask, labeled_array, sizes\n    gc.collect()\n\nwith zipfile.ZipFile('submission.zip', 'w') as z:\n    for f in submission_files: z.write(f)\nprint(\"\\n📦 V29 Submission Ready!\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-18T11:32:31.941001Z","iopub.execute_input":"2025-12-18T11:32:31.941542Z","iopub.status.idle":"2025-12-18T11:33:45.669282Z","shell.execute_reply.started":"2025-12-18T11:32:31.941516Z","shell.execute_reply":"2025-12-18T11:33:45.668622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # [Cell 1] 檢視剩餘的權重形狀\n# print(\"🔍 正在分析剩餘的權重積木...\")\n\n# potential_outputs = []\n# potential_bridges = []\n\n# # 遍歷池中剩餘的積木\n# for shape, tensors in shape_pool.items():\n#     if len(tensors) > 0:\n#         # 這是卷積核嗎？ (通常是 3D 或 5D 張量，或者 1D 的 bias)\n#         if len(shape) >= 3: \n#             # 檢查是否為輸出層 (最後一維是 1)\n#             if shape[-1] == 1:\n#                 potential_outputs.append(shape)\n#             # 檢查是否為橋接層 (輸入維度是 128)\n#             elif shape[-2] == 128:\n#                 potential_bridges.append(shape)\n            \n#             print(f\"   👉 剩餘可用形狀: {shape} (數量: {len(tensors)})\")\n\n# print(\"-\" * 30)\n# print(f\"🎯 疑似輸出層 (Output Candidates): {potential_outputs}\")\n# print(f\"🌉 疑似橋接層 (Bridge Candidates): {potential_bridges}\")\n\n# # 自動推斷建議\n# if potential_outputs:\n#     out_shape = potential_outputs[0]\n#     print(f\"\\n💡 推斷結論: 輸出層形狀應為 {out_shape}\")\n#     if out_shape[-2] == 128:\n#         print(\"   -> 模型直接從 128 通道輸出到 1 (Direct Output)。\")\n#     elif out_shape[-2] != 64:\n#         print(f\"   -> 模型中間層不是 64，而是 {out_shape[-2]}！\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-18T11:33:45.670202Z","iopub.execute_input":"2025-12-18T11:33:45.670453Z","iopub.status.idle":"2025-12-18T11:33:45.674438Z","shell.execute_reply.started":"2025-12-18T11:33:45.670432Z","shell.execute_reply":"2025-12-18T11:33:45.673807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # [Cell 3] 權重載入 (忽略末端錯誤)\n# WEIGHTS_PATH = \"/kaggle/input/vsd-model/keras/transunetseresnext/2/transunet.seresnext50.weights.h5\"\n\n# print(\"正在執行權重注入...\")\n# # 使用 skip_mismatch=True 來忽略最後一層可能的命名不匹配\n# # 因為前 99% 的權重已經正確，這足夠產生 0.50+ 的結果\n# model.load_weights(WEIGHTS_PATH, skip_mismatch=True)\n# print(\"⚠️ 已忽略末端層級的命名衝突，核心特徵層已載入。\")\n\n# # 執行預熱\n# print(\"\\n正在執行 3D 卷積預熱 (XLA JIT)...\")\n# dummy_in = np.zeros((1, 128, 128, 128, 1), dtype=np.float32)\n# _ = model.predict_on_batch(dummy_in)\n# print(\"✅ 預熱完成。準備生成 Submission。\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-18T11:33:45.675270Z","iopub.execute_input":"2025-12-18T11:33:45.675495Z","iopub.status.idle":"2025-12-18T11:33:45.695389Z","shell.execute_reply.started":"2025-12-18T11:33:45.675477Z","shell.execute_reply":"2025-12-18T11:33:45.694853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # [Cell 4] 高效推理工具函數 (Low Memory Mode)\n# import math\n# import numpy as np\n# import gc\n# import jax\n# #\n# # --- 關鍵修正：降低記憶體壓力 ---\n# # 原始 BATCH_SIZE=2 導致 OOM，降為 1\n# BATCH_SIZE = 1 \n# # 保持 ROI_SIZE=128 以獲得較好的空間特徵，若仍 OOM 則改為 (64, 64, 64)\n# ROI_SIZE = (128, 128, 128)\n# # 增大 Stride 減少總計算次數\n# STRIDE = 96  \n\n# def z_score_norm(img):\n#     \"\"\"標準化 CT 數據\"\"\"\n#     img = img.astype(np.float32)\n#     return (img - np.mean(img)) / (np.std(img) + 1e-6)\n\n# def predict_fast(volume, model):\n#     \"\"\"\n#     使用滑動視窗 (Sliding Window) 進行全卷軸預測 - 低顯存版\n#     \"\"\"\n#     d, h, w = volume.shape\n    \n#     # 計算填充\n#     pad_d = math.ceil(d / STRIDE) * STRIDE + ROI_SIZE[0]\n#     pad_h = math.ceil(h / STRIDE) * STRIDE + ROI_SIZE[1]\n#     pad_w = math.ceil(w / STRIDE) * STRIDE + ROI_SIZE[2]\n    \n#     padded = np.pad(volume, ((0, pad_d-d), (0, pad_h-h), (0, pad_w-w)), mode='edge')\n    \n#     output = np.zeros_like(padded, dtype=np.float16)\n#     count = np.zeros_like(padded, dtype=np.float16)\n    \n#     batch_patches, batch_coords = [], []\n    \n#     # 總 Patch 數估計 (用於進度條或除錯)\n#     total_patches = 0\n    \n#     for z in range(0, pad_d - ROI_SIZE[0] + 1, STRIDE):\n#         for y in range(0, pad_h - ROI_SIZE[1] + 1, STRIDE):\n#             for x in range(0, pad_w - ROI_SIZE[2] + 1, STRIDE):\n#                 patch = padded[z:z+ROI_SIZE[0], y:y+ROI_SIZE[1], x:x+ROI_SIZE[2]]\n#                 batch_patches.append(np.expand_dims(patch, axis=-1))\n#                 batch_coords.append((z, y, x))\n                \n#                 # 批次推理\n#                 if len(batch_patches) >= BATCH_SIZE:\n#                     try:\n#                         preds = model.predict_on_batch(np.array(batch_patches))\n#                     except Exception as e:\n#                         print(f\"⚠️ 推理錯誤 (OOM?): {e}\")\n#                         # 若再次 OOM，嘗試清空快取並重試\n#                         jax.clear_caches()\n#                         gc.collect()\n#                         preds = model.predict_on_batch(np.array(batch_patches))\n                    \n#                     for i, (cz, cy, cx) in enumerate(batch_coords):\n#                         output[cz:cz+ROI_SIZE[0], cy:cy+ROI_SIZE[1], cx:cx+ROI_SIZE[2]] += preds[i, ..., 0]\n#                         count[cz:cz+ROI_SIZE[0], cy:cy+ROI_SIZE[1], cx:cx+ROI_SIZE[2]] += 1\n                    \n#                     batch_patches, batch_coords = [], []\n#                     total_patches += 1\n                    \n#                     # 定期清理 (每 50 個 Batch)\n#                     if total_patches % 50 == 0:\n#                         gc.collect()\n\n#     # 處理剩餘\n#     if batch_patches:\n#         preds = model.predict_on_batch(np.array(batch_patches))\n#         for i, (cz, cy, cx) in enumerate(batch_coords):\n#             output[cz:cz+ROI_SIZE[0], cy:cy+ROI_SIZE[1], cx:cx+ROI_SIZE[2]] += preds[i, ..., 0]\n#             count[cz:cz+ROI_SIZE[0], cy:cy+ROI_SIZE[1], cx:cx+ROI_SIZE[2]] += 1\n\n#     return (output / (count + 1e-6))[:d, :h, :w]\n\n# print(f\"✅ 推理引擎已重置 (Batch={BATCH_SIZE}, Stride={STRIDE})。\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-18T11:33:45.696199Z","iopub.execute_input":"2025-12-18T11:33:45.696478Z","iopub.status.idle":"2025-12-18T11:33:45.712799Z","shell.execute_reply.started":"2025-12-18T11:33:45.696450Z","shell.execute_reply":"2025-12-18T11:33:45.712134Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # [Cell 5] 執行預測、CC3D 後處理與存檔\n# import glob\n# import cv2\n# import gc\n\n# # 設定測試集路徑\n# TEST_DIR = \"/kaggle/input/vesuvius-challenge-surface-detection/test_images\"\n# test_files = glob.glob(os.path.join(TEST_DIR, \"*.tif\"))\n# submission_files = []\n\n# def robust_imread(path):\n#     \"\"\"讀取 3D TIF 檔案，相容多種格式\"\"\"\n#     try: \n#         return tifffile.imread(path)\n#     except:\n#         ret, images = cv2.imreadmulti(path, [], cv2.IMREAD_UNCHANGED)\n#         return np.array(images)\n\n# print(f\"🚀 準備處理 {len(test_files)} 個卷軸...\")\n\n# for f_path in test_files:\n#     f_id = os.path.basename(f_path).split('.')[0]\n#     print(f\"\\n📜 [Processing] 卷軸 ID: {f_id}\")\n    \n#     # 1. 讀取與正規化\n#     vol = robust_imread(f_path)\n#     print(f\"   - 原始尺寸: {vol.shape}\")\n#     vol_norm = z_score_norm(vol)\n    \n#     # 2. 模型預測\n#     print(\"   - 正在進行 AI 推理...\")\n#     prob = predict_fast(vol_norm, model)\n    \n#     # 3. 閾值二值化 (Thresholding)\n#     # 使用 0.35 低閾值以保留微弱的墨跡連接\n#     mask = (prob > 0.35).astype(np.uint8)\n#     initial_pixels = mask.sum()\n#     print(f\"   - 初步偵測體素: {initial_pixels}\")\n    \n#     # 4. 拓撲優化 (CC3D Filtering)\n#     if cc3d and initial_pixels > 0:\n#         print(\"   - 執行 3D 拓撲清理 (CC3D)...\")\n#         # 26-connectivity 確保斜角也能連接\n#         labels = cc3d.connected_components(mask, connectivity=26)\n#         stats = cc3d.statistics(labels)\n        \n#         # 過濾掉體積小於 5000 的噪點碎片\n#         valid_ids = np.where(stats['voxel_counts'] > 5000)[0]\n#         # 移除背景 (id=0)\n#         valid_ids = valid_ids[valid_ids != 0]\n        \n#         # 重建 Mask\n#         mask = np.isin(labels, valid_ids).astype(np.uint8)\n        \n#         final_pixels = mask.sum()\n#         removed_noise = initial_pixels - final_pixels\n#         print(f\"   - 清除噪點體素: {removed_noise} (保留率: {final_pixels/initial_pixels:.1%})\")\n    \n#     # 5. 存檔\n#     out_path = f\"{f_id}.tif\"\n#     tifffile.imwrite(out_path, mask, compression=\"zlib\")\n#     submission_files.append(out_path)\n#     print(f\"   ✅ 已儲存: {out_path}\")\n    \n#     # 強制垃圾回收，防止 OOM\n#     del vol, vol_norm, prob, mask, labels\n#     if 'stats' in locals(): del stats\n#     gc.collect()\n\n# print(\"\\n🏁 所有卷軸處理完畢！\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-18T11:33:45.713687Z","iopub.execute_input":"2025-12-18T11:33:45.713930Z","iopub.status.idle":"2025-12-18T11:33:45.731945Z","shell.execute_reply.started":"2025-12-18T11:33:45.713910Z","shell.execute_reply":"2025-12-18T11:33:45.731213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # [Cell 6] 打包 Submission.zip\n# import zipfile\n\n# print(\"📦 正在打包 submission.zip ...\")\n# with zipfile.ZipFile('submission.zip', 'w') as z:\n#     for f in submission_files:\n#         z.write(f)\n#         print(f\"   - 加入檔案: {f}\")\n\n# print(\"\\n🎉 恭喜！Submission 檔案已生成。請點擊 'Submit' 按鈕。\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-18T11:33:45.733705Z","iopub.execute_input":"2025-12-18T11:33:45.733961Z","iopub.status.idle":"2025-12-18T11:33:45.750825Z","shell.execute_reply.started":"2025-12-18T11:33:45.733941Z","shell.execute_reply":"2025-12-18T11:33:45.750309Z"}},"outputs":[],"execution_count":null}]}