{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.9","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":23823,"databundleVersionId":1920183,"sourceType":"competition"},{"sourceId":1911681,"sourceType":"datasetVersion","datasetId":1128406},{"sourceId":1934626,"sourceType":"datasetVersion","datasetId":1128710},{"sourceId":3075714,"sourceType":"datasetVersion","datasetId":849808},{"sourceId":12127290,"sourceType":"datasetVersion","datasetId":7636434},{"sourceId":12148879,"sourceType":"datasetVersion","datasetId":7651552},{"sourceId":244667974,"sourceType":"kernelVersion"},{"sourceId":433812,"sourceType":"modelInstanceVersion","modelInstanceId":353724,"modelId":375030},{"sourceId":438557,"sourceType":"modelInstanceVersion","modelInstanceId":357800,"modelId":379137},{"sourceId":441190,"sourceType":"modelInstanceVersion","modelInstanceId":358793,"modelId":380105}],"dockerImageVersionId":30056,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This is simple mmdetection infrence script as a base line.\nTraining part can be foud [here](https://www.kaggle.com/its7171/mmdetection-for-segmentation-training).","metadata":{"papermill":{"duration":0.009854,"end_time":"2021-02-02T02:49:13.549001","exception":false,"start_time":"2021-02-02T02:49:13.539147","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!pip install \"../input/landmark-additional-packages/timm-0.3.4-py3-none-any.whl\" # already in the system\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-20T01:48:03.049736Z","iopub.execute_input":"2025-06-20T01:48:03.050038Z","iopub.status.idle":"2025-06-20T01:48:41.573336Z","shell.execute_reply.started":"2025-06-20T01:48:03.050016Z","shell.execute_reply":"2025-06-20T01:48:41.572395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom PIL import Image\nfrom torchvision import transforms\nimport timm\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-20T01:49:03.773498Z","iopub.execute_input":"2025-06-20T01:49:03.773805Z","iopub.status.idle":"2025-06-20T01:49:03.881153Z","shell.execute_reply.started":"2025-06-20T01:49:03.773772Z","shell.execute_reply":"2025-06-20T01:49:03.880589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nmask_root = '/kaggle/input/mmdetec-mask-20-40/masks_20_40_thr05/masks_thr05'\nunique_filenames = set()\n\nfor image_id in os.listdir(mask_root):\n    folder_path = os.path.join(mask_root, image_id)\n    if not os.path.isdir(folder_path):\n        continue\n\n    for fname in os.listdir(folder_path):\n        if fname.endswith('.png'):\n            unique_filenames.add(os.path.join(image_id, fname))  # 保留 image_id + 文件名 的完整路径\n\nprint(f\"唯一掩膜文件（考虑 class）总数为: {len(unique_filenames)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-20T01:54:59.901246Z","iopub.execute_input":"2025-06-20T01:54:59.901536Z","iopub.status.idle":"2025-06-20T01:55:04.762657Z","shell.execute_reply.started":"2025-06-20T01:54:59.901511Z","shell.execute_reply":"2025-06-20T01:55:04.761806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nmask_root = '/kaggle/input/mmdetec-mask-20-40/masks_20_40_thr05/masks_thr05'\nunique_image_cell_pairs = set()\n\nfor image_id in os.listdir(mask_root):\n    folder_path = os.path.join(mask_root, image_id)\n    if not os.path.isdir(folder_path):\n        continue\n\n    for fname in os.listdir(folder_path):\n        if not fname.endswith('.png'):\n            continue\n        try:\n            # ✅ 文件名格式如 class6_cell1.png\n            cell_part = fname.split('_')[1]  # e.g., 'cell1.png'\n            cell_id = cell_part.replace('cell', '').replace('.png', '')\n            unique_image_cell_pairs.add((image_id, cell_id))\n        except Exception as e:\n            print(f\"[跳过] 无法处理文件名: {fname}，错误: {e}\")\n\nprint(f\"唯一的 image_id + cell_id 对数量为: {len(unique_image_cell_pairs)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-20T01:53:27.024159Z","iopub.execute_input":"2025-06-20T01:53:27.024477Z","iopub.status.idle":"2025-06-20T01:53:31.624063Z","shell.execute_reply.started":"2025-06-20T01:53:27.024451Z","shell.execute_reply":"2025-06-20T01:53:31.623163Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"因为有的file在weak supervision的时候被赋予了两遍value，导致有一些mask是完全重复的。我们在这里把它去掉","metadata":{}},{"cell_type":"markdown","source":"# 从掩膜提取 crop 图像","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nfrom PIL import Image\nfrom tqdm import tqdm\n\nmask_root = '/kaggle/input/mmdetec-mask-20-40/masks_20_40_thr05/masks_thr05'\nimage_root = '/kaggle/input/hpa-single-cell-image-classification/train'\nsave_dir = '/kaggle/working/cell_crops_test'\nos.makedirs(save_dir, exist_ok=True)\n\ndef load_rgb_image(image_id, image_root):\n    colors = ['red', 'green', 'blue']\n    imgs = []\n    for c in colors:\n        path = os.path.join(image_root, f\"{image_id}_{c}.png\")\n        if not os.path.exists(path):\n            print(f\"[跳过] 通道图像缺失: {path}\")\n            return None\n        img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n        imgs.append(img)\n    return np.stack(imgs, axis=-1)\n\nall_image_ids = sorted(os.listdir(mask_root))\nnum_limit = int(len(all_image_ids))\nselected_image_ids = all_image_ids[:num_limit]\n\nfor image_id in tqdm(selected_image_ids, desc=\"正在处理图像\"):\n    mask_dir = os.path.join(mask_root, image_id)\n    if not os.path.isdir(mask_dir):\n        continue\n\n    rgb_img = load_rgb_image(image_id, image_root)\n    if rgb_img is None or len(rgb_img.shape) != 3:\n        continue\n\n    # 每个 image_id 内，记录已处理的 cell_id，避免重复\n    seen_cell_ids = set()\n\n    for fname in os.listdir(mask_dir):\n        if not fname.endswith('.png'):\n            continue\n\n        try:\n            class_id = int(fname.split('_')[0].replace('class', ''))\n            cell_id = int(fname.split('_')[1].replace('cell', '').replace('.png', ''))\n        except:\n            print(f\"[跳过] 无法解析类名或细胞编号：{fname}\")\n            continue\n\n        # 只保留第一个遇到的 cell_id（忽略重复的不同 class）\n        if cell_id in seen_cell_ids:\n            continue\n        seen_cell_ids.add(cell_id)\n\n        mask_path = os.path.join(mask_dir, fname)\n        mask = cv2.imread(mask_path, cv2.IMREAD_GRAYSCALE)\n        if mask is None:\n            continue\n\n        ys, xs = np.where(mask > 0)\n        if len(xs) == 0 or len(ys) == 0:\n            continue\n\n        x_min, x_max = xs.min(), xs.max()\n        y_min, y_max = ys.min(), ys.max()\n        crop = rgb_img[y_min:y_max+1, x_min:x_max+1]\n\n        if crop.size == 0 or len(crop.shape) != 3:\n            continue\n\n        try:\n            crop_pil = Image.fromarray(cv2.cvtColor(crop, cv2.COLOR_BGR2RGB))\n        except:\n            continue\n\n        # ✅ 文件名仍保留 class_id，但只存一个\n        out_name = f\"{image_id}_class{class_id}_cell{cell_id}.png\"\n        crop_pil.save(os.path.join(save_dir, out_name))\n\nprint(\"所有唯一细胞裁剪图像生成完毕，已保存至：\", save_dir)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-20T02:00:28.591675Z","iopub.execute_input":"2025-06-20T02:00:28.591969Z","execution_failed":"2025-06-20T02:04:00.285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# 设置路径\ncrop_dir = '/kaggle/working/cell_crops_test'\n\n# 获取所有 .png 文件\nall_files = [f for f in os.listdir(crop_dir) if f.endswith('.png')]\n\n# 提取唯一的 image_id（从文件名中提取 image_id）\nunique_image_ids = set(f.split('_')[0] for f in all_files)\n\n# 输出结果\nprint(f\"总共裁剪图像数: {len(all_files)}\")\nprint(f\"唯一 image_id 数量: {len(unique_image_ids)}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!cd /kaggle/working && zip -r cell_crops_test.zip cell_crops_test\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}