{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.9","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":23823,"databundleVersionId":1920183,"sourceType":"competition"},{"sourceId":12045790,"sourceType":"datasetVersion","datasetId":7580336},{"sourceId":12045833,"sourceType":"datasetVersion","datasetId":7580365},{"sourceId":244610628,"sourceType":"kernelVersion"},{"sourceId":244611379,"sourceType":"kernelVersion"}],"dockerImageVersionId":30055,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!unzip -o /kaggle/input/mmdetection-for-segmentation-20-inference/masks_thr05.zip -d /kaggle/working/ | head -n 10\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T08:43:09.813062Z","iopub.execute_input":"2025-06-10T08:43:09.813449Z","iopub.status.idle":"2025-06-10T08:43:25.038143Z","shell.execute_reply.started":"2025-06-10T08:43:09.813411Z","shell.execute_reply":"2025-06-10T08:43:25.036644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pkl_path = '/kaggle/input/mmdetection-for-segmentation-20-inference/mask_rcnn_resnest101_v5_ep9_train20.pkl'\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T08:43:44.751364Z","iopub.execute_input":"2025-06-10T08:43:44.7518Z","iopub.status.idle":"2025-06-10T08:43:44.757141Z","shell.execute_reply.started":"2025-06-10T08:43:44.751762Z","shell.execute_reply":"2025-06-10T08:43:44.755768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport os\nimport cv2\n\nimport torch\nimport torch.nn as nn\nimport torchvision\nimport torchvision.transforms as transforms\nfrom torch.utils.data import Dataset, DataLoader\nimport torch.nn.functional as nnf\n\nimport pickle\n","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":1.838023,"end_time":"2021-01-31T14:33:44.653299","exception":false,"start_time":"2021-01-31T14:33:42.815276","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T08:43:47.729045Z","iopub.execute_input":"2025-06-10T08:43:47.729428Z","iopub.status.idle":"2025-06-10T08:43:49.644639Z","shell.execute_reply.started":"2025-06-10T08:43:47.729392Z","shell.execute_reply":"2025-06-10T08:43:49.64335Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# real single cell with lable, doing by mask","metadata":{}},{"cell_type":"code","source":"import os\nimport pickle\n\n# 加载 .pkl 文件\nwith open('/kaggle/input/mmdetection-for-segmentation-20-inference/mask_rcnn_resnest101_v5_ep9_train20.pkl', 'rb') as f:\n    data = pickle.load(f)\n\n# 获取解压后的掩膜目录下所有 image_id 文件夹\nmask_root = '/kaggle/working/masks_thr05'\nall_image_ids = sorted(os.listdir(mask_root))  # 按名称排序确保顺序一致\n\n# 仅保留前 18% 的图像\nkeep_count = int(len(all_image_ids) * 0.83)\nall_image_ids = all_image_ids[:keep_count]\n\n# 构建 index → image_id 的映射\nindex_to_imageid = {f\"{i:04d}\": image_id for i, image_id in enumerate(all_image_ids)}\n\n# 打印检查前 3 个映射结果\nfor i in range(3):\n    print(f\"{i:04d} → {index_to_imageid[f'{i:04d}']}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T11:08:16.583983Z","iopub.execute_input":"2025-06-10T11:08:16.584631Z","iopub.status.idle":"2025-06-10T11:08:17.76929Z","shell.execute_reply.started":"2025-06-10T11:08:16.58456Z","shell.execute_reply":"2025-06-10T11:08:17.767111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 读取 pkl\nwith open('/kaggle/input/mmdetection-for-segmentation-20-inference/mask_rcnn_resnest101_v5_ep9_train20.pkl', 'rb') as f:\n    data = pickle.load(f)\n\n# 获取 mask 文件夹中的 image_id\nmask_root = '/kaggle/working/masks_thr05'\nall_image_ids = sorted(os.listdir(mask_root))\n\n# 只保留前 18% 图像\nkeep_count = int(len(all_image_ids) * 0.83)\nall_image_ids = all_image_ids[:keep_count]\n\n# 验证是否一一对应\nprint(f\"len(data): {len(data)}\")\nprint(f\"len(image_ids): {len(all_image_ids)}\")\n\n# 如果数量一致，再检查一个掩膜文件夹是否存在 mask\nsample_idx = 0\nimage_id = all_image_ids[sample_idx]\nmask_path = os.path.join(mask_root, image_id)\nprint(f\"第一个样本对应 mask 文件夹: {mask_path}\")\nprint(\"文件夹中前几个掩膜文件:\", os.listdir(mask_path)[:3])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T11:08:22.545738Z","iopub.execute_input":"2025-06-10T11:08:22.546492Z","iopub.status.idle":"2025-06-10T11:08:23.27823Z","shell.execute_reply.started":"2025-06-10T11:08:22.546438Z","shell.execute_reply":"2025-06-10T11:08:23.27726Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# correct to save the version fit weak supervision","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pickle\nfrom tqdm import tqdm\nimport pandas as pd\n\n# 路径设定\nmask_root = '/kaggle/working/masks_thr05'\npkl_path = '/kaggle/input/mmdetection-for-segmentation-20-inference/mask_rcnn_resnest101_v5_ep9_train20.pkl'\nimage_root = '/kaggle/input/hpa-single-cell-image-classification/train'\nout_crop_dir = '/kaggle/working/cell_crops_0.83'\nos.makedirs(out_crop_dir, exist_ok=True)\n\n# 加载 pkl 文件（可省略，但用于校验顺序）\nwith open(pkl_path, 'rb') as f:\n    data = pickle.load(f)\n\n# 建立 index → image_id 映射\nall_image_ids = sorted([f for f in os.listdir(mask_root) if os.path.isdir(os.path.join(mask_root, f))])\n\n# 只保留前 80%\nkeep_count = int(len(all_image_ids) * 0.83)\nall_image_ids = all_image_ids[:keep_count]\n\nindex_to_imageid = {f\"{i:04d}\": image_id for i, image_id in enumerate(all_image_ids)}\n\n# 加载图像级标签\ndf = pd.read_csv('/kaggle/input/hpa-single-cell-image-classification/train.csv')\nimage_label_map = {\n    row['ID']: list(map(int, row['Label'].split('|')))\n    for _, row in df.iterrows()\n}\n\n# 读取 RGB 图像\ndef load_RGB_image(image_id):\n    def read_channel(c):\n        path = os.path.join(image_root, f\"{image_id}_{c}.png\")\n        if not os.path.exists(path):\n            raise FileNotFoundError(f\"通道 {c} 缺失: {path}\")\n        return cv2.imread(path, cv2.IMREAD_UNCHANGED)\n    r = read_channel('red')\n    g = read_channel('green')\n    b = read_channel('blue')\n    return np.stack([r, g, b], axis=-1)\n\n# 提取并保存\ncount = 0\nmatched = 0\nunmatched = 0\n\nfor folder_id in tqdm(all_image_ids):\n    image_id = folder_id\n    image_dir = os.path.join(mask_root, image_id)\n\n    # 图像级标签\n    weak_class_ids = image_label_map.get(image_id, [])\n    if len(weak_class_ids) == 0:\n        print(f\"找不到标签: {image_id}\")\n        unmatched += 1\n        continue\n\n    try:\n        img = load_RGB_image(image_id)\n    except Exception as e:\n        print(f\"图像 {image_id} 通道读取失败: {e}\")\n        continue\n\n    for fname in os.listdir(image_dir):\n        if not fname.startswith(\"class\") or not fname.endswith(\".png\"):\n            continue\n        class_id = fname.split('_')[0].replace('class', '')\n        cell_id = fname.split('_')[1].replace('cell', '').replace('.png', '')\n        mask_path = os.path.join(image_dir, fname)\n        mask = cv2.imread(mask_path, cv2.IMREAD_GRAYSCALE)\n\n        if mask is None or mask.sum() == 0:\n            continue\n\n        # 提取单个细胞区域\n        ys, xs = np.where(mask > 0)\n        y1, y2 = ys.min(), ys.max()\n        x1, x2 = xs.min(), xs.max()\n        crop = img[y1:y2+1, x1:x2+1]\n        mask_crop = mask[y1:y2+1, x1:x2+1]\n        crop = cv2.bitwise_and(crop, crop, mask=mask_crop)\n\n        # 对每个 weak class 复制保存\n        for weak_class_id in weak_class_ids:\n            save_name = f\"{image_id}_class{weak_class_id}_cell{cell_id}.png\"\n            save_path = os.path.join(out_crop_dir, save_name)\n            cv2.imwrite(save_path, crop)\n            count += 1\n\n    matched += 1\n\nprint(f\"\\n提取完成：共保存 {count} 个单细胞 patch\")\nprint(f\"匹配成功 image_id 数量: {matched}\")\nprint(f\"匹配失败（跳过）image_id 数量: {unmatched}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T11:10:25.270448Z","iopub.execute_input":"2025-06-10T11:10:25.270901Z","iopub.status.idle":"2025-06-10T11:10:39.368916Z","shell.execute_reply.started":"2025-06-10T11:10:25.270861Z","shell.execute_reply":"2025-06-10T11:10:39.366494Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"从分割掩膜中提取出的单个细胞图像的数量，每一个 crop 对应一个单细胞区域，保存为一张 .png 图像。","metadata":{}},{"cell_type":"markdown","source":"# check result for this step","metadata":{}},{"cell_type":"code","source":"# check the result for this step\nimport os\nfrom PIL import Image\nimport matplotlib.pyplot as plt\n\n# 随机选择6个 patch 进行展示\ncrop_dir = '/kaggle/working/cell_crops_0.83'\nsample_files = sorted(os.listdir(crop_dir))[:6]\n\nplt.figure(figsize=(12, 6))\nfor i, fname in enumerate(sample_files):\n    img = Image.open(os.path.join(crop_dir, fname))\n    plt.subplot(2, 3, i + 1)\n    plt.imshow(img)\n    plt.title(fname)\n    plt.axis('off')\nplt.tight_layout()\nplt.show()\n# picture title menas imageid+mask predicted cell type+该图像中该类别下的第 2 个细胞（序号）","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T10:25:26.39218Z","iopub.execute_input":"2025-06-10T10:25:26.39295Z","iopub.status.idle":"2025-06-10T10:25:27.297575Z","shell.execute_reply.started":"2025-06-10T10:25:26.392813Z","shell.execute_reply":"2025-06-10T10:25:27.296049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#check has which file\n!ls -lh /kaggle/input/file-109/\n!head /kaggle/input/file-109/cell_level_weak_labels.csv\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T10:25:33.74431Z","iopub.execute_input":"2025-06-10T10:25:33.744716Z","iopub.status.idle":"2025-06-10T10:25:36.025926Z","shell.execute_reply.started":"2025-06-10T10:25:33.744681Z","shell.execute_reply":"2025-06-10T10:25:36.024012Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# from the cell_level_weak_labels.csv","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom tqdm import tqdm  # 加入进度条支持\n\n# 设置路径\ncrop_dir = '/kaggle/working/cell_crops_0.83'\ntrain_csv = '/kaggle/input/hpa-single-cell-image-classification/train.csv'\noutput_csv = '/kaggle/working/cell_level_weak_labels_0.83.csv'\n\n# 读取图像级标签\ndf = pd.read_csv(train_csv)\nimage_label_map = {}  # image_id → list of class_ids\nfor _, row in df.iterrows():\n    image_id = row['ID']\n    labels = list(map(int, row['Label'].split('|')))\n    image_label_map[image_id] = labels\n\n# 构建 cell-level weak label\nrecords = []\nfor fname in tqdm(os.listdir(crop_dir), desc=\"处理 crop 图像\"):\n    if not fname.endswith('.png'):\n        continue\n    fname_no_ext = fname.replace('.png', '')\n    \n    # 拆分文件名：以最后两个字段为 class & cell，其余部分为 image_id\n    parts = fname_no_ext.split('_')\n    image_id = '_'.join(parts[:-2])  # e.g. 5b88d5e8-bb99-11e8-b2b9-ac1f6b6435d0\n    class_str = parts[-2]            # e.g. class14\n    cell_str = parts[-1]             # e.g. cell10\n\n    try:\n        cell_id = int(cell_str.replace('cell', ''))\n    except:\n        continue  # 忽略非法文件名\n\n    if image_id not in image_label_map:\n        continue\n\n    for class_id in image_label_map[image_id]:\n        records.append({\n            'filename': fname,\n            'image_id': image_id,\n            'cell_id': cell_id,\n            'class_id': class_id\n        })\n\n# 保存 CSV\nweak_df = pd.DataFrame(records).drop_duplicates()\nweak_df.to_csv(output_csv, index=False)\nprint(f\"\\n已生成 cell-level weak label，共 {len(weak_df)} 条记录，保存至 {output_csv}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T10:26:01.433336Z","iopub.execute_input":"2025-06-10T10:26:01.433862Z","iopub.status.idle":"2025-06-10T10:26:04.83172Z","shell.execute_reply.started":"2025-06-10T10:26:01.433817Z","shell.execute_reply":"2025-06-10T10:26:04.829106Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"for example:如果这张图有 3 个图像级标签（如 class 0, 3, 16），那么会为每个 crop 复制 3 个标签：\n\n总记录数 = crop 数 × 标签数 ≈ 15 × 3 = 45","metadata":{}},{"cell_type":"markdown","source":"# check the weal superision lable are same as the train.csv","metadata":{}},{"cell_type":"code","source":"# check the result\nimport pandas as pd\n\n# 路径设定\ntrain_csv = '/kaggle/input/hpa-single-cell-image-classification/train.csv'\nweak_csv = '/kaggle/working/cell_level_weak_labels_0.83.csv'\n\n# 指定图像 ID\ntarget_id = '5c27f04c-bb99-11e8-b2b9-ac1f6b6435d0'\n\n# 读取源文件标签\ntrain_df = pd.read_csv(train_csv)\nraw_labels = train_df.loc[train_df['ID'] == target_id, 'Label'].values[0]\nraw_class_ids = sorted(set(int(x) for x in raw_labels.split('|')))\nprint(f\"原始标签中的 class_id：{raw_class_ids}\")\n\n# 读取弱监督文件\nweak_df = pd.read_csv(weak_csv)\npred_class_ids = sorted(weak_df.loc[weak_df['image_id'] == target_id, 'class_id'].unique())\nprint(f\"弱监督标签中的 class_id：{pred_class_ids}\")\n\n# 差异比较\nmissing = sorted(set(raw_class_ids) - set(pred_class_ids))\nextra = sorted(set(pred_class_ids) - set(raw_class_ids))\n\nprint(\"\\n弱监督中缺失的 class_id：\", missing if missing else \"无缺失\")\nprint(\"弱监督中多出的 class_id：\", extra if extra else \"无多余\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T07:23:24.057997Z","iopub.execute_input":"2025-06-10T07:23:24.05847Z","iopub.status.idle":"2025-06-10T07:23:24.119857Z","shell.execute_reply.started":"2025-06-10T07:23:24.058418Z","shell.execute_reply":"2025-06-10T07:23:24.118534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# 文件路径\nweak_csv = '/kaggle/working/cell_level_weak_labels_0.83.csv'\ndf = pd.read_csv(weak_csv)\n\n# 指定 image_id\nimage_id = '5c27f04c-bb99-11e8-b2b9-ac1f6b6435d0'\n\n# 过滤并查看该图像的所有细胞记录\nfiltered = df[df['image_id'] == image_id].sort_values(by='cell_id')\nprint(f\"该图像在 weak supervision 文件中的细胞记录（共 {len(filtered)} 条）：\")\ndisplay(filtered.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T07:23:29.419345Z","iopub.execute_input":"2025-06-10T07:23:29.419745Z","iopub.status.idle":"2025-06-10T07:23:29.454699Z","shell.execute_reply.started":"2025-06-10T07:23:29.419714Z","shell.execute_reply":"2025-06-10T07:23:29.453235Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"完全合理，对于 HPA 图像来说，20 个细胞左右/图是常见密度","metadata":{}},{"cell_type":"code","source":"\n#chck what does the csv has\n\nimport pandas as pd\n\n# 文件路径\ncsv_path = '/kaggle/working/cell_level_weak_labels_0.83.csv'\n\n# 读取并展示前五行\ndf = pd.read_csv(csv_path)\nprint(\"弱监督标签文件前五行如下：\")\nprint(df.head(10))\n\n# 每张图像中细胞的数量\ncell_counts = df.groupby('image_id')['cell_id'].nunique().reset_index(name='num_cells')\nprint(\"每张图像中细胞数量（前5张图像）：\")\nprint(cell_counts.head())\n\n# 各类 class_id 出现频率\nclass_counts = df['class_id'].value_counts().sort_index()\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(10, 5))\nclass_counts.plot(kind='bar')\nplt.xlabel('Class ID')\nplt.ylabel('Number of Cells')\nplt.title('Distribution of Predicted Cell Classes')\nplt.grid(True)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T07:39:55.73558Z","iopub.execute_input":"2025-06-10T07:39:55.735927Z","iopub.status.idle":"2025-06-10T07:39:56.157968Z","shell.execute_reply.started":"2025-06-10T07:39:55.735898Z","shell.execute_reply":"2025-06-10T07:39:56.15671Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# save the cell crops as zip","metadata":{}},{"cell_type":"code","source":"# 打包 cell_crops_0.83 文件夹\n!cd /kaggle/working && zip -r cell_crops_0.83.zip cell_crops_0.83 | head -n 5\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-10T03:41:16.038164Z","iopub.execute_input":"2025-06-10T03:41:16.038609Z","iopub.status.idle":"2025-06-10T03:41:25.252054Z","shell.execute_reply.started":"2025-06-10T03:41:16.038576Z","shell.execute_reply":"2025-06-10T03:41:25.250934Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# weak supervivion","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}