{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.9","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":23823,"databundleVersionId":1920183,"sourceType":"competition"},{"sourceId":1154184,"sourceType":"datasetVersion","datasetId":652413},{"sourceId":1885540,"sourceType":"datasetVersion","datasetId":1123128},{"sourceId":1888569,"sourceType":"datasetVersion","datasetId":1125071},{"sourceId":1888577,"sourceType":"datasetVersion","datasetId":1125092},{"sourceId":2201182,"sourceType":"datasetVersion","datasetId":1318466},{"sourceId":2218048,"sourceType":"datasetVersion","datasetId":1324198},{"sourceId":2221384,"sourceType":"datasetVersion","datasetId":1227730},{"sourceId":12506227,"sourceType":"datasetVersion","datasetId":7893315},{"sourceId":12507247,"sourceType":"datasetVersion","datasetId":7893785},{"sourceId":251505556,"sourceType":"kernelVersion"}],"dockerImageVersionId":30086,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install ../input/pycocotools/pycocotools-2.0-cp37-cp37m-linux_x86_64.whl\n!pip install ../input/hpapytorchzoozip/pytorch_zoo-master\n!pip install ../input/hpacellsegmentatormaster/HPA-Cell-Segmentation-master","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-24T09:49:54.697443Z","iopub.execute_input":"2025-07-24T09:49:54.69765Z","iopub.status.idle":"2025-07-24T09:51:49.888867Z","shell.execute_reply.started":"2025-07-24T09:49:54.697597Z","shell.execute_reply":"2025-07-24T09:51:49.887791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ Step 0：复制上传的 py 文件到 /kaggle/working/\n!cp /kaggle/input/d/weiyingsvk/generate-test-mask-kernel-faster7-correct/generate_train_mask_kernel_faster7.py /kaggle/working/\n\n# ✅ Step 1：切换到 working 目录（确保后续 py 可以访问 sample_submission.csv）\n%cd /kaggle/working\n\n# ✅ Step 2：复制你构造的 sample_submission.csv 到当前工作目录\n!cp /kaggle/working/sample_submission.csv ./sample_submission.csv\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T09:51:49.89286Z","iopub.execute_input":"2025-07-24T09:51:49.893688Z","iopub.status.idle":"2025-07-24T09:51:50.711718Z","shell.execute_reply.started":"2025-07-24T09:51:49.893606Z","shell.execute_reply":"2025-07-24T09:51:50.710248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import base64\nimport gc\nimport os\nimport pickle\nimport random\nimport sys\nimport typing as t\nimport zlib\nfrom itertools import groupby\nfrom multiprocessing import Pool\nfrom operator import itemgetter\nfrom pathlib import Path\n\nfrom IPython.display import display\nimport albumentations as A\nimport cv2\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport torch\nfrom albumentations.pytorch import ToTensorV2\nfrom pycocotools import _mask as coco_mask\nfrom pycocotools import mask as mutils\nfrom torch.utils.data import DataLoader, Dataset\nfrom tqdm import tqdm\n\nrandom.seed(0)\n\n# =============================================================================\n# setting\n# =============================================================================\n\"\"\"\nPrecomputed Public + Dummy Private: COMPUTE_PUBLIC=False, COMPUTE_PRIVATE=False\nPrecomputed Public + Private      : COMPUTE_PUBLIC=False, COMPUTE_PRIVATE=True\nPublic + Private                  : COMPUTE_PUBLIC=True,  COMPUTE_PRIVATE=True\n\"\"\"\n\nLOCAL = False\nCOMPUTE_PUBLIC = False\nCOMPUTE_PRIVATE = True\n\n# ==========================\n# 1 st stage\n# ==========================\nEXP_NAME_1ST = [\"exp049\", \"exp050\"]\nMODEL_NAMES_1ST = [\n    \"model_best_0.pth\", # \"model_tmp_0.pth\", \n    \"model_best_1.pth\", # \"model_tmp_1.pth\", \n    \"model_best_2.pth\", # \"model_tmp_2.pth\",\n    \"model_best_3.pth\", # \"model_tmp_3.pth\"\n              ]\nTEST_LOCAL_COMPUTED_1ST = [\n    \"pred_0.csv\", \n    \"pred_1.csv\", \n    \"pred_2.csv\", \n    \"pred_3.csv\"\n]\n\n\n# ==========================\n# 2nd stage\n# ==========================\nBACKBONE_NAME = \"seresnet152d\"\n\nEXP_NAME = [\"exp068\", \"exp071\", \"exp072\", \"exp073\"]\nMODEL_NAMES = [\n    # index 0, exp068\n    [\n        \"model_best_0.pth\", \"model_tmp_0.pth\", \n        \"model_best_1.pth\", \"model_tmp_1.pth\", \n        \"model_best_2.pth\", \"model_tmp_2.pth\",\n        \"model_best_3.pth\", \"model_tmp_3.pth\"\n\n    ],\n    # index 1, exp071\n    [\n        \"model_0_25.pth\", \"model_0_21.pth\", \n        \"model_1_25.pth\", \"model_1_21.pth\", \n    ],\n    # index 2, exp072\n    [\n        \"model_0_25.pth\", \"model_0_21.pth\", \n#         \"model_1_25.pth\", \"model_1_21.pth\", \n    ],\n    # index 3, exp073\n    [\n        \"model_0_25.pth\", \"model_0_21.pth\", \n#         \"model_1_25.pth\", \"model_1_21.pth\", \n    ],\n]\nTEST_LOCAL_COMPUTED = [\n    # index 0, exp068\n    [\n        \"pred_0.csv\", \n        \"pred_1.csv\", \n        \"pred_2.csv\", \n        \"pred_3.csv\"\n    ],\n    # index 1, exp071\n    [\n        \"pred_0.csv\", \n        \"pred_1.csv\", \n#         \"pred_2.csv\", \n#         \"pred_3.csv\"\n    ],\n    # index 2, exp072\n    [\n        \"pred_0.csv\", \n#         \"pred_1.csv\", \n#         \"pred_2.csv\", \n#         \"pred_3.csv\"\n    ],\n    # index 3, exp073\n    [\n        \"pred_0.csv\", \n#         \"pred_1.csv\", \n#         \"pred_2.csv\", \n#         \"pred_3.csv\"\n    ],\n]\n\n\n# %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%\n# image level\n# %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%\nMODEL_NAMES_IMAGE = [\n    \"model_best_0.pth\", \"model_tmp_0.pth\", \n    \"model_best_1.pth\", \"model_tmp_1.pth\", \n    \"model_best_2.pth\", \"model_tmp_2.pth\",\n    \"model_best_3.pth\", \"model_tmp_3.pth\",\n    \"model_best_4.pth\", \"model_tmp_4.pth\",\n    \"model_best_5.pth\", \"model_tmp_5.pth\",\n    \"model_best_6.pth\", \"model_tmp_6.pth\",\n    \"model_best_7.pth\", \"model_tmp_7.pth\",\n              ]\n\nMODEL_PATHS_IMAGE = [f\"../input/hpa-image-level-weight/exp102/{p}\" for p in MODEL_NAMES_IMAGE]\n\nTEST_LOCAL_COMPUTED_IMAGE = [\n    \"pred_0.csv\", \n    \"pred_1.csv\", \n    \"pred_2.csv\", \n    \"pred_3.csv\",\n    \"pred_4.csv\",\n    \"pred_5.csv\",\n    \"pred_6.csv\",\n    \"pred_7.csv\",\n]\n\nTEST_LOCAL_COMPUTED_PATHS_IMAGE = [f\"../input/hpa-image-level-weight/exp102/{p}\" for p in TEST_LOCAL_COMPUTED_IMAGE]\n\n\n\nCOLS_TARGET = [f\"label_{i}\" for i in range(19)]\n\nBATCH_SIZE = 32\nIMAGE_SIZE = 512\n\nMARGIN = 100\nW_MASK = True\nIN_CHANS = 4\n\nKEEP_CELL_AREA_MIN = 0.005\nKEEP_NUC_AREA_MIN = 0.001\nKEEP_EDGE_CELL_AREA_MIN = 0.01\nNUC_AREA_MIN_0to5 = 0.12\n\nWEIGHT_CELL_LEVEL_VS_IMAGE_MEAN_VS_IMAGE_PRED = [0.6, 0., 0.4]\n\nRATE_OF_WEIGHT_1ST_2ND = [0.2, 0.8]\n\n\nGPUS = torch.cuda.device_count()\nGPU = 0\n\nROOT = Path.cwd().parent\nif LOCAL:\n    INPUT = ROOT / \"input\"\n    MODEL_PATHS = [ROOT / \"output\" / EXP_NAME / p for p in MODEL_NAMES]\n    TEST_LOCAL_COMPUTED_PATH = ROOT / \"output\" / EXP_NAME / TEST_LOCAL_COMPUTED\n    TEST_IMG_DIR = ROOT / \"data\" / \"test_rgby_images\"\n    MASK_DIR = ROOT / \"data\" / \"mask\"\n\n    NUC_MODEL = MASK_DIR / \"dpn_unet_nuclei_v1.pth\"\n    CELL_MODEL = MASK_DIR / \"dpn_unet_cell_3ch_v1.pth\"\n\n    MAX_THRE = 40\n\n    import timm\nelse:\n    INPUT = Path(\"/kaggle/working\")\n    LIB_DIR = ROOT / \"input\" / \"hpa2021-libs\"\n    \n    # ============================\n    # 1 st stage\n    # ============================\n    MODEL_PATHS = []\n    TEST_LOCAL_COMPUTED_PATHS_1ST = []\n    for e_name in EXP_NAME_1ST:\n        MODEL_PATHS += [LIB_DIR / e_name / p for p in MODEL_NAMES_1ST]\n        TEST_LOCAL_COMPUTED_PATHS_1ST += [LIB_DIR / e_name / p for p in TEST_LOCAL_COMPUTED_1ST]\n        \n    LEN_1ST = len(MODEL_PATHS)\n        \n    # ============================\n    # 2nd stage\n    # ============================\n    TEST_LOCAL_COMPUTED_PATHS = []\n    for i, e_name in enumerate(EXP_NAME):\n        MODEL_PATHS += [LIB_DIR / e_name / p for p in MODEL_NAMES[i]]\n        TEST_LOCAL_COMPUTED_PATHS += [LIB_DIR / e_name / p for p in TEST_LOCAL_COMPUTED[i]]\n        \n    # ===========================\n    # Weight 1st vs 2nd\n    # ===========================\n    WEIGHT_1ST_2ND = np.ones(len(MODEL_PATHS)).reshape(-1 , 1, 1)\n    WEIGHT_1ST_2ND[:LEN_1ST] = len(MODEL_PATHS) * (1 / LEN_1ST) * RATE_OF_WEIGHT_1ST_2ND[0]\n    WEIGHT_1ST_2ND[LEN_1ST:] = len(MODEL_PATHS) * (1 / (len(MODEL_PATHS) - LEN_1ST)) * RATE_OF_WEIGHT_1ST_2ND[1]\n\n\n    OUTPUT = ROOT / \"temp\"\n    MASK_DIR = OUTPUT / \"mask\"\n    MASK_DIR.mkdir(exist_ok=True, parents=True)\n    NUCEIL_DIR = MASK_DIR / \"test\" / \"nuclei\"\n    NUCEIL_DIR.mkdir(exist_ok=True, parents=True)\n    CELL_DIR = MASK_DIR / \"test\" / \"cell\"\n    CELL_DIR.mkdir(exist_ok=True, parents=True)\n\n    NUC_MODEL = (\n        ROOT / \"input\" / \"hpacellsegmentatormodelweights\" / \"dpn_unet_nuclei_v1.pth\"\n    )\n    CELL_MODEL = (\n        ROOT / \"input\" / \"hpacellsegmentatormodelweights\" / \"dpn_unet_cell_3ch_v1.pth\"\n    )\n\n    MAX_THRE = 2\n\n    sys.path.append(str(ROOT / \"input\" / \"hpa2021-libs\"))\n    import timm\n    from tqdm.notebook import tqdm\n\n\nsample_submission = pd.read_csv(\"/kaggle/input/hpa-single-cell-image-classification/train.csv\")\n\n\nprint(\"NUC_MODEL:\", NUC_MODEL.exists())\nprint(\"CELL_MODEL:\", CELL_MODEL.exists())\nprint(\"MODEL_PATHS:\", [p.exists() for p in MODEL_PATHS])\nprint(\"TEST_LOCAL_COMPUTED_PATHS:\", [p.exists() for p in TEST_LOCAL_COMPUTED_PATHS])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T09:51:50.714999Z","iopub.execute_input":"2025-07-24T09:51:50.715349Z","iopub.status.idle":"2025-07-24T09:51:56.227037Z","shell.execute_reply.started":"2025-07-24T09:51:50.715321Z","shell.execute_reply":"2025-07-24T09:51:56.225616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# def\n# =============================================================================\ndef decode_binary_mask(decoded_base64_str, width, height):\n    \"\"\"Converts a OID challenge encoding ascii text into binary mask.\"\"\"\n\n    binary_str = base64.b64decode(decoded_base64_str)\n    rle_encoded_mask = zlib.decompress(binary_str)\n    # print(rle_encoded_mask)\n    decoding_dict = {\n        \"size\": [height, width],  # [im_height, im_width],\n        \"counts\": rle_encoded_mask,\n    }\n    mask_tensor = mutils.decode(decoding_dict).astype(bool)\n    return mask_tensor\n\n\ndef coco_rle_encode(mask):\n    rle = {\"counts\": [], \"size\": list(mask.shape)}\n    counts = rle.get(\"counts\")\n    for i, (value, elements) in enumerate(groupby(mask.ravel(order=\"F\"))):\n        if i == 0 and value == 1:\n            counts.append(0)\n        counts.append(len(list(elements)))\n    return rle\n\n\ndef read_img(image_id, color, train_or_test=\"train\"):\n    filename = f\"{INPUT}/{train_or_test}/{image_id}_{color}.png\"\n    assert os.path.exists(filename), f\"not found {filename}\"\n    img = cv2.imread(filename, cv2.IMREAD_UNCHANGED)\n    if img.dtype == \"uint16\":\n        img = (img / 256).astype(\"uint8\")\n    return img\n\n\ndef load_RGB_image(image_id, train_or_test=\"test\"):\n    red = read_img(image_id, \"red\", train_or_test)\n    green = read_img(image_id, \"green\", train_or_test)\n    blue = read_img(image_id, \"blue\", train_or_test)\n    # using rgb only here\n    # yellow = read_img(image_id, \"yellow\", train_or_test, image_size)\n    stacked_images = np.transpose(np.array([red, green, blue]), (1, 2, 0))\n    return stacked_images\n\n\ndef load_RGBY_image(image_id, train_or_test=\"test\"):\n    red = read_img(image_id, \"red\", train_or_test)\n    green = read_img(image_id, \"green\", train_or_test)\n    blue = read_img(image_id, \"blue\", train_or_test)\n    # using rgb only here\n    yellow = read_img(image_id, \"yellow\", train_or_test)\n    stacked_images = np.transpose(np.array([red, green, blue, yellow]), (1, 2, 0))\n    return stacked_images\n\n\ndef print_masked_img(image_id, mask):\n    img = load_RGB_image(image_id, \"test\")\n\n    plt.figure(figsize=(15, 15))\n    plt.subplot(1, 3, 1)\n    plt.imshow(img)\n    plt.title(\"Image\")\n    plt.axis(\"off\")\n\n    plt.subplot(1, 3, 2)\n    plt.imshow(mask)\n    plt.title(\"Mask\")\n    plt.axis(\"off\")\n\n    plt.subplot(1, 3, 3)\n    plt.imshow(img)\n    plt.imshow(mask, alpha=0.6)\n    plt.title(\"Image + Mask\")\n    plt.axis(\"off\")\n    plt.show()\n\n\ndef split_list(l, n):\n    for idx in range(0, len(l), n):\n        yield l[idx : idx + n]\n\n\n# =============================================================================\n# Transforms\n# =============================================================================\ndef get_transforms():\n    return A.Compose(\n        [\n            A.Resize(IMAGE_SIZE, IMAGE_SIZE),\n            A.Normalize(\n                mean=[0.485, 0.456, 0.406, 0.456], std=[0.229, 0.224, 0.225, 0.225],\n            ),\n            ToTensorV2(),\n        ]\n    )\n\n\n# =============================================================================\n# Dataset\n# =============================================================================\ndef load_bmask(cell_mask_dir, image_id, cell_id):\n    mask = np.load(f\"{cell_mask_dir}/{image_id}.npz\")[\"arr_0\"]\n    bmask = mask == cell_id\n    return bmask * 1\n\n\nclass MyDataset(Dataset):\n    def __init__(self, df, mode, w_mask=False):\n        self.df = df.reset_index(drop=True)\n        self.mode = mode\n        self.w_mask = w_mask\n        self.transform = get_transforms()\n\n        if self.mode in [\"train\", \"valid\"]:\n            self.targets = self.df[COLS_TARGET].values\n            self.cell_mask_dir = MASK_DIR / \"train\" / \"cell\"\n        else:\n            self.cell_mask_dir = MASK_DIR / \"test\" / \"cell\"\n\n    def crop(self, image, idx):\n        y0, x0, y1, x1 = self.df.loc[idx, [\"y0\", \"x0\", \"y1\", \"x1\"]].values.astype(int)\n\n        y0 = max(0, y0 - MARGIN)\n        x0 = max(0, x0 - MARGIN)\n        y1 = min(image.shape[0], y1 + MARGIN)\n        x1 = min(image.shape[1], x1 + MARGIN)\n\n        image = image[y0:y1, x0:x1]\n        return image\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        image_id = self.df.loc[idx, \"image_id\"]\n\n        image = load_RGBY_image(image_id, \"test\")\n        image = self.crop(image, idx)\n\n        if self.w_mask:\n            cell_id = self.df.loc[idx, \"cell_id\"]\n            bmask = load_bmask(self.cell_mask_dir, image_id, cell_id)\n            bmask = self.crop(bmask, idx)\n\n            image = image * np.stack([bmask] * IN_CHANS, 2)\n\n        else:\n            pass\n\n        augmented = self.transform(image=image.astype(\"uint8\"))\n        image = augmented[\"image\"]\n\n        if self.mode in [\"train\", \"valid\"]:\n            targets = self.df.loc[idx, COLS_TARGET].values\n            return image, torch.FloatTensor(targets.astype(\"float32\"))\n        else:\n            return image\n\n\n# %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%\n# image level\n# %%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%\n\ndef get_transforms_image():\n    return A.Compose(\n        [\n            A.Resize(768, 768),\n            A.Normalize(\n                mean=[0.485, 0.456, 0.406, 0.456], std=[0.229, 0.224, 0.225, 0.225],\n            ),\n            ToTensorV2(),\n        ]\n    )\n\n\nclass ImageDataset(Dataset):\n    def __init__(self, df):\n        self.df = df.reset_index(drop=True)\n        self.transform = get_transforms_image()\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        image_id = self.df.loc[idx, \"image_id\"]\n\n        image = load_RGBY_image(image_id, \"test\")\n\n        augmented = self.transform(image=image.astype(\"uint8\"))\n        image = augmented[\"image\"]\n\n        return image\n\n# =============================================================================\n# Network\n# =============================================================================\ndef get_model(model_path):\n    model = timm.create_model(\n        BACKBONE_NAME, pretrained=False, in_chans=IN_CHANS, num_classes=19\n    )\n    load_model(model_path, model, GPU)\n    return model.cuda(GPU)\n\n\ndef load_model(model_path, model, rank):\n    # print(f\"loading... {model_path}\")\n    model.load_state_dict(torch.load(model_path, map_location=torch.device(rank)))\n    return\n\n\ndef predict(models, test_loader, rank):\n    [model.eval() for model in models]\n    pred_list1 = []\n    with torch.no_grad():\n        for images in tqdm(test_loader):\n            images = images.cuda(rank)\n            pred_list2 = []\n            for model in models:\n                preds = model(images).cpu().sigmoid()\n\n                pred_list2.append(preds.numpy())\n\n            pred_list1.append(np.stack(pred_list2, 0).mean(0))\n\n        preds = np.concatenate(pred_list1)\n    # torch.cuda.empty_cache()\n    return preds\n\n\ndef predict_weighted(models, test_loader, rank, weights:np.ndarray=None):\n    [model.eval() for model in models]\n    if weights is None:\n        weights = np.ones(len(models)).reshape(-1 , 1, 1)\n    pred_list1 = []\n    with torch.no_grad():\n        for images in tqdm(test_loader):\n            images = images.cuda(rank)\n            pred_list2 = []\n            for i, model in enumerate(models):\n                preds = model(images).cpu().sigmoid().numpy()\n                if i < LEN_1ST:\n                    preds[:, 11] = 0\n                pred_list2.append(preds)\n            stack = np.stack(pred_list2, 0)\n            preds = np.multiply(stack, weights).mean(0)\n            pred_list1.append(preds)\n        preds = np.concatenate(pred_list1)\n    # torch.cuda.empty_cache()\n    return preds\n\n\ndef get_test_shortage(test):\n    test_shortage = sample_submission[~sample_submission[\"ID\"].isin(test[\"image_id\"])]\n    return test_shortage\n\n\ndef write_submission(test, fill_shortage=False):\n    if test.isnull().sum().sum() > 0:\n        sample_submission.to_csv(\"submission.csv\", index=False)\n        return\n\n    gr = test[[\"image_id\", \"w\", \"h\", \"encoded_mask\"] + COLS_TARGET].groupby(\"image_id\")\n\n    if fill_shortage:\n        test_shortage = get_test_shortage(test)\n        len_test = test[\"image_id\"].nunique()\n        len_test_shortage = test_shortage[\"ID\"].nunique()\n        print(f\"fill shortage: {len_test} => {len_test + len_test_shortage}\")\n        test_shortage = get_test_shortage(test)\n    else:\n        test_shortage = []\n        if len(sample_submission) == 559:\n            pass\n        elif sample_submission[\"ID\"].isin(test[\"image_id\"]).mean() != 1.0:\n            sample_submission.to_csv(\"submission.csv\", index=False)\n            return\n\n    classes = list(map(str, range(19)))\n\n    with open(\"submission.csv\", \"w\") as outf:\n        print(\"ID,ImageWidth,ImageHeight,PredictionString\", file=outf)\n        for image_id, df in gr:\n            if len(df) == 0:\n                continue\n            w = df.iloc[0, 1]\n            h = df.iloc[0, 2]\n\n            pred_strs = []\n            for i, row in df.iterrows():\n                cnfs = [row[c] for c in COLS_TARGET]\n                emasks = [row[\"encoded_mask\"]] * len(cnfs)\n                pred_strs += list(zip(classes, cnfs, emasks))\n\n            pred_strs = sorted(pred_strs, key=itemgetter(1), reverse=True)\n            pred_strs = \" \".join(map(lambda x: f\"{x[0]} {x[1]} {x[2]}\", pred_strs))\n\n            print(f\"{image_id},{w},{h},{pred_strs}\", file=outf)\n\n        if len(test_shortage) > 0:\n            for i, row in test_shortage.iterrows():\n                print(\n                    f\"{row['ID']},{row['ImageWidth']},{row['ImageHeight']},{row['PredictionString']}\",\n                    file=outf,\n                )\n\n    return\n\n\ndef post_process1(test):\n    len1 = len(test)\n\n    keep_condition = (\n        (test[\"cell_area_ratio\"] > KEEP_CELL_AREA_MIN) &\n        (test[\"nuc_area_ratio\"] > KEEP_NUC_AREA_MIN)\n    )\n    test = test[keep_condition].reset_index(drop=True)\n\n    # edge\n    edge = (\n        (test['y0'].between(0,1)) |\n        (test['h'] - test['y1']).between(0,1) |\n        (test['x0'].between(0,1)) |\n        (test['w'] - test['x1']).between(0,1)\n    )\n    drop_condition = (edge &\n      (test[\"cell_area_ratio\"] < KEEP_EDGE_CELL_AREA_MIN)\n      )\n    test = test[~drop_condition].reset_index(drop=True)\n\n    len2 = len(test)\n\n    print(f\"remove cell with \")\n    print(f\"cell_area_ratio({KEEP_CELL_AREA_MIN:.6f}) and \")\n    print(f\"nuc_area_ratio ({KEEP_NUC_AREA_MIN:.6f}), \")\n    print(f\"small cell in edge({KEEP_EDGE_CELL_AREA_MIN:.6f}), \")\n    print(f\"{len1} => {len2}\")\n\n    return test\n\n\ndef post_process2(test):\n    condition = (test[\"nuc_area_ratio\"] / test[\"cell_area_ratio\"]) < NUC_AREA_MIN_0to5\n\n    test.loc[condition, COLS_TARGET[:6]] = test.loc[condition, COLS_TARGET[:6]] * 0.5\n\n    print(\n        f\"decrease conf with NUC_AREA_MIN_0to5({NUC_AREA_MIN_0to5:.6f}) / \"\n        f\"subject to update: {condition.sum()}\"\n    )\n\n    return test\n\n\n# def post_process3(test):\n#     # groupby image id -> mean\n#     image_mean = test.groupby(\"image_id\")[COLS_TARGET].transform('mean')\n#     image_mean[\"label_11\"] = test[\"label_11\"]\n#     w_cell = WEIGHT_CELL_LEVEL_VS_IMAGE_MEAN[0]\n#     w_image = WEIGHT_CELL_LEVEL_VS_IMAGE_MEAN[1]\n\n#     test[COLS_TARGET] = w_cell * test[COLS_TARGET] + w_image * image_mean\n\n#     print(\n#         \"Add image level prediction (aggregating-average of image_id) to cell level prediction\"\n#     )\n#     print(f\"cell-level-pred : image-level-pred = {w_cell} : {w_image}\")\n\n#     return test\n\n\ndef post_process3(test, test_image):\n    # groupby image id -> mean\n    image_mean = test.groupby(\"image_id\")[COLS_TARGET].transform('mean')\n    image_mean[\"label_11\"] = test[\"label_11\"]\n\n    # match shape of image_level_prediction to cell_level_prediction\n    image_pred = test.loc[:, [\"image_id\"]].merge(test_image, on=\"image_id\", how=\"left\")[COLS_TARGET]\n    image_pred[\"label_11\"] = test[\"label_11\"]\n\n    w_cell = WEIGHT_CELL_LEVEL_VS_IMAGE_MEAN_VS_IMAGE_PRED[0]\n    w_image_mean = WEIGHT_CELL_LEVEL_VS_IMAGE_MEAN_VS_IMAGE_PRED[1]\n    w_image_pred = WEIGHT_CELL_LEVEL_VS_IMAGE_MEAN_VS_IMAGE_PRED[2]\n\n    test[COLS_TARGET] = w_cell * test[COLS_TARGET] + w_image_mean * image_mean + w_image_pred * image_pred\n\n    print(\n        \"Add image level prediction (aggregating-average of image_id) to cell level prediction\"\n    )\n    print(f\"cell-level-pred : image-level-mean : image-level-pred = {w_cell} : {w_image_mean} : {w_image_pred}\")\n\n    return test","metadata":{"_kg_hide-input":true,"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T09:51:56.22821Z","iopub.execute_input":"2025-07-24T09:51:56.228422Z","iopub.status.idle":"2025-07-24T09:51:56.276496Z","shell.execute_reply.started":"2025-07-24T09:51:56.228397Z","shell.execute_reply":"2025-07-24T09:51:56.275179Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 1st Stage","metadata":{}},{"cell_type":"code","source":"# ✅ Step 1：构造 sample_submission.csv（只包含 private 图像 ID）\n\nimport os\nimport pandas as pd\nimport cv2\nfrom tqdm import tqdm\nimport random\n\ntrain_dir = \"/kaggle/input/hpa-single-cell-image-classification/train\"\npublic_path = \"/kaggle/input/hpa2021-libs/public_sample_submission.csv\"\nused_path = \"/kaggle/input/hpa2021-p-all-1stx8-2ndx16-imgx16/train_bbox_filtered.csv\"  # ✅ 你提供的 CSV 路径\n\n# ✅ 读取 public ID 列表\npublic_df = pd.read_csv(public_path)\npublic_ids = set(public_df[\"ID\"].unique())\n\n# ✅ 获取所有训练图像 ID\nall_image_ids = sorted(set(fname.split(\"_\")[0] for fname in os.listdir(train_dir)))\n\n# ✅ 过滤掉 public ID，保留 private 图像\nprivate_image_ids = [iid for iid in all_image_ids if iid not in public_ids]\n\n# ✅ 读取已使用过的 ID 并去重\nused_df = pd.read_csv(used_path)\nused_ids = set(used_df[\"image_id\"].unique())\n\n# ✅ 从 private ID 中排除已使用的，剩余可供抽样的 ID\navailable_ids = [iid for iid in private_image_ids if iid not in used_ids]\nprint(f\"🧹 从 private 图像中排除已使用的，共剩余 {len(available_ids)} 张\")\n\n# ✅ 随机抽取 10%\nproportion = 0.001\nnum_keep = int(len(private_image_ids) * proportion)  # 注意：用总数的10%而不是剩下的10%\nrandom.seed(42)\nimage_ids = random.sample(available_ids, num_keep)\n\nprint(f\"⚙️ 最终随机抽样 private 图像 {len(image_ids)} 张\")\n\n# ✅ 构建 CSV\ndef get_image_size(image_id):\n    img_path = f\"{train_dir}/{image_id}_red.png\"\n    img = cv2.imread(img_path, cv2.IMREAD_UNCHANGED)\n    return img.shape[1], img.shape[0]\n\ndata = []\nfor image_id in tqdm(image_ids, desc=\"🛠️ 构建 submission 数据\"):\n    width, height = get_image_size(image_id)\n    data.append({\n        \"ID\": image_id,\n        \"ImageWidth\": width,\n        \"ImageHeight\": height,\n        \"PredictionString\": \"\"\n    })\n\ndf_sub = pd.DataFrame(data)\ndf_sub.to_csv(\"/kaggle/working/sample_submission.csv\", index=False)\nprint(f\"✅ 保存 sample_submission.csv，共 {len(df_sub)} 行\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T09:56:14.760854Z","iopub.execute_input":"2025-07-24T09:56:14.761229Z","iopub.status.idle":"2025-07-24T09:56:18.474389Z","shell.execute_reply.started":"2025-07-24T09:56:14.761196Z","shell.execute_reply":"2025-07-24T09:56:18.473456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndf = pd.read_csv(\"/kaggle/working/sample_submission.csv\")\nprint(f\"✅ 共读取 {len(df)} 行\")\ndf.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T09:56:18.476694Z","iopub.execute_input":"2025-07-24T09:56:18.476965Z","iopub.status.idle":"2025-07-24T09:56:18.503077Z","shell.execute_reply.started":"2025-07-24T09:56:18.476934Z","shell.execute_reply":"2025-07-24T09:56:18.502107Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"这个一会儿可以替换成全部的","metadata":{}},{"cell_type":"code","source":"\n# ✅ Step 3：执行 py 文件，读取 working 区构造好的 sample_submission.csv\n!python generate_train_mask_kernel_faster7.py --compute_private\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-24T09:56:18.504119Z","iopub.execute_input":"2025-07-24T09:56:18.504312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/working/sample_submission.csv\")\npublic_df = pd.read_csv(\"/kaggle/input/hpa2021-libs/public_sample_submission.csv\")\n\npub_ids = set(public_df[\"ID\"].values)\nprint(\"✅ 当前 sample_submission.csv 中 public ID 的数量：\", df[\"ID\"].isin(pub_ids).sum())\nprint(\"✅ 当前 sample_submission.csv 中 private ID 的数量：\", (~df[\"ID\"].isin(pub_ids)).sum())\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/working/sample_submission.csv\")\nprint(\"🧾 sample_submission.csv 行数：\", len(df))\nprint(df[\"ImageWidth\"].value_counts())\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 显示全部切割的结果","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(MASK_DIR / \"train_bbox.csv\")\ntrain","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 进行筛选（去掉小细胞、边缘细胞） 进行结果展示","metadata":{}},{"cell_type":"code","source":"train = post_process1(train)\ntrain","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ 保存为 CSV（可在右侧 Output 面板下载）\ntrain.to_csv(\"/kaggle/working/train_bbox_filtered.csv\", index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 读取筛选后的结果（post_process1 后）\nn_images = train[\"image_id\"].nunique()\nunique_ids = train[\"image_id\"].unique()\n\nprint(f\"共生成了 {n_images} 张图像的掩膜\")\nprint(\"部分 image_id 示例:\", unique_ids[:5])\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom pathlib import Path\nimport shutil\n\nsrc_dir = Path.cwd().parent / \"temp\" / \"mask\" / \"train\" / \"cell\"\ndst_dir = Path(\"/kaggle/working/mask/train/cell\")\n\ndst_dir.mkdir(parents=True, exist_ok=True)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 复制 .npz 掩膜文件到 working\nmask_files = list(src_dir.glob(\"*.npz\"))\n\nfor file in mask_files:\n    shutil.copy(file, dst_dir / file.name)\n\nprint(f\"✅ 已成功复制 {len(mask_files)} 个 mask 文件到 {dst_dir}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 可视化mask","metadata":{}},{"cell_type":"markdown","source":"可能有一些肉眼看起来不清楚，但是不影响切割的结果","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport cv2\nimport pandas as pd\nfrom pathlib import Path\n\n# ✅ 设置路径\nimage_root = \"/kaggle/input/hpa-single-cell-image-classification/train\"\nmask_root = \"/kaggle/working/mask/train/cell\"  # 你前面复制过来的 .npz 掩膜文件\ntrain_csv_path = Path.cwd().parent / \"temp\" / \"mask\" / \"train_bbox.csv\"\n\n# ✅ 加载并 post-process 数据（如果你已经做过这一步可以跳过）\ntrain = pd.read_csv(train_csv_path)\ntrain = train[(train[\"cell_area_ratio\"] > 0.005) & (train[\"nuc_area_ratio\"] > 0.001)].reset_index(drop=True)\n\n# ✅ 伽马增强图像（解决亮度过曝）\ndef enhance_contrast(img, gamma=0.7):\n    img = img.astype(np.float32) / 255.0\n    img = np.power(img, gamma)\n    img = (img * 255).clip(0, 255).astype(np.uint8)\n    return img\n\n# ✅ 可视化函数（展示 RGB + Mask + Overlay）\ndef show_overlay(image_id, cell_id):\n    r = cv2.imread(f\"{image_root}/{image_id}_red.png\", cv2.IMREAD_UNCHANGED)\n    g = cv2.imread(f\"{image_root}/{image_id}_green.png\", cv2.IMREAD_UNCHANGED)\n    b = cv2.imread(f\"{image_root}/{image_id}_blue.png\", cv2.IMREAD_UNCHANGED)\n    rgb = np.stack([r, g, b], axis=-1)\n    rgb = enhance_contrast(rgb, gamma=0.7)\n\n    mask = np.load(f\"{mask_root}/{image_id}.npz\")[\"arr_0\"]\n    cell_mask = (mask == cell_id).astype(np.uint8)\n\n    plt.figure(figsize=(15, 5))\n    plt.subplot(1, 3, 1)\n    plt.imshow(rgb)\n    plt.title(\"Original Image\")\n    plt.axis(\"off\")\n\n    plt.subplot(1, 3, 2)\n    plt.imshow(cell_mask, cmap='gray')\n    plt.title(f\"Mask (cell_id={cell_id})\")\n    plt.axis(\"off\")\n\n    plt.subplot(1, 3, 3)\n    plt.imshow(rgb)\n    plt.imshow(cell_mask, alpha=0.7, cmap='Reds')\n    plt.title(\"Overlay\")\n    plt.axis(\"off\")\n\n    plt.tight_layout()\n    plt.show()\n\n# ✅ 控制展示：前 3 张图像，每张前 3 个细胞\nN = 3\nM = 3\nshown = 0\n\nfor image_id, group in train.groupby(\"image_id\"):\n    if shown >= M:\n        break\n    print(f\"🧬 图像: {image_id} - 展示前 {N} 个细胞\")\n    for _, row in group.head(N).iterrows():\n        show_overlay(image_id, int(row[\"cell_id\"]))\n    shown += 1\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ✅ 统计每张图像中有多少个 cell\nfrom collections import defaultdict\n\ncell_counts = {}\n\nfor image_id in train[\"image_id\"].unique()[:10]:\n    npz_path = Path(mask_root) / f\"{image_id}.npz\"\n    if npz_path.exists():\n        mask = np.load(npz_path)[\"arr_0\"]\n        unique_cells = np.unique(mask)\n        unique_cells = unique_cells[unique_cells != 0]  # 排除背景\n        cell_counts[image_id] = len(unique_cells)\n    else:\n        print(f\"⚠️ 没有找到对应掩膜文件: {image_id}\")\n\n# ✅ 打印结果（按 image_id 排序）\nprint(\"📊 每张图像包含的细胞数量：\\n\")\nfor image_id in sorted(cell_counts.keys()):\n    print(f\"🧬 图像 {image_id}：{cell_counts[image_id]} 个 cell\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 想要运行所有的cell mask","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# 读取 CSV\nsample_path = \"/kaggle/input/hpa-single-cell-image-classification/sample_submission.csv\"\ndf = pd.read_csv(sample_path)\n\n# 统计唯一 ID 数量\nunique_ids = df[\"ID\"].unique()\nn_unique = len(unique_ids)\n\nprint(f\"📦 sample_submission.csv 中共包含 {n_unique} 个唯一图像 ID\")\nprint(\"🔍 前 5 个 image ID 示例：\", unique_ids[:5])\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# 文件路径\nfiltered_csv = \"/kaggle/working/train_bbox_filtered.csv\"\ntrain_csv = \"/kaggle/input/hpa-single-cell-image-classification/train.csv\"\n\n# 读取文件\ndf_filtered = pd.read_csv(filtered_csv)\ndf_train = pd.read_csv(train_csv)\n\n# 如果列名是 image_id，转换为统一 ID\nif \"image_id\" in df_filtered.columns:\n    df_filtered.rename(columns={\"image_id\": \"ID\"}, inplace=True)\nif \"Image\" in df_train.columns:\n    df_train.rename(columns={\"Image\": \"ID\"}, inplace=True)\n\n# 提取 ID 集合\nfiltered_ids = set(df_filtered[\"ID\"].unique())\ntrain_ids = set(df_train[\"ID\"].unique())\n\n# 计算交集、差集\nshared_ids = filtered_ids & train_ids\nonly_in_filtered = filtered_ids - train_ids\nonly_in_train = train_ids - filtered_ids\n\n# 输出统计信息\nprint(f\"📁 train_bbox_filtered.csv 中 ID 总数: {len(filtered_ids)}\")\nprint(f\"📁 train.csv 中 ID 总数: {len(train_ids)}\")\nprint(f\"🔁 重合 ID 数量: {len(shared_ids)}\")\nprint(f\"📉 重合比例（相对于 filtered）: {len(shared_ids)/len(filtered_ids)*100:.2f}%\")\nprint(f\"📉 重合比例（相对于 train）   : {len(shared_ids)/len(train_ids)*100:.2f}%\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}