{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## summary\n\n* 2.5d segmentation\n    *  segmentation_models_pytorch \n    *  Unet\n* use only 6 slices\n* slide inference\n* add rotate TTA","metadata":{}},{"cell_type":"code","source":"import hashlib\nimport pandas as pd\ndata_dir = '/kaggle/input/vesuvius-challenge-ink-detection/test'\n\nskip_fake_test = True\n\n## code to skip test if commit run:\na_file = f'{data_dir}/a/mask.png'\nwith open(a_file,'rb') as f:\n    hash_md5 = hashlib.md5(f.read()).hexdigest()\nis_skip_test = hash_md5 == '0b0fffdc0e88be226673846a143bb3e0'\nprint('is_skip_test:',is_skip_test) ## bu demek ki, test ici bos, daha dogrusu dummy test var.\n\nif skip_fake_test and is_skip_test:\n    skipp = True\n    print('skipping test')\n    submit_df = pd.DataFrame({'Id': ['a', 'b'], 'Predicted':['1 2', '1 2']})\n    submit_df.to_csv('submission.csv', index=False)\nelse: skipp = False\n    \nprint(skipp)","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:14:59.885251Z","iopub.execute_input":"2023-06-14T17:14:59.885637Z","iopub.status.idle":"2023-06-14T17:14:59.908116Z","shell.execute_reply.started":"2023-06-14T17:14:59.885602Z","shell.execute_reply":"2023-06-14T17:14:59.907160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\ntorch.__version__","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:14:59.909648Z","iopub.execute_input":"2023-06-14T17:14:59.910401Z","iopub.status.idle":"2023-06-14T17:15:01.509663Z","shell.execute_reply.started":"2023-06-14T17:14:59.910368Z","shell.execute_reply":"2023-06-14T17:15:01.508596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import DataLoader\nfrom torch.cuda.amp import autocast, GradScaler\nimport sys\nimport time\nimport torch as tc\nimport random\nfrom torch.utils.data import DataLoader, Dataset\nimport torch\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport numpy as np\n\nfrom tqdm.auto import tqdm\n\nimport torch\nimport torch.nn as nn\nfrom torch.optim import AdamW\n\nfrom torch.utils.data import DataLoader, Dataset\nimport matplotlib.pyplot as plt\nimport cv2,gc\nimport os,warnings\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:01.511224Z","iopub.execute_input":"2023-06-14T17:15:01.511995Z","iopub.status.idle":"2023-06-14T17:15:03.539135Z","shell.execute_reply.started":"2023-06-14T17:15:01.511958Z","shell.execute_reply":"2023-06-14T17:15:03.538166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    import segmentation_models_pytorch as smp\nexcept:   \n    !pip install --no-index --find-links=\"/kaggle/input/segmentation-models-pytorch-offline-installation/\" segmentation-models-pytorch\n    import segmentation_models_pytorch as smp","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:03.545738Z","iopub.execute_input":"2023-06-14T17:15:03.548221Z","iopub.status.idle":"2023-06-14T17:15:22.053078Z","shell.execute_reply.started":"2023-06-14T17:15:03.548185Z","shell.execute_reply":"2023-06-14T17:15:22.052052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## config","metadata":{}},{"cell_type":"code","source":"import os\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\nclass CFG:\n    # ============== comp exp name =============\n    comp_name = 'vesuvius'\n\n    # comp_dir_path = './'\n    comp_dir_path = '/kaggle/input/'\n    comp_folder_name = 'vesuvius-challenge-ink-detection'\n    # comp_dataset_path = f'{comp_dir_path}datasets/{comp_folder_name}/'\n    comp_dataset_path = f'{comp_dir_path}{comp_folder_name}/'\n    \n    exp_name = 'vesuvius_2d_slide_exp002'\n\n    # ============== pred target =============\n    target_size = 1\n    TTA=True\n    \n    # ============== model cfg =============\n    model_name = 'Unet'\n    # backbone = 'efficientnet-b0'\n    backbone = 'se_resnext50_32x4d'\n\n    in_chans = 6 # 65\n    # ============== training cfg =============\n    size = 448\n    tile_size = 448\n    stride = tile_size // 2\n\n    batch_size = 16 # 32\n    use_amp = True\n\n    scheduler = 'GradualWarmupSchedulerV2'\n    # scheduler = 'CosineAnnealingLR'\n    epochs = 15\n\n    warmup_factor = 10\n    lr = 1e-4 / warmup_factor\n\n    # ============== fold =============\n    valid_id = 2\n\n    objective_cv = 'binary'  # 'binary', 'multiclass', 'regression'\n    metric_direction = 'maximize'  # maximize, 'minimize'\n    # metrics = 'dice_coef'\n\n    # ============== fixed =============\n    pretrained = True\n    inf_weight = 'best'  # 'best'\n\n    min_lr = 1e-6\n    weight_decay = 1e-6\n    max_grad_norm = 1000\n\n    print_freq = 50\n    num_workers = 4\n\n    seed = 42\n\n    # ============== augmentation =============\n    train_aug_list = [\n        # A.RandomResizedCrop(\n        #     size, size, scale=(0.85, 1.0)),\n        A.Resize(size, size),\n        A.HorizontalFlip(p=0.5),\n        A.VerticalFlip(p=0.5),\n        A.RandomBrightnessContrast(p=0.75),\n        A.ShiftScaleRotate(p=0.75),\n        A.OneOf([\n                A.GaussNoise(var_limit=[10, 50]),\n                A.GaussianBlur(),\n                A.MotionBlur(),\n                ], p=0.4),\n        A.GridDistortion(num_steps=5, distort_limit=0.3, p=0.5),\n        A.CoarseDropout(max_holes=1, max_width=int(size * 0.3), max_height=int(size * 0.3), \n                        mask_fill_value=0, p=0.5),\n        # A.Cutout(max_h_size=int(size * 0.6),\n        #          max_w_size=int(size * 0.6), num_holes=1, p=1.0),\n        A.Normalize(\n            mean= [0] * in_chans,\n            std= [1] * in_chans\n        ),\n        ToTensorV2(transpose_mask=True),\n    ]\n\n    valid_aug_list = [\n        A.Resize(size, size),\n        A.Normalize(\n            mean= [0] * in_chans,\n            std= [1] * in_chans\n        ),\n        ToTensorV2(transpose_mask=True),\n    ]\n","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:22.054992Z","iopub.execute_input":"2023-06-14T17:15:22.055412Z","iopub.status.idle":"2023-06-14T17:15:22.071480Z","shell.execute_reply.started":"2023-06-14T17:15:22.055372Z","shell.execute_reply":"2023-06-14T17:15:22.070251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IS_DEBUG = False\nmode = 'train' if IS_DEBUG else 'test'\nTH = 0.25","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:22.073097Z","iopub.execute_input":"2023-06-14T17:15:22.073481Z","iopub.status.idle":"2023-06-14T17:15:22.084766Z","shell.execute_reply.started":"2023-06-14T17:15:22.073447Z","shell.execute_reply":"2023-06-14T17:15:22.083691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:22.087967Z","iopub.execute_input":"2023-06-14T17:15:22.088398Z","iopub.status.idle":"2023-06-14T17:15:22.119215Z","shell.execute_reply.started":"2023-06-14T17:15:22.088371Z","shell.execute_reply":"2023-06-14T17:15:22.118291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## helper","metadata":{}},{"cell_type":"code","source":"# ref.: https://www.kaggle.com/stainsby/fast-tested-rle\ndef rle(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels = img.flatten()\n    # pixels = (pixels >= thr).astype(int)\n    \n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:22.120855Z","iopub.execute_input":"2023-06-14T17:15:22.122089Z","iopub.status.idle":"2023-06-14T17:15:22.133998Z","shell.execute_reply.started":"2023-06-14T17:15:22.122055Z","shell.execute_reply":"2023-06-14T17:15:22.133077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## dataset","metadata":{}},{"cell_type":"code","source":"def read_image(fragment_id):\n    images = []\n\n    # idxs = range(65)\n    mid = 65 // 2\n    start = mid - CFG.in_chans // 2\n    end = mid + CFG.in_chans // 2\n    idxs = range(start, end)\n\n    for i in tqdm(idxs):\n        \n        image = cv2.imread(CFG.comp_dataset_path + f\"{mode}/{fragment_id}/surface_volume/{i:02}.tif\", 0)\n\n        pad0 = (CFG.tile_size - image.shape[0] % CFG.tile_size)\n        pad1 = (CFG.tile_size - image.shape[1] % CFG.tile_size)\n\n        image = np.pad(image, [(0, pad0), (0, pad1)], constant_values=0)\n\n        images.append(image)\n    images = np.stack(images, axis=2)\n    \n    return images","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:22.135649Z","iopub.execute_input":"2023-06-14T17:15:22.136074Z","iopub.status.idle":"2023-06-14T17:15:22.144780Z","shell.execute_reply.started":"2023-06-14T17:15:22.136042Z","shell.execute_reply":"2023-06-14T17:15:22.144064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_transforms(data, cfg):\n    if data == 'train':\n        aug = A.Compose(cfg.train_aug_list)\n    elif data == 'valid':\n        aug = A.Compose(cfg.valid_aug_list)\n\n    # print(aug)\n    return aug\n\nclass CustomDataset(Dataset):\n    def __init__(self, images, cfg, labels=None, transform=None):\n        self.images = images\n        self.cfg = cfg\n        self.labels = labels\n        self.transform = transform\n\n    def __len__(self):\n        # return len(self.xyxys)\n        return len(self.images)\n\n    def __getitem__(self, idx):\n        # x1, y1, x2, y2 = self.xyxys[idx]\n        image = self.images[idx]\n        data = self.transform(image=image)\n        image = data['image']\n        return image\n","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:22.145783Z","iopub.execute_input":"2023-06-14T17:15:22.147337Z","iopub.status.idle":"2023-06-14T17:15:22.157792Z","shell.execute_reply.started":"2023-06-14T17:15:22.147304Z","shell.execute_reply":"2023-06-14T17:15:22.156931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_test_dataset(fragment_id):\n    test_images = read_image(fragment_id)\n    \n    x1_list = list(range(0, test_images.shape[1]-CFG.tile_size+1, CFG.stride))\n    y1_list = list(range(0, test_images.shape[0]-CFG.tile_size+1, CFG.stride))\n    \n    test_images_list = []\n    xyxys = []\n    for y1 in y1_list:\n        for x1 in x1_list:\n            y2 = y1 + CFG.tile_size\n            x2 = x1 + CFG.tile_size\n            \n            test_images_list.append(test_images[y1:y2, x1:x2])\n            xyxys.append((x1, y1, x2, y2))\n    xyxys = np.stack(xyxys)\n            \n    test_dataset = CustomDataset(test_images_list, CFG, transform=get_transforms(data='valid', cfg=CFG))\n    \n    test_loader = DataLoader(test_dataset,\n                          batch_size=CFG.batch_size,\n                          shuffle=False,\n                          num_workers=CFG.num_workers, pin_memory=True, drop_last=False)\n    \n    return test_loader, xyxys","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:22.161388Z","iopub.execute_input":"2023-06-14T17:15:22.161656Z","iopub.status.idle":"2023-06-14T17:15:22.170828Z","shell.execute_reply.started":"2023-06-14T17:15:22.161633Z","shell.execute_reply":"2023-06-14T17:15:22.169847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## model","metadata":{}},{"cell_type":"code","source":"class CustomModel(nn.Module):\n    def __init__(self, cfg, weight=None):\n        super().__init__()\n        self.cfg = cfg\n\n        self.encoder = smp.Unet(\n            encoder_name=cfg.backbone, \n            encoder_weights=weight,\n            in_channels=cfg.in_chans,\n            classes=cfg.target_size,\n            activation=None,\n        )\n\n    def forward(self, image):\n        output = self.encoder(image)\n        output = output.squeeze(-1)\n        return output\n\ndef build_model(cfg, weight=\"imagenet\"):\n    print('model_name', cfg.model_name)\n    print('backbone', cfg.backbone)\n\n    model = CustomModel(cfg, weight)\n    return model\n","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:22.172340Z","iopub.execute_input":"2023-06-14T17:15:22.172727Z","iopub.status.idle":"2023-06-14T17:15:22.184158Z","shell.execute_reply.started":"2023-06-14T17:15:22.172689Z","shell.execute_reply":"2023-06-14T17:15:22.183254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class EnsembleModel(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.model = nn.ModuleList()\n        for fold in [1, 2, 3]:\n            _model = build_model(CFG, weight=None)\n            #_model.to(device)\n\n            model_path = f'/kaggle/input/vesuvius-models-public/{CFG.exp_name}/vesuvius-models/Unet_fold{fold}_best.pth'\n            state = torch.load(model_path)['model']\n            _model.load_state_dict(state)\n            _model.eval()\n\n            self.model.append(_model)\n    \n    def forward(self,x):\n        output=[]\n        for m in self.model:\n            output.append(m(x))\n        output=torch.stack(output,dim=0).mean(0)\n        return output\n        \n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:22.190845Z","iopub.execute_input":"2023-06-14T17:15:22.191203Z","iopub.status.idle":"2023-06-14T17:15:22.198879Z","shell.execute_reply.started":"2023-06-14T17:15:22.191168Z","shell.execute_reply":"2023-06-14T17:15:22.197429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def TTA(x:tc.Tensor,model:nn.Module):\n    #x.shape=(batch,c,h,w)\n    if CFG.TTA:\n        shape=x.shape\n        x=[x,*[tc.rot90(x,k=i,dims=(-2,-1)) for i in range(1,4)]]\n        x=tc.cat(x,dim=0)\n        x=model(x)\n        x=torch.sigmoid(x)\n        x=x.reshape(4,shape[0],*shape[2:])\n        x=[tc.rot90(x[i],k=-i,dims=(-2,-1)) for i in range(4)]\n        x=tc.stack(x,dim=0)\n        return x.mean(0)\n    else :\n        x=model(x)\n        x=torch.sigmoid(x)\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:22.200649Z","iopub.execute_input":"2023-06-14T17:15:22.201189Z","iopub.status.idle":"2023-06-14T17:15:22.213087Z","shell.execute_reply.started":"2023-06-14T17:15:22.201158Z","shell.execute_reply":"2023-06-14T17:15:22.212171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if mode == 'test':\n    fragment_ids = sorted(os.listdir(CFG.comp_dataset_path + mode))\nelse:\n    fragment_ids = [3]","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:22.214677Z","iopub.execute_input":"2023-06-14T17:15:22.215053Z","iopub.status.idle":"2023-06-14T17:15:22.224182Z","shell.execute_reply.started":"2023-06-14T17:15:22.215021Z","shell.execute_reply":"2023-06-14T17:15:22.223267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = EnsembleModel()\n# model = nn.DataParallel(model) # nn.DataParallel(model, device_ids=[0, 1])\nmodel = model.cuda()","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:22.225690Z","iopub.execute_input":"2023-06-14T17:15:22.226098Z","iopub.status.idle":"2023-06-14T17:15:57.919085Z","shell.execute_reply.started":"2023-06-14T17:15:22.225996Z","shell.execute_reply":"2023-06-14T17:15:57.913248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## main","metadata":{}},{"cell_type":"code","source":"results = []\nif not skipp:\n    for fragment_id in fragment_ids:\n\n        test_loader, xyxys = make_test_dataset(fragment_id)\n\n        binary_mask = cv2.imread(CFG.comp_dataset_path + f\"{mode}/{fragment_id}/mask.png\", 0)\n        binary_mask = (binary_mask / 255).astype(int)\n\n        ori_h = binary_mask.shape[0]\n        ori_w = binary_mask.shape[1]\n        # mask = mask / 255\n\n        pad0 = (CFG.tile_size - binary_mask.shape[0] % CFG.tile_size)\n        pad1 = (CFG.tile_size - binary_mask.shape[1] % CFG.tile_size)\n\n        binary_mask = np.pad(binary_mask, [(0, pad0), (0, pad1)], constant_values=0)\n\n        mask_pred = np.zeros(binary_mask.shape)\n        mask_count = np.zeros(binary_mask.shape)\n\n        for step, (images) in tqdm(enumerate(test_loader), total=len(test_loader)):\n            images = images.cuda()\n            batch_size = images.size(0)\n\n            with torch.no_grad():\n                y_preds = TTA(images,model).cpu().numpy()\n\n            start_idx = step*CFG.batch_size\n            end_idx = start_idx + batch_size\n            for i, (x1, y1, x2, y2) in enumerate(xyxys[start_idx:end_idx]):\n                mask_pred[y1:y2, x1:x2] += y_preds[i].reshape(mask_pred[y1:y2, x1:x2].shape)\n                mask_count[y1:y2, x1:x2] += np.ones((CFG.tile_size, CFG.tile_size))\n\n\n        print(f'mask_count_min: {mask_count.min()}')\n        mask_pred /= mask_count\n\n        fig, axes = plt.subplots(1, 3, figsize=(15, 8))\n        axes[0].imshow(mask_count)\n        axes[1].imshow(mask_pred.copy())\n\n\n\n        mask_pred = mask_pred[:ori_h, :ori_w]\n        binary_mask = binary_mask[:ori_h, :ori_w]\n\n        mask_pred = (mask_pred >= TH).astype(int)\n        mask_pred *= binary_mask\n        axes[2].imshow(mask_pred)\n        plt.show()\n\n        inklabels_rle = rle(mask_pred)\n\n        results.append((fragment_id, inklabels_rle))\n\n\n        del mask_pred, mask_count\n        del test_loader\n\n        gc.collect()\n        torch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.920540Z","iopub.status.idle":"2023-06-14T17:15:57.921087Z","shell.execute_reply.started":"2023-06-14T17:15:57.920792Z","shell.execute_reply":"2023-06-14T17:15:57.920818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## submission","metadata":{}},{"cell_type":"code","source":"if not skipp:\n    sub = pd.DataFrame(results, columns=['Id', 'Predicted'])\n    sample_sub = pd.read_csv(CFG.comp_dataset_path + 'sample_submission.csv')\n    sample_sub = pd.merge(sample_sub[['Id']], sub, on='Id', how='left')\n    sample_sub.to_csv(\"submission.csv\", index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.922787Z","iopub.status.idle":"2023-06-14T17:15:57.923275Z","shell.execute_reply.started":"2023-06-14T17:15:57.923040Z","shell.execute_reply":"2023-06-14T17:15:57.923062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.924913Z","iopub.status.idle":"2023-06-14T17:15:57.925421Z","shell.execute_reply.started":"2023-06-14T17:15:57.925170Z","shell.execute_reply":"2023-06-14T17:15:57.925191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p submission_ensemble","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.927068Z","iopub.status.idle":"2023-06-14T17:15:57.927546Z","shell.execute_reply.started":"2023-06-14T17:15:57.927306Z","shell.execute_reply":"2023-06-14T17:15:57.927328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp submission.csv submission_ensemble/submission25.csv","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.934310Z","iopub.status.idle":"2023-06-14T17:15:57.935120Z","shell.execute_reply.started":"2023-06-14T17:15:57.934865Z","shell.execute_reply":"2023-06-14T17:15:57.934888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nfor f in glob.glob('*'):\n    if not f.startswith('submission_ensemble'):\n        !rm -rf {f}","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.936522Z","iopub.status.idle":"2023-06-14T17:15:57.937305Z","shell.execute_reply.started":"2023-06-14T17:15:57.937045Z","shell.execute_reply":"2023-06-14T17:15:57.937067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## birinci solution is over","metadata":{}},{"cell_type":"code","source":"%reset -f","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.938665Z","iopub.status.idle":"2023-06-14T17:15:57.939444Z","shell.execute_reply.started":"2023-06-14T17:15:57.939184Z","shell.execute_reply":"2023-06-14T17:15:57.939208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import hashlib\nimport pandas as pd\ndata_dir = '/kaggle/input/vesuvius-challenge-ink-detection/test'\n\n\nskip_fake_test = True\n\n## code to skip test if commit run:\na_file = f'{data_dir}/a/mask.png'\nwith open(a_file,'rb') as f:\n    hash_md5 = hashlib.md5(f.read()).hexdigest()\nis_skip_test = hash_md5 == '0b0fffdc0e88be226673846a143bb3e0'\nprint('is_skip_test:',is_skip_test) ## bu demek ki, test ici bos, daha dogrusu dummy test var.\n\nif skip_fake_test and is_skip_test:\n    skipp = True\n    print('skipping test')\n    submit_df = pd.DataFrame({'Id': ['a', 'b'], 'Predicted':['1 2', '1 2']})\n    submit_df.to_csv('submission.csv', index=False)\nelse: skipp = False\n    \nprint(skipp)","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.940843Z","iopub.status.idle":"2023-06-14T17:15:57.941661Z","shell.execute_reply.started":"2023-06-14T17:15:57.941396Z","shell.execute_reply":"2023-06-14T17:15:57.941421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import segmentation_models_pytorch as smp","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.943056Z","iopub.status.idle":"2023-06-14T17:15:57.943824Z","shell.execute_reply.started":"2023-06-14T17:15:57.943580Z","shell.execute_reply":"2023-06-14T17:15:57.943604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3d_resnet_size384_stride2_bs3","metadata":{}},{"cell_type":"code","source":"import os,cv2\nimport gc\nimport sys\nimport random\nfrom glob import glob\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.cuda import amp\nfrom torch.utils.data import Dataset, DataLoader\n\nsys.path.append(\"/kaggle/input/resnet3d\")\nfrom resnet3d import generate_model\nimport torch as tc","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.945182Z","iopub.status.idle":"2023-06-14T17:15:57.945937Z","shell.execute_reply.started":"2023-06-14T17:15:57.945698Z","shell.execute_reply":"2023-06-14T17:15:57.945720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    # ============== comp exp name =============\n    comp_name = 'vesuvius'\n\n    # comp_dir_path = './'\n    comp_dir_path = '/kaggle/input/'\n    comp_folder_name = 'vesuvius-challenge-ink-detection'\n    # comp_dataset_path = f'{comp_dir_path}datasets/{comp_folder_name}/'\n    comp_dataset_path = f'{comp_dir_path}{comp_folder_name}/'\n    \n    exp_name = 'vesuvius_2d_slide_exp002'\n\n    # ============== pred target =============\n    target_size = 1\n\n    # ============== model cfg =============\n    model_name = 'Unet'\n    #backbone = 'efficientnet-b5'\n    #backbone = 'mit_b5'\n    backbone = 'resnet3d'\n    #backbone = 'resnext50_32x4d'\n    pretrained = True\n\n    in_chans = 16 # 65\n    load_chans=16\n    # ============== training cfg =============\n    prd_size=384\n    stride = prd_size // 2\n\n    batch_size = 3 # 32\n    use_amp = True\n\n    seed = 42\n    num_workers=2","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.947391Z","iopub.status.idle":"2023-06-14T17:15:57.948154Z","shell.execute_reply.started":"2023-06-14T17:15:57.947894Z","shell.execute_reply":"2023-06-14T17:15:57.947917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ref.: https://www.kaggle.com/stainsby/fast-tested-rle\ndef rle(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels = img.flatten()\n    # pixels = (pixels >= thr).astype(int)\n    \n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\ndef normalization(x:tc.Tensor)->tc.Tensor:\n    \"\"\"input.shape=(batch,f1,f2,...)\"\"\"\n    #[batch,f1,f2]->dim[1,2]\n    dim=list(range(1,x.ndim))\n    mean=x.mean(dim=dim,keepdim=True)\n    std=x.std(dim=dim,keepdim=True)\n    return (x-mean)/(std+1e-9)\n\ndef get_folder_size(folder_path):\n    total_size = 0\n    for dirpath, dirnames, filenames in os.walk(folder_path):\n        for f in filenames:\n            fp = os.path.join(dirpath, f)\n            total_size += os.path.getsize(fp)\n    return total_size","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.949540Z","iopub.status.idle":"2023-06-14T17:15:57.950303Z","shell.execute_reply.started":"2023-06-14T17:15:57.950039Z","shell.execute_reply":"2023-06-14T17:15:57.950062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cupy as cp\nxp = cp\n\ndelta_lookup = {\n    \"xx\": xp.array([[1, -2, 1]], dtype=float),\n    \"yy\": xp.array([[1], [-2], [1]], dtype=float),\n    \"xy\": xp.array([[1, -1], [-1, 1]], dtype=float),\n}\n\ndef operate_derivative(img_shape, pair):\n    assert len(img_shape) == 2\n    delta = delta_lookup[pair]\n    fft = xp.fft.fftn(delta, img_shape)\n    return fft * xp.conj(fft)\n\ndef soft_threshold(vector, threshold):\n    return xp.sign(vector) * xp.maximum(xp.abs(vector) - threshold, 0)\n\ndef back_diff(input_image, dim):\n    assert dim in (0, 1)\n    r, n = xp.shape(input_image)\n    size = xp.array((r, n))\n    position = xp.zeros(2, dtype=int)\n    temp1 = xp.zeros((r+1, n+1), dtype=float)\n    temp2 = xp.zeros((r+1, n+1), dtype=float)\n    \n    temp1[position[0]:size[0], position[1]:size[1]] = input_image\n    temp2[position[0]:size[0], position[1]:size[1]] = input_image\n    \n    size[dim] += 1\n    position[dim] += 1\n    temp2[position[0]:size[0], position[1]:size[1]] = input_image\n    temp1 -= temp2\n    size[dim] -= 1\n    return temp1[0:size[0], 0:size[1]]\n\ndef forward_diff(input_image, dim):\n    assert dim in (0, 1)\n    r, n = xp.shape(input_image)\n    size = xp.array((r, n))\n    position = xp.zeros(2, dtype=int)\n    temp1 = xp.zeros((r+1, n+1), dtype=float)\n    temp2 = xp.zeros((r+1, n+1), dtype=float)\n        \n    size[dim] += 1\n    position[dim] += 1\n\n    temp1[position[0]:size[0], position[1]:size[1]] = input_image\n    temp2[position[0]:size[0], position[1]:size[1]] = input_image\n    \n    size[dim] -= 1\n    temp2[0:size[0], 0:size[1]] = input_image\n    temp1 -= temp2\n    size[dim] += 1\n    return -temp1[position[0]:size[0], position[1]:size[1]]\n\ndef iter_deriv(input_image, b, scale, mu, dim1, dim2):\n    g = back_diff(forward_diff(input_image, dim1), dim2)\n    d = soft_threshold(g + b, 1 / mu)\n    b = b + (g - d)\n    L = scale * back_diff(forward_diff(d - b, dim2), dim1)\n    return L, b\n\ndef iter_xx(*args):\n    return iter_deriv(*args, dim1=1, dim2=1)\n\ndef iter_yy(*args):\n    return iter_deriv(*args, dim1=0, dim2=0)\n\ndef iter_xy(*args):\n    return iter_deriv(*args, dim1=0, dim2=1)\n\ndef iter_sparse(input_image, bsparse, scale, mu):\n    d = soft_threshold(input_image + bsparse, 1 / mu)\n    bsparse = bsparse + (input_image - d)\n    Lsparse = scale * (d - bsparse)\n    return Lsparse, bsparse\n\ndef denoise_image(input_image, iter_num=100, fidelity=150, sparsity_scale=10, continuity_scale=0.5, mu=1):\n    image_size = xp.shape(input_image)\n    #print(\"Initialize denoising\")\n    norm_array = (\n        operate_derivative(image_size, \"xx\") + \n        operate_derivative(image_size, \"yy\") + \n        2 * operate_derivative(image_size, \"xy\")\n    )\n    norm_array += (fidelity / mu) + sparsity_scale ** 2\n    b_arrays = {\n        \"xx\": xp.zeros(image_size, dtype=float),\n        \"yy\": xp.zeros(image_size, dtype=float),\n        \"xy\": xp.zeros(image_size, dtype=float),\n        \"L1\": xp.zeros(image_size, dtype=float),\n    }\n    g_update = xp.multiply(fidelity / mu, input_image)\n    for i in tqdm(range(iter_num), total=iter_num):\n        #print(f\"Starting iteration {i+1}\")\n        g_update = xp.fft.fftn(g_update)\n        if i == 0:\n            g = xp.fft.ifftn(g_update / (fidelity / mu)).real\n        else:\n            g = xp.fft.ifftn(xp.divide(g_update, norm_array)).real\n        g_update = xp.multiply((fidelity / mu), input_image)\n        \n        #print(\"XX update\")\n        L, b_arrays[\"xx\"] = iter_xx(g, b_arrays[\"xx\"], continuity_scale, mu)\n        g_update += L\n        \n        #print(\"YY update\")\n        L, b_arrays[\"yy\"] = iter_yy(g, b_arrays[\"yy\"], continuity_scale, mu)\n        g_update += L\n        \n        #print(\"XY update\")\n        L, b_arrays[\"xy\"] = iter_xy(g, b_arrays[\"xy\"], 2 * continuity_scale, mu)\n        g_update += L\n        \n        #print(\"L1 update\")\n        L, b_arrays[\"L1\"] = iter_sparse(g, b_arrays[\"L1\"], sparsity_scale, mu)\n        g_update += L\n        \n    g_update = xp.fft.fftn(g_update)\n    g = xp.fft.ifftn(xp.divide(g_update, norm_array)).real\n    \n    g[g < 0] = 0\n    g -= g.min()\n    g /= g.max()\n    return g","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.951732Z","iopub.status.idle":"2023-06-14T17:15:57.952497Z","shell.execute_reply.started":"2023-06-14T17:15:57.952245Z","shell.execute_reply":"2023-06-14T17:15:57.952267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_image(fragment_id):\n    images = []\n\n    # idxs = range(65)\n    mid = 65 // 2\n    start = mid - CFG.in_chans // 2\n    end = mid + CFG.in_chans // 2\n    idxs = range(start, end)\n\n    for i in tqdm(idxs):\n        image = cv2.imread(CFG.comp_dataset_path + f\"{mode}/{fragment_id}/surface_volume/{i:02}.tif\", 0)\n\n        pad0 = (CFG.prd_size - image.shape[0] % CFG.prd_size)\n        pad1 = (CFG.prd_size - image.shape[1] % CFG.prd_size)\n\n        image = np.pad(image, [(0, pad0), (0, pad1)], constant_values=0)\n\n        images.append(image)\n    images = np.stack(images, axis=2)\n    \n    return images","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.953878Z","iopub.status.idle":"2023-06-14T17:15:57.954647Z","shell.execute_reply.started":"2023-06-14T17:15:57.954401Z","shell.execute_reply":"2023-06-14T17:15:57.954424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomDataset(Dataset):\n    def __init__(self, images, cfg,xys, labels=None):\n        self.images = images\n        self.cfg = cfg\n        self.labels = labels\n        self.xys=xys\n\n    def __len__(self):\n        # return len(self.xyxys)\n        return len(self.images)\n\n    def __getitem__(self, idx):\n        # x1, y1, x2, y2 = self.xyxys[idx]\n        image = self.images[idx]\n        image=tc.from_numpy(image).permute(2,0,1).to(tc.float32)/255\n        image = (image - 0.45)/0.225\n        return image,self.xys[idx]\n","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.955996Z","iopub.status.idle":"2023-06-14T17:15:57.956773Z","shell.execute_reply.started":"2023-06-14T17:15:57.956532Z","shell.execute_reply":"2023-06-14T17:15:57.956555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_test_dataset(fragment_id):\n    test_images = read_image(fragment_id)\n    \n    x1_list = list(range(0, test_images.shape[1]-CFG.prd_size+1, CFG.stride))\n    y1_list = list(range(0, test_images.shape[0]-CFG.prd_size+1, CFG.stride))\n    \n    test_images_list = []\n    xyxys = []\n    for y1 in y1_list:\n        for x1 in x1_list:\n            y2 = y1 + CFG.prd_size\n            x2 = x1 + CFG.prd_size\n            if np.all(test_images[y1:y2, x1:x2]==0):\n                continue\n            test_images_list.append(test_images[y1:y2, x1:x2])\n            xyxys.append((x1, y1, x2, y2))\n    xyxys = np.stack(xyxys)\n            \n    test_dataset = CustomDataset(test_images_list, CFG,xys=xyxys)\n    \n    test_loader = DataLoader(test_dataset,\n                          batch_size=CFG.batch_size,\n                          shuffle=False,\n                          num_workers=CFG.num_workers, pin_memory=True, drop_last=False)\n    \n    return test_loader, xyxys","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.958160Z","iopub.status.idle":"2023-06-14T17:15:57.958918Z","shell.execute_reply.started":"2023-06-14T17:15:57.958673Z","shell.execute_reply":"2023-06-14T17:15:57.958695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Decoder(nn.Module):\n    def __init__(self, encoder_dims, upscale):\n        super().__init__()\n        self.convs = nn.ModuleList([\n            nn.Sequential(\n                nn.Conv2d(encoder_dims[i]+encoder_dims[i-1], encoder_dims[i-1], 3, 1, 1, bias=False),\n                nn.BatchNorm2d(encoder_dims[i-1]),\n                nn.ReLU(inplace=True)\n            ) for i in range(1, len(encoder_dims))])\n\n        self.logit = nn.Conv2d(encoder_dims[0], 1, 1, 1, 0)\n        self.up = nn.Upsample(scale_factor=upscale, mode=\"bilinear\")\n\n    def forward(self, feature_maps):\n        for i in range(len(feature_maps)-1, 0, -1):\n            f_up = F.interpolate(feature_maps[i], scale_factor=2, mode=\"bilinear\")\n            f = torch.cat([feature_maps[i-1], f_up], dim=1)\n            f_down = self.convs[i-1](f)\n            feature_maps[i-1] = f_down\n\n        x = self.logit(feature_maps[0])\n        mask = self.up(x)\n        return mask\n    \nclass SegModel(nn.Module):\n    def __init__(self,model_depth=34):\n        super().__init__()\n        self.encoder = generate_model(model_depth=model_depth, n_input_channels=1)\n        self.decoder = Decoder(encoder_dims=[64, 128, 256, 512], upscale=4)\n        \n    def forward(self, x):\n        if x.ndim==4:\n            x=x[:,None]\n        \n        feat_maps = self.encoder(x)\n        feat_maps_pooled = [torch.mean(f, dim=2) for f in feat_maps]\n        pred_mask = self.decoder(feat_maps_pooled)\n        return pred_mask\n        \nclass CustomModel(nn.Module):\n    def __init__(self, cfg=CFG, weight=None):\n        super().__init__()\n        self.cfg = cfg\n\n        if cfg.backbone==\"resnet3d\":\n            self.encoder=SegModel()\n        elif cfg.backbone[:3]!=\"mit\":\n            self.encoder = smp.Unet(\n                encoder_name=cfg.backbone, \n                encoder_weights=weight,\n                in_channels=cfg.in_chans,\n                classes=cfg.target_size,\n                activation=None,\n            )\n        else :\n            self.encoder = smp.Unet(\n                encoder_name=cfg.backbone, \n                encoder_weights=weight,\n                classes=cfg.target_size,\n                activation=None,\n            )\n            print(self.encoder.encoder.patch_embed1.proj)\n            out_channels=self.encoder.encoder.patch_embed1.proj.out_channels\n            self.encoder.encoder.patch_embed1.proj=nn.Conv2d(cfg.in_chans,out_channels,7,4,3)\n\n    def forward(self, images:torch.Tensor):\n        #image.shape=(b,C,H,W)\n        if images.ndim==4:\n            images=images[:,None]\n        images=normalization(images)\n        output = self.encoder(images)\n        return output\n\ndef build_model(cfg, weight=\"imagenet\"):\n    print('model_name', cfg.model_name)\n    print('backbone', cfg.backbone)\n\n    model = CustomModel(cfg, weight)\n    return model\n","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.960337Z","iopub.status.idle":"2023-06-14T17:15:57.961082Z","shell.execute_reply.started":"2023-06-14T17:15:57.960833Z","shell.execute_reply":"2023-06-14T17:15:57.960855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def TTA(x:tc.Tensor,model:nn.Module):\n    #x.shape=(batch,c,h,w)\n    shape=x.shape\n    x=[x,*[tc.rot90(x,k=i,dims=(-2,-1)) for i in range(1,4)]]\n    x=tc.cat(x,dim=0)\n    x=model(x)\n    x=torch.sigmoid(x)\n    x=x.reshape(4,shape[0],*shape[2:])\n    x=[tc.rot90(x[i],k=-i,dims=(-2,-1)) for i in range(4)]\n    x=tc.stack(x,dim=0)\n    return x.mean(0)","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.962464Z","iopub.status.idle":"2023-06-14T17:15:57.963243Z","shell.execute_reply.started":"2023-06-14T17:15:57.962967Z","shell.execute_reply":"2023-06-14T17:15:57.963017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#in_submission=get_folder_size(\"/kaggle/input/vesuvius-challenge-ink-detection/test\")!=6732244267\nin_submission=True\nIS_DEBUG = False\nmode = 'train' if IS_DEBUG else 'test'\nTH = 0.55\nif mode == 'test':\n    fragment_ids = sorted(os.listdir(CFG.comp_dataset_path + mode))\nelse:\n    fragment_ids = [3]\n\nmodel = build_model(CFG)\nmodel.load_state_dict(tc.load(\"/kaggle/input/3d-resnet-baseline-inference-model-data/resnet3d-34_3d_seg_epoch_14.pth\"))\n# model = nn.DataParallel(model, device_ids=[0, 1])\nmodel = model.cuda()#.eval()\nmodel.training","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.964630Z","iopub.status.idle":"2023-06-14T17:15:57.965394Z","shell.execute_reply.started":"2023-06-14T17:15:57.965138Z","shell.execute_reply":"2023-06-14T17:15:57.965160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = []\nif not skipp:\n    for fragment_id in fragment_ids:\n        if not in_submission:\n            break\n        test_loader, xyxys = make_test_dataset(fragment_id)\n\n        binary_mask = cv2.imread(CFG.comp_dataset_path + f\"{mode}/{fragment_id}/mask.png\", 0)\n        binary_mask = (binary_mask / 255).astype(int)\n\n        ori_h = binary_mask.shape[0]\n        ori_w = binary_mask.shape[1]\n\n        pad0 = (CFG.prd_size - binary_mask.shape[0] % CFG.prd_size)\n        pad1 = (CFG.prd_size - binary_mask.shape[1] % CFG.prd_size)\n\n        binary_mask = np.pad(binary_mask, [(0, pad0), (0, pad1)], constant_values=0)\n\n        mask_pred = np.zeros(binary_mask.shape)\n        mask_count = np.zeros(binary_mask.shape)\n        for step, (images,xys) in tqdm(enumerate(test_loader), total=len(test_loader)):\n            images = images.cuda()\n            batch_size = images.size(0)\n\n            with torch.no_grad():\n                y_preds=TTA(images,model)\n\n\n            for k, (x1, y1, x2, y2) in enumerate(xys):\n                mask_pred[y1:y2, x1:x2] += y_preds[k].squeeze(0).cpu().numpy()\n                mask_count[y1:y2, x1:x2] += 1\n\n        print(f'mask_count_min: {mask_count.min()}')\n        mask_pred /= (mask_count+1e-7)\n\n\n        fig, axes = plt.subplots(1, 4, figsize=(15, 8))\n        axes[0].imshow(mask_count)\n        axes[1].imshow(mask_pred.copy())\n\n        mask_pred=xp.array(mask_pred)\n        mask_pred=denoise_image(mask_pred, iter_num=250)\n        mask_pred=mask_pred.get()\n        axes[2].imshow(mask_pred)\n\n        mask_pred = mask_pred[:ori_h, :ori_w]\n        binary_mask = binary_mask[:ori_h, :ori_w]\n\n        mask_pred = (mask_pred >= TH).astype(np.uint8)\n        mask_pred=mask_pred.astype(int)\n        mask_pred *= binary_mask\n\n        axes[3].imshow(mask_pred)\n        plt.show()\n\n        inklabels_rle = rle(mask_pred)\n\n        results.append((fragment_id, inklabels_rle))\n\n\n        del mask_pred, mask_count\n        del test_loader\n\n        gc.collect()\n        torch.cuda.empty_cache()\n        plt.clf()\n        fig.clear()\n        plt.close(fig)","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.966800Z","iopub.status.idle":"2023-06-14T17:15:57.967573Z","shell.execute_reply.started":"2023-06-14T17:15:57.967331Z","shell.execute_reply":"2023-06-14T17:15:57.967354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! cp /kaggle/input/vesuvius-challenge-ink-detection/sample_submission.csv submission.csv\nif not skipp:\n    sub = pd.DataFrame(results, columns=['Id', 'Predicted'])\n    #sub\n    sample_sub = pd.read_csv(CFG.comp_dataset_path + 'sample_submission.csv')\n    sample_sub = pd.merge(sample_sub[['Id']], sub, on='Id', how='left')\n    #sample_sub\n    sample_sub.to_csv(\"submission.csv\", index=False)\n    print(\"ok\")","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.968939Z","iopub.status.idle":"2023-06-14T17:15:57.969703Z","shell.execute_reply.started":"2023-06-14T17:15:57.969466Z","shell.execute_reply":"2023-06-14T17:15:57.969488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp submission.csv submission_ensemble/submission30.csv","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.971089Z","iopub.status.idle":"2023-06-14T17:15:57.971857Z","shell.execute_reply.started":"2023-06-14T17:15:57.971613Z","shell.execute_reply":"2023-06-14T17:15:57.971636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for f in glob('*'):\n    if not f.startswith('submission_ensemble'):\n        !rm -rf {f}","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.973229Z","iopub.status.idle":"2023-06-14T17:15:57.973992Z","shell.execute_reply.started":"2023-06-14T17:15:57.973746Z","shell.execute_reply":"2023-06-14T17:15:57.973768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## inheritance inference","metadata":{}},{"cell_type":"code","source":"%reset -f","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.975381Z","iopub.status.idle":"2023-06-14T17:15:57.976143Z","shell.execute_reply.started":"2023-06-14T17:15:57.975888Z","shell.execute_reply":"2023-06-14T17:15:57.975911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import hashlib\nimport pandas as pd\ndata_dir = '/kaggle/input/vesuvius-challenge-ink-detection/test'\n\n\nskip_fake_test = True\n\n## code to skip test if commit run:\na_file = f'{data_dir}/a/mask.png'\nwith open(a_file,'rb') as f:\n    hash_md5 = hashlib.md5(f.read()).hexdigest()\nis_skip_test = hash_md5 == '0b0fffdc0e88be226673846a143bb3e0'\nprint('is_skip_test:',is_skip_test) ## bu demek ki, test ici bos, daha dogrusu dummy test var.\n\nif skip_fake_test and is_skip_test:\n    skipp = True\n    print('skipping test')\n    submit_df = pd.DataFrame({'Id': ['a', 'b'], 'Predicted':['1 2', '1 2']})\n    submit_df.to_csv('submission.csv', index=False)\nelse: skipp = False\n    \nprint(skipp)","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.977533Z","iopub.status.idle":"2023-06-14T17:15:57.978311Z","shell.execute_reply.started":"2023-06-14T17:15:57.978046Z","shell.execute_reply":"2023-06-14T17:15:57.978068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.979904Z","iopub.status.idle":"2023-06-14T17:15:57.980672Z","shell.execute_reply.started":"2023-06-14T17:15:57.980428Z","shell.execute_reply":"2023-06-14T17:15:57.980451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import segmentation_models_pytorch as smp","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.982053Z","iopub.status.idle":"2023-06-14T17:15:57.982812Z","shell.execute_reply.started":"2023-06-14T17:15:57.982576Z","shell.execute_reply":"2023-06-14T17:15:57.982599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport math\nimport PIL.Image as Image\nimport torch\nfrom torch.utils.data import DataLoader\nimport torch.nn as nn\nfrom torch.optim import AdamW\nfrom torch.utils.data import DataLoader, Dataset, random_split\nimport torchvision.transforms as T\nimport cv2\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport torch.optim as optim\nfrom torch.optim import lr_scheduler\nimport torch.backends.cudnn as cudnn\nimport torchvision\nimport time\nimport copy\nimport gc\nfrom tqdm.notebook import tqdm\nfrom tqdm import tqdm, trange\nfrom tqdm.auto import tqdm\n\ncudnn.benchmark = True\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nprint(torch.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.984228Z","iopub.status.idle":"2023-06-14T17:15:57.984980Z","shell.execute_reply.started":"2023-06-14T17:15:57.984741Z","shell.execute_reply":"2023-06-14T17:15:57.984764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#config\nSUBMIT = True #if set to False, evaluate labels accuracy\nTILE_SIZE = 224\nSIZE = TILE_SIZE\nSTRIDE = TILE_SIZE// 2\n\n#specify scan images to load start / end (algorithm will find best positions automatically)\nSTART_SCAN = 28\nEND_SCAN = 36\nINPUT_CHANNELS = END_SCAN - START_SCAN + 1\n\nLR = 0.0001\nLOSS_FUNCTION = torch.nn.BCEWithLogitsLoss()\nTHRESHOLD = 0.45","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.986377Z","iopub.status.idle":"2023-06-14T17:15:57.987137Z","shell.execute_reply.started":"2023-06-14T17:15:57.986880Z","shell.execute_reply":"2023-06-14T17:15:57.986903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_image(path):\n    images = []\n\n    slideids = range(START_SCAN, END_SCAN + 1)\n    \n    for i in slideids:\n        #read images specified by start_scan and end_scan numbers\n        image = cv2.imread(path + f\"/surface_volume/{i:02}.tif\", 0)\n        \n        pad0 = (TILE_SIZE- image.shape[0] % TILE_SIZE)\n        pad1 = (TILE_SIZE- image.shape[1] % TILE_SIZE)\n        \n        #increase borders to match tile size\n        image = np.pad(image, [(0, pad0), (0, pad1)], constant_values=0)\n        \n        images.append(image)\n        del image\n        gc.collect()\n    #stack images (2D arrays) on axis 2\n    images = np.stack(images, axis=2)\n    \n    return images\n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.988521Z","iopub.status.idle":"2023-06-14T17:15:57.989273Z","shell.execute_reply.started":"2023-06-14T17:15:57.989031Z","shell.execute_reply":"2023-06-14T17:15:57.989053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomDataset(Dataset):\n    def __init__(self, images, transform = None):\n        self.images = images\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.images)\n\n    def __getitem__(self, idx):\n        image = self.images[idx]\n        if self.transform:\n            data = self.transform(image = image)\n            image = data['image']\n\n        return image","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.990661Z","iopub.status.idle":"2023-06-14T17:15:57.991450Z","shell.execute_reply.started":"2023-06-14T17:15:57.991190Z","shell.execute_reply":"2023-06-14T17:15:57.991213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_dataset(paths):\n    \n    images_list = []\n    xyxys = []\n    \n    for path in paths:\n        imageset = read_image(path)\n        print(\"Input images - amount:\" + str(imageset.shape[2]) + \" height: \" + str(imageset.shape[0]) + \" width: \" + str(imageset.shape[1]))\n\n        #split images into multiple smaller images (TILE_SIZE)\n        x1_list = list(range(0, imageset.shape[1] - TILE_SIZE+ 1, TILE_SIZE))\n        y1_list = list(range(0, imageset.shape[0] - TILE_SIZE+ 1, TILE_SIZE))\n\n        for y1 in tqdm(y1_list,desc=\"Building dataset \"):\n            for x1 in x1_list:\n                y2 = y1 + TILE_SIZE\n                x2 = x1 + TILE_SIZE\n                images_list.append(imageset[y1:y2, x1:x2,:])\n                xyxys.append((x1, y1, x2, y2))                 \n                \n    xyxys = np.stack(xyxys)\n    image_dataset = CustomDataset(images = images_list,transform=getTransforms())\n    print(\"Dataset includes \" + str(len(images_list)) + \" - \" + str(TILE_SIZE) + \"x\"  + str(TILE_SIZE) + \" images\")\n\n    return image_dataset, xyxys","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.992830Z","iopub.status.idle":"2023-06-14T17:15:57.993598Z","shell.execute_reply.started":"2023-06-14T17:15:57.993355Z","shell.execute_reply":"2023-06-14T17:15:57.993377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#trained different pretrained models and selected different architectures to build an average prediction\n#this function is used to create specified models and load the stored weights\ndef createModel(modelnr):\n    if modelnr == 1: #Unet++ with better augmentation\n        model = smp.UnetPlusPlus(\n            encoder_name='resnext50_32x4d', \n            encoder_weights=None, \n            classes=1, \n            in_channels=INPUT_CHANNELS,\n            activation=None)\n        optimizer = optim.Adam(model.parameters(), lr=LR) \n        model = model.to(device)\n        optimizer_ft = optim.SGD(model.parameters(), lr=0.001, momentum=0.9)\n        # Decay LR by a factor of 0.1 every 7 EPOCHS\n        exp_lr_scheduler = lr_scheduler.StepLR(optimizer_ft, step_size=7, gamma=0.1)\n        model.load_state_dict(torch.load(\"/kaggle/input/trained-models/best-unetpp-resnext50_32x4d.pth\",map_location=torch.device(device)))\n        \n    elif modelnr == 2: #stride 3 Unet\n        model = smp.Unet(\n            encoder_name='resnext50_32x4d', \n            encoder_weights=None, \n            classes=1, \n            in_channels=INPUT_CHANNELS,\n            activation=None)\n        optimizer = optim.Adam(model.parameters(), lr=LR) \n        model = model.to(device)\n        optimizer_ft = optim.SGD(model.parameters(), lr=0.001, momentum=0.9)\n        # Decay LR by a factor of 0.1 every 7 EPOCHS\n        exp_lr_scheduler = lr_scheduler.StepLR(optimizer_ft, step_size=7, gamma=0.1)\n        model.load_state_dict(torch.load(\"/kaggle/input/trained-models/best-unet-resnext50_32x4d.pth\",map_location=torch.device(device)))\n        \n    elif modelnr == 3: #stride 3 Linknet\n        model = smp.Linknet(\n            encoder_name='resnet34', \n            encoder_weights=None, \n            classes=1, \n            in_channels=INPUT_CHANNELS,\n            activation=None)\n        optimizer = optim.Adam(model.parameters(), lr=LR) \n        model = model.to(device)\n        optimizer_ft = optim.SGD(model.parameters(), lr=0.001, momentum=0.9)\n        # Decay LR by a factor of 0.1 every 7 EPOCHS\n        exp_lr_scheduler = lr_scheduler.StepLR(optimizer_ft, step_size=7, gamma=0.1)\n        model.load_state_dict(torch.load(\"/kaggle/input/trained-models/best-linknet-resnet34.pth\",map_location=torch.device(device)))\n        \n    elif modelnr == 4: #stride 3 FPN\n        model = smp.FPN(\n            encoder_name='resnet34', \n            encoder_weights=None, \n            classes=1, \n            in_channels=INPUT_CHANNELS,\n            activation=None)\n        optimizer = optim.Adam(model.parameters(), lr=LR) \n        model = model.to(device)\n        optimizer_ft = optim.SGD(model.parameters(), lr=0.001, momentum=0.9)\n        # Decay LR by a factor of 0.1 every 7 EPOCHS\n        exp_lr_scheduler = lr_scheduler.StepLR(optimizer_ft, step_size=7, gamma=0.1)\n        model.load_state_dict(torch.load(\"/kaggle/input/trained-models/best-fpn-resnet34.pth\",map_location=torch.device(device)))\n    \n    return model ,optimizer","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.995108Z","iopub.status.idle":"2023-06-14T17:15:57.995875Z","shell.execute_reply.started":"2023-06-14T17:15:57.995632Z","shell.execute_reply":"2023-06-14T17:15:57.995655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def getTransforms():\n    #to apply normalization\n    augList = [A.Normalize(mean= [0] * INPUT_CHANNELS,std= [1] * INPUT_CHANNELS),ToTensorV2(transpose_mask=True),]\n    aug = A.Compose(augList)\n    return aug","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.997271Z","iopub.status.idle":"2023-06-14T17:15:57.998039Z","shell.execute_reply.started":"2023-06-14T17:15:57.997786Z","shell.execute_reply":"2023-06-14T17:15:57.997809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predictTest(model,x):\n    yMask = []\n    #set model to evaluation mode\n    model.eval()\n\n    #turn off gradient tracking\n    with torch.no_grad():\n        for i in tqdm(range(len(x)), desc = \"Predict \" + str(model.name)):\n\n            xPart = x[i]\n            xPart = xPart.unsqueeze(0).to('cuda')\n            yPred = model(xPart)\n            \n            predMask = yPred.squeeze()\n            predMask = torch.sigmoid(predMask)            \n\n            #append predictions to list\n            yMask.append(predMask.cpu().numpy())\n            \n    #convert list to array\n    yMask = np.array(yMask)\n   \n    return yMask","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:57.999419Z","iopub.status.idle":"2023-06-14T17:15:58.000180Z","shell.execute_reply.started":"2023-06-14T17:15:57.999931Z","shell.execute_reply":"2023-06-14T17:15:57.999953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adapted from https://www.kaggle.com/code/stainsby/fast-tested-rle/notebook\n# and https://www.kaggle.com/code/kotaiizuka/faster-rle/notebook\ndef rle(output):\n    #pixels = np.where(output.flatten().cpu() > THRESHOLD, 1, 0).astype(np.uint8)\n    pixels = np.where(output.flatten() > THRESHOLD, 1, 0).astype(np.uint8)\n    pixels[0] = 0\n    pixels[-1] = 0\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 2\n    runs[1::2] = runs[1::2] - runs[:-1:2]\n    return ' '.join(str(x) for x in runs)","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.001616Z","iopub.status.idle":"2023-06-14T17:15:58.002410Z","shell.execute_reply.started":"2023-06-14T17:15:58.002148Z","shell.execute_reply":"2023-06-14T17:15:58.002171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#best accuracy models 6,7,8,9 with different architectures \n\nmodel1, _ = createModel(1) #unet++\nmodel2, _ = createModel(2) #unet\nmodel3, _ = createModel(3) #Linknet\nmodel4, _ = createModel(4) #FPN","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.003796Z","iopub.status.idle":"2023-06-14T17:15:58.004571Z","shell.execute_reply.started":"2023-06-14T17:15:58.004334Z","shell.execute_reply":"2023-06-14T17:15:58.004356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preparePlot(predMaskAvg, predMask1, predMask2, predMask3, predMask4, labelMask = None, threshold = 0.45):\n\n    #initialize our figure\n    cols = 6\n    if labelMask is None:\n        cols -= 1\n    figure, ax = plt.subplots(nrows=1, ncols=cols, figsize=(10, 10))\n    \n    \n    #plot \n    ax[0].imshow((predMaskAvg > threshold) * 255)\n    ax[1].imshow((predMask1 > threshold) * 255)\n    ax[2].imshow((predMask2 > threshold) * 255)\n    ax[3].imshow((predMask3 > threshold) * 255)\n    ax[4].imshow((predMask4 > threshold) * 255)\n    if labelMask is not None:\n        ax[5].imshow((labelMask * 255).astype('int8'))\n        \n    #set the titles of the subplots\n    ax[0].set_title(\"Pred Avg \" + str(threshold))\n    ax[1].set_title(\"Unet++ \"  + str(threshold))\n    ax[2].set_title(\"Unet \"  + str(threshold))\n    ax[3].set_title(\"Linknet \"  + str(threshold))\n    ax[4].set_title(\"FPN \"  + str(threshold))\n    \n    if labelMask is not None:\n        ax[5].set_title(\"Label\")\n\n    figure.tight_layout()\n \n    figure.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.005938Z","iopub.status.idle":"2023-06-14T17:15:58.006728Z","shell.execute_reply.started":"2023-06-14T17:15:58.006482Z","shell.execute_reply":"2023-06-14T17:15:58.006505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def unpackImage(imageList,pad_h,pad_w):\n    #assemble predicted image parts to one big image\n    image = np.zeros(shape=(pad_h, pad_w))\n    \n    len_h = pad_h / TILE_SIZE\n    len_w = pad_w / TILE_SIZE\n    \n    h = 0\n    w = 0\n\n    #fill predicted image with partial predictions\n    for i in range(len(imageList)):\n        image[h:h + TILE_SIZE,w:w + TILE_SIZE] = imageList[i,:,:]\n        w += TILE_SIZE\n        if w >= pad_w:\n            h += TILE_SIZE\n            w = 0\n    \n    return image","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.008144Z","iopub.status.idle":"2023-06-14T17:15:58.008906Z","shell.execute_reply.started":"2023-06-14T17:15:58.008667Z","shell.execute_reply":"2023-06-14T17:15:58.008690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predictFolder(path):\n    startTime = time.time()\n    #predict for submission\n    TILE_SIZE= 224\n    SIZE= TILE_SIZE\n    \n    #record accuracies\n    acTreshold = []\n    acAvg = []\n    acUnetpp = []\n    acUnet = []\n    acLink = []\n    acFdn = []\n    \n    #first testset\n    paths = []\n    paths.append(path)\n    \n    if os.path.isfile(path + \"/inklabels.png\"):\n        labelExist = True\n        labelPath = path + \"/inklabels.png\"\n        label_mask = cv2.imread(labelPath, 0)\n        labelImage = label_mask\n        label_mask = (label_mask / 255).astype('float32')\n    else:\n        label_mask = None\n        labelImage = label_mask\n        labelExist = False\n\n    test_set, _ = create_dataset(paths)\n\n    yMask1 = predictTest(model1,test_set) #unet++\n    yMask2 = predictTest(model2,test_set) #unet\n    yMask3 = predictTest(model3,test_set) #linknet\n    yMask4 = predictTest(model4,test_set) #fdn\n    \n    #build average prediction  \n    yMask = (yMask1 + yMask2 + yMask3 + yMask4) /4\n\n    print(\"Predicted mask shape :\" + str(yMask.shape) )\n\n    binary_mask = cv2.imread(path + \"/mask.png\", 0)\n    binary_mask = (binary_mask / 255).astype('float32')\n\n    #original size\n    ori_h = binary_mask.shape[0]\n    ori_w = binary_mask.shape[1]\n\n    #padding again to fit with our tile size\n    pad0 = (TILE_SIZE - binary_mask.shape[0] % TILE_SIZE)\n    pad1 = (TILE_SIZE - binary_mask.shape[1] % TILE_SIZE)\n\n    binary_mask_padded = np.pad(binary_mask, [(0, pad0), (0, pad1)], constant_values=0)\n\n    pad_h = binary_mask_padded.shape[0]\n    pad_w = binary_mask_padded.shape[1]\n    \n    #assemble predicted image parts to big image\n    yPred = unpackImage(yMask,pad_h,pad_w)\n    yPred1 = unpackImage(yMask1,pad_h,pad_w)\n    yPred2 = unpackImage(yMask2,pad_h,pad_w)\n    yPred3 = unpackImage(yMask3,pad_h,pad_w)\n    yPred4 = unpackImage(yMask4,pad_h,pad_w)\n    \n    del yMask1\n    del yMask2\n    del yMask3\n    del yMask4\n\n    gc.collect()\n\n    #back to original size\n    yPred = yPred[:ori_h, :ori_w] \n    yPred1 = yPred1[:ori_h, :ori_w] \n    yPred2 = yPred2[:ori_h, :ori_w]\n    yPred3 = yPred3[:ori_h, :ori_w] \n    yPred4 = yPred4[:ori_h, :ori_w] \n\n    #multiply with mask to set predictions to zero if corresponding mask is zero\n    yPred = np.multiply(binary_mask, yPred)\n    yPred1 = np.multiply(binary_mask, yPred1)\n    yPred2 = np.multiply(binary_mask, yPred2)\n    yPred3 = np.multiply(binary_mask, yPred3)\n    yPred4 = np.multiply(binary_mask, yPred4)\n\n    #show predictions \n    preparePlot(yPred, yPred1, yPred2, yPred3, yPred4, labelImage, 0.5)\n\n\n    \n    #if labels exist, display accuracy for multiple thresholds\n    if labelExist == True:\n        for t in range(20,70,5):\n            tact = t/100\n            \n            accAvg = ((((yPred > tact) * 1 == label_mask).astype('float32').sum())/label_mask.size) * 100\n            acc1 = ((((yPred1 > tact) * 1 == label_mask).astype('float32').sum())/label_mask.size) * 100\n            acc2 = ((((yPred2 > tact) * 1 == label_mask).astype('float32').sum())/label_mask.size) * 100\n            acc3 = ((((yPred3 > tact) * 1 == label_mask).astype('float32').sum())/label_mask.size) * 100\n            acc4 = ((((yPred4 > tact) * 1 == label_mask).astype('float32').sum())/label_mask.size) * 100\n            \n            acTreshold.append(tact)\n            acAvg.append(accAvg)\n            acUnetpp.append(acc1)\n            acUnet.append(acc2)\n            acLink.append(acc3)\n            acFdn.append(acc4)           \n            \n            #print(\"Average Accuracy: {:.2f} threshold: {:.2f}\".format(accAvg,tact))\n            #print(\"Accuracy Unet: {:.2f} threshold: {:.2f}\".format(acc2,tact))\n            #print(\"Accuracy Linknet: {:.2f} threshold: {:.2f}\".format(acc3,tact))\n            #print(\"Accuracy FPN: {:.2f} threshold: {:.2f}\".format(acc4,tact))\n            #print(\"*************************************************************\")\n            \n        #plot accuracy \n        plt.figure(2,(15,15))\n        plt.subplot(111)\n        plt.plot(acTreshold, acAvg, label=\"Average\")\n        plt.plot(acTreshold, acUnetpp, label=\"Unet++\")\n        plt.plot(acTreshold, acUnet, label=\"Unet\")\n        plt.plot(acTreshold, acLink, label=\"LinkNet\")\n        plt.plot(acTreshold, acFdn, label=\"FDN\")\n\n        plt.xlabel(\"Activation threshold\")\n        plt.ylabel(\"Accuracy\")\n\n        plt.title(\"Accuracy/Activation\")\n\n        plt.legend()\n\n        plt.show()\n            \n    #multiply with 1 to receive 0,1 values instead of True, False values\n    yPred = (yPred > THRESHOLD) * 1\n    yPred1 = (yPred1 > THRESHOLD) * 1\n    yPred2 = (yPred2 > THRESHOLD) * 1\n    yPred3 = (yPred3 > THRESHOLD) * 1\n    yPred4 = (yPred4 > THRESHOLD) * 1\n\n    #some statistics to compare data structure\n    print(\"yPrediction:\" + \" min: \" + str(yPred.min()) + \" max: \" + str(yPred.max()) + \" shape: \" + str(yPred.shape)  + \" size: \" + str(yPred.size))\n    print(\"Binary Mask:\" + \" min: \" + str(binary_mask.min()) + \" max: \" + str(binary_mask.max()) + \" shape: \" + str(binary_mask.shape) + \" size: \" + str(binary_mask.size))\n    if labelExist == True:\n        print(\"Label Mask:\" + \" min: \" + str(label_mask.min()) + \" max: \" + str(label_mask.max()) + \" shape: \" + str(label_mask.shape) + \" size: \" + str(label_mask.size))\n    print(\"**************************************************************************************************\")\n    #use the average predictions for submission\n    rleOutput = rle(yPred)\n\n    endTime = time.time()\n    totalTime = endTime - startTime\n    print(\"Total time: {:.2f} seconds\".format(totalTime))\n    \n\n    \n    return rleOutput","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.010393Z","iopub.status.idle":"2023-06-14T17:15:58.011142Z","shell.execute_reply.started":"2023-06-14T17:15:58.010886Z","shell.execute_reply":"2023-06-14T17:15:58.010908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if SUBMIT == False:\n    _ = predictFolder(\"/kaggle/input/vesuvius-challenge-ink-detection/train/1\")","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.012518Z","iopub.status.idle":"2023-06-14T17:15:58.013277Z","shell.execute_reply.started":"2023-06-14T17:15:58.013039Z","shell.execute_reply":"2023-06-14T17:15:58.013061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if SUBMIT == False:\n    _ = predictFolder(\"/kaggle/input/vesuvius-challenge-ink-detection/train/2\")","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.014646Z","iopub.status.idle":"2023-06-14T17:15:58.015419Z","shell.execute_reply.started":"2023-06-14T17:15:58.015153Z","shell.execute_reply":"2023-06-14T17:15:58.015176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if SUBMIT == False:\n    _ = predictFolder(\"/kaggle/input/vesuvius-challenge-ink-detection/train/3\")","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.016786Z","iopub.status.idle":"2023-06-14T17:15:58.017564Z","shell.execute_reply.started":"2023-06-14T17:15:58.017324Z","shell.execute_reply":"2023-06-14T17:15:58.017347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not skipp:\n    rleOutput1 = predictFolder(\"/kaggle/input/vesuvius-challenge-ink-detection/test/a\")","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.018912Z","iopub.status.idle":"2023-06-14T17:15:58.019681Z","shell.execute_reply.started":"2023-06-14T17:15:58.019438Z","shell.execute_reply":"2023-06-14T17:15:58.019460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not skipp:\n    rleOutput2 = predictFolder(\"/kaggle/input/vesuvius-challenge-ink-detection/test/b\")","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.021061Z","iopub.status.idle":"2023-06-14T17:15:58.021827Z","shell.execute_reply.started":"2023-06-14T17:15:58.021586Z","shell.execute_reply":"2023-06-14T17:15:58.021608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not skipp:\n    results = []\n    results.append(('a',rleOutput1))\n    results.append(('b',rleOutput2))\n    submission = pd.DataFrame(results, columns=['Id', 'Predicted'])\n    submission.to_csv(\"submission.csv\", index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.023218Z","iopub.status.idle":"2023-06-14T17:15:58.023973Z","shell.execute_reply.started":"2023-06-14T17:15:58.023732Z","shell.execute_reply":"2023-06-14T17:15:58.023755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp submission.csv submission_ensemble/submission_Inherit.csv","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.025392Z","iopub.status.idle":"2023-06-14T17:15:58.026151Z","shell.execute_reply.started":"2023-06-14T17:15:58.025898Z","shell.execute_reply":"2023-06-14T17:15:58.025920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nfor f in glob.glob('*'):\n    if not f.startswith('submission_ensemble'):\n        !rm -rf {f}","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.027554Z","iopub.status.idle":"2023-06-14T17:15:58.028321Z","shell.execute_reply.started":"2023-06-14T17:15:58.028065Z","shell.execute_reply":"2023-06-14T17:15:58.028090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## finalizing ","metadata":{}},{"cell_type":"code","source":"%reset -f","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.029654Z","iopub.status.idle":"2023-06-14T17:15:58.030428Z","shell.execute_reply.started":"2023-06-14T17:15:58.030168Z","shell.execute_reply":"2023-06-14T17:15:58.030190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## 4 submission oku ve decode","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.031783Z","iopub.status.idle":"2023-06-14T17:15:58.032557Z","shell.execute_reply.started":"2023-06-14T17:15:58.032313Z","shell.execute_reply":"2023-06-14T17:15:58.032337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## and this is for... the rle\nimport numba\n\n@numba.njit()\ndef rle_numba(pixels):\n    size = len(pixels)\n    points = []\n    if pixels[0] == 1:\n        flag = False\n        points.append(1)\n    else:\n        flag = True\n    for i in range(1, size):\n        if pixels[i] != pixels[i-1]:\n            if flag:\n                points.append(i+1)\n                flag = False\n            else:\n                points.append(i+1 - points[-1])\n                flag = True\n    if pixels[-1] == 1: points.append(size-points[-1]+1)    \n    return points\n\ndef rle_numba_encode(image):\n#     pixels = image.flatten(order = 'F')\n    pixels = image.flatten(order = 'C') ## bizden istedigi default olan C galiba .. bu comp icin\n\n    points = rle_numba(pixels)\n    return ' '.join(str(x) for x in points)\n\ndef rle_decode(mask_rle, shape=(384, 384)):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img.reshape(shape, order='C')","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.033928Z","iopub.status.idle":"2023-06-14T17:15:58.034697Z","shell.execute_reply.started":"2023-06-14T17:15:58.034455Z","shell.execute_reply":"2023-06-14T17:15:58.034477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas\nfrom PIL import Image\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.036104Z","iopub.status.idle":"2023-06-14T17:15:58.036866Z","shell.execute_reply.started":"2023-06-14T17:15:58.036624Z","shell.execute_reply":"2023-06-14T17:15:58.036648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## a ve b nin shapine gore is yapacaksin..\n## direk masklari alip bakabilirsin\ndata_dir = '/kaggle/input/vesuvius-challenge-ink-detection/test'\nrle = {}\n# weights = [1,1,1,1,1]\nweights = [1,1,1]\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.038241Z","iopub.status.idle":"2023-06-14T17:15:58.039022Z","shell.execute_reply.started":"2023-06-14T17:15:58.038762Z","shell.execute_reply":"2023-06-14T17:15:58.038784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask_a = np.array(Image.open(os.path.join(f'{data_dir}/a', 'mask.png')))\nmask_b = np.array(Image.open(os.path.join(f'{data_dir}/b', 'mask.png')))\n\nheight_a,width_a = mask_a.shape\nheight_b,width_b = mask_b.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.040414Z","iopub.status.idle":"2023-06-14T17:15:58.041175Z","shell.execute_reply.started":"2023-06-14T17:15:58.040917Z","shell.execute_reply":"2023-06-14T17:15:58.040941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"height_a,width_a,height_b,width_b","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.042724Z","iopub.status.idle":"2023-06-14T17:15:58.043512Z","shell.execute_reply.started":"2023-06-14T17:15:58.043255Z","shell.execute_reply":"2023-06-14T17:15:58.043277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# simdi submisionlari teker teker okuyaoim..","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.044875Z","iopub.status.idle":"2023-06-14T17:15:58.045652Z","shell.execute_reply.started":"2023-06-14T17:15:58.045410Z","shell.execute_reply":"2023-06-14T17:15:58.045433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rle_25 = pd.read_csv('/kaggle/working/submission_ensemble/submission25.csv')\nrle_30 = pd.read_csv('/kaggle/working/submission_ensemble/submission30.csv')\nrle_ii = pd.read_csv('/kaggle/working/submission_ensemble/submission_Inherit.csv')\n# rle_13 = pd.read_csv('/kaggle/working/submission_ensemble/submission_13Haz.csv')\n# rle_14 = pd.read_csv('/kaggle/working/submission_ensemble/submission_14Haz.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.047025Z","iopub.status.idle":"2023-06-14T17:15:58.047789Z","shell.execute_reply.started":"2023-06-14T17:15:58.047544Z","shell.execute_reply":"2023-06-14T17:15:58.047568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rle_25.Predicted[0]","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.049158Z","iopub.status.idle":"2023-06-14T17:15:58.049911Z","shell.execute_reply.started":"2023-06-14T17:15:58.049670Z","shell.execute_reply":"2023-06-14T17:15:58.049693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## now decode them.... adan baslayalim....\npreds_25 = rle_decode(rle_25.Predicted[0],(height_a,width_a))\npreds_30 = rle_decode(rle_30.Predicted[0],(height_a,width_a))\npreds_ii = rle_decode(rle_ii.Predicted[0],(height_a,width_a))\n# preds_13 = rle_decode(rle_13.Predicted[0],(height_a,width_a))\n# preds_14 = rle_decode(rle_14.Predicted[0],(height_a,width_a))\n\n## bunlari float yapip np.average with weights yapip sonra tekrar >.5 ile sonra orda rle ve\n## bu rle a olacak","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.051297Z","iopub.status.idle":"2023-06-14T17:15:58.052048Z","shell.execute_reply.started":"2023-06-14T17:15:58.051796Z","shell.execute_reply":"2023-06-14T17:15:58.051819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preds = np.stack([preds_25,preds_30,preds_ii,preds_13,preds_14]).astype(np.float32)\npreds = np.stack([preds_25,preds_30,preds_ii]).astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.053423Z","iopub.status.idle":"2023-06-14T17:15:58.054182Z","shell.execute_reply.started":"2023-06-14T17:15:58.053931Z","shell.execute_reply":"2023-06-14T17:15:58.053954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# np.sum(preds)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.055616Z","iopub.status.idle":"2023-06-14T17:15:58.056637Z","shell.execute_reply.started":"2023-06-14T17:15:58.056395Z","shell.execute_reply":"2023-06-14T17:15:58.056418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## 25 30 inherit 13 ve 14 haziran modelleri sirasi ile weighted..\npreds = np.average(preds,axis=0,weights=weights)\npreds.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.058022Z","iopub.status.idle":"2023-06-14T17:15:58.058783Z","shell.execute_reply.started":"2023-06-14T17:15:58.058541Z","shell.execute_reply":"2023-06-14T17:15:58.058563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = (preds>.5).astype(np.uint8)\nnp.sum(preds)","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.060150Z","iopub.status.idle":"2023-06-14T17:15:58.060924Z","shell.execute_reply.started":"2023-06-14T17:15:58.060682Z","shell.execute_reply":"2023-06-14T17:15:58.060705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rle['a'] = rle_numba_encode(preds)","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.062302Z","iopub.status.idle":"2023-06-14T17:15:58.063075Z","shell.execute_reply.started":"2023-06-14T17:15:58.062800Z","shell.execute_reply":"2023-06-14T17:15:58.062822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## simdi b icin:\n## now decode them.... adan baslayalim....\npreds_25 = rle_decode(rle_25.Predicted[1],(height_b,width_b))\npreds_30 = rle_decode(rle_30.Predicted[1],(height_b,width_b))\npreds_ii = rle_decode(rle_ii.Predicted[1],(height_b,width_b))\n# preds_13 = rle_decode(rle_13.Predicted[1],(height_b,width_b))\n# preds_14 = rle_decode(rle_14.Predicted[1],(height_b,width_b))\n\n# preds = np.stack([preds_25,preds_30,preds_ii,preds_13,preds_14]).astype(np.float32)\npreds = np.stack([preds_25,preds_30,preds_ii]).astype(np.float32)\n\n# mreds = preds.copy() ## burada anlirsan... averagelari da denersin ~~~~\npreds = np.average(preds,axis=0,weights=weights)\npreds = (preds>.5).astype(np.uint8)\nrle['b'] = rle_numba_encode(preds)\n\n## bunlari float yapip np.average with weights yapip sonra tekrar >.5 ile sonra orda rle ve\n## bu rle a olacak","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.064474Z","iopub.status.idle":"2023-06-14T17:15:58.065229Z","shell.execute_reply.started":"2023-06-14T17:15:58.064979Z","shell.execute_reply":"2023-06-14T17:15:58.065012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nfor f in glob.glob('*'):\n    !rm -rf {f}","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.066620Z","iopub.status.idle":"2023-06-14T17:15:58.067383Z","shell.execute_reply.started":"2023-06-14T17:15:58.067125Z","shell.execute_reply":"2023-06-14T17:15:58.067147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Id,Predicted\\na,\" + rle['a'] + \"\\nb,\" + rle['b'], file=open('submission.csv', 'w'))    ","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.068723Z","iopub.status.idle":"2023-06-14T17:15:58.069506Z","shell.execute_reply.started":"2023-06-14T17:15:58.069253Z","shell.execute_reply":"2023-06-14T17:15:58.069275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.070869Z","iopub.status.idle":"2023-06-14T17:15:58.071636Z","shell.execute_reply.started":"2023-06-14T17:15:58.071394Z","shell.execute_reply":"2023-06-14T17:15:58.071417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image.fromarray(preds*255)","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.072980Z","iopub.status.idle":"2023-06-14T17:15:58.073760Z","shell.execute_reply.started":"2023-06-14T17:15:58.073520Z","shell.execute_reply":"2023-06-14T17:15:58.073543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# labels = np.array(Image.open('/kaggle/input/vesuvius-challenge-ink-detection/train/1/inklabels.png'))\n# labels = labels[-5454:]\n# labels.shape,np.unique(labels),preds.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.075140Z","iopub.status.idle":"2023-06-14T17:15:58.075901Z","shell.execute_reply.started":"2023-06-14T17:15:58.075661Z","shell.execute_reply":"2023-06-14T17:15:58.075683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # # # NEDEN AYNISI OLMUHYOR ONA BAKMAK LAZIM.....\n# from sklearn.metrics import fbeta_score\n# fbeta_score(labels.flatten(), preds.flatten(), beta=0.5)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.077483Z","iopub.status.idle":"2023-06-14T17:15:58.078367Z","shell.execute_reply.started":"2023-06-14T17:15:58.078096Z","shell.execute_reply":"2023-06-14T17:15:58.078120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tholds = [.05,.1,.15,.2,.25,.30,.35,.40,.45,.50,.55,.60,.65,.70, .75, .80]\n\n# for thi in tholds:\n#     preds = mreds.copy()\n# #     preds = mask*preds ## buna gerek kalmadi cunku mreds'i mask * preds sonrasina aldim...\n#     preds = (preds>thi).astype(np.uint8)\n    \n#     score = fbeta_score(labels.flatten(), preds.flatten(), beta=0.5)\n#     print(f'for thold of {thi},, the fbeta score is:{score}')\n\n\n# # for thold of 0.05,, the fbeta score is:0.4138839027303663\n# # for thold of 0.1,, the fbeta score is:0.46358726046431376\n# # for thold of 0.15,, the fbeta score is:0.501133748144812\n# # for thold of 0.2,, the fbeta score is:0.5320235519067721\n# # for thold of 0.25,, the fbeta score is:0.5604565047334683\n# # for thold of 0.3,, the fbeta score is:0.5691042296051885\n# # for thold of 0.35,, the fbeta score is:0.5719583491913619\n# # for thold of 0.4,, the fbeta score is:0.5692280584683906\n# # for thold of 0.45,, the fbeta score is:0.560449774441391\n# # for thold of 0.5,, the fbeta score is:0.5512158841316568\n# # for thold of 0.55,, the fbeta score is:0.5339695495590538\n# # for thold of 0.6,, the fbeta score is:0.5108621823429387\n# # for thold of 0.65,, the fbeta score is:0.48228067416494846\n# # for thold of 0.7,, the fbeta score is:0.44896378975670603\n# # for thold of 0.75,, the fbeta score is:0.4075783422540912\n# # for thold of 0.8,, the fbeta score is:0.3617596448857491","metadata":{"execution":{"iopub.status.busy":"2023-06-14T17:15:58.079783Z","iopub.status.idle":"2023-06-14T17:15:58.080553Z","shell.execute_reply.started":"2023-06-14T17:15:58.080314Z","shell.execute_reply":"2023-06-14T17:15:58.080339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}