{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":61446,"databundleVersionId":6962461,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd \nimport numpy as np \nimport cv2 \nimport os ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = '/kaggle/input/blood-vessel-segmentation/train'\ncsv_path = '/kaggle/input/blood-vessel-segmentation/train_rles.csv'\ntest_dir = '/kaggle/input/blood-vessel-segmentation/test'\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(csv_path)\ntrain_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[['label_folder', 'filename']] = train_df['id'].str.rsplit('_', expand=True, n = 1)\ntrain_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.to_csv('/kaggle/working/train_rles_split_id.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # decode and encode \n# # ref.: https://www.kaggle.com/stainsby/fast-tested-rle\n# def mask2rle(img):\n#     \"\"\"\n#     img: numpy array, 1 - mask, 0 - background\n#     Returns run length as string formatted\n#     \"\"\"\n#     pixels = img.flatten()\n#     pixels = np.concatenate([[0], pixels, [0]])\n#     runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n#     runs[1::2] -= runs[::2]\n#     return ' '.join(str(x) for x in runs)\n\n\n# def rle2mask(mask_rle, shape, label=1):\n#     \"\"\"\n#     mask_rle: run-length as string formatted (start length)\n#     shape: (height,width) of array to return\n#     Returns numpy array, 1 - mask, 0 - background\n\n#     \"\"\"\n#     s = mask_rle.split()\n#     starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n#     starts -= 1\n#     ends = starts + lengths\n#     img = np.zeros(shape[0] * shape[1], dtype=np.uint8)\n#     for lo, hi in zip(starts, ends):\n#         img[lo:hi] = label\n#     return img.reshape(shape)  # Needed to align to RLE direction\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# encode: correct! rle == rle from label\ndef label2rle(img):\n    '''\n    img: numpy array, 1 - mask, 0 - background\n    Returns run length as string formated\n    '''\n    pixels = img.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    rle = ' '.join(str(x) for x in runs)\n    if rle=='':\n        rle = '1 0'\n    return rle\n\n \n\ndef rle2label(mask_rle, shape):\n    '''\n    mask_rle: run-length as string formated (start length)\n    shape: (height,width) of array to return \n    Returns numpy array, 1 - mask, 0 - background\n\n    '''\n    s = mask_rle.split()\n    starts, lengths = [np.asarray(x, dtype=int) for x in (s[0:][::2], s[1:][::2])]\n    starts -= 1\n    ends = starts + lengths\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    img = img.astype('float32')\n    return img.reshape(shape)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load image \ndef load_img(path, scale = True):\n    img = cv2.imread(path, cv2.IMREAD_UNCHANGED)\n    img = np.tile(img[...,None], [1, 1, 3]) # gray to rgb\n    img = img.astype('float32') # original is uint16\n#     print(max(img.reshape(-1)))\n\n    if scale:\n        mx = np.max(img)\n        img/=mx # scale image to [0, 1]\n    return img\n\ndef load_mask(path):\n    msk = cv2.imread(path, cv2.IMREAD_UNCHANGED)\n    \n    msk = msk.astype('float32')\n    msk/=255.0\n    return msk","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef compare(img_path, test_imgs_path):\n    \n    for test_path in test_imgs_path:\n        a_img = load_img(img_path)\n        b_img = load_img(test_path)\n#         print(a_img.shape, b_img.shape)\n        is_similar(a_img, b_img)\n#             print('a_img', img_path)\n#             print('b_img', test_path)\n        \n        \n    \n\ndef is_similar(image1, image2):\n    assert (max(image1.reshape(-1))<= 1 and max(image2.reshape(-1))) or (max(image1.reshape(-1))> 1 and max(image2.reshape(-1)) > 1)\n    difference = cv2.subtract(image1, image2)\n    result = not np.any(difference)\n\n    return result\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\nimport matplotlib.pyplot as plt \nimport os\n\ncategories = np.unique(train_df['label_folder'].values)\ncho = 3 \nfor cate in categories:\n    print(f'----------{cate}--------')\n    indexes = np.where(train_df['label_folder'] == cate)[0]\n    print(f'number of files: {len(indexes)}')\n    \n    rand_indexes = np.random.choice(indexes, cho)\n    \n    \n\n    plt.axis('off')\n    \n    for i, ind in enumerate(rand_indexes):\n        _, rle, label_folder, filename = train_df.iloc[ind, :]\n        assert label_folder == cate \n\n        label_path = os.path.join(train_dir, cate, 'labels', filename + '.tif')\n        \n\n        label = load_mask(label_path)\n        \n        \n        if cate == 'kidney_3_dense':\n            img_path =  os.path.join(train_dir, 'kidney_3_sparse', 'images', filename + '.tif')\n        else:\n            img_path = os.path.join(train_dir, cate, 'images', filename + '.tif')\n        img = load_img(img_path)\n        \n        rle_from_label = label2rle(label)\n        assert rle == rle_from_label, f'label_path: {label_path}, img_path: {img_path}'\n        \n        label_from_rle = rle2label(rle, label.shape)\n        \n        if not is_similar(label_from_rle, label):\n            print(f'label_path: {label_path} not similar to labels from rle')\n  \n        \n           \n\n        \n        plt.figure(figsize = (9, 12))\n        plt.suptitle(label_path, y=0.92)\n        plt.subplot(i+1, 3, 1)\n        plt.imshow(img)\n        plt.title(f'image: {img.shape}')\n        \n        plt.subplot(i+1, 3, 2)\n        plt.imshow(label)\n        plt.title(f'label: {label.shape}')\n        \n\n        plt.subplot(i+1, 3, 3)\n        plt.imshow(label_from_rle)\n        plt.title(f'rles: {label_from_rle.shape}')     \n        \n\n    \n    # conclusion: rles --> label vs label from data folder: would be the same \n    # label --> rles vs rles from csv: not the same \n    # => had better use rles as ground truth \n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.where(train_df['rle'] == '1 0')[0]\nrle2label('1 0', shape = (224, 224))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check whether kidney_3_sparse == kidney_3_dense\n\ndense_indexes = np.where(train_df['label_folder'] == 'kidney_3_dense')[0]\n# print(dense_indexes)\nrand_dense_indexes = np.random.choice(dense_indexes, 5)\n\nrle_equal = 0\nlabel_equal = 0\nfor d_ind in dense_indexes: # rand_dense_indexes:\n    _, dense_rle, label_folder, filename = train_df.iloc[d_ind, :]\n    img_path = os.path.join(train_dir, 'kidney_3_sparse', 'images', f'{filename}.tif')\n    dense_label_path = os.path.join(train_dir, label_folder, 'labels', f'{filename}.tif')\n    sparse_label_path = os.path.join(train_dir, 'kidney_3_sparse', 'labels', f'{filename}.tif')\n\n    # compare difference based ong grey image\n    assert os.path.exists(dense_label_path)\n    assert os.path.exists(sparse_label_path)\n#     compare(dense_label_path, [sparse_label_path])\n\n\n    dense_label = load_mask(dense_label_path)\n    sparse_label = load_mask(sparse_label_path)\n    \n    dense_rle_from_l = label2rle(dense_label)\n    sparse_rle_from_l = label2rle(sparse_label)\n    \n\n    if dense_rle_from_l == sparse_rle_from_l and sparse_rle_from_l != '1 0':\n#         print(f'img_path: {img_path}\\nd: {dense_rle_from_l}\\ns:{sparse_rle_from_l}')\n        label_equal += 1 \n        \n    \n    \n    # compare the difference on rles \n    sparse_rle = train_df['rle'].loc[(train_df['label_folder'] == 'kidney_3_sparse') & (train_df['filename'] == filename)].values\n\n    if sparse_rle == dense_rle and dense_rle != '1 0':\n#         print(img_path, 'sparse_rle =dense_rle:')\n        rle_equal += 1 \n    \n    \n#     if d_ind % 5000 == 0:\n\n#         plt.figure(figsize = (9, 12))\n#         plt.suptitle(dense_label_path)\n#         plt.subplot(i+1, 3, 1)\n#         plt.imshow(img)\n#         plt.title(f'image: {img.shape}')\n\n#         plt.subplot(i+1, 3, 2)\n#         plt.imshow(dense_label)\n#         plt.title(f'dense_label: {dense_label.shape}')\n\n#         plt.subplot(i+1, 3, 3)\n#         plt.imshow(sparse_label)\n#         plt.title(f'sparse_label: {sparse_label.shape}')      \n\n\n#         conclusion: except '1 0'. there would be 10 labels from kidney_3_dense = kidney_3_sparse (either from rles from csv or labels from image folder)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_equal, rle_equal","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test images \ncho = 5 \ntest_imgs_path = []\nfor dir, folder, files in os.walk(test_dir):\n    for ff in files:\n        img_path = os.path.join(dir, ff)\n        img = load_img(img_path)\n        print('img.shape', img.shape)\n        print('img path', img_path)\n        plt.figure(figsize = (9, 9))\n\n        plt.imshow(img)\n        plt.title(img_path)\n\n        compare(img_path, test_imgs_path)\n\n        test_imgs_path.append(img_path)\n        \n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}