{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":61446,"databundleVersionId":6962461,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"67b73484-28be-4642-9a7b-5cb29098fb54","_cell_guid":"1de76d26-f652-4416-b6e8-4bbdccfbb780","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tifffile\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check the tiff format \nimg_path = '/kaggle/input/blood-vessel-segmentation/train/kidney_1_dense/images/0529.tif'\nlabel_path = '/kaggle/input/blood-vessel-segmentation/train/kidney_1_dense/labels/0529.tif'\nimport tifffile as tiff\nimport numpy as np\nimport cv2\n\na = tiff.imread(img_path)  # x,y, 16 bits \nprint(a.shape)\na.dtype\n\n# load_img(img_path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# refer: https://blog.csdn.net/a486259/article/details/131603554?utm_medium=distribute.pc_relevant.none-task-blog-2~default~baidujs_baidulandingword~default-0-131603554-blog-123264071.235^v38^pc_relevant_sort_base3&spm=1001.2101.3001.4242.1&utm_relevant_index=3\ndef cut_normalization_img(img):\n    \n\n    #统计每个值出现的频率\n    img=img.astype(np.int32)\n    frequence=np.bincount(img.reshape(-1)) \n    #设置用于计算最大值阈值，最小值阈值的参数\n    all_pixs=frequence.sum()\n    min_rate=0.005 #从左往右累加统计各个值域的频率，频率值大于min_rate时，则为最小值阈值\n    max_rate=0.005#从右往左累加统计各个值域的频率，频率值大于min_rate时，则为最大值阈值\n    now_min_rate=0\n    now_max_rate=0\n\n    min_index=0 #最小值阈值\n    while now_min_rate>=min_rate:\n        now_pixs=frequence[:min_index].sum()\n        now_min_rate=now_pixs/all_pixs\n        min_index+=1\n\n    max_index=-1 #最小值阈值\n    while now_max_rate>=max_rate:\n        now_pixs=frequence[max_index:].sum()\n        now_max_rate=now_pixs/all_pixs\n        max_index-=1\n    max_index=max_index+frequence.shape[0]#修正最小值阈值的表达方式\n\n    #将数据的值域调整为0~(max_index-min_index)\n    img[img>max_index]=max_index\n    img[img<min_index]=min_index\n    img=img-min_index\n\n    #将数据的值域调整为0~255\n    img=img.astype(np.float32)\n    img=img/(max_index-min_index)\n    img=img*255\n    img=img.astype(np.uint8)\n    \n#     print(img.shape)\n    return img\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# load image \ndef load_img(path, reshape = (1024, 1024)):\n    img = cv2.imread(path, cv2.IMREAD_UNCHANGED)\n\n    img = cut_normalization_img(img)\n    img = np.tile(img[...,None], [1, 1, 3]) # gray to rgb\n    if reshape:\n        img = cv2.resize(img, reshape)\n    return img\n\n\norigin_img = cv2.imread(img_path, cv2.IMREAD_UNCHANGED)\nimg = load_img(img_path)\nimport matplotlib.pyplot as plt \n\nplt.figure(figsize = (9, 12))\nplt.subplot(1, 2, 1)\nplt.imshow(origin_img)\n\nplt.subplot(1, 2, 2)\nplt.imshow(img)\n\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_label(path, reshape = (1024, 1024)):\n    msk = cv2.imread(path, cv2.IMREAD_UNCHANGED)\n    if reshape:\n        msk = cv2.resize(msk, reshape)\n    msk = msk.astype('float16')\n    msk/=255.0\n\n    return msk\n\norigin_label = cv2.imread(label_path, cv2.IMREAD_UNCHANGED)\n\nlabel = load_label(label_path)\nplt.subplot(1, 2, 1)\nplt.imshow(origin_label)\n\nplt.subplot(1, 2, 2)\nplt.imshow(label)\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# size distribution \nimport glob \nfrom collections import Counter\nfrom PIL import Image\n\ndef get_size_distribution(directory):\n    \n    # Get a list of all files in the directory\n    files = glob.glob(os.path.join(directory, '*.tif'))\n\n    # Initialize a list to store the sizes\n    sizes = []\n\n    # For each file\n    for file in files:\n        # Construct the full file path\n        filepath = os.path.join(directory, file)\n\n        # Open the image and get its size\n        try:\n            with Image.open(filepath) as img:\n                sizes.append(img.size)\n        except Exception as e:\n            print(f\"Unable to open image {filepath}: {e}\")\n\n    # Convert the list of sizes to a numpy array\n    sizes = np.array(sizes)\n\n    # Use collections.Counter to get the frequency of each unique size\n    size_distribution = Counter(map(str, sizes))\n\n    # Now 'size_distribution' contains the frequency of each unique size\n    \n    for size, count in size_distribution.items():\n        print(f\"Size: {size}, Count: {count}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# resize the whole dataset \ninput_dir = '/kaggle/input/blood-vessel-segmentation/train'\noutput_dir = '/kaggle/working/train'\n\nfor directory in os.listdir(input_dir):\n    image_directory = os.path.join(input_dir, directory, 'images')\n    print(f'**{image_directory}**', end = ' ')\n    get_size_distribution(image_directory)\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !rm -rf /kaggle/working/*\n    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = '/kaggle/input/blood-vessel-segmentation/train'\noutput_dir = '/kaggle/working/dataset/train'\ni = 0 \nimport collections\ncount_dict = collections.defaultdict(int)\n\nfor directory, folder, files in os.walk(train_dir):\n        \n    for ff in files:\n        img_path = os.path.join(directory, ff)\n        count_dict[directory] += 1 \n        assert os.path.exists(img_path), img_path\n        if 'images' in img_path:\n            img = load_img(img_path)\n        else:\n            img = load_label(img_path)\n        \n        saved_path = img_path.replace(train_dir, output_dir)\n        saved_path = saved_path.replace('tif', 'png')\n#         print(saved_path, 'saved_path')\n        \n        os.makedirs(os.path.dirname(saved_path), exist_ok = True)\n        cv2.imwrite(saved_path, img)\n        i += 1 \n        \n#         if i >= 3:\n#             break","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_dict","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !zip -r cleaned_dataset.zip /kaggle/working/dataset\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import FileLink\nFileLink(r'cleaned_dataset.zip')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}