{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n\n# Versions\n* V1-V3 efficientnetb4 + unet\n* v4 efficientnetb0 + linknet + dce_jacobian loss\n* v5 efficientnetb0 + unet + bce_jacobian \n* v6-v8 add guassian data augmentation\n* v14 efficientnetb0 + unet 256 tiles\n* v15 efficientnetb0 + linknet\n* v9 fix shuffbug\n* v10 - v49 \n* try to find best learning rate for batchsize 1024 is about 1e-3 to 5e-4 ( version 43 )\n* best backbone is efficientnet b2 due to image resolution \n* gpu with batchsize 32 run over 9 hours limit in version 48\n* v50 read paper and try more loss function\n* v54 add extennal data https://www.kaggle.com/baesiann/glomeruli-hubmap-external-1024x1024\n* v60 tuning params for optimizer adam look forward\n* v64 ~tpu tensorboard for analysis~ only find tpu profiler for gcp","metadata":{}},{"cell_type":"markdown","source":"# Refferences:\n1. https://www.kaggle.com/wrrosa/hubmap-tf-with-tpu-efficientunet-512x512-tfrecs (how to create training and inference tfrecords)\n2. https://www.kaggle.com/wrrosa/hubmap-tf-with-tpu-efficientunet-512x512-train (training pipeline)\n3. https://www.kaggle.com/wrrosa/hubmap-tf-with-tpu-efficientunet-512x512-subm (inference with submission)\n4. https://www.kaggle.com/vgarshin/kidney-unet-model-keras-inference?scriptVersionId=58536838 and https://github.com/vgarshin/kaggle_kidney/blob/master/kidney_train.ipynb\n","metadata":{}},{"cell_type":"markdown","source":"# tpu\n\n- [Issue]tpu doesn't support numpy_function \n1. issue: https://github.com/tensorflow/tensorflow/issues/38762\n\n- tpu traning guide\n1. https://www.tensorflow.org/guide/data_performance\n2. https://www.tensorflow.org/guide/distributed_training\n\n\n# todo \n1. change tf.image to tf sequence layer https://www.tensorflow.org/tutorials/images/data_augmentation#resizing_and_rescaling\n2. 语言特性, tf 特性 和 np 的差异 \n3. tf 的特性 流的概念 session 的概念 维度对齐\n4. tpu 的特性 tpu的效率\n5. 尝试多种loss https://github.com/JunMa11/SegLoss","metadata":{}},{"cell_type":"markdown","source":"# Init - parameters, packages, gcs_paths, tpu","metadata":{}},{"cell_type":"code","source":"# tpu v3-8 https://www.kaggle.com/docs/tpu#tpu2\nP = {}\nP['EPOCHS'] = 120\n# 不同base 的efficeinet 似乎只有模型规模的指数不同! \n# 不是的！！！！！ 不同 baseline 对应了 不同分辨率 https://keras.io/examples/vision/image_classification_efficientnet_fine_tuning/\n# 暂时换成b0 调试起来更快\nP['BACKBONE'] = 'efficientnetb0' \n#P['BACKBONE'] = 'efficientnetb2' \nP['NFOLDS'] = 5\nP['SEED'] = 7788\nP['VERBOSE'] = 0\nP['DISPLAY_PLOT'] = True \nP['BATCH_COE'] = 8 # BATCH_SIZE = P['BATCH_COE'] * strategy.num_replicas_in_sync\n#P['BATCH_COE'] = 32 # for gpu\n\nP['TILING'] = [1024,256] # 1024,512 1024,256 1024,128 1536,512 768,384\nP['DIM'] = P['TILING'][1] \nP['DIM_FROM'] = P['TILING'][0]\n\n#P['LR'] = 5e-4\nP['LR'] = 5e-4 # for tpu\nP['OVERLAPP'] = True\nP['STEPS_COE'] = 1\n\nP['patience'] = 20\n# 不用外部数据 加速调试\nP['extenal'] = False\n\nimport yaml\nwith open(r'params.yaml', 'w') as file:\n    yaml.dump(P, file)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(P)\n#!ls\n#!cat params.yaml","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! pip install segmentation_models -q\n%matplotlib inline\n\n%load_ext tensorboard\n# Clear any logs from previous runs\n!rm -rf ./logs/\n\nimport os\nos.environ['SM_FRAMEWORK'] = 'tf.keras'\nimport glob\n\nfrom segmentation_models.losses import bce_jaccard_loss\nimport segmentation_models as sm\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom sklearn.model_selection import KFold\n\nimport tensorflow as tf\nfrom tensorflow.keras import backend as K\nfrom tensorflow.keras.utils import get_custom_objects\n\nfrom kaggle_datasets import KaggleDatasets\n\nAUTO = tf.data.experimental.AUTOTUNE\n\nimport tensorflow_addons as tfa\n\n#import albumentations as albu\n#from functools import partial","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try: # detect TPUs\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() # TPU detection\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept ValueError: # no TPU found, detect GPUs\n    #strategy = tf.distribute.MirroredStrategy() # for GPU or multi-GPU machines\n    strategy = tf.distribute.get_strategy() # default strategy that works on CPU and single GPU\n    #strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy() # for clusters of multi-GPU machines\n\nBATCH_SIZE = P['BATCH_COE'] * strategy.num_replicas_in_sync\n\nprint(\"Number of accelerators: \", strategy.num_replicas_in_sync)\nprint(\"BATCH_SIZE: \", str(BATCH_SIZE))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#strategy","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## GCS_PATHS","metadata":{}},{"cell_type":"code","source":"#import tensorflow_datasets as tfds\nGCS_PATH = KaggleDatasets().get_gcs_path('hubmaptfwithtpuefficientunet256tfrecord')\n#GCS_PATH = '../input/hubmap-tf-with-tpu-efficientunet-256-tfrecord'\nALL_TRAINING_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/train/*.tfrec')\n#ALL_TRAINING_FILENAMES","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 使用shift过1024的 train2数据 这里如果overlap为true会在训练阶段把train2加进去一起训练\nif P['OVERLAPP']:\n    OVERLAPP = tf.io.gfile.glob(GCS_PATH + '/train2/*.tfrec')\n    ALL_TRAINING_FILENAMES += OVERLAPP\n    \nif P['extenal']:\n    extenal = tf.io.gfile.glob(GCS_PATH + '/extenal/*.tfrec')\n    ALL_TRAINING_FILENAMES += extenal","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\ndef count_data_items(filenames):\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) for filename in filenames]\n    return np.sum(n)\nprint('NUM_TRAINING_IMAGES:' )\n#if P['OVERLAPP']:\n#    print(count_data_items(ALL_TRAINING_FILENAMES2)+count_data_items(ALL_TRAINING_FILENAMES)+count_data_items(ALL_TRAINING_FILENAMES3))\n#else:\nprint(count_data_items(ALL_TRAINING_FILENAMES))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nfrom datetime import datetime","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(\"if without overlapp\", count_data_items(ALL_TRAINING_FILENAMES))\n#print(\"train2\", count_data_items(ALL_TRAINING_FILENAMES2))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Datasets pipeline","metadata":{}},{"cell_type":"code","source":"#_r = tf.random.stateless_uniform(())\n\"\"\"\n# 测试一下图片自增对多图片\nx = [[[1.0, 2.0, 3.0],\n      [4.0, 5.0, 6.0]],\n     [[7.0, 8.0, 9.0],\n      [10.0, 11.0, 12.0]]]\n_seed = (1, 2)\n_a = tf.image.stateless_random_saturation([x, x], 0.7, 1.3\n                                         #, seed\n                                         )\n\"\"\"\n\n#print(_r)\n#plt.imshow(a)\n#plt.imshow(b)\n\n#g1 = tf.random.Generator.from_seed(1)\n#print(g1.normal(shape=[]))\n#g2 = tf.random.get_global_generator()\n#print(g2.normal(shape=[]))\n\n#seed = g2.make_seeds(2)[0]\n#print(seed)\n#color_random = tf.random.uniform(shape=[], minval=0, maxval=10, dtype=tf.int64)\n#color_random % 3 == 1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DIM = P['DIM']\n#cutDIM = 128\n\n\"\"\"\n# 哎 弄了两天 albu 在tpu 下彻底不兼容 如果强行开session转tensor.ops到np会影响性能 这个目前还是没辙\n# 数据自增\n# https://www.kaggle.com/kool777/training-hubmap-eda-tf-keras-tpu\ntransforms = albu.Compose([\n    albu.OneOf([\n        albu.RandomBrightness(limit=.2, p=1), \n        albu.RandomContrast(limit=.2, p=1), \n        albu.RandomGamma(p=1)\n    ], p=.5),\n    albu.OneOf([\n        albu.Blur(blur_limit=3, p=1),\n        albu.MedianBlur(blur_limit=3, p=1)\n    ], p=.25),\n    albu.OneOf([\n        albu.GaussNoise(0.002, p=.5),\n        albu.IAAAffine(p=.5),\n    ], p=.25),\n    albu.RandomRotate90(p=.5),\n    albu.HorizontalFlip(p=.5),\n    albu.VerticalFlip(p=.5),\n    albu.Cutout(num_holes=10, \n                max_h_size=int(.1 * DIM), max_w_size=int(.1 * DIM), \n                p=.25),\n    albu.ShiftScaleRotate(p=.25)\n])\n\n\ndef aug_fn(image, img_size, mask):\n    data = {\"image\":image, \"mask\": mask}\n    aug_data = transforms(**data)\n    aug_img = aug_data[\"image\"]\n    aug_img = tf.cast(aug_img/255.0, tf.float32)\n    aug_img = tf.image.resize(aug_img, size=[img_size, img_size])\n    \n    aug_mask = aug_data[\"mask\"]\n    aug_mask = tf.cast(aug_mask, tf.float32)\n    aug_mask = tf.image.resize(aug_mask, size=[img_size, img_size])    \n    return aug_img, aug_mask\n    #return image, mask\n\"\"\"\n\n#https://www.tensorflow.org/tutorials/images/data_augmentation\ndef _parse_image_function(example_proto, seed, augment = True):\n    image_feature_description = {\n        'image': tf.io.FixedLenFeature([], tf.string),\n        'mask': tf.io.FixedLenFeature([], tf.string)\n    }\n    single_example = tf.io.parse_single_example(example_proto, image_feature_description)\n    image = tf.reshape( tf.io.decode_raw(single_example['image'],out_type=np.dtype('uint8')), (DIM,DIM, 3))\n    mask =  tf.reshape(tf.io.decode_raw(single_example['mask'],out_type='bool'),(DIM,DIM,1))\n        \n    if augment:\n        # 这里代码要改一下 和取 tfrecord分开\n        # https://www.tensorflow.org/tutorials/images/data_augmentation#apply_the_preprocessing_layers_to_the_datasets\n        \n        if tf.random.uniform(()) > 0.5:\n            image = tf.image.flip_left_right(image)\n            mask = tf.image.flip_left_right(mask)    \n        # shiftscalerotate 0.25\n        if tf.random.uniform(()) > 0.4:\t        \n            image = tf.image.flip_up_down(image)\t        #if tf.random.uniform(()) > 0.75:\n            mask = tf.image.flip_up_down(mask)\n            \n        #image = tf.image.stateless_random_flip_left_right(image, seed) \n        #mask = tf.image.stateless_random_flip_left_right(mask, seed) \n\n        #image = tf.image.stateless_random_flip_up_down(image, seed)\n        #mask = tf.image.stateless_random_flip_up_down(mask, seed)\n\n        if tf.random.uniform(()) > 0.5:\n            image = tf.image.rot90(image)\n            mask = tf.image.rot90(mask)      \n        # shiftscalerotate 0.25\n        \n        #if tf.random.uniform(()) > 0.75:\n        #    image = tf.image.stateless_random_crop(image, size=[cutDIM, cutDIM, 3], seed=seed)\n        #    mask = tf.image.stateless_random_crop(mask, size=[cutDIM, cutDIM, 1], seed=seed)\n        \n        #  随机调整色调\n        if tf.random.uniform(()) > 0.75:\n            image = tf.image.stateless_random_hue(image, 0.2, seed=seed)    \n        \n        #  图片质量 这个好像会造成反向优化\n        #if tf.random.uniform(()) > 0.7:\n        #    image = tf.image.stateless_random_jpeg_quality(image, 75, 95, seed)\n            \n        if tf.random.uniform(()) > 0.75:\n            image = tf.image.stateless_random_saturation(image, 0.7, 1.3, seed=seed)\n            \n        if tf.random.uniform(()) > 0.75:\n            image = tf.image.stateless_random_contrast(image, lower=0.8, upper=1.2, seed=seed)\n        \n        if tf.random.uniform(()) > 0.75:\n            image = tf.image.stateless_random_brightness(image, max_delta=0.95, seed=seed)\n        \n        \"\"\"\n            #  颜色\n            color_random = tf.random.uniform(shape=[], minval=0, maxval=10, dtype=tf.int64)\n            color_random_resd = color_random % 3;\n            if color_random > 5:\n                # 三选一\n                if color_random_resd == 2:\n                    image = tf.image.stateless_random_saturation(image, 0.7, 1.3, seed=seed)    \n                elif color_random_resd == 1:\n                    image = tf.image.stateless_random_brightness(image, max_delta=0.95, seed=seed)\n                else: \n                    # color_random_resd == 0:\n                    image = tf.image.stateless_random_contrast(\n                      image, lower=0.8, upper=1.2, seed=seed)\n                image = tf.clip_by_value(image, 0, 1)\n        \"\"\"\n        \n\n        # Blur MedianBlur 二选一 \n        \n        \n        #noise = tf.random.normal(shape=tf.shape(image), mean=0.0, stddev=(50)/(255), dtype=tf.float32)\n        #image = image + noise\n        \n        # albu.GaussNoise(0.002, p=.5),albu.IAAAffine(p=.5), 二选一 0.25\n        #image = tf.image.resize(image, [DIM, DIM])\n        #mask = tf.image.resize(mask, [DIM, DIM])\n    return tf.cast(image, tf.float32), tf.cast(mask, tf.float32)\n    \"\"\"\n    # 自增工具 https://albumentations.ai/docs/examples/example_multi_target/\n    # map numpy_function 等的说明 https://www.tensorflow.org/api_docs/python/tf/data/Dataset\n    aug_image, aug_mask = tf.numpy_function(func=aug_fn, inp=[image, DIM, mask], Tout=[tf.float32, tf.float32])\n    # 需要恢复shape https://albumentations.ai/docs/examples/tensorflow-example/#restoring-dataset-shapes\n    aug_image.set_shape((DIM,DIM, 3))\n    aug_mask.set_shape((DIM,DIM, 1))\n    return aug_image, aug_mask\n    \"\"\"\n\ndef load_dataset(filenames, ordered=False, augment = True):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    dataset = dataset.with_options(ignore_order)\n    \n    counter = tf.data.experimental.Counter()\n    dataset = tf.data.Dataset.zip((dataset, (counter, counter)))\n    \n    dataset = dataset.map(lambda ex, i: _parse_image_function(ex, i, augment = augment), num_parallel_calls=AUTO)\n    #dataset = dataset.map(partial(_parse_image_function, augment = augment), num_parallel_calls=AUTO)\n    # 这一行计数会极大的影响性能\n    #dataset_length = [i for i,_ in enumerate(dataset)][-1] + 1        \n    #print(\"dataset_length\", dataset_length)            \n    return dataset\n# 之前这里值都一样 现在改成不一样的\ndef get_training_dataset(index= P['SEED']):\n    print(\"trainning data load\")\n    dataset = load_dataset(TRAINING_FILENAMES)\n    dataset = dataset.repeat()\n    dataset = dataset.shuffle(128, seed = index)\n    dataset = dataset.batch(BATCH_SIZE,drop_remainder=True)\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\ndef get_validation_dataset(ordered=True):\n    print(\"validate data load\")\n    dataset = load_dataset(VALIDATION_FILENAMES, ordered=ordered, augment = False)\n    dataset = dataset.batch(BATCH_SIZE,drop_remainder=True)\n    #dataset = dataset.cache()\n    dataset = dataset.prefetch(AUTO)\n    return dataset","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dataset = load_dataset(\"gs://kds-d633b6ed58b88a8006c19265c98f17c827c34cb7d20242fa7111baa9/train/0486052bb-355.tfrec\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n#遇到了一个浅拷贝的问题：get_training_dataset > load_dataset 如果重复运行 “似乎会取错 dataset 的并行执行单元的指针“，取之前被重复赋值之前的指针，而那个指针指向的”并行计算单元“已经被销毁了，就会导致报错“cannot connect alladress”错误\n\nDIM = P['DIM']\n\ntransforms = albu.Compose([\n    albu.OneOf([\n        albu.RandomBrightness(limit=.2, p=1), \n        albu.RandomContrast(limit=.2, p=1), \n        albu.RandomGamma(p=1)\n    ], p=.5),\n    albu.OneOf([\n        albu.Blur(blur_limit=3, p=1),\n        albu.MedianBlur(blur_limit=3, p=1)\n    ], p=.25),\n    albu.OneOf([\n        albu.GaussNoise(0.002, p=.5),\n        albu.IAAAffine(p=.5),\n    ], p=.25),\n    albu.RandomRotate90(p=.5),\n    albu.HorizontalFlip(p=.5),\n    albu.VerticalFlip(p=.5),\n    albu.Cutout(num_holes=10, \n                max_h_size=int(.1 * DIM), max_w_size=int(.1 * DIM), \n                p=.25),\n    albu.ShiftScaleRotate(p=.25)\n])\n\ndef aug_fn(image, img_size, mask):\n    data = {\"image\":image, \"mask\": mask}\n    aug_data = transforms(**data)\n    aug_img = aug_data[\"image\"]\n    #aug_img = tf.cast(aug_img/255.0, tf.float32)\n    #aug_img = tf.image.resize(aug_img, size=[img_size, img_size])\n    \n    aug_mask = aug_data[\"mask\"]\n    #aug_mask = tf.cast(aug_mask, tf.float32)\n    #aug_mask = tf.image.resize(aug_mask, size=[img_size, img_size])\n    return aug_img, aug_mask\n    #return image, mask\n\nfilenames = \"gs://kds-d633b6ed58b88a8006c19265c98f17c827c34cb7d20242fa7111baa9/train/0486052bb-355.tfrec\"\n\ntestArr = []\ndef _parse_image(example_proto):\n    image_feature_description = {\n        'image': tf.io.FixedLenFeature([], tf.string),\n        'mask': tf.io.FixedLenFeature([], tf.string)\n    }\n    single_example = tf.io.parse_single_example(example_proto, image_feature_description)\n    image = tf.reshape( tf.io.decode_raw(single_example['image'],out_type=np.dtype('uint8')), (DIM,DIM, 3))\n    mask =  tf.reshape(tf.io.decode_raw(single_example['mask'],out_type='bool'),(DIM,DIM,1))\n    image = tf.cast(image, tf.float32)\n    mask = tf.cast(mask, tf.float32)\n    testArr.append(image)\n    testArr.append(mask)\n    #aug_image, aug_mask = tf.numpy_function(func=aug_fn, inp=[image, size, mask], Tout=[tf.float32, tf.float32])\n    aug_image, aug_mask = aug_fn(image, DIM, mask)\n    \n    aug_image.set_shape((DIM,DIM, 3))\n    aug_mask.set_shape((DIM,DIM, 1))\n    return aug_image, aug_mask \n    #return image, mask \n    \n\ndataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n\n#dataset = dataset.map(partial(_parse_image), num_parallel_calls=AUTO)\ndataset_alb = dataset.map(partial(_parse_image), num_parallel_calls=AUTO)\n\"\"\"\n#print(testArr)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import PIL\n#import PIL.Image\n#type(testArr[0])\n\n#arr = np.ndarray(testArr[0])\n#arr_ = np.squeeze(arr)\n\n#plt.imshow(testArr[0])\n#plt.show()\n#PIL.Image.open(str(testArr[0]))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#type(element)\n#get_training_dataset()\n#get_validation_dataset()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nplt.imshow(element[0])\nplt.show()\n\"\"\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get_validation_dataset()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"# https://tensorlayer.readthedocs.io/en/latest/_modules/tensorlayer/cost.html#dice_coe\n\"\"\"\ndef dice_coe(output, target, axis = None, smooth=1e-10):\n    output = tf.dtypes.cast( tf.math.greater(output, 0.5), tf. float32 )\n    target = tf.dtypes.cast( tf.math.greater(target, 0.5), tf. float32 )\n    inse = tf.reduce_sum(output * target, axis=axis)\n    l = tf.reduce_sum(output, axis=axis)\n    r = tf.reduce_sum(target, axis=axis)\n\n    dice = (2. * inse + smooth) / (l + r + smooth)\n    dice = tf.reduce_mean(dice, name='dice_coe')\n    return dice\n\"\"\"\n\ndef dice_coe(y_true, y_pred, smooth=1):\n    y_true_f = K.flatten(y_true)\n    y_pred_f = K.flatten(y_pred)\n    intersection = K.sum(y_true_f * y_pred_f)\n    return (2 * intersection + smooth) / (K.sum(y_true_f) + K.sum(y_pred_f) + smooth)\n\n# https://www.kaggle.com/kool777/training-hubmap-eda-tf-keras-tpu\ndef tversky(y_true, y_pred, alpha=0.7, beta=0.3, smooth=1):\n    y_true_pos = K.flatten(y_true)\n    y_pred_pos = K.flatten(y_pred)\n    true_pos = K.sum(y_true_pos * y_pred_pos)\n    false_neg = K.sum(y_true_pos * (1 - y_pred_pos))\n    false_pos = K.sum((1 - y_true_pos) * y_pred_pos)\n    return (true_pos + smooth) / (true_pos + alpha * false_neg + beta * false_pos + smooth)\ndef tversky_loss(y_true, y_pred):\n    return 1 - tversky(y_true, y_pred)\ndef focal_tversky_loss(y_true, y_pred, gamma=0.75):\n    tv = tversky(y_true, y_pred)\n    return K.pow((1 - tv), gamma)\n\ndef dice_loss(y_true, y_pred, smooth=1):\n    return (1 - dice_coe(y_true, y_pred, smooth))\n\ndef bce_dice_loss(y_true, y_pred):\n    return PARAMS['bce_weight'] * binary_crossentropy(y_true, y_pred) + \\\n        (1 - PARAMS['bce_weight']) * dice_loss(y_true, y_pred)\n\nget_custom_objects().update({\"focal_tversky\": focal_tversky_loss})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model fit","metadata":{}},{"cell_type":"code","source":"%%time\n\nM = {}\nmetrics = ['loss','dice_coe'\n           #,'accuracy'\n          ]\nfor fm in metrics:\n    M['val_'+fm] = []\n\nfold = KFold(n_splits=P['NFOLDS'], shuffle=True, random_state=P['SEED'])\nfor fold,(tr_idx, val_idx) in enumerate(fold.split(ALL_TRAINING_FILENAMES)):\n    print('#'*35); print('############ FOLD ',fold+1,' #############'); print('#'*35);\n    print(f'Image Size: {DIM}, Batch Size: {BATCH_SIZE}')\n\n    # CREATE TRAIN AND VALIDATION SUBSETS\n    TRAINING_FILENAMES = [ALL_TRAINING_FILENAMES[fi] for fi in tr_idx]\n    \n    #这段似乎不太对\n    \"\"\"    \n    if P['OVERLAPP']:\n        TRAINING_FILENAMES += [ALL_TRAINING_FILENAMES2[fi] for fi in tr_idx]\n    \"\"\"\n\n    VALIDATION_FILENAMES = [ALL_TRAINING_FILENAMES[fi] for fi in val_idx]\n    STEPS_PER_EPOCH = P['STEPS_COE'] * count_data_items(TRAINING_FILENAMES) // BATCH_SIZE\n\n    # BUILD MODEL\n    K.clear_session()\n    with strategy.scope():   \n        model = sm.Unet(P['BACKBONE'], encoder_weights='imagenet')\n        #model = sm.Linknet(P['BACKBONE'], encoder_weights='imagenet')\n        #https://www.kaggle.com/bigironsphere/loss-function-library-keras-pytorch 目前最好的lr 为 1e-3 到  5e-4 附近\n        \"\"\"\n        model.compile(optimizer=tfa.optimizers.Lookahead(\n            tf.keras.optimizers.Adam(learning_rate=P['LR']),\n            sync_period=max(6, int(P['patience'] / 4))\n          ),\n        \n        \"\"\"\n\n        model.compile(optimizer = tf.keras.optimizers.Adam(lr = P['LR']),\n                      loss = tf.keras.losses.BinaryCrossentropy(),\n                      \n                      #loss='focal_tversky',\n                      \n                      #loss = bce_jaccard_loss,\n                      metrics=[dice_coe\n                              # ,'accuracy'\n                              ])\n    \"\"\"\n      optimizer=tfa.optimizers.Lookahead(\n            tf.keras.optimizers.Adam(learning_rate=P['LR']),\n            sync_period=max(6, int(P['patience'] / 4))\n      ),\n    \"\"\"\n\n    # CALLBACKS\n    checkpoint = tf.keras.callbacks.ModelCheckpoint('/kaggle/working/model-fold-%i.h5'%fold,\n                                 verbose=P['VERBOSE'],monitor='val_dice_coe',\n                                                    #patience = 10,\n                                 mode='max',save_best_only=True)\n\n    early_stop = tf.keras.callbacks.EarlyStopping(monitor='val_dice_coe',mode = 'max', patience=P['patience'], restore_best_weights=True)\n    reduce = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.1, patience=int(P['patience'] / 2), min_lr=P['LR'] / 1e3)\n\n    print(f'Training Model Fold {fold+1}...')\n\n    log_dir = \"logs/fit/\" + datetime.now().strftime(\"%Y%m%d-%H%M%S\")\n    tensorboard_callback = tf.keras.callbacks.TensorBoard(log_dir=log_dir, histogram_freq=1)\n\n    history = model.fit(\n        get_training_dataset(index=fold),\n        epochs = P['EPOCHS'],\n        steps_per_epoch = STEPS_PER_EPOCH,\n        callbacks = [\n            #tensorboard_callback, \n            checkpoint, reduce,early_stop],\n        validation_data = get_validation_dataset(),\n        verbose=P['VERBOSE']\n    )   \n\n    #with strategy.scope():\n    #    model = tf.keras.models.load_model('/kaggle/working/model-fold-%i.h5'%fold, custom_objects = {\"dice_coe\": dice_coe})\n\n    # SAVE METRICS\n    m = model.evaluate(get_validation_dataset(),return_dict=True)\n    for fm in metrics:\n        M['val_'+fm].append(m[fm])\n\n    # PLOT TRAINING\n    # https://www.kaggle.com/cdeotte/triple-stratified-kfold-with-tfrecords\n    if P['DISPLAY_PLOT']:        \n        plt.figure(figsize=(15,5))\n        n_e = np.arange(len(history.history['dice_coe']))\n        plt.plot(n_e,history.history['dice_coe'],'-o',label='Train dice_coe',color='#ff7f0e')\n        plt.plot(n_e,history.history['val_dice_coe'],'-o',label='Val dice_coe',color='#1f77b4')\n        x = np.argmax( history.history['val_dice_coe'] ); y = np.max( history.history['val_dice_coe'] )\n        xdist = plt.xlim()[1] - plt.xlim()[0]; ydist = plt.ylim()[1] - plt.ylim()[0]\n        plt.scatter(x,y,s=200,color='#1f77b4'); plt.text(x-0.03*xdist,y-0.13*ydist,'max dice_coe\\n%.2f'%y,size=14)\n        plt.ylabel('dice_coe',size=14); plt.xlabel('Epoch',size=14)\n        plt.legend(loc=2)\n        plt2 = plt.gca().twinx()\n        plt2.plot(n_e,history.history['loss'],'-o',label='Train Loss',color='#2ca02c')\n        plt2.plot(n_e,history.history['val_loss'],'-o',label='Val Loss',color='#d62728')\n        x = np.argmin( history.history['val_loss'] ); y = np.min( history.history['val_loss'] )\n        ydist = plt.ylim()[1] - plt.ylim()[0]\n        plt.scatter(x,y,s=200,color='#d62728'); plt.text(x-0.03*xdist,y+0.05*ydist,'min loss',size=14)\n        plt.ylabel('Loss',size=14)\n        plt.legend(loc=3)\n        plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### WRITE METRICS\nM['datetime'] = str(datetime.now())\nfor fm in metrics:\n    M['oof_'+fm] = np.mean(M['val_'+fm])\n    print('OOF '+ fm + ' '+ str(M['oof_'+fm]))\nwith open('metrics.json', 'w') as outfile:\n    json.dump(M, outfile)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# this need some special configurature for tpu \n#%tensorboard --logdir logs/fit","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!ls","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}