{"cells":[{"metadata":{},"cell_type":"markdown","source":"# It is my \"hello word\" code for this competition. \n\nused the adapted version of  https://blogs.rstudio.com/ai/posts/2019-08-23-unet/"},{"metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","trusted":true},"cell_type":"code","source":"library(tensorflow)\nlibrary(keras)\nlibrary(tfdatasets)\nlibrary(tidyverse)\nlibrary(rsample)\nlibrary(reticulate)\nlibrary(magick)\n\ndata.train<- file.path(\"../input/512x512/train_images/train_images\")\ndata.mask<- file.path(\"../input/512x512/mask_images/mask_images\")\nTRAIN <- file.path(\"../input/hubmap-kidney-segmentation/train\")\nbasedir<- file.path(\"../input/hubmap-kidney-segmentation\")\nOUTPUT<-file.path(\"./\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#train<- read_delim(file.path(basedir,\"train.csv\"), delim = \",\")\n#test<-read_csv(file.path(basedir,\"sample_submission.csv\"))\n#dataset<- read_csv(file.path(basedir,\"HuBMAP-20-dataset_information.csv\"))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#ff<- list.files(TRAIN, full)\n#anato<-grep(\"anato\",ff, value=T)\n#img<-grep(\"tiff\",ff, value=T)\n#jsons<- setdiff(ff,c(anato,img))\n\n\n#data_json <- tibble(\n # jsons= lapply(jsons ,fromJSON),\n # anato = lapply(anato,fromJSON)\n#)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"mask2rle<- function(x){\n   pixel <- x %>% as.vector\n   xx<- rle(pixel)\n   ids<- which (xx$values != 0 )\n   lls<- xx$lengths[ids]\n   cs<- cumsum(xx$lengths)\n   css<-cs[ids-1]+1\n   return( paste(css,lls,sep = \" \", collapse = \" \") )\n}\n\nrle2mask <- function(mask_rle,shape){\n  sl<- mask_rle   %>% strsplit(\" \") %>% .[[1]] %>% as.numeric\n  starts<-sl[seq(1,(length(sl)/2 -1),2)] \n  lengths<-sl[seq(2,length(sl)/2,2)]\n  ends<-starts + lengths -1\n  img<- matrix(0,shape[1] * shape[2]  )\n  ids<-lapply(1:length(starts),function(i){seq(starts[i],ends[i],1)}) %>% unlist\n  img[ids,1]<-1\n  #img<-rep(img,3)\n  dim(img)<-c(shape[1],shape[2],1)\n  return((img))\n}\n\nmakeMask<-function(i){\nmask_rle<- train %>% subset(id == ffnames[i]) %>% dplyr::select(encoding) %>% pull\nshape_rle<-dataset %>% subset(image_file == paste0(ffnames[i], \".tiff\")) %>% dplyr::select(width_pixels, height_pixels) %>% unlist %>% as.vector\nmask<- rle2mask(mask_rle, shape_rle)\nmd<- dim(mask)\ndim(mask)<-c(md[2],md[1],md[3])\nout<-image_read(mask)\nreturn(out)\n\n}\nffnames<-list.files(TRAIN, full.names = F,pattern = \"*.tiff\") %>% gsub(pattern = \".tiff\",replacement =  \"\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#patht<-file.path(basedir,\"train_images\")\n#if(!dir.exists(patht)) {\n#dir.create(patht)\n#}\n#pathm<-file.path(basedir,\"mask_images\")\n#if(!dir.exists(pathm)) {\n#dir.create(pathm)\n#}\n\nmakePng<-function(i,j,k){\n  shape<-dataset %>% subset(image_file == paste0(ffnames[k], \".tiff\")) %>% dplyr::select(width_pixels, height_pixels) %>% unlist %>% as.vector\n  im<- image_read(img[[k]])\n  mask<- makeMask(k)\n  gc()\n  nl<-c( seq(0,shape[1]-SIZE, SIZE), shape[1]-SIZE)\n  nc<-c( seq(0,shape[2]-SIZE, SIZE) , shape[2]-SIZE)\n  nn<-expand.grid(nl,nc)\n  \n  dest<- paste0(SIZE,\"x\",SIZE, \"+\",i, \"+\", j)\n  imager::save.image(\n      image_crop(im,dest) %>% .[[1]] %>% as.numeric %>% array(dim=c (SIZE,SIZE,1,3)) %>% as.cimg(cc=3),\n        file = file.path(patht, paste0(\"image\",\"_\" ,ffnames[k],\"_\",i,\"_\",j,\"_\" ,\".png\") ))\n\n  imager::save.image(\n      image_crop(mask,dest)  %>% .[[1]] %>% as.numeric %>% array(dim=c (SIZE,SIZE,1,3)) %>% as.cimg(cc=3),\n      file = file.path(pathm, paste0(\"mask\",\"_\" ,ffnames[k],\"_\",i,\"_\",j, \"_\" ,\".png\") ))\n}\n#walk2(nn[,1],nn[,2],makePng,k)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"The functions above were used to make the png, with size = 2 ^9 =512, so it was run previously.  Tested both rel encode and decode and they match and worked fine. "},{"metadata":{"trusted":true},"cell_type":"code","source":"conv2d_block <- function(inputs, use_batch_norm = TRUE, dropout = 0.3,\n                         filters = 16, kernel_size = c(3, 3), activation = \"relu\",\n                         kernel_initializer = \"he_normal\", padding = \"same\") {\n\n  x <- keras::layer_conv_2d(\n    inputs,\n    filters = filters,\n    kernel_size = kernel_size,\n    activation = activation,\n    kernel_initializer = kernel_initializer,\n    padding = padding\n  )\n\n  if (use_batch_norm) {\n    x <- keras::layer_batch_normalization(x)\n  }\n\n  if (dropout > 0) {\n    x <- keras::layer_dropout(x, rate = dropout)\n  }\n\n  x <- keras::layer_conv_2d(\n    x,\n    filters = filters,\n    kernel_size = kernel_size,\n    activation = activation,\n    kernel_initializer = kernel_initializer,\n    padding = padding\n  )\n\n  if (use_batch_norm) {\n    x <- keras::layer_batch_normalization(x)\n  }\n\n  x\n}\n\n#' U-Net: Convolutional Networks for Biomedical Image Segmentation\n#'\n#' @param input_shape Dimensionality of the input (integer) not including the\n#'   samples axis. Must be length 3 numeric vector.\n#' @param num_classes Number of classes.\n#' @param dropout Dropout rate applied between downsampling and upsampling phases.\n#' @param filters Number of filters of the first convolution.\n#' @param num_layers Number of downsizing blocks in the encoder.\n#' @param  output_activation Activation in the output layer.\n#'\n#' @export\nunet <- function(input_shape, num_classes = 1, dropout = 0.5, filters = 64,\n                 num_layers = 4, output_activation = \"sigmoid\") {\n\n\n  input <- keras::layer_input(shape = input_shape)\n\n  x <- input\n  down_layers <- list()\n\n  for (i in seq_len(num_layers)) {\n\n    x <- conv2d_block(\n      inputs = x,\n      filters = filters,\n      use_batch_norm = FALSE,\n      dropout = 0,\n      padding = \"same\"\n    )\n\n    down_layers[[i]] <- x\n\n    x <- keras::layer_max_pooling_2d(x, pool_size = c(2,2), strides = c(2,2))\n\n    filters <- filters * 2\n\n  }\n\n  if (dropout > 0) {\n    x <- keras::layer_dropout(x, rate = dropout)\n  }\n\n  x <- conv2d_block(\n    inputs = x,\n    filters = filters,\n    use_batch_norm = FALSE,\n    dropout = 0.0,\n    padding = 'same'\n  )\n\n  for (conv in rev(down_layers)) {\n\n    filters <- filters / 2L\n\n    x <- keras::layer_conv_2d_transpose(\n      x,\n      filters = filters,\n      kernel_size = c(2,2),\n      padding = \"same\",\n      strides = c(2,2)\n    )\n\n    x <- keras::layer_concatenate(list(conv, x))\n    x <- conv2d_block(\n      inputs = x,\n      filters = filters,\n      use_batch_norm = FALSE,\n      dropout = 0.0,\n      padding = 'same'\n    )\n\n  }\n\n  output <- keras::layer_conv_2d(\n    x,\n    filters = num_classes,\n    kernel_size = c(1,1),\n    activation = output_activation\n  )\n\n  model <- keras::keras_model(input, output)\n\n  model\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nBATCH_SIZE = 8\n\n\n# libraries we're going to need later\nfnames<-list.files((data.train)) %>% map( . %>% gsub(pattern = \"image\", replacement = \"\"))\n\n\nimages <- tibble(\n  img = file.path(data.train,paste0(\"image\",fnames)),  #list.files((data.train), full.names = TRUE),\n  mask =  file.path(data.mask, paste0(\"mask\",fnames) ) #list.files((data.mask), full.names = TRUE)\n  ) %>% \n  sample_n(2) %>%\n  map(. %>% magick::image_read() %>% magick::image_resize(\"128x128\"))\n\nout <- magick::image_append(c(\n  magick::image_append(images$img, stack = TRUE), \n  magick::image_append(images$mask, stack = TRUE)\n  )\n)\n\n\n\ndata <- tibble(\n  img = file.path(data.train,paste0(\"image\",fnames)),  #list.files((data.train), full.names = TRUE),\n  mask =  file.path(data.mask, paste0(\"mask\",fnames) ) #list.files((data.mask), full.names = TRUE)\n)\ndata <- initial_split(data, prop = 0.8)\n\n\n\n\n\n\ntraining_dataset <- training(data) %>%  \n  tensor_slices_dataset() %>% \n  dataset_map(~.x %>% list_modify(\n    # decode_jpeg yields a 3d tensor of shape (1280, 1918, 3)\n      img =tf$image$decode_png(tf$io$read_file(.x$img)),\n\n    # decode_gif yields a 4d tensor of shape (1, 1280, 1918, 3),\n    # so we remove the unneeded batch dimension and all but one \n    # of the 3 (identical) channels\n    mask = tf$image$decode_png(tf$io$read_file(.x$mask))[,,1,drop=F]\n\n  ))%>% \n  dataset_map(~.x %>% list_modify(\n    img = tf$image$convert_image_dtype(.x$img, dtype = tf$float32),\n    mask = tf$image$convert_image_dtype(.x$mask, dtype = tf$float32)\n  ))\n\nexample <- training_dataset %>% as_iterator() %>% iter_next()\nexample\n\n\n# training_dataset <- training_dataset %>% \n#   dataset_map(~.x %>% list_modify(\n#     img = tf$image$convert_image_dtype(.x$img, dtype = tf$float32),\n#     mask = tf$image$convert_image_dtype(.x$mask, dtype = tf$float32)\n#   ))\n# \n\nrandom_bsh <- function(img) {\n  img %>%\n    tf$image$random_brightness(max_delta = 0.3) %>%\n    tf$image$random_contrast(lower = 0.5, upper = 0.7) %>%\n    tf$image$random_saturation(lower = 0.5, upper = 0.7) %>%\n    # make sure we still are between 0 and 1\n    tf$clip_by_value(0, 1)\n}\n\ntraining_dataset <- training_dataset %>%\n  dataset_map(~.x %>% list_modify(\n    img = random_bsh(.x$img)\n  ))\n\n\n\n\nexample <- training_dataset %>% as_iterator() %>% iter_next()\nexample$img %>% as.array() %>% image_read() %>% plot()\n\ncreate_dataset <- function(data, train, batch_size = BATCH_SIZE) {\n  \n  dataset <- data %>% \n    tensor_slices_dataset() %>% \n   dataset_map(~.x %>% list_modify(\n    # decode_jpeg yields a 3d tensor of shape (1280, 1918, 3)\n    img = tf$image$decode_png(tf$io$read_file(.x$img)),\n    # decode_gif yields a 4d tensor of shape (1, 1280, 1918, 3),\n    # so we remove the unneeded batch dimension and all but one \n    # of the 3 (identical) channels\n    mask =tf$image$decode_png(tf$io$read_file(.x$mask))\n  ))%>% \n  dataset_map(~.x %>% list_modify(\n    img = tf$image$convert_image_dtype(.x$img, dtype = tf$float32),\n    mask = tf$image$convert_image_dtype(.x$mask, dtype = tf$float32)\n  )) %>%\n      dataset_map(~.x %>% list_modify(\n      img = tf$image$resize(.x$img, size = shape(256, 256)),\n      mask = tf$image$resize(.x$mask, size = shape(256, 256))[,,1,drop=F]\n    ))\n  \n\n\n#  data augmentation performed on training set only\n  if (train) {\n    dataset <- dataset %>%\n      dataset_map(~.x %>% list_modify(\n        img = random_bsh(.x$img)\n      ))\n  }\n\n  # shuffling on training set only\n  if (train) {\n    dataset <- dataset %>% \n      dataset_shuffle(buffer_size = batch_size*2^5)\n  }\n  \n  # train in batches; batch size might need to be adapted depending on\n  # available memory\n  dataset <- dataset %>% \n    dataset_batch(batch_size)\n  \n  dataset %>% \n    # output needs to be unnamed\n    dataset_map(unname) \n}\n\ntraining_dataset <- create_dataset(training(data), train = TRUE)\nvalidation_dataset <- create_dataset(testing(data), train = FALSE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n\nmodel <- unet(input_shape = c(256, 256, 3))\nsummary(model)\n\n\ndice <- custom_metric(\"dice\", function(y_true, y_pred, smooth = 0) {\n  y_true_f <- k_flatten(y_true)\n  y_pred_f <- k_flatten(y_pred)\n  intersection <- k_sum(y_true_f * y_pred_f)\n  (2 * intersection + smooth) / (k_sum(y_true_f) + k_sum(y_pred_f) + smooth)\n})\n\nmodel %>% compile(\n  optimizer = optimizer_rmsprop(lr = 1e-5),\n  loss = \"binary_crossentropy\",\n  metrics = list(dice, metric_binary_accuracy)\n)\n\n\nhist<- model %>% \n  fit(\n  training_dataset,\n  epochs = 5,\n  #batch_size=2^10,\n  verbose = 1,\n  #validation_split = 0.2\n  validation_data = validation_dataset\n    )\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x <- list.files(data.mask,full.names = TRUE)\nx<- x[sapply(x, file.size) > 1500]\next<- x %>% map( . %>% gsub(pattern = file.path(data.mask,\"mask\"), replacement = \"\"))\nims<-   paste0(file.path(data.train,\"image\"),ext) %>% unlist\nimages <- tibble(\n  img = x, \n  mask =ims \n  ) %>% \n  sample_n(5) %>%\n  map(. %>% magick::image_read() %>% magick::image_resize(\"128x128\"))\n\nout <- magick::image_append(c(\n  magick::image_append(images$img, stack = TRUE), \n  magick::image_append(images$mask, stack = TRUE)\n  )\n)\nplot(out)\nii <- tibble(\n  img = x, \n  mask =ims \n  )\nevaldata<-create_dataset(ii, train = F, batch_size=4)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nbatch <- evaldata %>% as_iterator() %>% iter_next()\npredictions <- predict(model, batch)\n\nimages <- tibble(\n  image = batch[[1]] %>% array_branch(1),\n  predicted_mask = predictions[,,,1] %>% array_branch(1),\n  mask = batch[[2]][,,,1]  %>% array_branch(1)\n) %>% \n  sample_n(3) %>% \n  map_depth(2, function(x) {\n    as.raster(x) %>% magick::image_read()\n  }) %>% \n  map(~do.call(c, .x))\n\n\nout <- magick::image_append(c(\n  magick::image_append(images$mask, stack = TRUE),\n  magick::image_append(images$image, stack = TRUE), \n  magick::image_append(images$predicted_mask, stack = TRUE)\n  )\n)\n\nplot(out)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"3.6.3"}},"nbformat":4,"nbformat_minor":4}