{"cells":[{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import os\nimport PIL\nimport json\nimport pickle\nimport numpy as np\nfrom math import log, exp\nfrom tqdm import tqdm\nfrom random import shuffle\nfrom PIL import ImageEnhance, ImageFont, ImageDraw\nfrom IPython.display import Image, display\nfrom multiprocessing import Pool\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.regularizers import l2\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.python.keras.utils.data_utils import Sequence\n\ntf.keras.backend.clear_session()  # For easy reset of notebook state.\n\ncat_list = ['tops', 'trousers', 'shirt', 'dresses', 'skirts']\n\ninput_shape = (224,224,3)\nwt_decay = 5e-4\n\ndims_list = [(7,7),(14,14)]\naspect_ratios = [(1,1), (1,2), (2,1)]\n\n#base_folder = \"/content/gdrive/My Drive/CV_Pictures/til2020\"\nbase_folder = '/kaggle'\ndata_folder = os.path.join( base_folder, 'data')\ntrain_imgs_folder = os.path.join( data_folder, 'train', 'train' )\ntrain_annotations = os.path.join( data_folder, 'train.json' )\nval_imgs_folder = os.path.join( data_folder, 'val', 'val' )\nval_annotations = os.path.join( data_folder, 'val.json' )\n\ntrain_pickle = os.path.join( data_folder,'train.p' )\nval_pickle = os.path.join( data_folder ,'val.p' )\n\nsave_model_folder = os.path.join( base_folder, 'model' )\nload_model_folder = data_folder\n\n# Precomputed detections json file for tutorial\nmodel_detections = os.path.join(data_folder, 'detections-7x7-14x14-top100.json')\n\nprint (train_pickle)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nfrom PIL import Image\nimport json\nimport pickle\nimport numpy as np\nfrom math import log, exp\nfrom tqdm import tqdm\nfrom random import shuffle\nfrom PIL import ImageEnhance, ImageFont, ImageDraw,Image\nfrom IPython.display import Image, display\nfrom multiprocessing import Pool\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.regularizers import l2\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.python.keras.utils.data_utils import Sequence\n\nbase_folder='/kaggle'\n\ndata_folder = os.path.join( base_folder, 'data')\ntrain_imgs_folder = os.path.join( data_folder, 'train')\n\ntrain_imgs_folder2 = os.path.join( train_imgs_folder, 'train')\n\n#print (train_imgs_folder2)\nimage_filepath = os.path.join( train_imgs_folder2, '6')\n#img=Image.open('5')\n#img.show()\ndisplay(image_filepath)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.python.client import device_lib\n\ndef get_available_devices():\n    local_device_protos = device_lib.list_local_devices()\n    return [x.name for x in local_device_protos]\n\nprint(get_available_devices())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''\nAugmentation methods. We need to implement our own augmentation because native support in keras does not change the bounding box \nlabels for us as the image is altered. We need to do it ourselves.\n'''\n# Helper method: Computes the boundary of the image that includes all bboxes\ndef compute_reasonable_boundary(labels):\n  bounds = [ (x-w/2, x+w/2, y-h/2, y+h/2) for _,x,y,w,h in labels]\n  xmin = min([bb[0] for bb in bounds])\n  xmax = max([bb[1] for bb in bounds])\n  ymin = min([bb[2] for bb in bounds])\n  ymax = max([bb[3] for bb in bounds])\n  return xmin, xmax, ymin, ymax\n\ndef aug_horizontal_flip(img, labels):\n  flipped_labels = []\n  for c,x,y,w,h in labels:\n    flipped_labels.append( (c,1-x,y,w,h) )\n  return img.transpose(PIL.Image.FLIP_LEFT_RIGHT), np.array(flipped_labels)\n\ndef aug_crop(img, labels):\n  # Compute bounds such that no boxes are cut out\n  xmin, xmax, ymin, ymax = compute_reasonable_boundary(labels)\n  # Choose crop_xmin from [0, xmin]\n  crop_xmin = max( np.random.uniform() * xmin, 0 )\n  # Choose crop_xmax from [xmax, 1]\n  crop_xmax = min( xmax + (np.random.uniform() * (1-xmax)), 1 )\n  # Choose crop_ymin from [0, ymin]\n  crop_ymin = max( np.random.uniform() * ymin, 0 )\n  # Choose crop_ymax from [ymax, 1]\n  crop_ymax = min( ymax + (np.random.uniform() * (1-ymax)), 1 )\n  # Compute the \"new\" width and height of the cropped image\n  crop_w = crop_xmax - crop_xmin\n  crop_h = crop_ymax - crop_ymin\n  cropped_labels = []\n  for c,x,y,w,h in labels:\n    c_x = (x - crop_xmin) / crop_w\n    c_y = (y - crop_ymin) / crop_h\n    c_w = w / crop_w\n    c_h = h / crop_h\n    cropped_labels.append( (c,c_x,c_y,c_w,c_h) )\n\n  W,H = img.size\n  # Compute the pixel coordinates and perform the crop\n  impix_xmin = int(W * crop_xmin)\n  impix_xmax = int(W * crop_xmax)\n  impix_ymin = int(H * crop_ymin)\n  impix_ymax = int(H * crop_ymax)\n  return img.crop( (impix_xmin, impix_ymin, impix_xmax, impix_ymax) ), np.array( cropped_labels )\n\ndef aug_translate(img, labels):\n  # Compute bounds such that no boxes are cut out\n  xmin, xmax, ymin, ymax = compute_reasonable_boundary(labels)\n  trans_range_x = [-xmin, 1 - xmax]\n  tx = trans_range_x[0] + (np.random.uniform() * (trans_range_x[1] - trans_range_x[0]))\n  trans_range_y = [-ymin, 1 - ymax]\n  ty = trans_range_y[0] + (np.random.uniform() * (trans_range_y[1] - trans_range_y[0]))\n\n  trans_labels = []\n  for c,x,y,w,h in labels:\n    trans_labels.append( (c,x+tx,y+ty,w,h) )\n\n  W,H = img.size\n  tx_pix = int(W * tx)\n  ty_pix = int(H * ty)\n  return img.rotate(0, translate=(tx_pix, ty_pix)), np.array( trans_labels )\n\ndef aug_colorbalance(img, labels, color_factors=[0.2,2.0]):\n  factor = color_factors[0] + np.random.uniform() * (color_factors[1] - color_factors[0])\n  enhancer = ImageEnhance.Color(img)\n  return enhancer.enhance(factor), labels\n\ndef aug_contrast(img, labels, contrast_factors=[0.2,2.0]):\n  factor = contrast_factors[0] + np.random.uniform() * (contrast_factors[1] - contrast_factors[0])\n  enhancer = ImageEnhance.Contrast(img)\n  return enhancer.enhance(factor), labels\n\ndef aug_brightness(img, labels, brightness_factors=[0.2,2.0]):\n  factor = brightness_factors[0] + np.random.uniform() * (brightness_factors[1] - brightness_factors[0])\n  enhancer = ImageEnhance.Brightness(img)\n  return enhancer.enhance(factor), labels\n\ndef aug_sharpness(img, labels, sharpness_factors=[0.2,10.0]):\n  factor = sharpness_factors[0] + np.random.uniform() * (sharpness_factors[1] - sharpness_factors[0])\n  enhancer = ImageEnhance.Sharpness(img)\n  return enhancer.enhance(factor), labels\n\n# Performs no augmentations and returns the original image and bbox. Used for the validation images.\ndef aug_identity(pil_img, label_arr):\n  return np.array(pil_img), label_arr\n\n# This is the default augmentation scheme that we will use for each training image.\ndef aug_default(img, labels, p={'flip':0.5, 'crop':0.5, 'translate':0.5, 'color':0.2, 'contrast':0.2, 'brightness':0.2, 'sharpness':0.2}):\n  if p['color'] > np.random.uniform():\n    img, labels = aug_colorbalance(img, labels)\n  if p['contrast'] > np.random.uniform():\n    img, labels = aug_contrast(img, labels)\n  if p['brightness'] > np.random.uniform():\n    img, labels = aug_brightness(img, labels)\n  if p['sharpness'] > np.random.uniform():\n    img, labels = aug_sharpness(img, labels)\n  \n  if p['flip'] > np.random.uniform():\n    img, labels = aug_horizontal_flip(img, labels)\n  if p['crop'] > np.random.uniform():\n    img, labels = aug_crop(img, labels)\n  if p['translate'] > np.random.uniform():\n    img, labels = aug_translate(img, labels)\n  return np.array(img), labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Shape of ypred: ( batch, i, j, aspect_ratios, 1+4+numclasses ). For a batch,i,j, we get #aspect_ratios vectors of length 7.\n# Shape of ytrue: ( batch, i, j, aspect_ratios, 1+4+numclasses+2 ). For a batch,i,j, we get #aspect_ratios vectors of length 9 (two more for objectness and cat/loc indicators)\ndef custom_loss(ytrue, ypred):\n  obj_loss_weight = 1.0\n  cat_loss_weight = 1.0\n  loc_loss_weight = 1.0\n\n  end_cat = len(cat_list) + 1\n\n  objloss_indicators = ytrue[:,:,:,:,-2:-1]\n  catlocloss_indicators = ytrue[:,:,:,:,-1:]\n\n  ytrue_obj, ypred_obj = ytrue[:,:,:,:,:1], ypred[:,:,:,:,:1]\n  ytrue_obj = tf.where( objloss_indicators != 0, ytrue_obj, 0 )\n  ypred_obj = tf.where( objloss_indicators != 0, ypred_obj, 0 )\n  objectness_loss = tf.keras.losses.BinaryCrossentropy(from_logits=True)( ytrue_obj, ypred_obj )\n\n  ytrue_cat, ypred_cat = ytrue[:,:,:,:,1:end_cat], ypred[:,:,:,:,1:end_cat]\n  ytrue_cat = tf.where( catlocloss_indicators != 0, ytrue_cat, 0 )\n  ypred_cat = tf.where( catlocloss_indicators != 0, ypred_cat, 0 )\n  categorical_loss = tf.keras.losses.CategoricalCrossentropy(from_logits=True) ( ytrue_cat, ypred_cat )\n\n  # Remember that ytrue is longer than ypred, so we will need to stop at index -2, which is where the indicators are stored\n  ytrue_loc, ypred_loc = ytrue[:,:,:,:,end_cat:-2], ypred[:,:,:,:,end_cat:]\n  ytrue_loc = tf.where( catlocloss_indicators != 0, ytrue_loc, 0 )\n  ypred_loc = tf.where( catlocloss_indicators != 0, ypred_loc, 0 )\n  localisation_loss = tf.keras.losses.Huber() ( ytrue_loc, ypred_loc )\n\n  return obj_loss_weight*objectness_loss + cat_loss_weight*categorical_loss + loc_loss_weight*localisation_loss\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Computes the intersection-over-union (IoU) of two bounding boxes\ndef iou(bb1, bb2):\n  x1,y1,w1,h1 = bb1\n  xmin1 = x1 - w1/2\n  xmax1 = x1 + w1/2\n  ymin1 = y1 - h1/2\n  ymax1 = y1 + h1/2\n\n  x2,y2,w2,h2 = bb2\n  xmin2 = x2 - w2/2\n  xmax2 = x2 + w2/2\n  ymin2 = y2 - h2/2\n  ymax2 = y2 + h2/2\n\n  area1 = w1*h1\n  area2 = w2*h2\n\n  # Compute the boundary of the intersection\n  xmin_int = max( xmin1, xmin2 )\n  xmax_int = min( xmax1, xmax2 )\n  ymin_int = max( ymin1, ymin2 )\n  ymax_int = min( ymax1, ymax2 )\n  intersection = max(xmax_int - xmin_int, 0) * max( ymax_int - ymin_int, 0 )\n\n  # Remove the double counted region\n  union = area1+area2-intersection\n\n  return intersection / union\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Sampling schemes\ndef yolo_posneg_sampling(iou_scores_dict, label_tensor, gtclass, cat_list, iou_threshold=0.5):\n  iou_scores = []\n  for _, scores in iou_scores_dict.items():\n    iou_scores.extend(scores)\n  iou_scores.sort( key=lambda x: x[0], reverse=True )\n  \n  top_iou_score = iou_scores.pop(0)\n  _, key, i, j, k, dx, dy, dw, dh = top_iou_score\n  zeros = [0] * len(cat_list)\n  payload = [1, *zeros, dx,dy,dw,dh]\n  payload[gtclass + 1] = 1\n  # Train objectness, class and loc for the positive\n  label_tensor[key][i,j,k,-2:] = 1\n  label_tensor[key][i,j,k,:len(payload)] = payload\n\n  # Train objectness only for the negatives\n  low_iou_scores = [iou_score for iou_score in iou_scores if iou_score[0] < iou_threshold]\n  for _, key, i, j, k, _, _, _, _ in low_iou_scores:\n    label_tensor[key][i,j,k,-2] = 1\n\ndef modified_yolo_posneg_sampling(iou_scores_dict, label_tensor, gtclass, cat_list, iou_threshold=0.5):\n  iou_scores = []\n  zeros = [0] * len(cat_list)\n\n  for _, scores in iou_scores_dict.items():\n    iou_scores.extend(scores)\n  iou_scores.sort( key=lambda x: x[0], reverse=True )\n  \n  top_iou_score = iou_scores.pop(0)\n  _, key, i, j, k, dx, dy, dw, dh = top_iou_score\n  payload = [1, *zeros, dx,dy,dw,dh]\n  payload[gtclass + 1] = 1\n  # Train objectness, class and loc for the positive\n  label_tensor[key][i,j,k,-2:] = 1\n  label_tensor[key][i,j,k,:len(payload)] = payload\n\n  # Train objectness only for the negatives\n  low_iou_scores = [iou_score for iou_score in iou_scores if iou_score[0] < iou_threshold]\n  for _, key, i, j, k, _, _, _, _ in low_iou_scores:\n    label_tensor[key][i,j,k,-2] = 1\n\n  # Train cat/loc only for the in-betweens - those with high IoU but not positive\n  high_iou_scores = [iou_score for iou_score in iou_scores if iou_score[0] >= iou_threshold]\n  for _, key, i, j, k, dx, dy, dw, dh in high_iou_scores:\n    label_tensor[key][i,j,k,-1] = 1\n    payload = [0,*zeros,dx,dy,dw,dh]\n    payload[gtclass + 1] = 1\n    label_tensor[key][i,j,k,:len(payload)] = payload\n\ndef top_ratio_sampling(iou_scores_dict, label_tensor, gtclass, cat_list, positive_ratio=0.25):\n  iou_scores = []\n  # Let all tensors learn objectness score\n  for v in label_tensor.values():\n    v[:,:,:,-2] = 1\n  \n  for _, iou_score_list in iou_scores_dict.items():\n    iou_score_list.sort( key=lambda x: x[0], reverse=True )\n    top_percentile_iou_scores = iou_score_list[:round(len(iou_score_list) * positive_ratio)]\n    # Include the rest that cross the IoU threshold\n    iou_score_list = top_percentile_iou_scores + [iou_score for iou_score in iou_score_list[len(top_percentile_iou_scores):] if iou_score[0] >= self.iou_threshold]\n    iou_scores.extend( iou_score_list )\n\n  for iou_score in iou_scores:\n    IoU, key, i, j, k, dx, dy, dw, dh = iou_score\n    zeros = [0] * len(cat_list)\n    payload = [IoU, *zeros, dx,dy,dw,dh]\n    payload[gtclass + 1] = 1\n    label_tensor[key][i,j,k,:len(payload)] = payload\n    # Set the classification/localisation indicator at this location to positive\n    label_tensor[key][i,j,k,-1] = 1\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''\nEncoder: label -> tensor\nlabel_arr: np array like:\n[[class_idx x y w h]]: num_labels x 5\n...  \nUsed to figure out for each label line, which tensor entry to shove it into.\nIf the box corresponding to the tensor entry overlaps the ground truth by at least a predefined threshold, then we shove it in.\n'''\ndef encode_label(label_arr, dims_list, aspect_ratios, iou_fn, sampling_fn, cat_list):\n  num_entries = 7 + len(cat_list) # objectness, ... len(cat_list) ..., dx, dy, dw, dh, obj_indicator, catloc_indicator\n  np_labels = {}\n  for dims in dims_list:\n    dimkey = '{}x{}'.format(*dims)\n    np_labels[dimkey] = np.zeros( (*dims, len(aspect_ratios), num_entries ) )\n\n  for label in label_arr:\n    gtclass, gtx, gty, gtw, gth = label\n    gtclass = int(gtclass)\n    gt_bbox = [gtx, gty, gtw, gth]\n    \n    iou_scores_dict = {}\n\n    for dims in dims_list:\n      key = '{}x{}'.format(*dims)\n    \n      kx,ky = dims\n      gapx = 1.0 / kx\n      gapy = 1.0 / ky\n      '''\n      There are kx x ky tiles. \n      For now, all have the same w,h of gapx,gapy. \n      For the (i,j)-th tile, x = 0.5*gapx + i*gapx = (0.5+i)*gapx | y = (0.5+j)*gapy\n      '''\n      for i in range(kx):\n        for j in range(ky):\n          for k in range( len(aspect_ratios) ):\n            dims_aspect_key = (*dims, k) # a 3-tuple: (dim1,dim2,ar)\n            if dims_aspect_key not in iou_scores_dict:\n              iou_scores_dict[dims_aspect_key] = []\n            x = (0.5+i)*gapx\n            y = (0.5+j)*gapy\n\n            # Different aspect ratios alter the anchor box default dimensions\n            w = gapx * aspect_ratios[k][0]\n            h = gapy * aspect_ratios[k][1]\n            cand_bbox = [x,y,w,h]\n\n            # SSD formulation\n            dx = (gtx - x) / w \n            dy = (gty - y) / h\n            dw = log( gtw / w )\n            dh = log( gth / h )\n            \n            int_over_union = iou_fn( cand_bbox, gt_bbox )\n            iou_scores_dict[dims_aspect_key].append( (int_over_union, key, i, j, k, dx, dy, dw, dh) )\n      sampling_fn( iou_scores_dict, np_labels, gtclass, cat_list )\n  return np_labels\n\ndef decode_tensor(pred_dict, aspect_ratios):\n  results = []\n  for dim_str, pred_tensor in pred_dict.items():\n    pred_tensor = pred_tensor[0] # remove the batch\n    kx, ky = [int(g) for g in dim_str.split('x')]\n    gapx = 1. / kx\n    gapy = 1. / ky\n\n    # We trained without activations, so we need to process the logits into probabilities/scores\n    pred_arr = np.array(pred_tensor)\n    obj_logits = pred_arr[:,:,:,0]\n    obj_scores = 1. / (1 + np.exp(-obj_logits))\n    pred_arr[:,:,:,0] = obj_scores\n    \n    cls_logits = pred_arr[:,:,:,1:-4]\n    cls_scores = np.exp(cls_logits)\n    cls_scores = cls_scores / cls_scores.sum(axis=-1)[...,np.newaxis]\n    pred_arr[:,:,:,1:-4] = cls_scores\n\n    for k, ar in enumerate(aspect_ratios):\n      for i in range(kx):\n        for j in range(ky):\n          cx = (0.5+i)*gapx\n          cy = (0.5+j)*gapy\n          w = gapx * ar[0]\n          h = gapy * ar[1]\n\n          payload = pred_arr[i,j,k]\n          obj_score = payload[0]\n          dx, dy, dw, dh = payload[-4:]\n          cls_probs = payload[1:-4]\n\n          predx = (dx * w) + cx\n          predy = (dy * h) + cy\n          predw = w * exp( dw )\n          predh = h * exp( dh )\n          max_cls_idx = np.argmax( cls_probs )\n          max_cls_prob = cls_probs[max_cls_idx]\n          category_id = max_cls_idx + 1\n          det_score = obj_score * max_cls_prob\n          results.append( (det_score, category_id, predx, predy, predw, predh) )\n  return results","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def transfer_model(backbone_model, input_shape, dims_list, num_aspect_ratios, wt_decay, model_name='transfer-objdet-model'):\n  inputs = keras.Input(shape=input_shape)\n  backbone_output = backbone_model(inputs) #7\n\n  x = layers.Conv2D(512, 1, padding='same', kernel_regularizer=l2(wt_decay))(backbone_output) #7\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n  x = layers.Conv2D(512, 3, padding='valid', kernel_regularizer=l2(wt_decay))(x) #5\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n  x = layers.Conv2D(256, 1, padding='same', kernel_regularizer=l2(wt_decay))(x) #5\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n  x = layers.Conv2D(512, 3, padding='valid', kernel_regularizer=l2(wt_decay))(x) #3\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n\n  x = layers.Conv2D(256, 1, padding='same', kernel_regularizer=l2(wt_decay))(x) #3\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n  x = layers.Conv2D(512, 3, padding='same', kernel_regularizer=l2(wt_decay))(x) #3\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n\n  # You can accumulate more scales via shortcut. Imagine each (n,m) is a grid super-imposed on the original image.\n  # See the next cell for an example for more scales.\n  dim_tensor_map = {'3x3': x}\n\n  # For each dimension, construct a predictions tensor. Accumulate them into a dictionary for keras to understand multiple labels.\n  preds_dict = {}\n  for dims in dims_list:\n    dimkey = '{}x{}'.format(*dims)\n    tens = dim_tensor_map[dimkey]\n    ar_preds = []\n    for _ in range(num_aspect_ratios):\n      objectness_preds = layers.Conv2D(1, 1, kernel_regularizer=l2(wt_decay))( tens )\n      class_preds = layers.Conv2D(len(cat_list), 1, kernel_regularizer=l2(wt_decay))( tens )\n      bbox_preds = layers.Conv2D(4, 1, kernel_regularizer=l2(wt_decay))( tens )\n      ar_preds.append( layers.Concatenate()([objectness_preds, class_preds, bbox_preds]) )\n\n    if num_aspect_ratios > 1:\n      predictions = layers.Concatenate()(ar_preds)\n    elif num_aspect_ratios == 1:\n      predictions = ar_preds[0]\n    \n    predictions = layers.Reshape( (*dims, num_aspect_ratios, 5+len(cat_list)), name=dimkey )(predictions)\n    preds_dict[dimkey] = predictions\n\n  model = keras.Model(inputs, preds_dict, name=model_name)\n\n  model.compile( optimizer=tf.keras.optimizers.Adam(1e-5),\n                 loss=custom_loss )\n  return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def transfer_model_7x7_14x14(backbone_model, input_shape, dims_list, num_aspect_ratios, wt_decay, model_name='transfer-objdet-model-7x7-14x14'):\n  inputs = keras.Input(shape=input_shape)\n  intermediate_layer_model = keras.Model(inputs=backbone_model.input,\n                                         outputs=backbone_model.get_layer('conv4_block6_out').output)\n  intermediate_output = intermediate_layer_model(inputs) #14\n  backbone_output = backbone_model(inputs) #7\n\n  x = layers.Conv2D(512, 1, padding='same', kernel_regularizer=l2(wt_decay))(backbone_output) #7\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n  x = layers.Conv2D(1024, 3, padding='same', kernel_regularizer=l2(wt_decay))(x) #7\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n  x = layers.Conv2D(512, 1, padding='same', kernel_regularizer=l2(wt_decay))(x) #7\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n  x = layers.Conv2D(1024, 3, padding='same', kernel_regularizer=l2(wt_decay))(x) #7\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n  x = layers.Conv2D(512, 1, padding='same', kernel_regularizer=l2(wt_decay))(x) #7\n  x = layers.BatchNormalization()(x)\n  upsample = layers.LeakyReLU(0.01)(x)\n  x = layers.Conv2D(2048, 3, padding='same', kernel_regularizer=l2(wt_decay))(upsample) #7\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n  tens_7x7 = layers.Add()([x,backbone_output])\n\n  x = layers.Conv2D(256, 1, padding='same', kernel_regularizer=l2(wt_decay))(upsample) #7\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n  x = layers.Conv2DTranspose(512, 5, strides=(2, 2), padding='same')(x) #14\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n\n  x = layers.Concatenate()([x,intermediate_output])\n\n  x = layers.Conv2D(256, 1, padding='same', kernel_regularizer=l2(wt_decay))(x) #14\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n  x = layers.Conv2D(512, 3, padding='same', kernel_regularizer=l2(wt_decay))(x) #14\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n  x = layers.Conv2D(256, 1, padding='same', kernel_regularizer=l2(wt_decay))(x) #14\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n  x = layers.Conv2D(512, 3, padding='same', kernel_regularizer=l2(wt_decay))(x) #14\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n  x = layers.Conv2D(256, 1, padding='same', kernel_regularizer=l2(wt_decay))(x) #14\n  x = layers.BatchNormalization()(x)\n  x = layers.LeakyReLU(0.01)(x)\n  x = layers.Conv2D(512, 3, padding='same', kernel_regularizer=l2(wt_decay))(x) #14\n  x = layers.BatchNormalization()(x)\n  tens_14x14 = layers.LeakyReLU(0.01)(x)\n\n  dim_tensor_map = {'7x7': tens_7x7, '14x14': tens_14x14}\n\n  # For each dimension, construct a predictions tensor. Accumulate them into a dictionary for keras to understand multiple labels.\n  preds_dict = {}\n  for dims in dims_list:\n    dimkey = '{}x{}'.format(*dims)\n    tens = dim_tensor_map[dimkey]\n    ar_preds = []\n    for _ in range(num_aspect_ratios):\n      objectness_preds = layers.Conv2D(1, 1, kernel_regularizer=l2(wt_decay))( tens )\n      class_preds = layers.Conv2D(len(cat_list), 1, kernel_regularizer=l2(wt_decay))( tens )\n      bbox_preds = layers.Conv2D(4, 1, kernel_regularizer=l2(wt_decay))( tens )\n      ar_preds.append( layers.Concatenate()([objectness_preds, class_preds, bbox_preds]) )\n\n    if num_aspect_ratios > 1:\n      predictions = layers.Concatenate()(ar_preds)\n    elif num_aspect_ratios == 1:\n      predictions = ar_preds[0]\n    \n    predictions = layers.Reshape( (*dims, num_aspect_ratios, 5+len(cat_list)), name=dimkey )(predictions)\n    preds_dict[dimkey] = predictions\n\n  model = keras.Model(inputs, preds_dict, name=model_name)\n\n  model.compile( optimizer=tf.keras.optimizers.Adam(1e-5),\n                 loss=custom_loss )\n  return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class TILPickle(Sequence):\n  def __init__(self, pickle_file, batch_size, augment_fn, input_size, label_encoder, preprocess_fn, testmode=False):\n\n#    with open(pickle_file, 'rb') as p:\n#      self.ids, self.x, self.y = pickle.load(p)\n    self.ids, self.x, self.y = pickle.load(open(pickle_file, 'rb'))\n    self.batch_size = batch_size\n    self.augment_fn = augment_fn\n    self.input_wh = (*input_size[:2][::-1],input_size[2])\n    self.label_encoder = label_encoder\n    self.preprocess_fn = preprocess_fn\n    self.testmode = testmode\n    \n  def __len__(self):\n    return int(np.ceil(len(self.x) / float(self.batch_size)))\n  \n  def __getitem__(self, idx):\n    batch_x = self.x[idx * self.batch_size:(idx + 1) * self.batch_size]\n    batch_y = self.y[idx * self.batch_size:(idx + 1) * self.batch_size]\n    batch_ids = self.ids[idx * self.batch_size:(idx + 1) * self.batch_size]\n\n    x_acc, y_acc = [], {}\n    \n    for x,y in zip( batch_x, batch_y ):\n      x_aug, y_aug = self.augment_fn( x, y )\n      if x_aug.size != self.input_wh[:2]:\n        x_aug.resize( self.input_wh )\n      x_acc.append( np.array(x_aug) )\n      y_dict = self.label_encoder( y_aug )\n      for dimkey, label in y_dict.items():\n        if dimkey not in y_acc:\n          y_acc[dimkey] = []\n        y_acc[dimkey].append( label )\n\n    if self.testmode:\n      return batch_ids, self.preprocess_fn( np.array( x_acc ) ), { dimkey: np.array( gt_tensor ) for dimkey, gt_tensor in y_acc.items() }\n    return self.preprocess_fn( np.array( x_acc ) ), { dimkey: np.array( gt_tensor ) for dimkey, gt_tensor in y_acc.items() }\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class TILSequence(Sequence):\n  def __init__(self, img_folder, json_annotation_file, batch_size, augment_fn, input_size, label_encoder, preprocess_fn, testmode=False):\n\n    self._prepare_data(img_folder, json_annotation_file)\n    self.batch_size = batch_size\n    self.augment_fn = augment_fn\n    self.input_wh = (*input_size[:2][::-1],input_size[2])\n    self.label_encoder = label_encoder\n    self.preprocess_fn = preprocess_fn\n    self.testmode = testmode\n    \n  def _prepare_data(self, img_folder, json_annotation_file):\n    imgs_dict = {im.split('.')[0]:im for im in os.listdir(img_folder) if im.endswith('.jpg')}\n    data_dict = {}\n    with open(json_annotation_file, 'r') as f:\n      annotations_dict = json.load(f)\n    annotations_list = annotations_dict['annotations']\n    for annotation in annotations_list:\n      img_id = str(annotation['image_id'])\n      c = annotation['category_id'] - 1 # TODO: make sure that category ids start from 1, not 0\n      boxleft,boxtop,boxwidth,boxheight = annotation['bbox']\n      if img_id in imgs_dict:\n        img_fp = os.path.join(img_folder, imgs_dict[img_id])\n        imwidth,imheight = PIL.Image.open(img_fp).size\n        if img_id not in data_dict:\n          data_dict[img_id] = []\n        box_cenx = boxleft + boxwidth/2.\n        box_ceny = boxtop + boxheight/2.\n        x,y,w,h = box_cenx/imwidth, box_ceny/imheight, boxwidth/imwidth, boxheight/imheight\n\n        data_dict[img_id].append( [c,x,y,w,h] )\n    self.x, self.y, self.ids = [], [], []\n    for img_id, labels in data_dict.items():\n      self.x.append( os.path.join(img_folder, imgs_dict[img_id]) )\n      self.y.append( np.array(labels) )\n      self.ids.append( img_id )\n\n  def __len__(self):\n    return int(np.ceil(len(self.x) / float(self.batch_size)))\n  \n  def __getitem__(self, idx):\n    batch_x = self.x[idx * self.batch_size:(idx + 1) * self.batch_size]\n    batch_y = self.y[idx * self.batch_size:(idx + 1) * self.batch_size]\n\n    x_acc, y_acc = [], {}\n    original_img_dims = []\n    with Pool(self.batch_size) as p:\n      # Read in the PIL objects from filepaths\n      batch_x = p.map(load_img, batch_x)\n    \n    for x,y in zip( batch_x, batch_y ):\n      W,H = x.size\n      original_img_dims.append( (W,H) )\n\n      x = x.resize( self.input_wh[:2] )\n      x_aug, y_aug = self.augment_fn( x, y )\n      x_acc.append( np.array(x_aug) )\n      y_dict = self.label_encoder( y_aug )\n      for dimkey, label in y_dict.items():\n        if dimkey not in y_acc:\n          y_acc[dimkey] = []\n        y_acc[dimkey].append( label )\n\n    return self.get_batch_test(idx, x_acc, y_acc, original_img_dims) if self.testmode else self.get_batch(x_acc, y_acc)\n\n  def get_batch_test(self, idx, x_acc, y_acc, original_img_dims):\n    batch_ids = self.ids[idx * self.batch_size:(idx + 1) * self.batch_size]\n    return batch_ids, original_img_dims, self.preprocess_fn( np.array( x_acc ) ), { dimkey: np.array( gt_tensor ) for dimkey, gt_tensor in y_acc.items() }\n\n  def get_batch(self, x_acc, y_acc):\n    return self.preprocess_fn( np.array( x_acc ) ), { dimkey: np.array( gt_tensor ) for dimkey, gt_tensor in y_acc.items() }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Choose whether to start a new model or load a previously trained one\nmodel_context = 'model-7x7-14x14-3aspect-modyoloposneg-wd{}'.format(wt_decay)\n# load_model_path = os.path.join( load_model_folder, '{}-best_val_loss.h5'.format(model_context) )\nload_model_path = None\n\nif load_model_path is None:\n  backbone_model = tf.keras.applications.ResNet50(input_shape=input_shape, include_top=False)\n  model = transfer_model_7x7_14x14(backbone_model, input_shape=input_shape, dims_list=dims_list, num_aspect_ratios=len(aspect_ratios), wt_decay=wt_decay, model_name=model_context+'-res50')\nelse:\n  model = tf.keras.models.load_model(load_model_path, custom_objects={'custom_loss':custom_loss})\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''\n- There is overfitting now that I set top 25% (of each dim-ar combination) as positives. How?\n- Larger image size - maybe 448\n- Transfer learning\n- Change weights of losses?\n\n# Also add more callbacks, such as tensorboard \ndataset, batch_size, augment_fn, input_size, label_encoder, preprocess_fn\nencode_label(label_arr, dims_list, aspect_ratios, iou_fn, sampling_fn, cat_list)\nimg_folder, json_annotation_file, batch_size, augment_fn, input_size, label_encoder, preprocess_fn\n'''\nbs=16\nn_epochs_warmup = 300\nn_epochs_after = 300\n\nlabel_encoder = lambda y: encode_label(y, dims_list, aspect_ratios, iou, modified_yolo_posneg_sampling, cat_list)\npreproc_fn = lambda x: x / 255.\n\nprint('Creating training sequence...')\n# train_sequence = TILSequence(train_imgs_folder, train_annotations, bs, aug_default, input_shape, label_encoder, preproc_fn)\ntrain_sequence = TILPickle(train_pickle, bs, aug_default, input_shape, label_encoder, preproc_fn)\nprint('Creating validation sequence...')\n# val_sequence = TILSequence(val_imgs_folder, val_annotations, bs, aug_identity, input_shape, label_encoder, preproc_fn)\nval_sequence = TILPickle(val_pickle, bs, aug_identity, input_shape, label_encoder, preproc_fn)\n\nsave_model_path = os.path.join( save_model_folder, '{}-best_val_loss.h5'.format(model_context) )\nmodel_checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(\n                                                                filepath=save_model_path,\n                                                                save_weights_only=False,\n                                                                monitor='val_loss',\n                                                                mode='auto',\n                                                                save_best_only=True)\nearlystopping = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=30)\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=5, min_lr=1e-8)\n\nfor layer in backbone_model.layers:\n  layer.trainable = False\n\nprint('Warming up the model...')\nmodel.fit(x=train_sequence, \n          epochs=n_epochs_warmup, \n          validation_data=val_sequence, \n          callbacks=[model_checkpoint_callback, earlystopping, reduce_lr])\n\n# Fine tuning\nprint('Model warmed. Loading best val version of model...')\nload_model_path = os.path.join( load_model_folder, '{}-best_val_loss.h5'.format(model_context) )\ndel model\nmodel = tf.keras.models.load_model(load_model_path, custom_objects={'custom_loss':custom_loss})\n\nfor layer in model.get_layer('resnet50').layers:\n  layer.trainable = True\n\nmodel.compile(optimizer=tf.keras.optimizers.Adam(1e-5), loss=custom_loss)\nmodel_context = 'ft-' + model_context\nsave_model_path = os.path.join( save_model_folder, '{}-best_val_loss.h5'.format(model_context) )\nmodel_checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(\n                                                                filepath=save_model_path,\n                                                                save_weights_only=False,\n                                                                monitor='val_loss',\n                                                                mode='auto',\n                                                                save_best_only=True)\nearlystopping = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=30)\nreduce_lr = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=5, min_lr=1e-8)\n\nmodel.fit(x=train_sequence, \n          epochs=n_epochs_after, \n          validation_data=val_sequence, \n          callbacks=[model_checkpoint_callback, earlystopping, reduce_lr])\n\n# Final save\nmodel.save(os.path.join(save_model_folder, '{}-final.h5'.format(model_context)))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# To fix multiple, we introduce non-maximum suppression, or NMS for short\ndef nms(detections, iou_thresh=0.):\n  dets_by_class = {}\n  final_result = []\n  for det in detections:\n    cls = det[1]\n    if cls not in dets_by_class:\n      dets_by_class[cls] = []\n    dets_by_class[cls].append( det )\n  for _, dets in dets_by_class.items():\n    candidates = list(dets)\n    candidates.sort( key=lambda x:x[0], reverse=True )\n    while len(candidates) > 0:\n      candidate = candidates.pop(0)\n      _,_,cx,cy,cw,ch = candidate\n      copy = list(candidates)\n      for other in candidates:\n        # Compute the IoU. If it exceeds thresh, we remove it\n        _,_,ox,oy,ow,oh = other\n        if iou( (cx,cy,cw,ch), (ox,oy,ow,oh) ) > iou_thresh:\n          copy.remove(other)\n      candidates = list(copy)\n      final_result.append(candidate)\n  return final_result","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# load the model\nload_model_path = os.path.join( load_model_folder, 'model-7x7-14x14-3aspect-modyoloposneg-wd0.0005-best_val_loss.h5' )\nmodel = tf.keras.models.load_model(load_model_path, custom_objects={'custom_loss':custom_loss})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# load the test data\nlabel_encoder = lambda y: encode_label(y, dims_list, aspect_ratios, iou, modified_yolo_posneg_sampling, cat_list)\npreproc_fn = lambda x: x / 255.\n\ntest_sequence_pickle = TILPickle(val_pickle, 1, aug_identity, input_shape, label_encoder, preproc_fn, testmode=True)\ntest_sequence = TILSequence(val_imgs_folder, val_annotations, 1, aug_identity, input_shape, label_encoder, preproc_fn, testmode=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Test to make sure that both dispensers dispense the same data\nimg_idx = 42\nids_pickle, x_pickle, y_pickle = test_sequence_pickle[img_idx]\nids_seq, dims_seq, x_seq, y_seq = test_sequence[img_idx]\n\nprint('Image ids of pickle and seq:', ids_pickle[0], ',', ids_seq[0])\nprint('Are input arrays same?:', np.allclose( x_pickle, x_seq ))\nfor dimkey, ylabel_pickle in y_pickle.items():\n  ylabel_seq = y_seq[dimkey]\n  print('Are labels same for key=(', dimkey, ')?:', np.allclose( ylabel_pickle, ylabel_seq ))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Run this to visualize\nrank_colors = ['cyan', 'magenta', 'pink']\ndet_threshold=0.\ntop_dets=3\n\nstart=0\nend=20\nfor k in range(start,end):\n  _, img_arr, label_cxywh = test_sequence_pickle[k]\n  img_arr = img_arr[0]\n  pil_img = PIL.Image.fromarray( (img_arr * 255.).astype(np.uint8) )\n  W,H = pil_img.size\n  pred_dict = model(np.array([img_arr]))\n  preds = decode_tensor( pred_dict, aspect_ratios )\n    \n  # Post-processing\n  preds.sort( key=lambda x:x[0], reverse=True )\n  preds = [pred for pred in preds if pred[0] >= det_threshold]\n  preds = preds[:top_dets]\n  preds = nms(preds, iou_thresh=0.5)\n\n  draw_img = pil_img.copy()\n  draw = ImageDraw.Draw(draw_img)\n  for i, pred in enumerate(preds):\n    conf,cls,x,y,w,h = pred\n    bb_x = int(x * W)\n    bb_y = int(y * H)\n    bb_w = int(w * W)\n    bb_h = int(h * H)\n    left = int(bb_x - bb_w / 2)\n    top = int(bb_y - bb_h / 2)\n    right = int(bb_x + bb_w / 2)\n    bot = int(bb_y + bb_h / 2)\n    cls_str = cat_list[cls-1]\n\n    draw.rectangle(((left, top), (right, bot)), outline=rank_colors[i])\n    draw.text((bb_x, bb_y), cls_str, fill=rank_colors[i])\n    draw.text( ( int(left + bb_w*.1), int(top + bb_h*.1) ), '{:.2f}'.format(conf), fill=rank_colors[i] )\n\n  display(draw_img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}