{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":13032,"databundleVersionId":408833,"sourceType":"competition"}],"dockerImageVersionId":29507,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"The environment setup is copied from [Pednoi](https://www.kaggle.com/pednoi), and roughly follows the setup prescribed [here.](https://github.com/matterport/Mask_RCNN/blob/master/samples/shapes/train_shapes.ipynb>) What follows after is a thorough analysis of Mask R-CNN trained on the iMaterialist dataset.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom pathlib import Path\nimport os\nimport sys\n# !pip install tensorflow==2.0.0-beta1\nimport tensorflow as tf\nimport json\nprint(sys.version)\n!pip freeze\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.python.client import device_lib\n\ndef get_available_gpus():\n    local_device_protos = device_lib.list_local_devices()\n    return [x.name for x in local_device_protos]\nget_available_gpus()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"INPUT_DIR = Path('/kaggle/input')\nDATA_DIR = INPUT_DIR/\"imaterialist-fashion-2019-FGVC6\"\nROOT_DIR = Path('/kaggle/working')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Actually, unfamiliar with how Kaggle handles their filesystem. Let's check.","metadata":{}},{"cell_type":"code","source":"print(os.listdir(\"/kaggle/\"))\nprint(os.listdir(\"/kaggle/config\"))\nprint(os.listdir(INPUT_DIR))\nprint(os.listdir(ROOT_DIR))\nprint(os.listdir(DATA_DIR))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Get the Matterport implementation of Mask R-CNN from github","metadata":{}},{"cell_type":"code","source":"!git clone https://www.github.com/matterport/Mask_RCNN.git\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Current Directory: {}\".format(os.curdir))\nprint(\"Current Directory Contents: {}\".format(os.listdir()))\nprint(\"Changing dir...\")\nos.chdir('Mask_RCNN')\nprint(\"Current Directory: {}\".format(os.curdir))\nprint(\"Current Directory Contents: {}\".format(os.listdir()))\n\n!rm -rf .git # to prevent an error when the kernel is committed\n!rm -rf images assets # to prevent displaying images at the bottom of a kernel","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"### Matterport Imports ###\nimport os\nimport sys\nimport random\nimport math\nimport re\nimport time\nimport numpy as np\nimport cv2\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport PIL\nfrom PIL import Image\n\nfrom imgaug import augmenters as iaa\nfrom sklearn.model_selection import StratifiedKFold, KFold\n\n# Import Mask RCNN\nsys.path.append(ROOT_DIR/'Mask_RCNN')\nfrom mrcnn.config import Config\nfrom mrcnn import utils\nimport mrcnn.model as modellib\nfrom mrcnn import visualize\nfrom mrcnn.model import log\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!wget --quiet https://github.com/matterport/Mask_RCNN/releases/download/v2.0/mask_rcnn_coco.h5\nprint(os.listdir())\n!ls -lh mask_rcnn_coco.h5\n\n\nCOCO_WEIGHTS_PATH = 'mask_rcnn_coco.h5'","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### The Matterport library has two primary structures to work with: Config & Dataset","metadata":{}},{"cell_type":"markdown","source":"Config first...","metadata":{}},{"cell_type":"code","source":"class iMaterialistConfig(Config):\n    \"\"\"Configuration for training on the iMaterialist Fashion 2019 dataset.\n    Derives from the base Config class and overrides values specific to the \n    iMaterialist dataset.\"\"\"\n    \n    # Give config a name\n    NAME = \"fashion\"\n    \n    # Train with 1 img/GPU, as the images are large, and I haven't done the\n    # calculation to determine if we can fit the largest image in memory. \n    # Trial by fire!\n    GPU_COUNT = 1\n    IMAGES_PER_GPU = 1\n    \n    # At first, we will ignore attributes to simplify the classification\n    NUM_CLASSES = 46 + 1 # background + 27 main articles + 19 apparel parts\n    \n    # We will try to train on the images as is. This may fail, as the\n    # largest image in the set is (6824, 10717). Yet, we push on.\n    IMAGE_MIN_DIM = 151\n    IMAGE_MAX_DIM = 16384\n    \n    # The objects in the images were analyzed in my other notebook on this\n    # challenge. Good sizes to start seemed to be 128, 384, and 896.\n    RPN_ANCHOR_SCALES = (128, 384, 896)\n    \n    # Apparently the Mask RCNN paper prescribes 512, but this is alot, especially\n    # for this dataset. We trim this down to a more reasonable number. \n    TRAIN_ROIS_PER_IMAGE = 45\n    \n    # Use a small step size at first (don't need validation updates very often)\n    # Ideally this uses all of the instances in the training set. So this should \n    # equal (len(training_dataset)/(GPU_COUNT*IMAGES_PER_GPU)). Validation steps\n    # should similarly be the same as this number.\n    STEPS_PER_EPOCH = 1000\n    \n    # Validation steps to run after each epoch\n    VALIDATION_STEPS = 100\n    \n    BACKBONE = \"resnet50\"\n    \nconfig = iMaterialistConfig()\nconfig.display()\n    ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Dataset second...  \n  \nWe need to override the following methods:\n* load_image\n* load_mask\n* image_reference","metadata":{}},{"cell_type":"code","source":"import json\nfrom pandas.io.json import json_normalize\n\ntrain_df = pd.read_csv(DATA_DIR/\"train.csv\")\n\nprint()\nprint(\"Training df:\")\nprint(train_df.head())\nprint()\n\n# Reading the json as a dict\nwith open(DATA_DIR/\"label_descriptions.json\") as json_data:\n    data = json.load(json_data)\n\nJSON_COLUMNS = list(data.keys())\ncategories_df = json_normalize(data['categories'])\n# To be able to match this data with img_df data, rename 'id' column to 'ClassId'\ncategories_df = categories_df.rename(columns={\"id\": \"ClassId\"})\nattributes_df = json_normalize(data['attributes'])\ninfo_df = json_normalize(data['info'])\n# Make a list matching up category string with id (as list index)\ncategory_names = list(categories_df[\"name\"])\n\nprint()\nprint(\"Columns in info JSON:\")\nprint(JSON_COLUMNS)\nprint()\nprint(\"Categories df:\")\nprint(categories_df.head())\nprint()\nprint(\"Attributes df:\")\nprint(attributes_df.head())\nprint()\nprint(\"Info df:\")\nprint(info_df.head())\nprint()\nprint(\"Category Names, length:\")\nprint(category_names)\nprint(len(category_names))\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Let's remove the attributes from the ClassId column so \n# that the dataset is simpler to deal with at first\ntrain_df[\"ClassId\"] = train_df[\"ClassId\"].apply(lambda x: x.split(\"_\")[0])\ntrain_df_img_groups = train_df.groupby(\"ImageId\")[\"EncodedPixels\", \"ClassId\"].agg(lambda x: list(x))\nsize_df = train_df.groupby(\"ImageId\")[\"Height\", \"Width\"].mean()\ntrain_df_img_groups = train_df_img_groups.join(size_df, on=\"ImageId\")\ntrain_df_img_groups = train_df_img_groups.rename(columns={\"ClassId\": \"CategoryIds\"})\n\nprint(\"Train df grouped by image:\")\nprint(type(train_df_img_groups))\nprint(train_df_img_groups.head())\nprint()\nprint(\"First group of train df grouped by image:\")\ndf_iterable = list(zip(*train_df_img_groups.iterrows()))\nimg_ids, row_list = df_iterable[0], df_iterable[1]\nprint(row_list[0])\nprint(train_df_img_groups.sort_values())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class iMaterialistDataset(utils.Dataset):\n    \"\"\"Generates the iMaterialist Fashion 2019 dataset\"\"\"\n    \n    def __init__(self, df, category_names):\n        super().__init__(self)\n        self.category_names = category_names\n        \n        ##########################################################\n        # add_class() fills in self.class_info, a list of dicts,\n        # so we can keep track of labels in training and test\n        #\n        # The entries in self.class_info list take the form:\n        # {\n        #     \"source\": source,\n        #     \"id\": class_id,\n        #     \"name\": class_name,\n        # }\n        #\n        for i, category_name in enumerate(category_names):\n            # i+1 because background is id = 0\n            self.add_class(source = \"fashion\", \n                           class_id = i+1, \n                           class_name = category_name\n                          )\n        \n        ##########################################################\n        # add_image() fills in self.image_info, a list of \n        # dicts, so we can generate appropriate masks\n        #\n        # The entries in self.image_info template looks like this:\n        # {\n        #     \"id\": image_id,\n        #     \"source\": source,\n        #     \"path\": path,\n        # }\n        #\n        # You can add custom fields to keep track of things like\n        # mask pixel encodings, etc.\n        #\n        for img_id, row in df.iterrows():\n            self.add_image(source = \"fashion\",\n                           image_id = row.name,\n                           path = DATA_DIR/(\"train/\"+row.name),\n                           labels = row[\"CategoryIds\"],\n                           width = row[\"Width\"],\n                           height = row[\"Height\"],\n                           annotations = row[\"EncodedPixels\"]\n                          )\n        \n    def load_image(self, img_num):\n        \"\"\"Load the specified image and return a [H,W,3] Numpy array.\n        \"\"\"\n        # Given image_name (<string>.jpg), load the image into array\n        info = self.image_info[img_num]\n        img = np.asarray(Image.open(info[\"path\"]))\n        return img\n    \n    def RLE_to_submask(self, mask, encoded_pixels):\n        \"\"\"Fills in 1's in mask array for segmented object that is\n        encoded with values in encoded pixels.\n        \"\"\"\n        mask_height = mask.shape[0]\n        start_pixels = encoded_pixels[::2]\n        lengths = encoded_pixels[1:][::2]\n        for start_pixel, length in zip(start_pixels, lengths):\n            start_pixel, length = int(start_pixel), int(length)\n            # Pixels are numbered in y (not left to right in x)\n            y_start = (start_pixel-1) % mask_height\n            x_start = int((start_pixel - 1) / mask_height)\n            mask[y_start: y_start + length, x_start] = 1\n        return mask\n\n    def load_mask(self, img_num):\n        \"\"\"Load instance masks for the given image.\n        Different datasets use different ways to store masks. Override this\n        method to load instance masks and return them in the form of an\n        array of binary masks of shape [height, width, instances].\n        \n        Returns:\n            masks: A bool array of shape [height, width, instance count] with\n                a binary mask per instance.\n            class_ids: a 1D array of class IDs of the instance masks.\n        \"\"\"\n        info = self.image_info[img_num]\n        masks = np.full((info[\"height\"], info[\"width\"], len(info[\"labels\"])), fill_value=0, dtype=np.uint8)\n        category_ids = []\n        for instance_num, category_annotation in enumerate(zip(info[\"labels\"], info[\"annotations\"])):\n            category_id, encoded_pixels = category_annotation\n            encoded_pixels = encoded_pixels.split(' ')\n            mask = masks[:, :, instance_num]\n            mask = self.RLE_to_submask(mask, encoded_pixels)\n            masks[:, :, instance_num] = mask\n            category_ids.append(int(category_id)+1) # Offset by 1 for BG at index 0.\n        category_ids = np.array(category_ids, dtype=np.int32)\n        return masks,category_ids\n\n    def image_reference(self, img_num):\n        info = self.image_info[img_num]\n        if info[\"source\"] == \"fashion\":\n            # Match the category_id (x for x in info.labels) with the string in\n            # the category names list.\n            return info[\"path\"], [self.category_names[int(x)] for x in info[\"labels\"]]\n        else:\n            super(self.__class__).image_reference(self, image_id)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset = iMaterialistDataset(train_df_img_groups, category_names)\ndataset.prepare()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def sample_dataset(dataset):\n    for i in range(6):\n        image_id = random.choice(dataset.image_ids)\n        print(\"ImageId: {}\\n\".format(image_id))\n        print(dataset.image_reference(image_id))\n\n        image = dataset.load_image(image_id)\n        mask, class_ids = dataset.load_mask(image_id)\n        visualize.display_top_masks(image, mask, class_ids, dataset.class_names, limit=4)\n        \nsample_dataset(dataset)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Let's train this muthasucka!\n![alt text](https://thenypost.files.wordpress.com/2014/10/2220943-e1413627640425.jpg \"George Clinton of Parliament! Check out that fashion though.\")","metadata":{}},{"cell_type":"code","source":"# This code partially supports k-fold training, \n# you can specify the fold to train and the total number of folds here\nFOLD = 0\nN_FOLDS = 5\n\nkf = KFold(n_splits=N_FOLDS, random_state=42, shuffle=True)\nsplits = kf.split(train_df_img_groups) # ideally, this should be multilabel stratification\n\ndef get_fold():    \n    for i, (train_index, valid_index) in enumerate(splits):\n        if i == FOLD:\n            return train_df_img_groups.iloc[train_index], train_df_img_groups.iloc[valid_index]\n        \ntrain_fold, valid_fold = get_fold()\n\ntrain_dataset = iMaterialistDataset(train_fold, category_names)\ntrain_dataset.prepare()\n\nvalid_dataset = iMaterialistDataset(valid_fold, category_names)\nvalid_dataset.prepare()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initial go. These are the parameters that Pednoi uses.\nLR = 2e-3\nEPOCHS = [2, 6, 8]\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# EDIT: This won't work with current config\n\n# model = modellib.MaskRCNN(mode='training', config=config, model_dir=ROOT_DIR)\n\n# model.load_weights(COCO_WEIGHTS_PATH, \n#                    by_name=True, \n#                    exclude=['mrcnn_class_logits', \n#                             'mrcnn_bbox_fc', \n#                             'mrcnn_bbox', \n#                             'mrcnn_mask'])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"First we train the heads, which is a type of transfer learning known as <u>Feature Extraction</u>. This trains the classification pieces of the network and freezes any and all convolutional layers. The idea is that the network uses features already learned by the backbone network (in this case ResNet50), which are likely useful to learn features about various articles of clothing/fashion.","metadata":{}},{"cell_type":"markdown","source":"Below doesn't seem to work (kernel dies). Let's try to resize images as Pednoi does.","metadata":{}},{"cell_type":"code","source":"# Train the head branches\n# Passing layers=\"heads\" freezes all layers except the head\n# layers. You can also pass a regular expression to select\n# which layers to train by name pattern.\n\n\n# model.train(train_dataset, valid_dataset, \n#             learning_rate=LR, \n#             epochs=EPOCHS[0], \n#             layers='heads')\n\n# history = model.keras_model.history.history","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We will implement a resize for images that <u>need</u> it. Let's do Pednoi's resize to a max image dimension of 512 for now. That means we will rescale images that have max dimension > 512 so that their max dimension is 512. Lose some detail here, probably OK.  \n  \nThis will include edits to the following objects/methods:\n* iMaterialistConfig() - IMAGE_MAX_DIM, RPN_ANCHOR_SCALES, IMAGES_PER_GPU\n* iMaterialistDataset() - load_image(), load_mask()","metadata":{}},{"cell_type":"code","source":"# Tiny image resize experiment to determine which is most appropriate\n# Seek out the largest images in the data set and resize to where the max dimension is 512\ninterp_methods = [PIL.Image.NEAREST, \n                  PIL.Image.BILINEAR, \n                  PIL.Image.BICUBIC, \n                  PIL.Image.LANCZOS, \n                 ]\nMAX_DIM = 512\n\nmax_width_img_id = train_df_img_groups[\"Width\"].idxmax()\nmax_height_img_id = train_df_img_groups[\"Height\"].idxmax()\nmax_categories_img_id = train_df_img_groups[\"CategoryIds\"].apply(len).idxmax()\nprint(max_width_img_id)\nprint(max_height_img_id)\nprint(max_categories_img_id)\n\n\ndef resize_interp(img_id):\n    img_info = train_df_img_groups.ix[img_id]\n    print(len(img_info[\"CategoryIds\"]))\n    img = Image.open(DATA_DIR/(\"train/\"+img_id))\n\n    fig, ax = plt.subplots(len(interp_methods), 1, figsize=(16, 30))\n    for i, interp_method in enumerate(interp_methods):\n        width, height = img_info[\"Width\"], img_info[\"Height\"]\n        ratio = min(MAX_DIM/width, MAX_DIM/height)\n        img_resize = img.resize((int(ratio*width), int(ratio*height)), \n                                                    resample=interp_method)\n        ax[i].imshow(img_resize)\n\nresize_interp(max_width_img_id)\nresize_interp(max_height_img_id)\nresize_interp(max_categories_img_id)\nresize_interp(\"94f8ed8e4f2ff4662c720a3e160dee09.jpg\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Honestly, Lanczos looks like the best representation. Let's use it!","metadata":{}},{"cell_type":"markdown","source":"Mask scaling is a slightly different beast. In `load_image()`, masks are arrays and not PIL Images. Let's see what zoom can do for us:","metadata":{}},{"cell_type":"code","source":"from scipy.ndimage import zoom\n\nfig, ax = plt.subplots(2, 4, figsize=(16,16))\nimage_id = random.choice(dataset.image_ids)\nprint(\"ImageId: {}\\n\".format(image_id))\nprint(dataset.image_reference(image_id))\nimage = dataset.load_image(image_id)\nmask, class_ids = dataset.load_mask(image_id)\nmask = mask[:, :, 0]\nprint(mask.shape)\nwidth, height = mask.shape\n\n# Resize so the max dimension for each image is MAX_DIM in size\nratio = min(MAX_DIM/width, MAX_DIM/height)\n# resize_shape = (int(ratio*width), int(ratio*height))\n\nax[0, 0].imshow(mask)\nzoom_mask = zoom(mask, ratio, order=1, mode='nearest')\nax[0, 1].imshow(zoom_mask)\nzoom_mask = zoom(mask, ratio, order=3, mode='nearest')\nax[0, 2].imshow(zoom_mask)\nzoom_mask = zoom(mask, ratio, order=5, mode='nearest')\nax[0, 3].imshow(zoom_mask)\n\nax[1, 0].imshow(mask)\nzoom_mask = zoom(mask, ratio, order=3, mode='reflect')\nax[1, 1].imshow(zoom_mask)\nzoom_mask = zoom(mask, ratio, order=3, mode='nearest')\nax[1, 2].imshow(zoom_mask)\nzoom_mask = zoom(mask, ratio, order=3, mode='wrap')\nax[1, 3].imshow(zoom_mask)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There are weird slim margins appearing for masks that have objects touching the border. Looks like there's not much reason to pick hairs here though, so let's go with default settings","metadata":{}},{"cell_type":"code","source":"class iMaterialistMaxDimConfig(iMaterialistConfig):\n    \"\"\"Configuration for training on the iMaterialist Fashion 2019 dataset.\n    Derives from the base Config class and overrides values specific to the \n    iMaterialist dataset. In addition\"\"\"\n    \n    # Give config a name\n    NAME = \"fashion\"\n    \n    # Train with 4 imgs/GPU, as the images are large as Pednoi prescribes.\n    GPU_COUNT = 1\n    IMAGES_PER_GPU = 4\n    \n    # We will try to train on resized image data.\n    IMAGE_MIN_DIM = MAX_DIM\n    IMAGE_MAX_DIM = MAX_DIM\n    \n    # The objects in the images were analyzed in my other notebook on this\n    # challenge. Good sizes to start seemed to be 128, 384, and 896 for \n    # non-resized images. Let's try \n    RPN_ANCHOR_SCALES = (32, 64, 128, 256, 384)\n    \n    MAX_GT_INSTANCES = 100\n    \nconfig = iMaterialistMaxDimConfig()\nconfig.display()\n    ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class iMaterialistMaxDimDataset(iMaterialistDataset):\n    \n    def load_image(self, img_num):\n        \"\"\"Load the specified image and return a [H,W,3] Numpy array.\n        \"\"\"\n        # Given image_name (<string>.jpg), load the image into array\n        info = self.image_info[img_num]\n        width, height = info[\"width\"], info[\"height\"]\n        raw_img = Image.open(info[\"path\"])\n        # Resize so the max dimension for each image is MAX_DIM in size\n        ratio = min(MAX_DIM/info[\"width\"], MAX_DIM/info[\"height\"])\n        if ratio < 1.:\n            resize_shape = (int(round(ratio*width)), int(round(ratio*height)))\n            raw_img = raw_img.resize(resize_shape, \n                                     resample=PIL.Image.LANCZOS)\n        img = np.asarray(raw_img)\n        return img\n\n    def load_mask(self, img_num):\n        \"\"\"Load instance masks for the given image.\n        Different datasets use different ways to store masks. Override this\n        method to load instance masks and return them in the form of an\n        array of binary masks of shape [height, width, instances].\n        \n        Returns:\n            masks: A bool array of shape [height, width, instance count] with\n                a binary mask per instance.\n            class_ids: a 1D array of class IDs of the instance masks.\n        \"\"\"\n        info = self.image_info[img_num]\n        width, height = info[\"width\"], info[\"height\"]\n        ratio = min(MAX_DIM/width, MAX_DIM/height)\n        masks = np.full((height, \n                         width, \n                         len(info[\"labels\"])),\n                        fill_value=0, \n                        dtype=np.uint8)\n        if ratio < 1.:\n            resize_shape = int(round(ratio*width)), int(round(ratio*height))\n            masks_resize = np.full((resize_shape[1], \n                                    resize_shape[0], \n                                    len(info[\"labels\"])), \n                                   fill_value=0, \n                                   dtype=np.uint8)        \n        category_ids = []\n        for instance_num, category_annotation in enumerate(zip(info[\"labels\"], info[\"annotations\"])):\n            category_id, encoded_pixels = category_annotation\n            encoded_pixels = encoded_pixels.split(' ')\n            mask = masks[:, :, instance_num]\n            mask = self.RLE_to_submask(mask, encoded_pixels)\n            if ratio < 1.:\n                #  mask = zoom(mask, ratio, order=1, mode='nearest')\n                mask = zoom(mask, ratio)\n                masks_resize[:, :, instance_num] = mask\n            else:\n                masks[:, :, instance_num] = mask\n            category_ids.append(int(category_id)+1) # Offset by 1 for BG at index 0.\n        if ratio < 1.:\n            masks = masks_resize\n        category_ids = np.array(category_ids, dtype=np.int32)\n        return masks, category_ids\n    ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"max_dim_data = iMaterialistMaxDimDataset(train_df_img_groups, category_names)\nmax_dim_data.prepare()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_dataset(max_dim_data)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Let's train this muthasucka, vol. 2!\n\n![alt text](https://image-ticketfly.imgix.net/00/00/37/45/29-og.jpg \"Bootsy Collins of Parliament! Check out that fashion though.\")","metadata":{}},{"cell_type":"code","source":"# This code partially supports k-fold training, \n# you can specify the fold to train and the total number of folds here\nFOLD = 0\nN_FOLDS = 5\n\nkf = KFold(n_splits=N_FOLDS, random_state=42, shuffle=True)\nsplits = kf.split(train_df_img_groups) # ideally, this should be multilabel stratification\n\ndef get_fold():    \n    for i, (train_index, valid_index) in enumerate(splits):\n        if i == FOLD:\n            return train_df_img_groups.iloc[train_index], train_df_img_groups.iloc[valid_index]\n        \ntrain_fold, valid_fold = get_fold()\n\ntrain_dataset = iMaterialistMaxDimDataset(train_fold, category_names)\ntrain_dataset.prepare()\n\nvalid_dataset = iMaterialistMaxDimDataset(valid_fold, category_names)\nvalid_dataset.prepare()\n\nconfig = iMaterialistMaxDimConfig()\n\n# Now that we have the fold, let's figure out the STEPS_PER_EPOCH\n# that we need to cover every instance. It turns out you need thousands\n# to do this. Instead we will just pick a high number.\n# config.STEPS_PER_EPOCH = len(train_fold)/(config.GPU_COUNT*config.IMAGES_PER_GPU)\nconfig.STEPS_PER_EPOCH = 1000\nprint(\"Using {} steps per epoch\".format(config.STEPS_PER_EPOCH))\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Second go round. Let's increase epoch numbers (might break kernel with hangup)\n# LR = 1e-3\nLR = 5e-4\nEPOCHS = [5, 15, 25]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = modellib.MaskRCNN(mode='training', config=config, model_dir=ROOT_DIR)\n\nmodel.load_weights(COCO_WEIGHTS_PATH, \n                   by_name=True, \n                   exclude=['mrcnn_class_logits', \n                            'mrcnn_bbox_fc', \n                            'mrcnn_bbox', \n                            'mrcnn_mask'])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# Train the head branches\n# Passing layers=\"heads\" freezes all layers except the head\n# layers. You can also pass a regular expression to select\n# which layers to train by name pattern.\n\nmodel.train(train_dataset, \n            valid_dataset, \n            learning_rate=LR, \n            epochs=EPOCHS[0], \n            layers='heads')\n\nhistory = model.keras_model.history.history","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"epochs = range(EPOCHS[0])\n\nfig, axes = plt.subplots(1, 3, figsize=(16,8))\naxes[0].plot(epochs, history['loss'], label=\"train loss\")\naxes[0].plot(epochs, history['val_loss'], label=\"valid loss\")\naxes[0].legend()\naxes[1].plot(epochs, history['mrcnn_class_loss'], label=\"train class loss\")\naxes[1].plot(epochs, history['val_mrcnn_class_loss'], label=\"valid class loss\")\naxes[1].legend()\naxes[2].plot(epochs, history['mrcnn_mask_loss'], label=\"train mask loss\")\naxes[2].plot(epochs, history['val_mrcnn_mask_loss'], label=\"valid mask loss\")\naxes[2].legend()\n\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# The logged model can be found in root directory (next to /Mask_RCNN/)\n# Look for /config_name><date_of_training><time_of_training>/\nroot_contents = os.listdir(ROOT_DIR)\nprint(\"Root Directory contents: {}\".format(root_contents))\nlog_dirs = [f for f in root_contents if config.NAME in f]\nprint(\"Log Directory contents: {}\".format(os.listdir(ROOT_DIR/log_dirs[0])))\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The log directory contains the weights after each epoch. No information on config, so we will want to save that information somehow too.","metadata":{}}]}