{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.6"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":13032,"databundleVersionId":862545,"sourceType":"competition"},{"sourceId":7376095,"sourceType":"datasetVersion","datasetId":4283274}],"dockerImageVersionId":25160,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Welcome to the world where fashion meets computer vision! This is a starter kernel that applies Mask R-CNN with COCO pretrained weights to the task of [iMaterialist (Fashion) 2019 at FGVC6](https://www.kaggle.com/c/imaterialist-fashion-2019-FGVC6).","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport sys\nimport json\nimport glob\nimport random\nfrom pathlib import Path\n\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport itertools\nfrom tqdm import tqdm\n\nfrom imgaug import augmenters as iaa\nfrom sklearn.model_selection import StratifiedKFold, KFold","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T22:57:48.495958Z","iopub.execute_input":"2025-05-19T22:57:48.496238Z","iopub.status.idle":"2025-05-19T22:57:49.755549Z","shell.execute_reply.started":"2025-05-19T22:57:48.496174Z","shell.execute_reply":"2025-05-19T22:57:49.754599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = Path('/kaggle/input')\nROOT_DIR = Path('/kaggle/working')\n\n# For demonstration purpose, the classification ignores attributes (only categories),\n# and the image size is set to 512, which is the same as the size of submission masks\nNUM_CATS = 13\nIMAGE_SIZE = 1024","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T22:57:52.89197Z","iopub.execute_input":"2025-05-19T22:57:52.892288Z","iopub.status.idle":"2025-05-19T22:57:52.896761Z","shell.execute_reply.started":"2025-05-19T22:57:52.892226Z","shell.execute_reply":"2025-05-19T22:57:52.895893Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dowload Libraries and Pretrained Weights","metadata":{}},{"cell_type":"code","source":"!git clone https://www.github.com/matterport/Mask_RCNN.git\nos.chdir('Mask_RCNN')\n\n!rm -rf .git # to prevent an error when the kernel is committed\n!rm -rf images assets # to prevent displaying images at the bottom of a kernel","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true,"execution":{"iopub.status.busy":"2025-05-19T22:57:55.917495Z","iopub.execute_input":"2025-05-19T22:57:55.917738Z","iopub.status.idle":"2025-05-19T22:58:03.611919Z","shell.execute_reply.started":"2025-05-19T22:57:55.9177Z","shell.execute_reply":"2025-05-19T22:58:03.610903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sys.path.append(ROOT_DIR/'Mask_RCNN')\nfrom mrcnn.config import Config\nfrom mrcnn import utils\nimport mrcnn.model as modellib\nfrom mrcnn import visualize\nfrom mrcnn.model import log","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"execution":{"iopub.status.busy":"2025-05-19T22:58:05.339957Z","iopub.execute_input":"2025-05-19T22:58:05.340224Z","iopub.status.idle":"2025-05-19T22:58:06.292143Z","shell.execute_reply.started":"2025-05-19T22:58:05.340182Z","shell.execute_reply":"2025-05-19T22:58:06.291431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!wget --quiet https://github.com/matterport/Mask_RCNN/releases/download/v2.0/mask_rcnn_coco.h5\n!ls -lh mask_rcnn_coco.h5\n\nCOCO_WEIGHTS_PATH = 'mask_rcnn_coco.h5'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T22:58:07.865974Z","iopub.execute_input":"2025-05-19T22:58:07.866251Z","iopub.status.idle":"2025-05-19T22:58:12.541981Z","shell.execute_reply.started":"2025-05-19T22:58:07.866208Z","shell.execute_reply":"2025-05-19T22:58:12.541215Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Set Config","metadata":{}},{"cell_type":"markdown","source":"Mask R-CNN has a load of hyperparameters. I only adjust some of them.","metadata":{}},{"cell_type":"code","source":"class FashionConfig(Config):\n    NAME = \"deepfashion2\"\n    NUM_CLASSES = NUM_CATS + 1 # +1 for the background class\n    \n    GPU_COUNT = 1\n    IMAGES_PER_GPU = 4 # a memory error occurs when IMAGES_PER_GPU is too high\n    \n    BACKBONE = 'resnet50'\n    \n    IMAGE_MIN_DIM = IMAGE_SIZE\n    IMAGE_MAX_DIM = IMAGE_SIZE    \n    IMAGE_RESIZE_MODE = 'none'\n    \n    RPN_ANCHOR_SCALES = (16, 32, 64, 128, 256)\n    #DETECTION_NMS_THRESHOLD = 0.0\n    \n    # STEPS_PER_EPOCH should be the number of instances \n    # divided by (GPU_COUNT*IMAGES_PER_GPU), and so should VALIDATION_STEPS;\n    # however, due to the time limit, I set them so that this kernel can be run in 9 hours\n    STEPS_PER_EPOCH = 1000\n    VALIDATION_STEPS = 200\n    \nconfig = FashionConfig()\nconfig.display()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T22:58:30.393284Z","iopub.execute_input":"2025-05-19T22:58:30.393559Z","iopub.status.idle":"2025-05-19T22:58:30.403195Z","shell.execute_reply.started":"2025-05-19T22:58:30.393524Z","shell.execute_reply":"2025-05-19T22:58:30.40198Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Make Datasets","metadata":{}},{"cell_type":"code","source":"# with open(DATA_DIR/\"label_descriptions.json\") as f:\n#     label_descriptions = json.load(f)\n\n# label_names = [x['name'] for x in label_descriptions['categories']]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/deepfashion2-original-with-dataframes/DeepFashion2/img_info_dataframes/train.csv')\nunique_categories = train_df['category_name'].unique()#.to_list()\nprint(unique_categories)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T22:59:03.825225Z","iopub.execute_input":"2025-05-19T22:59:03.825466Z","iopub.status.idle":"2025-05-19T22:59:18.604287Z","shell.execute_reply.started":"2025-05-19T22:59:03.825429Z","shell.execute_reply":"2025-05-19T22:59:18.60349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_df = pd.read_csv('/kaggle/input/deepfashion2-original-with-dataframes/DeepFashion2/img_info_dataframes/validation.csv')\nval_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T23:01:23.700956Z","iopub.execute_input":"2025-05-19T23:01:23.701202Z","iopub.status.idle":"2025-05-19T23:01:30.348414Z","shell.execute_reply.started":"2025-05-19T23:01:23.701164Z","shell.execute_reply":"2025-05-19T23:01:30.347592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# segment_df = pd.read_csv('/kaggle/input/imaterialist-fashion-2019-FGVC6/train.csv')\n\n# multilabel_percent = len(segment_df[segment_df['ClassId'].str.contains('_')])/len(segment_df)*100\n# print(f\"Segments that have attributes: {multilabel_percent:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T13:40:19.877924Z","iopub.execute_input":"2025-05-19T13:40:19.878177Z","iopub.status.idle":"2025-05-19T13:40:47.111528Z","shell.execute_reply.started":"2025-05-19T13:40:19.878141Z","shell.execute_reply":"2025-05-19T13:40:47.110825Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Segments that contain attributes are only 3.46% of data, and [according to the host](https://www.kaggle.com/c/imaterialist-fashion-2019-FGVC6/discussion/90643#523135), 80% of images have no attribute. So, in the first step, we can only deal with categories to reduce the complexity of the task.","metadata":{}},{"cell_type":"code","source":"# segment_df['CategoryId'] = segment_df['ClassId'].str.split('_').str[0]\n\n# print(\"Total segments: \", len(segment_df))\n# # segment_df['EncodedPixels'][0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T14:39:23.457495Z","iopub.execute_input":"2025-05-19T14:39:23.457792Z","iopub.status.idle":"2025-05-19T14:39:24.008555Z","shell.execute_reply.started":"2025-05-19T14:39:23.457736Z","shell.execute_reply":"2025-05-19T14:39:24.007759Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def resize_image(image_path):\n#     img = cv2.imread(image_path)\n#     img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n#     img = cv2.resize(img, (IMAGE_SIZE, IMAGE_SIZE), interpolation=cv2.INTER_AREA)  \n#     return img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T23:25:58.996389Z","iopub.execute_input":"2025-05-19T23:25:58.996689Z","iopub.status.idle":"2025-05-19T23:25:59.001056Z","shell.execute_reply.started":"2025-05-19T23:25:58.996634Z","shell.execute_reply":"2025-05-19T23:25:59.000058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from mrcnn.utils import Dataset\nimport pandas as pd\nimport numpy as np\nimport cv2\nfrom PIL import Image\nfrom tqdm import tqdm\nclass CustomDataset(Dataset):\n    def __init__(self, annotation_df, class_map, image_size):\n        super().__init__()  # Важно: убрать `self` здесь\n        self.annotation_df = annotation_df\n        self.class_map = class_map\n        self.image_size = image_size  # (height, width)\n        \n    def image_reference(self, image_id):\n        info = self.image_info[image_id]\n        return info['path'], [label_names[int(x)] for x in info['labels']]\n\n    def prepare(self):\n        # Добавление классов\n        for category_id, category_name in self.class_map.items():\n            self.add_class(\"dataset\", category_id, category_name)\n\n        # Группировка по уникальным изображениям\n        grouped = self.annotation_df.groupby('path')\n\n        # Добавление каждого изображения один раз\n        for idx, (image_path, group) in tqdm(enumerate(grouped)):\n            try:\n                img = Image.open(image_path)\n                self.add_image(\n                    \"dataset\",\n                    image_id=idx,\n                    path=image_path,\n                    width=img.width,\n                    height=img.height\n                )\n            except Exception as e:\n                print(f\"Ошибка при добавлении {image_path}: {str(e)}\")\n                \n    def load_image(self, image_id):\n        # Загрузка и ресайз изображения\n        image = Image.open(self.image_info[image_id][\"path\"])\n        image = image.resize(\n            (self.image_size[1], self.image_size[0]),  # PIL требует (width, height)\n            Image.BILINEAR  # Для изображений используем билинейную интерполяцию\n        )\n        return np.array(image)\n    \n    def load_mask(self, image_id):\n        image_info = self.image_info[image_id]\n        records = self.annotation_df[self.annotation_df['path'] == image_info[\"path\"]]\n    \n        masks = []\n        class_ids = []\n        for _, record in tqdm(records.iterrows()):\n            # Создание маски оригинального размера\n            segmentation = eval(record['segmentation'])\n            mask = self.polygon_to_mask(\n                segmentation, \n                image_info['width'], \n                image_info['height']\n            )\n            \n            # Ресайз маски до IMAGE_SIZE\n            mask = Image.fromarray(mask)\n            mask = mask.resize(\n                (self.image_size[1], self.image_size[0]), \n                Image.NEAREST  # Для масок используем NEAREST\n            )\n            mask = np.array(mask)\n            \n            # Проверка бинарности (значения 0 или 1)\n            mask = (mask > 0.5).astype(np.uint8)\n            masks.append(mask)\n            class_ids.append(record['category_id'])\n    \n        if masks:\n            masks = np.stack(masks, axis=-1)\n        else:\n            masks = np.empty((0, 0))\n        \n        return masks, np.array(class_ids, dtype=np.int32)\n    \n    @staticmethod\n    def polygon_to_mask(polygon, width, height):\n        mask = np.zeros((height, width), dtype=np.uint8)\n        pts = np.array(polygon).reshape((-1, 2))\n        cv2.fillPoly(mask, [pts.astype(int)], color=1)\n        return mask","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-20T00:11:38.952484Z","iopub.execute_input":"2025-05-20T00:11:38.95274Z","iopub.status.idle":"2025-05-20T00:11:38.965326Z","shell.execute_reply.started":"2025-05-20T00:11:38.952699Z","shell.execute_reply":"2025-05-20T00:11:38.964485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndef create_class_mapping(csv_path):\n    # Загрузка данных\n    df = pd.read_csv(csv_path)\n    \n    # Проверка уникальности category_id -> category_name\n    unique_pairs = df[['category_id', 'category_name']].drop_duplicates()\n    \n    # Проверка на конфликты (если один category_id имеет разные имена)\n    if len(unique_pairs) != unique_pairs['category_id'].nunique():\n        conflicting = unique_pairs[unique_pairs.duplicated('category_id', keep=False)]\n        raise ValueError(f\"Конфликт в именах категорий:\\n{conflicting}\")\n    \n    # Создание словаря {category_id: category_name}\n    class_map = pd.Series(\n        unique_pairs.category_name.values, \n        index=unique_pairs.category_id\n    ).to_dict()\n    \n    # Проверка что ID начинаются с 1 (если требуется)\n    min_id = min(class_map.keys())\n    if min_id < 1:\n        raise ValueError(f\"category_id должен быть ≥ 1. Минимальный ID: {min_id}\")\n    \n    print(f\"Создан словарь для {len(class_map)} классов:\")\n    for idx, name in class_map.items():\n        print(f\"ID: {idx:3} → {name}\")\n        \n    return class_map\nclass_map = create_class_mapping(\"/kaggle/input/deepfashion2-original-with-dataframes/DeepFashion2/img_info_dataframes/train.csv\")\nclass_map","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T23:06:25.457905Z","iopub.execute_input":"2025-05-19T23:06:25.458169Z","iopub.status.idle":"2025-05-19T23:06:28.751402Z","shell.execute_reply.started":"2025-05-19T23:06:25.45813Z","shell.execute_reply":"2025-05-19T23:06:28.750727Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# dataset_train = CustomDataset(train_df, class_map)\ndataset_train = CustomDataset(train_df, class_map, (IMAGE_SIZE, IMAGE_SIZE))\ndataset_train.prepare()\ndataset_val = CustomDataset(val_df, class_map, (IMAGE_SIZE, IMAGE_SIZE))\ndataset_val.prepare()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-20T00:19:01.930526Z","iopub.execute_input":"2025-05-20T00:19:01.930787Z","iopub.status.idle":"2025-05-20T00:50:48.56153Z","shell.execute_reply.started":"2025-05-20T00:19:01.930746Z","shell.execute_reply":"2025-05-20T00:50:48.560721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_resized(dataset):\n    image_id = 154\n    image = dataset.load_image(image_id)\n    masks, _ = dataset.load_mask(image_id)\n    \n    plt.figure(figsize=(10, 5))\n    plt.subplot(1, 3, 1)\n    plt.imshow(image)\n    plt.title(f\"Resized to {IMAGE_SIZE}\")\n    \n    plt.subplot(1, 3, 2)\n    if masks.shape[-1] > 0:\n        plt.imshow(image)\n        plt.imshow(masks[:, :, 0], alpha=1, cmap='viridis')\n    plt.title(\"Mask\")\n    # plt.subplot(1, 3, 3)\n    # if masks.shape[-1] > 0:\n    #     plt.imshow(image)\n    #     plt.imshow(masks[:, :, 1], alpha=1, cmap='viridis')\n    # plt.title(\"Mask\")\n    \n    plt.show()\n\nvisualize_resized(dataset_val)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-20T00:13:36.646325Z","iopub.execute_input":"2025-05-20T00:13:36.646563Z","iopub.status.idle":"2025-05-20T00:13:37.146929Z","shell.execute_reply.started":"2025-05-20T00:13:36.646527Z","shell.execute_reply":"2025-05-20T00:13:37.146131Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Rows with the same image are grouped together because the subsequent operations perform in an image level.","metadata":{}},{"cell_type":"code","source":"# image_df = segment_df.groupby('ImageId')['EncodedPixels', 'CategoryId'].agg(lambda x: list(x))\n# size_df = segment_df.groupby('ImageId')['Height', 'Width'].mean()\n# image_df = image_df.join(size_df, on='ImageId')\n\n# print(\"Total images: \", len(image_df))\n# image_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T22:59:30.872211Z","iopub.execute_input":"2025-05-19T22:59:30.872517Z","iopub.status.idle":"2025-05-19T22:59:30.876069Z","shell.execute_reply.started":"2025-05-19T22:59:30.872462Z","shell.execute_reply":"2025-05-19T22:59:30.875129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# len(image_df['EncodedPixels'][0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T22:59:36.394295Z","iopub.execute_input":"2025-05-19T22:59:36.39454Z","iopub.status.idle":"2025-05-19T22:59:36.39781Z","shell.execute_reply.started":"2025-05-19T22:59:36.394503Z","shell.execute_reply":"2025-05-19T22:59:36.397144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_df['segmentation'][0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T22:59:40.664458Z","iopub.execute_input":"2025-05-19T22:59:40.66476Z","iopub.status.idle":"2025-05-19T22:59:40.66784Z","shell.execute_reply.started":"2025-05-19T22:59:40.664697Z","shell.execute_reply":"2025-05-19T22:59:40.666948Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n# Загрузка изображения\nimage_path = train_df['path'][0]\nimage = cv2.imread(image_path)\nimage = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)  # Конвертация в RGB\n\n# Данные маски\n# mask_data = [[360, 172, 364, 166, 371, 163, 325, 168, 299, 202, 279, 234, 258, 263, 233, 299, 256, 311, 273, 273, 288, 246, 304, 223, 308, 203, 298, 211, 279, 239, 258, 269, 287, 292, 306, 242, 341, 213, 360, 172],\n#              [410, 182, 409, 188, 406, 194, 383, 230, 380, 277, 366, 323, 402, 330, 414, 291, 424, 250, 424, 240, 418, 269, 409, 299, 397, 336, 385, 367, 418, 375, 422, 340, 428, 307, 435, 261, 433, 197, 410, 182],\n#              [325, 168, 299, 202, 279, 234, 258, 263, 233, 299, 256, 311, 273, 273, 288, 246, 304, 223, 308, 203, 325, 168],\n#              [424, 240, 418, 269, 409, 299, 397, 336, 385, 367, 418, 375, 422, 340, 428, 307, 435, 261, 433, 197, 424, 240],\n#              [279, 234, 258, 263, 233, 299, 256, 311, 273, 273, 288, 246, 279, 234],\n#              [409, 299, 397, 336, 385, 367, 418, 375, 422, 340, 428, 307, 409, 299]]\nmask_data = eval(train_df['segmentation'][0])\n\n# Создаем маску (в градациях серого)\nmask = np.zeros(image.shape[:2], dtype=np.uint8)  # Одноканальная маска\n\n# Рисуем полигоны на маске\nfor polygon in mask_data:\n    pts = np.array([[x, y] for x, y in zip(polygon[::2], polygon[1::2])], dtype=np.int32)\n    cv2.fillPoly(mask, [pts], color=255)  # Белые области на черном фоне\n\n# Создаем фигуру с двумя суб-графиками\nfig, axes = plt.subplots(1, 2, figsize=(15, 7))\n\n# Отображаем исходное изображение\naxes[0].imshow(image)\naxes[0].set_title('Original Image')\naxes[0].axis('off')\n\n# Отображаем маску\naxes[1].imshow(mask, cmap='gray')\naxes[1].set_title('Segmentation Mask')\naxes[1].axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-19T22:59:55.430737Z","iopub.execute_input":"2025-05-19T22:59:55.431029Z","iopub.status.idle":"2025-05-19T22:59:56.138833Z","shell.execute_reply.started":"2025-05-19T22:59:55.430987Z","shell.execute_reply":"2025-05-19T22:59:56.137984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# image_df = train_df.groupby('path')['EncodedPixels', 'CategoryId'].agg(lambda x: list(x))\n# size_df = segment_df.groupby('ImageId')['Height', 'Width'].mean()\n# image_df = image_df.join(size_df, on='ImageId')\n\n# print(\"Total images: \", len(image_df))\n# image_df.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Here is the custom function that resizes an image.","metadata":{}},{"cell_type":"code","source":"# def resize_image(image_path):\n#     img = cv2.imread(image_path)\n#     img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n#     img = cv2.resize(img, (IMAGE_SIZE, IMAGE_SIZE), interpolation=cv2.INTER_AREA)  \n#     return img","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The crucial part is to create a dataset for this task.","metadata":{}},{"cell_type":"code","source":"# class FashionDataset(utils.Dataset):\n\n#     def __init__(self, df):\n#         super().__init__(self)\n        \n#         # Add classes\n#         for i, name in enumerate(label_names):\n#             self.add_class(\"fashion\", i+1, name)\n        \n#         # Add images \n#         for i, row in df.iterrows():\n#             self.add_image(\"fashion\", \n#                            image_id=row.name, \n#                            path=str(DATA_DIR/'train'/row.name), \n#                            labels=row['CategoryId'],\n#                            annotations=row['EncodedPixels'], \n#                            height=row['Height'], width=row['Width'])\n\n#     def image_reference(self, image_id):\n#         info = self.image_info[image_id]\n#         return info['path'], [label_names[int(x)] for x in info['labels']]\n    \n#     def load_image(self, image_id):\n#         return resize_image(self.image_info[image_id]['path'])\n\n#     def load_mask(self, image_id):\n#         info = self.image_info[image_id]\n                \n#         mask = np.zeros((IMAGE_SIZE, IMAGE_SIZE, len(info['annotations'])), dtype=np.uint8)\n#         labels = []\n        \n#         for m, (annotation, label) in enumerate(zip(info['annotations'], info['labels'])):\n#             sub_mask = np.full(info['height']*info['width'], 0, dtype=np.uint8)\n#             annotation = [int(x) for x in annotation.split(' ')]\n            \n#             for i, start_pixel in enumerate(annotation[::2]):\n#                 sub_mask[start_pixel: start_pixel+annotation[2*i+1]] = 1\n\n#             sub_mask = sub_mask.reshape((info['height'], info['width']), order='F')\n#             sub_mask = cv2.resize(sub_mask, (IMAGE_SIZE, IMAGE_SIZE), interpolation=cv2.INTER_NEAREST)\n            \n#             mask[:, :, m] = sub_mask\n#             labels.append(int(label)+1)\n            \n#         return mask, np.array(labels)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's visualize some random images and their masks.","metadata":{}},{"cell_type":"code","source":"# dataset = FashionDataset(image_df)\n# dataset.prepare()\n\n# for i in range(6):\n#     image_id = random.choice(dataset.image_ids)\n#     print(dataset.image_reference(image_id))\n    \n#     image = dataset.load_image(image_id)\n#     mask, class_ids = dataset.load_mask(image_id)\n#     visualize.display_top_masks(image, mask, class_ids, dataset.class_names, limit=4)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now, the data are partitioned into train and validation sets.","metadata":{}},{"cell_type":"code","source":"# # This code partially supports k-fold training, \n# # you can specify the fold to train and the total number of folds here\n# FOLD = 0\n# N_FOLDS = 5\n\n# kf = KFold(n_splits=N_FOLDS, random_state=42, shuffle=True)\n# splits = kf.split(image_df) # ideally, this should be multilabel stratification\n\n# def get_fold():    \n#     for i, (train_index, valid_index) in enumerate(splits):\n#         if i == FOLD:\n#             return image_df.iloc[train_index], image_df.iloc[valid_index]\n        \n# train_df, valid_df = get_fold()\n\n# train_dataset = FashionDataset(train_df)\n# train_dataset.prepare()\n\n# valid_dataset = FashionDataset(valid_df)\n# valid_dataset.prepare()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's visualize class distributions of the train and validation data.","metadata":{}},{"cell_type":"code","source":"# train_segments = np.concatenate(train_df['CategoryId'].values).astype(int)\n# print(\"Total train images: \", len(train_df))\n# print(\"Total train segments: \", len(train_segments))\n\n# plt.figure(figsize=(12, 3))\n# values, counts = np.unique(train_segments, return_counts=True)\n# plt.bar(values, counts)\n# plt.xticks(values, label_names, rotation='vertical')\n# plt.show()\n\n# valid_segments = np.concatenate(valid_df['CategoryId'].values).astype(int)\n# print(\"Total train images: \", len(valid_df))\n# print(\"Total validation segments: \", len(valid_segments))\n\n# plt.figure(figsize=(12, 3))\n# values, counts = np.unique(valid_segments, return_counts=True)\n# plt.bar(values, counts)\n# plt.xticks(values, label_names, rotation='vertical')\n# plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"# Note that any hyperparameters here, such as LR, may still not be optimal\nLR = 1e-4\nEPOCHS = [2, 6, 8]\n\nimport warnings \nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-20T00:54:36.080623Z","iopub.execute_input":"2025-05-20T00:54:36.080953Z","iopub.status.idle":"2025-05-20T00:54:36.084663Z","shell.execute_reply.started":"2025-05-20T00:54:36.080892Z","shell.execute_reply":"2025-05-20T00:54:36.08399Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This section creates a Mask R-CNN model and specifies augmentations to be used.","metadata":{}},{"cell_type":"code","source":"model = modellib.MaskRCNN(mode='training', config=config, model_dir=ROOT_DIR)\n\nmodel.load_weights(COCO_WEIGHTS_PATH, by_name=True, exclude=[\n    'mrcnn_class_logits', 'mrcnn_bbox_fc', 'mrcnn_bbox', 'mrcnn_mask'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-20T00:54:39.204575Z","iopub.execute_input":"2025-05-20T00:54:39.205071Z","iopub.status.idle":"2025-05-20T00:54:47.252495Z","shell.execute_reply.started":"2025-05-20T00:54:39.204799Z","shell.execute_reply":"2025-05-20T00:54:47.251917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"augmentation = iaa.Sequential([\n    iaa.Fliplr(0.5) # only horizontal flip here\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-20T00:54:50.343701Z","iopub.execute_input":"2025-05-20T00:54:50.343988Z","iopub.status.idle":"2025-05-20T00:54:50.349036Z","shell.execute_reply.started":"2025-05-20T00:54:50.343947Z","shell.execute_reply":"2025-05-20T00:54:50.348Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"First, we train only the heads.","metadata":{}},{"cell_type":"code","source":"%%time\nmodel.train(dataset_train, dataset_val,\n            learning_rate=LR*2, # train heads with higher lr to speedup learning\n            epochs=EPOCHS[0],\n            layers='heads',\n            augmentation=None)\n\nhistory = model.keras_model.history.history","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-20T00:54:59.042468Z","iopub.execute_input":"2025-05-20T00:54:59.042749Z","iopub.status.idle":"2025-05-20T00:55:10.56022Z","shell.execute_reply.started":"2025-05-20T00:54:59.042707Z","shell.execute_reply":"2025-05-20T00:55:10.559322Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Then, all layers are trained.","metadata":{}},{"cell_type":"code","source":"%%time\nmodel.train(train_dataset, valid_dataset,\n            learning_rate=LR,\n            epochs=EPOCHS[1],\n            layers='all',\n            augmentation=augmentation)\n\nnew_history = model.keras_model.history.history\nfor k in new_history: history[k] = history[k] + new_history[k]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Afterwards, we reduce LR and train again.","metadata":{}},{"cell_type":"code","source":"%%time\nmodel.train(train_dataset, valid_dataset,\n            learning_rate=LR/5,\n            epochs=EPOCHS[2],\n            layers='all',\n            augmentation=augmentation)\n\nnew_history = model.keras_model.history.history\nfor k in new_history: history[k] = history[k] + new_history[k]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Let's visualize training history and choose the best epoch.","metadata":{}},{"cell_type":"code","source":"epochs = range(EPOCHS[-1])\n\nplt.figure(figsize=(18, 6))\n\nplt.subplot(131)\nplt.plot(epochs, history['loss'], label=\"train loss\")\nplt.plot(epochs, history['val_loss'], label=\"valid loss\")\nplt.legend()\nplt.subplot(132)\nplt.plot(epochs, history['mrcnn_class_loss'], label=\"train class loss\")\nplt.plot(epochs, history['val_mrcnn_class_loss'], label=\"valid class loss\")\nplt.legend()\nplt.subplot(133)\nplt.plot(epochs, history['mrcnn_mask_loss'], label=\"train mask loss\")\nplt.plot(epochs, history['val_mrcnn_mask_loss'], label=\"valid mask loss\")\nplt.legend()\n\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_epoch = np.argmin(history[\"val_loss\"]) + 1\nprint(\"Best epoch: \", best_epoch)\nprint(\"Valid loss: \", history[\"val_loss\"][best_epoch-1])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Predict","metadata":{}},{"cell_type":"code","source":"glob_list = glob.glob(f'/kaggle/working/fashion*/mask_rcnn_fashion_{best_epoch:04d}.h5')\nmodel_path = glob_list[0] if glob_list else ''","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class InferenceConfig(FashionConfig):\n    GPU_COUNT = 1\n    IMAGES_PER_GPU = 1\n\ninference_config = InferenceConfig()\n\nmodel = modellib.MaskRCNN(mode='inference', \n                          config=inference_config,\n                          model_dir=ROOT_DIR)\n\nassert model_path != '', \"Provide path to trained weights\"\nprint(\"Loading weights from \", model_path)\nmodel.load_weights(model_path, by_name=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in range(9):\n    image_id = sample_df.sample()['ImageId'].values[0]\n    image_path = str(DATA_DIR/'test'/image_id)\n    \n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n    \n    result = model.detect([resize_image(image_path)])\n    r = result[0]\n    \n    if r['masks'].size > 0:\n        masks = np.zeros((img.shape[0], img.shape[1], r['masks'].shape[-1]), dtype=np.uint8)\n        for m in range(r['masks'].shape[-1]):\n            masks[:, :, m] = cv2.resize(r['masks'][:, :, m].astype('uint8'), \n                                        (img.shape[1], img.shape[0]), interpolation=cv2.INTER_NEAREST)\n        \n        y_scale = img.shape[0]/IMAGE_SIZE\n        x_scale = img.shape[1]/IMAGE_SIZE\n        rois = (r['rois'] * [y_scale, x_scale, y_scale, x_scale]).astype(int)\n        \n        masks, rois = refine_masks(masks, rois)\n    else:\n        masks, rois = r['masks'], r['rois']\n        \n    visualize.display_instances(img, rois, masks, r['class_ids'], \n                                ['bg']+label_names, r['scores'],\n                                title=image_id, figsize=(12, 12))","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}