{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.6"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":7376095,"sourceType":"datasetVersion","datasetId":4283274}],"dockerImageVersionId":25160,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Cell 1: Imports\nimport os\nimport gc\nimport sys\nimport json\nimport glob\nimport random\nfrom pathlib import Path\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport itertools\nfrom tqdm import tqdm\nfrom imgaug import augmenters as iaa\nfrom sklearn.model_selection import StratifiedKFold, KFold\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:10:09.287423Z","iopub.execute_input":"2025-06-05T19:10:09.287761Z","iopub.status.idle":"2025-06-05T19:10:09.292857Z","shell.execute_reply.started":"2025-06-05T19:10:09.287693Z","shell.execute_reply":"2025-06-05T19:10:09.29213Z"}},"outputs":[],"execution_count":20},{"cell_type":"code","source":"# Cell 2: Paths and Basic Configuration\nDATA_DIR = Path('/kaggle/input')\nROOT_DIR = Path('/kaggle/working')\n\nNUM_CATS = 13 \nIMAGE_SIZE = 512 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:10:09.330819Z","iopub.execute_input":"2025-06-05T19:10:09.331065Z","iopub.status.idle":"2025-06-05T19:10:09.334607Z","shell.execute_reply.started":"2025-06-05T19:10:09.331013Z","shell.execute_reply":"2025-06-05T19:10:09.333818Z"}},"outputs":[],"execution_count":21},{"cell_type":"code","source":"# Cell 3: Clone Mask_RCNN and Setup (Original Working Version)\nif not (ROOT_DIR / 'Mask_RCNN').exists():\n    print(\"Cloning Mask_RCNN repository...\")\n    !git clone https://www.github.com/matterport/Mask_RCNN.git\n    os.chdir(ROOT_DIR / 'Mask_RCNN') \n    !rm -rf .git\n    !rm -rf images assets\n    os.chdir(ROOT_DIR) # Go back to the working root\n    print(\"Repository cloned and cleaned.\")\nelse:\n    print(\"Mask_RCNN directory already exists.\")\n\nmrcnn_library_path = str(ROOT_DIR / 'Mask_RCNN')\nif mrcnn_library_path not in sys.path:\n    sys.path.append(mrcnn_library_path)\n\nmrcnn_module_path = str(ROOT_DIR / 'Mask_RCNN' / 'mrcnn')\nif mrcnn_module_path not in sys.path:\n     sys.path.append(mrcnn_module_path)\n\n\nfrom mrcnn.config import Config\nfrom mrcnn import utils\nimport mrcnn.model as modellib\nfrom mrcnn import visualize\nfrom mrcnn.model import log\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:10:09.40291Z","iopub.execute_input":"2025-06-05T19:10:09.403158Z","iopub.status.idle":"2025-06-05T19:10:09.409331Z","shell.execute_reply.started":"2025-06-05T19:10:09.40312Z","shell.execute_reply":"2025-06-05T19:10:09.408597Z"}},"outputs":[{"name":"stdout","text":"Mask_RCNN directory already exists.\n","output_type":"stream"}],"execution_count":22},{"cell_type":"code","source":"# Cell 4: Download COCO Weights (Corrected wget and COCO_WEIGHTS_PATH to string)\n# Define COCO_WEIGHTS_PATH as a string\ncoco_weights_file = ROOT_DIR / 'Mask_RCNN' / 'mask_rcnn_coco.h5'\nCOCO_WEIGHTS_PATH = str(coco_weights_file) # Ensure it's a string\n\nif not coco_weights_file.exists(): # Check existence using the Path object\n    print(\"Downloading COCO weights to Mask_RCNN directory...\")\n    # Ensure Mask_RCNN directory exists\n    os.makedirs(ROOT_DIR / 'Mask_RCNN', exist_ok=True)\n    # Corrected wget command, using the string path for -O and correct URL\n    !wget --quiet https://github.com/matterport/Mask_RCNN/releases/download/v2.0/mask_rcnn_coco.h5 -O {COCO_WEIGHTS_PATH}\n    if coco_weights_file.exists():\n        print(\"COCO weights downloaded successfully.\")\n    else:\n        print(\"ERROR: COCO weights download failed. Please check the URL and network.\")\nelse:\n    print(f\"COCO weights already exist at {COCO_WEIGHTS_PATH}\")\n\nif os.path.exists(COCO_WEIGHTS_PATH): # os.path.exists also works with string paths\n    print(f\"Weights file confirmed to exist at: {COCO_WEIGHTS_PATH}\")\nelse:\n    print(f\"ERROR: Weights file NOT found at: {COCO_WEIGHTS_PATH} after attempting download. Check for errors above.\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:10:09.453313Z","iopub.execute_input":"2025-06-05T19:10:09.45364Z","iopub.status.idle":"2025-06-05T19:10:09.459174Z","shell.execute_reply.started":"2025-06-05T19:10:09.453584Z","shell.execute_reply":"2025-06-05T19:10:09.458547Z"}},"outputs":[{"name":"stdout","text":"COCO weights already exist at /kaggle/working/Mask_RCNN/mask_rcnn_coco.h5\nWeights file confirmed to exist at: /kaggle/working/Mask_RCNN/mask_rcnn_coco.h5\n","output_type":"stream"}],"execution_count":23},{"cell_type":"code","source":"# Cell 5: Model Configuration (Reverted to 1 GPU)\nclass FashionConfig(Config):\n    NAME = \"deepfashion2\"\n    NUM_CLASSES = NUM_CATS + 1\n    GPU_COUNT = 1\n    IMAGES_PER_GPU =  1\n    BACKBONE = 'resnet50'\n    IMAGE_MIN_DIM = IMAGE_SIZE\n    IMAGE_MAX_DIM = IMAGE_SIZE    \n    IMAGE_RESIZE_MODE = 'square' \n    # RPN_ANCHOR_SCALES = (16, 32, 64, 128, 256) \n    RPN_ANCHOR_SCALES = (8, 16, 32, 64, 128)\n    \n    STEPS_PER_EPOCH = 1000 \n    VALIDATION_STEPS = 200 \n    LEARNING_RATE = 0.001 \n\nconfig = FashionConfig()\nconfig.display()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:10:09.499625Z","iopub.execute_input":"2025-06-05T19:10:09.49993Z","iopub.status.idle":"2025-06-05T19:10:09.50841Z","shell.execute_reply.started":"2025-06-05T19:10:09.499873Z","shell.execute_reply":"2025-06-05T19:10:09.507731Z"}},"outputs":[{"name":"stdout","text":"\nConfigurations:\nBACKBONE                       resnet50\nBACKBONE_STRIDES               [4, 8, 16, 32, 64]\nBATCH_SIZE                     1\nBBOX_STD_DEV                   [0.1 0.1 0.2 0.2]\nCOMPUTE_BACKBONE_SHAPE         None\nDETECTION_MAX_INSTANCES        100\nDETECTION_MIN_CONFIDENCE       0.7\nDETECTION_NMS_THRESHOLD        0.3\nFPN_CLASSIF_FC_LAYERS_SIZE     1024\nGPU_COUNT                      1\nGRADIENT_CLIP_NORM             5.0\nIMAGES_PER_GPU                 1\nIMAGE_CHANNEL_COUNT            3\nIMAGE_MAX_DIM                  512\nIMAGE_META_SIZE                26\nIMAGE_MIN_DIM                  512\nIMAGE_MIN_SCALE                0\nIMAGE_RESIZE_MODE              square\nIMAGE_SHAPE                    [512 512   3]\nLEARNING_MOMENTUM              0.9\nLEARNING_RATE                  0.001\nLOSS_WEIGHTS                   {'rpn_class_loss': 1.0, 'rpn_bbox_loss': 1.0, 'mrcnn_class_loss': 1.0, 'mrcnn_bbox_loss': 1.0, 'mrcnn_mask_loss': 1.0}\nMASK_POOL_SIZE                 14\nMASK_SHAPE                     [28, 28]\nMAX_GT_INSTANCES               100\nMEAN_PIXEL                     [123.7 116.8 103.9]\nMINI_MASK_SHAPE                (56, 56)\nNAME                           deepfashion2\nNUM_CLASSES                    14\nPOOL_SIZE                      7\nPOST_NMS_ROIS_INFERENCE        1000\nPOST_NMS_ROIS_TRAINING         2000\nPRE_NMS_LIMIT                  6000\nROI_POSITIVE_RATIO             0.33\nRPN_ANCHOR_RATIOS              [0.5, 1, 2]\nRPN_ANCHOR_SCALES              (8, 16, 32, 64, 128)\nRPN_ANCHOR_STRIDE              1\nRPN_BBOX_STD_DEV               [0.1 0.1 0.2 0.2]\nRPN_NMS_THRESHOLD              0.7\nRPN_TRAIN_ANCHORS_PER_IMAGE    256\nSTEPS_PER_EPOCH                1000\nTOP_DOWN_PYRAMID_SIZE          256\nTRAIN_BN                       False\nTRAIN_ROIS_PER_IMAGE           200\nUSE_MINI_MASK                  True\nUSE_RPN_ROIS                   True\nVALIDATION_STEPS               200\nWEIGHT_DECAY                   0.0001\n\n\n","output_type":"stream"}],"execution_count":24},{"cell_type":"code","source":"# Cell 6: Load Full DataFrames\ntrain_df_full = pd.read_csv('/kaggle/input/deepfashion2-original-with-dataframes/DeepFashion2/img_info_dataframes/train.csv')\nval_df_full = pd.read_csv('/kaggle/input/deepfashion2-original-with-dataframes/DeepFashion2/img_info_dataframes/validation.csv')\nprint(f\"Full train_df loaded with {train_df_full['path'].nunique()} unique images and {len(train_df_full)} annotations.\")\nprint(f\"Full val_df loaded with {val_df_full['path'].nunique()} unique images and {len(val_df_full)} annotations.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:10:09.557607Z","iopub.execute_input":"2025-06-05T19:10:09.557859Z","iopub.status.idle":"2025-06-05T19:10:15.025111Z","shell.execute_reply.started":"2025-06-05T19:10:09.557819Z","shell.execute_reply":"2025-06-05T19:10:15.024423Z"}},"outputs":[{"name":"stdout","text":"Full train_df loaded with 191961 unique images and 312186 annotations.\nFull val_df loaded with 32153 unique images and 52490 annotations.\n","output_type":"stream"}],"execution_count":25},{"cell_type":"code","source":"# Cell 7: CustomDataset Class (Using ast.literal_eval with improved error reporting)\nimport ast # For ast.literal_eval\nfrom PIL import Image\n# tqdm, random, cv2, numpy are usually imported in Cell 1.\n# Ensure they are available. If not, uncomment the lines below or add to Cell 1.\n# from tqdm import tqdm\n# import random\n# import cv2\n# import numpy as np\n\n# Ensure utils is available from mrcnn (usually from Cell 3 imports)\n# from mrcnn import utils # This should already be effectively imported via 'import mrcnn.model as modellib'\n\nclass CustomDataset(utils.Dataset):\n    def __init__(self, annotation_df, class_map, max_images=None):\n        super().__init__()\n        self.annotation_df = annotation_df\n        self.class_map = class_map\n        self.max_images = max_images\n        self.parsing_error_count = 0 # Counter for parsing errors\n        self.fillpoly_error_count = 0 # Counter for fillPoly errors\n\n        for category_id, category_name in self.class_map.items():\n            self.add_class(\"dataset\", int(category_id), category_name)\n\n    def load_image(self, image_id):\n        info = self.image_info[image_id]\n        image = Image.open(info[\"path\"]).convert('RGB')\n        return np.array(image)\n\n    def load_mask(self, image_id):\n        info = self.image_info[image_id]\n        records = self.annotation_df[self.annotation_df['path'] == info[\"path\"]]\n\n        masks = []\n        class_ids = []\n\n        original_height = info['height']\n        original_width = info['width']\n\n        for _, record in records.iterrows():\n            segmentation_str = record['segmentation']\n            try:\n                # Use ast.literal_eval for safer and faster parsing\n                segmentation_data = ast.literal_eval(segmentation_str)\n            except (ValueError, SyntaxError) as e:\n                if self.parsing_error_count < 10: # Print only the first few errors\n                    print(f\"DEBUG: ast.literal_eval failed for image_id {image_id}, path {info['path']}.\")\n                    print(f\"   Problematic data string: '{segmentation_str}'\")\n                    print(f\"   Error: {e}\")\n                self.parsing_error_count += 1\n                continue # Skip this problematic annotation\n\n            instance_mask = np.zeros((original_height, original_width), dtype=np.uint8)\n            \n            # Flag to check if any polygon was successfully processed for this annotation\n            processed_polygon_for_annotation = False\n\n            try:\n                # Handle cases where segmentation_data is a list of polygons (list of lists of points)\n                if isinstance(segmentation_data, list) and segmentation_data and \\\n                   all(isinstance(poly, list) for poly in segmentation_data):\n                    for polygon in segmentation_data:\n                        if not polygon or not isinstance(polygon, list) or len(polygon) < 3: continue # Skip empty or invalid polygons\n                        pts = np.array(polygon).reshape((-1, 2)).astype(np.int32)\n                        if pts.shape[0] < 3 or pts.ndim != 2 or pts.shape[1] != 2: continue # Ensure valid points array\n                        cv2.fillPoly(instance_mask, [pts], 1)\n                        processed_polygon_for_annotation = True\n                # Handle cases where segmentation_data is a single polygon (list of points)\n                elif isinstance(segmentation_data, list) and segmentation_data and len(segmentation_data) >= 3:\n                    pts = np.array(segmentation_data).reshape((-1, 2)).astype(np.int32)\n                    if pts.shape[0] < 3 or pts.ndim != 2 or pts.shape[1] != 2: continue # Ensure valid points array\n                    cv2.fillPoly(instance_mask, [pts], 1)\n                    processed_polygon_for_annotation = True\n                else:\n                    if self.parsing_error_count < 10: # Count as a parsing/format error\n                         print(f\"DEBUG: Unexpected segmentation format (not list of lists or list of >=3 points) for image_id {image_id}, path {info['path']}. Data: {segmentation_data}\")\n                    self.parsing_error_count += 1\n                    continue # Skip this annotation\n            \n            except Exception as fill_e: # Catch errors specifically from np.array, reshape, or fillPoly\n                if self.fillpoly_error_count < 10:\n                     print(f\"DEBUG: cv2.fillPoly error for image_id {image_id}, path {info['path']}. Parsed data: {segmentation_data}. Error: {fill_e}\")\n                self.fillpoly_error_count +=1\n                continue # Skip this annotation if fillPoly fails\n\n            # Only append mask if at least one polygon was processed for this annotation\n            if processed_polygon_for_annotation:\n                masks.append(instance_mask)\n                class_ids.append(int(record['category_id']))\n\n        if not masks:\n            # This means all annotations for this image were skipped due to errors, or it had no valid annotations\n            return np.empty([original_height, original_width, 0], dtype=bool), np.empty([0], dtype=np.int32)\n\n        masks_np_original_size = np.stack(masks, axis=-1)\n        return masks_np_original_size.astype(bool), np.array(class_ids, dtype=np.int32)\n\n    def prepare(self, class_map_override=None):\n        self.parsing_error_count = 0 # Reset error counter for each prepare call\n        self.fillpoly_error_count = 0\n        \n        df_for_preparation = self.annotation_df\n        unique_paths_in_df = list(self.annotation_df['path'].unique())\n\n        if self.max_images is not None and self.max_images < len(unique_paths_in_df):\n            print(f\"Randomly selecting {self.max_images} images from {len(unique_paths_in_df)} available unique image paths.\")\n            selected_paths = random.sample(unique_paths_in_df, self.max_images)\n            df_for_preparation = self.annotation_df[self.annotation_df['path'].isin(selected_paths)].copy()\n        else:\n            if self.max_images is not None:\n                 print(f\"Requested {self.max_images} images, but only {len(unique_paths_in_df)} unique images are available. Using all available images.\")\n            else:\n                print(f\"Using all {len(unique_paths_in_df)} unique images for preparation.\")\n            df_for_preparation = self.annotation_df.copy()\n\n        grouped = df_for_preparation.groupby('path')\n\n        for idx, (image_path, group) in tqdm(enumerate(grouped), total=df_for_preparation['path'].nunique(), mininterval=0.5):\n            try:\n                image_record = group.iloc[0]\n                self.add_image(\n                    \"dataset\",\n                    image_id=idx,\n                    path=image_path,\n                    width=image_record['img_width'],\n                    height=image_record['img_height']\n                )\n            except Exception as e:\n                print(f\"Error adding image {image_path}: {str(e)}\")\n        \n        super().prepare()\n        \n        if self.parsing_error_count > 0:\n            print(f\"WARNING: Encountered {self.parsing_error_count} errors while parsing segmentation strings with ast.literal_eval.\")\n        if self.fillpoly_error_count > 0:\n            print(f\"WARNING: Encountered {self.fillpoly_error_count} errors during cv2.fillPoly operation.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:10:15.026512Z","iopub.execute_input":"2025-06-05T19:10:15.026808Z","iopub.status.idle":"2025-06-05T19:10:15.044993Z","shell.execute_reply.started":"2025-06-05T19:10:15.026749Z","shell.execute_reply":"2025-06-05T19:10:15.044159Z"}},"outputs":[],"execution_count":26},{"cell_type":"code","source":"# # Cell 7: CustomDataset Class (SPEED FIX APPLIED HERE)\n# from PIL import Image\n# from tqdm import tqdm\n# import random\n\n# class CustomDataset(utils.Dataset):\n#     def __init__(self, annotation_df, class_map, max_images=None):\n#         super().__init__()\n#         self.annotation_df = annotation_df\n#         self.class_map = class_map\n#         self.max_images = max_images\n\n#         for category_id, category_name in self.class_map.items():\n#             self.add_class(\"dataset\", int(category_id), category_name)\n\n#     def load_image(self, image_id):\n#         info = self.image_info[image_id]\n#         image = Image.open(info[\"path\"]).convert('RGB')\n#         return np.array(image)\n\n#     def load_mask(self, image_id):\n#         info = self.image_info[image_id]\n#         records = self.annotation_df[self.annotation_df['path'] == info[\"path\"]]\n\n#         masks = []\n#         class_ids = []\n\n#         original_height = info['height']\n#         original_width = info['width']\n\n#         for _, record in records.iterrows():\n#             segmentation_str = record['segmentation']\n#             try:\n#                 # CRITICAL SPEED FIX: Use ast.literal_eval instead of eval()\n#                 segmentation_data = ast.literal_eval(segmentation_str)\n#             except Exception as e:\n#                 # print(f\"Error evaluating segmentation for image_id {image_id}, path {info['path']}. Data: {segmentation_str}. Error: {e}\")\n#                 # It's better to log errors or count them rather than printing for every problematic row during training\n#                 continue\n\n#             instance_mask = np.zeros((original_height, original_width), dtype=np.uint8)\n\n#             if isinstance(segmentation_data, list) and all(isinstance(poly, list) for poly in segmentation_data):\n#                 for polygon in segmentation_data:\n#                     if not polygon: continue\n#                     pts = np.array(polygon).reshape((-1, 2)).astype(np.int32)\n#                     if pts.size == 0: continue\n#                     cv2.fillPoly(instance_mask, [pts], 1)\n#             elif isinstance(segmentation_data, list): # Handles cases where segmentation_data is a single polygon list\n#                 if not segmentation_data: continue\n#                 pts = np.array(segmentation_data).reshape((-1, 2)).astype(np.int32)\n#                 if pts.size == 0: continue\n#                 cv2.fillPoly(instance_mask, [pts], 1)\n#             else:\n#                 # print(f\"Unexpected segmentation format for image_id {image_id}, path {info['path']}.\")\n#                 continue\n\n#             masks.append(instance_mask)\n#             class_ids.append(int(record['category_id']))\n\n#         if not masks:\n#             return np.empty([original_height, original_width, 0], dtype=bool), np.empty([0], dtype=np.int32)\n\n#         masks_np_original_size = np.stack(masks, axis=-1)\n#         return masks_np_original_size.astype(bool), np.array(class_ids, dtype=np.int32)\n\n#     def prepare(self, class_map_override=None):\n#         df_for_preparation = self.annotation_df\n#         unique_paths_in_df = list(self.annotation_df['path'].unique())\n\n#         if self.max_images is not None and self.max_images < len(unique_paths_in_df):\n#             print(f\"Randomly selecting {self.max_images} images from {len(unique_paths_in_df)} available unique image paths.\")\n#             selected_paths = random.sample(unique_paths_in_df, self.max_images)\n#             df_for_preparation = self.annotation_df[self.annotation_df['path'].isin(selected_paths)].copy()\n#         else:\n#             if self.max_images is not None:\n#                  print(f\"Requested {self.max_images} images, but only {len(unique_paths_in_df)} unique images are available. Using all available images.\")\n#             else:\n#                 print(f\"Using all {len(unique_paths_in_df)} unique images for preparation.\")\n#             df_for_preparation = self.annotation_df.copy()\n\n#         grouped = df_for_preparation.groupby('path')\n\n#         for idx, (image_path, group) in tqdm(enumerate(grouped), total=df_for_preparation['path'].nunique(), mininterval=0.5):\n#             try:\n#                 image_record = group.iloc[0]\n#                 self.add_image(\n#                     \"dataset\",\n#                     image_id=idx,\n#                     path=image_path,\n#                     width=image_record['img_width'],\n#                     height=image_record['img_height']\n#                 )\n#             except Exception as e:\n#                 print(f\"Error adding image {image_path}: {str(e)}\")\n#         super().prepare()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:10:15.046028Z","iopub.execute_input":"2025-06-05T19:10:15.046236Z","iopub.status.idle":"2025-06-05T19:10:15.067464Z","shell.execute_reply.started":"2025-06-05T19:10:15.046192Z","shell.execute_reply":"2025-06-05T19:10:15.066626Z"}},"outputs":[],"execution_count":27},{"cell_type":"code","source":"\n# Cell 8: Create Class Mapping\ndef create_class_mapping(df_path):\n    df = pd.read_csv(df_path)\n    unique_pairs = df[['category_id', 'category_name']].drop_duplicates()\n    class_map_dict = pd.Series(unique_pairs.category_name.values, index=unique_pairs.category_id).to_dict()\n    print(f\"Created map for {len(class_map_dict)} classes:\")\n    for idx, name in class_map_dict.items():\n        print(f\"ID: {int(idx):3} -> {name}\")\n    return class_map_dict\n\nclass_map = create_class_mapping(\"/kaggle/input/deepfashion2-original-with-dataframes/DeepFashion2/img_info_dataframes/train.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:10:15.068762Z","iopub.execute_input":"2025-06-05T19:10:15.068993Z","iopub.status.idle":"2025-06-05T19:10:18.358439Z","shell.execute_reply.started":"2025-06-05T19:10:15.068935Z","shell.execute_reply":"2025-06-05T19:10:18.357734Z"}},"outputs":[{"name":"stdout","text":"Created map for 13 classes:\nID:   4 -> long sleeve outwear\nID:   8 -> trousers\nID:   1 -> short sleeve top\nID:   9 -> skirt\nID:  10 -> short sleeve dress\nID:   7 -> shorts\nID:   5 -> vest\nID:  12 -> vest dress\nID:   2 -> long sleeve top\nID:  11 -> long sleeve dress\nID:  13 -> sling dress\nID:   6 -> sling\nID:   3 -> short sleeve outwear\n","output_type":"stream"}],"execution_count":28},{"cell_type":"code","source":"\n# Cell 9: Prepare Datasets for Training (with subsetting option)\n\n# --- USER CONFIGURATION FOR SUBSETTING ---\n# Set to None to use all images from the respective dataframes (train_df_full, val_df_full).\n# For a quick test, you might set these to small numbers.\nNUM_TRAIN_IMAGES_TO_USE =100  # Example: Use up to 1000 unique training images\nNUM_VAL_IMAGES_TO_USE = 10     # Example: Use up to 200 unique validation images\n# --- END USER CONFIGURATION ---\n\n# train_df_full and val_df_full are loaded in Cell 6\n# class_map is created in Cell 8\n\nprint(f\"Full train_df (train_df_full) contains {train_df_full['path'].nunique()} unique images and {len(train_df_full)} annotations.\")\nprint(f\"Full val_df (val_df_full) contains {val_df_full['path'].nunique()} unique images and {len(val_df_full)} annotations.\")\n\nif NUM_TRAIN_IMAGES_TO_USE is not None:\n    print(f\"\\nAttempting to prepare training dataset with up to {NUM_TRAIN_IMAGES_TO_USE} unique images.\")\nelse:\n    print(f\"\\nAttempting to prepare training dataset with all available unique images.\")\n\nprint(\"Preparing train dataset (this may take some time depending on the number of images)...\")\n# Pass the full training DataFrame and the desired number of images to CustomDataset\ndataset_train = CustomDataset(train_df_full, class_map, max_images=NUM_TRAIN_IMAGES_TO_USE)\ndataset_train.prepare()\n\nif NUM_VAL_IMAGES_TO_USE is not None:\n    print(f\"\\nAttempting to prepare validation dataset with up to {NUM_VAL_IMAGES_TO_USE} unique images.\")\nelse:\n    print(f\"\\nAttempting to prepare validation dataset with all available unique images.\")\n\nprint(\"Preparing val dataset (this may take some time depending on the number of images)...\")\n# Pass the full validation DataFrame and the desired number of images to CustomDataset\ndataset_val = CustomDataset(val_df_full, class_map, max_images=NUM_VAL_IMAGES_TO_USE)\ndataset_val.prepare()\n\n# Verification of dataset sizes after preparation\nif hasattr(dataset_train, 'image_ids') and hasattr(dataset_val, 'image_ids'):\n    print(f\"\\nSuccessfully prepared train dataset with {len(dataset_train.image_ids)} images.\")\n    print(f\"Successfully prepared val dataset with {len(dataset_val.image_ids)} images.\")\n    if NUM_TRAIN_IMAGES_TO_USE is not None:\n        print(f\"(Requested up to {NUM_TRAIN_IMAGES_TO_USE} for training)\")\n    if NUM_VAL_IMAGES_TO_USE is not None:\n        print(f\"(Requested up to {NUM_VAL_IMAGES_TO_USE} for validation)\")\nelse:\n    print(\"\\nError: One or both datasets (dataset_train, dataset_val) were not prepared successfully or are empty.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:10:18.359711Z","iopub.execute_input":"2025-06-05T19:10:18.359924Z","iopub.status.idle":"2025-06-05T19:10:18.720087Z","shell.execute_reply.started":"2025-06-05T19:10:18.359888Z","shell.execute_reply":"2025-06-05T19:10:18.719161Z"}},"outputs":[{"name":"stdout","text":"Full train_df (train_df_full) contains 191961 unique images and 312186 annotations.\n","output_type":"stream"},{"name":"stderr","text":"100%|██████████| 100/100 [00:00<00:00, 2523.62it/s]","output_type":"stream"},{"name":"stdout","text":"Full val_df (val_df_full) contains 32153 unique images and 52490 annotations.\n\nAttempting to prepare training dataset with up to 100 unique images.\nPreparing train dataset (this may take some time depending on the number of images)...\nRandomly selecting 100 images from 191961 available unique image paths.\n\nAttempting to prepare validation dataset with up to 10 unique images.\nPreparing val dataset (this may take some time depending on the number of images)...\n","output_type":"stream"},{"name":"stderr","text":"\n100%|██████████| 10/10 [00:00<00:00, 1961.24it/s]","output_type":"stream"},{"name":"stdout","text":"Randomly selecting 10 images from 32153 available unique image paths.\n\nSuccessfully prepared train dataset with 100 images.\nSuccessfully prepared val dataset with 10 images.\n(Requested up to 100 for training)\n(Requested up to 10 for validation)\n","output_type":"stream"},{"name":"stderr","text":"\n","output_type":"stream"}],"execution_count":29},{"cell_type":"code","source":"# Cell 10: Adjust Config for Full Datasets and Define Training Epochs\nif not (hasattr(dataset_train, 'image_ids') and len(dataset_train.image_ids) > 0 and \\\n        hasattr(dataset_val, 'image_ids') and len(dataset_val.image_ids) > 0):\n    print(\"Full datasets are empty or not properly prepared. Training will be skipped.\")\n    CAN_TRAIN = False\nelse:\n    CAN_TRAIN = True\n    config.STEPS_PER_EPOCH = max(1, len(dataset_train.image_ids) // config.IMAGES_PER_GPU)\n    config.VALIDATION_STEPS = max(1, len(dataset_val.image_ids) // config.IMAGES_PER_GPU)\n    \n    EPOCHS_FULL_TRAINING = [2, 6, 8] # Define epochs for full training\n    \n    print(f\"Adjusted for full dataset: STEPS_PER_EPOCH = {config.STEPS_PER_EPOCH}, VALIDATION_STEPS = {config.VALIDATION_STEPS}\")\n    print(f\"Using full training epochs: {EPOCHS_FULL_TRAINING}\")\n\nhistory_accumulator = {} \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:10:18.721711Z","iopub.execute_input":"2025-06-05T19:10:18.721944Z","iopub.status.idle":"2025-06-05T19:10:18.727053Z","shell.execute_reply.started":"2025-06-05T19:10:18.721894Z","shell.execute_reply":"2025-06-05T19:10:18.726403Z"}},"outputs":[{"name":"stdout","text":"Adjusted for full dataset: STEPS_PER_EPOCH = 100, VALIDATION_STEPS = 10\nUsing full training epochs: [2, 6, 8]\n","output_type":"stream"}],"execution_count":30},{"cell_type":"code","source":"# Cell 11: Initialize Model and Load COCO Weights\nmodel = modellib.MaskRCNN(mode='training', config=config, model_dir=ROOT_DIR)\nif os.path.exists(COCO_WEIGHTS_PATH):\n    model.load_weights(COCO_WEIGHTS_PATH, by_name=True, exclude=[\n        'mrcnn_class_logits', 'mrcnn_bbox_fc', 'mrcnn_bbox', 'mrcnn_mask'])\n    print(\"COCO weights loaded.\")\nelse:\n    print(\"COCO weights not found. Model will train from scratch or imagenet if backbone supports.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:10:18.728306Z","iopub.execute_input":"2025-06-05T19:10:18.728552Z","iopub.status.idle":"2025-06-05T19:10:26.167516Z","shell.execute_reply.started":"2025-06-05T19:10:18.728504Z","shell.execute_reply":"2025-06-05T19:10:26.166792Z"}},"outputs":[{"name":"stdout","text":"COCO weights loaded.\n","output_type":"stream"}],"execution_count":31},{"cell_type":"code","source":"# Cell 12: Define Augmentation\naugmentation = iaa.Sequential([\n    iaa.Fliplr(0.5)\n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:10:26.169069Z","iopub.execute_input":"2025-06-05T19:10:26.169549Z","iopub.status.idle":"2025-06-05T19:10:26.174061Z","shell.execute_reply.started":"2025-06-05T19:10:26.169297Z","shell.execute_reply":"2025-06-05T19:10:26.173195Z"}},"outputs":[],"execution_count":32},{"cell_type":"code","source":"# Cell 13: Training Stage 1 (Heads)\nif CAN_TRAIN:\n    print(\"\\nTraining heads...\")\n    model.train(dataset_train, dataset_val,\n                learning_rate=config.LEARNING_RATE * 2, \n                epochs=EPOCHS_FULL_TRAINING[0],\n                layers='heads',\n                augmentation=None)\n    if model.keras_model.history and model.keras_model.history.history:\n        current_hist = model.keras_model.history.history\n        for k in current_hist: history_accumulator[k] = current_hist[k]\n    else:\n        print(\"No history from head training.\")\nelse:\n    print(\"Skipping head training.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:10:26.175124Z","iopub.execute_input":"2025-06-05T19:10:26.175308Z","iopub.status.idle":"2025-06-05T19:12:25.244722Z","shell.execute_reply.started":"2025-06-05T19:10:26.175273Z","shell.execute_reply":"2025-06-05T19:12:25.243822Z"}},"outputs":[{"name":"stdout","text":"\nTraining heads...\n\nStarting at epoch 0. LR=0.002\n\nCheckpoint Path: /kaggle/working/deepfashion220250605T1910/mask_rcnn_deepfashion2_{epoch:04d}.h5\nSelecting layers to train\nfpn_c5p5               (Conv2D)\nfpn_c4p4               (Conv2D)\nfpn_c3p3               (Conv2D)\nfpn_c2p2               (Conv2D)\nfpn_p5                 (Conv2D)\nfpn_p2                 (Conv2D)\nfpn_p3                 (Conv2D)\nfpn_p4                 (Conv2D)\nIn model:  rpn_model\n    rpn_conv_shared        (Conv2D)\n    rpn_class_raw          (Conv2D)\n    rpn_bbox_pred          (Conv2D)\nmrcnn_mask_conv1       (TimeDistributed)\nmrcnn_mask_bn1         (TimeDistributed)\nmrcnn_mask_conv2       (TimeDistributed)\nmrcnn_mask_bn2         (TimeDistributed)\nmrcnn_class_conv1      (TimeDistributed)\nmrcnn_class_bn1        (TimeDistributed)\nmrcnn_mask_conv3       (TimeDistributed)\nmrcnn_mask_bn3         (TimeDistributed)\nmrcnn_class_conv2      (TimeDistributed)\nmrcnn_class_bn2        (TimeDistributed)\nmrcnn_mask_conv4       (TimeDistributed)\nmrcnn_mask_bn4         (TimeDistributed)\nmrcnn_bbox_fc          (TimeDistributed)\nmrcnn_mask_deconv      (TimeDistributed)\nmrcnn_class_logits     (TimeDistributed)\nmrcnn_mask             (TimeDistributed)\n","output_type":"stream"},{"name":"stderr","text":"/opt/conda/lib/python3.6/site-packages/tensorflow/python/ops/gradients_impl.py:110: UserWarning: Converting sparse IndexedSlices to a dense Tensor of unknown shape. This may consume a large amount of memory.\n  \"Converting sparse IndexedSlices to a dense Tensor of unknown shape. \"\n/opt/conda/lib/python3.6/site-packages/keras/engine/training_generator.py:47: UserWarning: Using a generator with `use_multiprocessing=True` and multiple workers may duplicate your data. Please consider using the`keras.utils.Sequence class.\n  UserWarning('Using a generator with `use_multiprocessing=True`'\n","output_type":"stream"},{"name":"stdout","text":"Epoch 1/2\n100/100 [==============================] - 48s 479ms/step - loss: 4.2108 - rpn_class_loss: 0.0382 - rpn_bbox_loss: 2.4635 - mrcnn_class_loss: 0.2006 - mrcnn_bbox_loss: 0.8409 - mrcnn_mask_loss: 0.6676 - val_loss: 3.4203 - val_rpn_class_loss: 0.0171 - val_rpn_bbox_loss: 1.4135 - val_mrcnn_class_loss: 0.3339 - val_mrcnn_bbox_loss: 0.9736 - val_mrcnn_mask_loss: 0.6822\nEpoch 2/2\n100/100 [==============================] - 23s 232ms/step - loss: 3.4102 - rpn_class_loss: 0.0290 - rpn_bbox_loss: 1.8557 - mrcnn_class_loss: 0.2319 - mrcnn_bbox_loss: 0.6982 - mrcnn_mask_loss: 0.5955 - val_loss: 2.4893 - val_rpn_class_loss: 0.0178 - val_rpn_bbox_loss: 0.9987 - val_mrcnn_class_loss: 0.2208 - val_mrcnn_bbox_loss: 0.7214 - val_mrcnn_mask_loss: 0.5306\n","output_type":"stream"}],"execution_count":33},{"cell_type":"code","source":"# Cell 14: Training Stage 2 (All Layers, LR 1)\nif CAN_TRAIN:\n    print(\"\\nTraining all layers (stage 2)...\")\n    # Ensure model.epoch is correctly set from the previous stage\n    # model.train updates self.epoch internally to epochs (which is cumulative for Keras fit_generator)\n    next_stage_epochs = EPOCHS_FULL_TRAINING[0] + EPOCHS_FULL_TRAINING[1]\n    \n    model.train(dataset_train, dataset_val,\n                learning_rate=config.LEARNING_RATE,\n                epochs=next_stage_epochs, \n                \n                layers='all',\n                augmentation=augmentation)\n    if model.keras_model.history and model.keras_model.history.history:\n        new_hist = model.keras_model.history.history\n        for k in new_hist: \n            if k in history_accumulator: history_accumulator[k].extend(new_hist[k])\n            else: history_accumulator[k] = new_hist[k]\n    else:\n        print(\"No history from stage 2 training.\")\nelse:\n    print(\"Skipping stage 2 training.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:12:25.247212Z","iopub.execute_input":"2025-06-05T19:12:25.247576Z","iopub.status.idle":"2025-06-05T19:15:56.716139Z","shell.execute_reply.started":"2025-06-05T19:12:25.247504Z","shell.execute_reply":"2025-06-05T19:15:56.715254Z"}},"outputs":[{"name":"stdout","text":"\nTraining all layers (stage 2)...\n\nStarting at epoch 2. LR=0.001\n\nCheckpoint Path: /kaggle/working/deepfashion220250605T1910/mask_rcnn_deepfashion2_{epoch:04d}.h5\nSelecting layers to train\nconv1                  (Conv2D)\nbn_conv1               (BatchNorm)\nres2a_branch2a         (Conv2D)\nbn2a_branch2a          (BatchNorm)\nres2a_branch2b         (Conv2D)\nbn2a_branch2b          (BatchNorm)\nres2a_branch2c         (Conv2D)\nres2a_branch1          (Conv2D)\nbn2a_branch2c          (BatchNorm)\nbn2a_branch1           (BatchNorm)\nres2b_branch2a         (Conv2D)\nbn2b_branch2a          (BatchNorm)\nres2b_branch2b         (Conv2D)\nbn2b_branch2b          (BatchNorm)\nres2b_branch2c         (Conv2D)\nbn2b_branch2c          (BatchNorm)\nres2c_branch2a         (Conv2D)\nbn2c_branch2a          (BatchNorm)\nres2c_branch2b         (Conv2D)\nbn2c_branch2b          (BatchNorm)\nres2c_branch2c         (Conv2D)\nbn2c_branch2c          (BatchNorm)\nres3a_branch2a         (Conv2D)\nbn3a_branch2a          (BatchNorm)\nres3a_branch2b         (Conv2D)\nbn3a_branch2b          (BatchNorm)\nres3a_branch2c         (Conv2D)\nres3a_branch1          (Conv2D)\nbn3a_branch2c          (BatchNorm)\nbn3a_branch1           (BatchNorm)\nres3b_branch2a         (Conv2D)\nbn3b_branch2a          (BatchNorm)\nres3b_branch2b         (Conv2D)\nbn3b_branch2b          (BatchNorm)\nres3b_branch2c         (Conv2D)\nbn3b_branch2c          (BatchNorm)\nres3c_branch2a         (Conv2D)\nbn3c_branch2a          (BatchNorm)\nres3c_branch2b         (Conv2D)\nbn3c_branch2b          (BatchNorm)\nres3c_branch2c         (Conv2D)\nbn3c_branch2c          (BatchNorm)\nres3d_branch2a         (Conv2D)\nbn3d_branch2a          (BatchNorm)\nres3d_branch2b         (Conv2D)\nbn3d_branch2b          (BatchNorm)\nres3d_branch2c         (Conv2D)\nbn3d_branch2c          (BatchNorm)\nres4a_branch2a         (Conv2D)\nbn4a_branch2a          (BatchNorm)\nres4a_branch2b         (Conv2D)\nbn4a_branch2b          (BatchNorm)\nres4a_branch2c         (Conv2D)\nres4a_branch1          (Conv2D)\nbn4a_branch2c          (BatchNorm)\nbn4a_branch1           (BatchNorm)\nres4b_branch2a         (Conv2D)\nbn4b_branch2a          (BatchNorm)\nres4b_branch2b         (Conv2D)\nbn4b_branch2b          (BatchNorm)\nres4b_branch2c         (Conv2D)\nbn4b_branch2c          (BatchNorm)\nres4c_branch2a         (Conv2D)\nbn4c_branch2a          (BatchNorm)\nres4c_branch2b         (Conv2D)\nbn4c_branch2b          (BatchNorm)\nres4c_branch2c         (Conv2D)\nbn4c_branch2c          (BatchNorm)\nres4d_branch2a         (Conv2D)\nbn4d_branch2a          (BatchNorm)\nres4d_branch2b         (Conv2D)\nbn4d_branch2b          (BatchNorm)\nres4d_branch2c         (Conv2D)\nbn4d_branch2c          (BatchNorm)\nres4e_branch2a         (Conv2D)\nbn4e_branch2a          (BatchNorm)\nres4e_branch2b         (Conv2D)\nbn4e_branch2b          (BatchNorm)\nres4e_branch2c         (Conv2D)\nbn4e_branch2c          (BatchNorm)\nres4f_branch2a         (Conv2D)\nbn4f_branch2a          (BatchNorm)\nres4f_branch2b         (Conv2D)\nbn4f_branch2b          (BatchNorm)\nres4f_branch2c         (Conv2D)\nbn4f_branch2c          (BatchNorm)\nres5a_branch2a         (Conv2D)\nbn5a_branch2a          (BatchNorm)\nres5a_branch2b         (Conv2D)\nbn5a_branch2b          (BatchNorm)\nres5a_branch2c         (Conv2D)\nres5a_branch1          (Conv2D)\nbn5a_branch2c          (BatchNorm)\nbn5a_branch1           (BatchNorm)\nres5b_branch2a         (Conv2D)\nbn5b_branch2a          (BatchNorm)\nres5b_branch2b         (Conv2D)\nbn5b_branch2b          (BatchNorm)\nres5b_branch2c         (Conv2D)\nbn5b_branch2c          (BatchNorm)\nres5c_branch2a         (Conv2D)\nbn5c_branch2a          (BatchNorm)\nres5c_branch2b         (Conv2D)\nbn5c_branch2b          (BatchNorm)\nres5c_branch2c         (Conv2D)\nbn5c_branch2c          (BatchNorm)\nfpn_c5p5               (Conv2D)\nfpn_c4p4               (Conv2D)\nfpn_c3p3               (Conv2D)\nfpn_c2p2               (Conv2D)\nfpn_p5                 (Conv2D)\nfpn_p2                 (Conv2D)\nfpn_p3                 (Conv2D)\nfpn_p4                 (Conv2D)\nIn model:  rpn_model\n    rpn_conv_shared        (Conv2D)\n    rpn_class_raw          (Conv2D)\n    rpn_bbox_pred          (Conv2D)\nmrcnn_mask_conv1       (TimeDistributed)\nmrcnn_mask_bn1         (TimeDistributed)\nmrcnn_mask_conv2       (TimeDistributed)\nmrcnn_mask_bn2         (TimeDistributed)\nmrcnn_class_conv1      (TimeDistributed)\nmrcnn_class_bn1        (TimeDistributed)\nmrcnn_mask_conv3       (TimeDistributed)\nmrcnn_mask_bn3         (TimeDistributed)\nmrcnn_class_conv2      (TimeDistributed)\nmrcnn_class_bn2        (TimeDistributed)\nmrcnn_mask_conv4       (TimeDistributed)\nmrcnn_mask_bn4         (TimeDistributed)\nmrcnn_bbox_fc          (TimeDistributed)\nmrcnn_mask_deconv      (TimeDistributed)\nmrcnn_class_logits     (TimeDistributed)\nmrcnn_mask             (TimeDistributed)\n","output_type":"stream"},{"name":"stderr","text":"/opt/conda/lib/python3.6/site-packages/tensorflow/python/ops/gradients_impl.py:110: UserWarning: Converting sparse IndexedSlices to a dense Tensor of unknown shape. This may consume a large amount of memory.\n  \"Converting sparse IndexedSlices to a dense Tensor of unknown shape. \"\n/opt/conda/lib/python3.6/site-packages/keras/engine/training_generator.py:47: UserWarning: Using a generator with `use_multiprocessing=True` and multiple workers may duplicate your data. Please consider using the`keras.utils.Sequence class.\n  UserWarning('Using a generator with `use_multiprocessing=True`'\n","output_type":"stream"},{"name":"stdout","text":"Epoch 3/8\n100/100 [==============================] - 57s 571ms/step - loss: 2.7046 - rpn_class_loss: 0.0172 - rpn_bbox_loss: 1.4531 - mrcnn_class_loss: 0.2198 - mrcnn_bbox_loss: 0.5339 - mrcnn_mask_loss: 0.4805 - val_loss: 3.0046 - val_rpn_class_loss: 0.0157 - val_rpn_bbox_loss: 1.5433 - val_mrcnn_class_loss: 0.3415 - val_mrcnn_bbox_loss: 0.5295 - val_mrcnn_mask_loss: 0.5747\nEpoch 4/8\n100/100 [==============================] - 26s 259ms/step - loss: 2.3177 - rpn_class_loss: 0.0121 - rpn_bbox_loss: 1.2220 - mrcnn_class_loss: 0.2130 - mrcnn_bbox_loss: 0.4286 - mrcnn_mask_loss: 0.4420 - val_loss: 2.1509 - val_rpn_class_loss: 0.0086 - val_rpn_bbox_loss: 0.9801 - val_mrcnn_class_loss: 0.1655 - val_mrcnn_bbox_loss: 0.6412 - val_mrcnn_mask_loss: 0.3556\nEpoch 5/8\n100/100 [==============================] - 27s 267ms/step - loss: 2.0627 - rpn_class_loss: 0.0070 - rpn_bbox_loss: 0.9883 - mrcnn_class_loss: 0.1949 - mrcnn_bbox_loss: 0.4130 - mrcnn_mask_loss: 0.4595 - val_loss: 2.2356 - val_rpn_class_loss: 0.0108 - val_rpn_bbox_loss: 0.9358 - val_mrcnn_class_loss: 0.2959 - val_mrcnn_bbox_loss: 0.4798 - val_mrcnn_mask_loss: 0.5133\nEpoch 6/8\n100/100 [==============================] - 26s 258ms/step - loss: 1.8722 - rpn_class_loss: 0.0097 - rpn_bbox_loss: 0.7435 - mrcnn_class_loss: 0.2522 - mrcnn_bbox_loss: 0.4144 - mrcnn_mask_loss: 0.4525 - val_loss: 2.0583 - val_rpn_class_loss: 0.0073 - val_rpn_bbox_loss: 0.8268 - val_mrcnn_class_loss: 0.3403 - val_mrcnn_bbox_loss: 0.4808 - val_mrcnn_mask_loss: 0.4031\nEpoch 7/8\n100/100 [==============================] - 26s 265ms/step - loss: 1.6741 - rpn_class_loss: 0.0105 - rpn_bbox_loss: 0.8311 - mrcnn_class_loss: 0.1987 - mrcnn_bbox_loss: 0.3079 - mrcnn_mask_loss: 0.3259 - val_loss: 2.0822 - val_rpn_class_loss: 0.0091 - val_rpn_bbox_loss: 1.1736 - val_mrcnn_class_loss: 0.2398 - val_mrcnn_bbox_loss: 0.3184 - val_mrcnn_mask_loss: 0.3412\nEpoch 8/8\n100/100 [==============================] - 26s 261ms/step - loss: 1.4600 - rpn_class_loss: 0.0094 - rpn_bbox_loss: 0.6793 - mrcnn_class_loss: 0.1945 - mrcnn_bbox_loss: 0.2647 - mrcnn_mask_loss: 0.3120 - val_loss: 1.8628 - val_rpn_class_loss: 0.0056 - val_rpn_bbox_loss: 0.8108 - val_mrcnn_class_loss: 0.2577 - val_mrcnn_bbox_loss: 0.4392 - val_mrcnn_mask_loss: 0.3494\n","output_type":"stream"}],"execution_count":34},{"cell_type":"code","source":"# Cell 15: Training Stage 3 (All Layers, LR 2)\nif CAN_TRAIN:\n    print(\"\\nTraining all layers (stage 3)...\")\n    next_stage_epochs = EPOCHS_FULL_TRAINING[0] + EPOCHS_FULL_TRAINING[1] + EPOCHS_FULL_TRAINING[2]\n\n    model.train(dataset_train, dataset_val,\n                learning_rate=config.LEARNING_RATE / 5,\n                epochs=next_stage_epochs, \n                layers='all',\n                \n                augmentation=augmentation)\n    if model.keras_model.history and model.keras_model.history.history:\n        new_hist = model.keras_model.history.history\n        for k in new_hist: \n            if k in history_accumulator: history_accumulator[k].extend(new_hist[k])\n            else: history_accumulator[k] = new_hist[k]\n    else:\n        print(\"No history from stage 3 training.\")\nelse:\n    print(\"Skipping stage 3 training.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:15:56.718291Z","iopub.execute_input":"2025-06-05T19:15:56.71865Z","iopub.status.idle":"2025-06-05T19:20:25.692649Z","shell.execute_reply.started":"2025-06-05T19:15:56.718579Z","shell.execute_reply":"2025-06-05T19:20:25.691693Z"}},"outputs":[{"name":"stdout","text":"\nTraining all layers (stage 3)...\n\nStarting at epoch 8. LR=0.0002\n\nCheckpoint Path: /kaggle/working/deepfashion220250605T1910/mask_rcnn_deepfashion2_{epoch:04d}.h5\nSelecting layers to train\nconv1                  (Conv2D)\nbn_conv1               (BatchNorm)\nres2a_branch2a         (Conv2D)\nbn2a_branch2a          (BatchNorm)\nres2a_branch2b         (Conv2D)\nbn2a_branch2b          (BatchNorm)\nres2a_branch2c         (Conv2D)\nres2a_branch1          (Conv2D)\nbn2a_branch2c          (BatchNorm)\nbn2a_branch1           (BatchNorm)\nres2b_branch2a         (Conv2D)\nbn2b_branch2a          (BatchNorm)\nres2b_branch2b         (Conv2D)\nbn2b_branch2b          (BatchNorm)\nres2b_branch2c         (Conv2D)\nbn2b_branch2c          (BatchNorm)\nres2c_branch2a         (Conv2D)\nbn2c_branch2a          (BatchNorm)\nres2c_branch2b         (Conv2D)\nbn2c_branch2b          (BatchNorm)\nres2c_branch2c         (Conv2D)\nbn2c_branch2c          (BatchNorm)\nres3a_branch2a         (Conv2D)\nbn3a_branch2a          (BatchNorm)\nres3a_branch2b         (Conv2D)\nbn3a_branch2b          (BatchNorm)\nres3a_branch2c         (Conv2D)\nres3a_branch1          (Conv2D)\nbn3a_branch2c          (BatchNorm)\nbn3a_branch1           (BatchNorm)\nres3b_branch2a         (Conv2D)\nbn3b_branch2a          (BatchNorm)\nres3b_branch2b         (Conv2D)\nbn3b_branch2b          (BatchNorm)\nres3b_branch2c         (Conv2D)\nbn3b_branch2c          (BatchNorm)\nres3c_branch2a         (Conv2D)\nbn3c_branch2a          (BatchNorm)\nres3c_branch2b         (Conv2D)\nbn3c_branch2b          (BatchNorm)\nres3c_branch2c         (Conv2D)\nbn3c_branch2c          (BatchNorm)\nres3d_branch2a         (Conv2D)\nbn3d_branch2a          (BatchNorm)\nres3d_branch2b         (Conv2D)\nbn3d_branch2b          (BatchNorm)\nres3d_branch2c         (Conv2D)\nbn3d_branch2c          (BatchNorm)\nres4a_branch2a         (Conv2D)\nbn4a_branch2a          (BatchNorm)\nres4a_branch2b         (Conv2D)\nbn4a_branch2b          (BatchNorm)\nres4a_branch2c         (Conv2D)\nres4a_branch1          (Conv2D)\nbn4a_branch2c          (BatchNorm)\nbn4a_branch1           (BatchNorm)\nres4b_branch2a         (Conv2D)\nbn4b_branch2a          (BatchNorm)\nres4b_branch2b         (Conv2D)\nbn4b_branch2b          (BatchNorm)\nres4b_branch2c         (Conv2D)\nbn4b_branch2c          (BatchNorm)\nres4c_branch2a         (Conv2D)\nbn4c_branch2a          (BatchNorm)\nres4c_branch2b         (Conv2D)\nbn4c_branch2b          (BatchNorm)\nres4c_branch2c         (Conv2D)\nbn4c_branch2c          (BatchNorm)\nres4d_branch2a         (Conv2D)\nbn4d_branch2a          (BatchNorm)\nres4d_branch2b         (Conv2D)\nbn4d_branch2b          (BatchNorm)\nres4d_branch2c         (Conv2D)\nbn4d_branch2c          (BatchNorm)\nres4e_branch2a         (Conv2D)\nbn4e_branch2a          (BatchNorm)\nres4e_branch2b         (Conv2D)\nbn4e_branch2b          (BatchNorm)\nres4e_branch2c         (Conv2D)\nbn4e_branch2c          (BatchNorm)\nres4f_branch2a         (Conv2D)\nbn4f_branch2a          (BatchNorm)\nres4f_branch2b         (Conv2D)\nbn4f_branch2b          (BatchNorm)\nres4f_branch2c         (Conv2D)\nbn4f_branch2c          (BatchNorm)\nres5a_branch2a         (Conv2D)\nbn5a_branch2a          (BatchNorm)\nres5a_branch2b         (Conv2D)\nbn5a_branch2b          (BatchNorm)\nres5a_branch2c         (Conv2D)\nres5a_branch1          (Conv2D)\nbn5a_branch2c          (BatchNorm)\nbn5a_branch1           (BatchNorm)\nres5b_branch2a         (Conv2D)\nbn5b_branch2a          (BatchNorm)\nres5b_branch2b         (Conv2D)\nbn5b_branch2b          (BatchNorm)\nres5b_branch2c         (Conv2D)\nbn5b_branch2c          (BatchNorm)\nres5c_branch2a         (Conv2D)\nbn5c_branch2a          (BatchNorm)\nres5c_branch2b         (Conv2D)\nbn5c_branch2b          (BatchNorm)\nres5c_branch2c         (Conv2D)\nbn5c_branch2c          (BatchNorm)\nfpn_c5p5               (Conv2D)\nfpn_c4p4               (Conv2D)\nfpn_c3p3               (Conv2D)\nfpn_c2p2               (Conv2D)\nfpn_p5                 (Conv2D)\nfpn_p2                 (Conv2D)\nfpn_p3                 (Conv2D)\nfpn_p4                 (Conv2D)\nIn model:  rpn_model\n    rpn_conv_shared        (Conv2D)\n    rpn_class_raw          (Conv2D)\n    rpn_bbox_pred          (Conv2D)\nmrcnn_mask_conv1       (TimeDistributed)\nmrcnn_mask_bn1         (TimeDistributed)\nmrcnn_mask_conv2       (TimeDistributed)\nmrcnn_mask_bn2         (TimeDistributed)\nmrcnn_class_conv1      (TimeDistributed)\nmrcnn_class_bn1        (TimeDistributed)\nmrcnn_mask_conv3       (TimeDistributed)\nmrcnn_mask_bn3         (TimeDistributed)\nmrcnn_class_conv2      (TimeDistributed)\nmrcnn_class_bn2        (TimeDistributed)\nmrcnn_mask_conv4       (TimeDistributed)\nmrcnn_mask_bn4         (TimeDistributed)\nmrcnn_bbox_fc          (TimeDistributed)\nmrcnn_mask_deconv      (TimeDistributed)\nmrcnn_class_logits     (TimeDistributed)\nmrcnn_mask             (TimeDistributed)\n","output_type":"stream"},{"name":"stderr","text":"/opt/conda/lib/python3.6/site-packages/tensorflow/python/ops/gradients_impl.py:110: UserWarning: Converting sparse IndexedSlices to a dense Tensor of unknown shape. This may consume a large amount of memory.\n  \"Converting sparse IndexedSlices to a dense Tensor of unknown shape. \"\n/opt/conda/lib/python3.6/site-packages/keras/engine/training_generator.py:47: UserWarning: Using a generator with `use_multiprocessing=True` and multiple workers may duplicate your data. Please consider using the`keras.utils.Sequence class.\n  UserWarning('Using a generator with `use_multiprocessing=True`'\n","output_type":"stream"},{"name":"stdout","text":"Epoch 9/16\n100/100 [==============================] - 57s 573ms/step - loss: 1.7386 - rpn_class_loss: 0.0098 - rpn_bbox_loss: 0.9422 - mrcnn_class_loss: 0.1778 - mrcnn_bbox_loss: 0.2998 - mrcnn_mask_loss: 0.3089 - val_loss: 2.0861 - val_rpn_class_loss: 0.0105 - val_rpn_bbox_loss: 0.9987 - val_mrcnn_class_loss: 0.1994 - val_mrcnn_bbox_loss: 0.4800 - val_mrcnn_mask_loss: 0.3975\nEpoch 10/16\n100/100 [==============================] - 26s 261ms/step - loss: 1.4185 - rpn_class_loss: 0.0080 - rpn_bbox_loss: 0.7220 - mrcnn_class_loss: 0.1491 - mrcnn_bbox_loss: 0.2435 - mrcnn_mask_loss: 0.2960 - val_loss: 1.7172 - val_rpn_class_loss: 0.0119 - val_rpn_bbox_loss: 0.8053 - val_mrcnn_class_loss: 0.2008 - val_mrcnn_bbox_loss: 0.4044 - val_mrcnn_mask_loss: 0.2947\nEpoch 11/16\n100/100 [==============================] - 27s 267ms/step - loss: 1.2531 - rpn_class_loss: 0.0051 - rpn_bbox_loss: 0.5147 - mrcnn_class_loss: 0.1442 - mrcnn_bbox_loss: 0.2144 - mrcnn_mask_loss: 0.3747 - val_loss: 2.4785 - val_rpn_class_loss: 0.0061 - val_rpn_bbox_loss: 1.3221 - val_mrcnn_class_loss: 0.2052 - val_mrcnn_bbox_loss: 0.3794 - val_mrcnn_mask_loss: 0.5657\nEpoch 12/16\n100/100 [==============================] - 26s 262ms/step - loss: 1.2206 - rpn_class_loss: 0.0068 - rpn_bbox_loss: 0.4488 - mrcnn_class_loss: 0.1723 - mrcnn_bbox_loss: 0.2347 - mrcnn_mask_loss: 0.3580 - val_loss: 1.9062 - val_rpn_class_loss: 0.0060 - val_rpn_bbox_loss: 0.9285 - val_mrcnn_class_loss: 0.2909 - val_mrcnn_bbox_loss: 0.3389 - val_mrcnn_mask_loss: 0.3419\nEpoch 13/16\n100/100 [==============================] - 26s 262ms/step - loss: 0.9985 - rpn_class_loss: 0.0073 - rpn_bbox_loss: 0.4186 - mrcnn_class_loss: 0.1595 - mrcnn_bbox_loss: 0.1563 - mrcnn_mask_loss: 0.2567 - val_loss: 1.5705 - val_rpn_class_loss: 0.0070 - val_rpn_bbox_loss: 0.7229 - val_mrcnn_class_loss: 0.2021 - val_mrcnn_bbox_loss: 0.3268 - val_mrcnn_mask_loss: 0.3118\nEpoch 14/16\n100/100 [==============================] - 26s 258ms/step - loss: 0.8953 - rpn_class_loss: 0.0075 - rpn_bbox_loss: 0.3665 - mrcnn_class_loss: 0.1357 - mrcnn_bbox_loss: 0.1377 - mrcnn_mask_loss: 0.2478 - val_loss: 1.6798 - val_rpn_class_loss: 0.0049 - val_rpn_bbox_loss: 0.8067 - val_mrcnn_class_loss: 0.2376 - val_mrcnn_bbox_loss: 0.3177 - val_mrcnn_mask_loss: 0.3128\nEpoch 15/16\n100/100 [==============================] - 26s 258ms/step - loss: 1.3892 - rpn_class_loss: 0.0074 - rpn_bbox_loss: 0.5862 - mrcnn_class_loss: 0.1780 - mrcnn_bbox_loss: 0.2313 - mrcnn_mask_loss: 0.3863 - val_loss: 1.7911 - val_rpn_class_loss: 0.0047 - val_rpn_bbox_loss: 0.8728 - val_mrcnn_class_loss: 0.2710 - val_mrcnn_bbox_loss: 0.2829 - val_mrcnn_mask_loss: 0.3598\nEpoch 16/16\n100/100 [==============================] - 26s 260ms/step - loss: 1.3735 - rpn_class_loss: 0.0063 - rpn_bbox_loss: 0.6554 - mrcnn_class_loss: 0.1737 - mrcnn_bbox_loss: 0.2409 - mrcnn_mask_loss: 0.2973 - val_loss: 1.8328 - val_rpn_class_loss: 0.0056 - val_rpn_bbox_loss: 0.8342 - val_mrcnn_class_loss: 0.2859 - val_mrcnn_bbox_loss: 0.3700 - val_mrcnn_mask_loss: 0.3371\n","output_type":"stream"}],"execution_count":35},{"cell_type":"code","source":"# Cell 17: Select Best Epoch and Load Weights for Inference\nmodel_path_for_inference = ''\nif CAN_TRAIN and history_accumulator and 'val_loss' in history_accumulator and history_accumulator['val_loss']:\n    best_epoch_index = np.argmin(history_accumulator[\"val_loss\"])\n    best_epoch_number = best_epoch_index + 1 \n    print(f\"Best epoch based on val_loss: {best_epoch_number}\")\n    print(f\"Validation loss at best epoch: {history_accumulator['val_loss'][best_epoch_index]}\")\n    \n    model_path_glob_pattern = os.path.join(model.log_dir, f'mask_rcnn_{config.NAME.lower()}_{best_epoch_number:04d}.h5')\n    glob_list = glob.glob(model_path_glob_pattern)\n    if glob_list:\n        model_path_for_inference = glob_list[0]\n        print(f\"Model path for best epoch ({best_epoch_number}): {model_path_for_inference}\")\n    else:\n        print(f\"Could not find model for epoch {best_epoch_number} with pattern: {model_path_glob_pattern}\")\n        print(\"Attempting to load the last saved model instead.\")\n        model_path_for_inference = model.find_last()\n        if os.path.exists(model_path_for_inference):\n             print(f\"Using last model: {model_path_for_inference}\")\n        else:\n            print(\"No model files found.\")\n            model_path_for_inference = '' \nelif CAN_TRAIN:\n    print(\"Cannot determine best epoch. Attempting to load the last saved model.\")\n    model_path_for_inference = model.find_last()\n    if os.path.exists(model_path_for_inference):\n        print(f\"Using last model: {model_path_for_inference}\")\n    else:\n        print(\"No model files found.\")\n        model_path_for_inference = ''\nelse:\n    print(\"Training was skipped. No model to load for inference.\")\n\nclass InferenceConfig(FashionConfig):\n    GPU_COUNT = 1\n    IMAGES_PER_GPU = 1\ninference_config = InferenceConfig()\n\nmodel_inference = modellib.MaskRCNN(mode='inference', \n                                    config=inference_config,\n                                    model_dir=ROOT_DIR)\n\nif model_path_for_inference and os.path.exists(model_path_for_inference):\n    print(\"Loading weights for inference from: \", model_path_for_inference)\n    model_inference.load_weights(model_path_for_inference, by_name=True)\nelse:\n    print(\"No trained model weights loaded for inference. Model will use initial (COCO) weights if applicable.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:21:50.482921Z","iopub.execute_input":"2025-06-05T19:21:50.483163Z","iopub.status.idle":"2025-06-05T19:22:02.856683Z","shell.execute_reply.started":"2025-06-05T19:21:50.483125Z","shell.execute_reply":"2025-06-05T19:22:02.855927Z"}},"outputs":[{"name":"stdout","text":"Best epoch based on val_loss: 13\nValidation loss at best epoch: 1.5704970180988311\nModel path for best epoch (13): /kaggle/working/deepfashion220250605T1910/mask_rcnn_deepfashion2_0013.h5\nLoading weights for inference from:  /kaggle/working/deepfashion220250605T1910/mask_rcnn_deepfashion2_0013.h5\nRe-starting from epoch 13\n","output_type":"stream"}],"execution_count":43},{"cell_type":"code","source":"class InferenceConfig(FashionConfig):\n    GPU_COUNT = 1\n    IMAGES_PER_GPU = 1\n    DETECTION_MIN_CONFIDENCE = 0.0 # Try a lower value like 0.3 or even 0.1 for testing\ninference_config = InferenceConfig()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:20:37.439783Z","iopub.execute_input":"2025-06-05T19:20:37.440034Z","iopub.status.idle":"2025-06-05T19:20:37.44496Z","shell.execute_reply.started":"2025-06-05T19:20:37.439972Z","shell.execute_reply":"2025-06-05T19:20:37.443795Z"}},"outputs":[],"execution_count":37},{"cell_type":"code","source":"# Cell 18: Sample Prediction and Visualization\nif CAN_TRAIN and hasattr(dataset_val, 'image_ids') and dataset_val.image_ids.size > 0:\n    print(\"\\nRunning sample prediction on a validation image...\")\n\n    image_id_to_predict_idx = random.choice(dataset_val.image_ids)\n    image_info_to_predict = dataset_val.image_info[image_id_to_predict_idx]\n    image_path_to_predict = image_info_to_predict['path']\n\n    print(f\"Predicting on image_id (index): {image_id_to_predict_idx}, path: {image_path_to_predict}\")\n\n    original_image_np = dataset_val.load_image(image_id_to_predict_idx)\n\n    # The model.detect() method expects original images and handles molding/unmolding.\n    # Pass the original_image_np directly to detect().\n    results = model_inference.detect([original_image_np], verbose=0)\n    r = results[0]\n\n    class_names_for_display = ['BG'] + [ci['name'] for ci in dataset_val.class_info if ci['source'] == 'dataset']\n\n    visualize.display_instances(original_image_np, r['rois'], r['masks'], r['class_ids'],\n                                class_names_for_display, r['scores'],\n                                title=f\"Prediction for {os.path.basename(image_path_to_predict)}\",\n                                figsize=(12, 12))\n    plt.show()\n    \nelse:\n    print(\"Skipping sample prediction as validation dataset is empty or training was skipped.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"ROIs: {r['rois']}\")\nprint(f\"Class IDs: {r['class_ids']}\")\nprint(f\"Scores: {r['scores']}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-05T19:20:42.358762Z","iopub.execute_input":"2025-06-05T19:20:42.359023Z","iopub.status.idle":"2025-06-05T19:20:42.363905Z","shell.execute_reply.started":"2025-06-05T19:20:42.358952Z","shell.execute_reply":"2025-06-05T19:20:42.36309Z"}},"outputs":[{"name":"stdout","text":"ROIs: [[ 47  59 310 427]]\nClass IDs: [1]\nScores: [0.7492244]\n","output_type":"stream"}],"execution_count":39},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}