{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"},{"sourceId":448870,"sourceType":"datasetVersion","datasetId":204414},{"sourceId":8347168,"sourceType":"datasetVersion","datasetId":4958757},{"sourceId":28454,"sourceType":"modelInstanceVersion","modelInstanceId":23966}],"dockerImageVersionId":25160,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os \nimport sys\nimport random\nimport math\nimport numpy as np\nimport cv2\nimport matplotlib.pyplot as plt\nimport json\nimport pydicom\nfrom imgaug import augmenters as iaa\nfrom tqdm import tqdm\nimport pandas as pd \nimport glob\nimport keras\nimport tensorflow\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import precision_recall_fscore_support as prf\n","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:12:48.085235Z","iopub.execute_input":"2024-05-08T04:12:48.085561Z","iopub.status.idle":"2024-05-08T04:12:48.091536Z","shell.execute_reply.started":"2024-05-08T04:12:48.085497Z","shell.execute_reply":"2024-05-08T04:12:48.090381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install keras==2.1.3","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:12:54.170176Z","iopub.execute_input":"2024-05-08T04:12:54.170486Z","iopub.status.idle":"2024-05-08T04:13:00.720537Z","shell.execute_reply.started":"2024-05-08T04:12:54.17043Z","shell.execute_reply":"2024-05-08T04:13:00.719462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('keras:', keras.__version__)\nprint('tensorflow:',tensorflow.__version__)\n!python3 --version","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:20:25.220793Z","iopub.execute_input":"2024-05-08T01:20:25.221157Z","iopub.status.idle":"2024-05-08T01:20:26.242745Z","shell.execute_reply.started":"2024-05-08T01:20:25.221091Z","shell.execute_reply":"2024-05-08T01:20:26.241879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings \nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:20:28.010625Z","iopub.execute_input":"2024-05-08T01:20:28.010948Z","iopub.status.idle":"2024-05-08T01:20:28.016466Z","shell.execute_reply.started":"2024-05-08T01:20:28.010898Z","shell.execute_reply":"2024-05-08T01:20:28.015545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_DIR = '/kaggle/input'\n\n# Directory to save logs and trained model\nROOT_DIR = '/kaggle/working'","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:20:30.062234Z","iopub.execute_input":"2024-05-08T01:20:30.062636Z","iopub.status.idle":"2024-05-08T01:20:30.067349Z","shell.execute_reply.started":"2024-05-08T01:20:30.062549Z","shell.execute_reply":"2024-05-08T01:20:30.066107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!git clone https://www.github.com/matterport/Mask_RCNN.git\nos.chdir('Mask_RCNN')\n#!python setup.py -q install","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:20:32.219018Z","iopub.execute_input":"2024-05-08T01:20:32.219354Z","iopub.status.idle":"2024-05-08T01:20:38.255325Z","shell.execute_reply.started":"2024-05-08T01:20:32.219292Z","shell.execute_reply":"2024-05-08T01:20:38.254271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python /kaggle/working/Mask_RCNN/setup.py -q install","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:20:40.444721Z","iopub.execute_input":"2024-05-08T01:20:40.445111Z","iopub.status.idle":"2024-05-08T01:20:44.057798Z","shell.execute_reply.started":"2024-05-08T01:20:40.445016Z","shell.execute_reply":"2024-05-08T01:20:44.056947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import Mask RCNN\nsys.path.append(os.path.join(ROOT_DIR, 'Mask_RCNN'))  # To find local version of the library\nfrom mrcnn.config import Config\nfrom mrcnn import utils\nimport mrcnn.model as modellib\nfrom mrcnn import visualize\nfrom mrcnn.model import log","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:20:46.351093Z","iopub.execute_input":"2024-05-08T01:20:46.351417Z","iopub.status.idle":"2024-05-08T01:20:46.398155Z","shell.execute_reply.started":"2024-05-08T01:20:46.351362Z","shell.execute_reply":"2024-05-08T01:20:46.39733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dicom_dir = os.path.join(DATA_DIR, 'rsna-split/RSNA_Split/RSNA_Split/train/dcm')\nval_dicom_dir = os.path.join(DATA_DIR, 'rsna-split/RSNA_Split/RSNA_Split/val/dcm')\ntest_dicom_dir = os.path.join(DATA_DIR, 'rsna-split/RSNA_Split/RSNA_Split/test/dcm')","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:20:49.180276Z","iopub.execute_input":"2024-05-08T01:20:49.180572Z","iopub.status.idle":"2024-05-08T01:20:49.185004Z","shell.execute_reply.started":"2024-05-08T01:20:49.180528Z","shell.execute_reply":"2024-05-08T01:20:49.184104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Download COCO pre-trained weights\n!wget --quiet https://github.com/matterport/Mask_RCNN/releases/download/v2.0/mask_rcnn_coco.h5\n!ls -lh mask_rcnn_coco.h5\n\nCOCO_WEIGHTS_PATH = \"/kaggle/input/mask-rcnn-coco/mask_rcnn_coco.h5\"","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:20:50.860263Z","iopub.execute_input":"2024-05-08T01:20:50.860606Z","iopub.status.idle":"2024-05-08T01:20:54.885191Z","shell.execute_reply.started":"2024-05-08T01:20:50.860546Z","shell.execute_reply":"2024-05-08T01:20:54.883982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_dicom_fps(dicom_dir):\n    dicom_fps = glob.glob(dicom_dir+'/'+'*.dcm')\n    return list(set(dicom_fps))\n\ndef parse_dataset(dicom_dir, anns):\n    image_fps = get_dicom_fps(dicom_dir)\n    image_annotations = {fp: [] for fp in image_fps}\n    for index, row in anns.iterrows(): \n        fp = os.path.join(dicom_dir, row['patientId']+'.dcm')\n        image_annotations[fp].append(row)\n    return image_fps, image_annotations ","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:21:00.091292Z","iopub.execute_input":"2024-05-08T01:21:00.091624Z","iopub.status.idle":"2024-05-08T01:21:00.097623Z","shell.execute_reply.started":"2024-05-08T01:21:00.091572Z","shell.execute_reply":"2024-05-08T01:21:00.096497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hyper_paramters_comb={\n    'backbone':['resnet50'],\n    'learning_rate':[0.005],\n    'batch_size':[8],\n    'epochs':[10],\n    'det_min_conf':[0.9],\n    'det_nms_th':[0.8],\n    'rpn_nms_th':[0.7],\n    'steps_per_epoch':[250],\n    'layers': ['heads']\n}\n\nhpc=pd.DataFrame(hyper_paramters_comb)\n\nhpc['learning_rate'] = hpc['learning_rate'].astype(np.float32)\nhpc['det_min_conf'] = hpc['det_min_conf'].astype(np.float32)\nhpc['det_nms_th'] = hpc['det_nms_th'].astype(np.float32)\nhpc['rpn_nms_th'] = hpc['rpn_nms_th'].astype(np.float32)\n\nhpc.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:21:02.231985Z","iopub.execute_input":"2024-05-08T01:21:02.23237Z","iopub.status.idle":"2024-05-08T01:21:02.286601Z","shell.execute_reply.started":"2024-05-08T01:21:02.232309Z","shell.execute_reply":"2024-05-08T01:21:02.285722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training dataset\nanns = pd.read_csv(os.path.join(DATA_DIR, 'rsna-split/RSNA_Split/RSNA_Split/train/train_data.csv'))\nanns.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:21:07.425894Z","iopub.execute_input":"2024-05-08T01:21:07.426231Z","iopub.status.idle":"2024-05-08T01:21:07.485568Z","shell.execute_reply.started":"2024-05-08T01:21:07.426164Z","shell.execute_reply":"2024-05-08T01:21:07.48483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training dataset\nanns_val = pd.read_csv(os.path.join(DATA_DIR, 'rsna-split/RSNA_Split/RSNA_Split/val/val_data.csv'))\nanns_val.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:21:13.053096Z","iopub.execute_input":"2024-05-08T01:21:13.053392Z","iopub.status.idle":"2024-05-08T01:21:13.089851Z","shell.execute_reply.started":"2024-05-08T01:21:13.053347Z","shell.execute_reply":"2024-05-08T01:21:13.088979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training dataset\nanns_test = pd.read_csv(os.path.join(DATA_DIR, 'rsna-split/RSNA_Split/RSNA_Split/test/test_data.csv'))\nanns_test.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:21:17.619772Z","iopub.execute_input":"2024-05-08T01:21:17.620184Z","iopub.status.idle":"2024-05-08T01:21:17.676577Z","shell.execute_reply.started":"2024-05-08T01:21:17.620125Z","shell.execute_reply":"2024-05-08T01:21:17.675854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_fps, image_annotations = parse_dataset(train_dicom_dir, anns)\nimage_fps_valid, image_annotations_val = parse_dataset(val_dicom_dir, anns_val)\nimage_fps_testing, image_annotations_test = parse_dataset(test_dicom_dir, anns_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:21:20.391491Z","iopub.execute_input":"2024-05-08T01:21:20.391776Z","iopub.status.idle":"2024-05-08T01:21:24.147598Z","shell.execute_reply.started":"2024-05-08T01:21:20.391734Z","shell.execute_reply":"2024-05-08T01:21:24.146814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Original DICOM image size: 1024 x 1024\nORIG_SIZE = 1024","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:21:24.148885Z","iopub.execute_input":"2024-05-08T01:21:24.149147Z","iopub.status.idle":"2024-05-08T01:21:24.153168Z","shell.execute_reply.started":"2024-05-08T01:21:24.149097Z","shell.execute_reply":"2024-05-08T01:21:24.15233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Split the data into training and validation datasets that exctly macthes the dataset used for other models in the project**","metadata":{}},{"cell_type":"code","source":"def dataset_model():\n#     org_train, org_val = train_test_split(anns, test_size=0.30, random_state=32)\n    org_train=anns\n    org_val=anns_val\n    org_test=anns_test\n    #  taking subset of train and validation dataset\n    org_train_1=org_train[org_train.Target==1]\n    org_train_0=org_train[org_train.Target==0]\n    print(len(org_train_1),len(org_train_0))\n    image_fps_train=org_train_1.patientId[:].tolist() + org_train_0.patientId[:].tolist()\n    image_fps_train=[train_dicom_dir+'/'+x+'.dcm' for x in image_fps_train]\n\n    org_val_1=org_val[org_val.Target==1]\n    org_val_0=org_val[org_val.Target==0]\n    print(len(org_val_1),len(org_val_0))\n    image_fps_val=org_val_1.patientId[:].tolist() + org_val_0.patientId[:].tolist()\n    image_fps_val=[val_dicom_dir+'/'+x+'.dcm' for x in image_fps_val]\n    \n    org_test_1=org_test[org_test.Target==1]\n    org_test_0=org_test[org_test.Target==0]\n    image_fps_test=org_test_1.patientId[:].tolist() + org_test_0.patientId[:].tolist()\n    image_fps_test=[test_dicom_dir+'/'+x+'.dcm' for x in image_fps_test]\n    \n    \n    print(len(image_fps_train), len(image_fps_val), len(image_fps_test))\n\n    return image_fps_train, image_fps_val, image_fps_test","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:21:24.154319Z","iopub.execute_input":"2024-05-08T01:21:24.154564Z","iopub.status.idle":"2024-05-08T01:21:24.165482Z","shell.execute_reply.started":"2024-05-08T01:21:24.154506Z","shell.execute_reply":"2024-05-08T01:21:24.164676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_fps_train, image_fps_val, image_fps_test=dataset_model()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:21:26.103433Z","iopub.execute_input":"2024-05-08T01:21:26.103739Z","iopub.status.idle":"2024-05-08T01:21:26.134484Z","shell.execute_reply.started":"2024-05-08T01:21:26.103695Z","shell.execute_reply":"2024-05-08T01:21:26.133537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(image_fps_train), len(image_fps_val), len(image_fps_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:21:28.101783Z","iopub.execute_input":"2024-05-08T01:21:28.102173Z","iopub.status.idle":"2024-05-08T01:21:28.107815Z","shell.execute_reply.started":"2024-05-08T01:21:28.102104Z","shell.execute_reply":"2024-05-08T01:21:28.107108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DetectorDataset(utils.Dataset):\n    \"\"\"Dataset class for training pneumonia detection on the RSNA pneumonia dataset.\n    \"\"\"\n\n    def __init__(self, image_fps, image_annotations, orig_height, orig_width):\n        super().__init__(self)\n        \n        # Add classes\n        self.add_class('pneumonia', 1, 'Lung Opacity')\n        \n        # add images \n        for i, fp in enumerate(image_fps):\n            annotations = image_annotations[fp]\n            self.add_image('pneumonia', image_id=i, path=fp, \n                           annotations=annotations, orig_height=orig_height, orig_width=orig_width)\n            \n    def image_reference(self, image_id):\n        info = self.image_info[image_id]\n        return info['path']\n\n    def load_image(self, image_id):\n        info = self.image_info[image_id]\n        fp = info['path']\n        ds = pydicom.read_file(fp)\n        image = ds.pixel_array\n        # If grayscale. Convert to RGB for consistency.\n        if len(image.shape) != 3 or image.shape[2] != 3:\n            image = np.stack((image,) * 3, -1)\n        return image\n\n    def load_mask(self, image_id):\n        info = self.image_info[image_id]\n        annotations = info['annotations']\n        count = len(annotations)\n        if count == 0:\n            mask = np.zeros((info['orig_height'], info['orig_width'], 1), dtype=np.uint8)\n            class_ids = np.zeros((1,), dtype=np.int32)\n        else:\n            mask = np.zeros((info['orig_height'], info['orig_width'], count), dtype=np.uint8)\n            class_ids = np.zeros((count,), dtype=np.int32)\n            for i, a in enumerate(annotations):\n                if a['Target'] == 1:\n                    x = int(a['x'])\n                    y = int(a['y'])\n                    w = int(a['width'])\n                    h = int(a['height'])\n                    mask_instance = mask[:, :, i].copy()\n                    cv2.rectangle(mask_instance, (x, y), (x+w, y+h), 255, -1)\n                    mask[:, :, i] = mask_instance\n                    class_ids[i] = 1\n        return mask.astype(bool), class_ids.astype(np.int32)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:21:31.493047Z","iopub.execute_input":"2024-05-08T01:21:31.493408Z","iopub.status.idle":"2024-05-08T01:21:31.507781Z","shell.execute_reply.started":"2024-05-08T01:21:31.493347Z","shell.execute_reply":"2024-05-08T01:21:31.507072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare the training dataset\ndataset_train = DetectorDataset(image_fps_train, image_annotations, ORIG_SIZE, ORIG_SIZE)\ndataset_train.prepare()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:21:36.743824Z","iopub.execute_input":"2024-05-08T01:21:36.744162Z","iopub.status.idle":"2024-05-08T01:21:36.782162Z","shell.execute_reply.started":"2024-05-08T01:21:36.744109Z","shell.execute_reply":"2024-05-08T01:21:36.781181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare the validation dataset\ndataset_val = DetectorDataset(image_fps_val, image_annotations_val, ORIG_SIZE, ORIG_SIZE)\ndataset_val.prepare()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:21:38.456142Z","iopub.execute_input":"2024-05-08T01:21:38.456421Z","iopub.status.idle":"2024-05-08T01:21:38.469892Z","shell.execute_reply.started":"2024-05-08T01:21:38.45638Z","shell.execute_reply":"2024-05-08T01:21:38.469022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare the validation dataset\ndataset_test = DetectorDataset(image_fps_test, image_annotations_test, ORIG_SIZE, ORIG_SIZE)\ndataset_test.prepare()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:21:40.807986Z","iopub.execute_input":"2024-05-08T01:21:40.808329Z","iopub.status.idle":"2024-05-08T01:21:40.876174Z","shell.execute_reply.started":"2024-05-08T01:21:40.808278Z","shell.execute_reply":"2024-05-08T01:21:40.875323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Display a random image with bounding boxes**","metadata":{}},{"cell_type":"code","source":"# Load and display a random sample and their bounding boxes\n\nclass_ids = [0]\nwhile class_ids[0] == 0:  ## look for a mask\n    image_id = random.choice(dataset_train.image_ids)\n    image_fp = dataset_train.image_reference(image_id)\n    image = dataset_train.load_image(image_id)\n    mask, class_ids = dataset_train.load_mask(image_id)\n\nprint(image.shape)\n\nplt.figure(figsize=(10, 10))\nplt.subplot(1, 2, 1)\nplt.imshow(image)\nplt.axis('off')\n\nplt.subplot(1, 2, 2)\nmasked = np.zeros(image.shape[:2])\nfor i in range(mask.shape[2]):\n    masked += image[:, :, 0] * mask[:, :, i]\nplt.imshow(masked, cmap='gray')\nplt.axis('off')\n\n\nprint(image_fp)\nprint(class_ids)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:21:44.542313Z","iopub.execute_input":"2024-05-08T01:21:44.542627Z","iopub.status.idle":"2024-05-08T01:21:45.148133Z","shell.execute_reply.started":"2024-05-08T01:21:44.542579Z","shell.execute_reply":"2024-05-08T01:21:45.147386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image augmentation (light but constant)\naugmentation = iaa.Sequential([\n    iaa.OneOf([ ## geometric transform\n        iaa.Affine(\n            scale={\"x\": (0.98, 1.02), \"y\": (0.98, 1.04)},\n            translate_percent={\"x\": (-0.02, 0.02), \"y\": (-0.04, 0.04)},\n            rotate=(-2, 2),\n            shear=(-1, 1),\n        ),\n        iaa.PiecewiseAffine(scale=(0.001, 0.025)),\n    ]),\n    iaa.OneOf([ ## brightness or contrast\n        iaa.Multiply((0.9, 1.1)),\n        iaa.ContrastNormalization((0.9, 1.1)),\n    ]),\n    iaa.OneOf([ ## blur or sharpen\n        iaa.GaussianBlur(sigma=(0.0, 0.1)),\n        iaa.Sharpen(alpha=(0.0, 0.1)),\n    ]),\n])\n\n# test on the same image as above\nimggrid = augmentation.draw_grid(image[:, :, 0], cols=5, rows=2)\nplt.figure(figsize=(30, 12))\n_ = plt.imshow(imggrid[:, :, 0], cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:21:48.666157Z","iopub.execute_input":"2024-05-08T01:21:48.666462Z","iopub.status.idle":"2024-05-08T01:21:54.582488Z","shell.execute_reply.started":"2024-05-08T01:21:48.666415Z","shell.execute_reply":"2024-05-08T01:21:54.581397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The following parameters have been selected to reduce running time for demonstration purposes \n# These are not optimal \n\nclass DetectorConfig(Config):\n    \"\"\"Configuration for training pneumonia detection on the RSNA pneumonia dataset.\n    Overrides values in the base Config class.\n    \"\"\"\n    \n    # Give the configuration a recognizable name  \n    NAME = 'pneumonia'\n    \n    # Train on 1 GPU and 8 images per GPU. We can put multiple images on each\n    # GPU because the images are small. Batch size is 8 (GPUs * images/GPU).\n    GPU_COUNT = 1\n    IMAGES_PER_GPU = 8\n    \n    BACKBONE = hpc.iloc[0]['backbone']\n    BATCH_SIZE=hpc.iloc[0]['batch_size']\n    NUM_CLASSES = 2  # background + 1 pneumonia classes\n    \n    IMAGE_MIN_DIM = 256\n    IMAGE_MAX_DIM = 256\n    RPN_ANCHOR_SCALES = (32, 64, 128, 256)\n    TRAIN_ROIS_PER_IMAGE = 32\n    MAX_GT_INSTANCES = 3\n    DETECTION_MAX_INSTANCES = 3\n    DETECTION_MIN_CONFIDENCE = hpc.iloc[0]['det_min_conf']\n    DETECTION_NMS_THRESHOLD = hpc.iloc[0]['det_nms_th']\n    RPN_NMS_THRESHOLD = hpc.iloc[0]['rpn_nms_th']\n    STEPS_PER_EPOCH = hpc.iloc[0]['steps_per_epoch']\n","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:22:00.90951Z","iopub.execute_input":"2024-05-08T01:22:00.909815Z","iopub.status.idle":"2024-05-08T01:22:00.921339Z","shell.execute_reply.started":"2024-05-08T01:22:00.90977Z","shell.execute_reply":"2024-05-08T01:22:00.920383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config = DetectorConfig()\nconfig.display()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:22:03.441687Z","iopub.execute_input":"2024-05-08T01:22:03.441984Z","iopub.status.idle":"2024-05-08T01:22:03.450241Z","shell.execute_reply.started":"2024-05-08T01:22:03.441919Z","shell.execute_reply":"2024-05-08T01:22:03.448803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = modellib.MaskRCNN(mode='training', config=config, model_dir=ROOT_DIR)\n\n# Exclude the last layers because they require a matching number of classes\nmodel.load_weights(COCO_WEIGHTS_PATH, by_name=True, exclude=[\n    \"mrcnn_class_logits\", \"mrcnn_bbox_fc\", \"mrcnn_bbox\", \"mrcnn_mask\"])","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:22:41.599321Z","iopub.execute_input":"2024-05-08T01:22:41.599629Z","iopub.status.idle":"2024-05-08T01:22:56.765521Z","shell.execute_reply.started":"2024-05-08T01:22:41.599583Z","shell.execute_reply":"2024-05-08T01:22:56.764706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ncheckpoint_path = os.path.join(ROOT_DIR, \"mask_rcnn_{}_*epoch*.h5\".format(config.NAME.lower()))\ncheckpoint_path = checkpoint_path.replace(\"*epoch*\", \"{epoch:04d}\")\ncallbacks = [keras.callbacks.ModelCheckpoint(checkpoint_path,verbose=1, save_weights_only=True,period=1)]\n\nmodel.train(dataset_train, dataset_val, \n            learning_rate=hpc.iloc[0]['learning_rate'], \n            epochs=hpc.iloc[0]['epochs'], \n            custom_callbacks=callbacks,\n            layers=hpc.iloc[0]['layers'],\n            augmentation=augmentation)\n\nhistory = model.keras_model.history.history","metadata":{"execution":{"iopub.status.busy":"2024-05-08T01:22:59.104564Z","iopub.execute_input":"2024-05-08T01:22:59.104863Z","iopub.status.idle":"2024-05-08T03:46:08.800204Z","shell.execute_reply.started":"2024-05-08T01:22:59.104818Z","shell.execute_reply":"2024-05-08T03:46:08.796579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = range(1,len(next(iter(history.values())))+1)\npd.DataFrame(history, index=epochs)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T03:47:30.63559Z","iopub.execute_input":"2024-05-08T03:47:30.636527Z","iopub.status.idle":"2024-05-08T03:47:30.9321Z","shell.execute_reply.started":"2024-05-08T03:47:30.636072Z","shell.execute_reply":"2024-05-08T03:47:30.929511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,10))\nplt.subplot(111)\nplt.plot(epochs, history[\"loss\"], label=\"Train loss\")\nplt.plot(epochs, history[\"val_loss\"], label=\"Valid loss\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T03:47:33.32307Z","iopub.execute_input":"2024-05-08T03:47:33.323772Z","iopub.status.idle":"2024-05-08T03:47:34.795253Z","shell.execute_reply.started":"2024-05-08T03:47:33.323493Z","shell.execute_reply":"2024-05-08T03:47:34.791169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,10))\nplt.subplot(111)\nplt.plot(epochs, history[\"mrcnn_class_loss\"], label=\"Train class ce\")\nplt.plot(epochs, history[\"val_mrcnn_class_loss\"], label=\"Valid class ce\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T03:47:37.639746Z","iopub.execute_input":"2024-05-08T03:47:37.640379Z","iopub.status.idle":"2024-05-08T03:47:38.954682Z","shell.execute_reply.started":"2024-05-08T03:47:37.640292Z","shell.execute_reply":"2024-05-08T03:47:38.953434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,10))\nplt.subplot(111)\nplt.plot(epochs, history[\"mrcnn_bbox_loss\"], label=\"Train box loss\")\nplt.plot(epochs, history[\"val_mrcnn_bbox_loss\"], label=\"Valid box loss\")\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T03:47:45.742076Z","iopub.execute_input":"2024-05-08T03:47:45.742568Z","iopub.status.idle":"2024-05-08T03:47:47.377581Z","shell.execute_reply.started":"2024-05-08T03:47:45.742498Z","shell.execute_reply":"2024-05-08T03:47:47.366311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_epoch = np.argmin(history[\"val_loss\"])\nprint(\"Best Epoch:\", best_epoch + 1, history[\"val_loss\"][best_epoch])","metadata":{"execution":{"iopub.status.busy":"2024-05-08T03:47:53.504612Z","iopub.execute_input":"2024-05-08T03:47:53.505333Z","iopub.status.idle":"2024-05-08T03:47:53.512605Z","shell.execute_reply.started":"2024-05-08T03:47:53.504969Z","shell.execute_reply":"2024-05-08T03:47:53.511419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x=filter(lambda f : (f.startswith('mask_rcnn_pneumonia') and ((4-len(str(best_epoch+1)))*str(0)+str(best_epoch+1)) in f), os.listdir(model.model_dir))\nmodel_path=model.model_dir+'/'+list(x)[0]\nprint('Found model at {}'.format(model_path))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_path=\"/kaggle/working/mask_rcnn_pneumonia_0010.h5\"","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:13:46.139799Z","iopub.execute_input":"2024-05-08T04:13:46.140213Z","iopub.status.idle":"2024-05-08T04:13:46.14479Z","shell.execute_reply.started":"2024-05-08T04:13:46.140154Z","shell.execute_reply":"2024-05-08T04:13:46.143605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class InferenceConfig(DetectorConfig):\n    GPU_COUNT = 1\n    IMAGES_PER_GPU = 1\n\ninference_config = InferenceConfig()\n\n# Recreate the model in inference mode\nmodel = modellib.MaskRCNN(mode='inference', config=inference_config, model_dir=ROOT_DIR)\n\n# Load trained weights (fill in path to trained weights here)\nassert model_path != \"\", \"Provide path to trained weights\"\nprint(\"Loading weights from \", model_path)\nmodel.load_weights(model_path, by_name=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:13:50.786556Z","iopub.execute_input":"2024-05-08T04:13:50.786843Z","iopub.status.idle":"2024-05-08T04:14:00.867726Z","shell.execute_reply.started":"2024-05-08T04:13:50.786801Z","shell.execute_reply":"2024-05-08T04:14:00.866738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set color for class\ndef get_colors_for_class_ids(class_ids):\n    colors = []\n    for class_id in class_ids:\n        if class_id == 1:\n            colors.append((.941, .204, .204))\n    return colors","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:15:28.18206Z","iopub.execute_input":"2024-05-08T04:15:28.182387Z","iopub.status.idle":"2024-05-08T04:15:28.187079Z","shell.execute_reply.started":"2024-05-08T04:15:28.18234Z","shell.execute_reply":"2024-05-08T04:15:28.186053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**How does the predicted box compared to the expected value? Let's use the validation dataset to check.**","metadata":{}},{"cell_type":"code","source":"test_nolabel_dicom_dir = os.path.join(DATA_DIR, 'rsna-split/RSNA_Split/RSNA_Split/test_no_label/dcm')\n# Lấy danh sách tên tệp trong thư mục\nimage_fps_test = os.listdir(test_nolabel_dicom_dir)\n# In ra danh sách tên tệp với phần mở rộng .dcm\nprint(image_fps_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:09:55.066323Z","iopub.execute_input":"2024-05-08T04:09:55.066648Z","iopub.status.idle":"2024-05-08T04:09:55.074841Z","shell.execute_reply.started":"2024-05-08T04:09:55.066601Z","shell.execute_reply":"2024-05-08T04:09:55.074027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pydicom\n\ndef load_dicom_images(image_fps):\n    images = []\n    for fp in image_fps:\n        ds = pydicom.read_file(f'/kaggle/input/rsna-split/RSNA_Split/RSNA_Split/test_no_label/dcm/{fp}')\n        image = ds.pixel_array\n        # If grayscale, convert to RGB for consistency\n        if len(image.shape) != 3 or image.shape[2] != 3:\n            image = np.stack((image,) * 3, -1)\n        images.append(image)\n    return images\n\n# Giả sử image_fps là danh sách các đường dẫn tới các ảnh DICOM\nimages = load_dicom_images(image_fps_test)","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:12:03.881823Z","iopub.execute_input":"2024-05-08T04:12:03.882177Z","iopub.status.idle":"2024-05-08T04:12:04.109765Z","shell.execute_reply.started":"2024-05-08T04:12:03.882121Z","shell.execute_reply":"2024-05-08T04:12:04.10903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class UnlabeledDetectorDataset(utils.Dataset):\n    def __init__(self, image_fps, orig_height, orig_width):\n        super().__init__(self)\n        self.add_class('pneumonia', 1, 'Lung Opacity')\n        for i, fp in enumerate(image_fps):\n            self.add_image('pneumonia', image_id=i, path=fp, orig_height=orig_height, orig_width=orig_width)\n\n    def image_reference(self, image_id):\n        info = self.image_info[image_id]\n        return info['path']\n\n    def load_image(self, image_id):\n        info = self.image_info[image_id]\n        return images[image_id]\n\n    def load_mask(self, image_id):\n        info = self.image_info[image_id]\n        # Trả về mask và class_ids cho các mẫu không có nhãn\n        return np.zeros((info['orig_height'], info['orig_width'], 1), dtype=np.uint8), np.zeros((1,), dtype=np.int32)\n\n# Chuẩn bị dataset_test\ndataset_test = UnlabeledDetectorDataset(image_fps_test, ORIG_SIZE, ORIG_SIZE)\ndataset_test.prepare()","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:12:09.887885Z","iopub.execute_input":"2024-05-08T04:12:09.88819Z","iopub.status.idle":"2024-05-08T04:12:09.9152Z","shell.execute_reply.started":"2024-05-08T04:12:09.888143Z","shell.execute_reply":"2024-05-08T04:12:09.91444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(25, 30))\nypred = []\nfor i in range(len(dataset_test.image_ids)):\n    original_image = dataset_test.load_image(i)\n#     plt.subplot(6, 2, 2*i + 1)\n#     visualize.display_instances(original_image, gt_bbox, gt_mask, gt_class_id, \n#                                 dataset.class_names,\n#                                 colors=get_colors_for_class_ids(gt_class_id), ax=fig.axes[-\n    results = model.detect([original_image])\n    ypred.append(results[0]['rois'])\n    r = results[0]\n    plt.subplot(4, 3, i+1)                                                                              \n    visualize.display_instances(original_image, r['rois'], r['masks'], r['class_ids'], \n                                dataset_test.class_names, r['scores'], \n                                colors=get_colors_for_class_ids(r['class_ids']), ax=fig.axes[-1])   ","metadata":{"execution":{"iopub.status.busy":"2024-05-08T04:16:39.233192Z","iopub.execute_input":"2024-05-08T04:16:39.233482Z","iopub.status.idle":"2024-05-08T04:16:45.216146Z","shell.execute_reply.started":"2024-05-08T04:16:39.233438Z","shell.execute_reply":"2024-05-08T04:16:45.214976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ----------------------------------------------------------------------------------","metadata":{}},{"cell_type":"code","source":"test_nolabel_dicom_dir = os.path.join(DATA_DIR, 'rsna-split/RSNA_Split/RSNA_Split/test_no_label/dcm')\n_, _, test_nolabel_dicom_dir=dataset_model()\n\n# prepare the validation dataset\ndataset_test = DetectorDataset(test_nolabel_dicom_dir, image_annotations, ORIG_SIZE, ORIG_SIZE)\ndataset_test.prepare()","metadata":{"execution":{"iopub.status.busy":"2024-04-10T04:17:23.324215Z","iopub.execute_input":"2024-04-10T04:17:23.324503Z","iopub.status.idle":"2024-04-10T04:17:23.333773Z","shell.execute_reply.started":"2024-04-10T04:17:23.324449Z","shell.execute_reply":"2024-04-10T04:17:23.332944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images=dataset_test.image_ids\n\nytrue=[]\nypred=[]\n\nfor i in range(len(images)):\n        original_image,_, _, gt_bbox, gt_mask =\\\n            modellib.load_image_gt(dataset_test, inference_config, images[i], use_mini_mask=False)\n        ytrue.append(gt_bbox)\n        results=model.detect([original_image])\n        ypred.append(results[0]['rois'])","metadata":{"execution":{"iopub.status.busy":"2024-04-10T04:17:23.482816Z","iopub.execute_input":"2024-04-10T04:17:23.483247Z","iopub.status.idle":"2024-04-10T04:18:44.524687Z","shell.execute_reply.started":"2024-04-10T04:17:23.48316Z","shell.execute_reply":"2024-04-10T04:18:44.523407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show few example of ground truth vs. predictions on the test dataset \ndataset = dataset_test\nfig = plt.figure(figsize=(10, 30))\nimg_array=[]\nfor i in range(6):\n\n    image_id = random.choice(dataset.image_ids)\n    \n    original_image, image_meta, gt_class_id, gt_bbox, gt_mask =\\\n        modellib.load_image_gt(dataset_val, inference_config, \n                               image_id, use_mini_mask=False)\n    img_array.append(original_image)\n    print(original_image.shape)\n    plt.subplot(6, 2, 2*i + 1)\n    visualize.display_instances(original_image, gt_bbox, gt_mask, gt_class_id, \n                                dataset.class_names,\n                                colors=get_colors_for_class_ids(gt_class_id), ax=fig.axes[-1])\n    \n    plt.subplot(6, 2, 2*i + 2)\n    results = model.detect([original_image]) #, verbose=1)\n    r = results[0]\n    visualize.display_instances(original_image, r['rois'], r['masks'], r['class_ids'], \n                                dataset.class_names, r['scores'], \n                                colors=get_colors_for_class_ids(r['class_ids']), ax=fig.axes[-1])\n","metadata":{"execution":{"iopub.status.busy":"2024-04-10T04:21:12.423735Z","iopub.execute_input":"2024-04-10T04:21:12.424091Z","iopub.status.idle":"2024-04-10T04:21:14.994705Z","shell.execute_reply.started":"2024-04-10T04:21:12.424026Z","shell.execute_reply":"2024-04-10T04:21:14.993932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def loadmasks(bboxes):\n    mask = np.zeros((bboxes.shape[0], ORIG_SIZE, ORIG_SIZE))\n    for i in range(bboxes.shape[0]):\n        if bboxes[i]==[]:\n            continue\n        else:\n            for bbox in bboxes[i]:\n                x, y, w, h = bbox\n                x1=math.floor(x)\n                y1=math.floor(y)\n                x2=math.ceil(x+w)\n                y2=math.ceil(y+h)\n                mask[i, y1:y2,x1:x2] = 1\n\n    return mask","metadata":{"execution":{"iopub.status.busy":"2024-04-10T04:18:47.142192Z","iopub.execute_input":"2024-04-10T04:18:47.14245Z","iopub.status.idle":"2024-04-10T04:18:47.147996Z","shell.execute_reply.started":"2024-04-10T04:18:47.142399Z","shell.execute_reply":"2024-04-10T04:18:47.147103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def npmean_iou(bbox1, bbox2):\n    mask1=loadmasks(np.array(bbox1))\n    mask2=loadmasks(np.array(bbox2))\n    union=np.count_nonzero(mask1, 1).astype(np.float32) + np.count_nonzero(mask2, 1).astype(np.float32)\n    intersection = np.count_nonzero(np.logical_and(mask1, mask2), 1).astype(np.float32)\n    smooth = np.ones(intersection.shape)\n    iou = np.mean((intersection+smooth)/(union-intersection+smooth))\n    return iou\n\niou=npmean_iou(ytrue, ypred)\niou","metadata":{"execution":{"iopub.status.busy":"2024-04-10T04:18:47.149232Z","iopub.execute_input":"2024-04-10T04:18:47.149447Z","iopub.status.idle":"2024-04-10T04:18:53.889613Z","shell.execute_reply.started":"2024-04-10T04:18:47.149408Z","shell.execute_reply":"2024-04-10T04:18:53.888628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prec, rec, f1s, _ = prf([1 if np.sum(x)>1 else 0 for x in ytrue], [1 if np.sum(x)>1 else 0 for x in ypred], average='binary')\nprec, rec, f1s","metadata":{"execution":{"iopub.status.busy":"2024-04-09T20:13:16.817805Z","iopub.execute_input":"2024-04-09T20:13:16.818172Z","iopub.status.idle":"2024-04-09T20:13:16.833752Z","shell.execute_reply.started":"2024-04-09T20:13:16.818098Z","shell.execute_reply":"2024-04-09T20:13:16.832987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hpc.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-09T20:13:16.83536Z","iopub.execute_input":"2024-04-09T20:13:16.835685Z","iopub.status.idle":"2024-04-09T20:13:16.866251Z","shell.execute_reply.started":"2024-04-09T20:13:16.835611Z","shell.execute_reply":"2024-04-09T20:13:16.865544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rm -f mask_rcnn_pneumonia_*","metadata":{"execution":{"iopub.status.busy":"2024-04-09T20:13:16.867565Z","iopub.execute_input":"2024-04-09T20:13:16.867808Z","iopub.status.idle":"2024-04-09T20:13:16.873761Z","shell.execute_reply.started":"2024-04-09T20:13:16.867768Z","shell.execute_reply":"2024-04-09T20:13:16.872977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(keras.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-04-09T20:13:16.874983Z","iopub.execute_input":"2024-04-09T20:13:16.875267Z","iopub.status.idle":"2024-04-09T20:13:16.888536Z","shell.execute_reply.started":"2024-04-09T20:13:16.875214Z","shell.execute_reply":"2024-04-09T20:13:16.887614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}