{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"},{"sourceId":448870,"sourceType":"datasetVersion","datasetId":204414},{"sourceId":7606315,"sourceType":"datasetVersion","datasetId":4428574},{"sourceId":7606459,"sourceType":"datasetVersion","datasetId":4428672},{"sourceId":7606638,"sourceType":"datasetVersion","datasetId":4428808},{"sourceId":7607110,"sourceType":"datasetVersion","datasetId":4429156},{"sourceId":7607152,"sourceType":"datasetVersion","datasetId":4429183}],"dockerImageVersionId":27938,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os \nimport sys\n\nimport random\nimport math\nimport numpy as np\nimport cv2\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport pydicom\nfrom imgaug import augmenters as iaa\nfrom tqdm import tqdm\nimport pandas as pd \nimport glob \nimport keras\nfrom collections import Counter","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-11T18:34:18.989699Z","iopub.execute_input":"2024-02-11T18:34:18.990071Z","iopub.status.idle":"2024-02-11T18:34:18.996509Z","shell.execute_reply.started":"2024-02-11T18:34:18.990021Z","shell.execute_reply":"2024-02-11T18:34:18.995501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(os.listdir(\"/kaggle/input/mask-rcnn-coco\"))\nDATA_DIR = '/kaggle/input'\n# Directory to save logs and trained model\nROOT_DIR = '/kaggle/working'\n\n!pip install mask-rcnn-12rics\n!git clone https://github.com/matterport/Mask_RCNN\nos.chdir('Mask_RCNN')\n# !python setup.py install\n# # Import Mask RCNN\n# sys.path.append(os.path.join(ROOT_DIR, 'Mask_RCNN'))  # To find local version of the library\nfrom mrcnn.config import Config\nfrom mrcnn import utils\nimport mrcnn.model as modellib\nfrom mrcnn import visualize\nfrom mrcnn.model import log\n# os.chdir(ROOT_DIR)\ntrain_dicom_dir = os.path.join(DATA_DIR, 'rsna-pneumonia-detection-challenge/stage_2_train_images')\ntest_dicom_dir = os.path.join(DATA_DIR, 'rsna-pneumonia-detection-challenge/stage_2_test_images')","metadata":{"execution":{"iopub.status.busy":"2024-02-11T18:30:54.530355Z","iopub.execute_input":"2024-02-11T18:30:54.530631Z","iopub.status.idle":"2024-02-11T18:31:09.585904Z","shell.execute_reply.started":"2024-02-11T18:30:54.530569Z","shell.execute_reply":"2024-02-11T18:31:09.585116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Завантаження та візуалізація даних**","metadata":{}},{"cell_type":"code","source":"classes = pd.read_csv('/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv')\ntrain_label_df = pd.read_csv(\"/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv\")\nfinal_df = pd.concat([train_label_df, classes['class']], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T18:31:09.587395Z","iopub.execute_input":"2024-02-11T18:31:09.587647Z","iopub.status.idle":"2024-02-11T18:31:09.749206Z","shell.execute_reply.started":"2024-02-11T18:31:09.587597Z","shell.execute_reply":"2024-02-11T18:31:09.748530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_img_size = 256 \norigin_img_size = 1024\nscale_factor = train_img_size / origin_img_size\nisna_count = len(train_label_df[train_label_df.Target == 0])\ntrain_label_df = train_label_df[train_label_df.Target == 1]\ntrain_label_df.rename(columns={\"x\": \"X0\", \"y\": \"Y0\"}, inplace=True)\ntrain_label_df[\"X1\"] = train_label_df[\"X0\"] + train_label_df[\"width\"]\ntrain_label_df[\"Y1\"] = train_label_df[\"Y0\"] + train_label_df[\"height\"]\ntrain_label_df[[\"X0\", \"X1\", \"Y0\", \"Y1\"]] = train_label_df[[\"X0\", \"X1\", \"Y0\", \"Y1\"]] * scale_factor\ntrain_label_df[\"area\"] = train_label_df[\"width\"] * scale_factor * train_label_df[\"height\"] * scale_factor\ntrain_label_df.drop([\"width\", \"height\"], axis=1, inplace=True)\ntrain_label_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-11T18:31:09.750584Z","iopub.execute_input":"2024-02-11T18:31:09.750863Z","iopub.status.idle":"2024-02-11T18:31:09.926229Z","shell.execute_reply.started":"2024-02-11T18:31:09.750815Z","shell.execute_reply":"2024-02-11T18:31:09.925336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DataPreprocessing:\n    def __init__(self):\n        self.custom_fig = None\n        self.custom_ax = None\n        self.fig = None\n    \n    def build_percentage_bar(self, classes):\n        f, ax = plt.subplots(1, 1, figsize=(15, 5))\n        total = float(len(classes))\n        classes.groupby('class').size().plot.bar(color='skyblue')\n        for p in ax.patches:\n            height = p.get_height()\n            ax.text(p.get_x() + p.get_width() / 2., height + 3, f'{int(100 * height / total)} %', ha='center', bbox=dict(facecolor='red', alpha=0.5))\n        plt.show()\n        \n    def build_class_target(self, final_df):\n        self.custom_fig, self.custom_ax = plt.subplots(nrows=1, figsize=(12, 6))\n\n        custom_temp = final_df.groupby('Target')['class'].value_counts()\n        custom_data_target_class = pd.DataFrame(data={'Values': custom_temp.values}, index=custom_temp.index).reset_index()\n\n        sns.barplot(ax=self.custom_ax, x='Target', y='Values', hue='class', data=custom_data_target_class, palette='Set2')\n\n        plt.title('Клас і цільовий розподіл')\n        plt.show()\n    \n    def build_box_frequency(self, train_label_df, isna_count):\n        cnt = Counter(train_label_df.patientId)\n        sample_batch = [sample[0] + \".dcm\" for sample in cnt.most_common(2)]\n        counts = pd.Series(list(cnt.values())).value_counts()\n        counts[0] = isna_count\n\n        plt.figure(figsize=(10, 6))\n        plt.title(\"Частота обмежуючих рамок\")\n        plt.ylabel(\"Кількість\")\n        plt.xticks(counts.index)\n        plt.bar(counts.index, counts, color='skyblue')\n        plt.show()\n    \n    def build_area_analysis(self, train_label_df):\n        fig, axs = plt.subplots(1, 2, figsize=(18, 5))\n\n        # Гістограма\n        axs[0].hist(train_label_df[\"area\"], color='skyblue')\n        axs[0].set_xlabel(\"Площа\")\n        axs[0].set_ylabel(\"Кількість\")\n        axs[0].set_title(\"Гістограма площі\")\n\n        # Boxplot\n        axs[1].boxplot(train_label_df[\"area\"])\n        axs[1].set_ylabel(\"Площа\")\n        axs[1].set_title(\"Боксплот площі\")\n\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-11T18:31:09.929175Z","iopub.execute_input":"2024-02-11T18:31:09.929520Z","iopub.status.idle":"2024-02-11T18:31:09.951555Z","shell.execute_reply.started":"2024-02-11T18:31:09.929458Z","shell.execute_reply":"2024-02-11T18:31:09.950267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"graph_builder = DataPreprocessing()\n\ngraph_builder.build_class_target(final_df)\ngraph_builder.build_percentage_bar(classes)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T18:31:09.953501Z","iopub.execute_input":"2024-02-11T18:31:09.953883Z","iopub.status.idle":"2024-02-11T18:31:10.629840Z","shell.execute_reply.started":"2024-02-11T18:31:09.953823Z","shell.execute_reply":"2024-02-11T18:31:10.628589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"graph_builder.build_box_frequency(train_label_df, isna_count)\ngraph_builder.build_area_analysis(train_label_df)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T18:34:22.820423Z","iopub.execute_input":"2024-02-11T18:34:22.820789Z","iopub.status.idle":"2024-02-11T18:34:23.601332Z","shell.execute_reply.started":"2024-02-11T18:34:22.820725Z","shell.execute_reply":"2024-02-11T18:34:23.600062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  **Передобробка даних**","metadata":{}},{"cell_type":"code","source":"def get_dicom_fps(dicom_dir):\n    dicom_fps = glob.glob(dicom_dir+'/'+'*.dcm')\n    return list(set(dicom_fps))\n\ndef parse_dataset(dicom_dir, anns): \n    image_fps = get_dicom_fps(dicom_dir)\n    image_annotations = {fp: [] for fp in image_fps}\n    for index, row in anns.iterrows(): \n        fp = os.path.join(dicom_dir, row['patientId']+'.dcm')\n        image_annotations[fp].append(row)\n    return image_fps, image_annotations \n\n\nclass DetectorConfig(Config):\n    \"\"\"\n    Конфігурація основних параметрів\n    \"\"\" \n    NAME = 'pneumonia'\n    GPU_COUNT = 1\n    \n    BACKBONE = 'resnet50'\n    \n    NUM_CLASSES = 2  \n    \n    IMAGE_MIN_DIM = 256\n    IMAGE_MAX_DIM = 256\n    RPN_ANCHOR_SCALES = (16, 32, 64, 128)\n    TRAIN_ROIS_PER_IMAGE = 32\n    MAX_GT_INSTANCES = 4\n    DETECTION_MAX_INSTANCES = 3\n    DETECTION_MIN_CONFIDENCE = 0.78  \n    DETECTION_NMS_THRESHOLD = 0.01\n\n    STEPS_PER_EPOCH = 200    \n    \nconfig = DetectorConfig() \nconfig.display()","metadata":{"execution":{"iopub.status.busy":"2024-02-11T18:35:14.795415Z","iopub.execute_input":"2024-02-11T18:35:14.795752Z","iopub.status.idle":"2024-02-11T18:35:14.810278Z","shell.execute_reply.started":"2024-02-11T18:35:14.795703Z","shell.execute_reply":"2024-02-11T18:35:14.809472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DetectorDataset(utils.Dataset):\n    \"\"\"Dataset class for training pneumonia detection on the RSNA pneumonia dataset.\n    \"\"\"\n\n    def __init__(self, paths_to_images, annotations_per_image, image_height, image_width):\n        # Виклик конструктора базового класу\n        super().__init__()\n\n        # Додавання класу \"Легенева непрозорість\" до набору даних\n        self.add_class('pneumonia', 1, 'Lung Opacity')\n\n        # Додавання інформації про зображення в набір даних\n        for index, image_path in enumerate(paths_to_images):\n            # Отримання анотацій для конкретного зображення\n            current_annotations = annotations_per_image[image_path]\n            # Додавання зображення та його анотацій в набір даних\n            self.add_image(\n                'pneumonia',  # Джерело даних\n                image_id=index, \n                path=image_path,  \n                annotations=current_annotations,  \n                orig_height=image_height,  \n                orig_width=image_width \n            )\n\n    def image_reference(self, image_id):\n        image_info = self.image_info[image_id]\n        # Повертаємо шлях до зображення\n        return image_info['path']\n\n    def load_image(self, image_id):\n        # Отримання інформації про зображення за ID\n        image_info = self.image_info[image_id]\n        # Шлях до файлу зображення\n        image_path = image_info['path']\n        # Завантаження DICOM файла\n        dicom_file = pydicom.read_file(image_path)\n        # Перетворення DICOM у numpy масив\n        image_array = dicom_file.pixel_array\n        # Якщо зображення в градаціях сірого, конвертуємо у RGB для уніфікації\n        if len(image_array.shape) != 3 or image_array.shape[2] != 3:\n            image_array = np.stack((image_array,) * 3, axis=-1)\n        # Повертаємо зображення\n        return image_array\n\n    def load_mask(self, image_id):\n        # Отримання детальної інформації про зображення за його ідентифікатором\n        image_details = self.image_info[image_id]\n        # Витягуємо анотації, пов'язані з цим зображенням\n        image_annotations = image_details['annotations']\n        # Кількість анотацій\n        annotations_count = len(image_annotations)\n        # Ініціалізація маски та масиву ідентифікаторів класів\n        if annotations_count == 0:\n            # Створення порожньої маски, якщо анотацій немає\n            mask = np.zeros((image_details['orig_height'], image_details['orig_width'], 1), dtype=np.uint8)\n            class_ids = np.zeros((1,), dtype=np.int32)\n        else:\n            # Створення маски з анотаціями\n            mask = np.zeros((image_details['orig_height'], image_details['orig_width'], annotations_count), dtype=np.uint8)\n            class_ids = np.zeros((annotations_count,), dtype=np.int32)\n            for idx, annotation in enumerate(image_annotations):\n                if annotation['Target'] == 1:\n                    # Координати об'єкта анотації\n                    rect_start_x = int(annotation['x'])\n                    rect_start_y = int(annotation['y'])\n                    rect_width = int(annotation['width'])\n                    rect_height = int(annotation['height'])\n                    # Створення індивідуальної маски для анотації\n                    individual_mask = mask[:, :, idx].copy()\n                    cv2.rectangle(individual_mask, (rect_start_x, rect_start_y), (rect_start_x + rect_width, rect_start_y + rect_height), 255, -1)\n                    mask[:, :, idx] = individual_mask\n                    class_ids[idx] = 1\n        # Конвертація типу маски для використання у подальшій обробці\n        return mask.astype(np.bool_), class_ids.astype(np.int32)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T18:35:15.626505Z","iopub.execute_input":"2024-02-11T18:35:15.626832Z","iopub.status.idle":"2024-02-11T18:35:15.650820Z","shell.execute_reply.started":"2024-02-11T18:35:15.626775Z","shell.execute_reply":"2024-02-11T18:35:15.649924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# тренувальний набір\nanns = pd.read_csv(os.path.join(DATA_DIR, 'rsna-pneumonia-detection-challenge/stage_2_train_labels.csv'))\nimage_fps, image_annotations = parse_dataset(train_dicom_dir, anns=anns)\nanns.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-11T18:35:16.165059Z","iopub.execute_input":"2024-02-11T18:35:16.165373Z","iopub.status.idle":"2024-02-11T18:35:19.970288Z","shell.execute_reply.started":"2024-02-11T18:35:16.165327Z","shell.execute_reply":"2024-02-11T18:35:19.969387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Оригінальні DICOM картинки розміру: 1024 x 1024\nORIG_SIZE = 1024\nds = pydicom.read_file(image_fps[0]) \nimage = ds.pixel_array \nds","metadata":{"execution":{"iopub.status.busy":"2024-02-11T18:35:19.972544Z","iopub.execute_input":"2024-02-11T18:35:19.972904Z","iopub.status.idle":"2024-02-11T18:35:20.013491Z","shell.execute_reply.started":"2024-02-11T18:35:19.972843Z","shell.execute_reply":"2024-02-11T18:35:20.012733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"######################################################################\n# Модифікуйте цей рядок, щоб використовувати більше або менше зображень для тренування/валідації.\n# Щоб використати всі зображення, використовуйте: image_fps_list = list(image_fps)\n#image_fps_list = list(image_fps[:1000]) # Для прикладу, якщо потрібно обмежити кількість до 1000 зображень\nimage_fps_list = list(image_fps)\n#####################################################################\n\n# Розділення набору даних на тренувальний та валідаційний\n# Співвідношення розділення встановлене як 0.9 проти 0.1 \n# 0.8 проти 0.2 # для альтернативного варіанта розділення\nsorted(image_fps_list)  # Сортування списку зображень\nrandom.seed(42) \nrandom.shuffle(image_fps_list)  \n\nvalidation_split = 0.1 \n#validation_split = 0.2  \nsplit_index = int((1 - validation_split) * len(image_fps_list))  # Обчислення індексу, за яким відбудеться розділення\n\nimage_fps_train = image_fps_list[:split_index]  \nimage_fps_val = image_fps_list[split_index:]  \n\nprint(len(image_fps_train), len(image_fps_val))  ","metadata":{"execution":{"iopub.status.busy":"2024-02-11T18:35:20.015019Z","iopub.execute_input":"2024-02-11T18:35:20.015264Z","iopub.status.idle":"2024-02-11T18:35:20.080865Z","shell.execute_reply.started":"2024-02-11T18:35:20.015224Z","shell.execute_reply":"2024-02-11T18:35:20.079935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# підготовка тренувального набору\ndataset_train = DetectorDataset(image_fps_train, image_annotations, ORIG_SIZE, ORIG_SIZE)\ndataset_train.prepare()","metadata":{"execution":{"iopub.status.busy":"2024-02-11T18:35:20.082552Z","iopub.execute_input":"2024-02-11T18:35:20.082920Z","iopub.status.idle":"2024-02-11T18:35:20.169260Z","shell.execute_reply.started":"2024-02-11T18:35:20.082859Z","shell.execute_reply":"2024-02-11T18:35:20.168512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# вивід прикладів анотацій\n#test_fp = random.choice(image_fps_train) #xhb 20190525\ntest_fp='ef9fb572-2914-4d16-982a-59eb99f5567b.dcm'\nimage_annotations[DATA_DIR+'/rsna-pneumonia-detection-challenge/stage_2_train_images/'+test_fp]","metadata":{"execution":{"iopub.status.busy":"2024-02-11T18:35:20.171848Z","iopub.execute_input":"2024-02-11T18:35:20.172327Z","iopub.status.idle":"2024-02-11T18:35:20.179506Z","shell.execute_reply.started":"2024-02-11T18:35:20.172167Z","shell.execute_reply":"2024-02-11T18:35:20.178506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# підготовка валідаційного набору\ndataset_val = DetectorDataset(image_fps_val, image_annotations, ORIG_SIZE, ORIG_SIZE)\ndataset_val.prepare()","metadata":{"execution":{"iopub.status.busy":"2024-02-11T18:35:20.181370Z","iopub.execute_input":"2024-02-11T18:35:20.181725Z","iopub.status.idle":"2024-02-11T18:35:20.198985Z","shell.execute_reply.started":"2024-02-11T18:35:20.181662Z","shell.execute_reply":"2024-02-11T18:35:20.197995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Іллюстр\n# # Suggestion: Run this a few times to see different examples. \n\n# image_id = random.choice(dataset_train.image_ids)\n# image_fp = dataset_train.image_reference(image_id)\n# image = dataset_train.load_image(image_id)\n# mask, class_ids = dataset_train.load_mask(image_id)\n\n# print(image.shape)\n\n# plt.figure(figsize=(10, 10))\n# plt.subplot(1, 2, 1)\n# plt.imshow(image[:, :, 0], cmap='gray')\n# plt.axis('off')\n# plt.subplot(1, 2, 2)\n# masked = np.zeros(image.shape[:2])\n# for i in range(mask.shape[2]):\n#     masked += image[:, :, 0] * mask[:, :, i]\n# plt.imshow(masked, cmap='gray')\n# plt.axis('off')\n\n# print(image_fp)\n# print(class_ids)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T18:35:21.181330Z","iopub.execute_input":"2024-02-11T18:35:21.181676Z","iopub.status.idle":"2024-02-11T18:35:21.186492Z","shell.execute_reply.started":"2024-02-11T18:35:21.181616Z","shell.execute_reply":"2024-02-11T18:35:21.185403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Тренування моделі**\nЗавантаження моделі з попередньо визначеними вагами та встановлення зворотніх викликів для зберігання моделі.","metadata":{}},{"cell_type":"code","source":"\n\n# Аугментація зображень (вирівненя контрастності)\naugmentation = iaa.Sequential([\n    iaa.OneOf([ ## геометрична трасфрпмація\n        iaa.Affine(\n            scale={\"x\": (0.98, 1.02), \"y\": (0.98, 1.04)},\n            translate_percent={\"x\": (-0.02, 0.02), \"y\": (-0.04, 0.04)},\n            rotate=(-2, 2),\n            shear=(-1, 1),\n        ),\n        iaa.PiecewiseAffine(scale=(0.001, 0.025)),\n    ]),\n    iaa.OneOf([ ## яскравість та кон\n        iaa.Multiply((0.9, 1.1)),\n        iaa.ContrastNormalization((0.9, 1.1)),\n    ]),\n    iaa.OneOf([ \n        iaa.GaussianBlur(sigma=(0.0, 0.1)),\n        iaa.Sharpen(alpha=(0.0, 0.1)),\n    ]),\n])","metadata":{"execution":{"iopub.status.busy":"2024-02-11T18:35:25.246334Z","iopub.execute_input":"2024-02-11T18:35:25.246689Z","iopub.status.idle":"2024-02-11T18:35:25.260532Z","shell.execute_reply.started":"2024-02-11T18:35:25.246626Z","shell.execute_reply":"2024-02-11T18:35:25.259606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = modellib.MaskRCNN(mode='training', config=config, model_dir=ROOT_DIR)\nprint(model.model_dir)\nNUM_EPOCHS =100 # 2, для тесту\nCOCO_WEIGHTS_PATH = \"/kaggle/input/mask-rcnn-coco/mask_rcnn_coco.h5\"\nmodel.load_weights(COCO_WEIGHTS_PATH, by_name=True, exclude=[\n    \"mrcnn_class_logits\", \"mrcnn_bbox_fc\",\n    \"mrcnn_bbox\", \"mrcnn_mask\"])\n\n# Тренування моделі Mask-RCNN  \nimport warnings \nwarnings.filterwarnings(\"ignore\")\n\n# Callbacks\ncheckpoint_path = os.path.join(ROOT_DIR, \"mask_rcnn_{}_*epoch*.h5\".format(config.NAME.lower()))\ncheckpoint_path = checkpoint_path.replace(\"*epoch*\", \"{epoch:04d}\")\ncallbacks = [keras.callbacks.ModelCheckpoint(checkpoint_path,verbose=0, save_weights_only=True,period=5)]\n    \nmodel.train(dataset_train, dataset_val, \n            learning_rate=config.LEARNING_RATE, \n            epochs=NUM_EPOCHS, \n            custom_callbacks=callbacks,\n            layers='all',\n            augmentation=augmentation\n           )","metadata":{"execution":{"iopub.status.busy":"2024-02-11T18:35:25.904672Z","iopub.execute_input":"2024-02-11T18:35:25.905046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Вибір найкращих вагівмоделі зі збережених.**","metadata":{}},{"cell_type":"code","source":"# Вибір навченої моделі\n# Отримання списку директорій моделі\n#dir_names = next(os.walk(model.model_dir))[1]\ndir_names=[]\nfor root, dirs, files in os.walk(model.model_dir):\n    for name in files:\n        dir_names.append(name)\n# key = config.NAME.lower()\n\n# Ключ для пошуку моделей за назвою\nkey='mask_rcnn'\n# print('key is:',key)\n# print(dir_names)\n\n# Фільтрація та сортування директорій за ключем\ndir_names = filter(lambda f: f.startswith(key), dir_names)\ndir_names = sorted(dir_names)\n# print(dir_names)\n\n# Перевірка наявності директорій моделі\n# if not dir_names:\n#     import errno\n#     raise FileNotFoundError(\n#         errno.ENOENT,\n#         \"Could not find model directory under {}\".format(model.model_dir))\n \n# fps = []\n# # Pick last directory\n# for d in dir_names: \n#     dir_name = os.path.join(model.model_dir, d)\n#     # Find the last checkpoint\n#     checkpoints = next(os.walk(dir_name))[2]\n#     checkpoints = filter(lambda f: f.startswith(\"mask_rcnn\"), checkpoints)\n#     checkpoints = sorted(checkpoints)\n#     if not checkpoints:\n#         print('No weight files in {}'.format(dir_name))\n#     else:\n#         checkpoint = os.path.join(dir_name, checkpoints[-1])\n#         fps.append(checkpoint)\n\n# model_path = sorted(fps)[-1]\n\n#!!!!!! розкоментувати при тренуванні моделі       \nmodel_path=dir_names[-1]\nprint('Found model {}'.format(model_path))\nprint(model_path)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T16:07:50.555893Z","iopub.execute_input":"2024-02-11T16:07:50.556219Z","iopub.status.idle":"2024-02-11T16:07:50.567156Z","shell.execute_reply.started":"2024-02-11T16:07:50.556157Z","shell.execute_reply":"2024-02-11T16:07:50.566422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class InferenceConfig(DetectorConfig):\n    GPU_COUNT = 1\n    IMAGES_PER_GPU = 1\n\ninference_config = InferenceConfig()\n\n# перестворення моделі\nmodel = modellib.MaskRCNN(mode='inference', \n                          config=inference_config,\n                          model_dir=ROOT_DIR)\n\n#завантаження заздалегідь натренованої моделі, якщо закоментувати model_path, то завантажиться поточна модель\nmodel_path = \"/kaggle/input/model-0090/mask_rcnn_pneumonia_0090.h5\"\nprint(\"Loading weights from \", model_path)\nmodel.load_weights(model_path, by_name=True)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T16:07:50.568636Z","iopub.execute_input":"2024-02-11T16:07:50.568903Z","iopub.status.idle":"2024-02-11T16:08:00.710777Z","shell.execute_reply.started":"2024-02-11T16:07:50.568852Z","shell.execute_reply":"2024-02-11T16:08:00.709744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Приклади роботи на валідаційних наборах.**","metadata":{}},{"cell_type":"code","source":"def get_colors_for_class_ids(class_ids):\n    colors = []\n    for class_id in class_ids:\n        if class_id == 1:\n            colors.append((.941, .204, .204))\n    return colors\ndataset = dataset_val\nfig = plt.figure(figsize=(10, 30))\n\nfor i in range(4):\n\n    image_id = random.choice(dataset.image_ids)\n    \n    original_image, image_meta, gt_class_id, gt_bbox, gt_mask =\\\n        modellib.load_image_gt(dataset_val, inference_config, \n                               image_id, use_mini_mask=False)\n        \n    plt.subplot(6, 2, 2*i + 1)\n    visualize.display_instances(original_image, gt_bbox, gt_mask, gt_class_id, \n                                dataset.class_names,\n                                colors=get_colors_for_class_ids(gt_class_id), ax=fig.axes[-1])\n    \n    plt.subplot(6, 2, 2*i + 2)\n    results = model.detect([original_image]) #, verbose=1)\n    r = results[0]\n    visualize.display_instances(original_image, r['rois'], r['masks'], r['class_ids'], \n                                dataset.class_names, r['scores'], \n                                colors=get_colors_for_class_ids(r['class_ids']), ax=fig.axes[-1])","metadata":{"execution":{"iopub.status.busy":"2024-02-11T16:08:00.712416Z","iopub.execute_input":"2024-02-11T16:08:00.712676Z","iopub.status.idle":"2024-02-11T16:08:04.940905Z","shell.execute_reply.started":"2024-02-11T16:08:00.712635Z","shell.execute_reply":"2024-02-11T16:08:04.940080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get filenames of test dataset DICOM images\ntest_image_fps = get_dicom_fps(test_dicom_dir)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T16:08:04.942417Z","iopub.execute_input":"2024-02-11T16:08:04.942761Z","iopub.status.idle":"2024-02-11T16:08:04.964748Z","shell.execute_reply.started":"2024-02-11T16:08:04.942701Z","shell.execute_reply":"2024-02-11T16:08:04.964035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Визначення метаданих та записування результатів для кожного зображення у CSV.","metadata":{}},{"cell_type":"code","source":"# Зробити прогнози на тестових зображеннях, записати зразок подання\ndef predict(image_fps, filepath=ROOT_DIR+'/submission.csv', min_conf=0.98): \n    # припускаємо квадратне зображення   \n    with open(filepath, 'w') as file:\n        file.write('patientId,PredictionString'+\"\\n\")        \n        for image_id in tqdm(image_fps): \n            ds = pydicom.read_file(image_id)\n            image = ds.pixel_array\n        \n            # Якщо в чорно-білому. Конвертуємо в RGB для послідовності\n            if len(image.shape) != 3 or image.shape[2] != 3:\n                image = np.stack((image,) * 3, -1) \n            \n            patient_id = os.path.splitext(os.path.basename(image_id))[0]\n\n            results = model.detect([image])\n            r = results[0]\n\n            out_str = \"\"\n            out_str += patient_id \n            assert( len(r['rois']) == len(r['class_ids']) == len(r['scores']) )\n            if len(r['rois']) == 0:\n                out_str += \",\"\n            else: \n                num_instances = len(r['rois'])\n                out_str += \",\"\n                for i in range(num_instances): \n                    if r['scores'][i] > min_conf: \n                        out_str += ' '\n                        out_str += str(round(r['scores'][i], 2))\n                        out_str += ' '\n\n                        # x1, y1, ширина, висота \n                        x1 = r['rois'][i][1]\n                        y1 = r['rois'][i][0]\n                        width = r['rois'][i][3] - x1 \n                        height = r['rois'][i][2] - y1 \n                        bboxes_str = \"{} {} {} {}\".format(x1, y1, \\\n                                                          width, height)    \n                        out_str += bboxes_str\n            file.write(out_str+\"\\n\")","metadata":{"execution":{"iopub.status.busy":"2024-02-11T16:08:04.965876Z","iopub.execute_input":"2024-02-11T16:08:04.966096Z","iopub.status.idle":"2024-02-11T16:08:04.980881Z","shell.execute_reply.started":"2024-02-11T16:08:04.966058Z","shell.execute_reply":"2024-02-11T16:08:04.980134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_fp = ROOT_DIR+'/submission.csv'\npredict(test_image_fps, filepath=sample_submission_fp)\n#output = pd.read_csv(sample_submission_fp)\n# output.head(50)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T16:08:04.982049Z","iopub.execute_input":"2024-02-11T16:08:04.982277Z","iopub.status.idle":"2024-02-11T16:15:35.219146Z","shell.execute_reply.started":"2024-02-11T16:08:04.982228Z","shell.execute_reply":"2024-02-11T16:15:35.217679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Список використаних джерел:**\n<p>github.com/matterport/Mask_RCNN</p>\n<p>github.com/i-pan/kaggle-rsna18 </p>\n<p>kaggle.com/code/pneumonia-classification-fine-tuning-resnet-83-3 </p>\n<p>kaggle.com/code/rsna-pneumonia-detection-cnn-capstone-9 </p>\n<p></p>\n<p></p>\n<p></p>\n<p></p>","metadata":{}}]}