{"metadata":{"colab":{"name":"lesson-3-rsna-pneumonia-detection-challenge-kaggle","version":"0.3.2","provenance":[],"collapsed_sections":[]},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"accelerator":"GPU","language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"}],"dockerImageVersionId":27938,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os \nimport sys\nimport random\nimport math\nimport numpy as np\nimport cv2\nimport matplotlib.pyplot as plt\nimport json\nimport pydicom\nfrom imgaug import augmenters as iaa\nfrom tqdm import tqdm\nimport pandas as pd \nimport glob\nfrom sklearn.model_selection import KFold\nDATA_DIR = '/kaggle/input'\n\n# Directory to save logs and trained model\nROOT_DIR = '/kaggle/working'","metadata":{"id":"4kjcC6QqywWl","_uuid":"40c67b3ff0fa04587dec508363308adaa3ceaf34","execution":{"iopub.status.busy":"2024-02-11T21:23:05.733737Z","iopub.execute_input":"2024-02-11T21:23:05.734052Z","iopub.status.idle":"2024-02-11T21:23:07.171669Z","shell.execute_reply.started":"2024-02-11T21:23:05.733995Z","shell.execute_reply":"2024-02-11T21:23:07.170751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Встановлення моделі та всіх необхідних конфігів.**\n<p>Це все треба один раз запустити і потім це треба закоментити</p>","metadata":{"id":"kdYzLq1zfKL4","_uuid":"576df4c47a23d08b1bdb384245e09aa69f88bbd3"}},{"cell_type":"code","source":"!git clone https://www.github.com/matterport/Mask_RCNN.git\nos.chdir('Mask_RCNN')\n!wget --quiet https://github.com/matterport/Mask_RCNN/releases/download/v2.0/mask_rcnn_coco.h5\n!ls -lh mask_rcnn_coco.h5\n\n!pip install gdown\nimport gdown\n\n\nurl = 'https://drive.google.com/uc?id='+'1ZKSpyTya52QLb-ufq2rjbyyHoeuUn2gx'\noutput = '/kaggle/working/model.h5'\ngdown.download(url, output, quiet=False)\n\n#!python setup.py -q install","metadata":{"id":"KgllzLnDr7kF","outputId":"6c978df7-2013-437e-acd1-5011048dfb53","_uuid":"b37d22551d332f0f7b722cc7204eb614524b6c21","execution":{"iopub.status.busy":"2024-02-11T18:58:33.010809Z","iopub.execute_input":"2024-02-11T18:58:33.011158Z","iopub.status.idle":"2024-02-11T18:58:51.209320Z","shell.execute_reply.started":"2024-02-11T18:58:33.011102Z","shell.execute_reply":"2024-02-11T18:58:51.208362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import Mask RCNN\nsys.path.append(os.path.join(ROOT_DIR, 'Mask_RCNN'))  # To find local version of the library\nfrom mrcnn.config import Config\nfrom mrcnn import utils\nimport mrcnn.model as modellib\nfrom mrcnn import visualize\nfrom mrcnn.model import log","metadata":{"id":"-KZXyWwhzOVU","outputId":"2576cc17-7484-4311-ad72-3c5643dcb5bb","_uuid":"3acbbbe055b6a409d3c50ae0f893acf51b5ae7ba","execution":{"iopub.status.busy":"2024-02-11T18:58:51.211218Z","iopub.execute_input":"2024-02-11T18:58:51.211464Z","iopub.status.idle":"2024-02-11T18:58:51.216845Z","shell.execute_reply.started":"2024-02-11T18:58:51.211421Z","shell.execute_reply":"2024-02-11T18:58:51.215993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dicom_dir = os.path.join(DATA_DIR, 'stage_2_test_images')\n\nCOCO_WEIGHTS_PATH = \"mask_rcnn_coco.h5\"","metadata":{"id":"FghMmiMjzOX2","_uuid":"50089cc61791871cdf6a5c0037dc4f28b7b7d7cc","execution":{"iopub.status.busy":"2024-02-11T18:58:51.218609Z","iopub.execute_input":"2024-02-11T18:58:51.218954Z","iopub.status.idle":"2024-02-11T18:58:51.230031Z","shell.execute_reply.started":"2024-02-11T18:58:51.218888Z","shell.execute_reply":"2024-02-11T18:58:51.229253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Конфігурація моделі перед запуском**","metadata":{}},{"cell_type":"code","source":"def get_dicom_fps(dicom_dir):\n    dicom_fps = glob.glob(dicom_dir+'/'+'*.dcm')\n    return list(set(dicom_fps))\n\ndef parse_dataset(dicom_dir, anns): \n    image_fps = get_dicom_fps(dicom_dir)\n    image_annotations = {fp: [] for fp in image_fps}\n    for index, row in anns.iterrows(): \n        fp = os.path.join(dicom_dir, row['patientId']+'.dcm')\n        image_annotations[fp].append(row)\n    return image_fps, image_annotations \n\n\nclass DetectorConfig(Config):\n    \"\"\"\n    Конфігурація основних параметрів\n    \"\"\" \n    NAME = 'pneumonia'\n    GPU_COUNT = 1\n    \n    BACKBONE = 'resnet50'\n    \n    NUM_CLASSES = 2  \n    \n    IMAGE_MIN_DIM = 256\n    IMAGE_MAX_DIM = 256\n    RPN_ANCHOR_SCALES = (16, 32, 64, 128)\n    TRAIN_ROIS_PER_IMAGE = 32\n    MAX_GT_INSTANCES = 4\n    DETECTION_MAX_INSTANCES = 3\n    DETECTION_MIN_CONFIDENCE = 0.78  \n    DETECTION_NMS_THRESHOLD = 0.01\n\n    STEPS_PER_EPOCH = 200    \n    \nconfig = DetectorConfig() \nconfig.display()\n\n\n\nclass DetectorDataset(utils.Dataset):\n    \"\"\"\n    Dataset class for training pneumonia detection on the RSNA pneumonia dataset.\n    \"\"\"\n\n    def __init__(self, paths_to_images, annotations_per_image, image_height, image_width):\n        # Виклик конструктора базового класу\n        super().__init__()\n\n        # Додавання класу \"Легенева непрозорість\" до набору даних\n        self.add_class('pneumonia', 1, 'Lung Opacity')\n\n        # Додавання інформації про зображення в набір даних\n        for index, image_path in enumerate(paths_to_images):\n            # Отримання анотацій для конкретного зображення\n            current_annotations = annotations_per_image[image_path]\n            # Додавання зображення та його анотацій в набір даних\n            self.add_image(\n                'pneumonia',  # Джерело даних\n                image_id=index, \n                path=image_path,  \n                annotations=current_annotations,  \n                orig_height=image_height,  \n                orig_width=image_width \n            )\n\n    def image_reference(self, image_id):\n        image_info = self.image_info[image_id]\n        # Повертаємо шлях до зображення\n        return image_info['path']\n\n    def load_image(self, image_id):\n        # Отримання інформації про зображення за ID\n        image_info = self.image_info[image_id]\n        # Шлях до файлу зображення\n        image_path = image_info['path']\n        # Завантаження DICOM файла\n        dicom_file = pydicom.read_file(image_path)\n        # Перетворення DICOM у numpy масив\n        image_array = dicom_file.pixel_array\n        # Якщо зображення в градаціях сірого, конвертуємо у RGB для уніфікації\n        if len(image_array.shape) != 3 or image_array.shape[2] != 3:\n            image_array = np.stack((image_array,) * 3, axis=-1)\n        # Повертаємо зображення\n        return image_array\n\n    def load_mask(self, image_id):\n        # Отримання детальної інформації про зображення за його ідентифікатором\n        image_details = self.image_info[image_id]\n        # Витягуємо анотації, пов'язані з цим зображенням\n        image_annotations = image_details['annotations']\n        # Кількість анотацій\n        annotations_count = len(image_annotations)\n        # Ініціалізація маски та масиву ідентифікаторів класів\n        if annotations_count == 0:\n            # Створення порожньої маски, якщо анотацій немає\n            mask = np.zeros((image_details['orig_height'], image_details['orig_width'], 1), dtype=np.uint8)\n            class_ids = np.zeros((1,), dtype=np.int32)\n        else:\n            # Створення маски з анотаціями\n            mask = np.zeros((image_details['orig_height'], image_details['orig_width'], annotations_count), dtype=np.uint8)\n            class_ids = np.zeros((annotations_count,), dtype=np.int32)\n            for idx, annotation in enumerate(image_annotations):\n                if annotation['Target'] == 1:\n                    # Координати об'єкта анотації\n                    rect_start_x = int(annotation['x'])\n                    rect_start_y = int(annotation['y'])\n                    rect_width = int(annotation['width'])\n                    rect_height = int(annotation['height'])\n                    # Створення індивідуальної маски для анотації\n                    individual_mask = mask[:, :, idx].copy()\n                    cv2.rectangle(individual_mask, (rect_start_x, rect_start_y), (rect_start_x + rect_width, rect_start_y + rect_height), 255, -1)\n                    mask[:, :, idx] = individual_mask\n                    class_ids[idx] = 1\n        # Конвертація типу маски для використання у подальшій обробці\n        return mask.astype(np.bool_), class_ids.astype(np.int32)\n    \n    \n# Зробити прогнози на тестових зображеннях, записати зразок подання\ndef predict(image_fps, filepath=ROOT_DIR+'/submission.csv', min_conf=0.98): \n    # припускаємо квадратне зображення   \n    with open(filepath, 'w') as file:\n        file.write('patientId,PredictionString'+\"\\n\")        \n        for image_id in tqdm(image_fps): \n            ds = pydicom.read_file(image_id)\n            image = ds.pixel_array\n        \n            # Якщо в чорно-білому. Конвертуємо в RGB для послідовності\n            if len(image.shape) != 3 or image.shape[2] != 3:\n                image = np.stack((image,) * 3, -1) \n            \n            patient_id = os.path.splitext(os.path.basename(image_id))[0]\n\n            results = model.detect([image])\n            r = results[0]\n\n            out_str = \"\"\n            out_str += patient_id \n            assert( len(r['rois']) == len(r['class_ids']) == len(r['scores']) )\n            if len(r['rois']) == 0:\n                out_str += \",\"\n            else: \n                num_instances = len(r['rois'])\n                out_str += \",\"\n                for i in range(num_instances): \n                    if r['scores'][i] > min_conf: \n                        out_str += ' '\n                        out_str += str(round(r['scores'][i], 2))\n                        out_str += ' '\n\n                        # x1, y1, ширина, висота \n                        x1 = r['rois'][i][1]\n                        y1 = r['rois'][i][0]\n                        width = r['rois'][i][3] - x1 \n                        height = r['rois'][i][2] - y1 \n                        bboxes_str = \"{} {} {} {}\".format(x1, y1, \\\n                                                          width, height)    \n                        out_str += bboxes_str\n            file.write(out_str+\"\\n\")","metadata":{"id":"ivqC4cnszOaM","_uuid":"778cb19865d7cc63440491aef9202b71c61e8bb2","execution":{"iopub.status.busy":"2024-02-11T18:58:51.231466Z","iopub.execute_input":"2024-02-11T18:58:51.231710Z","iopub.status.idle":"2024-02-11T18:58:51.275787Z","shell.execute_reply.started":"2024-02-11T18:58:51.231662Z","shell.execute_reply":"2024-02-11T18:58:51.274983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Ініціалізація моделі**","metadata":{}},{"cell_type":"code","source":"class InferenceConfig(DetectorConfig):\n    GPU_COUNT = 1\n    IMAGES_PER_GPU = 1\n\ninference_config = InferenceConfig()\n\n# перестворення моделі\nmodel = modellib.MaskRCNN(mode='inference', \n                          config=inference_config,\n                          model_dir=ROOT_DIR)\n\n\n#завантаження заздалегідь натренованої моделі, якщо закоментувати model_path, то завантажиться поточна модель\nmodel_path = \"/kaggle/working/model.h5\"\nprint(\"Loading weights from \", model_path)\nmodel.load_weights(model_path, by_name=True)","metadata":{"id":"_SfzTa-1zOck","outputId":"91ae8935-bccb-4b8e-9a7e-aa690f95fd9b","_uuid":"dfcffc4eaa94a41497717851dee9f702d8a2a73b","execution":{"iopub.status.busy":"2024-02-11T18:58:51.278293Z","iopub.execute_input":"2024-02-11T18:58:51.278629Z","iopub.status.idle":"2024-02-11T18:58:57.384163Z","shell.execute_reply.started":"2024-02-11T18:58:51.278563Z","shell.execute_reply":"2024-02-11T18:58:57.383476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Обробка тестових даних**","metadata":{}},{"cell_type":"code","source":"\ntest_image_fps = get_dicom_fps(test_dicom_dir)\nsample_submission_fp = ROOT_DIR+'/submission.csv'\npredict(test_image_fps, filepath=sample_submission_fp)\n#output = pd.read_csv(sample_submission_fp)\n# output.head(50)","metadata":{"id":"EdhUEFDr0yDA","outputId":"1715a5df-a577-41fd-bf20-f1a27aadb28c","_uuid":"793b1c6c6ba4e5f0d51e130080aa799f230b5ef6","execution":{"iopub.status.busy":"2024-02-11T18:58:57.385950Z","iopub.execute_input":"2024-02-11T18:58:57.386289Z","iopub.status.idle":"2024-02-11T19:06:33.137821Z","shell.execute_reply.started":"2024-02-11T18:58:57.386226Z","shell.execute_reply":"2024-02-11T19:06:33.136661Z"},"trusted":true},"execution_count":null,"outputs":[]}]}