{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"}],"dockerImageVersionId":30512,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Phương hướng:\n* 1. Sử dụng đánh dấu trong bộ dữ liệu để train AI khoanh vùng khối viêm \n* 2. Sử dụng connected components để phân biệt các khối viêm ","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# Cấu trúc network:\n* 2 block chính gồm resblock (6 block, mỗi block 6 lớp) và downsample block (3 block, mỗi block 4 lớp)\n* Input -> giảm chiều ảnh -> trích xuất tính năng (nhiều lần) -> sigmoid để chẩn đoán có viêm hay không -> tăng chiều cho tính năng được trích xuất -> kết quả (tính năng được trích xuất, kết quả của hàm sigmoid)","metadata":{}},{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import os\nimport csv\nimport random\nfrom datetime import datetime\n\nimport pydicom\nimport numpy as np\nimport pandas as pd\nfrom skimage import io\nfrom skimage import measure\nfrom skimage.transform import resize\n\nimport tensorflow as tf\nfrom tensorflow import keras\n\nfrom matplotlib import pyplot as plt\nimport matplotlib.patches as patches","metadata":{"execution":{"iopub.status.busy":"2023-06-17T13:05:47.637710Z","iopub.execute_input":"2023-06-17T13:05:47.638072Z","iopub.status.idle":"2023-06-17T13:05:55.521159Z","shell.execute_reply.started":"2023-06-17T13:05:47.638042Z","shell.execute_reply":"2023-06-17T13:05:55.520136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Cấu trúc bộ dữ liệu \n* Cấu trúc file csv: [tên file : vị trí khối viêm] mỗi hàng.\n* Nếu 1 file chứa nhiều khối viêm, nhiều hàng có thể có cùng tên file những khác vị trí khối viêm  \n* Nếu hàng không có vị trí khối viêm => không bị viêm","metadata":{}},{"cell_type":"markdown","source":"Tạo dictionary (key-value pair) sử dụng tên file làm key và vị trí khối viêm làm value\n* Không bao gồm ảnh không bị viêm trong dictionary","metadata":{}},{"cell_type":"code","source":"# tạo dictionary trống\npneumonia_locations = {}\n# load file csv\nwith open(os.path.join('../input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv'), mode='r') as infile:\n    # đọc file\n    reader = csv.reader(infile)\n    # bỏ qua tên cột \n    next(reader, None)\n    # lặp qua mỗi hàng\n    for rows in reader:\n        # retrieve information\n        filename = rows[0]\n        location = rows[1:5]\n        pneumonia = rows[5]\n        # Nếu hàng chứa khối viêm thì add vào dictionary \n        # 1 file có nhiều khối viêm thì sẽ gộp vị trí thành 1 list \n        if pneumonia == '1':\n            # chuyển số thập phân thành số nguyên \n            location = [int(float(i)) for i in location]\n            # add vị trí khối viêm vào dictionary \n            if filename in pneumonia_locations:\n                pneumonia_locations[filename].append(location)\n            else:\n                pneumonia_locations[filename] = [location]","metadata":{"execution":{"iopub.status.busy":"2023-06-17T13:06:10.007571Z","iopub.execute_input":"2023-06-17T13:06:10.008724Z","iopub.status.idle":"2023-06-17T13:06:10.097202Z","shell.execute_reply.started":"2023-06-17T13:06:10.008686Z","shell.execute_reply":"2023-06-17T13:06:10.096117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load tên file","metadata":{}},{"cell_type":"code","source":"folder = '../input/rsna-pneumonia-detection-challenge/stage_2_train_images'\nlen(os.listdir(folder))","metadata":{"execution":{"iopub.status.busy":"2023-06-17T13:06:11.121436Z","iopub.execute_input":"2023-06-17T13:06:11.121822Z","iopub.status.idle":"2023-06-17T13:06:11.561502Z","shell.execute_reply.started":"2023-06-17T13:06:11.121794Z","shell.execute_reply":"2023-06-17T13:06:11.560561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load + tráo\nfolder = '../input/rsna-pneumonia-detection-challenge/stage_2_train_images'\nfilenames = os.listdir(folder)\nrandom.shuffle(filenames)\n# tách thành train và test\nn_valid_samples = 2684\ntrain_filenames = filenames[n_valid_samples:]\nvalid_filenames = filenames[:n_valid_samples]\nprint('n train samples', len(train_filenames))\nprint('n valid samples', len(valid_filenames))\nn_train_samples = len(filenames) - n_valid_samples","metadata":{"execution":{"iopub.status.busy":"2023-06-17T13:06:11.563373Z","iopub.execute_input":"2023-06-17T13:06:11.563748Z","iopub.status.idle":"2023-06-17T13:06:11.607856Z","shell.execute_reply.started":"2023-06-17T13:06:11.563715Z","shell.execute_reply":"2023-06-17T13:06:11.606763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tìm hiểu dữ liệu","metadata":{}},{"cell_type":"code","source":"print('Số ảnh:',len(filenames))\nprint('Số ảnh viêm:', len(pneumonia_locations))\n\nns = [len(value) for value in pneumonia_locations.values()]\nplt.figure()\nplt.hist(ns)\nplt.xlabel('số khối viêm / ảnh')\nplt.xticks(range(1, np.max(ns)+1))\nplt.show()\n\nheatmap = np.zeros((1024, 1024))\nws = []\nhs = []\nfor values in pneumonia_locations.values():\n    for value in values:\n        x, y, w, h = value\n        heatmap[y:y+h, x:x+w] += 1\n        ws.append(w)\n        hs.append(h)\nplt.figure()\nplt.title('Heatmap')\nplt.imshow(heatmap)\nplt.figure()\nplt.title('Độ dài khối viêm')\nplt.hist(hs, bins=np.linspace(0,1000,50))\nplt.show()\nplt.figure()\nplt.title('Độ rộng khối viêm')\nplt.hist(ws, bins=np.linspace(0,1000,50))\nplt.show()\nprint('Độ dài tối thiểu:', np.min(hs))\nprint('Độ rộng tối thiểu: ', np.min(ws))","metadata":{"execution":{"iopub.status.busy":"2023-06-17T13:06:12.244502Z","iopub.execute_input":"2023-06-17T13:06:12.245587Z","iopub.status.idle":"2023-06-17T13:06:15.074781Z","shell.execute_reply.started":"2023-06-17T13:06:12.245508Z","shell.execute_reply":"2023-06-17T13:06:15.073748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Generator \n* Data quá lớn để xử lý cùng một lúc \n* Sử dụng data generator để điều phối file \n* Khi yêu cầu dữ liệu thì generator sẽ chọn random 1 batch để đưa ra","metadata":{}},{"cell_type":"code","source":"class generator(keras.utils.Sequence):\n    \n    def __init__(self, folder, filenames, pneumonia_locations=None, batch_size=32, image_size=256, shuffle=True, augment=False, predict=False):\n        self.folder = folder\n        self.filenames = filenames\n        self.pneumonia_locations = pneumonia_locations\n        self.batch_size = batch_size\n        self.image_size = image_size\n        self.shuffle = shuffle\n        self.augment = augment\n        self.predict = predict\n        self.on_epoch_end()\n        \n    def __load__(self, filename):\n        # Mở file dicom dưới dạng array của numpy \n        img = pydicom.dcmread(os.path.join(self.folder, filename)).pixel_array\n        # tạo dữ liệu khoanh vùng khối viêm trống \n        # tạo 2d array có số pixel bằng của ảnh và đặt tất cả giá trị = 0\n        msk = np.zeros(img.shape)\n        # Lấy tên file và xoá đuôi \n        filename = filename.split('.')[0]\n        # Kiểm tra xem ảnh có khối viêm không \n        if filename in self.pneumonia_locations:\n            # Lặp qua các khối viêm \n            for location in self.pneumonia_locations[filename]:\n                # Những pixel có khối viêm sẽ được thay từ 0 thành 1 trong dữ liệu khoanh vùng\n                x, y, w, h = location\n                msk[y:y+h, x:x+w] = 1\n        # Đổi size ảnh và khoanh vùng\n        img = resize(img, (self.image_size, self.image_size), mode='reflect')\n        msk = resize(msk, (self.image_size, self.image_size), mode='reflect') > 0.5\n        # Nếu áp dụng augmentation thì  ảnh random \n        if self.augment and random.random() > 0.5:\n            img = np.fliplr(img)\n            msk = np.fliplr(msk)\n        # thêm chiều \n        img = np.expand_dims(img, -1)\n        msk = np.expand_dims(msk, -1)\n        return img, msk\n    \n    def __loadpredict__(self, filename):\n        # Mở file dicom dưới dạng array của numpy \n        img = pydicom.dcmread(os.path.join(self.folder, filename)).pixel_array\n        # resize image\n        img = resize(img, (self.image_size, self.image_size), mode='reflect')\n        # thêm chiều \n        img = np.expand_dims(img, -1)\n        return img\n        \n    def __getitem__(self, index):\n        # chọn batch \n        filenames = self.filenames[index*self.batch_size:(index+1)*self.batch_size]\n        # nếu đang chẩn đoán thì trả lại tên ảnh và ảnh \n        if self.predict:\n            # load file  \n            imgs = [self.__loadpredict__(filename) for filename in filenames]\n            # tạo batch \n            imgs = np.array(imgs)\n            return imgs, filenames\n        #  đang train thì trả lại ảnh và vị trí khối viêm \n        else:\n            # load file\n            items = [self.__load__(filename) for filename in filenames]\n            # tách ảnh và khoanh vùng thành 2 phần riêng biệt \n            imgs, msks = zip(*items)\n            # tạo batch \n            imgs = np.array(imgs)\n            msks = np.array(msks)\n            return imgs, msks\n        \n    def on_epoch_end(self):\n        if self.shuffle:\n            random.shuffle(self.filenames)\n        \n    def __len__(self):\n        if self.predict:\n            # trả lại mọi batch \n            return int(np.ceil(len(self.filenames) / self.batch_size))\n        else:\n            # trả lại những batch đã đầy \n            return int(len(self.filenames) / self.batch_size)","metadata":{"execution":{"iopub.status.busy":"2023-06-17T13:06:15.077160Z","iopub.execute_input":"2023-06-17T13:06:15.077562Z","iopub.status.idle":"2023-06-17T13:06:15.097443Z","shell.execute_reply.started":"2023-06-17T13:06:15.077505Z","shell.execute_reply":"2023-06-17T13:06:15.095982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Network ","metadata":{}},{"cell_type":"code","source":"def create_downsample(channels, inputs):\n    x = keras.layers.BatchNormalization(momentum=0.9)(inputs)\n    x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2D(channels, 1, padding='same', use_bias=False)(x)\n    x = keras.layers.MaxPool2D(2)(x)\n    return x\n\ndef create_resblock(channels, inputs):\n    x = keras.layers.BatchNormalization(momentum=0.9)(inputs)\n    x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2D(channels, 3, padding='same', use_bias=False)(x)\n    x = keras.layers.BatchNormalization(momentum=0.9)(x)\n    x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2D(channels, 3, padding='same', use_bias=False)(x)\n    return keras.layers.add([x, inputs])\n\ndef create_network(input_size, channels, n_blocks=2, depth=4):\n    # input\n    inputs = keras.Input(shape=(input_size, input_size, 1))\n    x = keras.layers.Conv2D(channels, 3, padding='same', use_bias=False)(inputs)\n    # áp dụng các block \n    for d in range(depth):\n        channels = channels * 2\n        x = create_downsample(channels, x)\n        for b in range(n_blocks):\n            x = create_resblock(channels, x)\n    # output\n    x = keras.layers.BatchNormalization(momentum=0.9)(x)\n    x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2D(1, 1, activation='sigmoid')(x)\n    outputs = keras.layers.UpSampling2D(2**depth)(x)\n    model = keras.Model(inputs=inputs, outputs=outputs)\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-06-17T13:06:15.099414Z","iopub.execute_input":"2023-06-17T13:06:15.099810Z","iopub.status.idle":"2023-06-17T13:06:15.115085Z","shell.execute_reply.started":"2023-06-17T13:06:15.099772Z","shell.execute_reply":"2023-06-17T13:06:15.113877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"# iou loss\ndef iou_loss(y_true, y_pred):\n    y_true = tf.reshape(y_true, [-1])\n    y_pred = tf.reshape(y_pred, [-1])\n    y_true = tf.cast(y_true, dtype=tf.float32 , name=None)\n    y_pred = tf.cast(y_pred, dtype=tf.float32 , name=None)\n    intersection = tf.reduce_sum(y_true * y_pred)\n    score = (intersection + 1.) / (tf.reduce_sum(y_true) + tf.reduce_sum(y_pred) - intersection + 1.)\n    return 1 - score\n\n# bce loss và iou loss \ndef iou_bce_loss(y_true, y_pred):\n    return 0.5 * keras.losses.binary_crossentropy(y_true, y_pred) + 0.5 * iou_loss(y_true, y_pred)\n\n# iou trung trung bình\ndef mean_iou(y_true, y_pred):\n    y_pred = tf.round(y_pred)\n    intersect = tf.reduce_sum(y_true * y_pred, axis=[1, 2, 3])\n    union = tf.reduce_sum(y_true, axis=[1, 2, 3]) + tf.reduce_sum(y_pred, axis=[1, 2, 3])\n    smooth = tf.ones(tf.shape(intersect))\n    return tf.reduce_mean((intersect + smooth) / (union - intersect + smooth))\n\n# tạo network \nmodel = create_network(input_size=256, channels=32, n_blocks=2, depth=4)\nmodel.compile(optimizer='adam',\n              loss=iou_bce_loss,\n              metrics=['accuracy', mean_iou])\n\n# chỉnh learning rate \ndef cosine_annealing(x):\n    lr = 0.001\n    epochs = 35\n    return lr*(np.cos(np.pi*x/epochs)+1.)/2\nlearning_rate = tf.keras.callbacks.LearningRateScheduler(cosine_annealing)\n\n# tạo generator cho train và test \nfolder = '../input/rsna-pneumonia-detection-challenge/stage_2_train_images'\ntrain_gen = generator(folder, train_filenames, pneumonia_locations, batch_size=32, image_size=256, shuffle=True, augment=True, predict=False)\nvalid_gen = generator(folder, valid_filenames, pneumonia_locations, batch_size=32, image_size=256, shuffle=False, predict=False)\n\n# Log cho tensorboard\nlogdir=\"/kaggle/working/logs/fit/\" + datetime.now().strftime(\"%Y%m%d-%H%M%S\")\ntensorboard_callback = keras.callbacks.TensorBoard(log_dir=logdir)\n\nhistory = model.fit(train_gen, validation_data=valid_gen, callbacks=[learning_rate, tensorboard_callback], epochs=30, workers=4, use_multiprocessing=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-17T13:07:36.941805Z","iopub.execute_input":"2023-06-17T13:07:36.942176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,4))\nplt.subplot(131)\nplt.plot(history.epoch, history.history[\"loss\"], label=\"Train loss\")\nplt.plot(history.epoch, history.history[\"val_loss\"], label=\"Valid loss\")\nplt.legend()\nplt.subplot(132)\nplt.plot(history.epoch, history.history[\"accuracy\"], label=\"Train accuracy\")\nplt.plot(history.epoch, history.history[\"val_accuracy\"], label=\"Valid accuracy\")\nplt.legend()\nplt.subplot(133)\nplt.plot(history.epoch, history.history[\"mean_iou\"], label=\"Train iou\")\nplt.plot(history.epoch, history.history[\"val_mean_iou\"], label=\"Valid iou\")\nplt.legend()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for imgs, msks in valid_gen:\n    # dự đoán\n    preds = model.predict(imgs)\n    # tạo đồ thị \n    f, axarr = plt.subplots(4, 8, figsize=(20,15))\n    axarr = axarr.ravel()\n    axidx = 0\n    # lặp qua batch \n    for img, msk, pred in zip(imgs, msks, preds):\n        # hiển thị hình ảnh \n        axarr[axidx].imshow(img[:, :, 0], cmap='gray')\n        # kiểm tra ảnh có viêm hay không (giá trị thật)\n        comp = msk[:, :, 0] > 0.5\n        # tách vị trí khối viêm \n        comp = measure.label(comp)\n        # tạo khoanh vùng \n        predictionString = ''\n        for region in measure.regionprops(comp):\n            # lấy giá trị x, y, dài, rộng \n            y, x, y2, x2 = region.bbox\n            height = y2 - y\n            width = x2 - x\n            axarr[axidx].add_patch(patches.Rectangle((x,y),width,height,linewidth=2,edgecolor='b',facecolor='none'))\n        # kiểm tra ảnh có viêm hay không (giá trị do mô hình dự đoán)\n        comp = pred[:, :, 0] > 0.5\n        # tách vị trí khối viêm \n        comp = measure.label(comp)\n        # tạo khoanh vùng \n        predictionString = ''\n        for region in measure.regionprops(comp):\n            # ấy giá trị x, y, dài, rộng \n            y, x, y2, x2 = region.bbox\n            height = y2 - y\n            width = x2 - x\n            axarr[axidx].add_patch(patches.Rectangle((x,y),width,height,linewidth=2,edgecolor='r',facecolor='none'))\n        axidx += 1\n    plt.show()\n    # chỉ hiển thị 1 batch \n    break","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('/kaggle/working/trained')","metadata":{},"execution_count":null,"outputs":[]}]}