{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":26680,"databundleVersionId":2283525,"sourceType":"competition"},{"sourceId":153296609,"sourceType":"kernelVersion"}],"dockerImageVersionId":30579,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## DATA PREPROCESSING SIIM-COVID19-DETECTION","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"Data Preprocessing for SIIM-COVID19-DETECTION DATASET. URL: https://www.kaggle.com/competitions/siim-covid19-detection/","metadata":{}},{"cell_type":"markdown","source":"The goal of this notebook is preparing the dataset to be used for implementing object-detection and multi-label classification algorithms.","metadata":{}},{"cell_type":"markdown","source":"## 0. Libraries and global variables:","metadata":{}},{"cell_type":"code","source":"!conda install gdcm -c conda-forge -y","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-12-09T18:10:18.184745Z","iopub.execute_input":"2023-12-09T18:10:18.187732Z","iopub.status.idle":"2023-12-09T18:11:33.507949Z","shell.execute_reply.started":"2023-12-09T18:10:18.187599Z","shell.execute_reply":"2023-12-09T18:11:33.50445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport math\nimport copy\nimport pydicom\nimport cv2\nimport shutil\nimport multiprocessing\nimport pandas as pd\nimport numpy as np\nimport warnings\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib.gridspec as gridspec\n\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom pydicom.encaps import encapsulate\nfrom multiprocessing import Manager\nfrom itertools import chain\nfrom random import randint","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:11:33.513629Z","iopub.execute_input":"2023-12-09T18:11:33.515878Z","iopub.status.idle":"2023-12-09T18:11:33.535283Z","shell.execute_reply.started":"2023-12-09T18:11:33.515768Z","shell.execute_reply":"2023-12-09T18:11:33.532144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Suppress pydicom warnings related to pixel data\nwarnings.filterwarnings(\"ignore\",\n                        category=UserWarning,\n                        module=\"pydicom.pixel_data_handlers.numpy_handler\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-09T18:11:33.538863Z","iopub.execute_input":"2023-12-09T18:11:33.539441Z","iopub.status.idle":"2023-12-09T18:11:34.265158Z","shell.execute_reply.started":"2023-12-09T18:11:33.539386Z","shell.execute_reply":"2023-12-09T18:11:34.262519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# GLOBAL variables\n\n# Paths to processed csvs.\nBASE_PATH = \"../input/eda-siim-covid19-detection\"\n\nTRAIN_CSV_PATH = os.path.join(BASE_PATH,\"train.csv\")\nTEST_CSV_PATH = os.path.join(BASE_PATH,\"test.csv\")\nBOXES_CSV_PATH = os.path.join(BASE_PATH,\"train_boxes.csv\")\n\n# Resizing shape.\nTARGET_SHAPE=(256, 256)\n# Used for reading and processing DICOM images.\nVOI_LUT=True\nFIX_MONOCHROME=True","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:11:34.271932Z","iopub.execute_input":"2023-12-09T18:11:34.273784Z","iopub.status.idle":"2023-12-09T18:11:34.285683Z","shell.execute_reply.started":"2023-12-09T18:11:34.273721Z","shell.execute_reply":"2023-12-09T18:11:34.283709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. Loading data:","metadata":{}},{"cell_type":"markdown","source":"### 1.1 Reading csvs","metadata":{}},{"cell_type":"code","source":"train_boxes = pd.read_csv(BOXES_CSV_PATH)\ntrain_boxes.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:11:34.288753Z","iopub.execute_input":"2023-12-09T18:11:34.289429Z","iopub.status.idle":"2023-12-09T18:11:34.479571Z","shell.execute_reply.started":"2023-12-09T18:11:34.289352Z","shell.execute_reply":"2023-12-09T18:11:34.476232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv(TRAIN_CSV_PATH)\ntrain_csv = train_csv.drop(columns=['boxes', 'label'])\ntrain_csv.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:11:34.483128Z","iopub.execute_input":"2023-12-09T18:11:34.483817Z","iopub.status.idle":"2023-12-09T18:11:34.622856Z","shell.execute_reply.started":"2023-12-09T18:11:34.483752Z","shell.execute_reply":"2023-12-09T18:11:34.61902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_csv = pd.read_csv(TEST_CSV_PATH)\ntest_csv.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:11:34.626463Z","iopub.execute_input":"2023-12-09T18:11:34.627442Z","iopub.status.idle":"2023-12-09T18:11:34.670764Z","shell.execute_reply.started":"2023-12-09T18:11:34.627382Z","shell.execute_reply":"2023-12-09T18:11:34.667359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"These are a custom CSVs generated from extractable information in SIIM-COVID19-DETECTION Dataset, and have been produced in the following notebook: https://www.kaggle.com/code/martacoll/eda-siim-covid19-detection/notebook","metadata":{}},{"cell_type":"markdown","source":"### 1.2 Utility functions for reading or visualizing images","metadata":{}},{"cell_type":"code","source":"def read_dicom_image(filename, dicom_data=False):\n    \"\"\"Credit: https://github.com/pydicom/pydicom/issues/319\n               https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n               \n    Function reads and corrects dicom pixel data to prevent x-rays from looking inverted.\n    args:\n        filename: DICOM file path (str)\n    returns:\n         modified_image_data: Image data. (pixel array)\n    \"\"\"\n    dicom_header = pydicom.dcmread(filename) \n\n    #====== DICOM IMAGE DATA ======\n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \"human-friendly\" view\n    if VOI_LUT:\n        data = apply_voi_lut(dicom_header.pixel_array, dicom_header)\n    else:\n        data = dicom_header.pixel_array\n    # depending on this value, X-ray may look inverted - fix that:\n    if FIX_MONOCHROME and dicom_header.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n\n    data = data - np.min(data)\n    data = data / np.max(data)\n    modified_image_data = (data * 255).astype(np.uint8)\n    \n    if dicom_data:\n        return modified_image_data, dicom_header\n\n    return modified_image_data","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:11:34.674275Z","iopub.execute_input":"2023-12-09T18:11:34.674938Z","iopub.status.idle":"2023-12-09T18:11:34.693222Z","shell.execute_reply.started":"2023-12-09T18:11:34.674905Z","shell.execute_reply":"2023-12-09T18:11:34.689641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_DICOM(images_paths, boxes=None, title=None,\n               images_titles=None, resize=False, resize_shape=(2000,2000), kar=False, thickness=5):\n    \"\"\"\n    Function to plot a variable number of DICOM images.\n    args:\n        images_paths: list of image paths. (list of strings)\n        boxes: List of bounding boxes for each image (list of lists of floats)\n        title: String for figure title. (str)\n        images_titles: List of strings for image titles. (list of strings)\n        resize: If true applies resize to image. (bool)\n        resize_shape: Resize shape. (tuple)\n        kar: If true keeps aspect ratio when resizing. (bool)\n    \"\"\"\n\n    num_images = len(images_paths)\n    \n    cols = int(math.ceil(math.sqrt(num_images)))\n    rows = int(math.ceil(num_images / cols))\n    \n    fig = plt.figure(figsize=(9, 8))\n    gs = gridspec.GridSpec(rows, cols,\n                           width_ratios=[1] * cols,\n                           height_ratios=[1] * rows)\n    \n    for i, image_path in enumerate(images_paths):\n        assert image_path.endswith(('.png', '.dcm')), \"Invalid image_path values found\"\n\n        if image_path.endswith('.dcm'):\n            image_data = read_dicom_image(image_path)\n            image_data = np.stack([image_data, image_data, image_data],\n                                  axis=-1) # To RGB.\n        elif image_path.endswith('.png'):\n            image_data = cv2.imread(image_path)\n\n        # Plot bounding boxes if provided.\n        if boxes is not None:\n            if not boxes[i]:\n                print(f'There is no ROI present in image {i + 1}.')\n            elif len(boxes[i])%4!=0:\n                print(f'Boxes provided have wrong format for image {i + 1}.')\n            else:\n                for n_box in range(0, int(len(boxes[i])/4)):\n                    color = [randint(0, 255), randint(0, 255), randint(0, 255)]\n\n                    x = int(boxes[i][0 + n_box*4])\n                    y = int(boxes[i][1 + n_box*4])\n                    w = int(boxes[i][2 + n_box*4])\n                    h = int(boxes[i][3 + n_box*4])\n                    \n                    image_data = cv2.rectangle(image_data,\n                                               (x, y),\n                                               (w, h),\n                                               color=color,\n                                               thickness=thickness)\n        if resize:\n            image_data = resize_image(image_data, resize_shape, kar)\n\n        ax = plt.subplot(gs[i])\n        ax.grid(False)\n        ax.imshow(image_data)\n\n        if images_titles is not None:\n            ax.set_title(images_titles[i])\n        else:\n            ax.set_title(f'Image {i + 1}')\n\n    if title is not None:\n        plt.suptitle(title)\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:11:34.696941Z","iopub.execute_input":"2023-12-09T18:11:34.697778Z","iopub.status.idle":"2023-12-09T18:11:34.739815Z","shell.execute_reply.started":"2023-12-09T18:11:34.697724Z","shell.execute_reply":"2023-12-09T18:11:34.73719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Data clean-up","metadata":{}},{"cell_type":"markdown","source":"From the [EDA](https://www.kaggle.com/code/martacoll/eda-siim-covid19-detection) performed, we observed that study label 'Negative for pneumonia' have no bounding boxes asociated. For the rest of study labels, a minor percentage of images had no bounding box. For ensuring that future analysis predict a bounding box for these classes, images with no bounding box for these categories will be deleted.","metadata":{}},{"cell_type":"code","source":"# Cleanup.\nno_bbx = train_boxes.loc[(train_boxes.box_label == 'none 1') &\n                         (train_boxes.y_label != 'negative')]\nno_bbx_ids = no_bbx.id.unique()\ntrain_csv = train_csv[~train_csv.id.isin(no_bbx_ids)].reset_index(drop = True)\ntrain_boxes = train_boxes[~train_boxes.id.isin(no_bbx_ids)].reset_index(drop = True)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:11:34.748502Z","iopub.execute_input":"2023-12-09T18:11:34.749238Z","iopub.status.idle":"2023-12-09T18:11:34.789003Z","shell.execute_reply.started":"2023-12-09T18:11:34.749181Z","shell.execute_reply":"2023-12-09T18:11:34.787388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Data Augmentation","metadata":{}},{"cell_type":"markdown","source":"From the [EDA](https://www.kaggle.com/code/martacoll/eda-siim-covid19-detection) performed, we have seen a need of balancing the target variable distribution. To achieve it, data augmentation will be applied for minoritary categories.","metadata":{}},{"cell_type":"markdown","source":"When augmenting data for X-ray images, it's important to consider the medical context and the specific characteristics of X-ray imaging. Recommended data augmentation techniques for X-ray images:\n\n**Rotation:**\nRotate the X-ray images by small angles to simulate different perspectives. This can help the model become more robust to variations in positioning during imaging.\n\n**Flip (Horizontal and Vertical):**\nHorizontal and vertical flips can be applied to simulate X-ray images taken from different orientations, enhancing the model's ability to generalize.\n\n**Translation:**\nShifting the X-ray images horizontally or vertically can simulate slight changes in the patient's position or the X-ray machine's alignment.\n\n**Brightness and Contrast Adjustments:**\nModifying the brightness and contrast of X-ray images can simulate variations in imaging conditions. Care should be taken to ensure that the adjustments do not compromise the diagnostic quality.\n\n\n**Noise Injection:**\nAdd different types of noise (e.g., Gaussian noise) to the X-ray images to make the model more robust to noisy data.\n\n**Blur:**\nApply slight blurring to simulate the inherent blurriness that might be present in X-ray images due to factors like motion or equipment limitations.\n\n**Cropping:**\nRandomly crop X-ray images to focus on specific regions of interest. This can help the model learn to detect features in different parts of the image.\n\n**Gamma Correction:**\nAdjust the gamma of the X-ray images to simulate variations in the intensity of the X-ray radiation.\n\nIt's crucial to note that any data augmentation applied to medical images should be done cautiously, with consideration for patient safety and the potential impact on diagnostic accuracy. Additionally, it's often beneficial to consult with medical professionals to ensure that the augmented data remains clinically relevant.","metadata":{}},{"cell_type":"markdown","source":"### 2.1 Study of different data augmentation strategies","metadata":{}},{"cell_type":"code","source":"def rotate_image(image, angle):\n    rows, cols = image.shape[:2]\n    M = cv2.getRotationMatrix2D((cols / 2, rows / 2), angle, 1)\n    return \"Rotation\", cv2.warpAffine(image, M, (cols, rows))\n\ndef flip_image(image, flip_code):\n    assert flip_code in [0, 1, -1], \"Wrong flip code, options are 0, 1, -1\"\n    if flip_code == 1:\n        flip_type = \"Horizontal flip\"\n    elif flip_code == 0:\n        flip_type = \"Vertical flip\"\n    else:\n        flip_type = \"Horizontal and vertical flip\"\n\n    return f\"{flip_type}\", cv2.flip(image, flip_code)\n\ndef translate_image(image, tx, ty): # Img size is the same after translation. Interpolates.\n    rows, cols = image.shape[:2]\n    M = np.float32([[1, 0, tx], [0, 1, ty]])\n    return \"Translation\", cv2.warpAffine(image, M, (cols, rows))\n\ndef adjust_brightness_contrast(image, alpha, beta):\n    return \"Brightness and Contrast Adjustments\", cv2.convertScaleAbs(image,\n                                                                      alpha=alpha,\n                                                                      beta=beta)\n\ndef add_noise(image, mean=0, sigma=25):\n    row, col = image.shape\n    gauss = np.random.normal(mean, sigma, (row, col))\n    noisy = np.clip(image + gauss, 0, 255)\n    return \"Noise Injection\", noisy.astype(np.uint8)\n\n\ndef apply_blur(image, kernel_size=(5, 5)):\n    # Ensure that the kernel size is a positive odd integer tuple\n    kernel_size = (kernel_size[0] if kernel_size[0] % 2 == 1 else kernel_size[0] + 1,\n                   kernel_size[1] if kernel_size[1] % 2 == 1 else kernel_size[1] + 1)\n    return \"Blur\", cv2.GaussianBlur(image, kernel_size, 0)\n\ndef crop_image(image, crop_percent=0.2):\n    height, width = image.shape[:2]\n    crop_pixels = int(min(height, width) * crop_percent)\n    return \"Cropping\", image[crop_pixels:height-crop_pixels,\n                             crop_pixels:width-crop_pixels]\n\ndef adjust_gamma(image, gamma=1.0):\n    inv_gamma = 1.0 / gamma\n    table = np.array([((i / 255.0) ** inv_gamma) * 255 for i in np.arange(0, 256)])\n    return \"Gamma correction\", cv2.LUT(image.astype(np.uint8),\n                                       table.astype(np.uint8))","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:11:34.791348Z","iopub.execute_input":"2023-12-09T18:11:34.791773Z","iopub.status.idle":"2023-12-09T18:11:34.817538Z","shell.execute_reply.started":"2023-12-09T18:11:34.79173Z","shell.execute_reply":"2023-12-09T18:11:34.815224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load an example X-ray image (replace 'your_image_path.jpg' with the actual path)\nimage_path = train_csv.image_path[0]\noriginal_image = read_dicom_image(image_path)\n\nfig = plt.figure(figsize=(15, 10))\n# Display the original image\nplt.subplot(3, 4, 1)\nplt.imshow(cv2.cvtColor(original_image, cv2.COLOR_BGR2RGB))\nplt.title('Original')\n\n# Apply data augmentation techniques and display the augmented images\naugmentations = [\n    (rotate_image, {'angle': 15}),\n    (flip_image, {'flip_code': 1}),  # 1 for horizontal flip\n    (flip_image, {'flip_code': 0}),  # 0 for vertical flip\n    (flip_image, {'flip_code': -1}),  # 1 for horizontal and vertical flip\n    (translate_image, {'tx': 30, 'ty': 20}),\n    (adjust_brightness_contrast, {'alpha': 1.2, 'beta': 20}),\n    (add_noise, {'mean': 0.1, 'sigma': 25}),\n    (apply_blur, {'kernel_size': (10, 10)}),\n    (crop_image, {'crop_percent': 0.15}),\n    (adjust_gamma, {'gamma': 1.5})\n]\n\nfor i, (augmentation, kwargs) in enumerate(augmentations, start=2):\n    title, augmented_image = augmentation(original_image, **kwargs)\n    plt.subplot(3, 4, i)\n    plt.imshow(cv2.cvtColor(augmented_image, cv2.COLOR_BGR2RGB))\n    plt.title(f'{title}')\n\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-09T18:11:34.820771Z","iopub.execute_input":"2023-12-09T18:11:34.821971Z","iopub.status.idle":"2023-12-09T18:11:58.121595Z","shell.execute_reply.started":"2023-12-09T18:11:34.821891Z","shell.execute_reply":"2023-12-09T18:11:58.12019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From different strategies analyzed, strateggies that imply a position shift will be selected for data augmentation; cropping, horizontal flip and translation. With the addition of not very noticeble noise ingection and blur, to avoid losing medical relevant data. Rotation is discarded due to the complexity on recalculating the bounding boxes. \n\nBrightness and contrast or gamma might be modified if image enhancing is applied later so we will discard them aswell. \n\nThe goal will be to duplicate images for 'Indeterminate' category, and triplicate images for 'Atypical'.","metadata":{}},{"cell_type":"markdown","source":"### 2.2 Adjust bounding boxes based on DA method used.","metadata":{}},{"cell_type":"markdown","source":"#### 2.2.1 Implementation","metadata":{}},{"cell_type":"code","source":"def recalculate_bounding_box(original_bbox, transformation, t_image_shape, o_image_shape, **kwargs):\n    \n    assert transformation in [\"Horizontal flip\",\n                              \"Cropping\",\n                              \"Translation\"], \"Transformation not implemented.\"\n\n    xmin, ymin, xmax, ymax = original_bbox\n    t_height, t_width = t_image_shape[:2]\n    \n    if transformation == \"Horizontal flip\":\n        new_xmin = t_width - xmax\n        new_ymin = ymin\n        new_xmax = t_width - xmin\n        new_ymax = ymax\n\n    elif transformation == \"Cropping\":\n        height, width = o_image_shape[:2]\n        crop_pixels = int(min(height, width) * kwargs.get(\"crop_percent\"))\n        \n        # Calculate the intersection between the original box and the crop region.\n        x_intersection = max(0, min(xmax, width - crop_pixels) - max(xmin, crop_pixels))\n        y_intersection = max(0, min(ymax, height - crop_pixels) - max(ymin, crop_pixels))\n\n        # If there's no intersection, create a 1px bbox 0, 0, 1, 1.\n        if x_intersection == 0 or y_intersection == 0:\n            new_xmin = 0\n            new_ymin = 0\n            new_xmax = 1\n            new_ymax = 1\n        else :\n            # Adjust coordinates based on the crop operation.\n            new_xmin = max(0, xmin - crop_pixels)\n            new_ymin = max(0, ymin - crop_pixels)\n            new_xmax = min(xmax - crop_pixels, t_width)\n            new_ymax = min(ymax - crop_pixels, t_height)\n     \n    else: # Translation.\n        tx = kwargs.get(\"tx\")\n        ty = kwargs.get(\"ty\")\n        # Adjust bounding box coordinates based on translation (tx, ty).\n        new_xmin = min(xmin + tx, t_width)\n        new_ymin = min(ymin + ty, t_height)\n        new_xmax = min(xmax + tx, t_width)\n        new_ymax = min(ymax + ty, t_height)\n        \n        # Case original bbox falls outside translated image. \n        if new_xmin == t_width or new_ymin == t_height:\n            new_xmin = 0\n            new_ymin = 0\n            new_xmax = 1\n            new_ymax = 1\n\n    return [new_xmin, new_ymin, new_xmax, new_ymax]","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:11:58.124457Z","iopub.execute_input":"2023-12-09T18:11:58.125852Z","iopub.status.idle":"2023-12-09T18:11:58.142037Z","shell.execute_reply.started":"2023-12-09T18:11:58.125765Z","shell.execute_reply":"2023-12-09T18:11:58.139767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TODO : delete\ndef save_array_to_dcm(image_array, dicom_data, out_path):\n    \"\"\"\n    Function to store a modified image_array back to file type dcm.\n    \"\"\"\n    # Modify dicom_data fields after transformations done in image_array.\n    dicom_data.PixelData = image_array.tobytes()\n    dicom_data.Rows, dicom_data.Columns = image_array.shape\n    dicom_data.BitsStored = 8  # Number of bits used for encoding pixel values.\n    dicom_data.BitsAllocated = 8  # Number of bits allocated in the pixel data.\n    dicom_data.SamplesPerPixel = 1\n    dicom_data.PhotometricInterpretation = \"\" # Avoids MONOCHROME1 correction to be re-applied.\n    try: \n        dicom_data.save_as(out_path)\n    except Exception:\n        return False\n    else:\n        return True","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:11:58.144563Z","iopub.execute_input":"2023-12-09T18:11:58.145248Z","iopub.status.idle":"2023-12-09T18:11:58.166429Z","shell.execute_reply.started":"2023-12-09T18:11:58.145202Z","shell.execute_reply":"2023-12-09T18:11:58.164127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_image_to_csvs(original_path, img_path, img_shape, boxes):\n    \"\"\"\n    Function adds new row to dataset dfs (csvs), with new img info.\n    \"\"\"\n    global train_csv, train_boxes\n\n    img_id = os.path.basename(img_path).replace(\".dcm\", \"_image\")\n    # Update train_csv.\n    original_img_row = train_csv.loc[train_csv.image_path == original_path]\n    new_row = copy.deepcopy(original_img_row)\n    new_row.image_path = img_path\n    new_row.id = img_id\n    train_csv = pd.concat([train_csv, new_row], ignore_index=True)\n    train_csv = train_csv.drop_duplicates()\n\n    # Update train_boxes.\n    original_img_row = train_boxes.loc[train_boxes.image_path == original_path].head(1)\n    new_row = copy.deepcopy(original_img_row)\n    new_row.image_path = img_path\n    new_row.id = img_id\n    new_row['rows'] = img_shape[0]\n    new_row['columns'] = img_shape[1]\n    for box in boxes:\n        new_row.xmin = box[0]\n        new_row.ymin = box[1]\n        new_row.xmax = box[2]\n        new_row.ymax = box[3]\n        train_boxes = pd.concat([train_boxes, new_row], ignore_index=True)\n    train_boxes = train_boxes.drop_duplicates()","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:11:58.172274Z","iopub.execute_input":"2023-12-09T18:11:58.172726Z","iopub.status.idle":"2023-12-09T18:11:58.18692Z","shell.execute_reply.started":"2023-12-09T18:11:58.172687Z","shell.execute_reply":"2023-12-09T18:11:58.184637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def apply_da_techniques(img_ids, transformations, update=True):\n    \"\"\"\n    Functions apply DA techniques listed in transformations to 'train' images, to\n    produce new images from them.\n    Note: Function wont save produced images with 'negative' label since they have no bboxes\n    and are not relevant for future tasks.\n    \"\"\"\n    global train_boxes\n    da_base_path = \"/kaggle/working/tmp/da\"\n    os.makedirs(da_base_path, exist_ok=True)\n\n    boxes = []\n    paths = []\n    # Obtain original image paths and boxes.\n    for im_id in img_ids:\n        slice_boxes = train_boxes.loc[train_boxes['id'] == im_id]\n        # This df can have more than one row with same id.\n        path = slice_boxes['image_path'].iloc[0] \n        img_boxes = [[row['xmin'], row['ymin'], row['xmax'], row['ymax']]\n                     if row['y_label'] != 'negative' else []\n                     for _,row in slice_boxes.iterrows()]\n        boxes.append(img_boxes)\n        paths.append(path)\n    \n    original_boxes = copy.deepcopy(boxes)\n    new_paths = []\n    new_boxes = []\n    \n    # Apply transformations to original image and store it.\n    for i, path in enumerate(paths):\n        im_boxes = [] # Initializing here for case where boxes need no transformation.\n        out_file_prefix = \"da_\"\n        \n        original_image, dicom_data = read_dicom_image(path, dicom_data=True)\n        o_im_shape = original_image.shape\n        da_image = original_image # Allows to acomulate transformations. \n\n        for _, (transformation, kwargs) in enumerate(transformations, start=2):\n            # Create new image from transforming the original.\n            title, da_image = transformation(da_image, **kwargs)\n            t_im_shape = da_image.shape\n            # Recalculate bboxes if needed:\n            if title in [\"Cropping\", \"Horizontal flip\", \"Translation\"]:\n                im_boxes = []\n                for box in boxes[i]:\n                    if not box: # Case label box == none\n                        new_box = []\n                    else:\n                        new_box = recalculate_bounding_box(box,\n                                                           title,\n                                                           t_im_shape,\n                                                           o_im_shape,\n                                                           **kwargs)\n                    if new_box != [0, 0, 1, 1] and new_box:\n                        im_boxes.append(new_box)\n                    \n                boxes[i] = im_boxes # Ensures multiple transformations can be applied.\n            out_file_prefix+= title.lower().replace(\" \", \"_\") + \"_\" # Add transformation to out filename prefix.\n        \n        new_boxes.append(im_boxes) # Appends transformed boxes for image.\n        \n        if im_boxes:\n            # Save new image if there are boxes in it or if test=True. \n            new_path = (os.path\n                        .join(da_base_path, out_file_prefix + os.path.basename(path))\n                        .replace('dcm', 'png'))\n            # Saving processed image.\n            cv2.imwrite(new_path,\n                da_image)\n            new_paths.append(new_path)\n            if update:\n                add_image_to_csvs(path, new_path, t_im_shape, im_boxes)\n    \n    return paths, new_paths, original_boxes, new_boxes","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:11:58.189302Z","iopub.execute_input":"2023-12-09T18:11:58.189803Z","iopub.status.idle":"2023-12-09T18:11:58.204277Z","shell.execute_reply.started":"2023-12-09T18:11:58.189757Z","shell.execute_reply":"2023-12-09T18:11:58.202899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 2.2.2 Visual check","metadata":{}},{"cell_type":"code","source":"# Id's of images to modify.\nimg_ids = train_csv['id'][8:12].tolist()","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:11:58.206146Z","iopub.execute_input":"2023-12-09T18:11:58.206615Z","iopub.status.idle":"2023-12-09T18:11:58.221564Z","shell.execute_reply.started":"2023-12-09T18:11:58.206571Z","shell.execute_reply":"2023-12-09T18:11:58.22049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transformations to apply to each image.\ntransformations = [\n    (flip_image, {'flip_code': 1}),  # 1 for horizontal flip\n    (add_noise, {'mean': 0.1, 'sigma': 25}),\n    (apply_blur, {'kernel_size': (10, 10)})\n]","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:11:58.223205Z","iopub.execute_input":"2023-12-09T18:11:58.224175Z","iopub.status.idle":"2023-12-09T18:11:58.231898Z","shell.execute_reply.started":"2023-12-09T18:11:58.224127Z","shell.execute_reply":"2023-12-09T18:11:58.230985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths, new_paths, boxes, new_boxes = apply_da_techniques(img_ids, transformations, update=False)\n\npaths_to_plot = paths + new_paths\nboxes_to_plot = [list(chain.from_iterable(img_boxes))\n                      for img_boxes in boxes + new_boxes]\nimages_titles = [f\"Original {i}\" if i < len(paths)\n                 else f\"Transformed {i-len(paths)}\"\n                 for i in range(len(paths_to_plot))]\n\nplot_DICOM(paths_to_plot,\n           boxes=boxes_to_plot,\n           title='Comparision before and after DA: H flip + noise + blur',\n           images_titles=images_titles,\n           thickness=20)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-09T18:11:58.232968Z","iopub.execute_input":"2023-12-09T18:11:58.235412Z","iopub.status.idle":"2023-12-09T18:12:10.464724Z","shell.execute_reply.started":"2023-12-09T18:11:58.235369Z","shell.execute_reply":"2023-12-09T18:12:10.463413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transformations = [\n    (translate_image, {'tx': 1800, 'ty': 20}),\n    (add_noise, {'mean': 0.1, 'sigma': 25}),\n    (apply_blur, {'kernel_size': (10, 10)})\n]","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:12:10.466335Z","iopub.execute_input":"2023-12-09T18:12:10.466899Z","iopub.status.idle":"2023-12-09T18:12:10.472025Z","shell.execute_reply.started":"2023-12-09T18:12:10.466862Z","shell.execute_reply":"2023-12-09T18:12:10.470878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths, new_paths, boxes, new_boxes = apply_da_techniques(img_ids, transformations, update=False)\n\npaths_to_plot = paths + new_paths\nboxes_to_plot = [list(chain.from_iterable(img_boxes))\n                      for img_boxes in boxes + new_boxes]\nimages_titles = [f\"Original {i}\" if i < len(paths)\n                 else f\"Transformed {i-len(paths)}\"\n                 for i in range(len(paths_to_plot))]\n\nplot_DICOM(paths_to_plot,\n           boxes=boxes_to_plot,\n           title='Comparision before and after DA: Translation + noise + blur',\n           images_titles=images_titles,\n           thickness=20)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-09T18:12:10.473953Z","iopub.execute_input":"2023-12-09T18:12:10.474601Z","iopub.status.idle":"2023-12-09T18:12:22.002564Z","shell.execute_reply.started":"2023-12-09T18:12:10.474558Z","shell.execute_reply":"2023-12-09T18:12:22.000328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transformations = [\n    (crop_image, {'crop_percent': 0.25}),\n    (add_noise, {'mean': 0.1, 'sigma': 25}),\n    (apply_blur, {'kernel_size': (10, 10)})\n]","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:12:22.004196Z","iopub.execute_input":"2023-12-09T18:12:22.004546Z","iopub.status.idle":"2023-12-09T18:12:22.012005Z","shell.execute_reply.started":"2023-12-09T18:12:22.004515Z","shell.execute_reply":"2023-12-09T18:12:22.010034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths, new_paths, boxes, new_boxes = apply_da_techniques(img_ids, transformations, update=False)\n\npaths_to_plot = paths + new_paths\nboxes_to_plot = [list(chain.from_iterable(img_boxes))\n                      for img_boxes in boxes + new_boxes]\nimages_titles = [f\"Original {i}\" if i < len(paths)\n                 else f\"Transformed {i-len(paths)}\"\n                 for i in range(len(paths_to_plot))]\n\nplot_DICOM(paths_to_plot,\n           boxes=boxes_to_plot,\n           title='Comparision before and after DA: Crop + noise + blur',\n           images_titles=images_titles,\n           thickness=20)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-09T18:12:22.014245Z","iopub.execute_input":"2023-12-09T18:12:22.014638Z","iopub.status.idle":"2023-12-09T18:12:29.341034Z","shell.execute_reply.started":"2023-12-09T18:12:22.014601Z","shell.execute_reply":"2023-12-09T18:12:29.338984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"directory_to_delete = \"/kaggle/working/tmp/da\"\ntry:\n    shutil.rmtree(directory_to_delete)\n    print(f\"Directory '{directory_to_delete}' and its contents successfully removed.\")\nexcept OSError as e:\n    print(f\"Error: {e}\")","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-12-09T18:12:29.342709Z","iopub.execute_input":"2023-12-09T18:12:29.343165Z","iopub.status.idle":"2023-12-09T18:12:29.360409Z","shell.execute_reply.started":"2023-12-09T18:12:29.343125Z","shell.execute_reply":"2023-12-09T18:12:29.35884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 2.2.3 Apply DA strategy","metadata":{}},{"cell_type":"markdown","source":"Now DA techniques will be applied to try to even out the target variable categories. Reaching aproximatelly 2k amount for atypical, indeterminate and typical classes.","metadata":{}},{"cell_type":"code","source":"# Create a new DataFrame with counts aggregated by 'y_label'.\nagg_df = train_csv.groupby('y_label').size().reset_index(name='count')\n\n# Create a bar plot.\nplt.figure(figsize=(8, 4))\nax = sns.barplot(x='y_label',\n                 y='count',\n                 data=agg_df,\n                 palette='viridis')\n\n# Add annotations with counts.\nfor p in ax.patches:\n    ax.annotate(f'{p.get_height()}', (p.get_x() + p.get_width() / 2., p.get_height()),\n                ha='center', va='center', xytext=(0, 10), textcoords='offset points')\n\n# Set labels and title.\nplt.xlabel('Study Labels')\nplt.ylabel('Counts')\nplt.title('Distribution of Study Labels')\n\n# Show the plot\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-09T18:12:29.362298Z","iopub.execute_input":"2023-12-09T18:12:29.363855Z","iopub.status.idle":"2023-12-09T18:12:29.592296Z","shell.execute_reply.started":"2023-12-09T18:12:29.363785Z","shell.execute_reply":"2023-12-09T18:12:29.589729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Atypical:**","metadata":{}},{"cell_type":"code","source":"# Id's of images to modify.\nimg_ids = train_csv.loc[train_csv.y_label == 'atypical']['id'].tolist()","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:12:29.594369Z","iopub.execute_input":"2023-12-09T18:12:29.594827Z","iopub.status.idle":"2023-12-09T18:12:29.605518Z","shell.execute_reply.started":"2023-12-09T18:12:29.59479Z","shell.execute_reply":"2023-12-09T18:12:29.603724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transformations to apply.\ntransformations_set_1 = [\n    (crop_image, {'crop_percent': 0.15}),\n    (add_noise, {'mean': 0, 'sigma': 20}),\n    (apply_blur, {'kernel_size': (3, 3)})\n]\n\ntransformations_set_2 = [\n    (translate_image, {'tx': 50, 'ty': 50}),\n    (add_noise, {'mean': 0, 'sigma': 20}),\n    (apply_blur, {'kernel_size': (3, 3)})\n]\n\ntransformations_set_3 =  [\n    (flip_image, {'flip_code': 1}),  # 1 for horizontal flip\n    (add_noise, {'mean': 0, 'sigma': 15}),\n    (apply_blur, {'kernel_size': (5, 5)})\n]","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:12:29.607579Z","iopub.execute_input":"2023-12-09T18:12:29.608021Z","iopub.status.idle":"2023-12-09T18:12:29.620666Z","shell.execute_reply.started":"2023-12-09T18:12:29.60798Z","shell.execute_reply":"2023-12-09T18:12:29.618778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = apply_da_techniques(img_ids, transformations_set_1)\n_ = apply_da_techniques(img_ids, transformations_set_2)\n_ = apply_da_techniques(img_ids, transformations_set_3)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:12:29.627646Z","iopub.execute_input":"2023-12-09T18:12:29.628074Z","iopub.status.idle":"2023-12-09T18:23:00.511602Z","shell.execute_reply.started":"2023-12-09T18:12:29.628037Z","shell.execute_reply":"2023-12-09T18:23:00.509257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Indeterminate:**","metadata":{}},{"cell_type":"code","source":"# Id's of images to modify.\nimg_ids = train_csv.loc[train_csv.y_label == 'indeterminate']['id'].tolist()","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:23:00.514248Z","iopub.execute_input":"2023-12-09T18:23:00.514731Z","iopub.status.idle":"2023-12-09T18:23:00.526193Z","shell.execute_reply.started":"2023-12-09T18:23:00.514688Z","shell.execute_reply":"2023-12-09T18:23:00.524268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transformations_set_3 =  [\n    (flip_image, {'flip_code': 1}),  # 1 for horizontal flip\n    (add_noise, {'mean': 0, 'sigma': 15}),\n    (apply_blur, {'kernel_size': (5, 5)})\n]","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:23:00.527962Z","iopub.execute_input":"2023-12-09T18:23:00.528742Z","iopub.status.idle":"2023-12-09T18:23:00.545523Z","shell.execute_reply.started":"2023-12-09T18:23:00.528686Z","shell.execute_reply":"2023-12-09T18:23:00.543807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = apply_da_techniques(img_ids, transformations_set_3)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:23:00.547069Z","iopub.execute_input":"2023-12-09T18:23:00.548239Z","iopub.status.idle":"2023-12-09T18:35:46.269121Z","shell.execute_reply.started":"2023-12-09T18:23:00.548192Z","shell.execute_reply":"2023-12-09T18:35:46.267977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Typical**:","metadata":{}},{"cell_type":"code","source":"# Random selection of rows.\nsample = train_csv[train_csv.y_label == 'typical'].sample(n=800)\n\n# Drop the selected rows from the DataFrame.\ntrain_csv = train_csv[~train_csv.id.isin(sample.id.to_list())]\n\n# Delete rows with id in sample.\ntrain_boxes = train_boxes[~train_boxes.id.isin(sample.id.to_list())]    ","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:35:46.270871Z","iopub.execute_input":"2023-12-09T18:35:46.271214Z","iopub.status.idle":"2023-12-09T18:35:46.292071Z","shell.execute_reply.started":"2023-12-09T18:35:46.271188Z","shell.execute_reply":"2023-12-09T18:35:46.289336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Resulting distribution:","metadata":{}},{"cell_type":"code","source":"# Create a new DataFrame with counts aggregated by 'y_label'.\nagg_df = train_csv.groupby('y_label').size().reset_index(name='count')\n\n# Create a bar plot.\nplt.figure(figsize=(8, 4))\nax = sns.barplot(x='y_label',\n                 y='count',\n                 data=agg_df,\n                 palette='viridis')\n\n# Add annotations with counts.\nfor p in ax.patches:\n    ax.annotate(f'{p.get_height()}', (p.get_x() + p.get_width() / 2., p.get_height()),\n                ha='center', va='center', xytext=(0, 10), textcoords='offset points')\n\n# Set labels and title.\nplt.xlabel('Study Labels')\nplt.ylabel('Counts')\nplt.title('Distribution of Study Labels')\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:35:46.29466Z","iopub.execute_input":"2023-12-09T18:35:46.295381Z","iopub.status.idle":"2023-12-09T18:35:46.553805Z","shell.execute_reply.started":"2023-12-09T18:35:46.295313Z","shell.execute_reply":"2023-12-09T18:35:46.551606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Image Enhancing","metadata":{}},{"cell_type":"markdown","source":"- If time, data enhancing and   https://www.kaggle.com/code/rerere/covid-19-image-enhancement \n","metadata":{}},{"cell_type":"markdown","source":"## 4. Image Resizing","metadata":{}},{"cell_type":"markdown","source":"### 4.1 Utility functions","metadata":{}},{"cell_type":"code","source":"def resize_image(image_array, target_shape, keep_aspect_ratio=False):\n    \"\"\"\n    Function resizes the image_array to the target shape maintaining the aspect\n    ratio or not depending on the value of 'keep_aspect_ratio'.\n    \"\"\"\n    if keep_aspect_ratio:\n        # Calculate the aspect ratio.\n        aspect_ratio = image_array.shape[1] / image_array.shape[0]\n\n        new_width = int(target_shape[0] * aspect_ratio)\n\n        resized_image = cv2.resize(image_array,\n                                   (new_width, target_shape[0]))\n    else:\n        # Resize the image without maintaining aspect ratio\n        resized_image = cv2.resize(image_array,\n                                   (target_shape[1], target_shape[0]))\n\n    return resized_image","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:35:46.555485Z","iopub.execute_input":"2023-12-09T18:35:46.555882Z","iopub.status.idle":"2023-12-09T18:35:46.563314Z","shell.execute_reply.started":"2023-12-09T18:35:46.555848Z","shell.execute_reply":"2023-12-09T18:35:46.562152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def scale_bbox(original_shape, resized_shape, bbox):\n    \"\"\"\n    Function scales the bounding boxes according to the size of the resized image. \n    \"\"\"\n    # Get scaling factor.\n    scale_x = resized_shape[1]/original_shape[1]\n    scale_y = resized_shape[0]/original_shape[0]\n\n    x = int(bbox[0]*scale_x) # xmin.\n    y = int(bbox[1]*scale_y) # ymin.\n    x1 = int(bbox[2]*scale_x) # xmax.\n    y1= int(bbox[3]*scale_y) # ymax.\n        \n    return [x, y, x1, y1]","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:35:46.56461Z","iopub.execute_input":"2023-12-09T18:35:46.564936Z","iopub.status.idle":"2023-12-09T18:35:46.579325Z","shell.execute_reply.started":"2023-12-09T18:35:46.564906Z","shell.execute_reply":"2023-12-09T18:35:46.577783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4.2 Examples of resizes:","metadata":{}},{"cell_type":"code","source":"images_paths = train_csv[\"image_path\"][:4].tolist()\nplot_DICOM(images_paths,\n           title=\"Resized images keeping aspect ratio\",\n           resize=True,\n           resize_shape=TARGET_SHAPE,\n           kar=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:35:46.582388Z","iopub.execute_input":"2023-12-09T18:35:46.583053Z","iopub.status.idle":"2023-12-09T18:35:48.220308Z","shell.execute_reply.started":"2023-12-09T18:35:46.583004Z","shell.execute_reply":"2023-12-09T18:35:48.218263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_DICOM(images_paths,\n           title=\"Resized images without keeping aspect ratio\",\n           resize=True,\n           resize_shape=TARGET_SHAPE,\n           kar=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:35:48.221957Z","iopub.execute_input":"2023-12-09T18:35:48.223069Z","iopub.status.idle":"2023-12-09T18:35:49.437887Z","shell.execute_reply.started":"2023-12-09T18:35:48.223028Z","shell.execute_reply":"2023-12-09T18:35:49.437126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5 Processing the full dataset:","metadata":{}},{"cell_type":"markdown","source":"The dataset processing includes:\n\n1. Resizing images and changing format to png.\n2. Performing image enhancing if option is set to True.\n3. Modifications in CSVs:\n    - Store new shape after resizing.\n    - Update image paths.\n    - Recalculate new bounding boxes after resizing. ","metadata":{}},{"cell_type":"code","source":"def process_and_save_image(file_path, save_dir, split_info, results, enhancing, target_shape, kar):\n    \"\"\"\n    Function processes an image and its corresponding rows, from dataset info dataframes,\n    and saves processed image in png format.\n    \"\"\"\n    def _update_rows(row, boxes=False):\n        \"\"\"\n        Updates df row after resizing, updating also the image path.\n        \"\"\"\n        if boxes:\n            if row['box_label'].startswith('opacity'):\n                bbox = [row['xmin'], row['ymin'], row['xmax'], row['ymax']]\n                row['xmin'], row['ymin'], row['xmax'], row['ymax'] = scale_bbox(original_shape,\n                                                                                resized_shape, bbox)\n        row['image_path'], row['rows_resized'], row['columns_resized'] = [f'{split_info[0]}/{filename}',\n                                                                            int(resized_shape[1]),\n                                                                            int(resized_shape[0])]\n        return row\n\n    filename = os.path.basename(file_path)\n    \n    if filename.endswith('dcm'):\n        # Reading image and resizing.\n        image_array = read_dicom_image(file_path)\n        filename = filename.replace('dcm', 'png') # Filname after processing.\n    else:\n        image_array = cv2.imread(file_path)\n    original_shape = list(image_array.shape)\n    image_array = resize_image(image_array,\n                               target_shape=target_shape,\n                               keep_aspect_ratio=kar)\n    resized_shape = list(image_array.shape)\n\n    if enhancing:\n        pass # Add it if there is time.\n    \n    # Changing image_path from dataframes and adding new im shape.\n    slice_df = split_info[1][split_info[1]['image_path'] == file_path]\n    rows_to_update = slice_df.copy()\n    rows_to_update = rows_to_update.apply(_update_rows, axis=1)\n    \n    # Adding modifications to the manager.\n    results[0].append(rows_to_update.to_dict(orient='records'))\n    \n    # Process train_boxes.csv.\n    if split_info[0] == 'train':\n        # Locating rows for processed image.  \n        slice_df = split_info[2][split_info[2]['image_path'] == file_path]\n\n        rows_to_update = slice_df.copy()\n        rows_to_update = rows_to_update.apply(_update_rows, axis=1, boxes=True)\n        # Adding modifications to the manager.\n        results[1].append(rows_to_update.to_dict(orient='records'))\n   \n    # Saving processed image.\n    cv2.imwrite(os.path.join(save_dir, filename),\n                image_array)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:35:49.438988Z","iopub.execute_input":"2023-12-09T18:35:49.440174Z","iopub.status.idle":"2023-12-09T18:35:49.456224Z","shell.execute_reply.started":"2023-12-09T18:35:49.440053Z","shell.execute_reply":"2023-12-09T18:35:49.453952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_split(split_info, enhancing=False, target_shape=(256,256), kar=False):\n    \"\"\"\n    Using multi processing, function processes the images asociated with a given split.\n    Updates provided split_info dataframes after the processing and stores them in a csv.\n    args:\n        split_info: List where first element is the name of the split (str), second\n                    element is a copy of the corresponding split df (df), and if the split\n                    is the 'train' split, there will be a third item for a copy of the\n                    train_boxes df (df).\n        enhancing: If true, image processing to improve image quality is applied. (bool)\n        target_shape: Target shape for the resizing (cols, rows) format. (tuple)\n        kar: If true keeps the aspect ratio for images when resizing (image shapes might vary\n             depending on the image). Default is false and it producs a square image. (bool)\n    \"\"\"\n\n    save_dir = f'/kaggle/working/tmp/{split_info[0]}/'\n    os.makedirs(save_dir, exist_ok=True)\n    \n    # Create new columns in df for resized shape.\n    split_info[1]['rows_resized'] = None\n    split_info[1]['columns_resized'] = None\n    split_info[1] = split_info[1].reset_index()\n    \n    if len(split_info)==3:\n        split_info[2]['rows_resized'] = None\n        split_info[2]['columns_resized'] = None\n        split_info[2] = split_info[2].reset_index()\n\n    with Manager() as manager:\n        # Dict to store the dataframes modifications in each process.\n        results = [manager.list(), manager.list()]\n    \n        with multiprocessing.Pool() as pool:\n            jobs = []\n            for file_path in split_info[1]['image_path']:\n                job = pool.apply_async(process_and_save_image,\n                                       args=(file_path,\n                                             save_dir,\n                                             split_info,\n                                             results,\n                                             enhancing,\n                                             target_shape,\n                                             kar))\n                jobs.append(job)\n            \n            # Wait for all processes to finish\n            for job in jobs:\n                job.get()\n\n        # Combine the results from each process.\n        updated_rows = pd.DataFrame([v for sublist in results[0] for v in sublist])\n        updated_rows_boxes = pd.DataFrame([v for sublist in results[1] for v in sublist])\n\n        if not updated_rows.empty:\n            \n            # Update original df.\n            columns_to_update = updated_rows.columns.difference(['index'])\n            split_info[1].loc[split_info[1]['index'].isin(updated_rows['index']),\n                              columns_to_update] = updated_rows[columns_to_update].values\n            split_info[1] = split_info[1].drop('index', axis=1, errors='ignore')\n\n        # Save split info csv with modifications.\n        split_info[1].to_csv(f'/kaggle/working/{split_info[0]}_p.csv',\n                            index=False)\n\n        # Concatenate all the updated rows\n        if not updated_rows_boxes.empty:\n            # Update the original DataFrame.\n            columns_to_update = updated_rows_boxes.columns.difference(['index'])\n            split_info[2].loc[split_info[2]['index'].isin(updated_rows_boxes['index']),\n                              columns_to_update] = updated_rows_boxes[columns_to_update].values\n            split_info[2] = split_info[2].drop('index', axis=1, errors='ignore')\n\n            # Save train_boxes csv with modifiecations.\n            split_info[2].to_csv(f'/kaggle/working/{split_info[0]}_boxes_p.csv',\n                                 index=False)    ","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:35:49.458634Z","iopub.execute_input":"2023-12-09T18:35:49.459144Z","iopub.status.idle":"2023-12-09T18:35:49.478645Z","shell.execute_reply.started":"2023-12-09T18:35:49.459105Z","shell.execute_reply":"2023-12-09T18:35:49.476847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"splits = [['train', train_csv.copy(),\n           train_boxes.copy()],\n          ['test', test_csv.copy()]]\n\n\nfor split in splits:\n    process_split(split, TARGET_SHAPE)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:35:49.481423Z","iopub.execute_input":"2023-12-09T18:35:49.482018Z","iopub.status.idle":"2023-12-09T18:45:46.810935Z","shell.execute_reply.started":"2023-12-09T18:35:49.481977Z","shell.execute_reply":"2023-12-09T18:45:46.808945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n!tar -zcf train.tar.gz -C \"/kaggle/working/tmp/train/\" .\n!tar -zcf test.tar.gz -C \"/kaggle/working/tmp/test/\" .","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:47:49.156853Z","iopub.execute_input":"2023-12-09T18:47:49.157336Z","iopub.status.idle":"2023-12-09T18:48:14.958384Z","shell.execute_reply.started":"2023-12-09T18:47:49.1573Z","shell.execute_reply":"2023-12-09T18:48:14.957134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5.1 Image comparison before and after processing.","metadata":{}},{"cell_type":"code","source":"img_ids = train_csv['id'][:4].tolist()","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:45:47.516721Z","iopub.execute_input":"2023-12-09T18:45:47.517316Z","iopub.status.idle":"2023-12-09T18:45:47.525555Z","shell.execute_reply.started":"2023-12-09T18:45:47.517269Z","shell.execute_reply":"2023-12-09T18:45:47.523972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 5.1.1 Images before processing","metadata":{}},{"cell_type":"code","source":"boxes = []\nfor im_id in img_ids:\n    slice_boxes = train_boxes.loc[train_boxes['id'] == im_id]\n    img_boxes = [[row['xmin'], row['ymin'], row['xmax'], row['ymax']]\n                 for _,row in slice_boxes.iterrows()] \n    boxes.append(list(chain.from_iterable(img_boxes))) # Reformat for plotting.","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:45:47.527731Z","iopub.execute_input":"2023-12-09T18:45:47.528207Z","iopub.status.idle":"2023-12-09T18:45:47.548729Z","shell.execute_reply.started":"2023-12-09T18:45:47.528164Z","shell.execute_reply":"2023-12-09T18:45:47.547167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_DICOM(train_csv['image_path'][:4],\n           boxes=boxes,\n           title='Images before processing',\n           thickness=20)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:45:47.551285Z","iopub.execute_input":"2023-12-09T18:45:47.55239Z","iopub.status.idle":"2023-12-09T18:45:53.633052Z","shell.execute_reply.started":"2023-12-09T18:45:47.552287Z","shell.execute_reply":"2023-12-09T18:45:53.631753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 5.1.2 Images after processing","metadata":{}},{"cell_type":"code","source":"train_csv_p = pd.read_csv(\"/kaggle/working/train_p.csv\")\ntrain_boxes_p = pd.read_csv(\"/kaggle/working/train_boxes_p.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:45:53.634876Z","iopub.execute_input":"2023-12-09T18:45:53.636143Z","iopub.status.idle":"2023-12-09T18:45:53.697658Z","shell.execute_reply.started":"2023-12-09T18:45:53.636045Z","shell.execute_reply":"2023-12-09T18:45:53.695917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"boxes = []\npaths = []\nfor im_id in img_ids:\n    slice_boxes = train_boxes_p.loc[train_boxes_p['id'] == im_id]\n    path = os.path.join(\"/kaggle/working/tmp\", slice_boxes['image_path'].iloc[0])\n    img_boxes = [[row['xmin'], row['ymin'], row['xmax'], row['ymax']]\n                 for _,row in slice_boxes.iterrows()]\n    boxes.append(list(chain.from_iterable(img_boxes))) # Reformat for plotting.\n    paths.append(path)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:45:53.699303Z","iopub.execute_input":"2023-12-09T18:45:53.699687Z","iopub.status.idle":"2023-12-09T18:45:53.718498Z","shell.execute_reply.started":"2023-12-09T18:45:53.699654Z","shell.execute_reply":"2023-12-09T18:45:53.716679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_DICOM(paths,\n           boxes=boxes,\n           title='Images after processing')","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:45:53.720771Z","iopub.execute_input":"2023-12-09T18:45:53.72126Z","iopub.status.idle":"2023-12-09T18:45:54.660449Z","shell.execute_reply.started":"2023-12-09T18:45:53.72119Z","shell.execute_reply":"2023-12-09T18:45:54.659176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"directory_to_delete = \"/kaggle/working/tmp\"\ntry:\n    shutil.rmtree(directory_to_delete)\n    print(f\"Directory '{directory_to_delete}' and its contents successfully removed.\")\nexcept OSError as e:\n    print(f\"Error: {e}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-12-09T18:56:18.135214Z","iopub.execute_input":"2023-12-09T18:56:18.135976Z","iopub.status.idle":"2023-12-09T18:56:19.10676Z","shell.execute_reply.started":"2023-12-09T18:56:18.135922Z","shell.execute_reply":"2023-12-09T18:56:19.105444Z"},"trusted":true},"execution_count":null,"outputs":[]}]}