{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":26680,"databundleVersionId":2283525,"sourceType":"competition"}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### EDA SIIM-COVID19-DETECTION","metadata":{}},{"cell_type":"markdown","source":"Exploratory data analysis for SIIM-COVID19-DETECTION DATASET. URL: https://www.kaggle.com/competitions/siim-covid19-detection/","metadata":{}},{"cell_type":"markdown","source":"## 0. Libraries and global variables:","metadata":{}},{"cell_type":"code","source":"!conda config --remove channels <file:///tmp/conda>\n!conda install gdcm -c conda-forge -y","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-12-02T09:56:05.357622Z","iopub.execute_input":"2023-12-02T09:56:05.358090Z","iopub.status.idle":"2023-12-02T09:57:57.898165Z","shell.execute_reply.started":"2023-12-02T09:56:05.358051Z","shell.execute_reply":"2023-12-02T09:57:57.897155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport glob\nimport math\nimport re\nimport warnings\nimport numpy as np\nimport pandas as pd \nimport cv2\nimport matplotlib.pyplot as plt\nimport matplotlib.gridspec as gridspec\nimport seaborn as sns\nimport pydicom\n#import gdcm # Needed for decoding images of De-identification methods != DICOM locally.\n\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom random import randint\nfrom tqdm.auto import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-12-02T09:57:57.900656Z","iopub.execute_input":"2023-12-02T09:57:57.901015Z","iopub.status.idle":"2023-12-02T09:57:59.473044Z","shell.execute_reply.started":"2023-12-02T09:57:57.900978Z","shell.execute_reply":"2023-12-02T09:57:59.472194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# GLOBAL variables\nbase_path = \"../input/siim-covid19-detection/\"\ntrain_path = os.path.join(base_path, \"train\")\ntest_path = os.path.join(base_path, \"test\")","metadata":{"execution":{"iopub.status.busy":"2023-12-02T09:57:59.474222Z","iopub.execute_input":"2023-12-02T09:57:59.474658Z","iopub.status.idle":"2023-12-02T09:57:59.479333Z","shell.execute_reply.started":"2023-12-02T09:57:59.474631Z","shell.execute_reply":"2023-12-02T09:57:59.478623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filter out the specific warning\nwarnings.filterwarnings(\"ignore\", message=\"Invalid value for VR UI\")","metadata":{"execution":{"iopub.status.busy":"2023-12-02T09:57:59.481413Z","iopub.execute_input":"2023-12-02T09:57:59.481968Z","iopub.status.idle":"2023-12-02T09:57:59.494566Z","shell.execute_reply.started":"2023-12-02T09:57:59.481939Z","shell.execute_reply":"2023-12-02T09:57:59.493655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. Data description:","metadata":{}},{"cell_type":"markdown","source":"### 1.1 Labels data:","metadata":{}},{"cell_type":"code","source":"train_roi_labels = pd.read_csv(os.path.join(base_path,\"train_image_level.csv\"))\ntrain_roi_labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T09:57:59.495901Z","iopub.execute_input":"2023-12-02T09:57:59.496183Z","iopub.status.idle":"2023-12-02T09:57:59.577847Z","shell.execute_reply.started":"2023-12-02T09:57:59.496158Z","shell.execute_reply":"2023-12-02T09:57:59.576828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of instances\nlen(train_roi_labels)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T09:57:59.579031Z","iopub.execute_input":"2023-12-02T09:57:59.580074Z","iopub.status.idle":"2023-12-02T09:57:59.585707Z","shell.execute_reply.started":"2023-12-02T09:57:59.580033Z","shell.execute_reply":"2023-12-02T09:57:59.584662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**boxes** - bounding boxes in easily-readable dictionary format\n\n**label** - the correct prediction label for the provided bounding boxes","metadata":{}},{"cell_type":"code","source":"train_study_labels = pd.read_csv(os.path.join(base_path,\"train_study_level.csv\"))\ntrain_study_labels.loc[:, \"StudyInstanceUID\"] = train_study_labels.id.str.split(\"_\", expand=True).loc[:, 0]\ndel train_study_labels['id']\ntrain_study_labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T09:57:59.587181Z","iopub.execute_input":"2023-12-02T09:57:59.587857Z","iopub.status.idle":"2023-12-02T09:57:59.631867Z","shell.execute_reply.started":"2023-12-02T09:57:59.587817Z","shell.execute_reply":"2023-12-02T09:57:59.630668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All images are stored in paths with the form study/series/image. The study ID here relates directly to the study-level predictions, and the image ID is the ID used for image-level predictions.","metadata":{}},{"cell_type":"markdown","source":"**Negative for Pneumonia** \t1 : if the study is negative for pneumonia, 0: otherwise\n\n**Typical Appearance** \t1: if the study has this appearance, 0: otherwise\n\n**Indeterminate Appearance** \t1: if the study has this appearance, 0: otherwise\n\n**Atypical Appearance** \t1: if the study has this appearance, 0: otherwise","metadata":{}},{"cell_type":"code","source":"# Number of instances\nlen(train_study_labels)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T09:57:59.633349Z","iopub.execute_input":"2023-12-02T09:57:59.633680Z","iopub.status.idle":"2023-12-02T09:57:59.640033Z","shell.execute_reply.started":"2023-12-02T09:57:59.633653Z","shell.execute_reply":"2023-12-02T09:57:59.639003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Probably some studies have more than one image. ","metadata":{}},{"cell_type":"markdown","source":"#### 1.1.1 Joining data labels:","metadata":{}},{"cell_type":"code","source":"train_labels = train_roi_labels.merge(train_study_labels, on='StudyInstanceUID')\ntrain_labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T09:57:59.641491Z","iopub.execute_input":"2023-12-02T09:57:59.641875Z","iopub.status.idle":"2023-12-02T09:57:59.675473Z","shell.execute_reply.started":"2023-12-02T09:57:59.641838Z","shell.execute_reply":"2023-12-02T09:57:59.674694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 1.1.2 Adding image paths:","metadata":{}},{"cell_type":"code","source":"def create_paths(row):\n    \"\"\"\n    Function to create image paths from df info.\n    \"\"\"\n\n    return glob.glob(os.path.join(train_path,\n                                  str(row.StudyInstanceUID),\n                                   \"*\",\n                                   str(row.id.split(\"_\")[0]) + \".dcm\"))[0]","metadata":{"execution":{"iopub.status.busy":"2023-12-02T09:57:59.678991Z","iopub.execute_input":"2023-12-02T09:57:59.679318Z","iopub.status.idle":"2023-12-02T09:57:59.684791Z","shell.execute_reply.started":"2023-12-02T09:57:59.679291Z","shell.execute_reply":"2023-12-02T09:57:59.683715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels['image_path'] = train_labels.apply(create_paths, axis=1)\ntrain_labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T09:57:59.686402Z","iopub.execute_input":"2023-12-02T09:57:59.686812Z","iopub.status.idle":"2023-12-02T10:00:37.750337Z","shell.execute_reply.started":"2023-12-02T09:57:59.686774Z","shell.execute_reply":"2023-12-02T10:00:37.749273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Adding integer label.\ndef label_row(row):\n    if row['Typical Appearance'] == 1:\n        return 0, 'typical'\n    elif row['Indeterminate Appearance'] == 1:\n        return 1, 'indeterminate'\n    elif row['Atypical Appearance'] == 1:\n        return 2, 'atypical'\n    else:\n        return 3, 'negative'\n\ntrain_labels['int_label'], train_labels['y_label'] = zip(*train_labels.apply(label_row,\n                                                                             axis=1))","metadata":{"execution":{"iopub.status.busy":"2023-12-02T11:11:48.279396Z","iopub.execute_input":"2023-12-02T11:11:48.280038Z","iopub.status.idle":"2023-12-02T11:11:48.359803Z","shell.execute_reply.started":"2023-12-02T11:11:48.279991Z","shell.execute_reply":"2023-12-02T11:11:48.358577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T11:11:59.679901Z","iopub.execute_input":"2023-12-02T11:11:59.680783Z","iopub.status.idle":"2023-12-02T11:11:59.696549Z","shell.execute_reply.started":"2023-12-02T11:11:59.680739Z","shell.execute_reply":"2023-12-02T11:11:59.695228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.to_csv(os.path.join(\"train.csv\"), index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T11:12:18.533701Z","iopub.execute_input":"2023-12-02T11:12:18.534157Z","iopub.status.idle":"2023-12-02T11:12:18.609157Z","shell.execute_reply.started":"2023-12-02T11:12:18.534120Z","shell.execute_reply":"2023-12-02T11:12:18.607909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 1.1.3 Missing values:","metadata":{}},{"cell_type":"code","source":"# Check for nulls.\ntrain_labels.info()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T11:12:23.111488Z","iopub.execute_input":"2023-12-02T11:12:23.111938Z","iopub.status.idle":"2023-12-02T11:12:23.128705Z","shell.execute_reply.started":"2023-12-02T11:12:23.111902Z","shell.execute_reply":"2023-12-02T11:12:23.127580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Variable **boxes** with an easily-readable dictionary format sometimes is empty. **label** info should be used instad. ","metadata":{}},{"cell_type":"code","source":"# Check for empty strings in each column.\nempty_string_counts = train_labels.apply(lambda col: col[col == ''].count())\nprint(empty_string_counts)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T11:12:37.783941Z","iopub.execute_input":"2023-12-02T11:12:37.784364Z","iopub.status.idle":"2023-12-02T11:12:37.800142Z","shell.execute_reply.started":"2023-12-02T11:12:37.784328Z","shell.execute_reply":"2023-12-02T11:12:37.798893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 1.1.4 Generating a dict to describe all the boxes in each image:","metadata":{}},{"cell_type":"code","source":"def process_row(row):\n    return train_labels.loc[int(row.name[0]), \"id\"]\n\n### Extraer los valores de 'label' usando una expresión regular\ntrain_boxes = (train_labels['label']\n                    .str.extractall(r'(opacity 1|none 1) (\\d+.\\d+|\\d+) (\\d+.\\d+|\\d+) (\\d+.\\d+|\\d+) (\\d+.\\d+|\\d+)'))\n\n# Renombrar las columnas extraídas\ntrain_boxes.columns = ['box_label', 'xmin', 'ymin', 'xmax', 'ymax']\n\ntrain_boxes[['xmin', 'ymin', 'xmax', 'ymax']] = train_boxes[['xmin', 'ymin', 'xmax', 'ymax']].astype(float)\n\ntrain_boxes[\"id\"] = train_boxes.apply(process_row, axis=1)\n\ntrain_boxes = train_boxes.reset_index()\n\ntrain_boxes = pd.merge(train_boxes, train_labels, on='id')\n\ndel train_boxes[\"level_0\"], train_boxes[\"match\"], train_boxes[\"boxes\"], train_boxes[\"label\"]\n\ntrain_boxes.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T11:12:42.089437Z","iopub.execute_input":"2023-12-02T11:12:42.089873Z","iopub.status.idle":"2023-12-02T11:12:42.645074Z","shell.execute_reply.started":"2023-12-02T11:12:42.089827Z","shell.execute_reply":"2023-12-02T11:12:42.643754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 1.1.5 Creating a csv for test data:","metadata":{}},{"cell_type":"code","source":"def get_filenames_df(folder_path):\n    \"\"\"\n    Function reads all image files in a given directory and creates\n    a dataframe with the image id and its full path.\n    \"\"\"\n    filenames_dict = {'id': [], 'image_path': []}\n\n    for dirname, _, filenames in tqdm(os.walk(folder_path)):\n        for file in filenames:\n            full_path = os.path.join(dirname, file)\n            filenames_dict['image_path'].append(full_path)\n            \n            filename_without_extension = file.split('.')[0]\n            filenames_dict['id'].append(f'{filename_without_extension}_image')\n\n    return pd.DataFrame(filenames_dict)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T11:13:08.629500Z","iopub.execute_input":"2023-12-02T11:13:08.630484Z","iopub.status.idle":"2023-12-02T11:13:08.637207Z","shell.execute_reply.started":"2023-12-02T11:13:08.630439Z","shell.execute_reply":"2023-12-02T11:13:08.636298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_csv = get_filenames_df(test_path)\ntest_csv.to_csv(os.path.join(\"test.csv\"), index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:00:38.120923Z","iopub.execute_input":"2023-12-02T10:00:38.121201Z","iopub.status.idle":"2023-12-02T10:01:06.242340Z","shell.execute_reply.started":"2023-12-02T10:00:38.121176Z","shell.execute_reply":"2023-12-02T10:01:06.241312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.2 Image level data:","metadata":{}},{"cell_type":"markdown","source":"#### 1.2.1 DICOM files exploration:","metadata":{}},{"cell_type":"code","source":"voi_lut=True\nfix_monochrome=True\n\n\ndef extract_dicom_data(filename, func='get_image'):\n    \"\"\"Credit: https://github.com/pydicom/pydicom/issues/319\n               https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n               \n    Function corrects dicom pixel data to prevent x-rays from looking inverted and extracts metadata.\n    args:\n        filename: DICOM file path (str)\n        func: 'get_image' to extract only image data from file. (str)\n              'get_metadata' to extract only metadata from file. (str)\n              'get_all' to extract metadata, and image data (str)\n    returns:\n         'get_image' to extract only image data from file. (pixel array)\n         'get_metadata' to extract only metadata from file. (dict)\n         'get_all' to extract metadata, and image data ((dict, pixel array))\n    \"\"\"\n    def _sanitise_unicode(s):\n        return s.replace(u\"\\u0000\", \"\").strip()\n\n    def _convert_value(v):\n        t = type(v)\n        if t in (list, int, float):\n            cv = v\n        elif t == str:\n            cv = _sanitise_unicode(v)\n        elif t == bytes:\n            s = v.decode('ascii', 'replace')\n            cv = _sanitise_unicode(s)\n        elif t == pydicom.valuerep.DSfloat:\n            cv = float(v)\n        elif t == pydicom.valuerep.IS:\n            cv = int(v)\n        else:\n            cv = repr(v)\n        return cv\n    \n    assert func in ['get_image', 'get_metadata', 'get_all'], f\"Variable has an unexpected value: {func}\"\n\n    dicom_header = pydicom.dcmread(filename) \n    \n    if func != 'get_metadata':\n        #====== DICOM IMAGE DATA ======\n        # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \"human-friendly\" view\n        if voi_lut:\n            data = apply_voi_lut(dicom_header.pixel_array, dicom_header)\n        else:\n            data = dicom_header.pixel_array\n        # depending on this value, X-ray may look inverted - fix that:\n        if fix_monochrome and dicom_header.PhotometricInterpretation == \"MONOCHROME1\":\n            data = np.amax(data) - data\n        data = data - np.min(data)\n        data = data / np.max(data)\n        modified_image_data = (data * 255).astype(np.uint8)\n        \n        if func == 'get_image':\n            return modified_image_data\n    \n    elif func != 'get_image':\n        #====== DICOM FILE DATA ======\n        dicom_dict = {}\n        repr(dicom_header)\n        for dicom_value in dicom_header.values():\n            if dicom_value.tag == (0x7fe0, 0x0010):\n                #discard pixel data\n                continue\n            if type(dicom_value.value) == pydicom.dataset.Dataset:\n                dicom_dict[dicom_value.name] = dicom_dataset_to_dict(dicom_value.value)\n            else:\n                v = _convert_value(dicom_value.value)\n                dicom_dict[dicom_value.name] = v\n\n        del dicom_dict['Pixel Representation']\n\n        if func == 'get_metadata':\n            return dicom_dict\n    else:\n        return dicom_dict, modified_image_data","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:01:06.243948Z","iopub.execute_input":"2023-12-02T10:01:06.244664Z","iopub.status.idle":"2023-12-02T10:01:06.256336Z","shell.execute_reply.started":"2023-12-02T10:01:06.244632Z","shell.execute_reply":"2023-12-02T10:01:06.255487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_DICOM(images_paths, boxes=None, title=None,\n               images_titles=None, resize_shape=(2000,2000)):\n    \"\"\"\n    Function to plot a variable number of DICOM images.\n    args:\n        images_paths: list of image paths. (list of strings)\n        boxes: List of bounding boxes for each image (list of lists of floats)\n        title: String for figure title. (str)\n        images_titles: List of strings for image titles. (list of strings)\n    \"\"\"\n\n    num_images = len(images_paths)\n    \n    cols = int(math.ceil(math.sqrt(num_images)))\n    rows = int(math.ceil(num_images / cols))\n    \n    fig = plt.figure(figsize=(9, 8))\n    gs = gridspec.GridSpec(rows, cols,\n                           width_ratios=[1] * cols,\n                           height_ratios=[1] * rows)\n    \n    for i, image_path in enumerate(images_paths):\n    \n        image_data = extract_dicom_data(image_path)\n        image_data = np.stack([image_data, image_data, image_data],\n                              axis=-1) # To RGB.\n        \n        # Plot bounding boxes if provided.\n        if boxes is not None:\n            if not boxes[i]:\n                print(f'There is no ROI present in image {i + 1}.')\n            elif len(boxes[i])%4!=0:\n                print(f'Boxes provided have wrong format for image {i + 1}.')\n            else:\n                for n_box in range(0, int(len(boxes[i])/4)):\n                    color = [randint(0, 255), randint(0, 255), randint(0, 255)]\n\n                    x = int(boxes[i][0 + n_box*4])\n                    y = int(boxes[i][1 + n_box*4])\n                    w = int(boxes[i][2 + n_box*4])\n                    h = int(boxes[i][3 + n_box*4])\n                    \n                    image_data = cv2.rectangle(image_data, (x, y), (w, h), color=color, thickness=20)\n\n        image_data = cv2.resize(image_data, resize_shape)\n        ax = plt.subplot(gs[i])\n        ax.grid(False)\n        ax.imshow(image_data)\n\n        if images_titles is not None:\n            ax.set_title(images_titles[i])\n        else:\n            ax.set_title(f'Image {i + 1}')\n\n    if title is not None:\n        plt.suptitle(title)\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:01:06.257627Z","iopub.execute_input":"2023-12-02T10:01:06.258389Z","iopub.status.idle":"2023-12-02T10:01:06.273221Z","shell.execute_reply.started":"2023-12-02T10:01:06.258360Z","shell.execute_reply":"2023-12-02T10:01:06.272398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Examples of DICOM files:","metadata":{}},{"cell_type":"code","source":"imgs_test_path = [train_labels.iloc[1]['image_path'],\n                  train_labels.iloc[3]['image_path'],\n                  train_labels.iloc[2]['image_path'],\n                  train_labels.iloc[155]['image_path']]\n\nplot_DICOM(imgs_test_path, title=\"Examples\")","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:01:06.274327Z","iopub.execute_input":"2023-12-02T10:01:06.274636Z","iopub.status.idle":"2023-12-02T10:01:10.790901Z","shell.execute_reply.started":"2023-12-02T10:01:06.274608Z","shell.execute_reply":"2023-12-02T10:01:10.789751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image shapes:\ndcm_1 = pydicom.dcmread(imgs_test_path[0])\nprint(dcm_1.pixel_array.shape)\n\ndcm_2 = pydicom.dcmread(imgs_test_path[2])\nprint(dcm_2.pixel_array.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:01:10.792297Z","iopub.execute_input":"2023-12-02T10:01:10.792623Z","iopub.status.idle":"2023-12-02T10:01:10.819805Z","shell.execute_reply.started":"2023-12-02T10:01:10.792596Z","shell.execute_reply":"2023-12-02T10:01:10.818831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Image data is given in DICOM format and have variable shapes but all of them are quite large. Resize will be needed for future classification tasks. ","metadata":{}},{"cell_type":"markdown","source":"DICOM files contain also metadata:","metadata":{}},{"cell_type":"code","source":"print(f'METADATA for DICOM file: {imgs_test_path[0]} \\n')\nextract_dicom_data(imgs_test_path[0], func=\"get_metadata\")","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:01:10.820860Z","iopub.execute_input":"2023-12-02T10:01:10.821152Z","iopub.status.idle":"2023-12-02T10:01:10.839608Z","shell.execute_reply.started":"2023-12-02T10:01:10.821125Z","shell.execute_reply":"2023-12-02T10:01:10.838665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"De-identification methods refer to techniques or processes used to remove or anonymize personally identifiable information (PII) from data, especially in the context of sensitive information like medical records or other private datasets. The goal is to protect individuals' privacy by ensuring that the data cannot be directly linked to specific individuals.\n\nIn the context of medical imaging and DICOM (Digital Imaging and Communications in Medicine) files, de-identification methods are applied to remove or anonymize patient-specific information, such as names, dates of birth, and other identifiers, while retaining the necessary clinical information for research or analysis purposes. This is crucial for complying with privacy regulations and standards in the healthcare and research domains.\n\nThese methods help balance the need for sharing and using data for research or analysis while protecting individuals' privacy and complying with legal and ethical standards.\n\n\nEvidently, parts of this data have undergone censorship. The upcoming data analysis will specifically focus on exploring the segments that have not been subject to censorship.","metadata":{}},{"cell_type":"markdown","source":"#### 1.2.2 DICOM metadata to CSV:","metadata":{}},{"cell_type":"code","source":"df_metadata = train_labels.apply(lambda row: pd.Series({**{'id': row['id']},\n                                                        **extract_dicom_data(row['image_path'],\n                                                                             func=\"get_metadata\")}),\n                                 axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:01:10.840619Z","iopub.execute_input":"2023-12-02T10:01:10.840887Z","iopub.status.idle":"2023-12-02T10:27:44.228823Z","shell.execute_reply.started":"2023-12-02T10:01:10.840864Z","shell.execute_reply":"2023-12-02T10:27:44.227621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_metadata.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T11:18:28.991008Z","iopub.execute_input":"2023-12-02T11:18:28.991498Z","iopub.status.idle":"2023-12-02T11:18:29.017201Z","shell.execute_reply.started":"2023-12-02T11:18:28.991455Z","shell.execute_reply":"2023-12-02T11:18:29.015875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the DataFrame to a CSV file.\ndf_metadata.to_csv(os.path.join(\"train_metadata.csv\"), index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:27:44.259154Z","iopub.execute_input":"2023-12-02T10:27:44.259590Z","iopub.status.idle":"2023-12-02T10:27:44.380889Z","shell.execute_reply.started":"2023-12-02T10:27:44.259558Z","shell.execute_reply":"2023-12-02T10:27:44.379957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 1.2.3 Add DICOM shape to train CSV and store it.","metadata":{}},{"cell_type":"code","source":"# Add image shape info to train DF and save it to CSV.\n# Merge only the 'name' column from df1 and 'age' column from df2 based on the 'id' column\ntrain_boxes = pd.merge(train_boxes, df_metadata[['id', 'Rows', 'Columns']], on='id')\ntrain_boxes.rename(columns={'Rows':'rows', 'Columns': 'columns'}, inplace=True)\ntrain_boxes.to_csv(os.path.join(\"train_boxes.csv\"), index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T11:15:09.682214Z","iopub.execute_input":"2023-12-02T11:15:09.682658Z","iopub.status.idle":"2023-12-02T11:15:09.820494Z","shell.execute_reply.started":"2023-12-02T11:15:09.682625Z","shell.execute_reply":"2023-12-02T11:15:09.819284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_boxes.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T11:15:12.773988Z","iopub.execute_input":"2023-12-02T11:15:12.775138Z","iopub.status.idle":"2023-12-02T11:15:12.799724Z","shell.execute_reply.started":"2023-12-02T11:15:12.775094Z","shell.execute_reply":"2023-12-02T11:15:12.798511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Data Analysis:","metadata":{}},{"cell_type":"markdown","source":"### 2.1 Study Labels distribution:","metadata":{}},{"cell_type":"markdown","source":"Distribution of the classes: ","metadata":{}},{"cell_type":"code","source":"# Create a new DataFrame with counts aggregated by 'y_label'.\nagg_df = train_labels.groupby('y_label').size().reset_index(name='count')\n\n# Create a bar plot.\nplt.figure(figsize=(8, 4))\nax = sns.barplot(x='y_label',\n                 y='count',\n                 data=agg_df,\n                 palette='viridis')\n\n# Add annotations with counts.\nfor p in ax.patches:\n    ax.annotate(f'{p.get_height()}', (p.get_x() + p.get_width() / 2., p.get_height()),\n                ha='center', va='center', xytext=(0, 10), textcoords='offset points')\n\n# Set labels and title.\nplt.xlabel('Study Labels')\nplt.ylabel('Counts')\nplt.title('Distribution of Study Labels')\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T12:22:32.560744Z","iopub.execute_input":"2023-12-02T12:22:32.561127Z","iopub.status.idle":"2023-12-02T12:22:32.866520Z","shell.execute_reply.started":"2023-12-02T12:22:32.561098Z","shell.execute_reply":"2023-12-02T12:22:32.865332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The majority of study cases exhibit a typical pneumonia appearance, followed by a substantial number of cases identified as negative for pneumonia and those with an indeterminate presentation. Cases characterized by an atypical pneumonia appearance are observed less frequently.\nThis showcase the need to balance the dataset if it has to be used for implementing automatic classification algorithms.","metadata":{}},{"cell_type":"markdown","source":"### 2.2 Image Labels distribution","metadata":{}},{"cell_type":"markdown","source":"It is useful to see how many images with no image labels (bounding boxes) do we have for each study label. ","metadata":{}},{"cell_type":"code","source":"# Check if there is any image that has both, a bbox identified\n# with box_label == 'opacity 1', and no bbox, identified with box_label == 'none 1'.\n\nboth_labels = train_boxes.groupby('id')['box_label'].nunique()\n# Get the list of images that have both labels.\nboth_labels = both_labels[both_labels == 2].index.tolist()\n\nprint(f'There are {len(both_labels)} images with both bbox and no bbox labels.')","metadata":{"execution":{"iopub.status.busy":"2023-12-02T12:33:51.897355Z","iopub.execute_input":"2023-12-02T12:33:51.897921Z","iopub.status.idle":"2023-12-02T12:33:51.920652Z","shell.execute_reply.started":"2023-12-02T12:33:51.897871Z","shell.execute_reply":"2023-12-02T12:33:51.919384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"no_bbx = train_boxes.loc[train_boxes.box_label == 'none 1']\nprint(f'There is a total of {len(no_bbx)} images without bbox'+\n      f' for a total of {len(train_labels)} images.')","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:42:28.524246Z","iopub.execute_input":"2023-12-02T10:42:28.525587Z","iopub.status.idle":"2023-12-02T10:42:28.536642Z","shell.execute_reply.started":"2023-12-02T10:42:28.525536Z","shell.execute_reply":"2023-12-02T10:42:28.535441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_no_duplicate_ids = train_boxes.drop_duplicates(subset='id',\n                                                  keep='first')","metadata":{"execution":{"iopub.status.busy":"2023-12-02T12:39:34.436077Z","iopub.execute_input":"2023-12-02T12:39:34.436525Z","iopub.status.idle":"2023-12-02T12:39:34.446907Z","shell.execute_reply.started":"2023-12-02T12:39:34.436486Z","shell.execute_reply":"2023-12-02T12:39:34.445793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a new DataFrame with counts aggregated by 'y_label', 'box_label'.\nagg_df = (df_no_duplicate_ids.groupby(['y_label', 'box_label'])\n                    .size()\n                    .reset_index(name='count'))\n\n# Pivot the DataFrame to have 'box_label' as columns.\npivot_df = agg_df.pivot_table(index='y_label',\n                              columns='box_label',\n                              values='count',\n                              fill_value=0).reset_index()\n\n# Create a stacked bar chart.\nfig, ax = plt.subplots()\n\n# Set the positions for each bar.\npositions = range(len(pivot_df))\n\nno_bbox_bar = ax.bar(positions,\n                     pivot_df['none 1'],\n                     label='No bboxes',\n                     color='blue')\n\nbbox_bar = ax.bar(positions,\n                  pivot_df['opacity 1'],\n                  label='Has bboxes',\n                  color='orange',\n                  bottom=pivot_df['none 1'])\n\n# Add annotations with counts.\nfor bar, count in zip(no_bbox_bar, pivot_df['none 1']):\n    ax.text(bar.get_x() + bar.get_width() / 2, bar.get_height(),\n            str(count),\n            ha='center',\n            va='bottom',\n            color=(0, 0, 1),\n            fontsize=12)\n\nfor bar, count in zip(bbox_bar, pivot_df['opacity 1']):\n    ax.text(bar.get_x() + bar.get_width() / 2, bar.get_height() + 100,\n            str(count),\n            ha='center',\n            va='bottom',\n            color=(1, 0.647, 0),\n            fontsize=12)\n\n# Set labels and title.\nax.set_xticks(positions)\nax.set_xticklabels(pivot_df['y_label'])\nax.set_ylabel('Counts')\nax.set_xlabel('Study Labels')\nax.set_title('Stacked Bar Chart of Counts by study labels and image bbox Labels')\nax.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T12:39:44.214692Z","iopub.execute_input":"2023-12-02T12:39:44.215121Z","iopub.status.idle":"2023-12-02T12:39:44.570625Z","shell.execute_reply.started":"2023-12-02T12:39:44.215086Z","shell.execute_reply":"2023-12-02T12:39:44.569488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As it can be seen, the category 'Negative for Pneumonia' has no bounding boxes on image labels. If the goal is to detect bounding boxes for the rest of categories, cleaning up those images that have no bbox is suggested.","metadata":{}},{"cell_type":"markdown","source":"### 2.3 Images' metadata distributions\":","metadata":{}},{"cell_type":"markdown","source":"DICOM files contain metadata that is worth exploring.","metadata":{}},{"cell_type":"code","source":"def label_sizes(col):\n    \"\"\"\n    Function to extract labels, sizes and unique values for a specific column.\n    \"\"\"\n    labels = df_metadata[col].value_counts().index\n    sizes = df_metadata[col].value_counts()\n    uc = df_metadata[col].nunique() #To be used in explode param.\n\n    return labels, sizes, uc\n\n\ncolumns = [\"De-identification Method\", \"Patient's Sex\",\n           \"Modality\",\"Photometric Interpretation\"]\n\ndata_list = [label_sizes(col) for col in columns]\n\n# Create pie charts\ncolors = ['#FF9999', '#FFD700', '#FF6347', '#FFB6C1', '#E9967A', '#F08080']\n\nfig, axes = plt.subplots(2, 2, figsize=(10, 8))\nfig.subplots_adjust(wspace=0.5, hspace=0.5)\n\naxes = axes.flatten()\n\nfor i, ax in enumerate(axes):\n    labels, sizes, uc = data_list[i]\n    explode = (0.05,) * uc\n    colors = colors[:len(labels)]\n\n    ax.pie(sizes,\n           labels=labels,\n           autopct='%1.1f%%',\n           colors=colors,\n           explode=explode,\n           startangle=90,\n           textprops={'fontsize': 7})\n    # Drawing a circle at the center to make it look like a donut chart.\n    ax.add_artist(plt.Circle((0, 0), 0.30, fc='white'))\n    ax.set_title(columns[i], fontsize=12)\n    ax.axis('equal')  # Equal aspect ratio ensures that pie is drawn as a circle.\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:27:45.486292Z","iopub.execute_input":"2023-12-02T10:27:45.486709Z","iopub.status.idle":"2023-12-02T10:27:45.980525Z","shell.execute_reply.started":"2023-12-02T10:27:45.486677Z","shell.execute_reply":"2023-12-02T10:27:45.979218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def horizontal_plot(df, ylabel, title):\n\n    colors = ['#66b3ff', '#99ff99', '#ffcc99', '#c2c2f0', '#ffb3e6', '#ff6666']\n\n    plt.figure(figsize=(8, 6))\n    sns.countplot(y=ylabel,\n                  data=df,\n                  palette=colors)\n\n    plt.xlabel('count', fontsize=12, color='gray')\n    plt.ylabel(ylabel, fontsize=12, color='gray')\n    plt.title(title, fontsize=16, color='#333333', weight='bold')\n\n    # Add grid for better readability\n    plt.grid(axis='x', linestyle='--', alpha=0.6)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:27:45.982206Z","iopub.execute_input":"2023-12-02T10:27:45.982675Z","iopub.status.idle":"2023-12-02T10:27:45.991115Z","shell.execute_reply.started":"2023-12-02T10:27:45.982634Z","shell.execute_reply":"2023-12-02T10:27:45.990022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"horizontal_plot(df_metadata,\n                \"Body Part Examined\",\n                'Body Part Examined Distribution')","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:27:45.992625Z","iopub.execute_input":"2023-12-02T10:27:45.992952Z","iopub.status.idle":"2023-12-02T10:27:46.395409Z","shell.execute_reply.started":"2023-12-02T10:27:45.992924Z","shell.execute_reply":"2023-12-02T10:27:46.394257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"horizontal_plot(df_metadata,\n                \"Private Creator\",\n                'Private Creator Distribution')","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:27:46.397061Z","iopub.execute_input":"2023-12-02T10:27:46.397554Z","iopub.status.idle":"2023-12-02T10:27:46.745326Z","shell.execute_reply.started":"2023-12-02T10:27:46.397508Z","shell.execute_reply":"2023-12-02T10:27:46.744505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3. Image examples for each case:","metadata":{}},{"cell_type":"markdown","source":"###  3.1 Images for each study case:","metadata":{}},{"cell_type":"code","source":"def format_boxes(image_list):\n    \"\"\"\n    Function recieves a list strings, that consist on the image 'id's for different images,\n    and reformat them in the following way:\n        - If box_label for the image starts with None, leave box empty (empty list).\n        - Whenever it encounters 'opacity' in box_label, append the box info to create one list of \n          boxes per each image. \n    Example : If we have the following boxes per image ['opacity 1 72 58 18 24 opacity 1 2 59 33 23',\n                               'none 1 0 0 1 1']\n              OUT result = [[72, 58, 18, 24, 2, 59, 33, 23], []]\n    Args: \n        image_list: list of image 'id's to extract boxes from. (list of strings)\n    Outputs:\n        result: List of Lists of reformated boxes. (list of lists)\n    \"\"\"\n    result = []\n    for i, im_id in enumerate(image_list):  # find id in train_boxes\n        boxes_info = train_boxes.loc[train_boxes.id == im_id]\n        boxes = []\n        for _, box in boxes_info.iterrows():\n            if box['box_label'].startswith('opacity'):\n                boxes.extend([box['xmin'], box['ymin'], box['xmax'], box['ymax']])\n            else: \n                boxes.extend([])\n        result.append(boxes)\n    return result","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:27:46.746478Z","iopub.execute_input":"2023-12-02T10:27:46.747388Z","iopub.status.idle":"2023-12-02T10:27:46.754910Z","shell.execute_reply.started":"2023-12-02T10:27:46.747352Z","shell.execute_reply":"2023-12-02T10:27:46.753780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 3.1.1 Negative for Pneumonia","metadata":{}},{"cell_type":"code","source":"trues = train_labels.loc[train_labels[\"Negative for Pneumonia\"] == 1][:4]\nimgs_test_path = trues.image_path.tolist()\nimg_test_boxes = format_boxes(trues.id.tolist())\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"Negative for Pneumonia Examples\")","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:27:46.756093Z","iopub.execute_input":"2023-12-02T10:27:46.756478Z","iopub.status.idle":"2023-12-02T10:27:50.933020Z","shell.execute_reply.started":"2023-12-02T10:27:46.756412Z","shell.execute_reply":"2023-12-02T10:27:50.932139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 3.1.2 Typical Appearance","metadata":{}},{"cell_type":"code","source":"trues = train_labels.loc[train_labels[\"Typical Appearance\"] == 1][:4]\nimgs_test_path = trues.image_path.tolist()\nimg_test_boxes = format_boxes(trues.id.tolist())\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"Typical Appearance Examples\")","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:27:50.934403Z","iopub.execute_input":"2023-12-02T10:27:50.934739Z","iopub.status.idle":"2023-12-02T10:27:55.952449Z","shell.execute_reply.started":"2023-12-02T10:27:50.934705Z","shell.execute_reply":"2023-12-02T10:27:55.951260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 3.1.3 Atypical Appearance","metadata":{}},{"cell_type":"code","source":"trues = train_labels.loc[train_labels[\"Atypical Appearance\"] == 1][:4]\nimgs_test_path = trues.image_path.tolist()\nimg_test_boxes = format_boxes(trues.id.tolist())\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"Atypical Appearance Examples\")","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:27:55.953998Z","iopub.execute_input":"2023-12-02T10:27:55.954302Z","iopub.status.idle":"2023-12-02T10:28:01.272498Z","shell.execute_reply.started":"2023-12-02T10:27:55.954275Z","shell.execute_reply":"2023-12-02T10:28:01.271180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 3.1.4 Indeterminate Appearance","metadata":{}},{"cell_type":"code","source":"trues = train_labels.loc[train_labels[\"Indeterminate Appearance\"] == 1][:4]\nimgs_test_path = trues.image_path.tolist()\nimg_test_boxes = format_boxes(trues.id.tolist())\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"Indeterminate Appearance Examples\")","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:28:01.274239Z","iopub.execute_input":"2023-12-02T10:28:01.275047Z","iopub.status.idle":"2023-12-02T10:28:05.946604Z","shell.execute_reply.started":"2023-12-02T10:28:01.275003Z","shell.execute_reply":"2023-12-02T10:28:05.945339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 3.1.5 Comparison between classes:","metadata":{}},{"cell_type":"code","source":"indeterminate = train_labels.loc[train_labels[\"Indeterminate Appearance\"] == 1][:1]\natypical = train_labels.loc[train_labels[\"Atypical Appearance\"] == 1][:1]\ntypical = train_labels.loc[train_labels[\"Typical Appearance\"] == 1][:1]\nnegative = train_labels.loc[train_labels[\"Negative for Pneumonia\"] == 1][:1]\n\nimgs_test_path = indeterminate.image_path.tolist() + \\\n                 atypical.image_path.tolist() + \\\n                 typical.image_path.tolist() + \\\n                 negative.image_path.tolist()\nimg_test_boxes = indeterminate.id.tolist() + \\\n                 atypical.id.tolist() + \\\n                 typical.id.tolist() + \\\n                 negative.id.tolist()\nimg_test_boxes = format_boxes(img_test_boxes)\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"Class comparison Examples\", [\"Indeterminate Appearance\", \"Atypical Appearance\",\n                                         \"Typical Appearance\", \"Negative for Pneumonia\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:28:05.948089Z","iopub.execute_input":"2023-12-02T10:28:05.948511Z","iopub.status.idle":"2023-12-02T10:28:09.848730Z","shell.execute_reply.started":"2023-12-02T10:28:05.948473Z","shell.execute_reply":"2023-12-02T10:28:09.847670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"indeterminate = train_labels.loc[train_labels[\"Indeterminate Appearance\"] == 1][4:5]\natypical = train_labels.loc[train_labels[\"Atypical Appearance\"] == 1][4:5]\ntypical = train_labels.loc[train_labels[\"Typical Appearance\"] == 1][4:5]\nnegative = train_labels.loc[train_labels[\"Negative for Pneumonia\"] == 1][4:5]\n\nimgs_test_path = indeterminate.image_path.tolist() + \\\n                 atypical.image_path.tolist() + \\\n                 typical.image_path.tolist() + \\\n                 negative.image_path.tolist()\nimg_test_boxes = indeterminate.id.tolist() + \\\n                 atypical.id.tolist() + \\\n                 typical.id.tolist() + \\\n                 negative.id.tolist()\nimg_test_boxes = format_boxes(img_test_boxes)\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"Class comparison Examples\", [\"Indeterminate Appearance\", \"Atypical Appearance\",\n                                         \"Typical Appearance\", \"Negative for Pneumonia\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:28:09.850307Z","iopub.execute_input":"2023-12-02T10:28:09.851393Z","iopub.status.idle":"2023-12-02T10:28:14.711923Z","shell.execute_reply.started":"2023-12-02T10:28:09.851358Z","shell.execute_reply":"2023-12-02T10:28:14.710973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"indeterminate = train_labels.loc[train_labels[\"Indeterminate Appearance\"] == 1][8:9]\natypical = train_labels.loc[train_labels[\"Atypical Appearance\"] == 1][8:9]\ntypical = train_labels.loc[train_labels[\"Typical Appearance\"] == 1][8:9]\nnegative = train_labels.loc[train_labels[\"Negative for Pneumonia\"] == 1][8:9]\n\nimgs_test_path = indeterminate.image_path.tolist() + \\\n                 atypical.image_path.tolist() + \\\n                 typical.image_path.tolist() + \\\n                 negative.image_path.tolist()\nimg_test_boxes = indeterminate.id.tolist() + \\\n                 atypical.id.tolist() + \\\n                 typical.id.tolist() + \\\n                 negative.id.tolist()\nimg_test_boxes = format_boxes(img_test_boxes)\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"Class comparison Examples\", [\"Indeterminate Appearance\", \"Atypical Appearance\",\n                                         \"Typical Appearance\", \"Negative for Pneumonia\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:28:14.713115Z","iopub.execute_input":"2023-12-02T10:28:14.714012Z","iopub.status.idle":"2023-12-02T10:28:19.354704Z","shell.execute_reply.started":"2023-12-02T10:28:14.713976Z","shell.execute_reply":"2023-12-02T10:28:19.353522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###  3.2 Images for different metadata attributes:","metadata":{}},{"cell_type":"code","source":"def extract_image_path_and_boxes(id_list):\n    \"\"\"\n    Function obtains a list of image paths and boxes (without format) from given image id list.\n    \"\"\"\n    imgs_test_path = []\n    img_test_boxes = []\n    for imgid in id_list:\n        loc = train_labels.loc[train_labels[\"id\"] == imgid]\n        imgs_test_path.append(loc.image_path.tolist()[0])\n        img_test_boxes.append(loc.id.tolist()[0])\n\n    return imgs_test_path, img_test_boxes\n    ","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:28:19.356201Z","iopub.execute_input":"2023-12-02T10:28:19.356552Z","iopub.status.idle":"2023-12-02T10:28:19.363502Z","shell.execute_reply.started":"2023-12-02T10:28:19.356522Z","shell.execute_reply":"2023-12-02T10:28:19.362392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 3.2.1 Modality Distribution","metadata":{}},{"cell_type":"code","source":"cr = df_metadata.loc[df_metadata[\"Modality\"] == \"CR\"][\"id\"][7:9].tolist()\ndx = df_metadata.loc[df_metadata[\"Modality\"] == \"DX\"][\"id\"][7:9].tolist()\n\ncr_path, cr_boxes = extract_image_path_and_boxes(cr)\ndx_path, dx_boxes = extract_image_path_and_boxes(dx)\n\nimgs_test_path = cr_path + dx_path     \nimg_test_boxes = cr_boxes + dx_boxes\nimg_test_boxes = format_boxes(img_test_boxes)\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"Modality\", [\"CR\", \"CR\", \"DX\", \"DX\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:28:19.365097Z","iopub.execute_input":"2023-12-02T10:28:19.366030Z","iopub.status.idle":"2023-12-02T10:28:23.448669Z","shell.execute_reply.started":"2023-12-02T10:28:19.366001Z","shell.execute_reply":"2023-12-02T10:28:23.447578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cr = df_metadata.loc[df_metadata[\"Modality\"] == \"CR\"][\"id\"][2:4].tolist()\ndx = df_metadata.loc[df_metadata[\"Modality\"] == \"DX\"][\"id\"][2:4].tolist()\n\ncr_path, cr_boxes = extract_image_path_and_boxes(cr)\ndx_path, dx_boxes = extract_image_path_and_boxes(dx)\n\nimgs_test_path = cr_path + dx_path     \nimg_test_boxes = cr_boxes + dx_boxes\nimg_test_boxes = format_boxes(img_test_boxes)\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"Modality\", [\"CR\", \"CR\", \"DX\", \"DX\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:28:23.450275Z","iopub.execute_input":"2023-12-02T10:28:23.451599Z","iopub.status.idle":"2023-12-02T10:28:27.532472Z","shell.execute_reply.started":"2023-12-02T10:28:23.451550Z","shell.execute_reply":"2023-12-02T10:28:27.531618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 3.2.2 De-identification Method","metadata":{}},{"cell_type":"code","source":"default = df_metadata.loc[df_metadata[\"De-identification Method\"] == \"CTP Default\"][\"id\"][1:3].tolist()\nrsna = df_metadata.loc[df_metadata[\"De-identification Method\"] == \"RSNA Covid-19 Dataset Default\"][\"id\"][1:3].tolist()\nmask = df_metadata[\"De-identification Method\"].str.startswith(\"CTP Default:\")\ndicom = df_metadata[mask][\"id\"][1:3].tolist()\n\ndefault_path, default_boxes = extract_image_path_and_boxes(default)\nrsna_path, rsna_boxes = extract_image_path_and_boxes(rsna)\ndicom_path, dicom_boxes = extract_image_path_and_boxes(dicom)\n\nimgs_test_path = default_path + rsna_path + dicom_path\nimg_test_boxes = default_boxes + rsna_boxes + dicom_boxes\nimg_test_boxes = format_boxes(img_test_boxes)\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"De-identification Method\", [\"CTP Default\", \"CTP Default\",\n                                        \"RSNA Covid-19 Dataset Default\", \"RSNA Covid-19 Dataset Default\",\n                                        \"CTP Default: based on DICOM PS3.15\", \"CTP Default: based on DICOM PS3.15\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T12:40:26.301752Z","iopub.execute_input":"2023-12-02T12:40:26.302212Z","iopub.status.idle":"2023-12-02T12:40:36.497227Z","shell.execute_reply.started":"2023-12-02T12:40:26.302166Z","shell.execute_reply":"2023-12-02T12:40:36.496030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"default = df_metadata.loc[df_metadata[\"De-identification Method\"] == \"CTP Default\"][\"id\"][4:6].tolist()\nrsna = df_metadata.loc[df_metadata[\"De-identification Method\"] == \"RSNA Covid-19 Dataset Default\"][\"id\"][4:6].tolist()\nmask = df_metadata[\"De-identification Method\"].str.startswith(\"CTP Default:\")\ndicom = df_metadata[mask][\"id\"][4:6].tolist()\n\ndefault_path, default_boxes = extract_image_path_and_boxes(default)\nrsna_path, rsna_boxes = extract_image_path_and_boxes(rsna)\ndicom_path, dicom_boxes = extract_image_path_and_boxes(dicom)\n\nimgs_test_path = default_path + rsna_path + dicom_path\nimg_test_boxes = default_boxes + rsna_boxes + dicom_boxes\nimg_test_boxes = format_boxes(img_test_boxes)\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"De-identification Method\", [\"CTP Default\", \"CTP Default\",\n                                        \"RSNA Covid-19 Dataset Default\", \"RSNA Covid-19 Dataset Default\",\n                                        \"CTP Default: based on DICOM PS3.15\", \"CTP Default: based on DICOM PS3.15\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T12:41:35.358118Z","iopub.execute_input":"2023-12-02T12:41:35.358559Z","iopub.status.idle":"2023-12-02T12:41:44.891462Z","shell.execute_reply.started":"2023-12-02T12:41:35.358519Z","shell.execute_reply":"2023-12-02T12:41:44.890153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 3.2.3 Photometric Interpretation","metadata":{}},{"cell_type":"code","source":"m1 = df_metadata.loc[df_metadata[\"Photometric Interpretation\"] == \"MONOCHROME1\"][\"id\"][7:9].tolist()\nm2 = df_metadata.loc[df_metadata[\"Photometric Interpretation\"] == \"MONOCHROME2\"][\"id\"][7:9].tolist()\n\nm1_path, m1_boxes = extract_image_path_and_boxes(m1)\nm2_path, m2_boxes = extract_image_path_and_boxes(m2)\n\nimgs_test_path = m1_path + m2_path     \nimg_test_boxes = m1_boxes + m2_boxes\nimg_test_boxes = format_boxes(img_test_boxes)\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"Photometric Interpretation\", [\"MONOCHROME1\", \"MONOCHROME1\", \"MONOCHROME2\", \"MONOCHROME2\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:28:27.543530Z","iopub.execute_input":"2023-12-02T10:28:27.543891Z","iopub.status.idle":"2023-12-02T10:28:31.968586Z","shell.execute_reply.started":"2023-12-02T10:28:27.543861Z","shell.execute_reply":"2023-12-02T10:28:31.967717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m1 = df_metadata.loc[df_metadata[\"Photometric Interpretation\"] == \"MONOCHROME1\"][\"id\"][2:4].tolist()\nm2 = df_metadata.loc[df_metadata[\"Photometric Interpretation\"] == \"MONOCHROME2\"][\"id\"][2:4].tolist()\n\nm1_path, m1_boxes = extract_image_path_and_boxes(m1)\nm2_path, m2_boxes = extract_image_path_and_boxes(m2)\n\nimgs_test_path = m1_path + m2_path     \nimg_test_boxes = m1_boxes + m2_boxes\nimg_test_boxes = format_boxes(img_test_boxes)\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"Photometric Interpretation\", [\"MONOCHROME1\", \"MONOCHROME1\", \"MONOCHROME2\", \"MONOCHROME2\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:28:31.970079Z","iopub.execute_input":"2023-12-02T10:28:31.970681Z","iopub.status.idle":"2023-12-02T10:28:35.826702Z","shell.execute_reply.started":"2023-12-02T10:28:31.970647Z","shell.execute_reply":"2023-12-02T10:28:35.825854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 3.2.4 Patient's Sex","metadata":{}},{"cell_type":"code","source":"m = df_metadata.loc[df_metadata[\"Patient's Sex\"] == \"M\"][\"id\"][7:9].tolist()\nf = df_metadata.loc[df_metadata[\"Patient's Sex\"] == \"F\"][\"id\"][7:9].tolist()\n\nm_path, m_boxes = extract_image_path_and_boxes(m)\nf_path, f_boxes = extract_image_path_and_boxes(f)\n\nimgs_test_path = m_path + f_path     \nimg_test_boxes = m_boxes + f_boxes\nimg_test_boxes = format_boxes(img_test_boxes)\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"Patient's Sex\", [\"M\", \"M\", \"F\", \"F\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:28:35.828249Z","iopub.execute_input":"2023-12-02T10:28:35.828608Z","iopub.status.idle":"2023-12-02T10:28:40.272378Z","shell.execute_reply.started":"2023-12-02T10:28:35.828577Z","shell.execute_reply":"2023-12-02T10:28:40.271081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df_metadata.loc[df_metadata[\"Patient's Sex\"] == \"M\"][\"id\"][30:32].tolist()\nf = df_metadata.loc[df_metadata[\"Patient's Sex\"] == \"F\"][\"id\"][30:32].tolist()\n\nm_path, m_boxes = extract_image_path_and_boxes(m)\nf_path, f_boxes = extract_image_path_and_boxes(f)\n\nimgs_test_path = m_path + f_path     \nimg_test_boxes = m_boxes + f_boxes\nimg_test_boxes = format_boxes(img_test_boxes)\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"Patient's Sex\", [\"M\", \"M\", \"F\", \"F\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:28:40.273810Z","iopub.execute_input":"2023-12-02T10:28:40.274139Z","iopub.status.idle":"2023-12-02T10:28:45.258892Z","shell.execute_reply.started":"2023-12-02T10:28:40.274110Z","shell.execute_reply":"2023-12-02T10:28:45.257638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 3.2.5 Body Part Examined","metadata":{}},{"cell_type":"code","source":"chest = df_metadata.loc[df_metadata[\"Body Part Examined\"] == \"CHEST\"][\"id\"][30:32].tolist()\nport = df_metadata.loc[df_metadata[\"Body Part Examined\"] == \"PORT CHEST\"][\"id\"][30:32].tolist()\ntorax = df_metadata.loc[df_metadata[\"Body Part Examined\"] == \"TORAX\"][\"id\"][0:2].tolist()\nskull = df_metadata.loc[df_metadata[\"Body Part Examined\"] == \"SKULL\"][\"id\"][0:2].tolist()\n\nchest_path, chest_boxes = extract_image_path_and_boxes(chest)\nport_path, port_boxes = extract_image_path_and_boxes(port)\ntorax_path, torax_boxes = extract_image_path_and_boxes(torax)\nskull_path, skull_boxes = extract_image_path_and_boxes(skull)\n\nimgs_test_path = chest_path + port_path + torax_path + skull_path   \nimg_test_boxes = chest_boxes + port_boxes + torax_boxes + skull_boxes\nimg_test_boxes = format_boxes(img_test_boxes)\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"Body Part Examined\", [\"CHEST\", \"CHEST\",\n                               \"PORT CHEST\", \"PORT CHEST\",\n                               \"TORAX\", \"TORAX\",\n                               \"SKULL\", \"SKULL\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:28:45.260515Z","iopub.execute_input":"2023-12-02T10:28:45.260956Z","iopub.status.idle":"2023-12-02T10:28:53.121664Z","shell.execute_reply.started":"2023-12-02T10:28:45.260914Z","shell.execute_reply":"2023-12-02T10:28:53.120657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chest = df_metadata.loc[df_metadata[\"Body Part Examined\"] == \"CHEST\"][\"id\"][0:2].tolist()\nport = df_metadata.loc[df_metadata[\"Body Part Examined\"] == \"PORT CHEST\"][\"id\"][0:2].tolist()\ntorax = df_metadata.loc[df_metadata[\"Body Part Examined\"] == \"TORAX\"][\"id\"][2:4].tolist()\nskull = df_metadata.loc[df_metadata[\"Body Part Examined\"] == \"SKULL\"][\"id\"][2:4].tolist()\n\nchest_path, chest_boxes = extract_image_path_and_boxes(chest)\nport_path, port_boxes = extract_image_path_and_boxes(port)\ntorax_path, torax_boxes = extract_image_path_and_boxes(torax)\nskull_path, skull_boxes = extract_image_path_and_boxes(skull)\n\nimgs_test_path = chest_path + port_path + torax_path + skull_path   \nimg_test_boxes = chest_boxes + port_boxes + torax_boxes + skull_boxes\nimg_test_boxes = format_boxes(img_test_boxes)\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"Body Part Examined\", [\"CHEST\", \"CHEST\",\n                               \"PORT CHEST\", \"PORT CHEST\",\n                               \"TORAX\", \"TORAX\",\n                               \"SKULL\", \"SKULL\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:28:53.122995Z","iopub.execute_input":"2023-12-02T10:28:53.124025Z","iopub.status.idle":"2023-12-02T10:29:00.820505Z","shell.execute_reply.started":"2023-12-02T10:28:53.123988Z","shell.execute_reply":"2023-12-02T10:29:00.819494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 3.2.6 Private Creator","metadata":{}},{"cell_type":"code","source":"gehiis = df_metadata.loc[df_metadata[\"Private Creator\"] == \"GEIIS\"][\"id\"][30:32].tolist()\nphilips = df_metadata.loc[df_metadata[\"Private Creator\"] == \"Philips RAD Imaging DD 097\"][\"id\"][30:32].tolist()\ngems = df_metadata.loc[df_metadata[\"Private Creator\"] == \"GEMS_GDXE_ATHENAV2_INTERNAL_USE\"][\"id\"][0:2].tolist()\n\ngehiis_path, gehiis_boxes = extract_image_path_and_boxes(gehiis)\nphilips_path, philips_boxes = extract_image_path_and_boxes(philips)\ngems_path, gems_boxes = extract_image_path_and_boxes(gems)\n\nimgs_test_path = gehiis_path + philips_path + gems_path    \nimg_test_boxes = gehiis_boxes + philips_boxes + gems_boxes\nimg_test_boxes = format_boxes(img_test_boxes)\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"Private Creator\", [\"GEIIS\", \"GEIIS\",\n                               \"Philips RAD Imaging DD 097\", \"Philips RAD Imaging DD 097\",\n                               \"GEMS_GDXE_ATHENAV2\", \"GEMS_GDXE_ATHENAV2\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:29:00.822347Z","iopub.execute_input":"2023-12-02T10:29:00.822783Z","iopub.status.idle":"2023-12-02T10:29:06.932338Z","shell.execute_reply.started":"2023-12-02T10:29:00.822743Z","shell.execute_reply":"2023-12-02T10:29:06.930832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gehiis = df_metadata.loc[df_metadata[\"Private Creator\"] == \"GEIIS\"][\"id\"][32:34].tolist()\nphilips = df_metadata.loc[df_metadata[\"Private Creator\"] == \"Philips RAD Imaging DD 097\"][\"id\"][32:34].tolist()\ngems = df_metadata.loc[df_metadata[\"Private Creator\"] == \"GEMS_GDXE_ATHENAV2_INTERNAL_USE\"][\"id\"][2:4].tolist()\n\ngehiis_path, gehiis_boxes = extract_image_path_and_boxes(gehiis)\nphilips_path, philips_boxes = extract_image_path_and_boxes(philips)\ngems_path, gems_boxes = extract_image_path_and_boxes(gems)\n\nimgs_test_path = gehiis_path + philips_path + gems_path    \nimg_test_boxes = gehiis_boxes + philips_boxes + gems_boxes\nimg_test_boxes = format_boxes(img_test_boxes)\n\nplot_DICOM(imgs_test_path, img_test_boxes,\n           \"Private Creator\", [\"GEIIS\", \"GEIIS\",\n                               \"Philips RAD Imaging DD 097\", \"Philips RAD Imaging DD 097\",\n                               \"GEMS_GDXE_ATHENAV2\", \"GEMS_GDXE_ATHENAV2\"])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T10:29:06.934143Z","iopub.execute_input":"2023-12-02T10:29:06.934521Z","iopub.status.idle":"2023-12-02T10:29:13.032049Z","shell.execute_reply.started":"2023-12-02T10:29:06.934489Z","shell.execute_reply":"2023-12-02T10:29:13.031003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Disclaimers:\n\nThe code and procedures are original unless otherwise specified. That being said, the work is inspired by the EDA : https://www.kaggle.com/code/ruchi798/siim-covid-19-detection-eda-data-augmentation \n","metadata":{}}]}