{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# %% [code] {\"_kg_hide-input\":true,\"_kg_hide-output\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:09:01.14924Z\",\"iopub.execute_input\":\"2021-06-15T11:09:01.149645Z\",\"iopub.status.idle\":\"2021-06-15T11:10:36.922263Z\",\"shell.execute_reply.started\":\"2021-06-15T11:09:01.149563Z\",\"shell.execute_reply\":\"2021-06-15T11:10:36.921151Z\"}}\n\n\nimport numpy as np \nimport pandas as pd \nimport os\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\n%matplotlib inline\nimport glob\nfrom pydicom import dcmread\nfrom pydicom.data import get_testdata_file\n\nfrom tqdm import tqdm\n\nimport ast\n\n!pip install hvplot\nimport hvplot.pandas \n\n!pip install pylibjpeg pylibjpeg-libjpeg pydicom python-gdcm\nimport gdcm\nimport pylibjpeg\n\n# %% [code]\n\n\n\n\n\n# %% [code] {\"_kg_hide-input\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:10:36.925533Z\",\"iopub.execute_input\":\"2021-06-15T11:10:36.92585Z\",\"iopub.status.idle\":\"2021-06-15T11:10:36.985107Z\",\"shell.execute_reply.started\":\"2021-06-15T11:10:36.925819Z\",\"shell.execute_reply\":\"2021-06-15T11:10:36.984315Z\"}}\n# Read the data\ndf_train_images = pd.read_csv('../input/siim-covid19-detection/train_image_level.csv')\ndf_train_study = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv')\n\n\n\n# %% [code] {\"_kg_hide-input\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:10:36.987558Z\",\"iopub.execute_input\":\"2021-06-15T11:10:36.988334Z\",\"iopub.status.idle\":\"2021-06-15T11:10:37.020478Z\",\"shell.execute_reply.started\":\"2021-06-15T11:10:36.988285Z\",\"shell.execute_reply\":\"2021-06-15T11:10:37.019459Z\"}}\nlabels = ['Negative for Pneumonia', 'Typical Appearance', 'Indeterminate Appearance', 'Atypical Appearance']\n\nprint(\"Number of study : \", len(df_train_study))\n\ndf_train_study.head()\n\n# ## See some sample images from the different categories\n\n\n# {\"_kg_hide-input\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:10:37.022516Z\",\"iopub.execute_input\":\"2021-06-15T11:10:37.02328Z\",\"iopub.status.idle\":\"2021-06-15T11:10:37.03261Z\",\"shell.execute_reply.started\":\"2021-06-15T11:10:37.023229Z\",\"shell.execute_reply\":\"2021-06-15T11:10:37.031421Z\"}}\nNUMBER_OF_SAMPLE = 5\n\ndef read_image_from_study(study_id):\n    study_name = study_id.split('_')[0]\n    file = glob.glob(\"../input/siim-covid19-detection/train/\" + study_name + \"/*/*.dcm\")\n    ds = dcmread(file[0])\n    return ds.pixel_array\n\ndef show_sample_data_from_study(sample_images, NB_SAMPLE = 5):\n    fig, axes = plt.subplots(nrows=1, ncols=NB_SAMPLE, figsize=(NB_SAMPLE * 4, 4))\n    i = 0\n    for index, row in sample_images.iterrows():\n        img = read_image_from_study(row['id'])\n        axes[i].imshow(img, cmap=plt.cm.gray, aspect='auto')\n        axes[i].axis('off')\n        i += 1\n    fig.show()\n\n# ### Negative for pneumonia\n\n#  {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:10:37.034277Z\",\"iopub.execute_input\":\"2021-06-15T11:10:37.034733Z\",\"iopub.status.idle\":\"2021-06-15T11:10:43.058371Z\",\"shell.execute_reply.started\":\"2021-06-15T11:10:37.034686Z\",\"shell.execute_reply\":\"2021-06-15T11:10:43.057315Z\"}}\nsample_negative_pneumonia = df_train_study[df_train_study['Negative for Pneumonia'] == 1].sample(n=NUMBER_OF_SAMPLE, random_state=42)\nshow_sample_data_from_study(sample_negative_pneumonia, NUMBER_OF_SAMPLE)\n\n# ### Typical appearance\n\n{\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:10:43.059631Z\",\"iopub.execute_input\":\"2021-06-15T11:10:43.059932Z\",\"iopub.status.idle\":\"2021-06-15T11:10:48.130043Z\",\"shell.execute_reply.started\":\"2021-06-15T11:10:43.059902Z\",\"shell.execute_reply\":\"2021-06-15T11:10:48.128941Z\"}}\nsample_negative_pneumonia = df_train_study[df_train_study['Typical Appearance'] == 1].sample(n=NUMBER_OF_SAMPLE, random_state=42)\nshow_sample_data_from_study(sample_negative_pneumonia, NUMBER_OF_SAMPLE)\n\n\n# ### Indeterminate appearance\n\n#{\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:10:48.13148Z\",\"iopub.execute_input\":\"2021-06-15T11:10:48.131778Z\",\"iopub.status.idle\":\"2021-06-15T11:10:53.633478Z\",\"shell.execute_reply.started\":\"2021-06-15T11:10:48.13175Z\",\"shell.execute_reply\":\"2021-06-15T11:10:53.632781Z\"}}\nsample_negative_pneumonia = df_train_study[df_train_study['Indeterminate Appearance'] == 1].sample(n=NUMBER_OF_SAMPLE, random_state=42)\nshow_sample_data_from_study(sample_negative_pneumonia, NUMBER_OF_SAMPLE)\n\n# ### Atypical appearance\n\n#{\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:10:53.635519Z\",\"iopub.execute_input\":\"2021-06-15T11:10:53.635917Z\",\"iopub.status.idle\":\"2021-06-15T11:10:58.582063Z\",\"shell.execute_reply.started\":\"2021-06-15T11:10:53.635887Z\",\"shell.execute_reply\":\"2021-06-15T11:10:58.581036Z\"}}\nsample_negative_pneumonia = df_train_study[df_train_study['Atypical Appearance'] == 1].sample(n=NUMBER_OF_SAMPLE, random_state=42)\nshow_sample_data_from_study(sample_negative_pneumonia, NUMBER_OF_SAMPLE)\n \n# Note : I think that could be interesting to use images enhancement in order to help the visualisation of x-rays images. [I found on github a library called *X-Ray Images Enhancement* that could be interesting](https://github.com/asalmada/x-ray-images-enhancement). I will try it in another kernel. If someone already applies this kind of techniques or used that library, feel free to share your experience with us :)\n\n# ## Distribution of the different categories\n\n {\"_kg_hide-input\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:10:58.583715Z\",\"iopub.execute_input\":\"2021-06-15T11:10:58.583992Z\",\"iopub.status.idle\":\"2021-06-15T11:10:58.890808Z\",\"shell.execute_reply.started\":\"2021-06-15T11:10:58.583964Z\",\"shell.execute_reply\":\"2021-06-15T11:10:58.890079Z\"}}\n# Count for each labels the number of occurence\nstudy_case = [df_train_study[label].value_counts()[1] for label in labels]\n\nplt.figure(figsize=(15, 6))\nplt.bar(labels, study_case)\nplt.title('Distribution of the different categories')\nplt.show()\n\nplt.figure(figsize=(8, 8))\nplt.pie(study_case, labels=labels, autopct='%1.1f%%')\nplt.title('Proportion of the different categories')\nplt.show()\n\n{\"_kg_hide-input\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:10:58.892054Z\",\"iopub.execute_input\":\"2021-06-15T11:10:58.892656Z\",\"iopub.status.idle\":\"2021-06-15T11:10:59.232603Z\",\"shell.execute_reply.started\":\"2021-06-15T11:10:58.892611Z\",\"shell.execute_reply\":\"2021-06-15T11:10:59.231493Z\"}}\ndef count_column(x):\n    return x.sum()\n    \ndf_train_count = df_train_study[labels].apply(count_column, axis=1)\nprint(\"Number of multiple categories ?\", df_train_count[df_train_count != 1].sum())\n\n# # Analysis of the images\n\n#{\"_kg_hide-input\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:10:59.2339Z\",\"iopub.execute_input\":\"2021-06-15T11:10:59.234217Z\",\"iopub.status.idle\":\"2021-06-15T11:10:59.247854Z\",\"shell.execute_reply.started\":\"2021-06-15T11:10:59.234174Z\",\"shell.execute_reply\":\"2021-06-15T11:10:59.246752Z\"}}\nprint(\"Number of images : \", len(df_train_images))\ndf_train_images.head()\n\n\n# ## Image analysis \n\n# To recall, we had seen in the previous part that we have multiple images for a given study. It could be interesting to visualize those data in order to understand why we have multiple images.\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:10:59.249415Z\",\"iopub.execute_input\":\"2021-06-15T11:10:59.249832Z\",\"iopub.status.idle\":\"2021-06-15T11:10:59.265054Z\",\"shell.execute_reply.started\":\"2021-06-15T11:10:59.249788Z\",\"shell.execute_reply\":\"2021-06-15T11:10:59.264169Z\"}}\nprint(\"Number of duplicate images :\", df_train_images.id.duplicated().sum())\nprint(\"Number of duplicate study :\", df_train_images.StudyInstanceUID.duplicated().sum())\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:10:59.266191Z\",\"iopub.execute_input\":\"2021-06-15T11:10:59.266555Z\",\"iopub.status.idle\":\"2021-06-15T11:10:59.281938Z\",\"shell.execute_reply.started\":\"2021-06-15T11:10:59.266523Z\",\"shell.execute_reply\":\"2021-06-15T11:10:59.280934Z\"}}\n# Count the number of duplicated images\nunique_study_duplicate = df_train_images[df_train_images.StudyInstanceUID.duplicated()].StudyInstanceUID.unique()\nprint(\"Some duplicated id : \", ' ; '.join(unique_study_duplicate[:10]))\n\nimages_with_duplicate_study = df_train_images[df_train_images.StudyInstanceUID.isin(unique_study_duplicate)]\nprint(\"Number of image concernd with duplication :\", len(images_with_duplicate_study))\n\n# %% [markdown]\n# ### Visualization of some studies\n\n# %% [code] {\"_kg_hide-input\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:10:59.284279Z\",\"iopub.execute_input\":\"2021-06-15T11:10:59.284703Z\",\"iopub.status.idle\":\"2021-06-15T11:11:13.174412Z\",\"shell.execute_reply.started\":\"2021-06-15T11:10:59.284658Z\",\"shell.execute_reply\":\"2021-06-15T11:11:13.173303Z\"}}\ndef read_image_from_image(study_name, image_id):\n    image_name = image_id.split('_')[0]\n    file = glob.glob(\"../input/siim-covid19-detection/train/\" + study_name + \"/*/\" + image_name + \".dcm\")\n    ds = dcmread(file[0])\n    return ds.pixel_array\n\ndef show_sample_duplicate(samples):\n    nb_show_sample = min(5, len(samples))\n    fig, axes = plt.subplots(nrows=1, ncols=nb_show_sample, figsize=(nb_show_sample * 4, 4))\n    i = 0\n    for index, row in samples.iterrows():\n        img = read_image_from_image(row['StudyInstanceUID'], row['id'])\n        axes[i].imshow(img, cmap=plt.cm.gray, aspect='auto')\n        axes[i].axis('off')\n        i += 1\n        if i == 5:\n            break\n        \n    fig.suptitle(samples.StudyInstanceUID.unique()[0], fontsize=20)\n    fig.show()\n    \n\n# Get some sample from duplicate study\nnp.random.seed(42)\nduplicated_study_sample = np.random.choice(unique_study_duplicate, 5)\n\n# See the different values\nfor sample_study_name in duplicated_study_sample:\n    sample_duplicate_image = df_train_images[df_train_images.StudyInstanceUID == sample_study_name]\n    \n    show_sample_duplicate(sample_duplicate_image)    \n\n# %% [markdown]\n# It seems that for the images present for a given study, there are duplicated but with a different quality. If we took the first, the second and the last, we clearly have different brightness. Nevertheless the fourth seems to be the same. Finally, the third one is 4 different images with different brightness and cropping.\n\n# %% [markdown]\n# ### The particular 0fd2db233deb ID\n# \n# During my research, I found a particular ID, which have duplicated images.\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:11:13.175885Z\",\"iopub.execute_input\":\"2021-06-15T11:11:13.176374Z\",\"iopub.status.idle\":\"2021-06-15T11:11:13.190878Z\",\"shell.execute_reply.started\":\"2021-06-15T11:11:13.176341Z\",\"shell.execute_reply\":\"2021-06-15T11:11:13.189922Z\"}}\ndf_train_images[df_train_images.StudyInstanceUID == \"0fd2db233deb\"]\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:11:13.192138Z\",\"iopub.execute_input\":\"2021-06-15T11:11:13.192457Z\",\"iopub.status.idle\":\"2021-06-15T11:11:13.214124Z\",\"shell.execute_reply.started\":\"2021-06-15T11:11:13.192428Z\",\"shell.execute_reply\":\"2021-06-15T11:11:13.213033Z\"}}\ndf_train_study[df_train_study.id == \"0fd2db233deb_study\"]\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:11:13.215442Z\",\"iopub.execute_input\":\"2021-06-15T11:11:13.215818Z\",\"iopub.status.idle\":\"2021-06-15T11:11:18.863537Z\",\"shell.execute_reply.started\":\"2021-06-15T11:11:13.215788Z\",\"shell.execute_reply\":\"2021-06-15T11:11:18.860906Z\"}}\nshow_sample_duplicate(df_train_images[df_train_images.StudyInstanceUID == \"0fd2db233deb\"])\n\n# %% [markdown]\n# Regarding the 0fd2db233deb ID, we have duplicated images. Moreover, regarding the image information, we have a box information for a unique rows. \n# \n# So, in our dataset, we have **duplicate images**. Some are different (with different brigthness, different cropping, different angle) and some are the same. They represent 512 images of our dataset. It's represent about 8% (512 * 100 / 6334) of our dataset. \n\n# %% [code] {\"_kg_hide-input\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:11:18.864891Z\",\"iopub.execute_input\":\"2021-06-15T11:11:18.86523Z\",\"iopub.status.idle\":\"2021-06-15T11:11:18.891849Z\",\"shell.execute_reply.started\":\"2021-06-15T11:11:18.865181Z\",\"shell.execute_reply\":\"2021-06-15T11:11:18.890411Z\"}}\n# Rename the 'StudyInstanceUID' column\ndf_train_study['StudyInstanceUID'] = df_train_study['id'].apply(lambda x : x.replace('_study', ''))\n\n# Get the duplicated study\ndf_study_from_duplicate = df_train_study[df_train_study['StudyInstanceUID'].isin(images_with_duplicate_study['StudyInstanceUID'].unique())]\n\n# Get the duplicated images\ndf_image_from_duplicate = df_train_images[df_train_images.StudyInstanceUID.isin(unique_study_duplicate)]\n\n\n# Count for each category the number of duplicated study\nduplicate_study_case = [df_study_from_duplicate[label].value_counts()[1] for label in labels]\ntotal_study_case = [df_train_study[label].value_counts()[1] for label in labels]\n\n# Get the percentage for each category\nratio_duplicate = [x / y for x, y in zip(duplicate_study_case, total_study_case)] \n\nprint(\"Ratio total duplicated image : \", len(df_image_from_duplicate) / len(df_train_images))\nprint(\"Ratio total duplicated study : \", sum(duplicate_study_case) / sum(total_study_case))\n\nprint()\nprint(\"Percentage of duplicated study for each category :\")\nprint() \n\nfor i in range(len(ratio_duplicate)):\n    print(labels[i], \" : \", ratio_duplicate[i])\n\n# %% [code] {\"_kg_hide-input\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:11:18.893299Z\",\"iopub.execute_input\":\"2021-06-15T11:11:18.89362Z\",\"iopub.status.idle\":\"2021-06-15T11:11:19.035352Z\",\"shell.execute_reply.started\":\"2021-06-15T11:11:18.893591Z\",\"shell.execute_reply\":\"2021-06-15T11:11:19.034315Z\"}}\nplt.figure(figsize=(8, 8))\nplt.pie(ratio_duplicate, labels=labels, autopct='%1.1f%%', normalize=True)\nplt.suptitle(\"Distribution of duplications for each category\", fontsize=20)\nplt.show()\n\n# %% [markdown]\n# In this section, we saw the different duplicated study and images. Those could be x-ray images that could be retaken, maybe duplicated from copy/past or even images that have been analyze multiple times. Radiography analisys is a complex task, and some errors are possible even for the most brillant doctor. So we should keep in mind that maybe we could have error in our dataset.\n# \n# Nevertheless, concerning the application of the duplicate images, multiple possibilites are available. \n# - The simplest solution is to decide to ignore these files. As this is a small percentage of our dataset, this might be feasible. With this, we could avoid the duplication problem.\n# \n# - The other possibility is that we could decide to get some of the data. I mean not all the data have to be throws away. Some of them are duplicate files. For them, it would be good if we could analyse the group of images and keep only the best ones, with all the metadata information collected from the others. We could do the same for other similar images, those with a different brightness and cropping.\n\n# %% [markdown]\n# ## See some image with their boxes\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:11:19.036637Z\",\"iopub.execute_input\":\"2021-06-15T11:11:19.036941Z\",\"iopub.status.idle\":\"2021-06-15T11:11:19.054343Z\",\"shell.execute_reply.started\":\"2021-06-15T11:11:19.036912Z\",\"shell.execute_reply\":\"2021-06-15T11:11:19.053156Z\"}}\nsample_images_with_boxes = df_train_images[df_train_images.boxes.notna()].sample(n=10, random_state=42)\nsample_images_with_boxes.head()\n\n# %% [code] {\"_kg_hide-input\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:11:19.055937Z\",\"iopub.execute_input\":\"2021-06-15T11:11:19.056525Z\",\"iopub.status.idle\":\"2021-06-15T11:11:28.622817Z\",\"shell.execute_reply.started\":\"2021-06-15T11:11:19.05648Z\",\"shell.execute_reply\":\"2021-06-15T11:11:28.621803Z\"}}\nfig, axes = plt.subplots(nrows=2, ncols=5, figsize=(5 * 4, 4 * 2))\n\ni = 0\n# Iterate through the sample\nfor index, row in sample_images_with_boxes.iterrows():\n    # Read and show image\n    img = read_image_from_image(row['StudyInstanceUID'], row['id'])\n    axes[i // NUMBER_OF_SAMPLE, i % NUMBER_OF_SAMPLE].imshow(img, cmap=plt.cm.gray, aspect='auto')\n    \n    # The boxes are saved as str, we need to translate them to array of dict\n    array_boxes = ast.literal_eval(row.boxes) \n    \n    # Now, show the boxes\n    for box in array_boxes:\n        rect = patches.Rectangle((box['x'], box['y']),\n                                 box['width'], \n                                 box['height'], \n                                 edgecolor='r', \n                                 facecolor=\"none\")\n        \n        axes[i // NUMBER_OF_SAMPLE, i % NUMBER_OF_SAMPLE].add_patch(rect)\n    \n    # Remove axis information\n    axes[i // NUMBER_OF_SAMPLE, i % NUMBER_OF_SAMPLE].axis('off')\n    i += 1\n\n# %% [markdown]\n# With these images, we can first see that the images in this sample are really different. If we take the second one, I can barely see the content (and I have to turn the brightness of my screen to the maximum!). The fourth one is also interesting because the image has been rotated and cropped. In this sample we can really see the different image contrasts we have.\n# \n# \n# Regarding the boxes, on this sample, we see that we usually have two boxes and are places on the left and on the right. Moreover, they are mainly between the inferior and the middle lobe. However, this is a sample, we cannot make generalization on this small amount of data.\n# \n# \n# <img src=\"https://cdn.lecturio.com/assets/Lobes-and-fissures-of-the-lungs-1200x570.jpg\" width=\"800\" />\n# \n# Credit : https://www.lecturio.com/concepts/lungs/ - Image by Lecturio.\n\n# %% [code]\n\n\n# %% [markdown]\n# ### Box size analysis\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:11:28.623978Z\",\"iopub.execute_input\":\"2021-06-15T11:11:28.624299Z\",\"iopub.status.idle\":\"2021-06-15T11:11:43.286982Z\",\"shell.execute_reply.started\":\"2021-06-15T11:11:28.624266Z\",\"shell.execute_reply\":\"2021-06-15T11:11:43.286008Z\"}}\nsample_images_with_boxes = df_train_images[df_train_images.boxes.notna()]\nbox_size = pd.DataFrame()\n\nfor boxes in sample_images_with_boxes.boxes:\n    array_boxes = ast.literal_eval(boxes) \n    for box in array_boxes:\n        box_size = box_size.append(box, ignore_index=True)\n\nbox_size.head()\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:11:43.291854Z\",\"iopub.execute_input\":\"2021-06-15T11:11:43.292224Z\",\"iopub.status.idle\":\"2021-06-15T11:11:43.32527Z\",\"shell.execute_reply.started\":\"2021-06-15T11:11:43.292176Z\",\"shell.execute_reply\":\"2021-06-15T11:11:43.324249Z\"}}\nbox_size.describe()\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:11:43.327002Z\",\"iopub.execute_input\":\"2021-06-15T11:11:43.327314Z\",\"iopub.status.idle\":\"2021-06-15T11:11:43.481885Z\",\"shell.execute_reply.started\":\"2021-06-15T11:11:43.327282Z\",\"shell.execute_reply\":\"2021-06-15T11:11:43.48073Z\"}}\n# Show image size\nsizes = box_size.groupby(['height', 'width']).size().reset_index().rename(columns={0 : 'count'})\nsizes.hvplot.scatter(\n    x='height', \n    y='width', \n    size='count',\n    title='Box size distribution',\n    xlim=(0,3141), ylim=(0,1920), \n    grid=True, \n    height=500, width=1000).options(scaling_factor=0.1, line_alpha=1, fill_alpha=0)\n\n# %% [markdown]\n# Regarding the size of the boxes, they seem to have similar shapes, but very variable sizes. \n\n# %% [markdown]\n# #### Annotation label\n# \n# For each box, we have an *opacity* tag. The question we might ask is whether we have any other tags.\n\n# %% [code] {\"_kg_hide-input\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:11:43.483281Z\",\"iopub.execute_input\":\"2021-06-15T11:11:43.483609Z\",\"iopub.status.idle\":\"2021-06-15T11:11:43.501144Z\",\"shell.execute_reply.started\":\"2021-06-15T11:11:43.483576Z\",\"shell.execute_reply\":\"2021-06-15T11:11:43.499967Z\"}}\no = []\nfor label in df_train_images.label.values:\n    a = label.split(' ')\n    o.append(a[0])\n    \npd.Series(o).value_counts()\n\n# %% [markdown]\n# Here, we know we have only two tags for boxes : *none* or *opacity*\n\n# %% [code]\n\n\n# %% [markdown]\n# \n\n# %% [code]\n\n\n# %% [code]\n\n\n# %% [code] {\"_kg_hide-input\":true,\"_kg_hide-output\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:11:43.502548Z\",\"iopub.execute_input\":\"2021-06-15T11:11:43.502985Z\",\"iopub.status.idle\":\"2021-06-15T11:11:43.522305Z\",\"shell.execute_reply.started\":\"2021-06-15T11:11:43.502822Z\",\"shell.execute_reply\":\"2021-06-15T11:11:43.520964Z\"}}\n# Merge the two dataframe\ndf_merged_data = df_train_study.merge(df_train_images, on=\"StudyInstanceUID\")\n\n# %% [markdown]\n# ## Visualize some DICOM image\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:11:43.523589Z\",\"iopub.execute_input\":\"2021-06-15T11:11:43.523885Z\",\"iopub.status.idle\":\"2021-06-15T11:11:44.552667Z\",\"shell.execute_reply.started\":\"2021-06-15T11:11:43.523855Z\",\"shell.execute_reply\":\"2021-06-15T11:11:44.551637Z\"}}\npath = '../input/siim-covid19-detection/train/00086460a852/9e8302230c91/65761e66de9f.dcm'\nds = dcmread(path)\nprint(ds)\nplt.imshow(ds.pixel_array, cmap=plt.cm.gray)\nplt.show()\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:11:44.55403Z\",\"iopub.execute_input\":\"2021-06-15T11:11:44.554351Z\",\"iopub.status.idle\":\"2021-06-15T11:11:45.717883Z\",\"shell.execute_reply.started\":\"2021-06-15T11:11:44.554321Z\",\"shell.execute_reply\":\"2021-06-15T11:11:45.716967Z\"}}\npath = '../input/siim-covid19-detection/train/057c02a959f1/6de2191aa170/ba463980acdb.dcm'\nds = dcmread(path)\nprint(ds)\nplt.imshow(ds.pixel_array, cmap=plt.cm.gray)\nplt.show()\n\n# %% [markdown]\n# ## Get meta-information from train images\n\n# %% [code] {\"_kg_hide-input\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:11:45.719378Z\",\"iopub.execute_input\":\"2021-06-15T11:11:45.719791Z\",\"iopub.status.idle\":\"2021-06-15T11:15:45.152807Z\",\"shell.execute_reply.started\":\"2021-06-15T11:11:45.719748Z\",\"shell.execute_reply\":\"2021-06-15T11:15:45.151904Z\"}}\ndef dcm2metadata(sample):\n    metadata = {}\n    for key in sample.keys():\n        if key.group < 50:\n            item = sample.get(key)\n        if hasattr(item, 'description') and hasattr(item, 'value'):\n            metadata[item.description()] = str(item.value)\n    return metadata\n\nTRAIN_PATH = \"../input/siim-covid19-detection/train\"\ntrain_images_path = glob.glob(TRAIN_PATH + \"/*/*/*.dcm\")\nimage_metadata = pd.DataFrame()\n\n\nfor image in tqdm(train_images_path):    \n    # Read only the metadata here\n    ds = dcmread(image, stop_before_pixels=True)\n    info = dcm2metadata(ds)\n    image_metadata = image_metadata.append(info, ignore_index=True)\n        \nimage_metadata.head()\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:15:45.154741Z\",\"iopub.execute_input\":\"2021-06-15T11:15:45.155282Z\",\"iopub.status.idle\":\"2021-06-15T11:15:45.162011Z\",\"shell.execute_reply.started\":\"2021-06-15T11:15:45.15524Z\",\"shell.execute_reply\":\"2021-06-15T11:15:45.161191Z\"}}\nimage_metadata.columns\n\n# %% [markdown]\n# Based on the metadata, I decide to focus only on the following columns :\n# \n# - Patient ID\n# - Patient's Sex \n# - Modality\n# - Body Part Examined\n# - Image type\n# - Columns\n# - Rows\n\n# %% [markdown]\n# ### Patient ID\n\n# %% [code] {\"_kg_hide-input\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:15:45.163325Z\",\"iopub.execute_input\":\"2021-06-15T11:15:45.163606Z\",\"iopub.status.idle\":\"2021-06-15T11:15:45.181458Z\",\"shell.execute_reply.started\":\"2021-06-15T11:15:45.16358Z\",\"shell.execute_reply\":\"2021-06-15T11:15:45.180408Z\"}}\nprint(\"Number of unique patient : \", len(image_metadata[\"Patient ID\"].unique()))\n\n# %% [markdown]\n# ### Patient's Sex\n\n# %% [code] {\"_kg_hide-input\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:15:45.182845Z\",\"iopub.execute_input\":\"2021-06-15T11:15:45.183151Z\",\"iopub.status.idle\":\"2021-06-15T11:15:45.309649Z\",\"shell.execute_reply.started\":\"2021-06-15T11:15:45.183125Z\",\"shell.execute_reply\":\"2021-06-15T11:15:45.308508Z\"}}\nnb_male = len(image_metadata[image_metadata[\"Patient's Sex\"] == 'M'])\nnb_female = len(image_metadata[image_metadata[\"Patient's Sex\"] == 'F'])\n\nplt.figure(figsize=(6,6))\nplt.title(\"Gender distribution\")\nplt.pie([nb_male, nb_female], labels=['Male', 'Female'], autopct='%1.1f%%', colors=['b', 'r'])\nplt.show()\n\n# %% [markdown]\n# ### Modality\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:15:45.310871Z\",\"iopub.execute_input\":\"2021-06-15T11:15:45.311165Z\",\"iopub.status.idle\":\"2021-06-15T11:15:45.321412Z\",\"shell.execute_reply.started\":\"2021-06-15T11:15:45.311138Z\",\"shell.execute_reply\":\"2021-06-15T11:15:45.320333Z\"}}\nimage_metadata[\"Modality\"].value_counts()\n\n# %% [markdown]\n# ### Body Part Examined\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:15:45.322979Z\",\"iopub.execute_input\":\"2021-06-15T11:15:45.323434Z\",\"iopub.status.idle\":\"2021-06-15T11:15:45.339723Z\",\"shell.execute_reply.started\":\"2021-06-15T11:15:45.323371Z\",\"shell.execute_reply\":\"2021-06-15T11:15:45.338587Z\"}}\nimage_metadata[\"Body Part Examined\"].value_counts()\n\n# %% [code]\n\n\n# %% [code] {\"_kg_hide-input\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:15:45.341341Z\",\"iopub.execute_input\":\"2021-06-15T11:15:45.34177Z\",\"iopub.status.idle\":\"2021-06-15T11:15:48.362344Z\",\"shell.execute_reply.started\":\"2021-06-15T11:15:45.341728Z\",\"shell.execute_reply\":\"2021-06-15T11:15:48.361342Z\"}}\ndef get_sample_body_part(sample, title):\n    fig, axes = plt.subplots(nrows=1, ncols=3, figsize=(3 * 4, 4))\n    fig.suptitle(title)\n    i = 0\n    for study_id in sample['Study Instance UID'].values:\n        path = glob.glob('../input/siim-covid19-detection/train/' + study_id + '/*/*.dcm')\n        ds = dcmread(path[0])\n        axes[i].imshow(ds.pixel_array, cmap=plt.cm.gray, aspect='auto')\n        axes[i].axis('off')\n        i+=1    \n\nsample_port_chest = image_metadata[image_metadata[\"Body Part Examined\"] == \"PORT CHEST\"].sample(n=3, random_state=42)\nget_sample_body_part(sample_port_chest, 'Radiography Port Chest')\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:15:48.363631Z\",\"iopub.execute_input\":\"2021-06-15T11:15:48.363934Z\",\"iopub.status.idle\":\"2021-06-15T11:15:51.790412Z\",\"shell.execute_reply.started\":\"2021-06-15T11:15:48.363904Z\",\"shell.execute_reply\":\"2021-06-15T11:15:51.789284Z\"}}\nsample_port_chest = image_metadata[image_metadata[\"Body Part Examined\"] == \"\"].sample(n=3, random_state=42)\nget_sample_body_part(sample_port_chest, 'Radiography Empty')\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:15:51.791614Z\",\"iopub.execute_input\":\"2021-06-15T11:15:51.791887Z\",\"iopub.status.idle\":\"2021-06-15T11:15:55.208667Z\",\"shell.execute_reply.started\":\"2021-06-15T11:15:51.79186Z\",\"shell.execute_reply\":\"2021-06-15T11:15:55.207611Z\"}}\nsample_port_chest = image_metadata[image_metadata[\"Body Part Examined\"] == \"SKULL\"].sample(n=3, random_state=42)\nget_sample_body_part(sample_port_chest, 'Radiography Skull')\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:15:55.209765Z\",\"iopub.execute_input\":\"2021-06-15T11:15:55.210173Z\",\"iopub.status.idle\":\"2021-06-15T11:15:58.986071Z\",\"shell.execute_reply.started\":\"2021-06-15T11:15:55.210141Z\",\"shell.execute_reply\":\"2021-06-15T11:15:58.984999Z\"}}\nsample_port_chest = image_metadata[image_metadata[\"Body Part Examined\"] == \"Pecho\"].sample(n=3, random_state=42)\nget_sample_body_part(sample_port_chest, 'Radiography Pecho')\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:15:58.987696Z\",\"iopub.execute_input\":\"2021-06-15T11:15:58.988089Z\",\"iopub.status.idle\":\"2021-06-15T11:16:03.56643Z\",\"shell.execute_reply.started\":\"2021-06-15T11:15:58.988049Z\",\"shell.execute_reply\":\"2021-06-15T11:16:03.56563Z\"}}\nsample_port_chest = image_metadata[image_metadata[\"Body Part Examined\"] == \"ABDOMEN\"].sample(n=3, random_state=42)\nget_sample_body_part(sample_port_chest, 'Radiography ABDOMEN')\n\n# %% [markdown]\n# ### Image type\n\n# %% [code] {\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:16:03.567546Z\",\"iopub.execute_input\":\"2021-06-15T11:16:03.567934Z\",\"iopub.status.idle\":\"2021-06-15T11:16:03.577443Z\",\"shell.execute_reply.started\":\"2021-06-15T11:16:03.567905Z\",\"shell.execute_reply\":\"2021-06-15T11:16:03.576702Z\"}}\nimage_metadata[\"Image Type\"].value_counts()\n\n# %% [code]\n\n\n# %% [markdown]\n# ### Image size analysis\n\n# %% [code] {\"_kg_hide-input\":true,\"execution\":{\"iopub.status.busy\":\"2021-06-15T11:16:03.578499Z\",\"iopub.execute_input\":\"2021-06-15T11:16:03.578959Z\",\"iopub.status.idle\":\"2021-06-15T11:16:03.714825Z\",\"shell.execute_reply.started\":\"2021-06-15T11:16:03.578921Z\",\"shell.execute_reply\":\"2021-06-15T11:16:03.713833Z\"}}\n# Convert dtype\nimage_metadata.Columns = np.array(image_metadata.Columns, dtype=int)\nimage_metadata.Rows = np.array(image_metadata.Rows, dtype=int)\n\n# Show image size\nsizes = image_metadata.groupby(['Columns', 'Rows']).size().reset_index().rename(columns={0 : 'count'})\nsizes.hvplot.scatter(\n    x='Columns', \n    y='Rows', \n    size='count',\n    title='Image size distribution',\n    xlim=(0,5000), ylim=(0,5000), \n    grid=True, \n    height=500, width=1000).options(scaling_factor=0.1, line_alpha=1, fill_alpha=0)\n\n# %% [markdown]\n# As we can see on the top diagram, the size of the images seems to follow a linear line starting from 0. It seems that we have globally square sized images. Moreover, we have a high concentration of images with a size between 2000 and 3000 pixels.\n\n# %% [markdown]\n# conclusion \n#  as finally i conclude the above were the covid19 patients on chest radio graphs","metadata":{"_uuid":"33388039-18d9-425f-80fd-cce34fc2c836","_cell_guid":"46cc4fff-e7aa-4db1-867e-c0328131fa7d","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]}]}