{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport warnings\nimport os\nimport pydicom\nimport cv2","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-09T15:14:54.460956Z","iopub.execute_input":"2021-06-09T15:14:54.461315Z","iopub.status.idle":"2021-06-09T15:14:54.467363Z","shell.execute_reply.started":"2021-06-09T15:14:54.461286Z","shell.execute_reply":"2021-06-09T15:14:54.466013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Explaining some confusing aspects of the data\nEvery image in train_image data is assigned to one study in train_study data (StudyInstanceUID column). But one study can contain multiple images (more info in the **STUDY DATA** section).  \n\nColumns 'boxes' and 'label' in train_image are somewhat redundant:\n* boxes contains location and size of the boxes\n* label contains location (xmin, ymin, xmax, ymax) of the boxes and confidence (which for the training data is always 1)\n\nSo we don't need 'boxes' column, we can just use the info from 'label' column.\n\n### The 'label' column in train_image data\nlabel column has a structure:  \na) for image with no boxes: `none 1 0 0 1 1`  \nb) for image with boxes: `opacity 1 <xmin> <ymin> <xmax> <ymax>`\n* if there is more than one box, there are multiple labels like this in a row, for example:  \n`opacity 1 789.29 582.43 1815.94 2499.73 opacity 1 2245.91 591.21 3340.57 2352.75`\n  \n\n### How should the submission file look like?\nFor images it is consistent with the label column!<br/>\na) for image with no boxes: `id_image,none 1 0 0 1 1`  \nb) for image with boxes: `id_image,opacity <confidence> <xmin> <ymin> <xmax> <ymax>`  \nAnd for studies: `id_study,<class> <confidence> 0 0 1 1`\n\nExample from the Kaggle Evaluation tab:\n```\nId,PredictionString\n2b95d54e4be65_study,negative 1 0 0 1 1\n2b95d54e4be66_study,typical 1 0 0 1 1\n2b95d54e4be67_study,indeterminate 1 0 0 1 1 atypical 1 0 0 1 1\n2b95d54e4be68_image,none 1 0 0 1 1\n2b95d54e4be69_image,opacity 0.5 100 100 200 200 opacity 0.7 10 10 20 20\n```\n\nAlso, if I understand correctly, every study in train and test data is assigned to exactly one class (the classes are mutually exclusive) and it's best if you predict that class. But if you are not sure about your prediction, you can also try multilabel classification, for example: `2b95d54e4be65_study,negative 0.7 0 0 1 1 indeterminate 0.6 0 0 1 1`.","metadata":{}},{"cell_type":"markdown","source":"## Let's look at the data!","metadata":{}},{"cell_type":"code","source":"train_image = pd.read_csv('../input/siim-covid19-detection/train_image_level.csv')\ntrain_study = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv')\nprint('shape of train_image:', train_image.shape)\ntrain_image.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:34:37.502616Z","iopub.execute_input":"2021-06-09T15:34:37.503033Z","iopub.status.idle":"2021-06-09T15:34:37.564918Z","shell.execute_reply.started":"2021-06-09T15:34:37.503001Z","shell.execute_reply":"2021-06-09T15:34:37.563659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('shape of train_study:', train_study.shape)\ntrain_study.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:14:54.530332Z","iopub.execute_input":"2021-06-09T15:14:54.530780Z","iopub.status.idle":"2021-06-09T15:14:54.545026Z","shell.execute_reply.started":"2021-06-09T15:14:54.530747Z","shell.execute_reply":"2021-06-09T15:14:54.543605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DIR = '../input/siim-covid19-detection/train'\nTEST_DIR = '../input/siim-covid19-detection/test'","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:14:54.546873Z","iopub.execute_input":"2021-06-09T15:14:54.547192Z","iopub.status.idle":"2021-06-09T15:14:54.552778Z","shell.execute_reply.started":"2021-06-09T15:14:54.547163Z","shell.execute_reply":"2021-06-09T15:14:54.551411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# IMAGE DATA\n\nThis is how you can get to an image:","metadata":{}},{"cell_type":"code","source":"# take first example and get path to it\nfrom os import walk, listdir\nimage_dir = train_image['StudyInstanceUID'][0]\npath_to_img = TRAIN_DIR + '/' + image_dir \npath_to_img = path_to_img +'/' + listdir(path_to_img)[0] \npath_to_img = path_to_img + '/' + next(walk(path_to_img))[2][0]","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:14:54.554245Z","iopub.execute_input":"2021-06-09T15:14:54.554750Z","iopub.status.idle":"2021-06-09T15:14:54.567846Z","shell.execute_reply.started":"2021-06-09T15:14:54.554718Z","shell.execute_reply":"2021-06-09T15:14:54.566718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The image files are in DICOM (.dcm) format. This format is often used in the medical images since it contains information about the patient (to avoid mismatching the patients' data). DICOM format can be handled using pydicom package (see more: https://pydicom.github.io/pydicom/stable/index.html )","metadata":{}},{"cell_type":"code","source":"data = pydicom.dcmread(path_to_img)\n# get the pixel information into a numpy array\nimg = data.pixel_array\nprint('The image has {} x {} voxels'.format(img.shape[0],\n                                            img.shape[1]))","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:14:54.569228Z","iopub.execute_input":"2021-06-09T15:14:54.569674Z","iopub.status.idle":"2021-06-09T15:14:54.613394Z","shell.execute_reply.started":"2021-06-09T15:14:54.569618Z","shell.execute_reply":"2021-06-09T15:14:54.612097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(img, img.shape)","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:14:54.615818Z","iopub.execute_input":"2021-06-09T15:14:54.616257Z","iopub.status.idle":"2021-06-09T15:14:54.623039Z","shell.execute_reply.started":"2021-06-09T15:14:54.616213Z","shell.execute_reply":"2021-06-09T15:14:54.621681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.imshow(img, cmap='gray')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:15:05.398243Z","iopub.execute_input":"2021-06-09T15:15:05.398621Z","iopub.status.idle":"2021-06-09T15:15:06.888856Z","shell.execute_reply.started":"2021-06-09T15:15:05.398589Z","shell.execute_reply":"2021-06-09T15:15:06.887745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# this can help with increasing the contrast\nfrom skimage import exposure\nequ_img = exposure.equalize_hist(img)\nplt.imshow(equ_img, cmap='gray')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:15:09.589982Z","iopub.execute_input":"2021-06-09T15:15:09.590310Z","iopub.status.idle":"2021-06-09T15:15:11.736013Z","shell.execute_reply.started":"2021-06-09T15:15:09.590283Z","shell.execute_reply":"2021-06-09T15:15:11.735203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"equ_img","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:15:11.737261Z","iopub.execute_input":"2021-06-09T15:15:11.737719Z","iopub.status.idle":"2021-06-09T15:15:11.743710Z","shell.execute_reply.started":"2021-06-09T15:15:11.737673Z","shell.execute_reply":"2021-06-09T15:15:11.742752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's see the boxes...","metadata":{}},{"cell_type":"code","source":"box1 = train_image.label[0].split()[:6]\nbox2 = train_image.label[0].split()[6:]\nimg = cv2.rectangle(img,(int(float(box1[2])), int(float(box1[3]))), \n                    (int(float(box1[4])), int(float(box1[5]))),\n                    color=(0, 0, 0), thickness=15)\n\nimg = cv2.rectangle(img,(int(float(box2[2])), int(float(box2[3]))), \n                    (int(float(box2[4])), int(float(box2[5]))),\n                    color=(0, 0, 0), thickness=15)\nplt.imshow(img, cmap='gray')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:29:40.886074Z","iopub.execute_input":"2021-06-09T15:29:40.886533Z","iopub.status.idle":"2021-06-09T15:29:42.277096Z","shell.execute_reply.started":"2021-06-09T15:29:40.886480Z","shell.execute_reply":"2021-06-09T15:29:42.276107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now on image with equalize_hist\nequ_img = cv2.rectangle(equ_img,(int(float(box1[2])), int(float(box1[3]))), \n                    (int(float(box1[4])), int(float(box1[5]))),\n                    color=(0, 0, 0), thickness=15)\n\nequ_img = cv2.rectangle(equ_img,(int(float(box2[2])), int(float(box2[3]))), \n                    (int(float(box2[4])), int(float(box2[5]))),\n                    color=(0, 0, 0), thickness=15)\nplt.imshow(equ_img, cmap='gray')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:29:31.847978Z","iopub.execute_input":"2021-06-09T15:29:31.848589Z","iopub.status.idle":"2021-06-09T15:29:33.468023Z","shell.execute_reply.started":"2021-06-09T15:29:31.848551Z","shell.execute_reply":"2021-06-09T15:29:33.467009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shape1, shape2, ratios = [], [], []\nfor study_id in set(train_image.StudyInstanceUID):\n    path1 = TRAIN_DIR + '/' + study_id \n    for p in listdir(path1):\n        path_to_img = path1 +'/' + p\n        path_to_img = path_to_img + '/' + next(walk(path_to_img))[2][0]\n        data = pydicom.dcmread(path_to_img)\n        sh1 = data.Rows\n        sh2 = data.Columns\n        shape1.append(sh1)\n        shape2.append(sh2)\n        ratios.append(sh1/sh2)","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:14:55.790818Z","iopub.status.idle":"2021-06-09T15:14:55.791263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(shape1, bins=20)\nplt.title('Height of the images')\nplt.show()\nplt.hist(shape2, bins=20)\nplt.title('Width of the images')\nplt.show()\nplt.hist(ratios, bins=20)\nplt.title('Height to width ratio of the images')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:14:55.792407Z","iopub.status.idle":"2021-06-09T15:14:55.792884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# STUDY DATA","metadata":{}},{"cell_type":"code","source":"# the studies in train_study_level.csv file aren't multilabeled:\nlabels_sum = train_study.iloc[:,1:].sum(axis=1)\nprint('Min number of labels:', min(labels_sum))\nprint('Max number of labels:', max(labels_sum))","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:14:55.793784Z","iopub.status.idle":"2021-06-09T15:14:55.794205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_study.iloc[:,1:].sum().plot(kind='bar', figsize=(10,6), grid=True, rot=0,\n                                  title='Frequency of the labels', width=2/3)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:14:55.795091Z","iopub.status.idle":"2021-06-09T15:14:55.795536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For most of the studies (5822 out of 6334) there is only 1 image but for some there are more (up to 9 images).  \nHere is an explanation from Competition Host:\n> Most of the studies only have 1 image.  \nIn some cases, however, there are studies with more than 1 image. In these cases, patients were imaged more than once on the same date/time (same StudyInstanceUID). In some cases, there is motion artifact, so the tech re-took the image. In other cases, different image processing is applied (the images look almost identical, but there is subtle change in contrast). In other cases, there are coverage, image penetration, or other technique issues, presumably resulting in the technologist needing to retake radiographs.  \n(...) We are addressing this issue currently regarding the duplicates and test set. We'll let you know when this process is completed.\n\n[source](https://www.kaggle.com/c/siim-covid19-detection/discussion/240250#1322940)","metadata":{}},{"cell_type":"code","source":"train_image.StudyInstanceUID.value_counts().plot(kind='hist', logy=True, bins=np.arange(1,11)-0.5,\n                                           xticks=range(1,10), title='Number of images per study',\n                                           figsize=(10,6), rwidth=0.9)\nplt.show()\nprint(train_image.StudyInstanceUID.value_counts().value_counts())","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:14:55.796595Z","iopub.status.idle":"2021-06-09T15:14:55.797037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Boxes\nAll of the images marked as \"Negative for Pneumonia\" don't have any boxes. But there are also some images from other categories with no boxes.","metadata":{}},{"cell_type":"code","source":"n_nan = train_image.boxes.isna().sum()\nn_boxes = len(train_image.boxes) - n_nan\nprint(f'There are {n_boxes} images with boxes and {n_nan} images without any boxes.')","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:31:38.095167Z","iopub.execute_input":"2021-06-09T15:31:38.095600Z","iopub.status.idle":"2021-06-09T15:31:38.103871Z","shell.execute_reply.started":"2021-06-09T15:31:38.095562Z","shell.execute_reply":"2021-06-09T15:31:38.102466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Labels of images with no boxes')\ntrain_study[train_study.id.str.split('_').str[0].isin(set(train_image[train_image.boxes.isna()].StudyInstanceUID))].iloc[:,1:].sum()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:32:37.762699Z","iopub.execute_input":"2021-06-09T15:32:37.763180Z","iopub.status.idle":"2021-06-09T15:32:37.791706Z","shell.execute_reply.started":"2021-06-09T15:32:37.763150Z","shell.execute_reply":"2021-06-09T15:32:37.790552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Labels of images with boxes')\ntrain_study[train_study.id.str.split('_').str[0].isin(set(train_image[~train_image.boxes.isna()].StudyInstanceUID))].iloc[:,1:].sum()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:32:46.951687Z","iopub.execute_input":"2021-06-09T15:32:46.952086Z","iopub.status.idle":"2021-06-09T15:32:46.984984Z","shell.execute_reply.started":"2021-06-09T15:32:46.952053Z","shell.execute_reply":"2021-06-09T15:32:46.983933Z"},"trusted":true},"execution_count":null,"outputs":[]}]}