{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Simple EDA process","metadata":{}},{"cell_type":"markdown","source":"install GDCM package","metadata":{}},{"cell_type":"code","source":"!conda install -c conda-forge gdcm -y","metadata":{"execution":{"iopub.status.busy":"2021-07-21T04:42:57.980369Z","iopub.execute_input":"2021-07-21T04:42:57.980791Z","iopub.status.idle":"2021-07-21T04:43:04.332520Z","shell.execute_reply.started":"2021-07-21T04:42:57.980709Z","shell.execute_reply":"2021-07-21T04:43:04.331664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Importing necessary packages","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport pydicom\nimport glob\nfrom tqdm.notebook import tqdm\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport matplotlib.pyplot as plt\nfrom skimage import exposure\nimport cv2\nimport warnings\nfrom fastai.vision.all import *\nfrom fastai.medical.imaging import *\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2021-07-16T10:44:20.993116Z","iopub.execute_input":"2021-07-16T10:44:20.993477Z","iopub.status.idle":"2021-07-16T10:44:21.000368Z","shell.execute_reply.started":"2021-07-16T10:44:20.993446Z","shell.execute_reply":"2021-07-16T10:44:20.99858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Listing all the files available","metadata":{}},{"cell_type":"code","source":"dataset_path = Path('../input/siim-covid19-detection')\nl1=dataset_path.ls()\nfor l in l1:\n    print(l)","metadata":{"execution":{"iopub.status.busy":"2021-07-16T10:44:22.107919Z","iopub.execute_input":"2021-07-16T10:44:22.108317Z","iopub.status.idle":"2021-07-16T10:44:22.117851Z","shell.execute_reply.started":"2021-07-16T10:44:22.108283Z","shell.execute_reply":"2021-07-16T10:44:22.116727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Viewing Study Level CSV","metadata":{}},{"cell_type":"code","source":"train_study_df = pd.read_csv(dataset_path/'train_study_level.csv')\nprint(train_study_df.shape)\ntrain_study_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-16T10:44:23.425998Z","iopub.execute_input":"2021-07-16T10:44:23.426369Z","iopub.status.idle":"2021-07-16T10:44:23.469778Z","shell.execute_reply.started":"2021-07-16T10:44:23.426335Z","shell.execute_reply":"2021-07-16T10:44:23.468805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Storing unique studyIDs","metadata":{}},{"cell_type":"code","source":"lst = np.unique(train_study_df.id)\nlen(lst)","metadata":{"execution":{"iopub.status.busy":"2021-07-16T10:44:24.625991Z","iopub.execute_input":"2021-07-16T10:44:24.626641Z","iopub.status.idle":"2021-07-16T10:44:24.63947Z","shell.execute_reply.started":"2021-07-16T10:44:24.626578Z","shell.execute_reply":"2021-07-16T10:44:24.638359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Print unique classes in Study Level\nNote: Target classes are One Hot Encoded","metadata":{}},{"cell_type":"code","source":"study_classes = ['Negative for Pneumonia', 'Typical Appearance', 'Indeterminate Appearance', 'Atypical Appearance']\nnp.unique(train_study_df[study_classes].values, axis=0)","metadata":{"execution":{"iopub.status.busy":"2021-07-16T10:44:25.470993Z","iopub.execute_input":"2021-07-16T10:44:25.471327Z","iopub.status.idle":"2021-07-16T10:44:25.485622Z","shell.execute_reply.started":"2021-07-16T10:44:25.471296Z","shell.execute_reply":"2021-07-16T10:44:25.484782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plotting the Class vs Frequency graph - Study Level","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\nplt.bar([1,2,3,4], train_study_df[study_classes].values.sum(axis=0))\nplt.xticks([1,2,3,4], study_classes)\nplt.xlabel('Class')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-16T10:44:26.468019Z","iopub.execute_input":"2021-07-16T10:44:26.468488Z","iopub.status.idle":"2021-07-16T10:44:26.616585Z","shell.execute_reply.started":"2021-07-16T10:44:26.468455Z","shell.execute_reply":"2021-07-16T10:44:26.615701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Viewing Image Level CSV","metadata":{}},{"cell_type":"code","source":"train_image_df = pd.read_csv(dataset_path/'train_image_level.csv')\nprint(train_image_df.shape)\ntrain_image_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-16T10:44:27.558748Z","iopub.execute_input":"2021-07-16T10:44:27.559257Z","iopub.status.idle":"2021-07-16T10:44:27.618021Z","shell.execute_reply.started":"2021-07-16T10:44:27.559224Z","shell.execute_reply":"2021-07-16T10:44:27.617343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Splitting the label attribute of image level CSV","metadata":{}},{"cell_type":"code","source":"train_image_df['split_label'] = train_image_df.label.apply(lambda x:[x.split()[offs:offs+6] for offs in range(0, len(x.split()),6)])\ntrain_image_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-16T10:44:28.920276Z","iopub.execute_input":"2021-07-16T10:44:28.920624Z","iopub.status.idle":"2021-07-16T10:44:28.956374Z","shell.execute_reply.started":"2021-07-16T10:44:28.920581Z","shell.execute_reply":"2021-07-16T10:44:28.955683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Finding Class Frequency and Area under box frequency","metadata":{}},{"cell_type":"code","source":"classes_freq = []\nbbox_areas = []\nfor i in range(len(train_image_df)):\n    for j in train_image_df.iloc[i].split_label:\n        classes_freq.append(j[0])\n        bbox_areas.append((float(j[4])-float(j[2]))*(float(j[5])*float(j[3])))\nplt.hist(classes_freq)\nplt.ylabel('Frequency')","metadata":{"execution":{"iopub.status.busy":"2021-07-16T10:44:29.868148Z","iopub.execute_input":"2021-07-16T10:44:29.868635Z","iopub.status.idle":"2021-07-16T10:44:30.664553Z","shell.execute_reply.started":"2021-07-16T10:44:29.868591Z","shell.execute_reply":"2021-07-16T10:44:30.663833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plotting Bounding Box Areas","metadata":{}},{"cell_type":"code","source":"plt.hist(bbox_areas)\nplt.ylabel('Frequency')","metadata":{"execution":{"iopub.status.busy":"2021-07-16T10:44:31.526175Z","iopub.execute_input":"2021-07-16T10:44:31.526741Z","iopub.status.idle":"2021-07-16T10:44:31.843499Z","shell.execute_reply.started":"2021-07-16T10:44:31.526705Z","shell.execute_reply":"2021-07-16T10:44:31.842826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Functions to convert .dcm to numpy arrays","metadata":{}},{"cell_type":"code","source":"def dicom2array(path, voi_lut=True, fix_monochrome=True):\n    dicom = pydicom.read_file(path)\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data*255).astype(np.uint8)\n    return data\n\ndef plot_img(img, size=(7,7), is_rgb=True, title=\"\", cmap='gray'):\n    plt.figure(figsize=size)\n    plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n    \ndef plot_imgs(imgs, cols=4, size=7, is_rgb=True, title=\"\", cmap='gray', img_size=(500, 500)):\n    rows = len(imgs)//cols + 1\n    fig = plt.figure(figsize=(cols*size, rows*size))\n    for i, img in enumerate(imgs):\n        if img_size is not None:\n            img = cv2.resize(img, img_size)\n        fig.add_subplot(rows, cols, i+1)\n        plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-16T10:44:32.663324Z","iopub.execute_input":"2021-07-16T10:44:32.663828Z","iopub.status.idle":"2021-07-16T10:44:32.672357Z","shell.execute_reply.started":"2021-07-16T10:44:32.663784Z","shell.execute_reply":"2021-07-16T10:44:32.671092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plotting x-ray images","metadata":{}},{"cell_type":"code","source":"dicom_paths = get_dicom_files(dataset_path/'train')\nimgs = [dicom2array(path) for path in dicom_paths[:4]]\nplot_imgs(imgs)","metadata":{"execution":{"iopub.status.busy":"2021-07-16T10:44:35.388341Z","iopub.execute_input":"2021-07-16T10:44:35.388698Z","iopub.status.idle":"2021-07-16T10:45:00.094287Z","shell.execute_reply.started":"2021-07-16T10:44:35.388666Z","shell.execute_reply":"2021-07-16T10:45:00.093371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Study with more than 3 images","metadata":{}},{"cell_type":"code","source":"num_images_per_study = []\nfor i in (dataset_path/'train').ls():\n    num_images_per_study.append(len(get_dicom_files(i)))\n    if len(get_dicom_files(i))>3:\n        print(f'Study {i} has {len(get_dicom_files(i))} images')\nplt.hist(num_images_per_study)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-16T10:45:23.196463Z","iopub.execute_input":"2021-07-16T10:45:23.197028Z","iopub.status.idle":"2021-07-16T10:45:33.791761Z","shell.execute_reply.started":"2021-07-16T10:45:23.196991Z","shell.execute_reply":"2021-07-16T10:45:33.790782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Extracting image paths","metadata":{}},{"cell_type":"code","source":"def image_path(row):\n    study_path = dataset_path/'train'/row.StudyInstanceUID\n    for i in get_dicom_files(study_path):\n        if row.id.split('_')[0] == i.stem:\n            return i\ntrain_image_df['image_path'] = train_image_df.apply(image_path, axis=1)\ntrain_image_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-16T10:45:33.868753Z","iopub.execute_input":"2021-07-16T10:45:33.869084Z","iopub.status.idle":"2021-07-16T10:45:40.386172Z","shell.execute_reply.started":"2021-07-16T10:45:33.869054Z","shell.execute_reply":"2021-07-16T10:45:40.385191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plotting bounding box in x-ray images","metadata":{}},{"cell_type":"code","source":"imgs = []\nimage_paths = train_image_df['image_path'].values\n# map label_id to specify color\nthickness = 10\nscale = 5\nfor i in range(8):\n    image_path = random.choice(image_paths)\n    print(image_path)\n    img = dicom2array(path=image_path)\n    img = cv2.resize(img, None, fx=1/scale, fy=1/scale)\n    img = np.stack([img, img, img], axis=-1)\n    for i in train_image_df.loc[train_image_df['image_path'] == image_path].split_label.values[0]:\n        if i[0] == 'opacity':\n            img = cv2.rectangle(img,\n                                (int(float(i[2])/scale), int(float(i[3])/scale)),\n                                (int(float(i[4])/scale), int(float(i[5])/scale)),\n                                [255,0,0], thickness)\n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\nplot_imgs(imgs, cmap=None)","metadata":{"execution":{"iopub.status.busy":"2021-07-16T10:45:53.740201Z","iopub.execute_input":"2021-07-16T10:45:53.740527Z","iopub.status.idle":"2021-07-16T10:45:59.832087Z","shell.execute_reply.started":"2021-07-16T10:45:53.740498Z","shell.execute_reply":"2021-07-16T10:45:59.831293Z"},"trusted":true},"execution_count":null,"outputs":[]}]}