{"cells":[{"metadata":{"_uuid":"6e1ab15f588fc6c54baacf68d87eb97b054a197b"},"cell_type":"markdown","source":"In this notebook, I visualize the dataset using OpenCV.\n\nYou may find the following links useful\n- [EDA](https://www.kaggle.com/peterchang77/exploratory-data-analysis)\n- [Learn OpenCV](https://docs.opencv.org/master/d6/d00/tutorial_py_root.html)\n- [Image Processing TGS ](https://www.kaggle.com/yushas/imageprocessingtips)\n- [Image Enhancement](https://www.kaggle.com/meaninglesslives/simple-feature-extraction-and-image-enhancement)\n\nI will keep updating the kernel as i explore further."},{"metadata":{"_uuid":"b8b636767610accc9d0e6b5896feca2bfb2ee2ba"},"cell_type":"markdown","source":"This dataset has three types of images namely\n\n*   No Lung Opacity / Not Normal\n*   Normal\n*   Lung Opacity\n\nI visualize each class separately to get better idea"},{"metadata":{"_uuid":"c2a0e0a494e7da38a56a05d1849d7a0a07f3ccde"},"cell_type":"markdown","source":"# Loading required libraries"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport sys\nimport random\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nfrom skimage.transform import resize\nfrom skimage.morphology import label\nfrom skimage.feature import hog\nfrom skimage import exposure\nfrom keras.preprocessing.image import ImageDataGenerator, array_to_img, img_to_array, load_img\nfrom skimage.feature import canny\nfrom skimage.filters import sobel\nfrom skimage.morphology import watershed\nfrom scipy import ndimage as ndi\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nfrom skimage.segmentation import mark_boundaries\nfrom scipy import signal\nimport cv2\nimport glob, pylab, pandas as pd\nimport pydicom, numpy as np\nimport tqdm\nimport gc\n# gc.enable()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6b89fc0c491ff50b02a282d174a29d2f7e295bba"},"cell_type":"markdown","source":"# Some utility functions"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"_kg_hide-input":false},"cell_type":"code","source":"# https://www.kaggle.com/peterchang77/exploratory-data-analysis\ndef parse_data(df):\n    \"\"\"\n    Method to read a CSV file (Pandas dataframe) and parse the \n    data into the following nested dictionary:\n\n      parsed = {\n        \n        'patientId-00': {\n            'dicom': path/to/dicom/file,\n            'label': either 0 or 1 for normal or pnuemonia, \n            'boxes': list of box(es)\n        },\n        'patientId-01': {\n            'dicom': path/to/dicom/file,\n            'label': either 0 or 1 for normal or pnuemonia, \n            'boxes': list of box(es)\n        }, ...\n\n      }\n\n    \"\"\"\n    # --- Define lambda to extract coords in list [y, x, height, width]\n    extract_box = lambda row: [row['y'], row['x'], row['height'], row['width']]\n\n    parsed = {}\n    for n, row in df.iterrows():\n        # --- Initialize patient entry into parsed \n        pid = row['patientId']\n        if pid not in parsed:\n            parsed[pid] = {\n                'dicom': '../input/stage_1_train_images/%s.dcm' % pid,\n                'label': row['Target'],\n                'boxes': []}\n\n        # --- Add box if opacity is present\n        if parsed[pid]['label'] == 1:\n            parsed[pid]['boxes'].append(extract_box(row))\n\n    return parsed\n# https://www.kaggle.com/peterchang77/exploratory-data-analysis\ndef draw(data,im):\n    \"\"\"\n    Method to draw single patient with bounding box(es) if present \n\n    \"\"\"\n\n    # --- Convert from single-channel grayscale to 3-channel RGB\n    im = np.stack([im] * 3, axis=2)\n\n    # --- Add boxes with random color if present\n    for box in data['boxes']:\n        rgb = np.floor(np.random.rand(3) * 256).astype('int')\n        im = overlay_box(im=im, box=box, rgb=rgb, stroke=6)\n        \n    return im\n\ndef overlay_box(im, box, rgb, stroke=1):\n    \"\"\"\n    Method to overlay single box on image\n\n    \"\"\"\n    # --- Convert coordinates to integers\n    box = [int(b) for b in box]\n    \n    # --- Extract coordinates\n    y1, x1, height, width = box\n    y2 = y1 + height\n    x2 = x1 + width\n\n    im[y1:y1 + stroke, x1:x2] = rgb\n    im[y2:y2 + stroke, x1:x2] = rgb\n    im[y1:y2, x1:x1 + stroke] = rgb\n    im[y1:y2, x2:x2 + stroke] = rgb\n\n    return im","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3ef5bea28573c765d8a276d5f1d45c87c8d0611f"},"cell_type":"code","source":"df = pd.read_csv('../input/stage_1_train_labels.csv')\nparsed = parse_data(df)\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2cba62ce828c3925f4930f00c8156a5a370ea6ba"},"cell_type":"code","source":"det_class_path = '../input/stage_1_detailed_class_info.csv'\ndet_class_df = pd.read_csv(det_class_path)\ndet_class_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"96a60d1855f762f06e6d1cb0fea3f2bc669cd04b"},"cell_type":"code","source":"import cv2\nfrom IPython.display import display, Image\ndef cvshow(image, format='.png', rate=255 ):\n    decoded_bytes = cv2.imencode(format, image*rate)[1].tobytes()\n    display(Image(data=decoded_bytes))\n    return","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0fd256eaf1b2c71d96129d40f4a3ad7846462ba6"},"cell_type":"markdown","source":"# Visualizing No Lung Opacity / Not Normal "},{"metadata":{"trusted":true,"_uuid":"dc5383cdd25f85df09013e16b53187d0a213ed33"},"cell_type":"code","source":"j = 0\ndf = det_class_df[det_class_df['class']=='No Lung Opacity / Not Normal']\n# nImg = df.shape[0]/3  # takes long time to load !!!\nnImg = 400\nimg_ar = np.empty(0)\ndf = df.reset_index()\nwhile img_ar.shape[0]!=nImg:\n# for j in range(nImg):\n    ind = np.random.randint(df.shape[0])\n    patientId = df['patientId'][ind]\n    dcm_file = '../input/stage_1_train_images/%s.dcm' % patientId\n    dcm_data = pydicom.read_file(dcm_file)\n    img = np.expand_dims(dcm_data.pixel_array,axis=0)    \n    if j==0:\n        img_ar = img\n    elif (j%100==0):\n        print(j,'images loaded')\n    else:\n        img_ar = np.concatenate([img_ar,img],axis=0)\n    j += 1\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3c780929c955aa651635a5cac816f299a30ad11e"},"cell_type":"code","source":"def imgtile(imgs,tile_w):\n    assert imgs.shape[0]%tile_w==0,\"'imgs' cannot divide by 'th'.\"\n    r=imgs.reshape((-1,tile_w)+imgs.shape[1:])\n    return np.hstack(np.hstack(r))\n\n#usage\ntiled = imgtile(img_ar,20)\n# cvshow(tiled)\ntiled.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3ed919d542159a5100da180ea9754ca0d565e2c0"},"cell_type":"code","source":"cvshow(cv2.resize( tiled, (1024,1024), interpolation=cv2.INTER_LINEAR ))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1d8e1dff329a4685371f75c4d18e1e84fe79f298"},"cell_type":"markdown","source":"# Switching images using the slide bar"},{"metadata":{"trusted":true,"_uuid":"49fce826aa7684d296446eb87c8840152206d0d4"},"cell_type":"code","source":"from ipywidgets import interact,IntSlider\n@interact\ndef f(i=IntSlider(min=1,max=18,step=1,value=0)):\n    cvshow(imgtile(img_ar[i*20:(i+1)*20],5))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d02212f6e6142983cbf1f6a8443b963912ca98af"},"cell_type":"markdown","source":"# Displaying Some Normal Images"},{"metadata":{"trusted":true,"_uuid":"ded5d8ee6bd7f9b9803feb995cd7b1e6a3ac63cc"},"cell_type":"code","source":"j = 0\ndf = det_class_df[det_class_df['class']=='Normal']\n# nImg = df.shape[0]/3  # takes long time to load !!!\nnImg = 400\ndf = df.reset_index()\nimg_ar = np.empty(0)\nwhile img_ar.shape[0]!=nImg:\n# for j in range(nImg):\n    ind = np.random.randint(df.shape[0])\n    patientId = df['patientId'][ind]\n    dcm_file = '../input/stage_1_train_images/%s.dcm' % patientId\n    dcm_data = pydicom.read_file(dcm_file)\n    img = np.expand_dims(dcm_data.pixel_array,axis=0)    \n    if j==0:\n        img_ar = img\n    elif (j%100==0):\n        print(j,'images loaded')\n    else:\n        img_ar = np.concatenate([img_ar,img],axis=0)\n    j += 1\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6c78a0de985105a3ec1c39e4cab2913e73946cbb"},"cell_type":"code","source":"#usage\ntiled = imgtile(img_ar,20)\ntiled.shape\ncvshow(cv2.resize( tiled, (1024,1024), interpolation=cv2.INTER_LINEAR ))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3ed28c7e7e511465709fbcba8fa0ccc3df4044e1"},"cell_type":"markdown","source":"# Displaying Lung Opacity Images"},{"metadata":{"_kg_hide-input":false,"trusted":true,"_uuid":"0e1e50fbacbf6f470afe57b8489584d09d32ccdf"},"cell_type":"code","source":"j = 0\ndf = det_class_df[det_class_df['class']=='Lung Opacity']\n# nImg = df.shape[0]/3  # takes long time to load !!!\nnImg = 400\nimg_ar = np.empty(0)\ndf = df.reset_index()\nimg_ar = np.empty(0)\nwhile img_ar.shape[0]!=nImg:\n# for j in range(nImg):\n    ind = np.random.randint(df.shape[0])\n    patientId = df['patientId'][ind]\n    dcm_file = '../input/stage_1_train_images/%s.dcm' % patientId\n    dcm_data = pydicom.read_file(dcm_file)\n    img = dcm_data.pixel_array\n    data = parsed[patientId]\n    img = draw(data,img)\n    img = np.expand_dims(img,axis=0)    \n    if j==0:\n        img_ar = img\n    elif (j%100==0):\n        print(j,'images loaded')\n    else:\n        img_ar = np.concatenate([img_ar,img],axis=0)\n    j += 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f7170920ce9973f9e335d3ef4201252d85790522"},"cell_type":"code","source":"#usage\ntiled = imgtile(img_ar,20)\ntiled.shape\ncvshow(cv2.resize( tiled, (1024,1024) ))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}