{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Part 1.0: Feature Extraction for Image Processing","metadata":{}},{"cell_type":"code","source":"# Imports\nimport os\nimport glob\nimport cv2\nimport pydicom\nimport skimage\nimport numpy as np\nimport pandas as pd \nfrom tqdm import tqdm\nfrom os import listdir\nfrom tqdm import tqdm_notebook\nfrom os.path import isfile, join\nfrom skimage import feature, filters\n\n%matplotlib inline \nimport matplotlib\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom matplotlib.patches import Rectangle","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-31T12:02:59.293809Z","iopub.execute_input":"2021-07-31T12:02:59.294076Z","iopub.status.idle":"2021-07-31T12:02:59.303359Z","shell.execute_reply.started":"2021-07-31T12:02:59.294052Z","shell.execute_reply":"2021-07-31T12:02:59.302504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load in filepaths\ntrainImagesPath = \"../input/rsna-pneumonia-detection-challenge/stage_2_train_images\"\ntestImagesPath = \"../input/rsna-pneumonia-detection-challenge/stage_2_test_images\"\nlabelsPath = \"../input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv\"\nclassInfoPath = \"../input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv\"\n\nlabels = pd.read_csv(labelsPath) # Read Labels\ndetails = pd.read_csv(classInfoPath) # Read classInfo","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:03:01.507572Z","iopub.execute_input":"2021-07-31T12:03:01.507849Z","iopub.status.idle":"2021-07-31T12:03:01.603066Z","shell.execute_reply.started":"2021-07-31T12:03:01.507826Z","shell.execute_reply":"2021-07-31T12:03:01.60223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Part 1.1: Define Functions for Extracing Image Features","metadata":{}},{"cell_type":"code","source":"\"\"\"\n@Description: This function will take an image and extract all of the above features\n@Input: Dicom image pixel array\n@Output: Returns the number of non-zero elements in the pixel array\n\"\"\"\ndef getImageArea(image):\n    return np.count_nonzero(image)\n\n\"\"\"\n@Description: This function gives us the equivalenet diameter from the area\n@Input: Takes the area of the given image\n@Output: Returns the equivalent image diameter\n\"\"\"\ndef getImageEquivalentDiameter(area):\n    return (np.sqrt((area*4) / np.pi))\n\n\n\"\"\"\n@Description: This function gets us the perimeter of an image\n@Input: Dicom image pixel array of edges of image\n@Output: Returns the number of non-zero elements in the perimeter pixel array\n\"\"\"\ndef getImagePerimeter(image):\n    return np.count_nonzero(image)\n\n\"\"\"\n@Description: This function gives us the irregularity in a given image\n@Input: Perimeter and Area of the image\n@Output: Returns the irregularity index\n\"\"\"\ndef getImageIrregularity(perimeter, area):\n    return ((area*4*np.pi) / (perimeter**2))\n\n\"\"\"\n@Description: This function gives us an images hu moments\n@Input: Takes in the contours of the given image\n@Output: Returns the various hu moments besides the 3rd and 7th\n\"\"\"\ndef getImageHuMoments(contour):\n    hu = cv2.HuMoments(cv2.moments(contour)).ravel().tolist() # Get the hu's\n    hu.pop(-1) # Remove last hu\n    hu.pop(2) # Remove third hu\n    return ([-np.sign(h)*np.log10(np.abs(h)) for h in hu]) # Return the log of the hu's","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:03:03.761824Z","iopub.execute_input":"2021-07-31T12:03:03.762281Z","iopub.status.idle":"2021-07-31T12:03:03.771657Z","shell.execute_reply.started":"2021-07-31T12:03:03.76225Z","shell.execute_reply":"2021-07-31T12:03:03.770147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Part 1.2: Define Extract Features Function","metadata":{}},{"cell_type":"code","source":"\"\"\"\n@Description: This function will take an image and extract all of the above features\n\n@Input: An image that has been read with pydicom\n\n@Output: Returns the extract features\n\n@Credit: This extraction function was borrowed from @suryathiru (https://www.kaggle.com/suryathiru/1-tradition-image-processing-feature-extraction/)\n\"\"\"\ndef extractFeatures(image):\n    \n    mean = image.mean() # Mean\n    stdDev = image.std() # Standard deviation\n    equalized = cv2.equalizeHist(image) # Hist Equalisation\n    \n    # Sharpening\n    hpf_kernel = np.full((3, 3), -1)\n    hpf_kernel[1,1] = 9\n    \n    sharpened = cv2.filter2D(equalized, -1, hpf_kernel)\n    \n    ret, binarized = cv2.threshold(cv2.GaussianBlur(sharpened,(7,7),0), 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU) # thresholding\n    edges = skimage.filters.sobel(binarized) # Edge detection for binarized image\n    \n    # Moments from contours\n    contours, hier = cv2.findContours((edges * 255).astype('uint8'),cv2.RETR_EXTERNAL,cv2.CHAIN_APPROX_SIMPLE)\n    select_contour = sorted(contours, key=lambda x: x.shape[0], reverse=True)[0]\n    \n    # Return extracted features\n    return (mean, stdDev, getImageArea(binarized), getImagePerimeter(edges), \n            getImageIrregularity(getImageArea(binarized), getImagePerimeter(edges)),\n            getImageEquivalentDiameter(getImageArea(binarized)), getImageHuMoments(select_contour))","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:06:14.408833Z","iopub.execute_input":"2021-07-31T12:06:14.40911Z","iopub.status.idle":"2021-07-31T12:06:14.416382Z","shell.execute_reply.started":"2021-07-31T12:06:14.409085Z","shell.execute_reply":"2021-07-31T12:06:14.415499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fileNames = [f for f in listdir(testImagesPath) if isfile(join(testImagesPath, f))] # Get test image filenames","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:03:10.582115Z","iopub.execute_input":"2021-07-31T12:03:10.582514Z","iopub.status.idle":"2021-07-31T12:03:17.85436Z","shell.execute_reply.started":"2021-07-31T12:03:10.58249Z","shell.execute_reply":"2021-07-31T12:03:17.853556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Part 1.3: Get Testing Data","metadata":{}},{"cell_type":"code","source":"\"\"\"\n@Description: This function goes through the dicom image information and returns 1 or 0\n              depending on whether the image contains Pneumonia or not\n\n@Inputs: A dataframe containing the metadata\n\n@Output: Returns our test y\n\"\"\"\ndef createY(df):\n    \n    y = (df['SeriesDescription'] == 'view: PA')\n    Y = np.zeros(len(y)) # Initialise Y\n    \n    for i in range(len(y)):\n        if(y[i] == True):\n            Y[i] = 1\n    \n    return Y","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:03:19.757859Z","iopub.execute_input":"2021-07-31T12:03:19.75815Z","iopub.status.idle":"2021-07-31T12:03:19.763503Z","shell.execute_reply.started":"2021-07-31T12:03:19.758113Z","shell.execute_reply":"2021-07-31T12:03:19.762577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reads each image path and puts it into a list\ndef readDicomData(data):\n    \n    res = []\n    \n    for filePath in tqdm(data): # Loop over data\n        f = pydicom.read_file(filePath, stop_before_pixels=True) # Read image and stop before pixels to save memory\n        res.append(f)\n    \n    return res","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:03:22.028261Z","iopub.execute_input":"2021-07-31T12:03:22.028534Z","iopub.status.idle":"2021-07-31T12:03:22.033185Z","shell.execute_reply.started":"2021-07-31T12:03:22.028512Z","shell.execute_reply":"2021-07-31T12:03:22.032257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n@Description: This function parses the medical images meta-data contained\n\n@Inputs: Takes in the dicom image after it has been read\n\n@Output: Returns the unpacked data and the group elements keywords\n\"\"\"\ndef parseMetadata(dcm):\n    \n    unpackedData = {}\n    groupElemToKeywords = {}\n    \n    for d in dcm: # Iterate here to force conversion from lazy RawDataElement to DataElement\n        pass\n    \n    # Un-pack Data\n    for tag, elem in dcm.items():\n        tagGroup = tag.group\n        tagElem = tag.elem\n        keyword = elem.keyword\n        groupElemToKeywords[(tagGroup, tagElem)] = keyword\n        value = elem.value\n        unpackedData[keyword] = value\n        \n    return unpackedData, groupElemToKeywords\n","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:03:23.949152Z","iopub.execute_input":"2021-07-31T12:03:23.949482Z","iopub.status.idle":"2021-07-31T12:03:23.955378Z","shell.execute_reply.started":"2021-07-31T12:03:23.949458Z","shell.execute_reply":"2021-07-31T12:03:23.954453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testFilepaths = glob.glob(f\"{testImagesPath}/*.dcm\") # Get test data file paths\ntestImages = readDicomData(testFilepaths) # Read test file paths\n\ntestMetaDicts, testKeyword = zip(*[parseMetadata(x) for x in tqdm(testImages)])\ntest_df = pd.DataFrame.from_dict(data = testMetaDicts) # Convert to dataframe\n\ntest_df['dataset'] = 'test' # Call it test\ntest_Y = createY(test_df) # Call the create Y function to get our test Y","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:03:27.620709Z","iopub.execute_input":"2021-07-31T12:03:27.62097Z","iopub.status.idle":"2021-07-31T12:03:57.13804Z","shell.execute_reply.started":"2021-07-31T12:03:27.620947Z","shell.execute_reply":"2021-07-31T12:03:57.137076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get testing images\nfeaturesTest = []\n\nfor fN in tqdm(fileNames): # Loop over file names\n\n    path = f\"{testImagesPath}/{fN}\" # Get Path\n    \n    image = pydicom.read_file(path).pixel_array # Read file and get pixel array\n    \n    featuresTest.append(extractFeatures(image)) # Extract features & append to array","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:06:18.417211Z","iopub.execute_input":"2021-07-31T12:06:18.417495Z","iopub.status.idle":"2021-07-31T12:09:33.663462Z","shell.execute_reply.started":"2021-07-31T12:06:18.417462Z","shell.execute_reply":"2021-07-31T12:09:33.66203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get training images\nfeaturesTrain = []\n\n# Loop over patient IDs\nfor patientId in tqdm(labels['patientId']):\n\n    path = f\"{trainImagesPath}/{patientId}.dcm\" # Get path\n    \n    image = pydicom.read_file(path).pixel_array # Read file and get pixel array\n    \n    featuresTrain.append(extractFeatures(image)) # Extract features & append to array","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:09:33.664528Z","iopub.execute_input":"2021-07-31T12:09:33.664711Z","iopub.status.idle":"2021-07-31T12:45:53.849751Z","shell.execute_reply.started":"2021-07-31T12:09:33.66469Z","shell.execute_reply":"2021-07-31T12:45:53.849267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testDf = pd.DataFrame(test_Y, columns = ['Target'])\ntestDf['features'] = featuresTest\n\nlabels['features'] = featuresTrain # Set features","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:46:04.697078Z","iopub.execute_input":"2021-07-31T12:46:04.697348Z","iopub.status.idle":"2021-07-31T12:46:04.734957Z","shell.execute_reply.started":"2021-07-31T12:46:04.697325Z","shell.execute_reply":"2021-07-31T12:46:04.733961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testDf.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:46:20.296334Z","iopub.execute_input":"2021-07-31T12:46:20.296792Z","iopub.status.idle":"2021-07-31T12:46:20.309784Z","shell.execute_reply.started":"2021-07-31T12:46:20.296763Z","shell.execute_reply":"2021-07-31T12:46:20.308771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-07-31T12:46:22.160493Z","iopub.execute_input":"2021-07-31T12:46:22.160835Z","iopub.status.idle":"2021-07-31T12:46:22.173884Z","shell.execute_reply.started":"2021-07-31T12:46:22.160812Z","shell.execute_reply":"2021-07-31T12:46:22.173093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Part 1.4: Download Data to CSV","metadata":{}},{"cell_type":"code","source":"# Download test data as csv\ntestDf.to_csv('testImageFeatures.csv')","metadata":{"execution":{"iopub.status.busy":"2021-07-29T07:04:22.718041Z","iopub.execute_input":"2021-07-29T07:04:22.718484Z","iopub.status.idle":"2021-07-29T07:04:22.785501Z","shell.execute_reply.started":"2021-07-29T07:04:22.71845Z","shell.execute_reply":"2021-07-29T07:04:22.784659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Download train data as csv\ndf = pd.DataFrame(labels)\ndf.to_csv('dicomImageFeatures.csv')","metadata":{"execution":{"iopub.status.busy":"2021-07-25T09:47:10.057058Z","iopub.execute_input":"2021-07-25T09:47:10.057445Z","iopub.status.idle":"2021-07-25T09:47:10.544504Z","shell.execute_reply.started":"2021-07-25T09:47:10.057396Z","shell.execute_reply":"2021-07-25T09:47:10.543517Z"},"trusted":true},"execution_count":null,"outputs":[]}]}