{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Part 1 - Tradition Image processing for classification: feature extraction\n使用传统图像处理方法构造更多特征","metadata":{}},{"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nimport matplotlib\nimport matplotlib.pyplot as plt\n# 进度条库\nfrom tqdm import tqdm_notebook\nfrom matplotlib.patches import Rectangle\nimport seaborn as sns\n# 医学图像处理库\nimport pydicom as dcm\n%matplotlib inline \nplt.set_cmap(plt.cm.bone)\nIS_LOCAL = True\nimport os\nimport cv2\n# sk的图像处理库\nimport skimage\nfrom skimage import feature, filters\nfrom tqdm import tqdm\n\nPATH=\"../input/rsna-pneumonia-detection-challenge\"\n\nprint(os.listdir(PATH))","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2021-06-23T13:40:38.904276Z","iopub.execute_input":"2021-06-23T13:40:38.904655Z","iopub.status.idle":"2021-06-23T13:40:40.847469Z","shell.execute_reply.started":"2021-06-23T13:40:38.904612Z","shell.execute_reply":"2021-06-23T13:40:40.846528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.rcParams['font.sans-serif'] = ['KaiTi'] # 指定默认字体\nplt.rcParams['axes.unicode_minus'] = False ","metadata":{"execution":{"iopub.status.busy":"2021-06-23T14:08:16.422898Z","iopub.execute_input":"2021-06-23T14:08:16.423316Z","iopub.status.idle":"2021-06-23T14:08:16.428017Z","shell.execute_reply.started":"2021-06-23T14:08:16.423278Z","shell.execute_reply":"2021-06-23T14:08:16.427231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 将训练集的特征-标签和图像的具体描述连接\nclass_info_df = pd.read_csv(PATH+'/stage_2_detailed_class_info.csv')\ntrain_labels_df = pd.read_csv(PATH+'/stage_2_train_labels.csv')\ntrain_class_df = train_labels_df.merge(class_info_df, left_on='patientId', right_on='patientId', how='inner')","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2021-06-23T13:41:33.363512Z","iopub.execute_input":"2021-06-23T13:41:33.363872Z","iopub.status.idle":"2021-06-23T13:41:33.541490Z","shell.execute_reply.started":"2021-06-23T13:41:33.363841Z","shell.execute_reply":"2021-06-23T13:41:33.540450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_class_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T13:41:35.731093Z","iopub.execute_input":"2021-06-23T13:41:35.731445Z","iopub.status.idle":"2021-06-23T13:41:35.753505Z","shell.execute_reply.started":"2021-06-23T13:41:35.731416Z","shell.execute_reply":"2021-06-23T13:41:35.752050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 看下图像的类别\ntrain_class_df['class'].unique()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T13:42:49.380910Z","iopub.execute_input":"2021-06-23T13:42:49.381290Z","iopub.status.idle":"2021-06-23T13:42:49.389821Z","shell.execute_reply.started":"2021-06-23T13:42:49.381254Z","shell.execute_reply":"2021-06-23T13:42:49.388823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 定义一些常用函数","metadata":{}},{"cell_type":"code","source":"\ndef load_images(data):\n    \"\"\"\n    加载图片，将data集中包含的id的图片全加载，返回一个列表\n    \"\"\"\n    imgs = []\n    for path in data['patientId']:\n        patientImage = path + '.dcm'\n        imagePath = os.path.join(PATH,\"stage_2_train_images/\", patientImage)\n        img = dcm.read_file(imagePath).pixel_array\n        imgs.append(img)\n    return imgs\n\ndef imshow_gray(img):\n    \"\"\"\n    用灰度显示单张图片\n    \"\"\"\n    plt.figure(figsize=(12,7))\n    return plt.imshow(img, cmap='gray')\n    \ndef imshow_with_labels(img, patient_id):\n    \"\"\"\n    在肺炎图片上标记出边界框，并显示图片\n    \"\"\"\n    # 从训练集中取出这id所对应的一行\n    rows = train_labels_df[train_labels_df['patientId'] == patient_id]\n    # 分别得到该图片的边界框特征\n    for row in rows.itertuples():        \n        x, y, w, h = row.x, row.y, row.width, row.height\n        x, y, w, h = map(int, [x,y,w,h])\n        cv2.rectangle(img, (x,y), (x+w,y+h), 255, 2)\n    plt.figure(figsize=(12,7))\n    return plt.imshow(img, cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2021-06-23T13:51:58.235455Z","iopub.execute_input":"2021-06-23T13:51:58.236058Z","iopub.status.idle":"2021-06-23T13:51:58.250609Z","shell.execute_reply.started":"2021-06-23T13:51:58.236026Z","shell.execute_reply":"2021-06-23T13:51:58.249748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 采样几张图片来进行图像处理","metadata":{}},{"cell_type":"code","source":"# 随机选取三种肺炎图像\ntest_df = train_class_df[train_class_df['Target']==1].sample(4)\nbox = test_df.loc[test_df.index, ['x', 'y', 'width', 'height']]\n# 读取随机采样的前三张肺炎图像\ntest = load_images(test_df[0:3])","metadata":{"execution":{"iopub.status.busy":"2021-06-23T13:53:18.837290Z","iopub.execute_input":"2021-06-23T13:53:18.837692Z","iopub.status.idle":"2021-06-23T13:53:18.965174Z","shell.execute_reply.started":"2021-06-23T13:53:18.837660Z","shell.execute_reply":"2021-06-23T13:53:18.964130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[0:3]","metadata":{"execution":{"iopub.status.busy":"2021-06-23T13:55:38.232556Z","iopub.execute_input":"2021-06-23T13:55:38.233121Z","iopub.status.idle":"2021-06-23T13:55:38.247201Z","shell.execute_reply.started":"2021-06-23T13:55:38.233072Z","shell.execute_reply":"2021-06-23T13:55:38.246418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 第一张的情况\nidx = 0\nimg = test[idx]\nimshow_with_labels(img.copy(), test_df.iloc[idx,0])","metadata":{"execution":{"iopub.status.busy":"2021-06-23T13:59:13.086042Z","iopub.execute_input":"2021-06-23T13:59:13.086582Z","iopub.status.idle":"2021-06-23T13:59:13.377494Z","shell.execute_reply.started":"2021-06-23T13:59:13.086532Z","shell.execute_reply":"2021-06-23T13:59:13.376398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 第二张的情况\nidx = 1\nimg = test[idx]\nimshow_with_labels(img.copy(), test_df.iloc[idx,0])","metadata":{"execution":{"iopub.status.busy":"2021-06-23T14:00:14.620691Z","iopub.execute_input":"2021-06-23T14:00:14.621051Z","iopub.status.idle":"2021-06-23T14:00:14.907190Z","shell.execute_reply.started":"2021-06-23T14:00:14.621020Z","shell.execute_reply":"2021-06-23T14:00:14.906074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 图像增强","metadata":{}},{"cell_type":"markdown","source":"### 1.直方图均衡化","metadata":{}},{"cell_type":"code","source":"equ = cv2.equalizeHist(test[idx])\nax = imshow_gray(equ)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T14:00:21.727209Z","iopub.execute_input":"2021-06-23T14:00:21.727579Z","iopub.status.idle":"2021-06-23T14:00:22.021235Z","shell.execute_reply.started":"2021-06-23T14:00:21.727547Z","shell.execute_reply":"2021-06-23T14:00:22.020423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imshow_with_labels(equ.copy(), test_df.iloc[idx,0])","metadata":{"execution":{"iopub.status.busy":"2021-06-23T14:01:17.857617Z","iopub.execute_input":"2021-06-23T14:01:17.857975Z","iopub.status.idle":"2021-06-23T14:01:18.159853Z","shell.execute_reply.started":"2021-06-23T14:01:17.857946Z","shell.execute_reply":"2021-06-23T14:01:18.158956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"均衡呈现肺部的对比度，图片明显变亮了一些，并进一步强调不透明度的存在","metadata":{}},{"cell_type":"markdown","source":"### 2.锐化","metadata":{}},{"cell_type":"code","source":"# 使用3*3的高通滤波器\nhpf_kernel = np.full((3, 3), -1)\nhpf_kernel[1,1] = 9\nim_hp = cv2.filter2D(equ, -1, hpf_kernel)\n\n# 使用虚光蒙版\nim_us = skimage.filters.unsharp_mask(equ)\n\n# 对比两种锐化的结果\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(16,9))\nax1.imshow(im_hp, cmap='gray')\nax1.set_title('高通 filter')\nax2.imshow(im_us, cmap='gray')\nax2.set_title('虚光 filter')\nfig.suptitle('锐化')","metadata":{"execution":{"iopub.status.busy":"2021-06-23T14:08:22.247069Z","iopub.execute_input":"2021-06-23T14:08:22.247549Z","iopub.status.idle":"2021-06-23T14:08:22.887686Z","shell.execute_reply.started":"2021-06-23T14:08:22.247518Z","shell.execute_reply":"2021-06-23T14:08:22.886905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"高通滤波器的锐化效果更明显","metadata":{}},{"cell_type":"markdown","source":"### 3.阈值化\n","metadata":{}},{"cell_type":"code","source":"# otsu 阈值化\nret, otsu = cv2.threshold(cv2.GaussianBlur(im_hp,(7,7),0),0,255,cv2.THRESH_BINARY+cv2.THRESH_OTSU)\n# 全局阈值化\nlocal = im_hp > skimage.filters.threshold_local(im_hp, 5)\n# 均值阈值化\nmean = im_hp > skimage.filters.threshold_mean(im_hp)\n\nfig, (ax1, ax2, ax3) = plt.subplots(1, 3, figsize=(16,9))\nax1.imshow(otsu, cmap='gray')\nax1.set_title('Otsu thresholding')\nax2.imshow(local, cmap='gray')\nax2.set_title('Local thresholding')\nfig.suptitle('Image sharpening')\nax3.imshow(mean, cmap='gray')\nax3.set_title('Mean thresholding')\nfig.suptitle('Image thresholding')","metadata":{"execution":{"iopub.status.busy":"2021-06-23T14:10:01.991561Z","iopub.execute_input":"2021-06-23T14:10:01.991961Z","iopub.status.idle":"2021-06-23T14:10:02.747922Z","shell.execute_reply.started":"2021-06-23T14:10:01.991928Z","shell.execute_reply":"2021-06-23T14:10:02.746758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Otsu的表现最好","metadata":{}},{"cell_type":"markdown","source":"### 边缘检测","metadata":{}},{"cell_type":"code","source":"# sobel算子\nsobel = filters.sobel(otsu)\n# canny算子\ncanny = feature.canny(otsu/255)\n\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(16,9))\nax1.imshow(canny, cmap='gray')\nax1.set_title('Canny edge detection')\nax2.imshow(sobel, cmap='gray')\nax2.set_title('Sobel operator')\nfig.suptitle('Edge detection')","metadata":{"execution":{"iopub.status.busy":"2021-06-23T14:12:55.408619Z","iopub.execute_input":"2021-06-23T14:12:55.408990Z","iopub.status.idle":"2021-06-23T14:12:56.265840Z","shell.execute_reply.started":"2021-06-23T14:12:55.408961Z","shell.execute_reply":"2021-06-23T14:12:56.264638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"sobel分割出的边界比canny的更清晰","metadata":{}},{"cell_type":"markdown","source":"### 肺部分割\n\n查找并画出肺部的轮廓","metadata":{}},{"cell_type":"code","source":"contours, hier = cv2.findContours((sobel * 255).astype('uint8'),cv2.RETR_EXTERNAL,cv2.CHAIN_APPROX_SIMPLE)\nprint('Contours found ', len(contours))\n\nsrt_contours = sorted(contours, key=lambda x: x.shape[0], reverse=True)\nselect_contour = srt_contours[0]  # probably not the best assumption\n\ntest = img.copy()\nimg_contour = cv2.drawContours(test, [select_contour], 0, (255,0,0), thickness=3)\n\nimshow_gray(img_contour)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T15:18:32.523473Z","iopub.execute_input":"2021-06-23T15:18:32.523859Z","iopub.status.idle":"2021-06-23T15:18:32.818438Z","shell.execute_reply.started":"2021-06-23T15:18:32.523826Z","shell.execute_reply":"2021-06-23T15:18:32.817353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"识别肺段后，我们可以提取该段的矩中心。由于所有 X 射线图像都来自同一维度，因此这可能是预测的有效特征","metadata":{}},{"cell_type":"code","source":"M = cv2.moments(select_contour)\ncx = int(M['m10'] / M['m00'])\ncy = int(M[\"m01\"] / M[\"m00\"])\n\ntest = img.copy()\ncv2.circle(test, (cx, cy), 7, (255, 255, 255), -1)\nimshow_gray(test)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T14:19:45.834706Z","iopub.execute_input":"2021-06-23T14:19:45.835105Z","iopub.status.idle":"2021-06-23T14:19:46.117907Z","shell.execute_reply.started":"2021-06-23T14:19:45.835074Z","shell.execute_reply":"2021-06-23T14:19:46.117124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"但是有些图像的主体略微旋转，并且大小不同，因此我们希望矩中心对旋转和缩放保持不变。所以我们选择hu矩。我们还将记录时刻以便于比较和删除第 3 个时刻，因为它取决于其他时刻和第7个时刻，因为它区分镜像并且数据集中没有翻转图像","metadata":{}},{"cell_type":"markdown","source":"But there are images were the subject is slightly rotated, and differently sized so we want center of moment to be invariant to rotation and scale. So we pick **Hu moments**. We will also log the moments to make it easy to compare and drop the 3rd moment as it depends on the other moments and 7th moment as it distinguishes mirror images and there are no flipped images in the dataset","metadata":{}},{"cell_type":"code","source":"def get_hu_moments(contour):\n    M = cv2.moments(select_contour)\n    hu = cv2.HuMoments(M).ravel().tolist()\n    del hu[2]\n    del hu[-1]\n    log_hu = [-np.sign(a)*np.log10(np.abs(a)) for a in hu]\n    return log_hu\n\nget_hu_moments(select_contour)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T14:21:28.253573Z","iopub.execute_input":"2021-06-23T14:21:28.253968Z","iopub.status.idle":"2021-06-23T14:21:28.265891Z","shell.execute_reply.started":"2021-06-23T14:21:28.253936Z","shell.execute_reply":"2021-06-23T14:21:28.264760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 特征\n1. 不透明区域\n2. 肺的周长\n3. 不规则程度\n4. 直径\n5. 原始图像的均值，标准差\n6. hu 矩\n","metadata":{}},{"cell_type":"code","source":"def area(img):\n    # 二值图像作为输入\n    return np.count_nonzero(img)\n\ndef perimeter(img):\n    # 边缘图作为输入\n    return np.count_nonzero(img)\n\ndef irregularity(area, perimeter):\n    # area and perimeter of the image as input, also called compactness\n    I = (4 * np.pi * area) / (perimeter ** 2)\n    return I\n\ndef equiv_diam(area):\n    # area of image as input\n    ed = np.sqrt((4 * area) / np.pi)\n    return ed\n\ndef get_hu_moments(contour):\n    # hu moments except 3rd and 7th (5 values)\n    M = cv2.moments(contour)\n    hu = cv2.HuMoments(M).ravel().tolist()\n    del hu[2]\n    del hu[-1]\n    log_hu = [-np.sign(a)*np.log10(np.abs(a)) for a in hu]\n    return log_hu","metadata":{"execution":{"iopub.status.busy":"2021-06-23T14:27:50.665416Z","iopub.execute_input":"2021-06-23T14:27:50.665869Z","iopub.status.idle":"2021-06-23T14:27:50.676482Z","shell.execute_reply.started":"2021-06-23T14:27:50.665833Z","shell.execute_reply":"2021-06-23T14:27:50.675164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 图像预处理和特征提取","metadata":{}},{"cell_type":"code","source":"def extract_features(img):\n    # 先求原始图像的统计值\n    mean = img.mean()\n    std_dev = img.std()\n    \n    # 直方图均衡化\n    equalized = cv2.equalizeHist(img)\n    \n    # 高通锐化\n    hpf_kernel = np.full((3, 3), -1)\n    hpf_kernel[1,1] = 9\n    sharpened = cv2.filter2D(equalized, -1, hpf_kernel)\n    \n    # 阈值化\n    ret, binarized = cv2.threshold(cv2.GaussianBlur(sharpened,(7,7),0),0,255,cv2.THRESH_BINARY+cv2.THRESH_OTSU)\n    \n    # 边缘检测\n    edges = skimage.filters.sobel(binarized)\n    \n    # 矩计算\n    contours, hier = cv2.findContours((edges * 255).astype('uint8'),cv2.RETR_EXTERNAL,cv2.CHAIN_APPROX_SIMPLE)\n    select_contour = sorted(contours, key=lambda x: x.shape[0], reverse=True)[0]\n    \n    # 特征提取\n    # 不透明区域面积\n    ar = area(binarized)\n    # 肺部边长\n    per = perimeter(edges)\n    # 不规则程度\n    irreg = irregularity(ar, per)\n    # 直径\n    eq_diam = equiv_diam(ar)\n    # hu矩\n    hu = get_hu_moments(select_contour)\n    # 最后一幅图被提取为(6+5)个特征,其中hu矩有5个特征\n    return (mean, std_dev, ar, per, irreg, eq_diam, *hu)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T14:32:23.617127Z","iopub.execute_input":"2021-06-23T14:32:23.617522Z","iopub.status.idle":"2021-06-23T14:32:23.629880Z","shell.execute_reply.started":"2021-06-23T14:32:23.617492Z","shell.execute_reply":"2021-06-23T14:32:23.629033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test the function\nextract_features(img)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T14:31:46.102399Z","iopub.execute_input":"2021-06-23T14:31:46.102805Z","iopub.status.idle":"2021-06-23T14:31:46.172418Z","shell.execute_reply.started":"2021-06-23T14:31:46.102759Z","shell.execute_reply":"2021-06-23T14:31:46.171433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 保存特征","metadata":{}},{"cell_type":"markdown","source":"## 加载数据\n只选择正常和肺炎图像进行模型构建","metadata":{}},{"cell_type":"code","source":"# 肺炎图像\npneumonia_ids = train_labels_df[train_labels_df['Target'] == 1]['patientId'].unique()\npneumonia_labels = [1] * len(pneumonia_ids)\n\n# 非肺炎患者的正常图像\nnormal_ids = class_info_df[class_info_df['class'] == 'Normal']['patientId'].unique()\nnormal_labels = [0] * len(normal_ids)\n\ndata = dict()\ndata['patientId'] = np.concatenate((pneumonia_ids, normal_ids))\ndata['target'] = np.concatenate((pneumonia_labels, normal_labels))\n\nprint(f'Pneumonia images: {len(pneumonia_ids)}\\nNormal images: {len(normal_ids)}')","metadata":{"execution":{"iopub.status.busy":"2021-06-23T14:34:30.456189Z","iopub.execute_input":"2021-06-23T14:34:30.456873Z","iopub.status.idle":"2021-06-23T14:34:30.482354Z","shell.execute_reply.started":"2021-06-23T14:34:30.456821Z","shell.execute_reply":"2021-06-23T14:34:30.481281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 生成特征\n对于每个 ID，从图像生成特征并将其存储在数据集中","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2021-06-23T14:35:07.682783Z","iopub.execute_input":"2021-06-23T14:35:07.683166Z","iopub.status.idle":"2021-06-23T14:35:07.687246Z","shell.execute_reply.started":"2021-06-23T14:35:07.683130Z","shell.execute_reply":"2021-06-23T14:35:07.686229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = []\n\nfor path in tqdm(data['patientId']):\n    # 加载数据集中的图像\n    patientImage = path + '.dcm'\n    imagePath = os.path.join(PATH,\"stage_2_train_images/\", patientImage)\n    img = dcm.read_file(imagePath).pixel_array\n    # 对该图像进行特征提取\n    feats = extract_features(img)\n    # 一行特征作为一个列表值，加到feats中\n    features.append(feats)\n\ndata['features'] = features\n# 最终data三个值（id,target,features）","metadata":{"execution":{"iopub.status.busy":"2021-06-23T14:35:09.994915Z","iopub.execute_input":"2021-06-23T14:35:09.995618Z","iopub.status.idle":"2021-06-23T14:41:32.338590Z","shell.execute_reply.started":"2021-06-23T14:35:09.995569Z","shell.execute_reply":"2021-06-23T14:41:32.336818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame(data)\ndf.to_csv('img_features.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"生成特征后，可以使用机器学习模型加载和训练它们以执行分类。","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}