{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfrom glob import glob\nfrom tqdm.notebook import tqdm\n!conda install gdcm -c conda-forge -y\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport matplotlib.pyplot as plt\nimport matplotlib\nimport seaborn as sns\nfrom skimage import exposure\nimport cv2\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.720041,"end_time":"2021-01-01T01:51:44.426457","exception":false,"start_time":"2021-01-01T01:51:43.706416","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-20T03:36:00.259046Z","iopub.execute_input":"2021-05-20T03:36:00.259505Z","iopub.status.idle":"2021-05-20T03:36:39.528151Z","shell.execute_reply.started":"2021-05-20T03:36:00.259464Z","shell.execute_reply":"2021-05-20T03:36:39.526545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset_dir = '../input/siim-covid19-detection'","metadata":{"papermill":{"duration":0.026418,"end_time":"2021-01-01T01:51:44.508062","exception":false,"start_time":"2021-01-01T01:51:44.481644","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-20T03:36:39.531369Z","iopub.execute_input":"2021-05-20T03:36:39.531737Z","iopub.status.idle":"2021-05-20T03:36:39.539089Z","shell.execute_reply.started":"2021-05-20T03:36:39.531696Z","shell.execute_reply":"2021-05-20T03:36:39.538073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dicom2array(path, voi_lut=True, fix_monochrome=True):\n    dicom = pydicom.read_file(path)\n    # VOI LUT (if available by DICOM device) is used to\n    # transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\n        \n    \ndef plot_img(img, size=(7, 7), is_rgb=True, title=\"\", cmap='gray'):\n    plt.figure(figsize=size)\n    plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n\n\ndef plot_imgs(imgs, cols=4, size=7, is_rgb=True, title=\"\", cmap='gray', img_size=(500,500)):\n    rows = len(imgs)//cols + 1\n    fig = plt.figure(figsize=(cols*size, rows*size))\n    for i, img in enumerate(imgs):\n        if img_size is not None:\n            img = cv2.resize(img, img_size)\n        fig.add_subplot(rows, cols, i+1)\n        plt.imshow(img, cmap=cmap)\n    plt.suptitle(title)\n    plt.show()\n","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","papermill":{"duration":0.040531,"end_time":"2021-01-01T01:51:44.567233","exception":false,"start_time":"2021-01-01T01:51:44.526702","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-20T03:36:39.541829Z","iopub.execute_input":"2021-05-20T03:36:39.542303Z","iopub.status.idle":"2021-05-20T03:36:39.557377Z","shell.execute_reply.started":"2021-05-20T03:36:39.542266Z","shell.execute_reply":"2021-05-20T03:36:39.556394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# simple plot","metadata":{}},{"cell_type":"code","source":"dicom_paths = glob(f'{dataset_dir}/train/*/*/*.dcm')\nimgs = [dicom2array(path) for path in dicom_paths[:4]]\nplot_imgs(imgs)","metadata":{"papermill":{"duration":9.946147,"end_time":"2021-01-01T01:51:54.532001","exception":false,"start_time":"2021-01-01T01:51:44.585854","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-20T03:36:39.559090Z","iopub.execute_input":"2021-05-20T03:36:39.559488Z","iopub.status.idle":"2021-05-20T03:36:46.761331Z","shell.execute_reply.started":"2021-05-20T03:36:39.559453Z","shell.execute_reply":"2021-05-20T03:36:46.760107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Maybe, you can try some preprocess like equalize histogram. You can see the difference between before and after","metadata":{"papermill":{"duration":0.029261,"end_time":"2021-01-01T01:51:54.592507","exception":false,"start_time":"2021-01-01T01:51:54.563246","status":"completed"},"tags":[]}},{"cell_type":"code","source":"imgs = [exposure.equalize_hist(img) for img in imgs]\nplot_imgs(imgs)","metadata":{"papermill":{"duration":2.163571,"end_time":"2021-01-01T01:51:56.7856","exception":false,"start_time":"2021-01-01T01:51:54.622029","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-20T03:36:46.762988Z","iopub.execute_input":"2021-05-20T03:36:46.763480Z","iopub.status.idle":"2021-05-20T03:36:48.935789Z","shell.execute_reply.started":"2021-05-20T03:36:46.763429Z","shell.execute_reply":"2021-05-20T03:36:48.934590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# load & preprocess","metadata":{}},{"cell_type":"code","source":"from bokeh.plotting import figure as bokeh_figure\nfrom bokeh.io import output_notebook, show, output_file\nfrom bokeh.models import ColumnDataSource, HoverTool, Panel\nfrom bokeh.models.widgets import Tabs\nimport pandas as pd\nfrom PIL import Image\nfrom sklearn import preprocessing\nimport random\nfrom random import randint","metadata":{"papermill":{"duration":1.37697,"end_time":"2021-01-01T01:51:58.374581","exception":false,"start_time":"2021-01-01T01:51:56.997611","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-20T03:36:48.937360Z","iopub.execute_input":"2021-05-20T03:36:48.937733Z","iopub.status.idle":"2021-05-20T03:36:48.943525Z","shell.execute_reply.started":"2021-05-20T03:36:48.937698Z","shell.execute_reply":"2021-05-20T03:36:48.942133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(f'{dataset_dir}/train_image_level.csv')\ntrain_study = pd.read_csv(f'{dataset_dir}/train_study_level.csv')\ntrain.head()","metadata":{"papermill":{"duration":1.675464,"end_time":"2021-01-01T01:52:00.092785","exception":false,"start_time":"2021-01-01T01:51:58.417321","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-20T03:36:48.945412Z","iopub.execute_input":"2021-05-20T03:36:48.945805Z","iopub.status.idle":"2021-05-20T03:36:49.011917Z","shell.execute_reply.started":"2021-05-20T03:36:48.945771Z","shell.execute_reply":"2021-05-20T03:36:49.010815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_study.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:36:49.014714Z","iopub.execute_input":"2021-05-20T03:36:49.015064Z","iopub.status.idle":"2021-05-20T03:36:49.027240Z","shell.execute_reply.started":"2021-05-20T03:36:49.015022Z","shell.execute_reply":"2021-05-20T03:36:49.026147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# merge study csv\ntrain_study['StudyInstanceUID'] = train_study['id'].apply(lambda x: x.replace('_study', ''))\ndel train_study['id']\ntrain = train.merge(train_study, on='StudyInstanceUID')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:36:49.030000Z","iopub.execute_input":"2021-05-20T03:36:49.030514Z","iopub.status.idle":"2021-05-20T03:36:49.067593Z","shell.execute_reply.started":"2021-05-20T03:36:49.030458Z","shell.execute_reply":"2021-05-20T03:36:49.066478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# add StudyInstanceUID_count column\ngroup_col = 'StudyInstanceUID'\ndf=pd.DataFrame(train.groupby(group_col)['id'].count())\ndf.columns = [f'{group_col}_count']\ntrain=train.merge(df.reset_index(), on=group_col)\none_study_multi_image_df = train[train[f'{group_col}_count'] > 1]\nprint(len(one_study_multi_image_df))\ntrain = train[train[f'{group_col}_count'] == 1] # delete 'StudyInstanceUID_count > 1' data\none_study_multi_image_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:36:49.069275Z","iopub.execute_input":"2021-05-20T03:36:49.069750Z","iopub.status.idle":"2021-05-20T03:36:49.114716Z","shell.execute_reply.started":"2021-05-20T03:36:49.069703Z","shell.execute_reply":"2021-05-20T03:36:49.113468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"we have 512images with 'StudyInstanceUID_count > 1'.  \nSince the problem has not been solved, I deleted this data.  \nhttps://www.kaggle.com/c/siim-covid19-detection/discussion/239980","metadata":{}},{"cell_type":"code","source":"def pie_plot(train_df, variable):\n    \"\"\"\n        input: variable ex: \"Sex\"\n        output: bar plot & value count\n    \"\"\"\n    # get feature\n    var = train_df[variable]\n    # count number of categorical variable(value/sample)\n    varValue = var.value_counts()\n    \n    # visualize\n    plt.figure(figsize = (20,10))\n    plt.pie(varValue, labels=varValue.index, autopct=\"%1.1f%%\")\n#     plt.xticks(varValue.index, varValue.index.values)\n#     plt.ylabel(\"Frequency\")\n    plt.title('target')\n    plt.show()\n    \ntrain['target'] = 'Negative for Pneumonia'\ntrain.loc[train['Typical Appearance']==1, 'target'] = 'Typical Appearance'\ntrain.loc[train['Indeterminate Appearance']==1, 'target'] = 'Indeterminate Appearance'\ntrain.loc[train['Atypical Appearance']==1, 'target'] = 'Atypical Appearance'\nprint(train.target.value_counts())\npie_plot(train, 'target')    ","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:36:49.116313Z","iopub.execute_input":"2021-05-20T03:36:49.116794Z","iopub.status.idle":"2021-05-20T03:36:49.269386Z","shell.execute_reply.started":"2021-05-20T03:36:49.116742Z","shell.execute_reply":"2021-05-20T03:36:49.268386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.boxes.values[0] # x_min, y_min, width, height","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:36:49.270799Z","iopub.execute_input":"2021-05-20T03:36:49.271144Z","iopub.status.idle":"2021-05-20T03:36:49.279867Z","shell.execute_reply.started":"2021-05-20T03:36:49.271109Z","shell.execute_reply":"2021-05-20T03:36:49.278356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.label.values[0] # x_min, y_min, x_max, y_max","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:36:49.282363Z","iopub.execute_input":"2021-05-20T03:36:49.282687Z","iopub.status.idle":"2021-05-20T03:36:49.297225Z","shell.execute_reply.started":"2021-05-20T03:36:49.282658Z","shell.execute_reply":"2021-05-20T03:36:49.296084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = ['Typical Appearance', 'Indeterminate Appearance', 'Atypical Appearance'] # we have 3 positive classes\nunique_classes = np.unique(train[class_names].values, axis=0)\nunique_classes # no multi label ","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:36:49.299139Z","iopub.execute_input":"2021-05-20T03:36:49.300036Z","iopub.status.idle":"2021-05-20T03:36:49.322945Z","shell.execute_reply.started":"2021-05-20T03:36:49.299972Z","shell.execute_reply":"2021-05-20T03:36:49.321641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plot img with bounding box","metadata":{"papermill":{"duration":0.045173,"end_time":"2021-01-01T01:52:00.181219","exception":false,"start_time":"2021-01-01T01:52:00.136046","status":"completed"},"tags":[]}},{"cell_type":"code","source":"imgs = []\nlabel2color = {\n    '[1, 0, 0]': [255,0,0], # Typical Appearance\n    '[0, 1, 0]': [0,255,0], # Indeterminate Appearance\n    '[0, 0, 1]': [0,0,255], # Atypical Appearance\n    '[0, 0, 0]': None, # negative\n}\nlabel2target = {\n    '[1, 0, 0]': 'typical',\n    '[0, 1, 0]': 'indeterminate',\n    '[0, 0, 1]': 'atypical'\n}\nthickness = 3\nscale = 5\nFONT = cv2.FONT_HERSHEY_SIMPLEX; FONT_SCALE = 1; FONT_THICKNESS = 2; FONT_LINE_TYPE = cv2.LINE_AA;\n\nfor _, row in train[train['Negative for Pneumonia']==0].iloc[:8].iterrows():\n    study_id = row['StudyInstanceUID']\n    img_path = glob(f'{dataset_dir}/train/{study_id}/*/*')[0]\n    img = dicom2array(path=img_path)\n    img = cv2.resize(img, None, fx=1/scale, fy=1/scale)\n    img = np.stack([img, img, img], axis=-1)\n    \n    claz = row[class_names].values\n    color = label2color[str(claz.tolist())]\n    target = label2target[str(claz.tolist())]\n\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row['label'].split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l)/scale)\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []    \n\n    text_width, text_height = cv2.getTextSize(target, FONT, FONT_SCALE, FONT_THICKNESS)[0]\n    for box in bboxes:\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, thickness\n        )\n        box_width = int(box[2]) - int(box[0])\n        img = cv2.putText(img, target, (int(box[0])-(text_width-box_width)//2, int(box[1])-10),\n                        FONT, FONT_SCALE, color, FONT_THICKNESS, FONT_LINE_TYPE)          \n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)\n","metadata":{"papermill":{"duration":17.072406,"end_time":"2021-01-01T01:52:17.298447","exception":false,"start_time":"2021-01-01T01:52:00.226041","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-05-20T03:36:49.326538Z","iopub.execute_input":"2021-05-20T03:36:49.326851Z","iopub.status.idle":"2021-05-20T03:36:52.095960Z","shell.execute_reply.started":"2021-05-20T03:36:49.326822Z","shell.execute_reply":"2021-05-20T03:36:52.094554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Typical Appearance only","metadata":{}},{"cell_type":"code","source":"imgs = []\nthickness = 3\nscale = 5\n\nfor _, row in train[train['Typical Appearance'] == 1].iloc[:16].iterrows():\n    study_id = row['StudyInstanceUID']\n    img_path = glob(f'{dataset_dir}/train/{study_id}/*/*')[0]\n    img = dicom2array(path=img_path)\n    img = cv2.resize(img, None, fx=1/scale, fy=1/scale)\n    img = np.stack([img, img, img], axis=-1)\n    \n    claz = row[class_names].values\n    color = label2color[str(claz.tolist())]\n\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row['label'].split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l)/scale)\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []    \n    \n    for box in bboxes:\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, thickness\n    )\n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:36:52.097757Z","iopub.execute_input":"2021-05-20T03:36:52.098491Z","iopub.status.idle":"2021-05-20T03:36:57.985793Z","shell.execute_reply.started":"2021-05-20T03:36:52.098419Z","shell.execute_reply":"2021-05-20T03:36:57.984803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Indeterminate Appearance only","metadata":{}},{"cell_type":"code","source":"imgs = []\nthickness = 3\nscale = 5\n\nfor _, row in train[train['Indeterminate Appearance'] == 1].iloc[:16].iterrows():\n    study_id = row['StudyInstanceUID']\n    img_path = glob(f'{dataset_dir}/train/{study_id}/*/*')[0]\n    img = dicom2array(path=img_path)\n    img = cv2.resize(img, None, fx=1/scale, fy=1/scale)\n    img = np.stack([img, img, img], axis=-1)\n    \n    claz = row[class_names].values\n    color = label2color[str(claz.tolist())]\n\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row['label'].split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l)/scale)\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []    \n    \n    for box in bboxes:\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, thickness\n    )\n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:36:57.987250Z","iopub.execute_input":"2021-05-20T03:36:57.987724Z","iopub.status.idle":"2021-05-20T03:37:03.555159Z","shell.execute_reply.started":"2021-05-20T03:36:57.987688Z","shell.execute_reply":"2021-05-20T03:37:03.553604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Atypical Appearance only","metadata":{}},{"cell_type":"code","source":"imgs = []\nthickness = 3\nscale = 5\n\nfor _, row in train[train['Atypical Appearance'] == 1].iloc[:16].iterrows():\n    study_id = row['StudyInstanceUID']\n    img_path = glob(f'{dataset_dir}/train/{study_id}/*/*')[0]\n    img = dicom2array(path=img_path)\n    img = cv2.resize(img, None, fx=1/scale, fy=1/scale)\n    img = np.stack([img, img, img], axis=-1)\n    \n    claz = row[class_names].values\n    color = label2color[str(claz.tolist())]\n\n    bboxes = []\n    bbox = []\n    \n    for i, l in enumerate(row['label'].split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l)/scale)\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []    \n    \n    for box in bboxes:\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, thickness\n    )\n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:37:03.557467Z","iopub.execute_input":"2021-05-20T03:37:03.557963Z","iopub.status.idle":"2021-05-20T03:37:09.899323Z","shell.execute_reply.started":"2021-05-20T03:37:03.557898Z","shell.execute_reply":"2021-05-20T03:37:09.898080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# heatmap","metadata":{}},{"cell_type":"code","source":"train_with_size = pd.read_csv('../input/siim-external/train.csv') # present for you\ntrain_with_size = train_with_size[train_with_size['Negative for Pneumonia'] == 0]\ntrain_with_size['class_id'] = 0\ntrain_with_size.loc[train_with_size['Indeterminate Appearance'] == 1, 'class_id'] = 1\ntrain_with_size.loc[train_with_size['Atypical Appearance'] == 1, 'class_id'] = 2\ntrain_with_size.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:37:09.900763Z","iopub.execute_input":"2021-05-20T03:37:09.901319Z","iopub.status.idle":"2021-05-20T03:37:09.974431Z","shell.execute_reply.started":"2021-05-20T03:37:09.901279Z","shell.execute_reply":"2021-05-20T03:37:09.973358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_2_label = {0: 'Typical Appearance', 1: 'Indeterminate Appearance', 2: 'Atypical Appearance'}\nHEATMAP_SIZE = (int(train_with_size.height.mean()), int(train_with_size.width.mean()), 3)\n\n# Initialize\nheatmap = np.zeros((HEATMAP_SIZE), dtype=np.int16)\nbbox_np = train_with_size[[\"class_id\", \"scaled_x_min\", \"scaled_x_max\", \"scaled_y_min\", \"scaled_y_max\"]].to_numpy()\nbbox_np[:, 1:3] *= int(train_with_size.width.mean())\nbbox_np[:, 3:5] *= int(train_with_size.height.mean())\nbbox_np = np.floor(bbox_np).astype(np.int16)\n\n# Color map stuff\ncustom_cmaps = [\n    matplotlib.colors.LinearSegmentedColormap.from_list(\n        colors=[(0.,0.,0.), c, (0.95,0.95,0.95)], \n        name=f\"custom_{i}\") for i,c in enumerate(sns.color_palette(\"Spectral\", 4))\n]\n\nfor row in tqdm(bbox_np, total=bbox_np.shape[0]):\n    heatmap[row[3]:row[4]+1, row[1]:row[2]+1, row[0]] += 1\n    \nfig = plt.figure(figsize=(20,25))\nplt.suptitle(\"Heatmaps Showing Bounding Box Placement\\n \", fontweight=\"bold\", fontsize=16)\nfor i in range(4):\n    plt.subplot(4, 4, i+1)\n    if i==0:\n        plt.imshow(heatmap.mean(axis=-1), cmap=\"bone\")\n        plt.title(f\"Average of All Classes\", fontweight=\"bold\")\n    else:\n        plt.imshow(heatmap[:, :, i-1], cmap=custom_cmaps[i-1])\n        plt.title(num_2_label[i-1], fontweight=\"bold\")\n        \n    plt.axis(False)\nfig.tight_layout(rect=[0, 0.03, 1, 0.97])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:37:09.976336Z","iopub.execute_input":"2021-05-20T03:37:09.976822Z","iopub.status.idle":"2021-05-20T03:37:32.604641Z","shell.execute_reply.started":"2021-05-20T03:37:09.976763Z","shell.execute_reply":"2021-05-20T03:37:32.598899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"left-right difference in indeterminate & atypical?","metadata":{}},{"cell_type":"markdown","source":"# box area distribution","metadata":{}},{"cell_type":"code","source":"# We need to compare on the same scale.\ntrain_with_size['area'] = (train_with_size['scaled_x_max'] - train_with_size['scaled_x_min']) * (train_with_size['scaled_y_max'] - train_with_size['scaled_y_min'])","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:37:32.609235Z","iopub.execute_input":"2021-05-20T03:37:32.610193Z","iopub.status.idle":"2021-05-20T03:37:32.628532Z","shell.execute_reply.started":"2021-05-20T03:37:32.610040Z","shell.execute_reply":"2021-05-20T03:37:32.626764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,5))\nplt.title('typical only: distribution of box area (max 1)')\nax = sns.distplot(train_with_size[train_with_size['Typical Appearance']==1]['area'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:37:32.631610Z","iopub.execute_input":"2021-05-20T03:37:32.632106Z","iopub.status.idle":"2021-05-20T03:37:32.984761Z","shell.execute_reply.started":"2021-05-20T03:37:32.632059Z","shell.execute_reply":"2021-05-20T03:37:32.983353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,5))\nplt.title('indeterminate only: distribution of box area (max 1)')\nax = sns.distplot(train_with_size[train_with_size['Indeterminate Appearance']==1]['area'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:37:32.991334Z","iopub.execute_input":"2021-05-20T03:37:32.992011Z","iopub.status.idle":"2021-05-20T03:37:33.268948Z","shell.execute_reply.started":"2021-05-20T03:37:32.991963Z","shell.execute_reply":"2021-05-20T03:37:33.266847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,5))\nplt.title('atypical only: distribution of box area (max 1)')\nax = sns.distplot(train_with_size[train_with_size['Atypical Appearance']==1]['area'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:37:33.271115Z","iopub.execute_input":"2021-05-20T03:37:33.271482Z","iopub.status.idle":"2021-05-20T03:37:33.520096Z","shell.execute_reply.started":"2021-05-20T03:37:33.271447Z","shell.execute_reply":"2021-05-20T03:37:33.518776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"・As we can see in the heat map, the 'typical' seems to have the largest area.  \n・In atypical, small boxes are in the majority.  \nDoes the box get bigger as the annotator confidently labels typical examples?  \nOr does the range get larger because the typical example has a more advanced disease state?  \nI don't know the truth, but it's fun to imagine the annotation process 😌  \n\nUpdated:\nThe following url describes annotations.  \nhttps://www.kaggle.com/c/siim-covid19-detection/discussion/240250","metadata":{}},{"cell_type":"markdown","source":"# plot small boxes only","metadata":{}},{"cell_type":"code","source":"small_images = train_with_size.query('x_min != 0').sort_values('area').image_id.values[:16]\ntrain['image_id'] = train['id'].apply(lambda x: x.replace('_image', ''))","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:37:33.522035Z","iopub.execute_input":"2021-05-20T03:37:33.522651Z","iopub.status.idle":"2021-05-20T03:37:33.552501Z","shell.execute_reply.started":"2021-05-20T03:37:33.522593Z","shell.execute_reply":"2021-05-20T03:37:33.550776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs = []\n\nfor _, row in train[train.image_id.isin(small_images)].iterrows():\n    study_id = row['StudyInstanceUID']\n    img_path = glob(f'{dataset_dir}/train/{study_id}/*/*')[0]\n    img = dicom2array(path=img_path)\n    img = cv2.resize(img, None, fx=1/scale, fy=1/scale)\n    img = np.stack([img, img, img], axis=-1)\n    \n    claz = row[class_names].values\n    color = label2color[str(claz.tolist())]\n    target = label2target[str(claz.tolist())]\n\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row['label'].split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l)/scale)\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []    \n\n    text_width, text_height = cv2.getTextSize(target, FONT, FONT_SCALE, FONT_THICKNESS)[0]\n    for box in bboxes:\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, thickness\n        )\n        box_width = int(box[2]) - int(box[0])\n        img = cv2.putText(img, target, (int(box[0])-(text_width-box_width)//2, int(box[1])-10),\n                        FONT, FONT_SCALE, color, FONT_THICKNESS, FONT_LINE_TYPE)          \n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:37:33.554641Z","iopub.execute_input":"2021-05-20T03:37:33.555296Z","iopub.status.idle":"2021-05-20T03:37:47.945970Z","shell.execute_reply.started":"2021-05-20T03:37:33.555256Z","shell.execute_reply":"2021-05-20T03:37:47.943357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"difficult... give me the doctor's eyes 😡","metadata":{}},{"cell_type":"markdown","source":"# compare negative / typical","metadata":{}},{"cell_type":"code","source":"imgs = []\nfor i in range(32):\n    if i % 2 == 0:\n        row = train[train['Negative for Pneumonia']==1].iloc[i]\n    else:\n        row = train[train['Typical Appearance']==1].iloc[i]\n        \n    study_id = row['StudyInstanceUID']\n    img_path = glob(f'{dataset_dir}/train/{study_id}/*/*')[0]\n    img = dicom2array(path=img_path)\n    img = cv2.resize(img, None, fx=1/scale, fy=1/scale)\n    img = np.stack([img, img, img], axis=-1)\n    \n    claz = row[class_names].values\n    color = label2color[str(claz.tolist())]\n\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row['label'].split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l)/scale)\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []    \n    \n    for box in bboxes:\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            color, thickness\n    )\n    img = cv2.resize(img, (500,500))\n    imgs.append(img)\n    \nplot_imgs(imgs, cmap=None)","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:37:47.947928Z","iopub.execute_input":"2021-05-20T03:37:47.948283Z","iopub.status.idle":"2021-05-20T03:38:01.074986Z","shell.execute_reply.started":"2021-05-20T03:37:47.948248Z","shell.execute_reply":"2021-05-20T03:38:01.069940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's learn to recognize the positive samples 💪","metadata":{}},{"cell_type":"markdown","source":"# plot from submission.csv","metadata":{}},{"cell_type":"code","source":"CONF_THRESHOLD = 0.3\n\ndef pred_str_to_confidence_box_maps(pred_str):\n    confidence_box_maps = []\n\n    for i, pred in enumerate(pred_str.split(' ')):\n        if i % 6 == 0:\n            confidence_box_map = {}\n            box = []\n        elif i % 6 == 1:\n            conf = float(pred)\n            confidence_box_map['conf'] = conf\n    #         print(confidence_box_map)\n        else:\n            box.append(int(float(pred)))\n        if i % 6 == 5:\n            confidence_box_map['box'] = box\n            if conf > CONF_THRESHOLD:\n                confidence_box_maps.append(confidence_box_map)\n    return confidence_box_maps\n\ndef add_bbox(img, confidence_box_maps):\n    for confidence_box_map in confidence_box_maps:\n        conf = confidence_box_map['conf']\n        box = confidence_box_map['box']\n        text = f'{round(conf, 4)}'\n        img = cv2.rectangle(\n            img,\n            (int(box[0]), int(box[1])),\n            (int(box[2]), int(box[3])),\n            [255,0,0], thickness\n        )    \n        box_width = int(box[2]) - int(box[0])\n        if img.shape[0] > 2000:\n#         if True:\n            font_scale = int(7 * img.shape[0] / 4000)\n            font_tickness = int(20 * img.shape[0] / 4000)\n        else:\n            font_scale = int(7 * img.shape[0] / 2000)\n            font_tickness = int(20 * img.shape[0] / 2000)            \n        text_width, text_height = cv2.getTextSize(text, FONT, font_scale, font_tickness)[0]\n#         text_width, text_height, font_scale, font_tickness = 677, 135, 6, 17,\n        img = cv2.putText(img, text, (int(box[0])-(text_width-box_width)//2, int(box[1])-40),\n                        FONT, font_scale, [255,0,0], font_tickness, FONT_LINE_TYPE)              \n    return img\n\ndef print_study_conf(sub, image_id):\n    study_id = glob(f'../input/siim-covid19-detection/test/*/*/{image_id}.dcm')[0].split('/')[-3]\n    study_pred_str = sub[sub['id'].str.startswith(study_id)].PredictionString.values[0]\n    names, confs = [], []\n    for i, pred in enumerate(study_pred_str.split(' ')):\n        if i % 6 == 0:\n            names.append(pred)\n        if i % 6 == 1:\n            confs.append(pred)\n    print('<↓study conficence↓>')\n    for name, conf in zip(names, confs):\n        print(f'{name} confidence: {conf}')\n\n\ndef plot_with_pred(sub, image_id):\n    print_study_conf(sub, image_id)\n    path = glob(f'../input/siim-covid19-detection/test/*/*/{image_id}.dcm')[0]\n    img = dicom2array(path)\n    pred_str = sub[sub['id'].str.startswith(image_id)].PredictionString.values[0]\n    confidence_box_maps = pred_str_to_confidence_box_maps(pred_str)\n    if len(confidence_box_maps) == 0:\n        print(f'{image_id}: There are no boxes with confidence greater than {CONF_THRESHOLD}.')\n    img = add_bbox(img, confidence_box_maps)\n    plot_img(img)\n","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:38:01.076901Z","iopub.execute_input":"2021-05-20T03:38:01.077319Z","iopub.status.idle":"2021-05-20T03:38:01.100457Z","shell.execute_reply.started":"2021-05-20T03:38:01.077282Z","shell.execute_reply":"2021-05-20T03:38:01.098976Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"part_of_my_sub = pd.read_csv('../input/test-subs/part_of_submission.csv')  # use your submission.csv\npart_of_my_sub","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:38:01.102097Z","iopub.execute_input":"2021-05-20T03:38:01.102447Z","iopub.status.idle":"2021-05-20T03:38:01.140630Z","shell.execute_reply.started":"2021-05-20T03:38:01.102412Z","shell.execute_reply":"2021-05-20T03:38:01.139485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_ids = ['d5911a060ee4', '25b281d5a9f3', '03a778f5a68b', '2aaab6a41f1a', '4f7f40e478b1', '88782677cbec']\nfor image_id in image_ids:\n    plot_with_pred(part_of_my_sub, image_id) # input: submission.csv, image_id","metadata":{"execution":{"iopub.status.busy":"2021-05-20T03:38:01.142231Z","iopub.execute_input":"2021-05-20T03:38:01.142588Z","iopub.status.idle":"2021-05-20T03:38:17.424549Z","shell.execute_reply.started":"2021-05-20T03:38:01.142551Z","shell.execute_reply":"2021-05-20T03:38:17.423329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 😊 Please upvote if you found this helpful 😊","metadata":{}}]}