{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## SIIM COVID-19: Visual verification of images and cleanup of the training dataset\nSelective verification of scaled images and bboxes.\n\nImages have been processed and converted to JPG here: https://www.kaggle.com/code/alexanderyyy/siim-covid-19-process-to-jpg-512-with-bboxes\n\nOriginal DICOM on the left, processed JPG with updated bboxes on the right.\n\nPlease remember to upvote, thanks!","metadata":{}},{"cell_type":"code","source":"!pip install python-gdcm -q","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-11-02T17:03:14.277478Z","iopub.execute_input":"2022-11-02T17:03:14.278045Z","iopub.status.idle":"2022-11-02T17:03:31.578010Z","shell.execute_reply.started":"2022-11-02T17:03:14.277939Z","shell.execute_reply":"2022-11-02T17:03:31.576373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom matplotlib.patches import Rectangle\nfrom matplotlib import patches\nimport pandas as pd\nimport numpy as np\nfrom fastai.medical.imaging import *\nimport pydicom","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:03:31.580740Z","iopub.execute_input":"2022-11-02T17:03:31.581276Z","iopub.status.idle":"2022-11-02T17:03:35.935345Z","shell.execute_reply.started":"2022-11-02T17:03:31.581224Z","shell.execute_reply":"2022-11-02T17:03:35.934040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_items = get_dicom_files('../input/siim-covid19-detection/train')\nprint(f'Got {len(train_items)} DICOM files from the TRAIN path')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:03:35.936967Z","iopub.execute_input":"2022-11-02T17:03:35.938204Z","iopub.status.idle":"2022-11-02T17:05:03.906019Z","shell.execute_reply.started":"2022-11-02T17:03:35.938161Z","shell.execute_reply":"2022-11-02T17:05:03.904190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read and prepare original train annotation file\ntr_df = pd.read_csv('../input/siim-covid19-detection/train_image_level.csv')\ntr_df.id = tr_df.apply(lambda r: r.id.replace('_image',''), axis=1)\ntr_df = tr_df.set_index('id')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:05:03.909855Z","iopub.execute_input":"2022-11-02T17:05:03.910421Z","iopub.status.idle":"2022-11-02T17:05:04.091012Z","shell.execute_reply.started":"2022-11-02T17:05:03.910376Z","shell.execute_reply":"2022-11-02T17:05:04.089649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read processed train annotation file with bboxes for the images with size 512x512\njdf = pd.read_csv('../input/siim-cov19-processed-jpg-512-with-updated-bboxes/train_512.csv')\njdf = jdf.set_index('id')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:05:04.092811Z","iopub.execute_input":"2022-11-02T17:05:04.093698Z","iopub.status.idle":"2022-11-02T17:05:04.151274Z","shell.execute_reply.started":"2022-11-02T17:05:04.093658Z","shell.execute_reply":"2022-11-02T17:05:04.149994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualization","metadata":{}},{"cell_type":"code","source":"def find_dcm_path(id, dicom_files):\n    for it in dicom_files:\n        if id == it.stem: return it","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-11-02T17:05:04.153304Z","iopub.execute_input":"2022-11-02T17:05:04.154097Z","iopub.status.idle":"2022-11-02T17:05:04.159623Z","shell.execute_reply.started":"2022-11-02T17:05:04.154058Z","shell.execute_reply":"2022-11-02T17:05:04.158287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# this function will plot DICOM on the left, JPG on the right\ndef plot_test( vids, figsize ):\n    ni = len(vids)\n    folder = '../input/siim-cov19-processed-jpg-512-with-updated-bboxes/train'\n    fig, ax = plt.subplots(ni,2, figsize=( figsize, (figsize*ni)//2 ))\n #   fig, ax = plt.subplots(ni,2, figsize=(figsize,figsize))\n    fig.suptitle('DICOM on the left, JPG on the right', fontsize=16)\n    \n    fig.tight_layout()\n    #plt.subplots_adjust(hspace=0, wspace=0)\n    \n    for vin in range(ni): \n        iid = vids[vin]\n        # original DICOM on the left\n        img = find_dcm_path(iid, train_items).dcmread()\n        pixels = img.pixel_array\n        if img.PhotometricInterpretation == \"MONOCHROME1\":\n            pixels = np.amax(pixels) - pixels\n        ax[vin,0].imshow(pixels, cmap='gray')\n        ax[vin,0].set_title(iid + '  Study:' + tr_df.loc[iid].StudyInstanceUID)\n        if pd.notnull(tr_df.loc[iid].boxes):\n            bxs = eval(tr_df.loc[iid].boxes)\n            for bx in bxs:\n                rct = patches.Rectangle( (bx['x'], bx['y']), bx['width'], bx['height'], edgecolor=\"yellow\", facecolor=\"none\", lw=2 )\n                ax[vin,0].add_patch(rct)\n\n        # processed JPG on the right\n        jmg = mpimg.imread(folder + '/' + iid + '.jpg')\n        ax[vin,1].imshow(jmg, cmap='gray')\n        ax[vin,1].set_title(iid)\n        if pd.notnull(jdf.loc[iid].boxes):\n            bxs = eval(jdf.loc[iid].boxes)\n            #print(iid, bxs)\n            for bx in bxs:\n                rct = patches.Rectangle( (bx['x'], bx['y']), bx['w'], bx['h'], edgecolor=\"yellow\", facecolor=\"none\", lw=2 )\n                ax[vin,1].add_patch(rct)\n    plt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-11-02T17:05:04.161680Z","iopub.execute_input":"2022-11-02T17:05:04.162618Z","iopub.status.idle":"2022-11-02T17:05:04.180616Z","shell.execute_reply.started":"2022-11-02T17:05:04.162572Z","shell.execute_reply":"2022-11-02T17:05:04.178702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting random samples - re-run  to see more\nnsamples = 3\nsample_idl = [idx for idx in tr_df.sample(nsamples).index]\nplot_test(sample_idl,12)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:05:04.182911Z","iopub.execute_input":"2022-11-02T17:05:04.183643Z","iopub.status.idle":"2022-11-02T17:05:10.064662Z","shell.execute_reply.started":"2022-11-02T17:05:04.183600Z","shell.execute_reply":"2022-11-02T17:05:10.063478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Demo of improvements by image processing","metadata":{}},{"cell_type":"code","source":"# list of image IDs to verify visually: with borders, low contrast, etc.\n# trouble images, improved by hist equalization and crop\nvis_ids = ['7d3112e63a37', '9c24e37a0ef5', '2909830fc390', '5b687c54d3fd', \n           '32579cfb5545', '89fd7f185d77', '7e99b9eb897b', '9a8d4bf6141a',]\n\nplot_test(vis_ids, 12)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:14:44.346065Z","iopub.execute_input":"2022-11-02T17:14:44.346729Z","iopub.status.idle":"2022-11-02T17:14:57.157459Z","shell.execute_reply.started":"2022-11-02T17:14:44.346676Z","shell.execute_reply":"2022-11-02T17:14:57.155628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparing a list of bad images\nSome hints found here: https://www.kaggle.com/code/dschettler8845/covid-detection-studies-with-multiple-images-viz/notebook","metadata":{}},{"cell_type":"code","source":"# helper function\ndef get_study(study_id):\n    # returns all images in this study\n    sidx = jdf[jdf.StudyInstanceUID == study_id].index\n    simgs = [idx for idx in sidx]\n    #print(simgs)\n    return simgs","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:05:10.066425Z","iopub.execute_input":"2022-11-02T17:05:10.067108Z","iopub.status.idle":"2022-11-02T17:05:10.073574Z","shell.execute_reply.started":"2022-11-02T17:05:10.067068Z","shell.execute_reply":"2022-11-02T17:05:10.071849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sdf = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv')\nsdf.id = sdf.apply(lambda r: r.id.replace('_study',''), axis=1)\nsdf = sdf.set_index('id')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:05:10.077701Z","iopub.execute_input":"2022-11-02T17:05:10.078095Z","iopub.status.idle":"2022-11-02T17:05:10.234869Z","shell.execute_reply.started":"2022-11-02T17:05:10.078062Z","shell.execute_reply":"2022-11-02T17:05:10.233523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for Study with multiple images - keep only image with boxes\n# if no boxes found - mark Study for the manual check\n\nmanual_check = []\nids_to_remove = []\nfor st_id in sdf.index:\n    si = get_study(st_id)\n    if len(si) > 1: \n        with_box = False\n        for iid in si:\n            if pd.notnull(jdf.loc[iid].boxes): \n                if with_box: # second image with box in the same study\n                    manual_check.append(st_id) # review this study manually\n                si.remove(iid) # except the one with box,\n                ids_to_remove += si # all other mark as 'bad'\n                with_box = True\n                \n        if not with_box: \n            manual_check.append(st_id) # review this study manually","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:05:10.236870Z","iopub.execute_input":"2022-11-02T17:05:10.237495Z","iopub.status.idle":"2022-11-02T17:05:15.167761Z","shell.execute_reply.started":"2022-11-02T17:05:10.237454Z","shell.execute_reply":"2022-11-02T17:05:15.166726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for manual_check Studies: when not \"negative\" - move to \"ids_to_remove\"\n# when study is positive, but contains no boxes - remove all images from this Study\nfor st_id in manual_check:\n    if sdf.loc[st_id]['Negative for Pneumonia'] == 0:\n        # this study is positive, but contains no boxes - remove all\n        si = get_study(st_id)\n        ids_to_remove += si # all other mark as 'bad'\n        manual_check.remove(st_id)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:05:15.169154Z","iopub.execute_input":"2022-11-02T17:05:15.169590Z","iopub.status.idle":"2022-11-02T17:05:15.186952Z","shell.execute_reply.started":"2022-11-02T17:05:15.169548Z","shell.execute_reply":"2022-11-02T17:05:15.185233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Manual visual check for the \"bad\" studies","metadata":{}},{"cell_type":"code","source":"for si in manual_check:\n    print(si, get_study(si))\n    plot_test( get_study(si), 8 )","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:05:15.188625Z","iopub.execute_input":"2022-11-02T17:05:15.189223Z","iopub.status.idle":"2022-11-02T17:08:02.098185Z","shell.execute_reply.started":"2022-11-02T17:05:15.189168Z","shell.execute_reply":"2022-11-02T17:08:02.096698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### I am not a radiologist, but some images do not contribute to science\nBadly cropped, unusual positions, obvious opacities, etc. I prefer to skip sample in doubt rather than keep confusing data.","metadata":{}},{"cell_type":"code","source":"bad_studies = [\n    '2edd69dd0934', '350cd603e64d', '3757a5a70aa8', '3e2d6faac5dd', '3f1e5e240522', \n    '4c1e5ca11b6e', '4c45ac349e3a', '4fe89b3c7225', '7416b5cbc531', '8c93b1d9363b', \n    '9855cbfdcb71', 'b185438c915c', 'ba27bcbd6881', 'bafb7e87a1b9', 'bddc88dfb40f', \n    'bf6f81be0705', 'c36cde17cd04', 'c6af19161721', 'cd082bbb2799', 'd172b5d65314', \n    'df07c0223107', 'e126d9d23457', 'e4b2c5f23e08', 'e4b50e7402c3', 'f07ec852f8c2', \n    'f109811743c5', 'f2b77c3c70c5', 'f6ffe212deeb',    \n]","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:08:02.099968Z","iopub.execute_input":"2022-11-02T17:08:02.100703Z","iopub.status.idle":"2022-11-02T17:08:02.107821Z","shell.execute_reply.started":"2022-11-02T17:08:02.100662Z","shell.execute_reply":"2022-11-02T17:08:02.106589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove all images from the \"bad\" Studies\nfor st_id in bad_studies:\n    si = get_study(st_id)\n    ids_to_remove += si # all images mark as 'bad'","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:08:02.109445Z","iopub.execute_input":"2022-11-02T17:08:02.109826Z","iopub.status.idle":"2022-11-02T17:08:02.144159Z","shell.execute_reply.started":"2022-11-02T17:08:02.109794Z","shell.execute_reply":"2022-11-02T17:08:02.143192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# other bad images noticed\nbad_images = [ '6d496f79fe93', '06f6423be3f9',\n    '06f6423be3f9', '8568c0c16033',\n    '7e99b9eb897b', '7ba4420fbb89']","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:25:11.640876Z","iopub.execute_input":"2022-11-02T17:25:11.642232Z","iopub.status.idle":"2022-11-02T17:25:11.651478Z","shell.execute_reply.started":"2022-11-02T17:25:11.642173Z","shell.execute_reply":"2022-11-02T17:25:11.648461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids_to_remove += bad_images\nlen(ids_to_remove)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:25:13.428836Z","iopub.execute_input":"2022-11-02T17:25:13.429665Z","iopub.status.idle":"2022-11-02T17:25:13.438096Z","shell.execute_reply.started":"2022-11-02T17:25:13.429616Z","shell.execute_reply":"2022-11-02T17:25:13.436935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Making cleaner annotation file","metadata":{}},{"cell_type":"code","source":"cdf = pd.DataFrame()\nnull_boxes = 0\nfor idx in jdf.index:\n    if idx in ids_to_remove: # skip bad image\n        continue\n    r = jdf.loc[idx]\n    if pd.isnull(r.boxes) and r.label_y != 'Neg':\n        null_boxes += 1\n        continue\n    cdf = cdf.append(r)\n\nprint(f'Amount of null_boxes detected:{null_boxes}')\nprint(f'Amount of records removed:{len(jdf) - len(cdf)}')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:38:56.291098Z","iopub.execute_input":"2022-11-02T17:38:56.291552Z","iopub.status.idle":"2022-11-02T17:39:12.857162Z","shell.execute_reply.started":"2022-11-02T17:38:56.291516Z","shell.execute_reply":"2022-11-02T17:39:12.855935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we still have duplicates from Studies with multiple images - remove duplicates\nbefore = len(cdf)\ncdf = cdf.drop_duplicates(subset=['StudyInstanceUID'], keep='first')\nprint(f'Amount of duplicate records removed:{before - len(cdf)}')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:46:56.897186Z","iopub.execute_input":"2022-11-02T17:46:56.897855Z","iopub.status.idle":"2022-11-02T17:46:56.912303Z","shell.execute_reply.started":"2022-11-02T17:46:56.897800Z","shell.execute_reply":"2022-11-02T17:46:56.910900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cdf.to_csv('train_512_clean.csv', index_label = 'id')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:47:07.834814Z","iopub.execute_input":"2022-11-02T17:47:07.835545Z","iopub.status.idle":"2022-11-02T17:47:07.865629Z","shell.execute_reply.started":"2022-11-02T17:47:07.835487Z","shell.execute_reply":"2022-11-02T17:47:07.864625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t = pd.read_csv('train_512_clean.csv')","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:48:29.636999Z","iopub.execute_input":"2022-11-02T17:48:29.637431Z","iopub.status.idle":"2022-11-02T17:48:29.657787Z","shell.execute_reply.started":"2022-11-02T17:48:29.637395Z","shell.execute_reply":"2022-11-02T17:48:29.656943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Final labels distribution","metadata":{}},{"cell_type":"code","source":"def show_dist(df):\n    labels = list(df.label_y.unique())\n    dist = [sum(df.label_y == lab) for lab in labels]\n    print(dict(zip(labels,dist)))\n    explode = [0.05 for _ in range(len(labels))]\n    plt.pie(dist, labels = labels, startangle = 270, explode=explode, autopct='%1.1f%%')\n    plt.legend()\n    plt.show()\n    \nshow_dist(t)","metadata":{"execution":{"iopub.status.busy":"2022-11-02T17:48:38.713654Z","iopub.execute_input":"2022-11-02T17:48:38.714027Z","iopub.status.idle":"2022-11-02T17:48:38.946911Z","shell.execute_reply.started":"2022-11-02T17:48:38.713996Z","shell.execute_reply":"2022-11-02T17:48:38.945217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Have a nice day! Upvote, please","metadata":{}}]}