{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div style=\"background-color:#fdb920; font-size:140%;  font-family:sans-serif; text-align:left\"><b>This notebook analysis the covid19 data and generate required tables</b></div>","metadata":{}},{"cell_type":"markdown","source":"### 07.29.2022 update: add one block to store the boxes information into a csv file","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport plotly.graph_objects as go \nimport warnings\nwarnings.filterwarnings('ignore')\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nfrom ast import literal_eval\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-29T20:02:43.101332Z","iopub.execute_input":"2022-07-29T20:02:43.101740Z","iopub.status.idle":"2022-07-29T20:02:43.120852Z","shell.execute_reply.started":"2022-07-29T20:02:43.101705Z","shell.execute_reply":"2022-07-29T20:02:43.119587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install python-gdcm","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:06:39.612574Z","iopub.execute_input":"2022-07-29T19:06:39.613237Z","iopub.status.idle":"2022-07-29T19:06:52.580536Z","shell.execute_reply.started":"2022-07-29T19:06:39.613198Z","shell.execute_reply":"2022-07-29T19:06:52.579168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv', index_col=None)\nimage_df = pd.read_csv('/kaggle/input/siim-covid19-detection/train_image_level.csv', index_col=None)\nstudy_df = pd.read_csv('/kaggle/input/siim-covid19-detection/train_study_level.csv', index_col=None)\nprint(f\"Train image level csv shape : {image_df.shape}\")\nprint(f\"Train study level csv shape : {study_df.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T20:04:17.689588Z","iopub.execute_input":"2022-07-29T20:04:17.690678Z","iopub.status.idle":"2022-07-29T20:04:17.799638Z","shell.execute_reply.started":"2022-07-29T20:04:17.690635Z","shell.execute_reply":"2022-07-29T20:04:17.797794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nall_files = []\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        all_files.append(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:06:52.671240Z","iopub.execute_input":"2022-07-29T19:06:52.671720Z","iopub.status.idle":"2022-07-29T19:08:19.441350Z","shell.execute_reply.started":"2022-07-29T19:06:52.671682Z","shell.execute_reply":"2022-07-29T19:08:19.440004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show the first 5 rows of train_study_leve\nstudy_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:08:19.442714Z","iopub.execute_input":"2022-07-29T19:08:19.443182Z","iopub.status.idle":"2022-07-29T19:08:19.473812Z","shell.execute_reply.started":"2022-07-29T19:08:19.443148Z","shell.execute_reply":"2022-07-29T19:08:19.472938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show the first 5 rows of train_image_level.csv\nimage_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:08:19.475327Z","iopub.execute_input":"2022-07-29T19:08:19.475793Z","iopub.status.idle":"2022-07-29T19:08:19.489015Z","shell.execute_reply.started":"2022-07-29T19:08:19.475748Z","shell.execute_reply":"2022-07-29T19:08:19.488063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Explore Study level dataset","metadata":{}},{"cell_type":"markdown","source":"## Group all studies by their categories","metadata":{}},{"cell_type":"code","source":"study_group = pd.melt(study_df, id_vars=list(study_df.columns)[:1], value_vars=list(study_df.columns)[1:],\n             var_name='label', value_name='value')\nstudy_group = study_group.loc[study_group['value']!=0]\nstudy_group = study_group.groupby('label').sum().sort_values('value',ascending=False).reset_index()\nstudy_group['percentage']= round((study_group['value'] / study_group['value'].sum())*100,2) # calculate the percentage of each category\nstudy_group.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:08:19.490361Z","iopub.execute_input":"2022-07-29T19:08:19.490733Z","iopub.status.idle":"2022-07-29T19:08:19.533212Z","shell.execute_reply.started":"2022-07-29T19:08:19.490688Z","shell.execute_reply":"2022-07-29T19:08:19.532287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize category distribution","metadata":{}},{"cell_type":"code","source":"# Define color for each category\ncolors = {'Typical Appearance' : '#DCD427',\n'Negative for Pneumonia' : '#0092CC',\n'Indeterminate Appearance' : '#CC3333',\n          'Atypical Appearance' : '#E6E6E6'\n         }\n# Incorporate into the table\nstudy_group['color'] = study_group['label'].apply(lambda x: colors[x])\nstudy_group.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:08:19.535588Z","iopub.execute_input":"2022-07-29T19:08:19.536062Z","iopub.status.idle":"2022-07-29T19:08:19.552561Z","shell.execute_reply.started":"2022-07-29T19:08:19.535995Z","shell.execute_reply":"2022-07-29T19:08:19.551702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot figure\nfig = go.Figure(data=[go.Pie(labels=study_group['label'],\n                             values=study_group['percentage'],\n                             hole=.3,\n                             pull=[0.1, 0.1, 0.1, 0.1]\n                            )\n                     ]\n               )\nfig.update_traces(hoverinfo='label+percent', textinfo='percent', textfont_size=16,\n                  marker=dict(colors=study_group['color'], line=dict(color='#000000', width=2))\n                 )\nfig.update_layout(title={'text': \"% of labels in training data\",\n        'y':0.9,\n        'x':0.45,\n        'xanchor': 'center',\n        'yanchor': 'top'})\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:08:19.554148Z","iopub.execute_input":"2022-07-29T19:08:19.554687Z","iopub.status.idle":"2022-07-29T19:08:19.739584Z","shell.execute_reply.started":"2022-07-29T19:08:19.554652Z","shell.execute_reply":"2022-07-29T19:08:19.738230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"background-color:#fdb913; font-size:120%;  font-family:sans-serif; text-align:center\"><b>DICOM Image</b></div>","metadata":{}},{"cell_type":"markdown","source":"### Overview \nDICOM - Digital Imaging and Communication in Medicine is a standard format for encoding and transmitting medical images. This format stores image metadata like patient information, image acquistion parameters, image size, pixel size, etc along with the actual image. The image metadata is stored in the DICOM header. The image pixel data may be compressed using various techniques like JPEG, lossless JPEG, run length encoding (RLE), etc. Let's load a sample image to look at its header (metadata) and the actual pixel_array.\n\n### Bit-Depth\n\nThe pixel range of an image format is determined by its bit-depth. The range is $ [0, 2^{bitdepth} -1] $. For example, an 8-bit image will have a range of $[0, 2^{8} -1] = [0, 255]$. Most of the common photographic formats such as JPEG, PNG, etc use 8 bits for storage and only have positive values. A JPEG image containing 3 channels (RGB) will have a bit-depth of 8 for each channel hence a total bit-depth of 24.\n\nHowever, medical images use a higher bit-depth since higher accuracy is needed. For example the sample image metadata shown below uses ((0028, 0100) Bits Allocated field) uses 16 bits. Hence the range of pixel values for a 16 bit image is $[0, 65535] $ for a total of $2^{16} = 65536$ values. \n","metadata":{}},{"cell_type":"markdown","source":"### Show the metadata of the first DICOM file in the train dataset","metadata":{}},{"cell_type":"code","source":"from pydicom import dcmread, read_file\nfrom pydicom.data import get_testdata_file\nPATH = '/kaggle/input/siim-covid19-detection/'\nfile_path = PATH+\"train/00086460a852/9e8302230c91/65761e66de9f.dcm\"\ndicom = read_file(file_path, stop_before_pixels=False)\ndicom","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:08:19.743262Z","iopub.execute_input":"2022-07-29T19:08:19.743666Z","iopub.status.idle":"2022-07-29T19:08:20.348174Z","shell.execute_reply.started":"2022-07-29T19:08:19.743632Z","shell.execute_reply":"2022-07-29T19:08:20.347237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = dicom.pixel_array\ntype(img), img.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:08:20.349496Z","iopub.execute_input":"2022-07-29T19:08:20.350709Z","iopub.status.idle":"2022-07-29T19:08:20.366996Z","shell.execute_reply.started":"2022-07-29T19:08:20.350664Z","shell.execute_reply":"2022-07-29T19:08:20.366200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Display samples","metadata":{}},{"cell_type":"code","source":"# for each class, get N = num samples\ndef get_samples(num):\n    study_df_grp = pd.melt(study_df, id_vars=list(study_df.columns)[:1], value_vars=list(study_df.columns)[1:],\n             var_name='label', value_name='value')\n    study_df_grp = study_df_grp.loc[study_df_grp['value']!=0].reset_index(drop=True)\n    labels = list(study_df_grp['label'].unique())\n    study_samples = {}\n    for label in labels:\n        study_ids = study_df_grp.loc[study_df_grp['label'] == label].sample(num)['id'].tolist() #Get num sample rows from the datafame\n        samples = []\n        for study_id in study_ids:\n            image = {}\n            study_instance_id = study_id.split('_')[0]\n            image_id = image_df.loc[image_df['StudyInstanceUID']==study_instance_id]['id'].values[0].split('_')[0] #Get the image matching study id\n            file_name = [string for string in all_files if image_id in string]\n            image['study_id'] = study_instance_id\n            image['dicom_file'] = file_name[0]\n            #Get the bounding boxes\n            box = None\n            try:\n                box = literal_eval(image_df.loc[image_df['StudyInstanceUID']==study_instance_id]['boxes'].values[0])\n            except ValueError:\n                pass\n            image['boxes'] = box\n            samples.append(image)\n        study_samples[label] = samples\n    return study_samples","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:08:20.368838Z","iopub.execute_input":"2022-07-29T19:08:20.369416Z","iopub.status.idle":"2022-07-29T19:08:20.380179Z","shell.execute_reply.started":"2022-07-29T19:08:20.369366Z","shell.execute_reply":"2022-07-29T19:08:20.379107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_all_class_samples():\n    ''' Input : List of samples \n    '''\n    all_class_samples = []\n    for key in samples:\n        sample_dict = samples[key][0]\n        sample_dict['class'] = key\n        all_class_samples.append(sample_dict)\n    fig1, ax1 = plt.subplots(1,4, figsize=(18, 5), facecolor='w', edgecolor='b')\n    fig1.subplots_adjust(hspace =.3, wspace=0.3)\n    axs = ax1.ravel()\n    for item, ax in zip(all_class_samples, axs):\n        dicom = read_file(item['dicom_file'], stop_before_pixels=False)\n        img = dicom.pixel_array\n        ax.imshow(img, cmap=\"gray\")\n        if 'boxes' in item and item['boxes'] is not None:\n            for box in item['boxes']:             \n                rect = patches.Rectangle((box['x'], box['y']), box['width'], box['height'], linewidth=1.5, edgecolor='r', facecolor='none')\n                ax.add_patch(rect)\n        ax.set_title('{}'.format(item['class']),fontsize = 18)    \n    plt.tight_layout(pad=3.0)\n    plt.subplots_adjust(top=0.91)\n    plt.suptitle('Samples across all classes',fontsize = 20)\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:08:20.381414Z","iopub.execute_input":"2022-07-29T19:08:20.382079Z","iopub.status.idle":"2022-07-29T19:08:20.397458Z","shell.execute_reply.started":"2022-07-29T19:08:20.382042Z","shell.execute_reply":"2022-07-29T19:08:20.395998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samples = get_samples(6) \ndisplay_all_class_samples()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:08:20.399200Z","iopub.execute_input":"2022-07-29T19:08:20.399564Z","iopub.status.idle":"2022-07-29T19:08:27.511983Z","shell.execute_reply.started":"2022-07-29T19:08:20.399521Z","shell.execute_reply":"2022-07-29T19:08:27.510600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"background-color:#fdb913; font-size:120%;  font-family:sans-serif; text-align:center\"><b>Negative for Pneumonia</b></div>\n","metadata":{}},{"cell_type":"code","source":"def display_samples(samples, title, draw_boxes=False):\n    ''' Input : List of samples \n    '''\n    fig1, ax1 = plt.subplots(2,3, figsize=(18, 12), facecolor='w', edgecolor='b')\n    fig1.subplots_adjust(hspace =.3, wspace=0.3)\n    axs = ax1.ravel()\n    for item, ax in zip(samples, axs):\n        dicom = read_file(item['dicom_file'], stop_before_pixels=False)\n        img = dicom.pixel_array\n        ax.imshow(img, cmap=\"gray\")\n        if draw_boxes == True and item['boxes'] is not None:\n            for box in item['boxes']:             \n                rect = patches.Rectangle((box['x'], box['y']), box['width'], box['height'], linewidth=1.5, edgecolor='r', facecolor='none')\n                ax.add_patch(rect)\n        ax.set_title('Study : {}'.format(item['study_id']),fontsize = 18)\n        \n    plt.tight_layout(pad=3.0)\n    plt.subplots_adjust(top=0.91)\n    plt.suptitle(title,fontsize = 20)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:08:27.513496Z","iopub.execute_input":"2022-07-29T19:08:27.514430Z","iopub.status.idle":"2022-07-29T19:08:27.524186Z","shell.execute_reply.started":"2022-07-29T19:08:27.514390Z","shell.execute_reply":"2022-07-29T19:08:27.522981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_samples(samples['Negative for Pneumonia'],'Negative for Pneumonia')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:08:27.525681Z","iopub.execute_input":"2022-07-29T19:08:27.526176Z","iopub.status.idle":"2022-07-29T19:08:36.423542Z","shell.execute_reply.started":"2022-07-29T19:08:27.526138Z","shell.execute_reply":"2022-07-29T19:08:36.422250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns # for histogram displaying\ndef display_histogram(samples, title):\n    ''' Input : List of samples \n    '''\n    fig1, ax1 = plt.subplots(2,3, figsize=(18, 12), facecolor='w', edgecolor='b')\n    fig1.subplots_adjust(hspace =.3, wspace=0.3)\n    axs = ax1.ravel()\n    for item, ax in zip(samples, axs):\n        dicom = read_file(item['dicom_file'], stop_before_pixels=False)\n        img = dicom.pixel_array\n        sub_plot = sns.histplot(img.flatten(), ax=ax)\n        ax.set_title('Study : {}'.format(item['study_id']),fontsize = 18)\n        \n    plt.tight_layout(pad=3.0)\n    plt.subplots_adjust(top=0.91)\n    plt.suptitle('Image Histogram - {}'.format(title),fontsize = 20)\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:08:36.425109Z","iopub.execute_input":"2022-07-29T19:08:36.425483Z","iopub.status.idle":"2022-07-29T19:08:36.922711Z","shell.execute_reply.started":"2022-07-29T19:08:36.425448Z","shell.execute_reply":"2022-07-29T19:08:36.921508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_histogram(samples['Negative for Pneumonia'],'Negative for Pneumonia')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:08:36.924451Z","iopub.execute_input":"2022-07-29T19:08:36.924808Z","iopub.status.idle":"2022-07-29T19:09:12.667654Z","shell.execute_reply.started":"2022-07-29T19:08:36.924774Z","shell.execute_reply":"2022-07-29T19:09:12.666303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"background-color:#fdb913; font-size:120%;  font-family:sans-serif; text-align:center\"><b>Typical Appearance</b></div>\n","metadata":{}},{"cell_type":"code","source":"display_samples(samples['Typical Appearance'],'Typical Appearance', draw_boxes=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:09:12.669193Z","iopub.execute_input":"2022-07-29T19:09:12.669668Z","iopub.status.idle":"2022-07-29T19:09:21.690657Z","shell.execute_reply.started":"2022-07-29T19:09:12.669631Z","shell.execute_reply":"2022-07-29T19:09:21.689835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_histogram(samples['Typical Appearance'],'Typical Appearance')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:09:21.691780Z","iopub.execute_input":"2022-07-29T19:09:21.692778Z","iopub.status.idle":"2022-07-29T19:10:00.149415Z","shell.execute_reply.started":"2022-07-29T19:09:21.692731Z","shell.execute_reply":"2022-07-29T19:10:00.148258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"background-color:#fdb913; font-size:120%;  font-family:sans-serif; text-align:center\"><b>Extract Metadata from DICOM Files</b></div>\n","metadata":{}},{"cell_type":"code","source":"def get_files(file_format):\n    files=[]\n    train_files = []\n    for file in all_files:\n        if file_format in file:\n            files.append(file)\n    return files","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:10:00.150956Z","iopub.execute_input":"2022-07-29T19:10:00.151334Z","iopub.status.idle":"2022-07-29T19:10:00.158294Z","shell.execute_reply.started":"2022-07-29T19:10:00.151299Z","shell.execute_reply":"2022-07-29T19:10:00.156897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_files = get_files('/train/')\ntest_files = get_files('/test/')\nprint(f\"Number of Train file : {len(train_files)}\")\nprint(f\"Number of Test file : {len(test_files)}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:10:00.160453Z","iopub.execute_input":"2022-07-29T19:10:00.161403Z","iopub.status.idle":"2022-07-29T19:10:00.177786Z","shell.execute_reply.started":"2022-07-29T19:10:00.161347Z","shell.execute_reply":"2022-07-29T19:10:00.176366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pydicom.tag import Tag\nfrom tqdm import tqdm\npvt_creator1 = Tag(0x2001, 0x10) #Private Creator 1\npvt_creator2 = Tag(0x0903, 0x10) #Private Creator 1\n\ncolumns = [ 'StudyID',\n 'StudyInstanceUID',\n 'PatientSex', \n 'BitsAllocated',\n 'BitsStored',\n 'Columns',\n 'Rows',\n 'BodyPartExamined', \n 'HighBit', \n 'ImageType',\n 'ImagerPixelSpacing',\n 'InstanceNumber',\n 'Modality',\n 'PatientID',\n 'PatientName',\n 'AccessionNumber',\n 'DeidentificationMethod',\n #'DeidentificationMethodCodeSequence',\n 'PhotometricInterpretation',\n 'PixelRepresentation',\n 'SOPClassUID',\n 'SOPInstanceUID',\n 'SamplesPerPixel',\n 'SeriesInstanceUID',\n 'SeriesNumber',\n 'SpecificCharacterSet',\n 'StudyDate',\n 'StudyTime',\n 'PrivateCreator1',\n 'PrivateCreator2']\n\ndef extract_metadata(columns, files):\n    df = pd.DataFrame(columns=columns)\n    for num, file in tqdm(enumerate(files)):\n        row = {}\n        dicom_img = read_file(file, stop_before_pixels=True)\n        for col in columns:\n            if col not in ['PrivateCreator1', 'PrivateCreator2']:\n                row[col] = dicom_img[col].value\n        try:        \n            row['PrivateCreator1'] = dicom_img.get_item(pvt_creator1).value\n            row['PrivateCreator2'] = dicom_img.get_item(pvt_creator2).value    \n        except AttributeError:\n            pass\n        df = df.append(row,ignore_index=True)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:10:00.179328Z","iopub.execute_input":"2022-07-29T19:10:00.180452Z","iopub.status.idle":"2022-07-29T19:10:00.193197Z","shell.execute_reply.started":"2022-07-29T19:10:00.180408Z","shell.execute_reply":"2022-07-29T19:10:00.191985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = extract_metadata(columns, train_files)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:10:00.194876Z","iopub.execute_input":"2022-07-29T19:10:00.195311Z","iopub.status.idle":"2022-07-29T19:13:41.278495Z","shell.execute_reply.started":"2022-07-29T19:10:00.195260Z","shell.execute_reply":"2022-07-29T19:13:41.276959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = extract_metadata(columns, test_files)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:13:41.280210Z","iopub.execute_input":"2022-07-29T19:13:41.280547Z","iopub.status.idle":"2022-07-29T19:14:23.688760Z","shell.execute_reply.started":"2022-07-29T19:13:41.280517Z","shell.execute_reply":"2022-07-29T19:14:23.687126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Meta data of Train images","metadata":{}},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:23.690415Z","iopub.execute_input":"2022-07-29T19:14:23.690851Z","iopub.status.idle":"2022-07-29T19:14:23.724783Z","shell.execute_reply.started":"2022-07-29T19:14:23.690813Z","shell.execute_reply":"2022-07-29T19:14:23.723587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['PatientSex'].value_counts().reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:23.726678Z","iopub.execute_input":"2022-07-29T19:14:23.727179Z","iopub.status.idle":"2022-07-29T19:14:23.743034Z","shell.execute_reply.started":"2022-07-29T19:14:23.727142Z","shell.execute_reply":"2022-07-29T19:14:23.741254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['BodyPartExamined'].value_counts().reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:23.749582Z","iopub.execute_input":"2022-07-29T19:14:23.749970Z","iopub.status.idle":"2022-07-29T19:14:23.764647Z","shell.execute_reply.started":"2022-07-29T19:14:23.749936Z","shell.execute_reply":"2022-07-29T19:14:23.763747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['BitsStored'] = train_df['BitsStored'].astype(int)\ndef combine_image_size(row):\n    return str(row['Rows']) + ',' + str(row['Columns'])\ntrain_df['ImageSize'] = train_df.apply(lambda x: combine_image_size(x), axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:23.765920Z","iopub.execute_input":"2022-07-29T19:14:23.766656Z","iopub.status.idle":"2022-07-29T19:14:23.873310Z","shell.execute_reply.started":"2022-07-29T19:14:23.766614Z","shell.execute_reply":"2022-07-29T19:14:23.872450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure(go.Scattergl(\n    x=train_df['Rows'], y=train_df['Columns'],\n    name='Image Size',\n    mode='markers',  \n    marker=dict(\n        color='#0092CC',\n    )\n))\nfig.update_layout(xaxis={'title' : 'Rows', \n                             'showgrid':False},\n                      yaxis={'showgrid':False,\n                            'title' : 'Columns'},\n                      showlegend=False,\n                     title = 'Train - image size')\nfig.update_traces(textfont_size=16)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:23.874398Z","iopub.execute_input":"2022-07-29T19:14:23.875236Z","iopub.status.idle":"2022-07-29T19:14:23.994087Z","shell.execute_reply.started":"2022-07-29T19:14:23.875196Z","shell.execute_reply":"2022-07-29T19:14:23.992794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Metadata of test images","metadata":{}},{"cell_type":"code","source":"test_df['PatientSex'].value_counts().reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:23.995818Z","iopub.execute_input":"2022-07-29T19:14:23.996465Z","iopub.status.idle":"2022-07-29T19:14:24.010501Z","shell.execute_reply.started":"2022-07-29T19:14:23.996421Z","shell.execute_reply":"2022-07-29T19:14:24.009079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['BodyPartExamined'].value_counts().reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:24.011754Z","iopub.execute_input":"2022-07-29T19:14:24.012648Z","iopub.status.idle":"2022-07-29T19:14:24.027651Z","shell.execute_reply.started":"2022-07-29T19:14:24.012600Z","shell.execute_reply":"2022-07-29T19:14:24.026780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure(go.Scattergl(\n    x=test_df['Rows'], y=test_df['Columns'],\n    name='Image Size',\n    mode='markers',  \n    marker=dict(\n        color='#DCD427',\n    )\n))\nfig.update_layout(xaxis={'title' : 'Rows', \n                             'showgrid':False},\n                      yaxis={'showgrid':False,\n                            'title' : 'Columns'},\n                      showlegend=False,\n                     title = 'Test - image size')\nfig.update_traces(textfont_size=16)\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:24.029232Z","iopub.execute_input":"2022-07-29T19:14:24.029908Z","iopub.status.idle":"2022-07-29T19:14:24.055659Z","shell.execute_reply.started":"2022-07-29T19:14:24.029858Z","shell.execute_reply":"2022-07-29T19:14:24.054857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['ImageSize'] = train_df['Rows'] * train_df['Columns']\ntest_df['ImageSize'] = test_df['Rows'] * test_df['Columns']\n\nfig = go.Figure()\nfig.add_trace(go.Box(y=train_df['ImageSize'], \n                         name='Train', \n                         jitter=0.5,\n                         whiskerwidth=0.6,\n                         fillcolor='#0092CC',\n                         marker_size=5,\n                         line_width=1))\nfig.add_trace(go.Box(y=test_df['ImageSize'], \n                         name='Test', \n                         jitter=0.5,\n                         whiskerwidth=0.6,\n                         fillcolor='#DCD427',\n                         marker_size=5,\n                         line_width=1))\n\nfig.update_layout(xaxis={'title' : None,'showgrid' :False},\n                  yaxis=dict(title='Image Size (Pixels)',showgrid=False,zeroline=False),\n                 title = 'Image Size IQR')    \nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:24.057339Z","iopub.execute_input":"2022-07-29T19:14:24.057929Z","iopub.status.idle":"2022-07-29T19:14:24.135589Z","shell.execute_reply.started":"2022-07-29T19:14:24.057895Z","shell.execute_reply":"2022-07-29T19:14:24.134404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Convert to excel files","metadata":{}},{"cell_type":"code","source":"train_df['Rows'] = train_df['Rows'].astype(int)\ntrain_df['Columns'] = train_df['Columns'].astype(int)\ntest_df['Rows'] = test_df['Rows'].astype(int)\ntest_df['Columns'] = test_df['Columns'].astype(int)\ntrain_df.to_csv('train_imgs_meta.csv', index=None)\ntest_df.to_csv('test_imgs_meta.csv', index=None)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:24.136680Z","iopub.execute_input":"2022-07-29T19:14:24.137037Z","iopub.status.idle":"2022-07-29T19:14:24.660235Z","shell.execute_reply.started":"2022-07-29T19:14:24.136988Z","shell.execute_reply":"2022-07-29T19:14:24.659116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image Modality\nDICOM file has an attribute called \"Modality\". This describes the technology used to capture the radiography images.\n* CR - Computed Radiography \n* DX - Digital Radiography \n\nQuick reference on these are [here](https://www.jpihealthcare.com/computed-radiography-cr-and-digital-radiography-dr-which-should-you-choose/)\n\nThe images given in this competitions have approximately equal distribution among these two techonologies. Digital Radiography has slightly more samples in both train and test images","metadata":{}},{"cell_type":"code","source":"modality_train = train_df['Modality'].value_counts().reset_index().rename(columns={'index':'Modality','Modality':'Count'})\nmodality_train['Type'] = 'Train'\nmodality_test = test_df['Modality'].value_counts().reset_index().rename(columns={'index':'Modality','Modality':'Count'})\nmodality_test['Type'] = 'Test'\nmodality_df = pd.concat([modality_train, modality_test]).reset_index(drop=True)\nmodality_map = {'DX' : 'Digital Radiography',\n'CR' : 'Computed Radiography'}\nmodality_df['Desc'] = modality_df['Modality'].apply(lambda x : modality_map[x])\nbar_colors = ['#FFA48E', '#4ACFAC']\nmodalities = list(modality_df.Desc.unique())\nfig = go.Figure()\nfor modality, color in zip(modalities, bar_colors):\n    df = modality_df.loc[modality_df['Desc'] == modality]\n    fig.add_trace(go.Bar(\n        x=df['Type'],\n        y=df['Count'],\n        name=modality,\n        marker_color = color,\n        textposition='auto',\n        marker_line_width=2.5, opacity=0.8,\n        marker_line_color = color        \n    ))\nfig.update_layout(barmode='stack',\n                  xaxis=dict(\n                                tickmode = 'array',\n                                 title=None,\n                                 showgrid=False,\n                                 zeroline=False,\n                            ),\n                      yaxis=dict(title='Count',\n                                 showgrid=False,\n                                 zeroline=False,\n                                ), \n                      title = dict(text = 'Modality of images',\n                                   xref = 'paper',\n                                  ),\n                      bargap=0.15, \n                    bargroupgap=0.1,\n                     )     \nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:24.661856Z","iopub.execute_input":"2022-07-29T19:14:24.662444Z","iopub.status.idle":"2022-07-29T19:14:24.747721Z","shell.execute_reply.started":"2022-07-29T19:14:24.662388Z","shell.execute_reply":"2022-07-29T19:14:24.746429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_modality_sample(df, modality):\n    np.random.seed(0)\n    studyinstance = df.loc[(df['Modality'] == modality) & (df['PatientSex'] == 'M')].sample(n=1)['StudyInstanceUID'].values[0]\n    return studyinstance\n    #files = [file for file in all_files if studyinstance in file]\n    #return files[0]\n\ndef get_modality_samples():\n    modality_samples = []\n    modality_samples.append(get_modality_sample(train_df,'CR'))\n    modality_samples.append(get_modality_sample(train_df,'DX'))\n    study_samples = []\n    for study_instance_id in modality_samples:\n        samples = []\n        image = {}\n        image_id = image_df.loc[image_df['StudyInstanceUID']==study_instance_id]['id'].values[0].split('_')[0] #Get the image matching study id\n        file_name = [string for string in all_files if image_id in string]\n        image['study_id'] = study_instance_id\n        image['dicom_file'] = file_name[0]\n        #Get the bounding boxes\n        box = None\n        try:\n            box = literal_eval(image_df.loc[image_df['StudyInstanceUID']==study_instance_id]['boxes'].values[0])\n        except ValueError:\n            pass\n        image['boxes'] = box\n        study_samples.append(image)\n        #study_samples[study_instance_id] = samples\n    return study_samples\n\ndef display_modality_samples(samples, title, draw_boxes=False):\n    ''' Input : List of samples \n    '''\n    fig1, ax1 = plt.subplots(1,2, figsize=(18, 12), facecolor='w', edgecolor='b')\n    fig1.subplots_adjust(hspace =.3, wspace=0.3)\n    axs = ax1.ravel()\n    for item, ax in zip(samples, axs):\n        dicom = read_file(item['dicom_file'], stop_before_pixels=False)\n        img = dicom.pixel_array\n        ax.imshow(img, cmap=\"gray\")\n        if draw_boxes == True and item['boxes'] is not None:\n            for box in item['boxes']:             \n                rect = patches.Rectangle((box['x'], box['y']), box['width'], box['height'], linewidth=1.5, edgecolor='r', facecolor='none')\n                ax.add_patch(rect)\n        ax.set_title('Study : {}'.format(item['study_id']),fontsize = 18)\n        \n    plt.tight_layout(pad=1.0)\n#    plt.subplots_adjust(top=0.99)\n    plt.suptitle(title,fontsize = 20)\n    plt.show()\n\ndisplay_modality_samples(get_modality_samples(), 'Computed Radiography vs Digital Radiography', True)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:24.749917Z","iopub.execute_input":"2022-07-29T19:14:24.750786Z","iopub.status.idle":"2022-07-29T19:14:27.799567Z","shell.execute_reply.started":"2022-07-29T19:14:24.750731Z","shell.execute_reply":"2022-07-29T19:14:27.798222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Photometric Interpretation\n\nIn DICOM, monochrome images are given a photometric interpretation of 'MONOCHROME1' (low values=bright, high values=dim) or 'MONOCHROME2' (low values=dark, high values=bright).","metadata":{}},{"cell_type":"code","source":"pi_train = train_df['PhotometricInterpretation'].value_counts().reset_index().rename(columns={'index':'PhotometricInterpretation','PhotometricInterpretation':'Count'})\npi_train['Type'] = 'Train'\npi_test = test_df['PhotometricInterpretation'].value_counts().reset_index().rename(columns={'index':'PhotometricInterpretation','PhotometricInterpretation':'Count'})\npi_test['Type'] = 'Test'\npi_df = pd.concat([pi_train, pi_test]).reset_index(drop=True)\nbar_colors = ['#FFA48E', '#4ACFAC']\npi_values = list(pi_df.PhotometricInterpretation.unique())\nfig = go.Figure()\nfor pi, color in zip(pi_values, bar_colors):\n    df = pi_df.loc[pi_df['PhotometricInterpretation'] == pi]\n    fig.add_trace(go.Bar(\n        x=df['Type'],\n        y=df['Count'],\n        name=pi,\n        marker_color = color,\n        #text = df['passenger_count_new'],\n        #texttemplate='%{text:.2s}', \n        textposition='auto',\n        marker_line_width=2.5, opacity=0.8,\n        marker_line_color = color        \n    ))\nfig.update_layout(barmode='stack',\n                  xaxis=dict(\n                                tickmode = 'array',\n                                 title=None,\n                                 showgrid=False,\n                                 zeroline=False,\n                            ),\n                      yaxis=dict(title='Count',\n                                 showgrid=False,\n                                 zeroline=False,\n                                ), \n                      title = dict(text = 'Photometric Interpretation of images',\n                                   xref = 'paper',\n                                  ),\n                      bargap=0.15, \n                    bargroupgap=0.1,\n                     )     \nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:27.801631Z","iopub.execute_input":"2022-07-29T19:14:27.802065Z","iopub.status.idle":"2022-07-29T19:14:27.844859Z","shell.execute_reply.started":"2022-07-29T19:14:27.802009Z","shell.execute_reply":"2022-07-29T19:14:27.843597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_photometricInterpretation_sample(df, modality):\n    np.random.seed(0)\n    studyinstance = df.loc[(df['PhotometricInterpretation'] == modality) & (df['PatientSex'] == 'M')].sample(n=1)['StudyInstanceUID'].values[0]\n    return studyinstance\n    #files = [file for file in all_files if studyinstance in file]\n    #return files[0]\n\ndef get_PhotometricInterpretation_samples():\n    modality_samples = []\n    modality_samples.append(get_photometricInterpretation_sample(train_df,'MONOCHROME2'))\n    modality_samples.append(get_photometricInterpretation_sample(train_df,'MONOCHROME1'))\n    study_samples = []\n    for study_instance_id in modality_samples:\n        samples = []\n        image = {}\n        image_id = image_df.loc[image_df['StudyInstanceUID']==study_instance_id]['id'].values[0].split('_')[0] #Get the image matching study id\n        file_name = [string for string in all_files if image_id in string]\n        image['study_id'] = study_instance_id\n        image['dicom_file'] = file_name[0]\n        #Get the bounding boxes\n        box = None\n        try:\n            box = literal_eval(image_df.loc[image_df['StudyInstanceUID']==study_instance_id]['boxes'].values[0])\n        except ValueError:\n            pass\n        image['boxes'] = box\n        study_samples.append(image)\n        #study_samples[study_instance_id] = samples\n    return study_samples\n\ndef display_photometricInterpretation_samples(samples, title, draw_boxes=False):\n    ''' Input : List of samples \n    '''\n    fig1, ax1 = plt.subplots(1,2, figsize=(18, 12), facecolor='w', edgecolor='b')\n    fig1.subplots_adjust(hspace =.3, wspace=0.3)\n    axs = ax1.ravel()\n    for item, ax in zip(samples, axs):\n        dicom = read_file(item['dicom_file'], stop_before_pixels=False)\n        img = dicom.pixel_array\n        ax.imshow(img, cmap=\"gray\")\n        if draw_boxes == True and item['boxes'] is not None:\n            for box in item['boxes']:             \n                rect = patches.Rectangle((box['x'], box['y']), box['width'], box['height'], linewidth=1.5, edgecolor='r', facecolor='none')\n                ax.add_patch(rect)\n        ax.set_title('Study : {}'.format(item['study_id']),fontsize = 18)\n        \n    plt.tight_layout(pad=1.0)\n#    plt.subplots_adjust(top=0.99)\n    plt.suptitle(title,fontsize = 20)\n    plt.show()\n\ndisplay_photometricInterpretation_samples(get_PhotometricInterpretation_samples(), 'MONOCHROME2 vs MONOCHROME1', True)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:27.846537Z","iopub.execute_input":"2022-07-29T19:14:27.847219Z","iopub.status.idle":"2022-07-29T19:14:30.498934Z","shell.execute_reply.started":"2022-07-29T19:14:27.847180Z","shell.execute_reply":"2022-07-29T19:14:30.497885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"background-color:#fdb913; font-size:120%;  font-family:sans-serif; text-align:center\"><b>Explore bounding boxes in training data</b></div>\n","metadata":{}},{"cell_type":"code","source":"img_df = image_df.copy()\n#from ast import literal_eval\n\ndef count_bb(x):\n    count=0\n    bb_lst = literal_eval(x)\n    for box in bb_lst:\n        count+=1\n    return count\n\ndef bb_size(x):\n    sizes= []\n    bb_lst = literal_eval(x)\n    for box in bb_lst:\n        w=box['width']\n        h=box['height']\n        sizes.append(w*h)\n    return sizes\n    \nimg_df['boxes'] = img_df['boxes'].fillna('[]')\nimg_df['bb_count'] = img_df['boxes'].apply(lambda x: count_bb(x))\nimg_df['bb_size'] = img_df['boxes'].apply(lambda x: bb_size(x))\n\nbbs_lst = []\nfor index, row in img_df.iterrows():\n    #lst = literal_eval(row['bb_size'])\n    bbs_lst.extend(row['bb_size'])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:30.500147Z","iopub.execute_input":"2022-07-29T19:14:30.501303Z","iopub.status.idle":"2022-07-29T19:14:31.081545Z","shell.execute_reply.started":"2022-07-29T19:14:30.501258Z","shell.execute_reply":"2022-07-29T19:14:31.080287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig1, ax1 = plt.subplots(1,1, figsize=(8, 6), facecolor='w', edgecolor='g')\nfig1.subplots_adjust(hspace =.3, wspace=0.3)\nsub_plot = sns.histplot(img_df['bb_count'], ax=ax1)\nax1.set_title('Number of bounding boxes in train images',fontsize = 18)\nax1.set_xlabel('Number of bounding boxes', fontsize=12)\nax1.set_ylabel('Count', fontsize=12)\nplt.tight_layout(pad=3.0)\nplt.subplots_adjust(top=0.91)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:31.083483Z","iopub.execute_input":"2022-07-29T19:14:31.083833Z","iopub.status.idle":"2022-07-29T19:14:31.416726Z","shell.execute_reply.started":"2022-07-29T19:14:31.083801Z","shell.execute_reply":"2022-07-29T19:14:31.415352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig1, ax1 = plt.subplots(1,1, figsize=(8, 6), facecolor='w', edgecolor='g')\nfig1.subplots_adjust(hspace =.3, wspace=0.3)\nsub_plot = sns.histplot(bbs_lst,bins=100, ax=ax1)\nax1.set_title('Size of bounding boxes in train images',fontsize = 18)\nax1.set_xlabel('Size', fontsize=12)\nax1.ticklabel_format(style='plain')\n#start, end = ax1.get_xlim()\n#ax1.xaxis.set_ticks(np.arange(start, end, 500000))\nax1.set_ylabel('Count', fontsize=12)\nplt.tight_layout(pad=3.0)\nplt.subplots_adjust(top=0.91)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:31.418289Z","iopub.execute_input":"2022-07-29T19:14:31.418654Z","iopub.status.idle":"2022-07-29T19:14:31.858012Z","shell.execute_reply.started":"2022-07-29T19:14:31.418623Z","shell.execute_reply":"2022-07-29T19:14:31.856781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_bb_df = pd.DataFrame(columns=['image_id','study_ins_id','bb_count', 'bb_size'])\n\nfor index, rec in img_df.iterrows():\n    row= {}\n    row['image_id'] = rec['id'].split('_')[0]\n    row['study_ins_id'] = rec['StudyInstanceUID']\n    row['bb_count'] = rec['bb_count']    \n    #bb_lst = literal_eval(rec['bb_size'])\n    for bb in rec['bb_size']:\n        row['bb_size'] = bb\n        img_bb_df = img_bb_df.append(row, ignore_index=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:31.862015Z","iopub.execute_input":"2022-07-29T19:14:31.862476Z","iopub.status.idle":"2022-07-29T19:14:52.511796Z","shell.execute_reply.started":"2022-07-29T19:14:31.862437Z","shell.execute_reply":"2022-07-29T19:14:52.510769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_bb_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:52.513110Z","iopub.execute_input":"2022-07-29T19:14:52.513702Z","iopub.status.idle":"2022-07-29T19:14:52.527243Z","shell.execute_reply.started":"2022-07-29T19:14:52.513666Z","shell.execute_reply":"2022-07-29T19:14:52.526073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure()\n#Sky Blue, Hyper Red, Sulphur Yellow, Green\nbb_colors = ['#0092CC','#FF3333','#DCD427','#779933']\n\nfor bb_count, color in zip([1,2,3,4], bb_colors):\n    df = img_bb_df.loc[img_bb_df['bb_count']==bb_count].reset_index(drop=True)\n    fig.add_trace(go.Box(y=df['bb_size'], \n                             name='BB count : {}'.format(bb_count), \n                             jitter=0.5,\n                             whiskerwidth=0.6,\n                             fillcolor=color,\n                             marker_size=5,\n                             line_width=1))\nfig.update_layout(xaxis={'title' : None,'showgrid' :False},\n                  yaxis=dict(title='Bounding Box Size (Pixels)',showgrid=False,zeroline=False),\n                 title = 'Train images - Bounding Box Size IQR')    \nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:52.528873Z","iopub.execute_input":"2022-07-29T19:14:52.529245Z","iopub.status.idle":"2022-07-29T19:14:52.561722Z","shell.execute_reply.started":"2022-07-29T19:14:52.529212Z","shell.execute_reply":"2022-07-29T19:14:52.560565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add labels to the image_bb dataframe\nsg_df = pd.melt(study_df, id_vars=list(study_df.columns)[:1], value_vars=list(study_df.columns)[1:],\n             var_name='label', value_name='value')\nsg_df = sg_df.loc[sg_df['value']!=0]\nsg_df['study_id'] = sg_df['id'].apply(lambda x : str(x).split('_')[0])\nlbl_map = {'Negative for Pneumonia' : 'Negative',\n 'Typical Appearance' : 'Typical',\n 'Indeterminate Appearance' : 'Indeterminate', \n 'Atypical Appearance' : 'Atypical'}\nsg_df['lbl'] = sg_df['label'].apply(lambda x: lbl_map[x])\nslbl_map = dict(zip(sg_df.study_id, sg_df.lbl)) \nimg_bb_df['label'] = img_bb_df['study_ins_id'].apply(lambda x: slbl_map[x])\n#slbl_map","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:52.563324Z","iopub.execute_input":"2022-07-29T19:14:52.564187Z","iopub.status.idle":"2022-07-29T19:14:52.594257Z","shell.execute_reply.started":"2022-07-29T19:14:52.564135Z","shell.execute_reply":"2022-07-29T19:14:52.593210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_bb_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:52.595552Z","iopub.execute_input":"2022-07-29T19:14:52.596055Z","iopub.status.idle":"2022-07-29T19:14:52.609245Z","shell.execute_reply.started":"2022-07-29T19:14:52.595995Z","shell.execute_reply":"2022-07-29T19:14:52.608004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"large_bb = img_bb_df.sort_values('bb_size', ascending=False).head(20).reset_index(drop=True)\ndef find_file(img_id):\n    imgs = [file for file in all_files if img_id in file]\n    return imgs[0]\n\nlarge_bb['file'] = large_bb['image_id'].apply(lambda x : find_file(x))\nlarge_bb = large_bb[['image_id','study_ins_id','label','file']]\nlg_bb_dict = large_bb.set_index('image_id').T.to_dict('list')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:52.611042Z","iopub.execute_input":"2022-07-29T19:14:52.611671Z","iopub.status.idle":"2022-07-29T19:14:52.642493Z","shell.execute_reply.started":"2022-07-29T19:14:52.611620Z","shell.execute_reply":"2022-07-29T19:14:52.641623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.patheffects as path_effects\nCOLOR='white'\ndef show_image(img, figsize=None, ax=None, cmap=\"gray\"):\n    if not ax: \n        fig, ax = plt.subplots(figsize=figsize)\n    ax.imshow(img, cmap=cmap)\n    ax.get_xaxis().set_visible(False)\n    ax.get_yaxis().set_visible(False)\n\ndef draw_outline(o, lw):\n    o.set_path_effects([patheffects.Stroke(linewidth=lw, foreground='black'), patheffects.Normal()])\n\n    \ndef draw_text(ax, x,y, txt, fontsize=14):\n    text = ax.text(x,y, \n                   txt,\n                  verticalalignment = 'top',\n                  color=COLOR,\n                  fontsize=fontsize,\n                  weight='bold')\n\ndef draw_bb(ax, bb, label):\n    for b in bb:\n        patch = ax.add_patch(patches.Rectangle((b['x'],\n                                           b['y']),\n                                           b['width'],\n                                           b['height'],\n                        fill=False,\n                         edgecolor='red',\n                         lw=2))\n        #draw_text(ax, b['x'], b['y'], label)\n\n\ndef get_image(file):\n    dicom = read_file(file, stop_before_pixels=False)\n    return dicom.pixel_array\n\ndef get_bb(img_id):\n    bb = literal_eval(image_df.loc[image_df['id'] == \"{}_image\".format(img_id)]['boxes'].values[0])\n    return bb\n        \ndef show_bb(samples, title, rows=3, cols=4):\n    ''' Input : Dict of samples\n    '''\n    fig, axs = plt.subplots(rows, cols, figsize=(18, 12), facecolor='w', edgecolor='b')\n    #fig.subplots_adjust(hspace =.3, wspace=0.3)\n    for i, ax in enumerate(axs.ravel()):\n        img_id = list(samples.keys())[i]\n        data = list(samples.values())[i]\n        img = get_image(data[2])\n        bb = get_bb(img_id)\n        show_image(img, ax=ax)\n        draw_bb(ax, bb, data[1])\n        ax.set_title('{}, {}'.format(data[0],data[1]),fontsize = 14)\n\n    plt.tight_layout(pad=3.0)\n    plt.subplots_adjust(top=0.91)\n    plt.suptitle(title,fontsize = 20)\n    plt.show()\nshow_bb(lg_bb_dict, 'Large Bounding Boxes')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:14:52.644007Z","iopub.execute_input":"2022-07-29T19:14:52.644554Z","iopub.status.idle":"2022-07-29T19:15:16.474011Z","shell.execute_reply.started":"2022-07-29T19:14:52.644519Z","shell.execute_reply":"2022-07-29T19:15:16.472198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"small_bb = img_bb_df.sort_values('bb_size', ascending=True).head(20).reset_index(drop=True)\nsmall_bb['file'] = small_bb['image_id'].apply(lambda x : find_file(x))\nsmall_bb = small_bb[['image_id','study_ins_id','label','file']]\nsm_bb_dict = small_bb.set_index('image_id').T.to_dict('list')\nshow_bb(sm_bb_dict, 'Small Bounding Boxes')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T19:15:16.475744Z","iopub.execute_input":"2022-07-29T19:15:16.476184Z","iopub.status.idle":"2022-07-29T19:15:26.303986Z","shell.execute_reply.started":"2022-07-29T19:15:16.476149Z","shell.execute_reply":"2022-07-29T19:15:26.302713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Search all boxes and store into boxes.csv file","metadata":{}},{"cell_type":"code","source":"boxes = []\nfor i, row in image_df.iterrows():\n    value = row['boxes']\n    if isinstance(value, float): # nan\n        #print('x', end='')\n        continue\n    values = eval(value)\n    #print('.', end='')\n    for val in values:\n        box = {\n            'ImageID': row['id'],\n            'StudyUID': row['StudyInstanceUID'],\n            'xmin': val['x'],\n            'ymin': val['y'],\n            'xmax': val['x'] + val['width'],\n            'ymax': val['y'] + val['height'],\n            'width': val['width'],\n            'height': val['height'],\n        }\n        boxes.append(box)\nboxes = pd.DataFrame(boxes)\nprint(len(boxes))\nboxes.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-29T20:05:46.935297Z","iopub.execute_input":"2022-07-29T20:05:46.936527Z","iopub.status.idle":"2022-07-29T20:05:47.617717Z","shell.execute_reply.started":"2022-07-29T20:05:46.936486Z","shell.execute_reply":"2022-07-29T20:05:47.616555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"boxes['ImageID'] = boxes.apply(lambda row: row.ImageID.split('_')[0], axis=1)\nboxes.to_csv('boxes.csv', index= False)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T20:06:54.711766Z","iopub.execute_input":"2022-07-29T20:06:54.712350Z","iopub.status.idle":"2022-07-29T20:06:54.884210Z","shell.execute_reply.started":"2022-07-29T20:06:54.712295Z","shell.execute_reply":"2022-07-29T20:06:54.883096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-07-29T20:07:29.287515Z","iopub.execute_input":"2022-07-29T20:07:29.288162Z","iopub.status.idle":"2022-07-29T20:07:29.383772Z","shell.execute_reply.started":"2022-07-29T20:07:29.288108Z","shell.execute_reply":"2022-07-29T20:07:29.382776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}