{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!conda install -c conda-forge gdcm -y","metadata":{"execution":{"iopub.status.busy":"2021-07-17T08:46:46.880776Z","iopub.execute_input":"2021-07-17T08:46:46.881092Z","iopub.status.idle":"2021-07-17T08:47:21.644773Z","shell.execute_reply.started":"2021-07-17T08:46:46.881056Z","shell.execute_reply":"2021-07-17T08:47:21.643646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nimport pydicom\nimport os\nimport numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2021-07-17T11:03:38.520166Z","iopub.execute_input":"2021-07-17T11:03:38.520590Z","iopub.status.idle":"2021-07-17T11:03:38.525445Z","shell.execute_reply.started":"2021-07-17T11:03:38.520552Z","shell.execute_reply":"2021-07-17T11:03:38.524450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pydicom.pixel_data_handlers.util import apply_voi_lut, apply_modality_lut\nfrom pydicom import dcmread\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n# functin to read dicom given the path , considers modality_lut and corrections w.r.t monochromes\ndef read_dicom( filepath ,modality_lut=True, fix_monochrome=True):\n    dcm = pydicom.read_file(filepath)\n    img = dcm.pixel_array\n    if modality_lut == True:\n        img = apply_modality_lut(dcm.pixel_array, dcm)\n    max_img = np.max(img)\n    min_img = np.min(img)\n    if fix_monochrome == True and dcm.PhotometricInterpretation=='MONOCHROME1':\n        img = max_img - img\n\n    img = (img - np.min(img))/(max_img - min_img)\n    img = (img * 255).astype(np.uint8)\n\n    return img ","metadata":{"execution":{"iopub.status.busy":"2021-07-17T11:03:59.878872Z","iopub.execute_input":"2021-07-17T11:03:59.879230Z","iopub.status.idle":"2021-07-17T11:03:59.890139Z","shell.execute_reply.started":"2021-07-17T11:03:59.879200Z","shell.execute_reply":"2021-07-17T11:03:59.889320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dirs = glob.glob(os.path.join('/kaggle/input/siim-covid19-detection','*'))\ndirs","metadata":{"execution":{"iopub.status.busy":"2021-07-17T11:04:11.651835Z","iopub.execute_input":"2021-07-17T11:04:11.652380Z","iopub.status.idle":"2021-07-17T11:04:11.665930Z","shell.execute_reply.started":"2021-07-17T11:04:11.652331Z","shell.execute_reply":"2021-07-17T11:04:11.665047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_study = pd.read_csv('/kaggle/input/siim-covid19-detection/train_study_level.csv')\nprint(train_study.columns)\ndf_sub = train_study[train_study['Negative for Pneumonia']==1]\nprint('length of negative for pnemonia',len(df_sub))\ndf_sub = train_study[train_study['Typical Appearance']==1]\nprint('length of Typical Appearance',len(df_sub))\ndf_sub = train_study[train_study['Indeterminate Appearance']==1]\nprint('length of Indeterminate Appearance',len(df_sub))\ndf_sub = train_study[train_study['Atypical Appearance']==1]\nprint('length of Atypical Appearance',len(df_sub))\n\n# train_study['id']","metadata":{"execution":{"iopub.status.busy":"2021-07-17T08:48:19.228636Z","iopub.execute_input":"2021-07-17T08:48:19.228998Z","iopub.status.idle":"2021-07-17T08:48:19.272270Z","shell.execute_reply.started":"2021-07-17T08:48:19.228967Z","shell.execute_reply":"2021-07-17T08:48:19.271275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Enable the line pertaining to    \"Atypical Appearance\", \"Inderminate Appearance\" , \"Typical Appearance\" , \"Negative for Pnuemonia\"","metadata":{}},{"cell_type":"code","source":"df_sub = train_study[train_study['Atypical Appearance']==1]\n# df_sub = train_study[train_study['Indeterminate Appearance']==1]\n# df_sub = train_study[train_study['Typical Appearance']==1]\n# df_sub = train_study[train_study['Negative for Pneumonia']==1]\nstudy_id = [ item.split('_')[0] for item in  df_sub['id'].values]\nprint(len(df_sub))\ndf_sub.head()\n# study_id_ap\n","metadata":{"execution":{"iopub.status.busy":"2021-07-17T11:26:08.768799Z","iopub.execute_input":"2021-07-17T11:26:08.769343Z","iopub.status.idle":"2021-07-17T11:26:08.785781Z","shell.execute_reply.started":"2021-07-17T11:26:08.769310Z","shell.execute_reply":"2021-07-17T11:26:08.784770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### train_study is the dataframe from the Study level. train_img is the dataframe from the image_level. we first select the study pertaining to one of the 4 category, then find the images belongs to all those studies of the selected category.","metadata":{}},{"cell_type":"code","source":"train_img = pd.read_csv('/kaggle/input/siim-covid19-detection/train_image_level.csv')\nfiltr  = [val in study_id   for val in train_img['StudyInstanceUID'].values]\ntrain_img = train_img[filtr]","metadata":{"execution":{"iopub.status.busy":"2021-07-17T11:26:10.686354Z","iopub.execute_input":"2021-07-17T11:26:10.686702Z","iopub.status.idle":"2021-07-17T11:26:10.800399Z","shell.execute_reply.started":"2021-07-17T11:26:10.686674Z","shell.execute_reply":"2021-07-17T11:26:10.799419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import ast\nimport json \nimport matplotlib.patches as patches\n\ndef convert_to_list(value):\n    try:\n        return ast.literal_eval(value)\n    except:\n        return []\n\ntrain_img['boxes'] = train_img['boxes'].apply(convert_to_list)\n","metadata":{"execution":{"iopub.status.busy":"2021-07-17T11:26:11.737784Z","iopub.execute_input":"2021-07-17T11:26:11.738134Z","iopub.status.idle":"2021-07-17T11:26:11.759898Z","shell.execute_reply.started":"2021-07-17T11:26:11.738103Z","shell.execute_reply":"2021-07-17T11:26:11.758891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_bounding_boxes(boxes):\n    boxes_list = []\n    for box in boxes:\n        boxes_list.append([box['x'], box['y'], box['width'], box['height']])\n    return boxes_list\n        \ndef read_image_bb_index(train_img, index):\n    df = train_img.iloc[index]\n    study_id = df['StudyInstanceUID']\n    image_id = df['id'].split('_')[0]\n    path1 = '/kaggle/input/siim-covid19-detection/train/'+study_id\n    path2 = image_id+'*dcm'\n    imagepath = glob.glob(os.path.join(path1,'*', path2))\n    img = read_dicom(imagepath[0])\n    \n    boxes_list = process_bounding_boxes(df['boxes'])\n    return img, boxes_list\n    \n    \ndef plot_examples(train_img, sz=16):\n#     fig, ax = plt.subplots(nrows=3, ncols=4, figsize=(16,12), gridspec_kw={'height_ratios': [1, 1, 1]})\n    fig, ax = plt.subplots(nrows=4, ncols=4, figsize=(16,16))\n    random_selection = np.random.choice(len(train_img), size=sz, replace=False)\n    for plt_idx,idx in enumerate(random_selection):\n        img, boxes_list = read_image_bb_index(train_img, idx)\n        row = int(plt_idx / 4)\n        col = int(plt_idx % 4)\n        ax[row][col].imshow(img, cmap='gray')\n\n        for bb in boxes_list:\n            rect = patches.Rectangle((bb[0], bb[1]), bb[2], bb[3], linewidth=1, \n                                     edgecolor='r', facecolor='none')\n            ax[row][col].add_patch(rect)\n    plt.subplots_adjust(hspace=0.1, wspace=0.2)\n\n    plt.show()\n        \n    ","metadata":{"execution":{"iopub.status.busy":"2021-07-17T11:01:52.848625Z","iopub.execute_input":"2021-07-17T11:01:52.848960Z","iopub.status.idle":"2021-07-17T11:01:52.860748Z","shell.execute_reply.started":"2021-07-17T11:01:52.848931Z","shell.execute_reply":"2021-07-17T11:01:52.859886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_examples(train_img)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T11:10:58.066363Z","iopub.execute_input":"2021-07-17T11:10:58.066767Z","iopub.status.idle":"2021-07-17T11:11:16.485383Z","shell.execute_reply.started":"2021-07-17T11:10:58.066724Z","shell.execute_reply":"2021-07-17T11:11:16.484403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def find_empty(item_list):\n#     return len(item_list)\n# num_bb = [ find_empty(item) for item in train_img['boxes'] ]\n# np.any([ item==0  for item in num_bb])\n# filtr = [ item!=0  for item in num_bb]\n# # train_img_ip0 = train_img[filtr]\n# # train_img_ta0 = train_img[filtr]\n# train_img_na0 = train_img[filtr]\n\n# # print(len(num_bb))\n# # num_bb","metadata":{},"execution_count":null,"outputs":[]}]}