{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## High-level info","metadata":{}},{"cell_type":"markdown","source":"* Ideal image resolution: https://pubs.rsna.org/doi/10.1148/ryai.2019190177\n    * Plateau's around 300 x 300 -- use higher resolution (512 x 512) for nodule pathologies\n* Outputs:\n    * We are going to output the label of the study\n        * One of: Atypical, Typical, Undeterminate, Negative\n    * Also the coordinates of the bounding box if possible\n    ","metadata":{}},{"cell_type":"markdown","source":"## What do we know about DICOM files\nIt seems to contain not only the image, but the image metadata:\n* Image type\n* Study date and time\n* Patient UID - name, ID, sex, \n* Body part examined\n* samples per pixel\n* Bit information (high bit)\n* Image size (number of rows and columns)\n* Number of elements (pixel data)","metadata":{}},{"cell_type":"markdown","source":"Information about Pydicom (full user guide is [here](https://pydicom.github.io/pydicom/stable/old/pydicom_user_guide.html)):\n\nImportant methods and properties:\n***FileDataset* object**\n* `Dataset.pixel_array` -- converts pydicom object into pixel array\n* `Dataset.PatientName` -- returns the value stored in the file_meta file with the variable name `PatientName` (works for other variables)\n* `Dataset.Rows` & `Dataset.Columns` -- returns the number of rows and columns, respectively","metadata":{}},{"cell_type":"markdown","source":"### Implementing YOLOV5 Preprocessing Pipeline","metadata":{"execution":{"iopub.status.busy":"2021-07-29T19:13:49.534389Z","iopub.execute_input":"2021-07-29T19:13:49.534771Z","iopub.status.idle":"2021-07-29T19:13:49.63034Z","shell.execute_reply.started":"2021-07-29T19:13:49.534722Z","shell.execute_reply":"2021-07-29T19:13:49.628819Z"}}},{"cell_type":"code","source":"from sklearn import model_selection\nimport cv2\nfrom matplotlib import patches\nimport ast\nfrom tqdm import tqdm\nimport os\nimport pydicom as dicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport numpy as np\nfrom pathlib import Path\nimport pandas as pd\nfrom fastai.medical.imaging import *\nfrom PIL import Image\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2021-08-09T05:30:34.967645Z","iopub.execute_input":"2021-08-09T05:30:34.968105Z","iopub.status.idle":"2021-08-09T05:30:38.961441Z","shell.execute_reply.started":"2021-08-09T05:30:34.967973Z","shell.execute_reply":"2021-08-09T05:30:38.960329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Run if you don't have file structure\n!rm -r labels images\ndirs = ['images/', 'labels/', 'images/train', 'images/validation', 'labels/train', 'labels/validation']\nfor dir in dirs:\n    os.makedirs(dir, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T06:23:57.297243Z","iopub.execute_input":"2021-08-09T06:23:57.297712Z","iopub.status.idle":"2021-08-09T06:23:58.224012Z","shell.execute_reply.started":"2021-08-09T06:23:57.297682Z","shell.execute_reply":"2021-08-09T06:23:58.223053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/siim-covid19-detection/train_image_level.csv')","metadata":{"execution":{"iopub.status.busy":"2021-08-09T05:30:39.715590Z","iopub.execute_input":"2021-08-09T05:30:39.715920Z","iopub.status.idle":"2021-08-09T05:30:39.778953Z","shell.execute_reply.started":"2021-08-09T05:30:39.715887Z","shell.execute_reply":"2021-08-09T05:30:39.777806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2021-08-09T05:30:39.780491Z","iopub.execute_input":"2021-08-09T05:30:39.780791Z","iopub.status.idle":"2021-08-09T05:30:39.814909Z","shell.execute_reply.started":"2021-08-09T05:30:39.780762Z","shell.execute_reply":"2021-08-09T05:30:39.814173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_bbox_format(row):\n    if row is np.nan: return [[0, 0, 1, 1]]\n    bbox_list = ast.literal_eval(row)\n    return [list(x.values()) for x in bbox_list]\n            \ntrain = pd.read_csv('../input/siim-covid19-detection/train_image_level.csv')\ntrain['bbox'] = train.boxes.apply(get_bbox_format)\ntrain['image'] = train['id'].str.extract('(.*)_image')\nimg_paths = get_dicom_files('../input/siim-covid19-detection/train')\nfor path in tqdm(img_paths, total=len(img_paths)):\n    study = path.parent.parent.name\n    series = path.parent.name\n    image = path.stem\n    train.loc[train.image == image, ['study', 'series', 'path']] = [study, series, path]","metadata":{"execution":{"iopub.status.busy":"2021-08-09T05:41:25.196971Z","iopub.execute_input":"2021-08-09T05:41:25.197377Z","iopub.status.idle":"2021-08-09T05:41:52.423668Z","shell.execute_reply.started":"2021-08-09T05:41:25.197341Z","shell.execute_reply":"2021-08-09T05:41:52.422668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.sample(5)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T05:41:58.187364Z","iopub.execute_input":"2021-08-09T05:41:58.187752Z","iopub.status.idle":"2021-08-09T05:41:58.217385Z","shell.execute_reply.started":"2021-08-09T05:41:58.187717Z","shell.execute_reply":"2021-08-09T05:41:58.216286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_df = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv')\nconditions = [\n    study_df['Negative for Pneumonia'] == 1,\n    study_df['Typical Appearance'] == 1,\n    study_df['Indeterminate Appearance'] == 1,\n    study_df['Atypical Appearance'] == 1,\n]\n\nchoices_str = [\n    \"negative\", \"typical\", \"indeterminate\", \"atypical\"\n]\n\nchoices_int = [\n    0, 1, 2, 3\n]\n\nstudy_df['label'] = np.select(conditions, choices_str, None)\nstudy_df['label_int'] = np.select(conditions, choices_int, None)\nstudy_df['study'] = study_df['id'].str.extract('(.*)_study')","metadata":{"execution":{"iopub.status.busy":"2021-08-09T05:42:08.482444Z","iopub.execute_input":"2021-08-09T05:42:08.482849Z","iopub.status.idle":"2021-08-09T05:42:08.522502Z","shell.execute_reply.started":"2021-08-09T05:42:08.482813Z","shell.execute_reply":"2021-08-09T05:42:08.521498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labeled_df = pd.merge(train[['image', 'series', 'study', 'path', 'bbox']], \n                      study_df[['study', 'label', 'label_int']],\n                      on='study', how='left')\nlabeled_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-09T05:45:22.186549Z","iopub.execute_input":"2021-08-09T05:45:22.186893Z","iopub.status.idle":"2021-08-09T05:45:22.219908Z","shell.execute_reply.started":"2021-08-09T05:45:22.186864Z","shell.execute_reply":"2021-08-09T05:45:22.218923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_images = len(labeled_df)\ntrain_df, valid_df = model_selection.train_test_split(labeled_df.sample(num_images, random_state=42), \n                                                      random_state=42,\n                                                      train_size=0.9, \n                                                      shuffle=True)\ntrain_df = train_df.reset_index(drop=True)\nvalid_df = valid_df.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T06:23:33.317854Z","iopub.execute_input":"2021-08-09T06:23:33.318230Z","iopub.status.idle":"2021-08-09T06:23:33.333910Z","shell.execute_reply.started":"2021-08-09T06:23:33.318198Z","shell.execute_reply":"2021-08-09T06:23:33.332738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def label_img(df, idx):\n    boxes = df.bbox.iloc[idx]\n    image_path = df.path.iloc[idx]\n    label = df.label.iloc[idx]\n    fig, ax = plt.subplots()\n    img_info = dicom.read_file(image_path)\n    img_arr = apply_voi_lut(img_info.pixel_array, img_info)\n    ax.imshow(img_arr);\n    if boxes != []:\n        for box in boxes:\n            x1, y1, w, h = box\n            ax.add_patch(patches.Rectangle((x1, y1), w, h, linewidth=1, edgecolor='r', facecolor='none', label=f'{label}'))\n            ax.annotate(label, (x1, y1), color='r', weight='bold', fontsize=15, ha='left', va='bottom')","metadata":{"execution":{"iopub.status.busy":"2021-08-09T05:45:27.921237Z","iopub.execute_input":"2021-08-09T05:45:27.921722Z","iopub.status.idle":"2021-08-09T05:45:27.930736Z","shell.execute_reply.started":"2021-08-09T05:45:27.921690Z","shell.execute_reply":"2021-08-09T05:45:27.929434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, row in labeled_df.iloc[4:5].iterrows():\n    label_img(labeled_df, i)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T06:01:19.640979Z","iopub.execute_input":"2021-08-09T06:01:19.641528Z","iopub.status.idle":"2021-08-09T06:01:21.244772Z","shell.execute_reply.started":"2021-08-09T06:01:19.641492Z","shell.execute_reply":"2021-08-09T06:01:21.243802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, row in train_df[:5].iterrows():\n    label_img(train_df, i)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T05:56:51.031058Z","iopub.execute_input":"2021-08-09T05:56:51.031455Z","iopub.status.idle":"2021-08-09T05:56:57.932102Z","shell.execute_reply.started":"2021-08-09T05:56:51.031423Z","shell.execute_reply":"2021-08-09T05:56:57.930837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Organize files into Yolo file structure\n```\n|-- main\n    |-- images\n    |   |-- train\n    |   |-- validation\n    |\n    |-- labels\n        |-- train\n        |-- validation\n```","metadata":{"execution":{"iopub.status.busy":"2021-07-29T20:17:38.496182Z","iopub.execute_input":"2021-07-29T20:17:38.496506Z","iopub.status.idle":"2021-07-29T20:17:38.503064Z","shell.execute_reply.started":"2021-07-29T20:17:38.496476Z","shell.execute_reply":"2021-07-29T20:17:38.501858Z"}}},{"cell_type":"code","source":"def get_bbox(df, idx, normalize=True): \n    img_path = df.iloc[idx].path\n    img_bbox = df.iloc[idx].bbox\n    img_class = df.iloc[idx].label_int\n    img = dicom.dcmread(img_path)\n    xsize = img.Columns\n    ysize = img.Rows\n\n    yolo_bboxes = []\n\n    for box in img_bbox:\n        x1, y1, w, h = box       \n        # yolo bboxes\n        bx = x1 + w / 2\n        by = y1 + h / 2\n        if normalize:\n            bx_norm = bx / xsize\n            by_norm = by / ysize\n            w_norm = w / xsize\n            h_norm = h / ysize\n\n            yolo_bboxes.append([img_class, bx_norm, by_norm, w_norm, h_norm])\n        else:\n            yolo_bboxes.append([img_class, bx, by, w, h])\n            \n    return yolo_bboxes\n\ndef convert_to_png(img_path, output_path):\n    filename = Path(img_path).stem\n    img_dicom = dicom.dcmread(img_path)\n    # Conversion to image using min-max scaling and conversion to 8-bit from [@heyytanay](https://www.kaggle.com/heyytanay)\n    data = apply_voi_lut(img_dicom.pixel_array, img_dicom)\n    if img_dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    img_arr = (data * 255).astype(np.uint8)\n    # Resize image when converting to png to save space\n    img_arr_resize = cv2.resize(img_arr, dsize=(512, 512), interpolation=cv2.INTER_CUBIC)\n    cv2.imwrite(output_path,img_arr_resize)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T06:22:15.699294Z","iopub.execute_input":"2021-08-09T06:22:15.699643Z","iopub.status.idle":"2021-08-09T06:22:15.711973Z","shell.execute_reply.started":"2021-08-09T06:22:15.699614Z","shell.execute_reply":"2021-08-09T06:22:15.710615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Test conversion to png","metadata":{"execution":{"iopub.status.busy":"2021-08-09T06:16:10.954261Z","iopub.execute_input":"2021-08-09T06:16:10.954608Z","iopub.status.idle":"2021-08-09T06:16:10.959184Z","shell.execute_reply.started":"2021-08-09T06:16:10.954578Z","shell.execute_reply":"2021-08-09T06:16:10.958200Z"}}},{"cell_type":"code","source":"path = r'../input/siim-covid19-detection/train/2a234c42eaac/a4dd021980a5/e24d2e46a243.dcm'\nconvert_to_png(path, './test_img.png')","metadata":{"execution":{"iopub.status.busy":"2021-08-09T06:15:42.575478Z","iopub.execute_input":"2021-08-09T06:15:42.575833Z","iopub.status.idle":"2021-08-09T06:15:42.817157Z","shell.execute_reply.started":"2021-08-09T06:15:42.575803Z","shell.execute_reply":"2021-08-09T06:15:42.816206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def file_organizer(df, dataset):\n    for i, row in tqdm(df.iterrows(), total=len(df)):\n        image_id = row.image\n        image_path = row.path\n        try:\n            img_dicom = dicom.dcmread(image_path)\n            img_arr = img_dicom.pixel_array\n            \n            # Save file in labels directory\n            labels = get_bbox(df, i, normalize=True)\n            np.savetxt(f\"./labels/{dataset}/{image_id}.txt\", labels)\n\n            # Copy image to local directory\n            convert_to_png(image_path, f\"./images/{dataset}/{image_id}.png\")\n        except RuntimeError:\n            print(f'could not process {image_id}')\n            continue\n\nfile_organizer(train_df, 'train')\nfile_organizer(valid_df, 'validation')","metadata":{"execution":{"iopub.status.busy":"2021-08-09T06:24:17.297664Z","iopub.execute_input":"2021-08-09T06:24:17.297998Z","iopub.status.idle":"2021-08-09T06:53:23.532972Z","shell.execute_reply.started":"2021-08-09T06:24:17.297966Z","shell.execute_reply":"2021-08-09T06:53:23.528386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip -r training_dataset.zip .","metadata":{"execution":{"iopub.status.busy":"2021-08-09T06:53:23.542882Z","iopub.execute_input":"2021-08-09T06:53:23.543277Z","iopub.status.idle":"2021-08-09T06:54:05.267749Z","shell.execute_reply.started":"2021-08-09T06:53:23.543243Z","shell.execute_reply":"2021-08-09T06:54:05.266595Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Inference for YoloV5 in future notebook!","metadata":{}},{"cell_type":"markdown","source":"---\n### Extra: Reading from files","metadata":{}},{"cell_type":"code","source":"Image.fromarray(np.uint8(Image.open('./images/train/a1b9944654af.png'))*255, 'L')","metadata":{"execution":{"iopub.status.busy":"2021-07-30T01:32:39.220845Z","iopub.execute_input":"2021-07-30T01:32:39.221283Z","iopub.status.idle":"2021-07-30T01:32:39.49174Z","shell.execute_reply.started":"2021-07-30T01:32:39.221244Z","shell.execute_reply":"2021-07-30T01:32:39.490839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('./labels/train/a1b9944654af.txt') as f:\n    content = f.read()\ncontent","metadata":{"execution":{"iopub.status.busy":"2021-07-30T01:32:39.493024Z","iopub.execute_input":"2021-07-30T01:32:39.493497Z","iopub.status.idle":"2021-07-30T01:32:39.50388Z","shell.execute_reply.started":"2021-07-30T01:32:39.493448Z","shell.execute_reply":"2021-07-30T01:32:39.502548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_organizer(train_df, 'train')","metadata":{"execution":{"iopub.status.busy":"2021-07-30T01:32:39.505409Z","iopub.execute_input":"2021-07-30T01:32:39.505754Z","iopub.status.idle":"2021-07-30T01:32:49.148786Z","shell.execute_reply.started":"2021-07-30T01:32:39.505724Z","shell.execute_reply":"2021-07-30T01:32:49.146746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}