{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"! pip install gdcm","metadata":{"execution":{"iopub.status.busy":"2021-06-24T04:39:48.290083Z","iopub.execute_input":"2021-06-24T04:39:48.290566Z","iopub.status.idle":"2021-06-24T04:39:58.524422Z","shell.execute_reply.started":"2021-06-24T04:39:48.290474Z","shell.execute_reply":"2021-06-24T04:39:58.523149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! pip install -q pylibjpeg pylibjpeg-libjpeg pydicom","metadata":{"execution":{"iopub.status.busy":"2021-06-24T04:41:09.60076Z","iopub.execute_input":"2021-06-24T04:41:09.601141Z","iopub.status.idle":"2021-06-24T04:41:18.375447Z","shell.execute_reply.started":"2021-06-24T04:41:09.601103Z","shell.execute_reply":"2021-06-24T04:41:18.374169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\n\nsys.path.append('../input/lovaszsoftmax/LovaszSoftmax-master/tensorflow')\nimport lovasz_losses_tf as L","metadata":{"execution":{"iopub.status.busy":"2021-07-05T04:38:44.078864Z","iopub.execute_input":"2021-07-05T04:38:44.079912Z","iopub.status.idle":"2021-07-05T04:38:44.161059Z","shell.execute_reply.started":"2021-07-05T04:38:44.079863Z","shell.execute_reply":"2021-07-05T04:38:44.160245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install efficientnet -q","metadata":{"execution":{"iopub.status.busy":"2021-07-05T04:44:39.537887Z","iopub.execute_input":"2021-07-05T04:44:39.538318Z","iopub.status.idle":"2021-07-05T04:44:47.526407Z","shell.execute_reply.started":"2021-07-05T04:44:39.538287Z","shell.execute_reply":"2021-07-05T04:44:47.525239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LOSS = L.lovasz_softmax","metadata":{"execution":{"iopub.status.busy":"2021-07-05T04:50:25.619499Z","iopub.execute_input":"2021-07-05T04:50:25.620023Z","iopub.status.idle":"2021-07-05T04:50:25.687981Z","shell.execute_reply.started":"2021-07-05T04:50:25.619993Z","shell.execute_reply":"2021-07-05T04:50:25.687217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport efficientnet.tfkeras as efn\n\nIMSIZE = (224, 240, 260, 300, 380, 456, 528, 600)\nIMS = 7\nn_labels = 4\n\nmodel = tf.keras.Sequential([\n    efn.EfficientNetB7(\n        input_shape=(IMSIZE[IMS], IMSIZE[IMS], 3),\n        weights='imagenet',\n        include_top=False),\n    tf.keras.layers.GlobalAveragePooling2D(),\n    tf.keras.layers.Dense(n_labels, activation='softmax')\n])\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(),\n    loss=LOSS,\n    metrics=[tf.keras.metrics.AUC(multi_label=True)])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-07-05T04:50:50.49909Z","iopub.execute_input":"2021-07-05T04:50:50.499591Z","iopub.status.idle":"2021-07-05T04:50:59.10071Z","shell.execute_reply.started":"2021-07-05T04:50:50.499551Z","shell.execute_reply":"2021-07-05T04:50:59.099622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport imageio\nimport glob\nimport matplotlib.pyplot as plt\nfrom PIL import ImageDraw, ImageFont, Image\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom pydicom import dcmread\n#import pylibjpeg\n#import gdcm\n\nRANDOM_STATE = 123","metadata":{"execution":{"iopub.status.busy":"2021-06-24T04:41:26.141744Z","iopub.execute_input":"2021-06-24T04:41:26.142115Z","iopub.status.idle":"2021-06-24T04:41:27.908206Z","shell.execute_reply.started":"2021-06-24T04:41:26.142078Z","shell.execute_reply":"2021-06-24T04:41:27.906365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# train_image_level.csvの確認","metadata":{}},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/siim-covid19-detection/\"\nTRAIN_PATH = os.path.join(BASE_PATH, \"train\")\nTEST_PATH = os.path.join(BASE_PATH, \"test\")\ntrain_image_path = os.path.join(BASE_PATH,\"train_image_level.csv\")\ntrain_study_path = os.path.join(BASE_PATH,\"train_study_level.csv\")\nsubmission_path = os.path.join(BASE_PATH,\"sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-06-06T09:05:31.137367Z","iopub.execute_input":"2021-06-06T09:05:31.137552Z","iopub.status.idle":"2021-06-06T09:05:31.142767Z","shell.execute_reply.started":"2021-06-06T09:05:31.137531Z","shell.execute_reply":"2021-06-06T09:05:31.14212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_image = pd.read_csv(train_image_path)\ndf_train_image","metadata":{"execution":{"iopub.status.busy":"2021-06-06T09:05:31.143932Z","iopub.execute_input":"2021-06-06T09:05:31.144267Z","iopub.status.idle":"2021-06-06T09:05:31.238667Z","shell.execute_reply.started":"2021-06-06T09:05:31.144237Z","shell.execute_reply":"2021-06-06T09:05:31.237789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_image.iloc[0][\"label\"].split()","metadata":{"execution":{"iopub.status.busy":"2021-06-06T09:06:24.58069Z","iopub.execute_input":"2021-06-06T09:06:24.580952Z","iopub.status.idle":"2021-06-06T09:06:24.589101Z","shell.execute_reply.started":"2021-06-06T09:06:24.58093Z","shell.execute_reply":"2021-06-06T09:06:24.588618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_image.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_train_image[\"label\"][0])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"boxesを[x_min, y_min, x_max, y_max]の形にする","metadata":{}},{"cell_type":"code","source":"# \"label\"列から[x_min, y_min, x_max, y_max]のリストを形成する関数\ndef get_bbox(row):\n    bboxes = []\n    bbox = []\n    for i, l in enumerate(row.split(' ')):\n        if (i % 6 == 0) | (i % 6 == 1):\n            continue\n        bbox.append(float(l))\n        if i % 6 == 5:\n            bboxes.append(bbox)\n            bbox = []  \n            \n    return bboxes","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_image[\"reformed_boxes\"] = df_train_image[\"label\"].apply(get_bbox)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_image.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_image.to_csv('new_train_image.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_image[\"reformed_boxes\"][0][0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(df_train_image[df_train_image[\"label\"]==\"none 1 0 0 1 1\"]))\ndf_train_image[\"boxes\"].isna().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# StudyInstanceUIDの重複を確認\ntrain_UID_list = []\ntrain_duplicate_list = []\nfor index, item in enumerate(df_train_image[\"StudyInstanceUID\"]):\n    if item in train_UID_list:\n        train_duplicate_list.append(item)\n        #print(\"index:\", index, \"UID:\", item)\n    else:\n        train_UID_list.append(item)\n        \nprint(len(train_UID_list))\nprint(len(train_duplicate_list))\n# duplicate_listに重複がないか確認\nprint(len(set(train_duplicate_list)))\ntrain_unique_dupulicate_list = list(set(train_duplicate_list))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"以上から、重複しているUIDがあることが分かる","metadata":{}},{"cell_type":"code","source":"print(train_duplicate_list[:3])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# opacityは1以外ないか確認\nfor index, item in enumerate(df_train_image[\"label\"][:100]):\n    if item.split(\" \")[1]!=\"1\":\n        print(item.split(\" \")[1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"opacityの番号に特に意味がないことが分かる","metadata":{}},{"cell_type":"markdown","source":"# train_study_level.csvの確認","metadata":{}},{"cell_type":"code","source":"import pandas as pd\ndf_train_study = pd.read_csv(\"../input/siim-covid19-detection/train_study_level.csv\")\ndf_train_study.columns\ndf_train_study","metadata":{"execution":{"iopub.status.busy":"2021-06-30T02:09:49.959324Z","iopub.execute_input":"2021-06-30T02:09:49.959771Z","iopub.status.idle":"2021-06-30T02:09:50.019696Z","shell.execute_reply.started":"2021-06-30T02:09:49.959677Z","shell.execute_reply":"2021-06-30T02:09:50.018866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_study[df_train_study[\"id\"]==\"13e21ab810f1_study\"]","metadata":{"execution":{"iopub.status.busy":"2021-06-28T07:59:08.453535Z","iopub.execute_input":"2021-06-28T07:59:08.453951Z","iopub.status.idle":"2021-06-28T07:59:08.469224Z","shell.execute_reply.started":"2021-06-28T07:59:08.453877Z","shell.execute_reply":"2021-06-28T07:59:08.468514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"negativeかpositiveかをまず判定するために、binary classの列を作る","metadata":{}},{"cell_type":"code","source":"def getbinary(df):\n    binary_target = []\n    for index, row in enumerate(df.iterrows()):\n        if row[1][\"Negative for Pneumonia\"]==1:\n            binary_target.append(0)\n        else:\n            binary_target.append(1)\n    return binary_target","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_study.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index, row in enumerate(df_train_study[:100].iterrows()):\n    num_classes = row[1][\"Negative for Pneumonia\"]+row[1][\"Typical Appearance\"]+row[1][\"Indeterminate Appearance\"]+row[1][\"Atypical Appearance\"]\n    if num_classes>=2:\n        print(i, \":\", num_classes)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"以上から、一つのstudyに二つ以上のクラスが混ざっていないことがわかる","metadata":{}},{"cell_type":"code","source":"# idがUIDと同じか確認\nprint(df_train_study[\"id\"][0].split(\"_\")[0])\ndf_train_study[\"id\"][0].split(\"_\")[0] in train_duplicate_list","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"train_study_levelのidとUIDは別物","metadata":{}},{"cell_type":"markdown","source":"# sample_submission.csvの確認","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ndf = pd.read_csv(\"../input/siim-covid19-detection/sample_submission.csv\")\ndf","metadata":{"execution":{"iopub.status.busy":"2021-06-26T07:25:43.705364Z","iopub.execute_input":"2021-06-26T07:25:43.7058Z","iopub.status.idle":"2021-06-26T07:25:43.755004Z","shell.execute_reply.started":"2021-06-26T07:25:43.705703Z","shell.execute_reply":"2021-06-26T07:25:43.753876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = pd.read_csv(submission_path)\ndf_submission.head()\ndf_submission[:10]","metadata":{"execution":{"iopub.status.busy":"2021-05-30T23:04:32.81017Z","iopub.execute_input":"2021-05-30T23:04:32.810716Z","iopub.status.idle":"2021-05-30T23:04:32.85289Z","shell.execute_reply.started":"2021-05-30T23:04:32.81068Z","shell.execute_reply":"2021-05-30T23:04:32.85185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_study = 0\nnum_image = 0\n\nfor index, row in enumerate(df_submission.iterrows()):\n    study_or_image = row[1][\"id\"].split(\"_\")[1]\n    if study_or_image==\"study\":\n        num_study+=1\n    elif study_or_image==\"image\":\n        num_image+=1\n    else:\n        print(index, \":else\")\n\nprint(\"num_study:\", num_study)\nprint(\"num_image:\", num_image)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_submissionをdf_test_studyとdf_test_imageに分ける\ndf_test_study = df_submission.iloc[:1214]\ndf_test_image = df_submission.iloc[1214:]\n\ndisplay(df_test_study.tail())\ndf_test_image.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 画像フォルダの確認","metadata":{}},{"cell_type":"code","source":"print(len(os.listdir(TRAIN_PATH)))\nprint(len(os.listdir(TEST_PATH)))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"以上から、1studyに対して複数の画像が格納されている場合がある。\n\ntrainの場合は、train_unique_dupulicate_listに格納。","metadata":{}},{"cell_type":"markdown","source":"# 実際に画像を確認してみる","metadata":{}},{"cell_type":"code","source":"UID = \"0fd2db233deb\"\nfile_id = \"000a312787f2\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_path_list = glob.glob(f\"/kaggle/input/siim-covid19-detection/train/{UID}/*/*.dcm\")\nimage_path_list","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_path = glob.glob(os.path.join(TRAIN_PATH, \"/*/*/{file_id}.dcm\"))\nimage_path","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = imageio.imread(image_path[0], \"DICOM\")\nprint(image.shape)\nplt.imshow(image, cmap=\"gray\")\nplt.axis(\"off\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"原画像とboxesを重ね合わせる関数を作成","metadata":{}},{"cell_type":"code","source":"# train_image_leveの画像確認\nimage_with_bboxes = []\nfor index, row in enumerate(df_train_image[:3].iterrows()):\n    file_id = row[1][\"id\"].split(\"_\")[0]\n    image_path = glob.glob(f\"/kaggle/input/siim-covid19-detection/train/*/*/{file_id}.dcm\")\n    image = imageio.imread(image_path[0], \"DICOM\")\n    #print(\"index:\", index, \"image_shape:\", image.shape)\n    image = Image.fromarray(image)\n    bboxes = row[1][\"reformed_boxes\"]\n    draw = ImageDraw.Draw(image)\n    for box in bboxes:\n        draw.rectangle(box, outline=\"black\", width=5)\n    image_with_bboxes.append(image)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 16))\nfor i, image in enumerate(image_with_bboxes):\n    plt.subplot(1, 3, i+1)\n    plt.imshow(image, cmap=\"gray\")\n    plt.axis(\"off\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"height_list = []\nwidth_list = []\n\nfor index, row in enumerate(df_train_study.iterrows()):\n    file_id = row[1][\"id\"].split(\"_\")[0]\n    image_path_list = glob.glob(f\"/kaggle/input/siim-covid19-detection/train/{file_id}/*/*.dcm\")\n    for path in image_path_list:\n        image = imageio.imread(path, \"DICOM\")\n        height, width = image.shape\n        height_list.append(height)\n        width_list.append(width)\n\nprint(len(width_list))\nlen(height_list)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"height_list = []\nwidth_list = []\n\nfor index, row in enumerate(df_train_study.iterrows()):\n    file_id = row[1][\"id\"].split(\"_\")[0]\n    image_path_list = glob.glob(f\"/kaggle/input/siim-covid19-detection/train/{file_id}/*/*.dcm\")\n    for path in image_path_list:\n        dicom = pydicom.read_file(path)\n        image = apply_voi_lut(dicom.pixel_array, dicom)\n        height, width = image.shape\n        height_list.append(height)\n        width_list.append(width)\n\nprint(len(width_list))\nlen(height_list)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\n\npaths = glob.glob(\"../input/512img840study/study/*.png\")","metadata":{"execution":{"iopub.status.busy":"2021-06-26T21:19:44.463604Z","iopub.execute_input":"2021-06-26T21:19:44.46444Z","iopub.status.idle":"2021-06-26T21:19:45.260676Z","shell.execute_reply.started":"2021-06-26T21:19:44.464279Z","shell.execute_reply":"2021-06-26T21:19:45.259483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport matplotlib.pyplot as plt\nimport cv2\n\nimage = cv2.imread(paths[1])\nplt.imshow(image)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-20T13:14:23.486535Z","iopub.execute_input":"2021-06-20T13:14:23.487022Z","iopub.status.idle":"2021-06-20T13:14:24.023676Z","shell.execute_reply.started":"2021-06-20T13:14:23.486981Z","shell.execute_reply":"2021-06-20T13:14:24.022632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-20T13:14:31.122713Z","iopub.execute_input":"2021-06-20T13:14:31.123155Z","iopub.status.idle":"2021-06-20T13:14:31.133911Z","shell.execute_reply.started":"2021-06-20T13:14:31.12312Z","shell.execute_reply":"2021-06-20T13:14:31.132776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport matplotlib.pyplot as plt\nimport cv2\n\nimage = cv2.imread(paths[0])\nthresh = 150\nimg_bin = cv2.threshold(image, thresh, 255, cv2.THRESH_BINARY)[1]\n\nimg_bin.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-26T21:21:18.451292Z","iopub.execute_input":"2021-06-26T21:21:18.451751Z","iopub.status.idle":"2021-06-26T21:21:18.483314Z","shell.execute_reply.started":"2021-06-26T21:21:18.451705Z","shell.execute_reply":"2021-06-26T21:21:18.482399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image = cv2.imread(paths[0])\nimg_gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)  \n\nimg_gray.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-26T21:22:54.049043Z","iopub.execute_input":"2021-06-26T21:22:54.049586Z","iopub.status.idle":"2021-06-26T21:22:54.08161Z","shell.execute_reply.started":"2021-06-26T21:22:54.049552Z","shell.execute_reply":"2021-06-26T21:22:54.080784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"im = Image.fromarray(img_gray)","metadata":{"execution":{"iopub.status.busy":"2021-06-26T21:26:04.482525Z","iopub.execute_input":"2021-06-26T21:26:04.483057Z","iopub.status.idle":"2021-06-26T21:26:04.489207Z","shell.execute_reply.started":"2021-06-26T21:26:04.483026Z","shell.execute_reply":"2021-06-26T21:26:04.48768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(im)","metadata":{"execution":{"iopub.status.busy":"2021-06-26T21:26:25.141432Z","iopub.execute_input":"2021-06-26T21:26:25.141867Z","iopub.status.idle":"2021-06-26T21:26:25.430492Z","shell.execute_reply.started":"2021-06-26T21:26:25.141821Z","shell.execute_reply":"2021-06-26T21:26:25.429709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\nimg = Image.open(paths[0])\nnew_image = np.array(img, dtype=np.uint8)\ntype(new_image)\nplt.imshow(new_image, cmap=\"gray\")","metadata":{"execution":{"iopub.status.busy":"2021-06-26T21:37:08.076814Z","iopub.execute_input":"2021-06-26T21:37:08.077311Z","iopub.status.idle":"2021-06-26T21:37:08.366215Z","shell.execute_reply.started":"2021-06-26T21:37:08.077272Z","shell.execute_reply":"2021-06-26T21:37:08.365111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_cropped = crop(new_image)\nplt.imshow(img_cropped, cmap=\"gray\")","metadata":{"execution":{"iopub.status.busy":"2021-06-26T21:39:38.250972Z","iopub.execute_input":"2021-06-26T21:39:38.251556Z","iopub.status.idle":"2021-06-26T21:39:38.455006Z","shell.execute_reply.started":"2021-06-26T21:39:38.251518Z","shell.execute_reply":"2021-06-26T21:39:38.454188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(type(img_cropped))\nimg_cropped.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-26T21:40:35.134543Z","iopub.execute_input":"2021-06-26T21:40:35.135173Z","iopub.status.idle":"2021-06-26T21:40:35.142604Z","shell.execute_reply.started":"2021-06-26T21:40:35.135134Z","shell.execute_reply":"2021-06-26T21:40:35.141806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def crop(data):\n    thresh = 150\n    img_bin = cv2.threshold(data, thresh, 255, cv2.THRESH_BINARY)[1]\n    img_bin = cv2.rotate(img_bin, cv2.cv2.ROTATE_90_CLOCKWISE)\n    right = 0;\n    left = 0;\n    line_thickness = 2\n\n    # This is the value that specifies how bright a row is to consider it 'not the edge (too bright)'\n    intensity_threshold = 190\n\n    # Start at the bottom and work upwards checking the mean of pixels in every 10th row, this is the right side of the image\n    for i in range(img_bin.shape[0]-1,0,-10):\n        row_mean = img_bin[i].mean()\n        if row_mean > intensity_threshold:\n            right = i\n\n            # Draw a line where we want to crop\n            cv2.line(img_bin, (0, i), (img_bin.shape[1], i), (0, img_bin.shape[1], 0), thickness=line_thickness)\n            break\n\n    # Start at the top and go down to find the left side\n    for i in range(0,img_bin.shape[0]-1,10):\n        row_mean = img_bin[i].mean()\n        if row_mean > intensity_threshold:\n            left = i\n\n            # Draw a line where we want to crop\n            cv2.line(img_bin, (0, i), (img_bin.shape[1], i), (0, img_bin.shape[1], 0), thickness=line_thickness)\n            break\n    # Rotate the image back to it's normal orientation\n    img_bin = cv2.rotate(img_bin, cv2.cv2.ROTATE_90_COUNTERCLOCKWISE)\n\n    x1 = left\n    y1 = 0\n    x2 = right\n    y2 = img_bin.shape[1]\n\n    # Grab the region we identified from the binarized image\n    img_cropped = img_bin[y1:y2, x1:x2]\n\n    top = 0;\n    bottom = 0;\n\n    # Set some threshold values to specify what we consider edge vs patient\n    bright_threshold = 240\n    dark_threshold = 100\n\n    # Start at the bottom and work upward\n    for i in range(img_cropped.shape[0]-1,0,-10):\n        row_mean = img_cropped[i].mean()\n        if row_mean < bright_threshold:\n            # Add 100 pixels of padding so we don't cut the costophrenic angles off\n            bottom = i + 100\n            cv2.line(img_cropped, (0, bottom), (img_cropped.shape[1], bottom), (0, img_cropped.shape[1], 0), thickness=line_thickness)\n            break\n\n    # Start at the top and go down\n    for i in range(0,img_cropped.shape[0]-1,10):\n        row_mean = img_cropped[i].mean()\n        if row_mean > dark_threshold:\n            top = i\n            cv2.line(img_cropped, (0, i), (img_cropped.shape[1], i), (0, img_cropped.shape[1], 0), thickness=line_thickness)\n            break\n            \n    x1 = 0\n    y1 = top\n    x2 = img_bin.shape[0]\n    y2 = bottom\n\n    img_cropped = img_cropped[y1:y2, x1:x2]\n    \n    img_cropped = data[top:bottom, left:right]\n    new_image = np.array(img_cropped, dtype=np.uint8)\n    \n    return new_image","metadata":{"execution":{"iopub.status.busy":"2021-06-26T21:39:33.2575Z","iopub.execute_input":"2021-06-26T21:39:33.258111Z","iopub.status.idle":"2021-06-26T21:39:33.274356Z","shell.execute_reply.started":"2021-06-26T21:39:33.25807Z","shell.execute_reply":"2021-06-26T21:39:33.273585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}