{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<center><h1> Smart cropping with YOLOv5</h1></center>","metadata":{}},{"cell_type":"code","source":"!pip install -U git+https://github.com/pydicom/pydicom.git\n!conda install pydicom --channel conda-forge -y\n!conda install -c conda-forge gdcm -y\n","metadata":{"execution":{"iopub.status.busy":"2021-07-21T08:25:41.849411Z","iopub.execute_input":"2021-07-21T08:25:41.849994Z","iopub.status.idle":"2021-07-21T08:27:18.784195Z","shell.execute_reply.started":"2021-07-21T08:25:41.84994Z","shell.execute_reply":"2021-07-21T08:27:18.782547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport json\nimport torch\nfrom skimage import exposure\nimport math\nfrom shutil import copyfile\nfrom IPython.display import Image, clear_output\nfrom PIL import Image\nfrom tqdm.auto import tqdm\nimport pydicom\n","metadata":{"execution":{"iopub.status.busy":"2021-07-21T08:27:18.786826Z","iopub.execute_input":"2021-07-21T08:27:18.787266Z","iopub.status.idle":"2021-07-21T08:27:20.786511Z","shell.execute_reply.started":"2021-07-21T08:27:18.78722Z","shell.execute_reply":"2021-07-21T08:27:20.78531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# D/L and install YOLOv5\n","metadata":{}},{"cell_type":"code","source":"!git clone https://github.com/ultralytics/yolov5\n%cd yolov5\n%pip install -qr requirements.txt\n%cd ../\nclear_output()\nprint(f\"setup complete, using torch {torch.__version__} ({torch.cuda.get_device_properties(0).name if torch.cuda.is_available() else 'CPU'})\")","metadata":{"execution":{"iopub.status.busy":"2021-07-21T08:27:20.788482Z","iopub.execute_input":"2021-07-21T08:27:20.788778Z","iopub.status.idle":"2021-07-21T08:27:29.09581Z","shell.execute_reply.started":"2021-07-21T08:27:20.788749Z","shell.execute_reply":"2021-07-21T08:27:29.0944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Copy the requirements file over so we don't have to change directory later. Is there an argument for this?\ncopyfile('yolov5/requirements.txt', '/kaggle/working/requirements.txt');\n\n# Load the image dataframe so we can get BB coords\nbase_path = \"/kaggle/input/siim-covid19-detection/\"\nimages_df = pd.read_csv(os.path.join(base_path,\"train_image_level.csv\"))","metadata":{"execution":{"iopub.status.busy":"2021-07-21T08:27:29.097885Z","iopub.execute_input":"2021-07-21T08:27:29.098207Z","iopub.status.idle":"2021-07-21T08:27:29.228251Z","shell.execute_reply.started":"2021-07-21T08:27:29.098174Z","shell.execute_reply":"2021-07-21T08:27:29.226987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## some function","metadata":{}},{"cell_type":"code","source":"\ndef auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    \n    return strategy\n\n\n\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    dicom = pydicom.read_file(path)\n    \n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data\n\n\n\n\ndef resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-07-21T08:27:29.229502Z","iopub.execute_input":"2021-07-21T08:27:29.229798Z","iopub.status.idle":"2021-07-21T08:27:29.241895Z","shell.execute_reply.started":"2021-07-21T08:27:29.229769Z","shell.execute_reply":"2021-07-21T08:27:29.240596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Convert yolo croods to pixels croods","metadata":{}},{"cell_type":"code","source":"def box2coord(x,y,w,h,image_w,image_h):\n    x1,y1 = round((x-w/2)*image_w), round((y-h/2)*image_h)\n    x2,y2 = round((x+w/2)*image_w), round((y+h/2)*image_h)\n    return x1, y1, x2, y2\n\n\n\ndef get_boxes(image_id):\n    image = image_id.replace('.dcm', '_image')\n    ti = images_df[images_df['id'] == image]\n    bx = [[],[]]\n    bx[0] = [0,0,0,0,'']\n    bx[1] = [0,0,0,0,'']\n    \n    if str(ti['boxes'].values[0]) != 'nan':\n        box = str(ti['boxes'].values[0].replace(\"'\",\"\\\"\"))\n        boxes = json.loads(box)\n        lab = ti['label'].values[0].split(\" \")\n        i = 0\n        for b in boxes:\n            bx[i] = [int(b['x']), int(b['y']), int(b['width']), int(b['height']), lab[0]]\n            i = i+1\n        return bx    \n    \n    \n    \ndef draw_boxes(boxes, z):\n    \n    for i in boxes:\n        x = [i[0]-z[0], i[0]+i[2]-z[0]]\n        y = [i[1]-z[1], i[1]-z[1]]\n        plt.plot(x,y , color = 'red', linewidth = 2)\n        \n        #bottom\n        y = [i[1]+i[3]-z[1], i[1]+i[3]-z[1]]\n        plt.plot(x,y , color = 'red', linewidth = 2)\n        \n        #left\n        x = [i[0] -z[0], i[0]-z[0]]\n        y = [i[1] - z[1], i[1] + i[3] - z[1]]\n        plt.plot(x,y, color='red', linewidth=2)\n        \n        x = [i[0]+i[2]-z[0], i[0]+i[2]-z[0]]\n        plt.plot(x,y, color='red', linewidth=2)\n            ","metadata":{"execution":{"iopub.status.busy":"2021-07-21T08:27:29.243743Z","iopub.execute_input":"2021-07-21T08:27:29.244257Z","iopub.status.idle":"2021-07-21T08:27:29.266502Z","shell.execute_reply.started":"2021-07-21T08:27:29.244211Z","shell.execute_reply":"2021-07-21T08:27:29.265265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nstrategy = auto_select_accelerator()","metadata":{"execution":{"iopub.status.busy":"2021-07-21T08:27:29.268207Z","iopub.execute_input":"2021-07-21T08:27:29.268524Z","iopub.status.idle":"2021-07-21T08:27:34.727409Z","shell.execute_reply.started":"2021-07-21T08:27:29.268487Z","shell.execute_reply":"2021-07-21T08:27:34.726196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    # Call YOLO detect.py with the image we exported\n    def detect():\n        # Clean up results from the last run\n        label_path = '/kaggle/working/runs/detect/exp/labels/'\n        if os.path.exists(label_path):\n            for txt_file in os.listdir(label_path):\n                txt_path = label_path+txt_file\n    #             print(txt_path)\n                os.remove(txt_path)\n    #     \n        # Call yolo detect\n        !python yolov5/detect.py --source /kaggle/working/imgs  --weights ../input/cxr-anatomy-detection/anatomy_detection.pt --img 640 --exist-ok --line-thickness 10 --save-txt","metadata":{"execution":{"iopub.status.busy":"2021-07-21T08:27:34.730561Z","iopub.execute_input":"2021-07-21T08:27:34.730985Z","iopub.status.idle":"2021-07-21T08:27:34.740869Z","shell.execute_reply.started":"2021-07-21T08:27:34.730952Z","shell.execute_reply":"2021-07-21T08:27:34.739223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.mkdir('/kaggle/working/imgs')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport numpy as np\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ni = 0\n\nwith strategy.scope():\n\n    for split in ['train']:\n        save_dir = f'/kaggle/working/{split}/'\n\n        os.makedirs(save_dir, exist_ok=True)\n\n        for dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/{split}')):          \n            for file in filenames:\n                i = i+1\n                if i > 3000:\n                    print(i)\n                     # set keep_ratio=True to have original aspect ratio\n                    pixels = read_xray(os.path.join(dirname, file))\n                    im = resize(pixels, size=600)  \n                    img_dir= dirname.split('/')[-2] + '_study.png'\n                    im.save(os.path.join('/kaggle/working/imgs/', img_dir))\n                    detect()  # detector run\n                    os.remove('/kaggle/working/imgs/'+img_dir)\n\n                    # # Get the predicted boxes into a dataframe\n                    label_dir = \"/kaggle/working/runs/detect/exp/labels/\"+img_dir.replace('png','txt')\n                    if os.path.isfile(label_dir): \n\n                        boxes = pd.read_csv(label_dir, delim_whitespace=True, header=None, index_col=False)\n\n                                    # Convert the normalized YOLO BB data of the predicted anatomy to pixel coords and add them to the dataframe\n                        for index, row in boxes.iterrows():\n\n                            x, y, xx, yy = box2coord(row[1], row[2], row[3], row[4], pixels.shape[1], pixels.shape[0])\n                            boxes.at[index,'x'] = x\n                            boxes.at[index,'y'] = y\n                            boxes.at[index,'xx'] = xx\n                            boxes.at[index,'yy'] = yy\n\n\n                        # Get only the lung coords .. class 0 and 1 .. you could choose shoulders or clavicles instead .. or all of them\n                        lungs = boxes[(boxes[0] == 0) | (boxes[0] == 1)]\n\n\n                        if len(lungs) != 0 :\n                            # Figure out the max dimensions of the predicted anatomy, and crop the image to those coords\n                            x1 = int(lungs['x'].min())\n                            x2 = int(lungs['xx'].max())\n                            y1 = int(lungs['y'].min())\n                            y2 = int(lungs['yy'].max())\n                            #crop image\n                            cropped = pixels[y1:y2, x1:x2]\n\n                            #croppeded save \n                            im = resize(cropped, size=600)\n                            im.save(save_dir+ img_dir)\n                        else:\n                            im.save(save_dir+ img_dir)\n                    else:\n                        im.save(save_dir+ img_dir)\n\n\n                \nprint('complete well done!')","metadata":{"execution":{"iopub.status.busy":"2021-07-21T08:58:39.072784Z","iopub.execute_input":"2021-07-21T08:58:39.073235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -r '/kaggle/working/yolov5'\n!rm -r '/kaggle/working/runs'\n!rm -r '/kaggle/working/imgs'","metadata":{},"execution_count":null,"outputs":[]}]}