{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install python-gdcm\nimport numpy as np\nimport pandas as pd\nimport cv2,random\nimport pydicom\nimport gdcm\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport matplotlib.pyplot as plt\nimport glob\nimport os\nfrom PIL import Image, ImageStat","metadata":{"execution":{"iopub.status.busy":"2021-07-10T09:50:09.450659Z","iopub.execute_input":"2021-07-10T09:50:09.45107Z","iopub.status.idle":"2021-07-10T09:50:19.910058Z","shell.execute_reply.started":"2021-07-10T09:50:09.450988Z","shell.execute_reply":"2021-07-10T09:50:19.908999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = glob.glob(\"../input/siim-covid19-detection/train/**/*.dcm\", recursive=True)\ndf_img = pd.read_csv(\"../input/siim-covid19-detection/train_image_level.csv\")\ndf_label = pd.read_csv(\"../input/siim-covid19-detection/train_study_level.csv\")\ndf_img.StudyInstanceUID = df_img.StudyInstanceUID + \"_study\"\ndf = pd.merge(df_img, df_label, left_on='StudyInstanceUID', right_on='id').drop(columns='id_y')\npath_df = pd.DataFrame(path, columns=['path'])\npath_df['filename'] = path_df.path.apply(lambda p: os.path.split(p)[1])\npath_df.filename = path_df.filename.str.replace('.dcm', '_image')\ndf = pd.merge(df, path_df, left_on='id_x', right_on='filename').drop(columns='filename')\ndf['fold'] = 0\ndf.fold = df.fold.apply(lambda x: random.randint(0,3))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-10T09:50:19.920777Z","iopub.execute_input":"2021-07-10T09:50:19.921064Z","iopub.status.idle":"2021-07-10T09:50:46.125568Z","shell.execute_reply.started":"2021-07-10T09:50:19.921038Z","shell.execute_reply":"2021-07-10T09:50:46.124524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data\n\ndef expand2square(pil_img, background_color):\n    width, height = pil_img.size\n    if width == height:\n        return pil_img\n    elif width > height:\n        result = Image.new(pil_img.mode, (width, width), background_color)\n        result.paste(pil_img, (0, (width - height) // 2))\n        return result\n    else:\n        result = Image.new(pil_img.mode, (height, height), background_color)\n        result.paste(pil_img, ((height - width) // 2, 0))\n        return result\n\ndef resize(array, size1,size2, keep_ratio=True, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size1, size2), resample)\n        im = expand2square(im, background_color=0)\n    else:\n        im = im.resize((size1, size2), resample)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-07-10T09:50:46.127439Z","iopub.execute_input":"2021-07-10T09:50:46.128034Z","iopub.status.idle":"2021-07-10T09:50:46.141062Z","shell.execute_reply.started":"2021-07-10T09:50:46.127989Z","shell.execute_reply":"2021-07-10T09:50:46.140048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def kuroobi_hrange(array, black_thr=50, white_thr=250):\n    H, W = np.shape(array)\n    cut_dh0, cut_dh1 = 0,0\n    for h0 in range(0, H//2):\n        row = array[h0, :]\n        mean = np.sum(row)/ len(row)\n        if mean < black_thr or white_thr < mean:\n            cut_dh0 += 1\n        else:\n            break\n    for h1 in range(1, H//2):\n        row = array[H-h1, :]\n        mean = np.sum(row) / len(row)\n        if mean < black_thr or white_thr < mean:\n            cut_dh1 += 1\n        else:\n            break\n    return cut_dh0, cut_dh1\n\ndef kuroobi_wrange(array, black_thr=50, white_thr=250):\n    H, W = np.shape(array)\n    cut_dw0, cut_dw1 = 0,0\n    for w0 in range(0, W//2):\n        col = array[:,w0]\n        mean = np.sum(col) / len(col)\n        if mean < black_thr or white_thr < mean:\n            cut_dw0 += 1\n        else:\n            break\n    for w1 in range(1, W//2):\n        col = array[:,W-w1]\n        mean = np.sum(col) / len(col)\n        if mean < black_thr or white_thr < mean:\n            cut_dw1 += 1\n        else:\n            break\n    return cut_dw0, cut_dw1\n\ndef check_kuroobi(image, black_thr=50, white_thr=250):\n    array = np.asarray(image)\n    H, W = np.shape(array)\n    dh0, dh1 = kuroobi_hrange(array)\n    dw0, dw1 = kuroobi_wrange(array)\n    return dh0,dw0,H-dh1,W-dw1\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs = []\ndfs.append(df[df.fold ==0])\ndfs.append(df[df.fold ==1])\ndfs.append(df[df.fold ==2])\ndfs.append(df[df.fold ==3])\nprint(len(dfs[0]),len(dfs[1]),len(dfs[2]),len(dfs[3]))\nprint(len(df))","metadata":{"execution":{"iopub.status.busy":"2021-07-10T09:50:46.142608Z","iopub.execute_input":"2021-07-10T09:50:46.14311Z","iopub.status.idle":"2021-07-10T09:50:46.167291Z","shell.execute_reply.started":"2021-07-10T09:50:46.143064Z","shell.execute_reply":"2021-07-10T09:50:46.166174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['mean'] = 0\ndf['var'] = 0\ndf['dim1'] = 0\ndf['dim2'] = 0\n\ndf['crop_xmin'] = 0\ndf['crop_ymin'] = 0\ndf['crop_xmax'] = 0\ndf['crop_ymax'] = 0\n\ndf['crop_fxmin'] = 0\ndf['crop_fymin'] = 0\ndf['crop_fxmax'] = 0\ndf['crop_fymax'] = 0","metadata":{"execution":{"iopub.status.busy":"2021-07-10T09:50:46.168799Z","iopub.execute_input":"2021-07-10T09:50:46.169214Z","iopub.status.idle":"2021-07-10T09:50:46.177992Z","shell.execute_reply.started":"2021-07-10T09:50:46.169183Z","shell.execute_reply":"2021-07-10T09:50:46.177115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run(fold):\n    for idx,(index,row) in enumerate(dfs[fold].iterrows()):\n        image = read_xray(row.path)\n        dim0, dim1 = image.shape\n        df.loc[index,'dim1'], df.loc[index,'dim2'] = dim0, dim1\n        \n        xmin, ymin, xmax, ymax = check_kuroobi(image)\n        \n        df.loc[index,\"crop_xmin\"] = xmin\n        df.loc[index,\"crop_ymin\"] = ymin\n        df.loc[index,\"crop_xmax\"] = xmax\n        df.loc[index,\"crop_ymax\"] = ymax\n        \n        df.loc[index,\"crop_fxmin\"] = xmin/dim0\n        df.loc[index,\"crop_fymin\"] = ymin/dim1\n        df.loc[index,\"crop_fxmax\"] = xmax/dim0\n        df.loc[index,\"crop_fymax\"] = ymax/dim1\n\n        df.loc[index,'mean'] = image.mean()\n        df.loc[index,'var'] = image.std()\n        \n        if idx%500==0:\n            print(idx)","metadata":{"execution":{"iopub.status.busy":"2021-07-10T09:50:46.179977Z","iopub.execute_input":"2021-07-10T09:50:46.180498Z","iopub.status.idle":"2021-07-10T09:50:46.195554Z","shell.execute_reply.started":"2021-07-10T09:50:46.180453Z","shell.execute_reply":"2021-07-10T09:50:46.194788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from joblib import Parallel, delayed\nParallel(n_jobs=4, backend=\"threading\")(delayed(run)(i) for i in range(4))","metadata":{"execution":{"iopub.status.busy":"2021-07-10T09:50:46.19684Z","iopub.execute_input":"2021-07-10T09:50:46.197408Z","iopub.status.idle":"2021-07-10T09:52:20.592768Z","shell.execute_reply.started":"2021-07-10T09:50:46.197376Z","shell.execute_reply":"2021-07-10T09:52:20.587927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv(\"imginfo.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-07-10T09:52:41.438168Z","iopub.execute_input":"2021-07-10T09:52:41.438482Z","iopub.status.idle":"2021-07-10T09:52:41.570045Z","shell.execute_reply.started":"2021-07-10T09:52:41.438451Z","shell.execute_reply":"2021-07-10T09:52:41.569011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}