{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# make dataset\n\n#### I got the imagelevel dataframe by [step2 make_dataframe](https://www.kaggle.com/kunihikofurugori/siim-step1-get-imginfo).\n\n#### Next step, I will make the dataset about image level.\n\n## my public notebook \n\n### [step1 get_imageinformation](https://www.kaggle.com/kunihikofurugori/siim-step1-get-imginfo).\n\n### [step2 make_dataframe](https://www.kaggle.com/kunihikofurugori/siim-step1-get-imginfo).\n\n### [step3-1 renew-imglev_ds](https://www.kaggle.com/kunihikofurugori/siim-step3-1-renew-imglev-ds)\n\n### [step3-2 renew-studylev_ds](https://www.kaggle.com/kunihikofurugori/siim-step3-1-renew-studylev-ds)","metadata":{}},{"cell_type":"code","source":"!pip install python-gdcm\n#!pip install pylibjpeg-libjpeg\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport cv2,random\nimport pydicom\nimport gdcm\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport matplotlib.pyplot as plt\nimport glob\nimport os\nfrom PIL import Image, ImageStat\nimport albumentations as A","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.mkdir(\"image640\")\nos.mkdir(\"image768\")","metadata":{"execution":{"iopub.status.busy":"2021-07-13T12:16:24.507729Z","iopub.execute_input":"2021-07-13T12:16:24.508026Z","iopub.status.idle":"2021-07-13T12:16:24.513512Z","shell.execute_reply.started":"2021-07-13T12:16:24.507996Z","shell.execute_reply":"2021-07-13T12:16:24.512451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"../input/step2-make-dataframe/train_image.csv\")\nprint(len(df))\ndf = df[~df.duplicated(keep='last',subset=['mean', 'var'])]\nprint(len(df))\ndf = df.sort_values(\"norm_var\",ascending=False)\ndf['fold'] = 0\ndf.fold = df.fold.apply(lambda x: random.randint(0,3))\ndf","metadata":{"execution":{"iopub.status.busy":"2021-07-13T12:16:24.536038Z","iopub.execute_input":"2021-07-13T12:16:24.536402Z","iopub.status.idle":"2021-07-13T12:16:24.704836Z","shell.execute_reply.started":"2021-07-13T12:16:24.536371Z","shell.execute_reply":"2021-07-13T12:16:24.703517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_transform(size):\n    transform = A.Compose([\n        #A.Rotate(p=1.0,limit=5),\n        A.RandomSizedCrop(\n                         min_max_height=[int(0.8*size), int(0.95*size)],\n                         height=size,\n                         width=size, \n                         p=1.0)\n    ],p=1.0,\n    bbox_params=A.BboxParams(\n                    format='albumentations',\n                    min_area=0, \n                    min_visibility=0,\n                    label_fields=['label'])\n    )\n    return transform\n\n\ndef exclude_blackflame(image, boxes, labels, size, trytimes=10):\n    best_var = image.std()\n    transform = get_transform(size)\n    for _ in range(trytimes):\n        sample = transform(image=image, bboxes=boxes, label=labels)\n        if sample['image'].std() < best_var:\n            best_var = sample['image'].std()\n            images = sample['image']\n            bboxes = sample['bboxes']\n    if best_var >= image.std():\n        images = image\n        bboxes = boxes\n    return images, bboxes","metadata":{"execution":{"iopub.status.busy":"2021-07-13T12:17:24.435011Z","iopub.execute_input":"2021-07-13T12:17:24.435683Z","iopub.status.idle":"2021-07-13T12:17:24.454023Z","shell.execute_reply.started":"2021-07-13T12:17:24.435628Z","shell.execute_reply":"2021-07-13T12:17:24.452264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data\n\ndef expand2square(pil_img, background_color):\n    width, height = pil_img.size\n    if width == height:\n        return pil_img\n    elif width > height:\n        result = Image.new(pil_img.mode, (width, width), background_color)\n        result.paste(pil_img, (0, (width - height) // 2))\n        return result\n    else:\n        result = Image.new(pil_img.mode, (height, height), background_color)\n        result.paste(pil_img, ((height - width) // 2, 0))\n        return result\n\ndef resize(array, size1,size2, keep_ratio=True, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size1, size2), resample)\n        im = expand2square(im, background_color=0)\n    else:\n        im = im.resize((size1, size2), resample)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-07-13T12:17:24.619015Z","iopub.execute_input":"2021-07-13T12:17:24.619622Z","iopub.status.idle":"2021-07-13T12:17:24.642356Z","shell.execute_reply.started":"2021-07-13T12:17:24.619563Z","shell.execute_reply":"2021-07-13T12:17:24.639977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs = []\ndfs.append(df[df.fold ==0])\ndfs.append(df[df.fold ==1])\ndfs.append(df[df.fold ==2])\ndfs.append(df[df.fold ==3])\nprint(len(dfs[0]),len(dfs[1]),len(dfs[2]),len(dfs[3]))\nprint(len(df))","metadata":{"execution":{"iopub.status.busy":"2021-07-13T12:17:25.018896Z","iopub.execute_input":"2021-07-13T12:17:25.019433Z","iopub.status.idle":"2021-07-13T12:17:25.038241Z","shell.execute_reply.started":"2021-07-13T12:17:25.019386Z","shell.execute_reply":"2021-07-13T12:17:25.035895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run(fold):\n    for idx,(index,row) in enumerate(dfs[fold].iterrows()):\n        image = read_xray(row.path)\n        boxes = np.zeros((row.boxnum,4))\n        labels =np.ones((boxes.shape[0],), dtype=np.int64)\n        for i in range(row.boxnum):\n            box = row[['fracxmin'+str(i+1), 'fracymin'+str(i+1), 'fracxmax'+str(i+1), 'fracymax'+str(i+1)]].values\n            boxes[i] = box\n        image = cv2.resize(image,(2048,2048),interpolation = cv2.INTER_AREA)\n        save_img = cv2.resize(image,(640,640),interpolation = cv2.INTER_AREA)\n        cv2.imwrite(\"image640\" + \"/\" + row.id + \".png\",save_img)\n        save_img = cv2.resize(image,(768,768),interpolation = cv2.INTER_AREA)\n        cv2.imwrite(\"image768\" + \"/\" + row.id + \".png\",save_img)\n                \n        #plt.imshow(images)\n        #plt.show()\n\n        if idx%500==0:\n            print(idx)","metadata":{"execution":{"iopub.status.busy":"2021-07-13T12:17:26.616965Z","iopub.execute_input":"2021-07-13T12:17:26.617332Z","iopub.status.idle":"2021-07-13T12:17:26.682216Z","shell.execute_reply.started":"2021-07-13T12:17:26.617303Z","shell.execute_reply":"2021-07-13T12:17:26.679884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#run(1)","metadata":{"execution":{"iopub.status.busy":"2021-07-13T12:17:26.687958Z","iopub.execute_input":"2021-07-13T12:17:26.690524Z","iopub.status.idle":"2021-07-13T12:17:26.713813Z","shell.execute_reply.started":"2021-07-13T12:17:26.690436Z","shell.execute_reply":"2021-07-13T12:17:26.711313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from joblib import Parallel, delayed\nParallel(n_jobs=4, backend=\"threading\")(delayed(run)(i) for i in range(4))","metadata":{"execution":{"iopub.status.busy":"2021-07-13T12:17:26.722409Z","iopub.execute_input":"2021-07-13T12:17:26.723117Z","iopub.status.idle":"2021-07-13T12:35:44.973119Z","shell.execute_reply.started":"2021-07-13T12:17:26.723025Z","shell.execute_reply":"2021-07-13T12:35:44.969756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip -qr image640.zip image640\n!rm -r image640\n!zip -qr image768.zip image768\n!rm -r image768","metadata":{"execution":{"iopub.status.busy":"2021-07-13T12:16:36.554464Z","iopub.status.idle":"2021-07-13T12:16:36.554934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv(\"train.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-07-13T12:16:36.556069Z","iopub.status.idle":"2021-07-13T12:16:36.556556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}