{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# make dataset\n\n#### I got the studylevel dataframe by [step2 make_dataframe](https://www.kaggle.com/kunihikofurugori/siim-step1-get-imginfo).\n\n#### Next step, I will make the dataset about study level.\n\n## my public notebook \n\n### [step1 get_imageinformation](https://www.kaggle.com/kunihikofurugori/siim-step1-get-imginfo).\n\n### [step2 make_dataframe](https://www.kaggle.com/kunihikofurugori/siim-step1-get-imginfo).\n\n### [step3-1 renew-imglev_ds](https://www.kaggle.com/kunihikofurugori/siim-step3-1-renew-imglev-ds)\n\n### [step3-2 renew-studylev_ds](https://www.kaggle.com/kunihikofurugori/siim-step3-1-renew-studylev-ds)","metadata":{}},{"cell_type":"markdown","source":"## study level strategy (pre augment image)\n\n### 1. exclude side black part\n### 2. adjust image histgram","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install python-gdcm\nimport numpy as np \nimport pandas as pd \nimport cv2,random\nimport pydicom\nimport gdcm\nfrom skimage.exposure import exposure, equalize_hist,equalize_adapthist\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport matplotlib.pyplot as plt\nimport glob\nimport os\nfrom PIL import Image, ImageStat\nimport albumentations as A","metadata":{"execution":{"iopub.status.busy":"2021-07-29T12:04:10.245843Z","iopub.execute_input":"2021-07-29T12:04:10.246224Z","iopub.status.idle":"2021-07-29T12:04:26.132413Z","shell.execute_reply.started":"2021-07-29T12:04:10.246191Z","shell.execute_reply":"2021-07-29T12:04:26.131241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.mkdir(\"image1024\")\nos.mkdir(\"augimage1024\")\n#os.mkdir(\"augcomparison\")","metadata":{"execution":{"iopub.status.busy":"2021-07-29T12:04:35.581717Z","iopub.execute_input":"2021-07-29T12:04:35.582129Z","iopub.status.idle":"2021-07-29T12:04:35.586064Z","shell.execute_reply.started":"2021-07-29T12:04:35.582094Z","shell.execute_reply":"2021-07-29T12:04:35.58505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"../input/step2-make-dataframe/train_study.csv\")\n\ndf[\"histflag\"] = 0\nindex = df[(df[\"mean\"]< 90) & (df[\"var\"]<40)].index\nprint(len(index))\ndf.loc[index,\"histflag\"] = 1\nprint(len(df))\ntargetmean, targetstd = df[\"mean\"].values.mean(),df[\"var\"].values.mean()\n#df = df.sort_values(\"norm_var\",ascending=False)\ndf['fold'] = 0\ndf.fold = df.fold.apply(lambda x: random.randint(0,3))\ndf","metadata":{"execution":{"iopub.status.busy":"2021-07-29T12:05:01.407663Z","iopub.execute_input":"2021-07-29T12:05:01.40804Z","iopub.status.idle":"2021-07-29T12:05:01.531061Z","shell.execute_reply.started":"2021-07-29T12:05:01.408011Z","shell.execute_reply":"2021-07-29T12:05:01.529986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targetmean, targetstd","metadata":{"execution":{"iopub.status.busy":"2021-07-29T12:10:38.707843Z","iopub.execute_input":"2021-07-29T12:10:38.708225Z","iopub.status.idle":"2021-07-29T12:10:38.71489Z","shell.execute_reply.started":"2021-07-29T12:10:38.708194Z","shell.execute_reply":"2021-07-29T12:10:38.713758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_transform(image_size):\n    return A.Compose(\n            [\n\n            A.CLAHE(clip_limit=(1,3),p=0.5),\n            A.RandomBrightnessContrast(\n                    brightness_limit=0.1, \n                    contrast_limit=0.1,\n                    p=1.0),\n            A.RandomGamma(gamma_limit = (90, 110),p=1.0)\n            ],p=1.0)\n\n            \ndef optimization_imagehist(row,image, image_crop, trytimes=100):\n    best_loss = targetmean**2 + targetstd**2\n    for _ in range(trytimes):\n        if image_crop:\n            if random.uniform(0,1)<0.5:\n                xmin, xmax = int(row.crop_xmin), int(row.crop_xmax)\n                ymin, ymax = int(row.crop_ymin), int(row.crop_ymax)\n                images = image[xmin:xmax,ymin:ymax]\n            else:\n                images = image\n        transform = get_transform((image.shape[0],image.shape[1]))\n        sample = transform(image=image)[\"image\"]\n        calc_loss = (sample.mean() - targetmean)**2 + (sample.std() - targetstd)**2\n        if calc_loss < best_loss:\n            best_loss = calc_loss\n            outimg = sample\n    return outimg","metadata":{"execution":{"iopub.status.busy":"2021-07-29T12:10:42.159559Z","iopub.execute_input":"2021-07-29T12:10:42.160171Z","iopub.status.idle":"2021-07-29T12:10:42.171105Z","shell.execute_reply.started":"2021-07-29T12:10:42.16012Z","shell.execute_reply":"2021-07-29T12:10:42.170284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_xray(path, voi_lut = True, fix_monochrome = True):\n    # Original from: https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    dicom = pydicom.read_file(path)\n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \n    # \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n               \n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n        \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n        \n    return data\n\ndef expand2square(pil_img, background_color):\n    width, height = pil_img.size\n    if width == height:\n        return pil_img\n    elif width > height:\n        result = Image.new(pil_img.mode, (width, width), background_color)\n        result.paste(pil_img, (0, (width - height) // 2))\n        return result\n    else:\n        result = Image.new(pil_img.mode, (height, height), background_color)\n        result.paste(pil_img, ((height - width) // 2, 0))\n        return result\n\ndef resize(array, size1,size2, keep_ratio=True, resample=Image.LANCZOS):\n    # Original from: https://www.kaggle.com/xhlulu/vinbigdata-process-and-resize-to-image\n    im = Image.fromarray(array)\n    \n    if keep_ratio:\n        im.thumbnail((size1, size2), resample)\n        im = expand2square(im, background_color=0)\n    else:\n        im = im.resize((size1, size2), resample)\n    \n    return im","metadata":{"execution":{"iopub.status.busy":"2021-07-29T12:10:42.437331Z","iopub.execute_input":"2021-07-29T12:10:42.437908Z","iopub.status.idle":"2021-07-29T12:10:42.450667Z","shell.execute_reply.started":"2021-07-29T12:10:42.43786Z","shell.execute_reply":"2021-07-29T12:10:42.449823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs = []\ndfs.append(df[df.fold ==0])\ndfs.append(df[df.fold ==1])\ndfs.append(df[df.fold ==2])\ndfs.append(df[df.fold ==3])\nprint(len(dfs[0]),len(dfs[1]),len(dfs[2]),len(dfs[3]))\nprint(len(df))","metadata":{"execution":{"iopub.status.busy":"2021-07-29T12:10:43.870415Z","iopub.execute_input":"2021-07-29T12:10:43.871104Z","iopub.status.idle":"2021-07-29T12:10:43.887077Z","shell.execute_reply.started":"2021-07-29T12:10:43.871053Z","shell.execute_reply":"2021-07-29T12:10:43.885764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run(fold):\n    for idx,(index,row) in enumerate(dfs[fold].iterrows()):\n        image = read_xray(row.path)\n        \n        \n\n        im = resize(image, size1=1024, size2 =1024, keep_ratio=False)\n        im.save(\"image1024\" + \"/\" + row.id + \".png\")\n        if row.histflag == 1:\n            image = optimization_imagehist(row,image, image_crop=False)\n        \n        if row.augflag == 1:\n            xmin, xmax = int(row.crop_xmin), int(row.crop_xmax)\n            ymin, ymax = int(row.crop_ymin), int(row.crop_ymax)\n            augimage = image[xmin:xmax,ymin:ymax]\n            \n            im = resize(augimage, size1=1024, size2 =1024, keep_ratio=False)\n            im.save(\"augimage1024\" + \"/\" + row.id + \".png\")\n            \n        if row.augflag == 0:\n            augimage = optimization_imagehist(row,image, image_crop=False)\n            im = resize(augimage, size1=1024, size2 =1024, keep_ratio=False)\n            im.save(\"augimage1024\" + \"/\" + row.id + \".png\")\n         \n        #if (image.mean() != augimage.mean()) | (image.std() != augimage.std()):\n            #fig, ax= plt.subplots(1,2)\n            #ax[0].imshow(image)\n            #ax[0].axis('off')\n            #ax[0].set_title(\"before\")\n            #ax[1].imshow(augimage)\n            #ax[1].axis('off')\n            #plt.title(f\"img_path:{row.id}\", x=0, y=1.3, size=15)\n            #plt.show()\n            #fig.savefig(\"augcomparison/\" + row.id +\".png\",dpi=80)\n            #plt.close()\n            #del fig, ax\n            \n        \n\n        if idx%500==0:\n            print(idx)","metadata":{"execution":{"iopub.status.busy":"2021-07-29T12:16:16.033381Z","iopub.execute_input":"2021-07-29T12:16:16.033765Z","iopub.status.idle":"2021-07-29T12:16:16.047007Z","shell.execute_reply.started":"2021-07-29T12:16:16.033733Z","shell.execute_reply":"2021-07-29T12:16:16.045763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#run(0)","metadata":{"execution":{"iopub.status.busy":"2021-07-29T12:16:43.135311Z","iopub.execute_input":"2021-07-29T12:16:43.135829Z","iopub.status.idle":"2021-07-29T12:16:43.139668Z","shell.execute_reply.started":"2021-07-29T12:16:43.135778Z","shell.execute_reply":"2021-07-29T12:16:43.138962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from joblib import Parallel, delayed\nParallel(n_jobs=4, backend=\"threading\")(delayed(run)(i) for i in range(4))","metadata":{"execution":{"iopub.status.busy":"2021-07-29T12:05:30.135024Z","iopub.status.idle":"2021-07-29T12:05:30.13556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip -qr image1024.zip image1024\n!rm -r image1024\n!zip -qr augimage1024.zip augimage1024\n!rm -r augimage1024\n#!zip -qr augcomparison.zip augcomparison\n#!rm -r augcomparison\n\n","metadata":{"execution":{"iopub.status.busy":"2021-07-29T12:17:16.183001Z","iopub.execute_input":"2021-07-29T12:17:16.183395Z","iopub.status.idle":"2021-07-29T12:17:20.83044Z","shell.execute_reply.started":"2021-07-29T12:17:16.183355Z","shell.execute_reply":"2021-07-29T12:17:20.829359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv(\"train.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-07-29T11:46:20.056961Z","iopub.status.idle":"2021-07-29T11:46:20.057348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}