{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Resizing image for faster training"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport matplotlib.image as immg\nfrom pathlib import Path\nimport os\nimport gc\nimport cv2\nfrom tqdm.notebook import tqdm\nimport zipfile\nimport io\nfrom sklearn.decomposition import PCA","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_dir = Path('../input/ranzcr-clip-catheter-line-classification/train')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"df = pd.read_csv(\"../input/ranzcr-clip-catheter-line-classification/train.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"files = [str(x)+'.jpg' for x in df.StudyInstanceUID.tolist()]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"files[0]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"* lets see a sample image size"},{"metadata":{"trusted":true},"cell_type":"code","source":"img = immg.imread(img_dir/files[0]);img.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Sample Image"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(1,figsize=(20,8))\nplt.subplot(121)\nplt.imshow(img);\nplt.subplot(122)\nplt.imshow(img,cmap='gray');","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## So I will reduce and save to 512x512 of its size"},{"metadata":{"trusted":true},"cell_type":"code","source":"OUT_TRAIN = 'trainXray_512.zip'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_tot,x2_tot = [],[]\nwith zipfile.ZipFile(OUT_TRAIN, 'w') as img_out:\n    for idx in tqdm(range(len(files))):\n        name = str(files[idx])[:-4]\n        img = cv2.imread(str(img_dir/files[idx]), 1) \n        img = cv2.resize(img, (512, 512))\n        x_tot.append((img/255.0).mean())\n        x2_tot.append(((img/255.0)**2).mean()) \n        \n        img = cv2.imencode('.jpg',img)[1]\n        img_out.writestr(name + '.jpg', img)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Image stats"},{"metadata":{"trusted":true},"cell_type":"code","source":"img_avr =  np.array(x_tot).mean()\nimg_std =  np.sqrt(np.array(x2_tot).mean() - img_avr**2)\nprint('mean:',img_avr, ', std:', img_std)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Saving 256x256"},{"metadata":{"trusted":true},"cell_type":"code","source":"OUT_TRAIN = 'trainXray_256.zip'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_tot,x2_tot = [],[]\nwith zipfile.ZipFile(OUT_TRAIN, 'w') as img_out:\n    for idx in tqdm(range(len(files))):\n        name = str(files[idx])[:-4]\n        img = cv2.imread(str(img_dir/files[idx]), 1) \n        img = cv2.resize(img, (256, 256))\n        x_tot.append((img/255.0).mean())\n        x2_tot.append(((img/255.0)**2).mean()) \n        \n        img = cv2.imencode('.jpg',img)[1]\n        img_out.writestr(name + '.jpg', img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_avr =  np.array(x_tot).mean()\nimg_std =  np.sqrt(np.array(x2_tot).mean() - img_avr**2)\nprint('mean:',img_avr, ', std:', img_std)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Resizing 128x128"},{"metadata":{"trusted":true},"cell_type":"code","source":"OUT_TRAIN = 'trainXray_128.zip'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_tot,x2_tot = [],[]\nwith zipfile.ZipFile(OUT_TRAIN, 'w') as img_out:\n    for idx in tqdm(range(len(files))):\n        name = str(files[idx])[:-4]\n        img = cv2.imread(str(img_dir/files[idx]), 1) \n        img = cv2.resize(img, (128, 128))\n        x_tot.append((img/255.0).mean())\n        x2_tot.append(((img/255.0)**2).mean()) \n        \n        img = cv2.imencode('.jpg',img)[1]\n        img_out.writestr(name + '.jpg', img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img_avr =  np.array(x_tot).mean()\nimg_std =  np.sqrt(np.array(x2_tot).mean() - img_avr**2)\nprint('mean:',img_avr, ', std:', img_std)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}