{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"> Resize images for classification task\n\nReferences : \n\n- [https://www.kaggle.com/xhlulu/siim-covid-19-convert-to-jpg-256px](https://www.kaggle.com/xhlulu/siim-covid-19-convert-to-jpg-256px)\n- [https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way](https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way)\n\ntrain.tar.gz consists of resized training images under respective label folders\n  - negative\n  - typical\n  - atypical\n  - indeterminate\n  ","metadata":{}},{"cell_type":"markdown","source":"<link href=\"https://fonts.googleapis.com/css?family=Merriweather:300,300i,400,400i,700,700i,900,900i\" rel='stylesheet' >\n<link href=\"https://fonts.googleapis.com/css?family=Source+Sans+Pro:300,300i,400,400i,700,700i\" rel='stylesheet' >\n<link href='http://fonts.googleapis.com/css?family=Source+Code+Pro:300,400' rel='stylesheet' >\n<style>\n@font-face {\n    font-family: \"Computer Modern\";\n    src: url('http://mirrors.ctan.org/fonts/cm-unicode/fonts/otf/cmunss.otf');\n}\n</style>","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport plotly.graph_objects as go\nfrom plotly.offline import iplot, init_notebook_mode\ninit_notebook_mode(connected=True)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom plotly.offline import iplot, init_notebook_mode\ninit_notebook_mode(connected=True)\nimport plotly_express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nfrom plotly.offline import init_notebook_mode\nimport plotly.io as pio\nfrom plotly.subplots import make_subplots\n# setting default template to plotly_white for all visualizations\npio.templates.default = \"plotly_white\"\n%matplotlib inline\nimport gc\n\nfrom colorama import Fore, Back, Style\n\ny_ = Fore.YELLOW\nr_ = Fore.RED\ng_ = Fore.GREEN\nb_ = Fore.BLUE\nm_ = Fore.MAGENTA\nc_ = Fore.CYAN\nres = Style.RESET_ALL\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2021-06-04T00:28:48.478306Z","iopub.execute_input":"2021-06-04T00:28:48.478656Z","iopub.status.idle":"2021-06-04T00:28:52.625255Z","shell.execute_reply.started":"2021-06-04T00:28:48.478584Z","shell.execute_reply":"2021-06-04T00:28:52.623707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install python-gdcm","metadata":{"execution":{"iopub.status.busy":"2021-06-04T00:28:52.627115Z","iopub.execute_input":"2021-06-04T00:28:52.62743Z","iopub.status.idle":"2021-06-04T00:29:05.34747Z","shell.execute_reply.started":"2021-06-04T00:28:52.627405Z","shell.execute_reply":"2021-06-04T00:29:05.346016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df = pd.read_csv('/kaggle/input/siim-covid19-detection/train_image_level.csv', index_col=None)\nstudy_df = pd.read_csv('/kaggle/input/siim-covid19-detection/train_study_level.csv', index_col=None)\npd.set_option('display.max_columns', None)  \npd.set_option('display.max_colwidth', None)\nimage_df.shape, study_df.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-04T00:29:16.670053Z","iopub.execute_input":"2021-06-04T00:29:16.670382Z","iopub.status.idle":"2021-06-04T00:29:16.710588Z","shell.execute_reply.started":"2021-06-04T00:29:16.670357Z","shell.execute_reply":"2021-06-04T00:29:16.709259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nall_files = []\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        all_files.append(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2021-06-04T00:29:17.922035Z","iopub.execute_input":"2021-06-04T00:29:17.922389Z","iopub.status.idle":"2021-06-04T00:29:54.986249Z","shell.execute_reply.started":"2021-06-04T00:29:17.922354Z","shell.execute_reply":"2021-06-04T00:29:54.985316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_files = [file for file in all_files if '/train/' in file]\ntest_files = [file for file in all_files if '/test/' in file] ","metadata":{"execution":{"iopub.status.busy":"2021-06-04T00:29:54.98836Z","iopub.execute_input":"2021-06-04T00:29:54.988688Z","iopub.status.idle":"2021-06-04T00:29:54.994236Z","shell.execute_reply.started":"2021-06-04T00:29:54.988661Z","shell.execute_reply":"2021-06-04T00:29:54.993434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_files),len(test_files)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T00:29:54.995728Z","iopub.execute_input":"2021-06-04T00:29:54.99596Z","iopub.status.idle":"2021-06-04T00:29:55.021594Z","shell.execute_reply.started":"2021-06-04T00:29:54.995937Z","shell.execute_reply":"2021-06-04T00:29:55.020079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pydicom import read_file\n\ndef show_image(img, figsize=None, ax=None, cmap=\"gray\"):\n    if not ax: \n        fig, ax = plt.subplots(figsize=figsize)\n    ax.imshow(img, cmap=cmap)\n    ax.get_xaxis().set_visible(False)\n    ax.get_yaxis().set_visible(False)\n\ndef get_image(file):\n    dicom = read_file(file, stop_before_pixels=False)\n    return dicom.pixel_array    \n\ndef get_dicom(file):\n    dicom = read_file(file, stop_before_pixels=False)\n    return dicom","metadata":{"execution":{"iopub.status.busy":"2021-06-04T00:36:48.585825Z","iopub.execute_input":"2021-06-04T00:36:48.586298Z","iopub.status.idle":"2021-06-04T00:36:48.593127Z","shell.execute_reply.started":"2021-06-04T00:36:48.586262Z","shell.execute_reply":"2021-06-04T00:36:48.591812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df","metadata":{"execution":{"iopub.status.busy":"2021-06-04T00:30:17.906549Z","iopub.execute_input":"2021-06-04T00:30:17.906841Z","iopub.status.idle":"2021-06-04T00:30:17.943749Z","shell.execute_reply.started":"2021-06-04T00:30:17.906816Z","shell.execute_reply":"2021-06-04T00:30:17.942557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_file(img_id):\n    imgs = [file for file in train_files if img_id in file]\n    return imgs[0]\nimage_df['img_id'] = image_df['id'].apply(lambda x: x.split('_')[0])\nimage_df['file'] = image_df['img_id'].apply(lambda x : find_file(x))","metadata":{"execution":{"iopub.status.busy":"2021-06-04T00:30:21.08877Z","iopub.execute_input":"2021-06-04T00:30:21.08913Z","iopub.status.idle":"2021-06-04T00:30:25.890771Z","shell.execute_reply.started":"2021-06-04T00:30:21.089099Z","shell.execute_reply":"2021-06-04T00:30:25.889643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_grp = pd.melt(study_df, id_vars=list(study_df.columns)[:1], value_vars=list(study_df.columns)[1:],\n             var_name='label', value_name='value')\nstudy_grp = study_grp.loc[study_grp['value']!=0]\nlbl_map = {'Negative for Pneumonia' : 'negative', 'Typical Appearance' : 'typical',\n       'Indeterminate Appearance' : 'indeterminate', 'Atypical Appearance' : 'atypical'}\n\nstudy_grp['StudyInstanceUID'] = study_grp['id'].apply(lambda x: x.split('_')[0])\nstudy_grp['label'] = study_grp['label'].apply(lambda x: lbl_map[x])\nstudy_map = dict(zip(study_grp.StudyInstanceUID, study_grp.label))","metadata":{"execution":{"iopub.status.busy":"2021-06-04T00:30:26.954391Z","iopub.execute_input":"2021-06-04T00:30:26.954937Z","iopub.status.idle":"2021-06-04T00:30:27.006134Z","shell.execute_reply.started":"2021-06-04T00:30:26.954895Z","shell.execute_reply":"2021-06-04T00:30:27.005082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_grp = image_df.groupby(['StudyInstanceUID'])['id'].count().reset_index()\nimg_grp","metadata":{"execution":{"iopub.status.busy":"2021-06-04T00:30:28.949952Z","iopub.execute_input":"2021-06-04T00:30:28.95026Z","iopub.status.idle":"2021-06-04T00:30:28.982283Z","shell.execute_reply.started":"2021-06-04T00:30:28.950229Z","shell.execute_reply":"2021-06-04T00:30:28.98066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-04T00:30:32.976884Z","iopub.execute_input":"2021-06-04T00:30:32.977253Z","iopub.status.idle":"2021-06-04T00:30:32.992894Z","shell.execute_reply.started":"2021-06-04T00:30:32.977207Z","shell.execute_reply":"2021-06-04T00:30:32.991815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df['class'] = image_df['StudyInstanceUID'].apply(lambda x: study_map[x])","metadata":{"execution":{"iopub.status.busy":"2021-06-04T00:30:36.498316Z","iopub.execute_input":"2021-06-04T00:30:36.498671Z","iopub.status.idle":"2021-06-04T00:30:36.509671Z","shell.execute_reply.started":"2021-06-04T00:30:36.498642Z","shell.execute_reply":"2021-06-04T00:30:36.508316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-04T00:30:38.842434Z","iopub.execute_input":"2021-06-04T00:30:38.842793Z","iopub.status.idle":"2021-06-04T00:30:38.860544Z","shell.execute_reply.started":"2021-06-04T00:30:38.842763Z","shell.execute_reply":"2021-06-04T00:30:38.85906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nimport cv2 \nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ndef resize_imagev_v0(img, image_size=(512,512)):\n    img = cv2.resize(img, image_size)\n    return img\n\ndef resize_image(dicom, image_size=(512,512)):\n    img = apply_voi_lut(dicom.pixel_array, dicom)\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = np.amax(img) - img\n    img = img - np.min(img)\n    img = img / np.max(img)\n    img = (img * 255).astype(np.uint8)    \n    img = cv2.resize(img, image_size)   \n    return img\n\nprint('Resizing train Images')\nBASE_PATH = '/kaggle/out_dir/train'\nos.makedirs(BASE_PATH, exist_ok=True)\nfor lbl in list(image_df['class'].unique()):\n    os.makedirs(BASE_PATH + '/' + lbl, exist_ok=True)\n    \ntrn_files_dict = image_df[['img_id','file','StudyInstanceUID','class']].set_index('img_id').T.to_dict()\nfor iid, rec in tqdm(trn_files_dict.items()):\n    out_file = BASE_PATH + '/' + rec['class'] + '/' + iid + '.jpg'\n    img = resize_image(get_dicom(rec['file']))\n    cv2.imwrite(out_file, img)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-04T00:37:38.813722Z","iopub.execute_input":"2021-06-04T00:37:38.814035Z","iopub.status.idle":"2021-06-04T00:37:39.722826Z","shell.execute_reply.started":"2021-06-04T00:37:38.814009Z","shell.execute_reply":"2021-06-04T00:37:39.721558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Resizing test Images')\nTEST_PATH = '/kaggle/out_dir/test'\nos.makedirs(TEST_PATH, exist_ok=True)\nfor file in tqdm(test_files):\n    out_file = TEST_PATH + '/' + file.split('/')[-1].split('.')[0] + '.jpg'\n    img = resize_image(get_dicom(file))\n    cv2.imwrite(out_file, img)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T00:41:52.303028Z","iopub.execute_input":"2021-06-04T00:41:52.303341Z","iopub.status.idle":"2021-06-04T00:41:53.172234Z","shell.execute_reply.started":"2021-06-04T00:41:52.303314Z","shell.execute_reply":"2021-06-04T00:41:53.171426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar -zcf train.tar.gz -C \"/kaggle/out_dir/train/\" .\n!tar -zcf test.tar.gz -C \"/kaggle/out_dir/test/\" .","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}