{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport gc\n\nfrom colorama import Fore, Back, Style\n\ny_ = Fore.YELLOW\nr_ = Fore.RED\ng_ = Fore.GREEN\nb_ = Fore.BLUE\nm_ = Fore.MAGENTA\nc_ = Fore.CYAN\nres = Style.RESET_ALL\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nPATH = '/kaggle/input/siim-covid19-detection/'\nsubmission = pd.read_csv('/kaggle/input/siim-covid19-detection/sample_submission.csv', index_col=None)\nimage_df = pd.read_csv('/kaggle/input/siim-covid19-detection/train_image_level.csv', index_col=None)\nstudy_df = pd.read_csv('/kaggle/input/siim-covid19-detection/train_study_level.csv', index_col=None)\npd.set_option('display.max_columns', None)  \npd.set_option('display.max_colwidth', None)\nprint(f\"{y_}Train image level csv shape : {image_df.shape}{res}\\n{g_}Train study level csv shape : {study_df.shape}{res}\")","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:24:33.123266Z","iopub.execute_input":"2021-08-03T13:24:33.123642Z","iopub.status.idle":"2021-08-03T13:24:33.292837Z","shell.execute_reply.started":"2021-08-03T13:24:33.12361Z","shell.execute_reply":"2021-08-03T13:24:33.291286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install python-gdcm","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:28:14.21726Z","iopub.execute_input":"2021-08-03T13:28:14.217691Z","iopub.status.idle":"2021-08-03T13:28:26.082727Z","shell.execute_reply.started":"2021-08-03T13:28:14.217653Z","shell.execute_reply":"2021-08-03T13:28:26.081609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df = pd.read_csv('/kaggle/input/siim-covid19-detection/train_image_level.csv', index_col=None)\nstudy_df = pd.read_csv('/kaggle/input/siim-covid19-detection/train_study_level.csv', index_col=None)\npd.set_option('display.max_columns', None)  \npd.set_option('display.max_colwidth', None)\nimage_df.shape, study_df.shape","metadata":{"execution":{"iopub.status.busy":"2021-08-03T14:29:20.979584Z","iopub.execute_input":"2021-08-03T14:29:20.979997Z","iopub.status.idle":"2021-08-03T14:29:21.064593Z","shell.execute_reply.started":"2021-08-03T14:29:20.979962Z","shell.execute_reply":"2021-08-03T14:29:21.063058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nall_files = []\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        all_files.append(os.path.join(dirname, filename))\n        \ntrain_files = [file for file in all_files if '/train/' in file]\ntest_files = [file for file in all_files if '/test/' in file] \n\nlen(train_files),len(test_files)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:29:00.557987Z","iopub.execute_input":"2021-08-03T13:29:00.558535Z","iopub.status.idle":"2021-08-03T13:29:30.758923Z","shell.execute_reply.started":"2021-08-03T13:29:00.558499Z","shell.execute_reply":"2021-08-03T13:29:30.757701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pydicom import read_file\n\ndef show_image(img, figsize=None, ax=None, cmap=\"gray\"):\n    if not ax: \n        fig, ax = plt.subplots(figsize=figsize)\n    ax.imshow(img, cmap=cmap)\n    ax.get_xaxis().set_visible(False)\n    ax.get_yaxis().set_visible(False)\n\ndef get_image(file):\n    dicom = read_file(file, stop_before_pixels=False)\n    return dicom.pixel_array    \n\ndef get_dicom(file):\n    dicom = read_file(file, stop_before_pixels=False)\n    return dicom","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:29:30.760257Z","iopub.execute_input":"2021-08-03T13:29:30.760681Z","iopub.status.idle":"2021-08-03T13:29:31.058022Z","shell.execute_reply.started":"2021-08-03T13:29:30.76064Z","shell.execute_reply":"2021-08-03T13:29:31.056561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_file(img_id):\n    imgs = [file for file in train_files if img_id in file]\n    return imgs[0]\nimage_df['img_id'] = image_df['id'].apply(lambda x: x.split('_')[0])\nimage_df['file'] = image_df['img_id'].apply(lambda x : find_file(x))","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:29:39.055659Z","iopub.execute_input":"2021-08-03T13:29:39.056094Z","iopub.status.idle":"2021-08-03T13:29:42.356148Z","shell.execute_reply.started":"2021-08-03T13:29:39.056062Z","shell.execute_reply":"2021-08-03T13:29:42.354691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_grp = pd.melt(study_df, id_vars=list(study_df.columns)[:1], value_vars=list(study_df.columns)[1:],\n             var_name='label', value_name='value')\nstudy_grp = study_grp.loc[study_grp['value']!=0]\nlbl_map = {'Negative for Pneumonia' : 'negative', 'Typical Appearance' : 'typical',\n       'Indeterminate Appearance' : 'indeterminate', 'Atypical Appearance' : 'atypical'}\n\nstudy_grp['StudyInstanceUID'] = study_grp['id'].apply(lambda x: x.split('_')[0])\nstudy_grp['label'] = study_grp['label'].apply(lambda x: lbl_map[x])\nstudy_map = dict(zip(study_grp.StudyInstanceUID, study_grp.label))","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:29:56.588124Z","iopub.execute_input":"2021-08-03T13:29:56.588509Z","iopub.status.idle":"2021-08-03T13:29:56.640393Z","shell.execute_reply.started":"2021-08-03T13:29:56.588477Z","shell.execute_reply":"2021-08-03T13:29:56.639529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_grp = image_df.groupby(['StudyInstanceUID'])['id'].count().reset_index()\nimg_grp","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:30:09.232817Z","iopub.execute_input":"2021-08-03T13:30:09.233355Z","iopub.status.idle":"2021-08-03T13:30:09.269755Z","shell.execute_reply.started":"2021-08-03T13:30:09.233304Z","shell.execute_reply":"2021-08-03T13:30:09.26864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_df['class'] = image_df['StudyInstanceUID'].apply(lambda x: study_map[x])\n","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:30:25.59637Z","iopub.execute_input":"2021-08-03T13:30:25.596742Z","iopub.status.idle":"2021-08-03T13:30:25.606293Z","shell.execute_reply.started":"2021-08-03T13:30:25.596712Z","shell.execute_reply":"2021-08-03T13:30:25.605492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nimport cv2 \nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\ndef resize_imagev_v0(img, image_size=(512,512)):\n    img = cv2.resize(img, image_size)\n    return img\n\ndef resize_image(dicom, image_size=(512,512)):\n    img = apply_voi_lut(dicom.pixel_array, dicom)\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = np.amax(img) - img\n    img = img - np.min(img)\n    img = img / np.max(img)\n    img = (img * 255).astype(np.uint8)    \n    img = cv2.resize(img, image_size)   \n    return img\n\nprint('Resizing train Images')\nBASE_PATH = '/kaggle/out_dir/train'\nos.makedirs(BASE_PATH, exist_ok=True)\nfor lbl in list(image_df['class'].unique()):\n    os.makedirs(BASE_PATH + '/' + lbl, exist_ok=True)\n    \ntrn_files_dict = image_df[['img_id','file','StudyInstanceUID','class']].set_index('img_id').T.to_dict()\nfor iid, rec in tqdm(trn_files_dict.items()):\n    out_file = BASE_PATH + '/' + rec['class'] + '/' + iid + '.jpg'\n    img = resize_image(get_dicom(rec['file']))\n    cv2.imwrite(out_file, img)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:30:56.191898Z","iopub.execute_input":"2021-08-03T13:30:56.192499Z","iopub.status.idle":"2021-08-03T14:02:33.759404Z","shell.execute_reply.started":"2021-08-03T13:30:56.192464Z","shell.execute_reply":"2021-08-03T14:02:33.758209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Resizing test Images')\nTEST_PATH = '/kaggle/out_dir/test'\nos.makedirs(TEST_PATH, exist_ok=True)\nfor file in tqdm(test_files):\n    out_file = TEST_PATH + '/' + file.split('/')[-1].split('.')[0] + '.jpg'\n    img = resize_image(get_dicom(file))\n    cv2.imwrite(out_file, img)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T14:05:02.964331Z","iopub.execute_input":"2021-08-03T14:05:02.964734Z","iopub.status.idle":"2021-08-03T14:11:03.071982Z","shell.execute_reply.started":"2021-08-03T14:05:02.964692Z","shell.execute_reply":"2021-08-03T14:11:03.070973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar -zcf train.tar.gz -C \"/kaggle/out_dir/train/\" .\n!tar -zcf test.tar.gz -C \"/kaggle/out_dir/test/\" .","metadata":{"execution":{"iopub.status.busy":"2021-08-03T14:11:08.537545Z","iopub.execute_input":"2021-08-03T14:11:08.538073Z","iopub.status.idle":"2021-08-03T14:11:38.270441Z","shell.execute_reply.started":"2021-08-03T14:11:08.538026Z","shell.execute_reply":"2021-08-03T14:11:38.268458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}