{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Note that all images are stored in paths with the form `set`/`study`/`series`/`image`\n\nWe will try to extract this information into a dataframe. This will be needed when inferencing on models and preparing the submission files!","metadata":{}},{"cell_type":"code","source":"! conda install -c conda-forge gdcm -y","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:35:36.730999Z","iopub.execute_input":"2021-06-04T15:35:36.731381Z","iopub.status.idle":"2021-06-04T15:36:37.234901Z","shell.execute_reply.started":"2021-06-04T15:35:36.731298Z","shell.execute_reply":"2021-06-04T15:36:37.233901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nIMG_FORMAT = \".dcm\"\nIMG_PATHS = []\nIMAGE_IDS = []\nIMAGE_NAMES = []\nSETS = []\nSERIES = []\nSTUDIES = []\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if filename.endswith(IMG_FORMAT):\n            img_path = os.path.join(dirname, filename)\n            Splitted = img_path.split('/')\n            # print(Splitted)\n            img_name = os.path.basename(img_path)\n            img_id = img_name.rstrip(IMG_FORMAT)\n            series_name = Splitted[-2]\n            study_name = Splitted[-3]\n            set_name = Splitted[-4]\n            IMG_PATHS.append(img_path)\n            IMAGE_NAMES.append(img_name)\n            IMAGE_IDS.append(img_id)\n            SETS.append(set_name)\n            SERIES.append(series_name)\n            STUDIES.append(study_name)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-04T15:36:37.240375Z","iopub.execute_input":"2021-06-04T15:36:37.243054Z","iopub.status.idle":"2021-06-04T15:36:59.565984Z","shell.execute_reply.started":"2021-06-04T15:36:37.243014Z","shell.execute_reply":"2021-06-04T15:36:59.565168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ext = pd.DataFrame.from_dict({\"Image_Path\":IMG_PATHS,\n                             \"Image_Name\":IMAGE_NAMES,\n                             \"Image_ID\":IMAGE_IDS,\n                             \"Set_Name\": SETS,\n                             \"Series_Name\":SERIES,\n                             \"Study_Name\":STUDIES})\ndf_ext.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.567784Z","iopub.execute_input":"2021-06-04T15:36:59.568106Z","iopub.status.idle":"2021-06-04T15:36:59.604404Z","shell.execute_reply.started":"2021-06-04T15:36:59.568072Z","shell.execute_reply":"2021-06-04T15:36:59.603542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_lvl_pth = \"../input/siim-covid19-detection/train_image_level.csv\"\ndf_img = pd.read_csv(img_lvl_pth)\ndf_img.sort_values(by=['id'],inplace=True)\ndf_img.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.605892Z","iopub.execute_input":"2021-06-04T15:36:59.606242Z","iopub.status.idle":"2021-06-04T15:36:59.655125Z","shell.execute_reply.started":"2021-06-04T15:36:59.606208Z","shell.execute_reply":"2021-06-04T15:36:59.654243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_img.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.656407Z","iopub.execute_input":"2021-06-04T15:36:59.656754Z","iopub.status.idle":"2021-06-04T15:36:59.6654Z","shell.execute_reply.started":"2021-06-04T15:36:59.656718Z","shell.execute_reply":"2021-06-04T15:36:59.664506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"std_lvl_pth = \"../input/siim-covid19-detection/train_study_level.csv\"\ndf_std = pd.read_csv(std_lvl_pth)\ndf_std['id'] = df_std['id'].str.replace('_study',\"\")\ndf_std.rename({'id': 'StudyInstanceUID'},axis=1, inplace=True)\ndf_std.head(3)\n# df_std.sort_values(by=['StudyInstanceUID'],inplace=True)\ndf_std.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.667013Z","iopub.execute_input":"2021-06-04T15:36:59.667418Z","iopub.status.idle":"2021-06-04T15:36:59.696739Z","shell.execute_reply.started":"2021-06-04T15:36:59.667381Z","shell.execute_reply":"2021-06-04T15:36:59.695843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_pth = \"../input/siim-covid19-detection/sample_submission.csv\"\ndf_sub = pd.read_csv(sub_pth)\ndf_sub.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.697957Z","iopub.execute_input":"2021-06-04T15:36:59.698288Z","iopub.status.idle":"2021-06-04T15:36:59.715884Z","shell.execute_reply.started":"2021-06-04T15:36:59.698255Z","shell.execute_reply":"2021-06-04T15:36:59.715179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df_img.merge(df_std, on='StudyInstanceUID')\ndf.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.719537Z","iopub.execute_input":"2021-06-04T15:36:59.719773Z","iopub.status.idle":"2021-06-04T15:36:59.740939Z","shell.execute_reply.started":"2021-06-04T15:36:59.719749Z","shell.execute_reply":"2021-06-04T15:36:59.74006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from copy import deepcopy\ndf_train = deepcopy(df)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.74285Z","iopub.execute_input":"2021-06-04T15:36:59.743171Z","iopub.status.idle":"2021-06-04T15:36:59.74711Z","shell.execute_reply.started":"2021-06-04T15:36:59.743138Z","shell.execute_reply":"2021-06-04T15:36:59.746358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.loc[df_train['Negative for Pneumonia']==1, 'study_label'] = 'negative'\ndf_train.loc[df_train['Typical Appearance']==1, 'study_label'] = 'typical'\ndf_train.loc[df_train['Indeterminate Appearance']==1, 'study_label'] = 'indeterminate'\ndf_train.loc[df_train['Atypical Appearance']==1, 'study_label'] = 'atypical'\ndf_train.drop(['Negative for Pneumonia','Typical Appearance', 'Indeterminate Appearance', 'Atypical Appearance'], axis=1, inplace=True)\ndf_train['id'] = df_train['id'].str.replace('_image', IMG_FORMAT)\n\ndef get_label(string):\n    return string.split()[0]\n\ndf_train['image_label'] = df_train['label'].map(get_label)\ndf_train.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.748365Z","iopub.execute_input":"2021-06-04T15:36:59.748917Z","iopub.status.idle":"2021-06-04T15:36:59.795972Z","shell.execute_reply.started":"2021-06-04T15:36:59.748881Z","shell.execute_reply":"2021-06-04T15:36:59.795294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.columns","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.797121Z","iopub.execute_input":"2021-06-04T15:36:59.797478Z","iopub.status.idle":"2021-06-04T15:36:59.802912Z","shell.execute_reply.started":"2021-06-04T15:36:59.797425Z","shell.execute_reply":"2021-06-04T15:36:59.802135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"image_label\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.804129Z","iopub.execute_input":"2021-06-04T15:36:59.804684Z","iopub.status.idle":"2021-06-04T15:36:59.814631Z","shell.execute_reply.started":"2021-06-04T15:36:59.804633Z","shell.execute_reply":"2021-06-04T15:36:59.81376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ext.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.815865Z","iopub.execute_input":"2021-06-04T15:36:59.816578Z","iopub.status.idle":"2021-06-04T15:36:59.830396Z","shell.execute_reply.started":"2021-06-04T15:36:59.816542Z","shell.execute_reply":"2021-06-04T15:36:59.829532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"string = \"a29c5a68b07b\"\nstring.zfill(15)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.831738Z","iopub.execute_input":"2021-06-04T15:36:59.832066Z","iopub.status.idle":"2021-06-04T15:36:59.837693Z","shell.execute_reply.started":"2021-06-04T15:36:59.832033Z","shell.execute_reply":"2021-06-04T15:36:59.836769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.839002Z","iopub.execute_input":"2021-06-04T15:36:59.839326Z","iopub.status.idle":"2021-06-04T15:36:59.846966Z","shell.execute_reply.started":"2021-06-04T15:36:59.839293Z","shell.execute_reply":"2021-06-04T15:36:59.84593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ext[\"id\"] = df_ext[\"Image_Name\"].str.zfill(19)\ndf_ext.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.848474Z","iopub.execute_input":"2021-06-04T15:36:59.848816Z","iopub.status.idle":"2021-06-04T15:36:59.864782Z","shell.execute_reply.started":"2021-06-04T15:36:59.848781Z","shell.execute_reply":"2021-06-04T15:36:59.8639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ext.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.866085Z","iopub.execute_input":"2021-06-04T15:36:59.866429Z","iopub.status.idle":"2021-06-04T15:36:59.880387Z","shell.execute_reply.started":"2021-06-04T15:36:59.866396Z","shell.execute_reply":"2021-06-04T15:36:59.879535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.881698Z","iopub.execute_input":"2021-06-04T15:36:59.882017Z","iopub.status.idle":"2021-06-04T15:36:59.89675Z","shell.execute_reply.started":"2021-06-04T15:36:59.881985Z","shell.execute_reply":"2021-06-04T15:36:59.896046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Create a dummy df with NaN values for test ","metadata":{}},{"cell_type":"code","source":"df_ext[\"Set_Name\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.899611Z","iopub.execute_input":"2021-06-04T15:36:59.899842Z","iopub.status.idle":"2021-06-04T15:36:59.908556Z","shell.execute_reply.started":"2021-06-04T15:36:59.89982Z","shell.execute_reply":"2021-06-04T15:36:59.907681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_ext[df_ext[\"Set_Name\"]==\"test\"]\ndf_test.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.909885Z","iopub.execute_input":"2021-06-04T15:36:59.910515Z","iopub.status.idle":"2021-06-04T15:36:59.929713Z","shell.execute_reply.started":"2021-06-04T15:36:59.910481Z","shell.execute_reply":"2021-06-04T15:36:59.928979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.930672Z","iopub.execute_input":"2021-06-04T15:36:59.930994Z","iopub.status.idle":"2021-06-04T15:36:59.9432Z","shell.execute_reply.started":"2021-06-04T15:36:59.930961Z","shell.execute_reply":"2021-06-04T15:36:59.942347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fill df_train_test with nan data","metadata":{}},{"cell_type":"code","source":"df_train_test = deepcopy(df_train)\nCOLS = list(df_train_test.columns)\ndef subtract_lists(x,y):\n    \"\"\"Subtract Two Lists (List Difference)\"\"\"\n    return [item for item in x if item not in y]\ndef merge_list_to_dict(test_keys,test_values):\n    \"\"\"Using dictionary comprehension to merge two lists to dictionary\"\"\"\n    merged_dict = {test_keys[i]: test_values[i] for i in range(len(test_keys))}\n    return merged_dict\n# NAN_COLS = subtract_lists(COLS,[\"id\"])\nTO_ATTACH = merge_list_to_dict(COLS,[np.nan]*len(COLS))\nfor index, row in df_test.iterrows():\n    TO_ATTACH[\"id\"] = row[\"id\"]\n    df_train_test = df_train_test.append(TO_ATTACH, ignore_index = True)\ndf_train_test.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:36:59.944635Z","iopub.execute_input":"2021-06-04T15:36:59.945363Z","iopub.status.idle":"2021-06-04T15:37:04.607085Z","shell.execute_reply.started":"2021-06-04T15:36:59.945285Z","shell.execute_reply":"2021-06-04T15:37:04.606307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_test.tail(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:37:04.611787Z","iopub.execute_input":"2021-06-04T15:37:04.612046Z","iopub.status.idle":"2021-06-04T15:37:04.622535Z","shell.execute_reply.started":"2021-06-04T15:37:04.612019Z","shell.execute_reply":"2021-06-04T15:37:04.621604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_test.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:37:04.624775Z","iopub.execute_input":"2021-06-04T15:37:04.625125Z","iopub.status.idle":"2021-06-04T15:37:04.631776Z","shell.execute_reply.started":"2021-06-04T15:37:04.625088Z","shell.execute_reply":"2021-06-04T15:37:04.630809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:37:04.633345Z","iopub.execute_input":"2021-06-04T15:37:04.633714Z","iopub.status.idle":"2021-06-04T15:37:04.640683Z","shell.execute_reply.started":"2021-06-04T15:37:04.633677Z","shell.execute_reply":"2021-06-04T15:37:04.639667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:37:04.642152Z","iopub.execute_input":"2021-06-04T15:37:04.642514Z","iopub.status.idle":"2021-06-04T15:37:04.648955Z","shell.execute_reply.started":"2021-06-04T15:37:04.642481Z","shell.execute_reply":"2021-06-04T15:37:04.647921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape[0] + df_test.shape[0]","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:37:04.651127Z","iopub.execute_input":"2021-06-04T15:37:04.651368Z","iopub.status.idle":"2021-06-04T15:37:04.660524Z","shell.execute_reply.started":"2021-06-04T15:37:04.651346Z","shell.execute_reply":"2021-06-04T15:37:04.659441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_test[\"id\"] = df_train_test[\"id\"].str.zfill(19)\ndf_train_test.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:37:04.661606Z","iopub.execute_input":"2021-06-04T15:37:04.662059Z","iopub.status.idle":"2021-06-04T15:37:04.683297Z","shell.execute_reply.started":"2021-06-04T15:37:04.662025Z","shell.execute_reply":"2021-06-04T15:37:04.682537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ext[\"id\"] = df_ext[\"id\"].str.zfill(19)\ndf_ext.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:37:04.684352Z","iopub.execute_input":"2021-06-04T15:37:04.684826Z","iopub.status.idle":"2021-06-04T15:37:04.705223Z","shell.execute_reply.started":"2021-06-04T15:37:04.684792Z","shell.execute_reply":"2021-06-04T15:37:04.70456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_test.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:37:04.706376Z","iopub.execute_input":"2021-06-04T15:37:04.706741Z","iopub.status.idle":"2021-06-04T15:37:04.712003Z","shell.execute_reply.started":"2021-06-04T15:37:04.706698Z","shell.execute_reply":"2021-06-04T15:37:04.710895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ext.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:37:04.713398Z","iopub.execute_input":"2021-06-04T15:37:04.713809Z","iopub.status.idle":"2021-06-04T15:37:04.721793Z","shell.execute_reply.started":"2021-06-04T15:37:04.713776Z","shell.execute_reply":"2021-06-04T15:37:04.720982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df_ext.merge(df_train_test, on='id')\ndf.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:37:04.723324Z","iopub.execute_input":"2021-06-04T15:37:04.723955Z","iopub.status.idle":"2021-06-04T15:37:04.750178Z","shell.execute_reply.started":"2021-06-04T15:37:04.72392Z","shell.execute_reply":"2021-06-04T15:37:04.749297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"Set_Name\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:37:04.751518Z","iopub.execute_input":"2021-06-04T15:37:04.751867Z","iopub.status.idle":"2021-06-04T15:37:04.761364Z","shell.execute_reply.started":"2021-06-04T15:37:04.751833Z","shell.execute_reply":"2021-06-04T15:37:04.760489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:37:04.762659Z","iopub.execute_input":"2021-06-04T15:37:04.763138Z","iopub.status.idle":"2021-06-04T15:37:04.769079Z","shell.execute_reply.started":"2021-06-04T15:37:04.763102Z","shell.execute_reply":"2021-06-04T15:37:04.768199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Utility Functions","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfrom glob import glob\nfrom tqdm.notebook import tqdm\nimport matplotlib.pyplot as plt\nfrom skimage import exposure\nimport cv2\nimport warnings\nwarnings.filterwarnings('ignore')\nimport shutil \nimport tensorflow as tf\n%matplotlib inline\n\n\nimport matplotlib.pylab as pylab\nimport seaborn as sns\nimport pprint\nimport pydicom as dicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport wandb\n\nimport PIL\nfrom PIL import Image\nfrom colorama import Fore, Back, Style\nviz_counter=0\n\ndef create_dir(dir, v=1):\n    \"\"\"\n    Creates a directory without throwing an error if directory already exists.\n    dir : The directory to be created.\n    v : Verbosity\n    \"\"\"\n    if not os.path.exists(dir):\n        os.makedirs(dir)\n        if v:\n            print(\"Created Directory : \", dir)\n        return 1\n    else:\n        if v:\n            print(\"Directory already existed : \", dir)\n        return 0\n\nvoi_lut=True\nfix_monochrome=True\n\ndef dicom_dataset_to_dict(filename):\n    \"\"\"Credit: https://github.com/pydicom/pydicom/issues/319\n               https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\n    \"\"\"\n    \n    dicom_header = dicom.dcmread(filename) \n    \n    #====== DICOM FILE DATA ======\n    dicom_dict = {}\n    repr(dicom_header)\n    for dicom_value in dicom_header.values():\n        if dicom_value.tag == (0x7fe0, 0x0010):\n            #discard pixel data\n            continue\n        if type(dicom_value.value) == dicom.dataset.Dataset:\n            dicom_dict[dicom_value.name] = dicom_dataset_to_dict(dicom_value.value)\n        else:\n            v = _convert_value(dicom_value.value)\n            dicom_dict[dicom_value.name] = v\n      \n    del dicom_dict['Pixel Representation']\n    \n    #====== DICOM IMAGE DATA ======\n    # VOI LUT (if available by DICOM device) is used to transform raw DICOM data to \"human-friendly\" view\n    if voi_lut:\n        data = apply_voi_lut(dicom_header.pixel_array, dicom_header)\n    else:\n        data = dicom_header.pixel_array\n    # depending on this value, X-ray may look inverted - fix that:\n    if fix_monochrome and dicom_header.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    data = data - np.min(data)\n    data = data / np.max(data)\n    modified_image_data = (data * 255).astype(np.uint8)\n    \n    return dicom_dict, modified_image_data\n\ndef _sanitise_unicode(s):\n    return s.replace(u\"\\u0000\", \"\").strip()\n\ndef _convert_value(v):\n    t = type(v)\n    if t in (list, int, float):\n        cv = v\n    elif t == str:\n        cv = _sanitise_unicode(v)\n    elif t == bytes:\n        s = v.decode('ascii', 'replace')\n        cv = _sanitise_unicode(s)\n    elif t == dicom.valuerep.DSfloat:\n        cv = float(v)\n    elif t == dicom.valuerep.IS:\n        cv = int(v)\n    else:\n        cv = repr(v)\n    return cv\n\n\nimport os, fnmatch\ndef find(pattern, path):\n    \"\"\"Utility to find files wrt a regex search\"\"\"\n    result = []\n    for root, dirs, files in os.walk(path):\n        for name in files:\n            if fnmatch.fnmatch(name, pattern):\n                result.append(os.path.join(root, name))\n    return result\ndef props(arr):\n    print(\"Shape :\",arr.shape,\"Maximum :\",arr.max(),\"Minimum :\",arr.min(),\"Data Type :\",arr.dtype)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:37:04.770488Z","iopub.execute_input":"2021-06-04T15:37:04.771025Z","iopub.status.idle":"2021-06-04T15:37:10.225163Z","shell.execute_reply.started":"2021-06-04T15:37:04.770988Z","shell.execute_reply":"2021-06-04T15:37:10.223872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"/kaggle/input/siim-covid19-detection/test/00188a671292/3eb5a506ccf3/3dcdfc352a06.dcm\"\ndicom_dict, modified_image_data = dicom_dataset_to_dict(path)\nprops(modified_image_data)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:37:10.227055Z","iopub.execute_input":"2021-06-04T15:37:10.227654Z","iopub.status.idle":"2021-06-04T15:37:10.932536Z","shell.execute_reply.started":"2021-06-04T15:37:10.227613Z","shell.execute_reply":"2021-06-04T15:37:10.93148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Define your Required Reshape Size Here!\n\n### Get Original Image Shapes from Dicom Images","metadata":{}},{"cell_type":"code","source":"from tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:37:10.934015Z","iopub.execute_input":"2021-06-04T15:37:10.934396Z","iopub.status.idle":"2021-06-04T15:37:10.94036Z","shell.execute_reply.started":"2021-06-04T15:37:10.934352Z","shell.execute_reply":"2021-06-04T15:37:10.937934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nShapes that you wish to resize to\n\"\"\"\n\nShape_X = 512\nShape_Y = 512\nimage_id = []\ndim0 = []\ndim1 = []\nsplits = []\nimg_paths = []\n\nfor split in ['test', 'train']:\n    # save_dir = f'/kaggle/tmp/{split}/'\n    save_dir = f'/kaggle/working/resized_data/{split}/'\n    print(split)\n    os.makedirs(save_dir, exist_ok=True)\n    \n    for dirname, _, filenames in tqdm(os.walk(f'/kaggle/input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            fpath = os.path.join(dirname, file)\n            dicom_dict, modified_image_data = dicom_dataset_to_dict(fpath)\n            res = cv2.resize(modified_image_data,(Shape_Y,Shape_X)) # cv2 has this opposite\n            save_path = os.path.join(save_dir, file.replace('dcm', 'png'))\n            cv2.imwrite(save_path,res)\n            img_id = file.replace('.dcm', '')\n            image_id.append(img_id)\n            dim0.append(modified_image_data.shape[0])\n            dim1.append(modified_image_data.shape[1])\n            img_paths.append(fpath)\n            splits.append(split)\n\"\"\"\n2475/?\n12386/?\n07:34 | 5.38it/s\n36:51 | 8.13it/s\n\"\"\"\nprint(\"Generation Complete!\")\n","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:37:10.941811Z","iopub.execute_input":"2021-06-04T15:37:10.942174Z","iopub.status.idle":"2021-06-04T15:38:07.837974Z","shell.execute_reply.started":"2021-06-04T15:37:10.942126Z","shell.execute_reply":"2021-06-04T15:38:07.836191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport zipfile\nimport shutil\n\n#taken from : https://www.kaggle.com/xhlulu/recursion-2019-load-resize-and-save-images\n\ndef zip_and_remove(path):\n    ziph = zipfile.ZipFile(f'{path}.zip', 'w', zipfile.ZIP_DEFLATED)\n    \n    for root, dirs, files in os.walk(path):\n        for file in files:\n            file_path = os.path.join(root, file)\n            ziph.write(file_path)\n            os.remove(file_path)\n    \n    ziph.close()\n    shutil.rmtree(path)\nsave_dir = 'resized_data'\nzip_and_remove(save_dir)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:38:07.839222Z","iopub.status.idle":"2021-06-04T15:38:07.83981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_df = pd.DataFrame.from_dict({'Image_Path': img_paths, 'dim0': dim0, 'dim1': dim1})\nnew_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:38:07.841003Z","iopub.status.idle":"2021-06-04T15:38:07.841641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:38:07.842792Z","iopub.status.idle":"2021-06-04T15:38:07.843396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:38:07.844547Z","iopub.status.idle":"2021-06-04T15:38:07.845159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_df.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:38:07.846394Z","iopub.status.idle":"2021-06-04T15:38:07.847105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df = df.merge(new_df,on=\"Image_Path\")\nfinal_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:38:07.848227Z","iopub.status.idle":"2021-06-04T15:38:07.848935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:38:07.850024Z","iopub.status.idle":"2021-06-04T15:38:07.850655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df.tail()","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:38:07.851658Z","iopub.status.idle":"2021-06-04T15:38:07.852344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###  Rescaling Imgs : Bounding Box Rescaling Needed\n\nRemember that if you're trying to rescale images, the bounding boxes need to be reshaped as well!\n\n##### Note : The BBox have shifted due to resizing\n\nA rectangle defined via an anchor point xy and its width and height.\n\nThe rectangle extends from xy[0] to xy[0] + width in x-direction and from xy[1] to xy[1] + height in y-direction.\n\n```\n:                +------------------+\n:                |                  |\n:              height               |\n:                |                  |\n:               (xy)---- width -----+\n```","metadata":{}},{"cell_type":"code","source":"# import cv2\n# how to load a string to json\n# import ast\n# jsonobj = ast.literal_eval(str(hdf))\n\nfrom ast import literal_eval","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:38:07.853374Z","iopub.status.idle":"2021-06-04T15:38:07.854016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = len(final_df)\n# Already defined above : Shape_Y,Shape_X = 512,512\nNEW_BOXES = []\nfor i in range(n):\n    if type(final_df['boxes'][i])==str:\n        boxes = literal_eval(final_df['boxes'][i])\n        BIG_BOX = []\n        for box in boxes:\n            xbase,ybase = (box['x']*(Shape_Y/final_df['dim1'][i]), box['y']*(Shape_X/final_df['dim0'][i]))\n            new_width,new_height = box['width']*(Shape_Y/final_df['dim1'][i]), box['height']*(Shape_X/final_df['dim0'][i])\n            CURR_BOX = {\"x\": xbase,\n                        \"y\" : ybase,\n                        \"width\" : new_width,\n                        \"height\" : new_height}\n            BIG_BOX.append(CURR_BOX)\n    else:\n        BIG_BOX = \"\"\n    NEW_BOXES.append(str(BIG_BOX))","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:38:07.855082Z","iopub.status.idle":"2021-06-04T15:38:07.855687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df['corrected_boxes'] = NEW_BOXES","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:38:07.856693Z","iopub.status.idle":"2021-06-04T15:38:07.857308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correction Factors\nfinal_df['cfy'] = Shape_Y/final_df['dim1']\nfinal_df['cfx'] = Shape_X/final_df['dim0']","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:38:07.858435Z","iopub.status.idle":"2021-06-04T15:38:07.859068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install openpyxl","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_df.to_csv('Extracted_Study_Series_Img.csv',index=False)\nfinal_df.to_excel('Extracted_Study_Series_Img.xlsx',index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualize","metadata":{}},{"cell_type":"markdown","source":"Incoming in the next Notebook!","metadata":{}},{"cell_type":"code","source":"\"\"\"\nsubset_df = final_df[final_df[\"Set_Name\"]==\"train\"].sample(n=20,random_state=2021)\nsubset_df.head()\nfor path in subset_dcm_files:\n    dicom_dict, modified_image_data = dicom_dataset_to_dict(path)\n    res = cv2.bitwise_and(resized_image_data,resized_image_data,mask = pred_img_preprocessed)\n    fig, ax = plt.subplots(1, 3, figsize=(20, 12))\n    ax[0].imshow(resized_image_data, cmap=\"viridis\")\n    ax[0].axis('off')\n    ax[1].imshow(pred_img_preprocessed, cmap=\"viridis\")    \n    ax[1].axis('off')\n    ax[2].imshow(res, cmap=\"viridis\")    \n    ax[2].axis('off')\n    plt.savefig(str(viz_counter)+\".png\",dpi=300)\n    viz_counter+=1\n    cv2.imwrite(str(viz_counter)+\".png\",res)\n    viz_counter+=1\n    plt.show()\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2021-06-04T15:38:07.860171Z","iopub.status.idle":"2021-06-04T15:38:07.860774Z"},"trusted":true},"execution_count":null,"outputs":[]}]}