{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":26680,"databundleVersionId":2283525,"sourceType":"competition"}],"dockerImageVersionId":30096,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing Libraries and Reading CSV Files","metadata":{}},{"cell_type":"code","source":"import time\nsince = time.time()\n!pip3 install python-gdcm","metadata":{"execution":{"iopub.status.busy":"2024-07-10T08:21:09.981277Z","iopub.execute_input":"2024-07-10T08:21:09.981721Z","iopub.status.idle":"2024-07-10T08:21:19.657210Z","shell.execute_reply.started":"2024-07-10T08:21:09.981623Z","shell.execute_reply":"2024-07-10T08:21:19.656216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport seaborn as sn\nimport pydicom as dicom # Dicom (Digital Imaging in Medicine) - medical image datasets, storage and transfer\nimport os\nfrom tqdm import tqdm # allows you to output a smart progress bar by wrapping around any iterable\nimport glob # retrieve files/pathnames matching a specified pattern\nimport pprint # pretty-print” arbitrary Python data structures\nimport ast # \nfrom pydicom.pixel_data_handlers.util import apply_voi_lut #\nimport wandb #\n\nfrom PIL import Image\nimport shutil\nimport gdcm\n\npath = '/kaggle/input/siim-covid19-detection/'\ntrain_image_level = pd.read_csv(path + \"train_image_level.csv\")\ntrain_study_level = pd.read_csv(path + \"train_study_level.csv\")\n\ntrain_image_level.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-10T08:21:19.658944Z","iopub.execute_input":"2024-07-10T08:21:19.659259Z","iopub.status.idle":"2024-07-10T08:21:21.520294Z","shell.execute_reply.started":"2024-07-10T08:21:19.659219Z","shell.execute_reply":"2024-07-10T08:21:21.519220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_level.describe()","metadata":{"execution":{"iopub.status.busy":"2024-07-10T08:21:21.522736Z","iopub.execute_input":"2024-07-10T08:21:21.523170Z","iopub.status.idle":"2024-07-10T08:21:21.575077Z","shell.execute_reply.started":"2024-07-10T08:21:21.523116Z","shell.execute_reply":"2024-07-10T08:21:21.574141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n2040 ids without box","metadata":{}},{"cell_type":"code","source":"train_study_level_key = train_study_level.id.str[:-6]\ntraining_set = pd.merge(left = train_study_level, right = train_image_level, how = 'right', left_on = train_study_level_key, right_on = 'StudyInstanceUID')\nprint(training_set.shape)","metadata":{"execution":{"iopub.status.busy":"2024-07-10T08:21:21.576673Z","iopub.execute_input":"2024-07-10T08:21:21.576982Z","iopub.status.idle":"2024-07-10T08:21:21.599632Z","shell.execute_reply.started":"2024-07-10T08:21:21.576949Z","shell.execute_reply":"2024-07-10T08:21:21.598648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(training_set.loc[0])","metadata":{"execution":{"iopub.status.busy":"2024-07-10T08:21:21.601037Z","iopub.execute_input":"2024-07-10T08:21:21.601451Z","iopub.status.idle":"2024-07-10T08:21:21.610612Z","shell.execute_reply.started":"2024-07-10T08:21:21.601406Z","shell.execute_reply":"2024-07-10T08:21:21.609606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read Image in DCM format\n\n## Searching Desired StudyInstanceUID in all Subfolders","metadata":{}},{"cell_type":"code","source":"def read_dcm(i):\n    path_train = path + 'train/' + training_set.loc[i, 'StudyInstanceUID']\n    #print(os.listdir(path_train))\n    img_id = training_set.loc[i, 'id_y'].replace('_image','.dcm')\n    \n    for dirname, _, filenames in os.walk(path_train):\n        for filename in filenames:\n            path_img_id = os.path.join(dirname, filename)\n            if path_img_id[-16:-4] == img_id:\n                #print(path_img_id[-16:-4])\n                break\n      \n    last_folder_in_path = os.listdir(path_train)[0]\n    path_train = path_train + '/{}/'.format(last_folder_in_path)\n    data_file = dicom.dcmread(path_img_id)#(path_train + img_id)\n    return img_id, data_file\n\ndef Image_resize(img,len_x):\n    img = img.resize((len_x, len_x), Image.ANTIALIAS)\n    return img\n\n_,img = read_dcm(1)\nprint(img)","metadata":{"execution":{"iopub.status.busy":"2024-07-10T08:21:21.611987Z","iopub.execute_input":"2024-07-10T08:21:21.612402Z","iopub.status.idle":"2024-07-10T08:21:21.879268Z","shell.execute_reply.started":"2024-07-10T08:21:21.612341Z","shell.execute_reply":"2024-07-10T08:21:21.878146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DCM Image as Pixel Array","metadata":{}},{"cell_type":"code","source":"img_arr = img.pixel_array\nplt.imshow(img_arr)\nprint('Shape:', img_arr.shape)","metadata":{"execution":{"iopub.status.busy":"2024-07-10T08:21:21.880623Z","iopub.execute_input":"2024-07-10T08:21:21.880943Z","iopub.status.idle":"2024-07-10T08:21:22.750718Z","shell.execute_reply.started":"2024-07-10T08:21:21.880910Z","shell.execute_reply":"2024-07-10T08:21:22.749608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(img_arr.ravel()) #calculating histogram","metadata":{"execution":{"iopub.status.busy":"2024-07-10T08:21:22.753357Z","iopub.execute_input":"2024-07-10T08:21:22.753768Z","iopub.status.idle":"2024-07-10T08:21:23.017325Z","shell.execute_reply.started":"2024-07-10T08:21:22.753725Z","shell.execute_reply":"2024-07-10T08:21:23.016422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_,img = read_dcm(5) # Another Image\nimg_arr = img.pixel_array\n\nplt.imshow(img_arr)\nprint('Image Shape:',img_arr.shape)","metadata":{"execution":{"iopub.status.busy":"2024-07-10T08:21:23.018860Z","iopub.execute_input":"2024-07-10T08:21:23.019160Z","iopub.status.idle":"2024-07-10T08:21:24.565317Z","shell.execute_reply.started":"2024-07-10T08:21:23.019128Z","shell.execute_reply":"2024-07-10T08:21:24.564222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(img_arr.ravel()) ","metadata":{"execution":{"iopub.status.busy":"2024-07-10T08:21:24.566542Z","iopub.execute_input":"2024-07-10T08:21:24.566852Z","iopub.status.idle":"2024-07-10T08:21:24.872513Z","shell.execute_reply.started":"2024-07-10T08:21:24.566813Z","shell.execute_reply":"2024-07-10T08:21:24.871415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Observation\n### Some images have an intensity range of 0 to 255\n### Some images have a higher intensity range.(can be 16 bit images/ 12 bit allocated)","metadata":{}},{"cell_type":"markdown","source":"\n# Histogram Equalization + Converting to RGB Images","metadata":{}},{"cell_type":"code","source":"def To_16bit(img_arr):\n    min_arr = np.amin(img_arr)\n    max_arr = np.amax(img_arr)\n    range_array = max_arr - min_arr\n\n    return np.round((img_arr-min_arr)/range_array*(np.power(2,16)*3-1)) \n\n\ndef To_RGB(img_arr): #Extend to 24 bit then segment by 8 bit\n    min_arr = np.amin(img_arr)\n    max_arr = np.amax(img_arr)\n    range_array = max_arr - min_arr\n    \n    lenx, leny = img_arr.shape\n    rgbArray = np.zeros((lenx,leny,3), 'uint8')\n\n    arr2 = np.round((img_arr-min_arr)/range_array*(np.power(2,24)-1))\n    rgbArray[:,:, 0] = arr2 % np.power(2,8)\n    arr2 = (arr2-rgbArray[:,:, 0])/np.power(2,8)\n    rgbArray[:,:, 1] = arr2 % np.power(2,8)\n    arr2 = (arr2-rgbArray[:,:, 1])/np.power(2,8)\n    rgbArray[:,:, 2] = arr2 % np.power(2,8)\n    \n    return rgbArray\n    \ndef To_RGB2(img_arr): #Extend to 16 bit then segment by 8 bit top(G), 8 bit bottom(R), 8 bit overall(B)\n    min_arr = np.amin(img_arr)\n    max_arr = np.amax(img_arr)\n    range_array = max_arr - min_arr\n    \n    lenx, leny = img_arr.shape\n    rgbArray = np.zeros((lenx,leny,3), 'uint8')\n\n    arr2 = np.round((img_arr-min_arr)/range_array*(np.power(2,16)-1))\n    rgbArray[:,:, 0] = arr2 % np.power(2,8)                 #8 bit bottom\n    arr2 = np.floor(arr2/np.power(2,8))                     #8 bit top\n    rgbArray[:,:, 1] = arr2 \n    \n    rgbArray[:,:, 2] = np.round((img_arr-min_arr)/range_array*(np.power(2,8)-1)) #8 bit overall\n    \n    \n    return rgbArray    ","metadata":{"execution":{"iopub.status.busy":"2024-07-10T08:21:24.873896Z","iopub.execute_input":"2024-07-10T08:21:24.874231Z","iopub.status.idle":"2024-07-10T08:21:24.886511Z","shell.execute_reply.started":"2024-07-10T08:21:24.874196Z","shell.execute_reply":"2024-07-10T08:21:24.885470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rgb_arr = To_RGB2(img_arr)\nplt.imshow(rgb_arr)\nprint('Image Shape:', rgb_arr.shape)","metadata":{"execution":{"iopub.status.busy":"2024-07-10T08:21:24.887851Z","iopub.execute_input":"2024-07-10T08:21:24.888145Z","iopub.status.idle":"2024-07-10T08:21:26.328528Z","shell.execute_reply.started":"2024-07-10T08:21:24.888116Z","shell.execute_reply":"2024-07-10T08:21:26.327568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plotting\n\n### A few images are still grey (All of R,G,B components have the same value)","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(3,3, figsize=(20,16))\nfig.subplots_adjust(hspace=.1, wspace=.1)\naxes = axes.ravel()\n\nplt.rcParams['font.size'] = '16'\nplt.rcParams['figure.dpi'] = 150\nstart_index = 20\n\nfor row in range(9):\n    img_id, img = read_dcm(row + start_index)\n    img = img.pixel_array\n    #img = To_RGB2(img)\n    #im = Image.fromarray(img)\n    #im = Image_resize(im,512)\n    #img = np.array(im)\n    print(img_id, training_set.loc[row, 'label'].split(' ')[0])\n    #if (training_set.loc[row + start_index,'boxes'] == training_set.loc[row + start_index,'boxes']):\n    #    boxes = ast.literal_eval(training_set.loc[row + start_index,'boxes'])\n    #    for box in boxes:\n    #        p = matplotlib.patches.Rectangle((box['x'], box['y']),\n    #                                          box['width'], box['height'],\n    #                                          ec = 'r', fc = 'none', lw = 2.\n    #                                        )\n    #        axes[row].add_patch(p)\n    axes[row].imshow(img, cmap = 'gray')\n    axes[row].set_title('Class: '+training_set.loc[row, 'label'].split(' ')[0] + ',  Shape:' +str(img.shape))\n    axes[row].set_xticklabels([])\n    axes[row].set_yticklabels([])\nplt.savefig('raw_data.pdf')  ","metadata":{"execution":{"iopub.status.busy":"2024-07-10T08:21:26.329745Z","iopub.execute_input":"2024-07-10T08:21:26.330058Z","iopub.status.idle":"2024-07-10T08:21:48.633749Z","shell.execute_reply.started":"2024-07-10T08:21:26.330026Z","shell.execute_reply":"2024-07-10T08:21:48.632574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"time_elapsed = time.time() - since\nprint('Time from start {:.0f}m {:.0f}s'.format(time_elapsed // 60, time_elapsed % 60))","metadata":{"execution":{"iopub.status.busy":"2024-07-10T08:21:48.635279Z","iopub.execute_input":"2024-07-10T08:21:48.635674Z","iopub.status.idle":"2024-07-10T08:21:48.641815Z","shell.execute_reply.started":"2024-07-10T08:21:48.635638Z","shell.execute_reply":"2024-07-10T08:21:48.640644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folderlocation = './Images'\nif not os.path.exists(folderlocation):\n        os.mkdir(folderlocation)\n\nfor iter_name in ['train', 'val', 'test']:\n    folderlocation = './Images/' + iter_name\n    if not os.path.exists(folderlocation):\n            os.mkdir(folderlocation)\n            \n    folderlocation = './Images/' + iter_name +'/none'\n    if not os.path.exists(folderlocation):\n            os.mkdir(folderlocation)\n            \n    folderlocation = './Images/' + iter_name +'/opacity'\n    if not os.path.exists(folderlocation):\n            os.mkdir(folderlocation)\n            \n    folderlocation = './Images/' + iter_name +'/side'\n    if not os.path.exists(folderlocation):\n            os.mkdir(folderlocation)\n    \n","metadata":{"execution":{"iopub.status.busy":"2024-07-10T08:21:48.643363Z","iopub.execute_input":"2024-07-10T08:21:48.643753Z","iopub.status.idle":"2024-07-10T08:21:48.654204Z","shell.execute_reply.started":"2024-07-10T08:21:48.643716Z","shell.execute_reply":"2024-07-10T08:21:48.653126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"iter_split = 0\ntrain_test_split = 0\nfolderlocation = './Images/'\n\nbox=0 #Except the initial one previous box weiil be considered\n\nfor row in range(len(training_set)): # len(training_set)\n    img_id, img = read_dcm(row)\n    img = img.pixel_array\n    \n    iter_split = iter_split +1\n    \n    \n    # for some opacity images, only taking the opacity portion\n    if (training_set.loc[row,'boxes'] == training_set.loc[row,'boxes']):\n        if iter_split%10 < 2:\n            box = ast.literal_eval(training_set.loc[row,'boxes']) [0]\n            img = img[int(box['y']):int(box['y']+box['height']), int(box['x']):int(box['x']+box['width'])]\n        \n    # for some none images, only taking a region or potential opacity\n    if training_set.loc[row, 'label'].split(' ')[0] == 'none':\n        if iter_split%10 < 2 and box:\n            img = img[int(box['y']):int(box['y']+box['height']), int(box['x']):int(box['x']+box['width'])]\n            \n            \n    # Some images of sides     \n    if iter_split%8 == 0 and box:\n        if (training_set.loc[row,'boxes'] == training_set.loc[row,'boxes']):\n            box = ast.literal_eval(training_set.loc[row,'boxes']) [0]\n        img_2 = img[0:int(box['y']), int(box['x']):int(box['x']+box['width'])].copy()\n        \n    if iter_split%8 == 1 and box:\n        if (training_set.loc[row,'boxes'] == training_set.loc[row,'boxes']):\n            box = ast.literal_eval(training_set.loc[row,'boxes']) [0]\n        img_2 = img[0:int(box['y']), 0:int(box['x'])].copy()\n        \n    if iter_split%8 == 2 and box:\n        if (training_set.loc[row,'boxes'] == training_set.loc[row,'boxes']):\n            box = ast.literal_eval(training_set.loc[row,'boxes']) [0]\n        img_2 = img[int(box['y']):int(box['y']+box['height']), 0:int(box['x'])].copy()\n        \n    if iter_split%8 == 3 and box:\n        if (training_set.loc[row,'boxes'] == training_set.loc[row,'boxes']):\n            box = ast.literal_eval(training_set.loc[row,'boxes']) [0]\n        img_2 = img[int(box['y']+box['height']):img.shape[0], int(box['x']):int(box['x']+box['width'])].copy()\n        \n                        \n    if iter_split%8 <4 and box:\n        if img_2.shape[0] !=0 and img_2.shape[1]!=0:\n            img_destination_2 = folderlocation +'train/side/'+img_id[:-4] +'.jpeg'\n            \n            if train_test_split ==1:\n                img_destination_2 = folderlocation +'val/side/'+img_id[:-4] +'.jpeg'\n                train_test_split = train_test_split+1\n                \n            if train_test_split ==0:\n                img_destination_2 = folderlocation +'test/side/'+img_id[:-4] +'.jpeg'\n                train_test_split = train_test_split+1\n                \n            img_3 = To_RGB2(img_2)\n            im = Image.fromarray(img_3)\n            im = Image_resize(im,512)\n            im.save(img_destination_2,  cmap='gray')\n    if img.shape[0] !=0 and img.shape[1]!=0:\n        img = To_RGB2(img)    \n    \n        img_destination = folderlocation +'train/'+ training_set.loc[row, 'label'].split(' ')[0] +'/'+img_id[:-4] +'.jpeg'\n\n        if iter_split%20 == 19:\n            img_destination = folderlocation +'test/'+ training_set.loc[row, 'label'].split(' ')[0] +'/'+img_id[:-4] +'.jpeg'\n        if iter_split%20 == 18:\n            img_destination = folderlocation +'val/'+ training_set.loc[row, 'label'].split(' ')[0] +'/'+img_id[:-4] +'.jpeg'\n\n        im = Image.fromarray(img)\n        im = Image_resize(im,512)\n        im.save(img_destination,  cmap='gray')\n    \n    if row%500 == 499:\n        time_elapsed = time.time() - since\n        print('Time from start {:.0f}m {:.0f}s'.format(time_elapsed // 60, time_elapsed % 60))\n        print('Percentage Complete:', np.round(10000*(row+1)/6334)/100)\n    \n    \nprint('100% Complete.')    ","metadata":{"execution":{"iopub.status.busy":"2024-07-10T08:28:16.848498Z","iopub.execute_input":"2024-07-10T08:28:16.848859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shutil.make_archive('SIIM-FISABIO-RSNA-JPEG', 'zip', folderlocation)\nshutil.rmtree(folderlocation)","metadata":{"execution":{"iopub.status.busy":"2024-07-10T08:21:48.678932Z","iopub.status.idle":"2024-07-10T08:21:48.679590Z"},"trusted":true},"execution_count":null,"outputs":[]}]}