{"cells":[{"metadata":{"_uuid":"620385c816bb2477c2a3456bd317433208e8489a"},"cell_type":"markdown","source":"##  **I.  Introduction:**\n\nThis kernel preprocesses RSNA Stage 2 inputs as follows:\n\n1.  DICOM images are converted into *.jpg files of shape (1024, 1024, 3)\n2.  Bounding Boxes for individual 'Lung Opacity' cases are saved in individual *.txt files in YOLOV3 format next to the *.jpg files.\n3.  RSNA metadata is saved.  Various collections of keys I could think of as being useful are saved in a *.npz archive.  A RSNA patient dictionary keyed by patientid and containing key DICOM attributes and bounding boxes in lists are saved in a *.h5 file.\n\nI pulled this kernel together and tested it on Kaggle (flags below allow for that), converted the notebook into a *.py file, and did the actual conversion on a CPU on the Google Cloud Platform (GCP).  I used the Kaggle API to download the competition files to GCP and upload the output datasets (releasing in parallel) back to Kaggle.\n\nComments welcome!"},{"metadata":{"_uuid":"145c8aadca1f018dd7d15fce695bd5eb551989bc"},"cell_type":"markdown","source":"## **II. Setup**"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os, random\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom PIL import Image\nimport pydicom\nimport glob\nimport h5py","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a24357ac0372476f6f526bafa151a2fb7f92365a"},"cell_type":"code","source":"# flags\nAT_KAGGLE=True\nQUICK_PROCESS=True\nQUICK_PROCESS_SIZE=16 # used if QUICK_PROCESS is True\nPROCESS=\"All\" #[\"All\", \"Images\", \"Yolov3Labels\",\"MetaData\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5a3feae33c6a27a76cf2535464e0ba620b173a3a"},"cell_type":"code","source":"# global variables (stage 2 variables do not have a stage prefix)\nTRAIN_DIR=\"../input/rsna-pneumonia-detection-challenge/stage_2_train_images\"\nSTAGE1_DETAILED_CLASSES_CSV_FILE=\"../input/rsna-stage1-archived-inputs/stage_1_detailed_class_info.csv\"\nDETAILED_CLASSES_CSV_FILE=\"../input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv\"\nDETAILED_CLASSES_CSV_COLUMN_NAMES=['patientId' , 'class']\n# dictionary to map string classes to numerical\nCLASSES_DICT={'Normal': 0, 'Lung Opacity' : 1, 'No Lung Opacity / Not Normal' : 2}\n\nSTAGE1_TRAIN_LABELS_CSV_FILE=\"../input/rsna-stage1-archived-inputs/stage_1_train_labels.csv\"\nTRAIN_LABELS_CSV_FILE=\"../input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv\"\n# pedantic nit: we are changing 'Target' to 'label' on the way in\nTRAIN_LABELS_CSV_COLUMN_NAMES=['patientId', 'x1', 'y1', 'bw', 'bh', 'label']\n\n# saved test ids from stage1\nSTAGE1_TEST_IDS_FILE=\"../input/rsna-stage1-archived-inputs/stage1_test_ids.npy\"\n\nTEST_DIR=\"../input/rsna-pneumonia-detection-challenge/stage_2_test_images\"\n# list of test images\nTEST_LIST=sorted(os.listdir(TEST_DIR))\n\nSAVED_KEYS_FILE=\"rsna-stage1-and-stage2-keys.npz\"\nSAVED_PATIENTDICT_FILE=\"rsna-patientdict.h5\"\n\nDICOM_IMAGE_SIZE=1024","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"85571cc1a0f41217114d2743738acac2b9daaf61"},"cell_type":"code","source":"# setup output directories\n# directory where we will put processed inputs\nif not AT_KAGGLE:\n    processedtraininputsdirectory=\"../input/rsna-stage2-processed-train-inputs\"\n    processedtestinputsdirectory=\"../input/rsna-stage2-processed-test-inputs\"\n    processedmetadatadirectory=\"../input/rsna-stage2-processed-metadata-inputs\"\nelse: # we are at Kaggle (have to write to the current directory to keep the Commit engine happy)\n    processedtraininputsdirectory=\"./rsna-stage2-processed-train-inputs\"\n    processedtestinputsdirectory=\"./rsna-stage2-processed-test-inputs\"\n    processedmetadatadirectory=\"./rsna-stage2-processed-metadata-inputs\"\n    \n# create directories (one-time)\nos.makedirs(processedtraininputsdirectory, exist_ok=False)\nos.makedirs(processedtestinputsdirectory, exist_ok=False)\nos.makedirs(processedmetadatadirectory, exist_ok=False)\n\nprint (\"Preprocessing training inputs into directory: {}\".format(processedtraininputsdirectory))\nprint (\"Preprocessing test inputs into directory: {}\".format(processedtestinputsdirectory))\nprint (\"Preprocessing meta data into directory: {}\".format(processedmetadatadirectory))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cfa6a40e776535fa0925dd402b2b075051d61b35"},"cell_type":"code","source":"# read STAGE1_DETAILED_CLASSES_CSV_FILE into a pandas dataframe\nclassesdf = pd.read_csv(STAGE1_DETAILED_CLASSES_CSV_FILE,\n                        names=DETAILED_CLASSES_CSV_COLUMN_NAMES,\n                        # skip the header line\n                        header=0,\n                        # index the dataframe on patientId\n                        index_col='patientId')\n#print (classesdf.shape)\n#print (classesdf.head(n=10))\n\n# remove duplicates\nclassesdf=classesdf.groupby(['patientId'])['class'].first()\n# make classes numerical based on CLASSES_DICT\nclassesdf=pd.DataFrame(classesdf.replace(to_replace=CLASSES_DICT), columns=['class'])\nprint (\"Stage 1:: {} lines read from {}\".format(len(classesdf), STAGE1_DETAILED_CLASSES_CSV_FILE))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"27526988024680e9c32ee086344bf89ff9ee1836"},"cell_type":"code","source":"# read list of stage1 test images\nstage1testkeys=sorted(list(np.load(STAGE1_TEST_IDS_FILE)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cd7a5e64eec9a7a477ca2dec191d8f4e2c0ef4dd"},"cell_type":"code","source":"# capture stage1 patientids for different classes\nstage1allkeys=classesdf.index.tolist()\nstage1lungopacitykeys=classesdf.index[classesdf['class']==1].tolist()\nstage1normalkeys=classesdf.index[classesdf['class']==0].tolist()\nstage1otherabnormalkeys=classesdf.index[classesdf['class']==2].tolist()\nprint (\"################STAGE 1 SUMMARY################\")\nprint (\"Total Training Samples: {}\".format(len(stage1allkeys)))\nprint (\">>Lung Opacity Samples: {}\".format(len(stage1lungopacitykeys)))\nprint (\">>Normal Samples: {}\".format(len(stage1normalkeys)))\nprint (\">>Not Normal / No Lung Opacity Samples: {}\".format(len(stage1otherabnormalkeys)))\nprint (\"Test Samples: {}\".format(len(stage1testkeys)))\nprint (\"##############################################\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"73e55406447f2be4ef3cb78d391993b0652a0a29"},"cell_type":"code","source":"# read stage2 DETAILED_CLASSES_CSV_FILE into a pandas dataframe\nclassesdf = pd.read_csv(DETAILED_CLASSES_CSV_FILE,\n                        names=DETAILED_CLASSES_CSV_COLUMN_NAMES,\n                        # skip the header line\n                        header=0,\n                        # index the dataframe on patientId\n                        index_col='patientId')\n#print (classesdf.shape)\n#print (classesdf.head(n=10))\n\n# remove duplicates\nclassesdf=classesdf.groupby(['patientId'])['class'].first()\n# make classes numerical based on CLASSES_DICT\nclassesdf=pd.DataFrame(classesdf.replace(to_replace=CLASSES_DICT), columns=['class'])\nprint (\"Stage 2:: {} lines read from {}\".format(len(classesdf), DETAILED_CLASSES_CSV_FILE))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"45a34fb0507d699f54d95c173a5b39db3bb20f06"},"cell_type":"code","source":"# capture stage2 test keys\ntestkeys=[]\nfor filename in TEST_LIST:\n    key=filename.split(\".\")[0]\n    testkeys.append(key)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"27458ca9a7176bab91ef8b6dc62d228f94508b36"},"cell_type":"code","source":"# capture stage2 patientids for different classes\nallkeys=classesdf.index.tolist()\nlungopacitykeys=classesdf.index[classesdf['class']==1].tolist()\nnormalkeys=classesdf.index[classesdf['class']==0].tolist()\notherabnormalkeys=classesdf.index[classesdf['class']==2].tolist()\nprint (\"################STAGE 2 SUMMARY################\")\nprint (\"Total Training Samples: {}\".format(len(allkeys)))\nprint (\">>Lung Opacity Samples: {}\".format(len(lungopacitykeys)))\nprint (\">>Normal Samples: {}\".format(len(normalkeys)))\nprint (\">>Not Normal / No Lung Opacity Samples: {}\".format(len(otherabnormalkeys)))\nprint (\"Test Samples: {}\".format(len(testkeys)))\nprint (\"##############################################\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c61e54cd7e7164d2bd25dfeef11d2a5d64d62a67"},"cell_type":"code","source":"print (\"{} test samples from Stage 1 were distributed into Stage 2 as:\".format(len(stage1testkeys)))\nprint (\">>{} additional Lung Opacity Samples\".format(len(lungopacitykeys)-len(stage1lungopacitykeys)))\nprint (\">>{} additional Normal Samples\".format(len(normalkeys)-len(stage1normalkeys)))\nprint (\">>{} additional Not Normal / No Lung Opacity Samples\".format(len(otherabnormalkeys)-len(stage1otherabnormalkeys)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"64c0baaddf2f51751369593990cfa910c548541b"},"cell_type":"code","source":"# check stage2 vs stage1 keys\nassert sorted(allkeys)==sorted(stage1normalkeys+stage1lungopacitykeys+stage1otherabnormalkeys+stage1testkeys), \"Keys Mismatch\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cb86aaab79e6bc7b4a87ad6512ab238f2fe56cef"},"cell_type":"code","source":"# read TRAIN_LABELS_CSV_FILE into a pandas dataframe\nlabelsbboxdf = pd.read_csv(TRAIN_LABELS_CSV_FILE,\n                           names=TRAIN_LABELS_CSV_COLUMN_NAMES,\n                           # skip the header line\n                           header=0,\n                           # index the dataframe on patientId\n                           index_col='patientId')\n\nlabelsbboxdf.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a8b9176dd2e52397706af80c5864149b33b87f81"},"cell_type":"code","source":"# compute and store bounding box centers\nbx=labelsbboxdf['x1']+labelsbboxdf['bw']/2\nby=labelsbboxdf['y1']+labelsbboxdf['bh']/2\nlabelsbboxdf=labelsbboxdf.assign(bx=bx, by=by)\nlabelsbboxdf.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a39326561e63c6fd792e743056d0fcd42f936ac1"},"cell_type":"code","source":"# drop labels and rearrange dataframe so we have bounding boxes in rsna format,\n# dropping all rows other than lungopacity ones\nrsnabboxesdf=labelsbboxdf[['x1', 'y1', 'bw', 'bh']].dropna()\nrsnabboxesdf.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a7a359e1556dfd5a343a013a95aeaac142757c4d"},"cell_type":"code","source":"# drop labels and top left coordinates and rearrange dataframe in yolov3 format,\n# dropping all rows other than lungopacity ones\nyolov3bboxesdf=labelsbboxdf[['bx', 'by', 'bw', 'bh']].dropna()\nyolov3bboxesdf.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9b536b0a02018b492a06b2a0d3dcfae144978012"},"cell_type":"code","source":"# yolov3 requires bounding box dimensions to be between 0 and 1\n# normalize to DICOM_IMAGE_SIZE\nyolov3bboxesdf=yolov3bboxesdf/DICOM_IMAGE_SIZE\nyolov3bboxesdf.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d970ea9592dce77bf558b3d6e295d9b3d4147e8b"},"cell_type":"code","source":"# save a copy of keys we are going to munge up when running quick checks\nif QUICK_PROCESS==True:\n    savedallkeys=allkeys\n    savedtestkeys=testkeys\n    savedlungopacitykeys=lungopacitykeys","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"47cba328f2d672a416a717d076b35a7b352c3cc7"},"cell_type":"code","source":"# setup a quick test to make sure everything is working before heading off to GCP\nif QUICK_PROCESS == True:\n    allkeys=random.sample(allkeys, QUICK_PROCESS_SIZE)\n    testkeys=random.sample(testkeys, QUICK_PROCESS_SIZE)\n    lungopacitykeys=random.sample(lungopacitykeys, QUICK_PROCESS_SIZE)\n    print (\"Quick check by preprocessing {} samples\".format(QUICK_PROCESS_SIZE))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f90e8d479a197d49b2c309bcebd965fae34f7dff"},"cell_type":"markdown","source":"## III.  Preprocess Images"},{"metadata":{"trusted":true,"_uuid":"9a1fc775159375ba2edb39eaa34e21703523df00"},"cell_type":"code","source":"def loadDicomImage (directory, patientid, mode=\"metadata\"):\n    imagergb=np.zeros([DICOM_IMAGE_SIZE, DICOM_IMAGE_SIZE, 3])\n    attributes=[]\n    filename=\"{}.dcm\".format(patientid)\n    \n    if mode==\"metadata\":\n        # load patient meta-data only from file\n        patientdata = pydicom.dcmread(os.path.join(directory, filename), stop_before_pixels=True)\n    elif mode==\"image\":\n        # load patient meta-data and image from file\n        patientdata = pydicom.dcmread(os.path.join(directory, filename))\n        imagegray=patientdata.pixel_array\n        # convert grayscale to rgb\n        imagegray=imagegray/imagegray.max()\n        imagegray = (255*imagegray).clip(0, 255).astype(np.uint8)\n        imagergb=np.stack([imagegray]*3, -1)\n    # make sure there isn't a mismatch\n    assert patientid==patientdata.PatientID, \"PatientId Mismatch\"\n    # grab attributes\n    attributes.append(patientdata.PatientSex)\n    attributes.append(patientdata.PatientAge)\n    attributes.append(patientdata.ViewPosition)\n    \n    #print (imagergb)\n    return attributes, imagergb","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2a2b821d33d4d67025a4868e324cac829017e117"},"cell_type":"code","source":"# save jpg images for train samples in original size\nif PROCESS == \"All\"  or PROCESS==\"Images\":\n    for patientid in tqdm(allkeys):\n        imagefilename=\"{}.jpg\".format(patientid)\n        imagepathname=os.path.join(processedtraininputsdirectory, imagefilename)\n        #print (imagepathname)\n        _, imagergb = loadDicomImage (TRAIN_DIR, patientid, mode=\"image\")\n        image=Image.fromarray(imagergb)\n        assert image.size==(DICOM_IMAGE_SIZE, DICOM_IMAGE_SIZE), \"Input Image Size Mismatch\"\n        image.save(imagepathname)\n        \n    # make sure all images made it through correctly\n    processedtrainkeys=[]\n    for filename in glob.glob(processedtraininputsdirectory+'/*.jpg'):\n        key=os.path.split(filename)[1].split(\".\")[0]\n        processedtrainkeys.append(key)\n    assert sorted(processedtrainkeys)==sorted(allkeys), \"Train Samples Missed\"\n    print (\"Preprocessed {} train images\".format(len(processedtrainkeys)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aa9115a227e098509e57b0e22ac43488a862e3b9"},"cell_type":"code","source":"# save jpg images for test inputs in original size\nif PROCESS == \"All\"  or PROCESS==\"Images\":\n    for patientid in tqdm(testkeys):\n        imagefilename=\"{}.jpg\".format(patientid)\n        imagepathname=os.path.join(processedtestinputsdirectory, imagefilename)\n        #print (imagepathname)\n        _, imagergb = loadDicomImage (TEST_DIR, patientid, mode=\"image\")\n        image=Image.fromarray(imagergb)\n        assert image.size==(DICOM_IMAGE_SIZE, DICOM_IMAGE_SIZE), \"Input Image Size Mismatch\"\n        image.save(imagepathname)\n        \n    # make sure all images made it through correctly\n    processedtestkeys=[]\n    for filename in glob.glob(processedtestinputsdirectory+'/*.jpg'):\n        key=os.path.split(filename)[1].split(\".\")[0]\n        processedtestkeys.append(key)\n    assert sorted(processedtestkeys)==sorted(testkeys), \"Test Samples Missed\"\n    print (\"Preprocessed {} test images\".format(len(processedtestkeys)))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1e9cd30dcfd1b0f4913544f4784bc24e2fbddeed"},"cell_type":"markdown","source":"## IV.  Preprocess YOLOV3 Labels"},{"metadata":{"trusted":true,"_uuid":"680f67986885e6847ec1f75e7eedee5d6a370f36"},"cell_type":"code","source":"# get yolov3 bounding boxes by patientid\ndef getyolov3BoundingBoxes (bboxesdf, key):\n    bboxarray=bboxesdf.loc[key][['bx', 'by', 'bw', 'bh']].values\n    # hack to detect and fix single bounding box case which\n    # comes in with shape (4,)\n    #print (bboxarray.shape)\n    bboxarray=np.expand_dims(bboxarray, -1)\n    if bboxarray.shape[1]==1:\n        bboxarray=bboxarray.reshape(1, bboxarray.shape[0])\n    else:\n        bboxarray=np.squeeze(bboxarray, axis=-1)\n    #print (bboxarray.shape)\n    return bboxarray","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d87994b30082f4c8ebc8d0e096df24ef85835459"},"cell_type":"code","source":"# write bounding box information for lungopacity cases in yolov3 format\nif PROCESS == \"All\"  or PROCESS==\"Yolov3labels\":\n    for patientid in tqdm(lungopacitykeys):\n        bboxarray=getyolov3BoundingBoxes(yolov3bboxesdf, patientid)\n        assert len(bboxarray) > 0, \"Missing Bounding Boxes for {}\".format(patientid)\n        bboxfilename=\"{}.txt\".format(patientid)\n        bboxpathname=os.path.join(processedtraininputsdirectory, bboxfilename)\n        #print (bboxpathname)\n        file=open(bboxpathname,'w')\n        for i in range(len(bboxarray)):\n            bx, by, bw, bh = bboxarray[i]\n            #print(bx, by, bw, bh)\n            boxrecord=\"0 {} {} {} {}\\n\".format(bx, by, bw, bh)\n            file.write(boxrecord)\n        file.close()\n        \n    # make sure all boxes made it through correctly\n    processedlungopacitykeys=[]\n    for filename in glob.glob(processedtraininputsdirectory+'/*.txt'):\n        key=os.path.split(filename)[1].split(\".\")[0]\n        processedlungopacitykeys.append(key)\n    assert sorted(processedlungopacitykeys)==sorted(lungopacitykeys), \"Lung Opacity Samples Missed\"\n    print (\"Saved bounding boxes for {} Lung Opacity cases\".format(len(processedlungopacitykeys)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2b9d38ebfc46cfec70a2e2960bf8334f22405288"},"cell_type":"code","source":"!ls -al \"./rsna-stage2-processed-train-inputs\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"db5b898ac2a2f7ab3396943261b3fdd798c2622d"},"cell_type":"code","source":"!ls -al \"./rsna-stage2-processed-test-inputs\"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3bd6d1c50226663d67cc1243d0dc8c8f76c58a24"},"cell_type":"markdown","source":"## V.  Preprocess RSNA Metadata"},{"metadata":{"trusted":true,"_uuid":"f823ecde02772ba0d83159d18658b7d931a052fe"},"cell_type":"code","source":"# we are going to run the RSNA Metadata for all samples so we don't clutter up the code\n# reset the keys we munged up for quick check\nif QUICK_PROCESS==True:\n    allkeys=savedallkeys\n    testkeys=savedtestkeys\n    lungopacitykeys=savedlungopacitykeys","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"021ffbe821c83c4e4c57f3dc3857d275a203f599"},"cell_type":"code","source":"# get rsna bounding boxes by patientid\ndef getrsnaBoundingBoxes (bboxesdf, key):\n    bboxarray=bboxesdf.loc[key][['x1', 'y1', 'bw', 'bh']].values\n    # hack to detect and fix single bounding box case which\n    # comes in with shape (4,)\n    #print (bboxarray.shape)\n    bboxarray=np.expand_dims(bboxarray, -1)\n    if bboxarray.shape[1]==1:\n        bboxarray=bboxarray.reshape(1, bboxarray.shape[0])\n    else:\n        bboxarray=np.squeeze(bboxarray, axis=-1)\n    #print (bboxarray.shape)\n    return bboxarray","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a72795ff95b7b44c872cb672071075952b29697d","scrolled":true},"cell_type":"code","source":"# save RSNA metadata (will take some time, you can kill the kernel if you have seen enough)\nif PROCESS == \"All\"  or PROCESS==\"MetaData\":\n    rsnapatientdict=pd.DataFrame()\n    oneboundingboxkeys=[]\n    twoboundingboxkeys=[]\n    threeboundingboxkeys=[]\n    fourboundingboxkeys=[]\n    \n    for patientid in tqdm(allkeys):\n        rsnaattributes, _ = loadDicomImage (TRAIN_DIR, patientid, mode=\"metadata\")\n        bboxlist=[]\n        if patientid in lungopacitykeys:\n            bboxarray=getrsnaBoundingBoxes(rsnabboxesdf, patientid)\n            assert len(bboxarray) > 0, \"Missing Bounding Boxes for {}\".format(patientid)\n            bboxlist=list(bboxarray)\n            \n            if len(bboxarray) == 1:\n                oneboundingboxkeys.append(patientid)\n            elif len(bboxarray) == 2:\n                twoboundingboxkeys.append(patientid)\n            elif len(bboxarray) == 3:\n                threeboundingboxkeys.append(patientid)\n            elif len(bboxarray) == 4:\n                fourboundingboxkeys.append(patientid)\n        \n        patientrecord=pd.DataFrame({\n            'patientId': [patientid],\n            'patientSex': [rsnaattributes[0]],\n            'patientAge': [rsnaattributes[1]],\n            'patientViewPosition': [rsnaattributes[2]],\n            'BoundingBoxes': [bboxlist]})\n        rsnapatientdict=rsnapatientdict.append(patientrecord)\n        \n    print (\"################STAGE 2 BOUNDING BOX SUMMARY################\")\n    print (\"Total Lung Opacity Samples: {}\".format(len(lungopacitykeys)))\n    print (\">>Samples with 1 Bounding Box: {}\".format(len(oneboundingboxkeys)))\n    print (\">>Samples with 2 Bounding Boxes: {}\".format(len(twoboundingboxkeys)))\n    print (\">>Samples with 3 Bounding Boxes: {}\".format(len(threeboundingboxkeys)))\n    print (\">>Samples with 4 Bounding Boxes: {}\".format(len(fourboundingboxkeys)))\n    print (\"#############################################################\")\n        \n    # save all stage1 and stage2 keys\n    np.savez(os.path.join(processedmetadatadirectory, SAVED_KEYS_FILE),\n             np.array(allkeys),\n             np.array(normalkeys),\n             np.array(lungopacitykeys),\n             np.array(otherabnormalkeys),\n             np.array(testkeys),\n             np.array(oneboundingboxkeys),\n             np.array(twoboundingboxkeys),\n             np.array(threeboundingboxkeys),\n             np.array(fourboundingboxkeys),\n             np.array(stage1allkeys),\n             np.array(stage1normalkeys),\n             np.array(stage1lungopacitykeys),\n             np.array(stage1otherabnormalkeys),\n             np.array(stage1testkeys))\n    \n    # save RSNA patient dictionary (work in progress, hdf5 is creaky about strings, may not be working )\n    rsnapatientdict.to_hdf(os.path.join(processedmetadatadirectory, SAVED_PATIENTDICT_FILE),\n                           key='rsnapatientdict',\n                           mode='w')\n    print (\">>>Saved RSNA patient dictionary to to {}\".format(os.path.join(processedmetadatadirectory, SAVED_PATIENTDICT_FILE)))\n        \n    # make sure we can get everything back\n    npzfile=np.load(os.path.join(processedmetadatadirectory, SAVED_KEYS_FILE))\n\n    assert allkeys==sorted(list(npzfile['arr_0'])), \"All Keys Mismatch\"\n    assert normalkeys==sorted(list(npzfile['arr_1'])), \"Normal Keys Mismatch\"\n    assert lungopacitykeys==sorted(list(npzfile['arr_2'])), \"Lung Opacity Keys Mismatch\"\n    assert otherabnormalkeys==sorted(list(npzfile['arr_3'])), \"Not Normal / No Lung Opacity Keys Mismatch\"\n    assert testkeys==sorted(list(npzfile['arr_4'])), \"Test Keys Mismatch\"\n\n    assert oneboundingboxkeys==sorted(list(npzfile['arr_5'])), \"One Bounding Box Keys Mismatch\"\n    assert twoboundingboxkeys==sorted(list(npzfile['arr_6'])), \"Two Bounding Box Keys Mismatch\"\n    assert threeboundingboxkeys==sorted(list(npzfile['arr_7'])), \"Three Bounding Box Keys Mismatch\"\n    assert fourboundingboxkeys==sorted(list(npzfile['arr_8'])), \"Four Bounding Box Keys Mismatch\"\n\n    assert stage1allkeys==sorted(list(npzfile['arr_9'])), \"Stage1 All Keys Mismatch\"\n    assert stage1normalkeys==sorted(list(npzfile['arr_10'])), \"Stage1 Normal Keys Mismatch\"\n    assert stage1lungopacitykeys==sorted(list(npzfile['arr_11'])), \"Stage1 Lung Opacity Keys Mismatch\"\n    assert stage1otherabnormalkeys==sorted(list(npzfile['arr_12'])), \"Stage1 Not Normal / No Lung Opacity Keys Mismatch\"\n    assert stage1testkeys==sorted(list(npzfile['arr_13'])), \"Stage1 Test Keys Mismatch\"\n        \n    ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"27710685d88c96bac746becf39baa6a9eaa00cce"},"cell_type":"markdown","source":"## VI.  Next Steps\nDownload the [RSNA Stage2 Processed Train Inputs](https://www.kaggle.com/kanwalinder/rsna-stage2-processed-train-inputs), [RSNA Stage2 Processed Test Inputs](https://www.kaggle.com/kanwalinder/rsna-stage2-processed-test-inputs), and [RSNA Stage 2 Processed Metadata Inputs](https://www.kaggle.com/kanwalinder/rsna-stage2-processed-metadata-inputs) datasets and proceed to Step 2 (coming soon), reviewing [RSNA Stage 2 Anchor Box Analysis](https://www.kaggle.com/kanwalinder/rsna-stage-2-anchor-box-analysis) along the way.\n"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}