{"cells":[{"metadata":{"trusted":true,"_uuid":"83c74cdf44b27e6b206556b8979bf54eda588e30"},"cell_type":"code","source":"# basic imports\nimport os, random\nfrom tqdm import tqdm\nimport pydicom\nimport pandas as pd\nimport numpy as np","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bcab09c98e7e02740e00ac5bd2e533658316484d"},"cell_type":"code","source":"# what do we have here?\nprint (os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"12173594123e32289906a8d18507a069337e1631"},"cell_type":"code","source":"# global variables\nTRAIN_DIR=\"../input/stage_1_train_images\"\n# list of training images\nTRAIN_LIST=sorted(os.listdir(TRAIN_DIR))\n\nTRAIN_LABELS_CSV_FILE=\"../input/stage_1_train_labels.csv\"\n# pedantic nit: we are changing 'Target' to 'label' on the way in\nTRAIN_LABELS_CSV_COLUMN_NAMES=['patientId', 'x1', 'y1', 'width', 'height', 'label']\n\nDETAILED_CLASSES_CSV_FILE=\"../input/stage_1_detailed_class_info.csv\"\nDETAILED_CLASSES_CSV_COLUMN_NAMES=['patientId' , 'class']\n# dictionary to map string classes to numerical\nCLASSES_DICT={'Normal': 0, 'Lung Opacity' : 1, 'No Lung Opacity / Not Normal' : 2}\n\nTEST_DIR=\"../input/stage_1_test_images\"\n# list of test images\nTEST_LIST=sorted(os.listdir(TEST_DIR))\n\nIMAGE_SIZE=[1024,1024]\n\n# intermediate meta-data files\nTRAIN_METADATA_CSV_FILE=\"stage_1_train_metadata.csv\"\nTRAIN_METADATA_CSV_COLUMN_NAMES=['patientId', 'sex', 'age', 'viewPosition']\nTEST_METADATA_CSV_FILE=\"stage_1_test_metadata.csv\"\nTEST_METADATA_CSV_COLUMN_NAMES=['patientId', 'sex', 'age', 'viewPosition']\n# dictionaries to map sex and viewPosition fields to numerical\nPATIENTSEX_DICT={'M': 0, 'F' : 1}\nVIEWPOSITION_DICT={'AP': 0, 'PA' : 1}\n\n# we will consolidate label, classes, and processed train meta-data into one csv file\nTRAIN_PROCESSED_CSV_FILE=\"stage_1_train_processed.csv\"\nTRAIN_PROCESSED_CSV_COLUMN_NAMES=['patientId', 'label', 'class', 'sex', 'age', 'viewPosition']\n\n# we will pre-process bounding boxes a bit and save in a separate csv file\nTRAIN_BOUNDINGBOX_CSV_FILE=\"stage_1_train_boundingboxes.csv\"\nTRAIN_BOUNDINGBOX_CSV_COLUMN_NAMES=['patientId', 'boundingbox']\n\n# we will save processed test meta-data into one csv file\nTEST_PROCESSED_CSV_FILE=\"stage_1_test_processed.csv\"\nTEST_PROCESSED_CSV_COLUMN_NAMES=['patientId', 'sex', 'age', 'viewPosition']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b033eed6942a80b15ec85df591e89705b6d06868"},"cell_type":"code","source":"# how many unique training examples (patient images) do we have?\nprint (\"Unique training examples provided: {}\".format(len(os.listdir(TRAIN_DIR))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ee9224f0348beb4926b2d682419671dbb9df24c0"},"cell_type":"code","source":"# remember 25684!","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c62da913fbcb81348d27f2ddfe38d07fb17d7a83"},"cell_type":"code","source":"# how many test cases we have to analyze?\nprint (\"Test cases to be predicted: {}\".format(len(os.listdir(TEST_DIR))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fef708c8a801cc0c6e8a07c923222ba1f370a589"},"cell_type":"code","source":"# read a random training patient data file\ntrainpatientdata = pydicom.dcmread(os.path.join(TRAIN_DIR, random.choice(TRAIN_LIST)))\n# print patient dataset\nprint(\"Training Patient Dataset: \\n.{}\\n\".format(trainpatientdata))\n# print patient dataset attributes\nprint(\"Training Patient Dataset Attributes: \\n{}\\n\".format(trainpatientdata.dir()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4dca5c6e65cd5eaf1cd8df0be8ea16414014ef20"},"cell_type":"code","source":"# PatientSex, PatientAge, and ViewPosition look like useful attributes of trainpatientdata.\n# the image is in trainpatientdata.pixel_array.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"47976b063c232239e96fdec8b3c1b9876d8fd47d"},"cell_type":"code","source":"# utility to check dicom files\ndef checkImages (directory, filelist):\n    for i, filename in tqdm(enumerate(filelist)):\n        # get patientid from filename\n        patientidfromfilename = filename.split(\".\")[0]\n        # load patient meta-data only from file\n        patientdata = pydicom.dcmread(os.path.join(directory, filename), stop_before_pixels=True)\n        # get patientid from patient data\n        patientidfromfile=patientdata.PatientID\n        # make sure everything is ok\n        assert patientidfromfilename == patientidfromfile, \"Error: Patient ID mismatch\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d76e2e7c39343f8e2dc7792a00d3eb501c439a5f"},"cell_type":"code","source":"# make sure training images are ok (no assertion errors)\n# need to check once only, comment out after first run\ncheckImages(TRAIN_DIR, TRAIN_LIST)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0441601c1d1d4be12a21faf339bb855e941e4dff"},"cell_type":"code","source":"# make sure test images are ok (no assertion errors)\n# need to check once only, comment out after first run\ncheckImages(TEST_DIR, TEST_LIST)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"703bb7088dfe36c0853b05f677c06d7d5178ba89"},"cell_type":"code","source":"# utility to load a dicom image and/or key attributes\ndef loadImage (directory, filename, mode=\"metadata\"):\n    imagearray=np.zeros(IMAGE_SIZE)\n    patientid= filename.split(\".\")[0]\n    \n    if mode==\"metadata\":\n        # load patient meta-data only from file\n        patientdata = pydicom.dcmread(os.path.join(directory, filename), stop_before_pixels=True)\n    elif mode==\"image\":\n        # load patient meta-data and image from file\n        patientdata = pydicom.dcmread(os.path.join(directory, filename))\n        imagearray=patientdata.pixel_array\n    patientid=patientdata.PatientID\n    attributes=\"{} {} {}\".format(patientdata.PatientSex,\n                                 patientdata.PatientAge,\n                                 patientdata.ViewPosition)\n    \n    return patientid, attributes, imagearray","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7018846578c9a960f8a0b22090633a0885a30d1b"},"cell_type":"code","source":"# utility to save meta data to csv files\ndef saveMetaData (directory, filelist, csvfilename):\n    csvrecord=\"patientId,sex,age,viewPosition\\n\"\n    for filename in tqdm(filelist):\n        patientid, attributes, imagearray=loadImage(directory, filename, mode=\"metadata\")\n        patientsex, patientage, patientposition = attributes.split()\n        csvrecord=\"{}{},{},{},{}\\n\".format(csvrecord, patientid, patientsex, patientage, patientposition)\n    # print (csvrecord)\n    with open(csvfilename,'w') as file:\n        file.write(csvrecord)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"639fff80408ecc1606850de9400488c6edd21bbc"},"cell_type":"code","source":"# save training meta data\nsaveMetaData(TRAIN_DIR, TRAIN_LIST, TRAIN_METADATA_CSV_FILE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3476c4d3eb28af5714222361db4dda4674cd4447"},"cell_type":"code","source":"# how many unique records did we save in stage_1_train_metadata.csv?\n!printf \"Number of unique training meta data records stored: \"; grep -v \"patientId,sex,age,viewPosition\" stage_1_train_metadata.csv | sort | uniq| wc -l","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fa277d9244c53b51caeb384d611d65ed9d0d70af"},"cell_type":"code","source":"# what does the stage_1_train_metadata.csv file look like?\n!printf \"First 10 rows, including header, in stage_1_train_metadata.csv:\\n\\n\"; \\\n         head -10 stage_1_train_metadata.csv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7b74549b28fb36467a9281a1f0d93bea845fce02"},"cell_type":"code","source":"# save test meta data\nsaveMetaData(TEST_DIR, TEST_LIST, TEST_METADATA_CSV_FILE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"822ed53c9e7ea34926d92838598c4b4173fad7a8"},"cell_type":"code","source":"# how many unique records did we save in stage_1_test_metadata.csv?\n!printf \"Number of unique test meta data records stored: \"; grep -v \"patientId,sex,age,viewPosition\" stage_1_test_metadata.csv | sort | uniq| wc -l","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bb072ced5260e5690d8df2fd8cfde2c90b783f8e"},"cell_type":"code","source":"# what does the stage_1_test_metadata.csv file look like?\n!printf \"First 10 rows, including header, in stage_1_test_metadata.csv:\\n\\n\"; \\\n         head -10 stage_1_test_metadata.csv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e83bdcdf1811770c21f48b2f090cbb2f2cc0a74c"},"cell_type":"code","source":"# what does the stage_1_train_labels.csv file look like?\n!printf \"First 10 rows, including header, in stage_1_train_labels.csv: \\n\\n\"; \\\n         head -10 ../input/stage_1_train_labels.csv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"df138eba0cf614b9528e671ecd161546e2a36dce"},"cell_type":"code","source":"# what does the stage_1_detailed_class_info.csv file look like?\n!printf \"First 10 rows, including header, in stage_1_detailed_class_info.csv:\\n\\n\"; \\\n         head -10 ../input/stage_1_detailed_class_info.csv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4f667693c726499c7fc3277c18e459896448bb7e"},"cell_type":"code","source":"def combineBoundingBoxes(bboxesin):\n    combinedbboxlist=[]\n    # convert incoming series into numpy array\n    bboxesinarray=bboxesin.values\n    #print (\"a: {}\".format(bboxesinarray))\n    # compute x1+width and y1+height for all rows\n    bboxesinarray[:,2]=bboxesinarray[:,0]+bboxesinarray[:,2]\n    bboxesinarray[:,3]=bboxesinarray[:,1]+bboxesinarray[:,3]\n    #print (\"b: {}\".format(bboxesinarray))\n    # compute smallest and the largest x,y values across rows\n    x1min, y1min, _, _ = bboxesinarray.min(axis=0)\n    _, _, x2max, y2max = bboxesinarray.max(axis=0)\n    #print (\"c: {},{},{},{}\".format(x1min, y1min, x2max, y2max))\n    # insert values back in x,y,width,height format\n    combinedbboxlist.insert(0, x1min)\n    combinedbboxlist.insert(1, y1min)\n    combinedbboxlist.insert(2, x2max-x1min)\n    combinedbboxlist.insert(3, y2max-y1min)\n    #print (\"d: {}\".format(combinedbboxlist))\n    return combinedbboxlist","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c98ce0e223f497a3c18951e1b0fa5e8863675d88"},"cell_type":"code","source":"# read TRAIN_LABELS_CSV_FILE into a pandas dataframe\nlabelsbboxdf = pd.read_csv(TRAIN_LABELS_CSV_FILE,\n                           names=TRAIN_LABELS_CSV_COLUMN_NAMES,\n                           # skip the header line\n                           header=0,\n                           # index the dataframe on patientId\n                           index_col='patientId')\nprint (labelsbboxdf.shape)\nprint (labelsbboxdf.head(n=10))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c7e024b147701f90fe8f14587a1814d3ad47feba"},"cell_type":"code","source":"# grab labels by unique patienId\nlabelsdf=pd.DataFrame(labelsbboxdf.pop('label'), columns=['label'])\n# remove duplicates\nlabelsdf=pd.DataFrame(labelsdf.groupby(['patientId'])['label'].first(), columns=['label'])\nprint (labelsdf.head(n=10))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b65e0eb0753e9eb5997539e7a1c107edc59dde0"},"cell_type":"code","source":"# after 'label' is popped off, x1,y1,width,height are left in labelsbboxdf\nprint(labelsbboxdf.head(n=10))\nbboxesdf=pd.DataFrame(labelsbboxdf.dropna())\n# drop rows with NaNs (will drop rows others than the one with class 'Lung Opacity')\nprint(bboxesdf.head(n=10))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e6ea986b8f4377da4fedccc3872eea7bf0180931"},"cell_type":"code","source":"# compute the largest bounding box by patientId\ncombinedbboxdf=pd.DataFrame(bboxesdf.groupby(['patientId']).apply(combineBoundingBoxes), columns=['combinedboundingbox'])\n# patch up bug in pandas resulting in the first entry being incorrect\ncombinedbboxdf.iat[0,0]=[264.0, 152.0, 554.0, 453.0]\ncombinedbboxdf.head(n=10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"753b3378e31f268861909a2862bc0330e1872ee2"},"cell_type":"code","source":"# consolidate train labels, classes, and meta-data by unique patientId\n# combine x,y,width,height records into one bounding box\n# process test meta-data by unique patientId\n\n# read TRAIN_LABELS_CSV_FILE into a pandas dataframe\nlabelsbboxdf = pd.read_csv(TRAIN_LABELS_CSV_FILE,\n                           names=TRAIN_LABELS_CSV_COLUMN_NAMES,\n                           # skip the header line\n                           header=0,\n                           # index the dataframe on patientId\n                           index_col='patientId')\n#print (labelsbboxdf.shape)\nprint (labelsbboxdf.head(n=10))\n\n\n\n# massage bounding box information\n\"\"\"x=pd.DataFrame(labelsbboxdf.pop('x'), columns=['x'])\ny=pd.DataFrame(labelsbboxdf.pop('y'), columns=['y'])\nwidth=pd.DataFrame(labelsbboxdf.pop('width'), columns=['width'])\nheight=pd.DataFrame(labelsbboxdf.pop('height'), columns=['height'])\"\"\"\n\n# check the data types the import happened in (should be float64)\n# print(\"{}\\n{}\\n{}\\n{}\".format(x.dtypes, y.dtypes, width.dtypes, height.dtypes))\n\n\n# combine x, y, width, height into a consolidated bounding box dataframe\nprint (labelsbboxdf.head(n=10))\nlabelsbboxdf['x2'] = labelsbboxdf['x1']+labelsbboxdf['width']\nlabelsbboxdf['y2'] = labelsbboxdf['y1']+labelsbboxdf['height']\nprint (labelsbboxdf.head(n=10))\n\"\"\"bboxdf=pd.DataFrame(x['x'].astype(np.float)+' '\n                    +y['y'].astype(np.float)+' '\n                    +width['width'].astype(np.float)+' '\n                    +height['height'].astype(np.float), columns=['boundingbox'])\n# cleanup the missing values\nbboxdf=bboxdf.replace('nan nan nan nan', np.NaN)\"\"\"\n#print (bboxdf.head(n=10))\n# we will not remove duplicates for now and we will save bounding boxes separately\n# save bounding boxes to csv file TRAIN_BOUNDINGBOX_CSV_FILE\nbboxdf.to_csv(TRAIN_BOUNDINGBOX_CSV_FILE)\n\n# read DETAILED_CLASSES_CSV_FILE into a pandas dataframe\nclassesdf = pd.read_csv(DETAILED_CLASSES_CSV_FILE,\n                        names=DETAILED_CLASSES_CSV_COLUMN_NAMES,\n                        # skip the header line\n                        header=0,\n                        # index the dataframe on patientId\n                        index_col='patientId')\n#print (classesdf.shape)\n#print (classesdf.head(n=10))\n\n# make classes numerical based on CLASSES_DICT\nclassesdf=classesdf.replace(to_replace=CLASSES_DICT)\n# remove duplicates\nclassesdf=pd.DataFrame(classesdf.groupby(['patientId'])['class'].first(), columns=['class'])\n#print (classesdf.head(n=10))\n\n# read TRAIN_METADATA_CSV_FILE into a pandas dataframe\ntrainmetadf = pd.read_csv(TRAIN_METADATA_CSV_FILE,\n                          names=TRAIN_METADATA_CSV_COLUMN_NAMES,\n                          # skip the header line\n                          header=0,\n                          # index the dataframe on patientId\n                          index_col='patientId')\n#print (trainmetadf.shape)\n#print (trainmetadf.head(n=10))\n\n# make sex numerical based on PATIENTSEX_DICT\ntrainmetadf['sex']=trainmetadf['sex'].replace(to_replace=PATIENTSEX_DICT)\n# make viewPosition numerical based on VIEWPOSITION_DICT\ntrainmetadf['viewPosition']=trainmetadf['viewPosition'].replace(to_replace=VIEWPOSITION_DICT)\n#print (trainmetadf.head(n=10))\n\n# consolidate label, class, and meta-data into a consolidated dataframe\ntrainprocesseddf=pd.concat([labelsdf, classesdf, trainmetadf], axis=1)\n# save to csv file TRAIN_PROCESSED_CSV_FILE\ntrainprocesseddf.to_csv(TRAIN_PROCESSED_CSV_FILE)\n\n# read TEST_METADATA_CSV_FILE into a pandas dataframe\ntestmetadf = pd.read_csv(TEST_METADATA_CSV_FILE,\n                         names=TEST_METADATA_CSV_COLUMN_NAMES,\n                         # skip the header line\n                         header=0,\n                         # index the dataframe on patientId\n                         index_col='patientId')\n#print (testmetadf.shape)\n#print (testmetadf.head(n=10))\n                                 \n# make sex numerical based on PATIENTSEX_DICT\ntestmetadf['sex']=testmetadf['sex'].replace(to_replace=PATIENTSEX_DICT)\n# make viewPosition numerical based on VIEWPOSITION_DICT\ntestmetadf['viewPosition']=testmetadf['viewPosition'].replace(to_replace=VIEWPOSITION_DICT)\n#print (testmetadf.head(n=10))\n# save to csv file TEST_PROCESSED_CSV_FILE\ntestmetadf.to_csv(TEST_PROCESSED_CSV_FILE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d87cb5f3006407858fede5a4269684083be6097c"},"cell_type":"code","source":"# how many unique records did we save in stage_1_train_boundingboxes.csv?\n!printf \"Number of bounding box data records (not unique) stored: \"; grep -v \"patientId,boundingbox\" stage_1_train_boundingboxes.csv | sort | uniq| wc -l","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d7725cbc34c5548cd730bdf0283dccd31a1eebfc"},"cell_type":"code","source":"# what does the stage_1_train_boundingboxes.csv file look like?\n!printf \"First 10 rows, including header, in stage_1_train_boundingboxes.csv:\\n\\n\"; \\\n         head -10 stage_1_train_boundingboxes.csv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7f428106119d9fd942db061ac418828e91498243"},"cell_type":"code","source":"# how many unique records did we save in stage_1_train_processed.csv?\n!printf \"Number of unique train processed data records stored: \"; grep -v \"patientId,label,class,sex,age,viewPosition\" stage_1_train_processed.csv | sort | uniq| wc -l","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b728de80699e1f576c11ef1996ceff256c9ea7ac"},"cell_type":"code","source":"# what does the stage_1_train_processed.csv file look like?\n!printf \"First 10 rows, including header, in stage_1_train_processed.csv:\\n\\n\"; \\\n         head -10 stage_1_train_processed.csv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"34818a3c92ddb301a2a2dc5a514740524f5b9b0a"},"cell_type":"code","source":"# how many unique records did we save in stage_1_test_processed.csv?\n!printf \"Number of unique train processed data records stored: \"; grep -v \"patientId,sex,age,viewPosition\" stage_1_test_processed.csv | sort | uniq| wc -l","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"353df679575368bc48efd622b6f8585448d05679"},"cell_type":"code","source":"# what does the stage_1_test_processed.csv file look like?\n!printf \"First 10 rows, including header, in stage_1_test_processed.csv:\\n\\n\"; \\\n         head -10 stage_1_test_processed.csv","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}