{"cells":[{"metadata":{"_uuid":"014ae21ecb3d73b7aa3c8b1c19524e60c035cf6d"},"cell_type":"markdown","source":"**Quick Note**\n\nThe purpose of this notebook is simply to create the tabular data that will be used in Part 2. At the end of this notebook the dataframes are saved as pickled files. For the details of how the Generators were built and how the Keras cnn was set up please go straight to Part 2.\n\nHowever, you will find this notebook useful if you would like to know how to extract meta data from the image files or see how to find and fix patient age errors."},{"metadata":{"_uuid":"0da06c3a698ff6716e643fa4ad467429a4b21cdb"},"cell_type":"markdown","source":"<hr>"},{"metadata":{"trusted":true,"_uuid":"65195e0cde60e90777ee948c53d57419c0000982"},"cell_type":"code","source":"from numpy.random import seed\nseed(101)\nfrom tensorflow import set_random_seed\nset_random_seed(101)\n\nimport pandas as pd\nimport numpy as np\nimport pydicom\nimport pylab\nimport os\nimport pickle\n\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n# Don't Show Warning Messages\nimport warnings\nwarnings.filterwarnings('ignore')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"857e6f2f4ce635a6bdf13366174a4b8979750bdc"},"cell_type":"code","source":"df_train = pd.read_csv('../input/stage_1_train_labels.csv')\ndf_test = pd.read_csv('../input/stage_1_sample_submission.csv')\ndf_info = pd.read_csv('../input/stage_1_detailed_class_info.csv')\n\nprint(df_train.shape)\nprint(df_test.shape)\nprint(df_info.shape)\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"06c791cdb3a97fff676eb01da08a8ad9b9e10b2a"},"cell_type":"markdown","source":"### Create new columns using df_info"},{"metadata":{"trusted":true,"_uuid":"99afb453572fdef882f140db24a6a22d9a7a508c"},"cell_type":"code","source":"# check df_info for missing values\ndf_info.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4fa03a075bccefb4db5f7389e56a64d67c470f29"},"cell_type":"code","source":"# New Col: num_bounding_boxes\ndf_train['num_bounding_boxes'] = 1","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5926eef7d6257c9e7f949005d9caf61850eba819"},"cell_type":"markdown","source":"### Extract the meta data from the images"},{"metadata":{"trusted":true,"_uuid":"a322abaa9d61f26fc13de01513e8fc59b3d07aee"},"cell_type":"code","source":"# create a dataframe with unique patientId's\ndf_group = \\\ndf_train.drop(['x','y','width','height','Target'], axis=1).groupby('patientId').sum()\n\n# reset the index\ndf_group.reset_index(inplace=True)\n\ndf_group.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b9b831d2d3075dd5e9d6597b3f2c4373aad813e7"},"cell_type":"code","source":"# check\nprint(df_train['patientId'].nunique())\nprint(len(df_group))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0e6a55d8d5a5fb2bba4a61ffb7eb50a8400ee0b1"},"cell_type":"code","source":"# create new columns\ndf_group['PatientAge'] = 0\ndf_group['PatientSex'] = 0\ndf_group['ViewPosition'] = 0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e966ae6ff1d5be1c2834c1a013d2b7b9688272b1"},"cell_type":"code","source":"# extract the meta data and store in df_group\n\nfor i in range(0,len(df_group)):\n    patientId = df_group.loc[i,'patientId']\n    \n    path = \\\n'../input/stage_1_train_images/%s.dcm' % patientId\n    \n    # get the meta data\n    dcm_data = pydicom.read_file(path)\n    \n    df_group.loc[i,'PatientAge'] = dcm_data.PatientAge\n    df_group.loc[i,'PatientSex'] = dcm_data.PatientSex\n    df_group.loc[i,'ViewPosition'] = dcm_data.ViewPosition\n    ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9c2ac5cc31313d75e11eb37723f764d81067a188"},"cell_type":"markdown","source":"### Extract meta data from the test images and store in df_test"},{"metadata":{"trusted":true,"_uuid":"841847c899c9d95263f95007058620d1e3d910c3"},"cell_type":"code","source":"# create new columns\ndf_test['PatientAge'] = 0\ndf_test['PatientSex'] = 0\ndf_test['ViewPosition'] = 0\n\nfor i in range(0,len(df_test)):\n    patientId = df_test.loc[i,'patientId']\n    \n    path = \\\n'../input/stage_1_test_images/%s.dcm' % patientId\n    \n    # get the meta data\n    dcm_data = pydicom.read_file(path)\n    \n    df_test.loc[i,'PatientAge'] = dcm_data.PatientAge\n    df_test.loc[i,'PatientSex'] = dcm_data.PatientSex\n    df_test.loc[i,'ViewPosition'] = dcm_data.ViewPosition\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0aeba93771e362060f17a44fd6811e0e7ed493e3"},"cell_type":"code","source":"# change the datatype to int16. now it is string\ndf_group['PatientAge'] = df_group['PatientAge'].astype(np.int16)\ndf_test['PatientAge'] = df_test['PatientAge'].astype(np.int16)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b32f55813a68e9d3b25298de171e0d218feb838f"},"cell_type":"markdown","source":"### Add the class column to df_group"},{"metadata":{"trusted":true,"_uuid":"6669c5043cb2a772cb3cf3c73172139304a97d48"},"cell_type":"code","source":"df = df_info\n\ndf.drop_duplicates(inplace=True)\n\n# reset the index or NaN's will be produced when trying to add the class col to df_group\ndf.reset_index(inplace=True)\n\ndf_group['class'] = df['class']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"94c1a9daa370303d36770373f3dd1be271fc1a4d"},"cell_type":"markdown","source":"### Fix errors in the age feature"},{"metadata":{"_uuid":"18cda02d54ea40d7e2a7e5a9506b23d65c38fb48"},"cell_type":"markdown","source":"The images all look like adults. Therefore, it could be that 1 was added by mistake to the beginning of the age. "},{"metadata":{"trusted":true,"_uuid":"937d9d85e47d0b08a8d7871ab694cb89af6d6843"},"cell_type":"code","source":"# check for age errors in the train set\ndf_group[df_group['PatientAge'] > 100]\n\n# 5 age errors found","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4ef24147a4f8e152582dc053e40234c9ef1e6aa1"},"cell_type":"code","source":"# check for age errors in the test set\ndf_test[df_test['PatientAge'] > 100]\n\n# no age errors found","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"71704248c1e38ea0f4c85af059cd150d11172d45"},"cell_type":"code","source":"# view the x-rays\n# load a patient's file\npatientId = df_train.loc[24537,'patientId']\npath = \\\n'../input/stage_1_train_images/%s.dcm' % patientId\n\ndcm_data = pydicom.read_file(path)\n\n# convert the image to a numpy array\nim = dcm_data.pixel_array\n\n# view an image\npylab.imshow(im, cmap=pylab.cm.gist_gray)\npylab.axis('off')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"adaa9828c0ed9d0ea22cd20b1867bdb240aa49c6"},"cell_type":"code","source":"# remove the 1 at the start of the age\n# assumes 1 was added by mistake to these ages\nage_errors = ['3b8b8777-a1f6-4384-872a-28b95f59bf0d', '73aeea88-fc48-4030-8564-0a9d7fdecac4',\n             'a4e8e96d-93a6-4251-b617-91382e610fab', 'ec3697bd-184e-44ba-9688-ff8d5fbf9bbc',\n             'f632328d-5819-4b29-b54f-adf4934bbee6']\n\ndf_group.loc[3175, 'PatientAge'] = 48\ndf_group.loc[9708, 'PatientAge'] = 51\ndf_group.loc[15273, 'PatientAge'] = 53\ndf_group.loc[23374, 'PatientAge'] = 50\ndf_group.loc[24537, 'PatientAge'] = 55","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0d70b82bc364d7bc2e9c6921e157e786d849cb4c"},"cell_type":"code","source":"# check the changes\ndf_group.loc[23374,:]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6948d3ef29cdf80e1e3bf9291707d7b54a68a922"},"cell_type":"markdown","source":"### Add a col to df_group for noopacity_but_not_normal"},{"metadata":{"trusted":true,"_uuid":"3b22401053fd3772f13d018d29cd339469779011"},"cell_type":"code","source":"df_group['noopacity_but_not_normal'] = df_group['class']\n\ndef noopacity_but_not_normal(x):\n    if x == 'No Lung Opacity / Not Normal':\n        return 1\n    else:\n        return 0\n\ndf_group['noopacity_but_not_normal'] = \\\ndf_group['noopacity_but_not_normal'].apply(noopacity_but_not_normal)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2df89650d32e2deb7d2b62b17fc7601d29eef18e"},"cell_type":"markdown","source":"### Map the 3 class labels to 0 and 1"},{"metadata":{"trusted":true,"_uuid":"0c9a433c361fe77f5ffcf1446cafd977d5f3c30d"},"cell_type":"code","source":"# New Col: target\ndf_group['target'] = \\\ndf_group['class'].map({'Lung Opacity':1, 'Normal':0, 'No Lung Opacity / Not Normal':0})","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ff533ed0f57d6dab2f3659d6f537f4183a5b263d"},"cell_type":"markdown","source":"### If no pneumonia set num_bounding_boxes to 0"},{"metadata":{"trusted":true,"_uuid":"76d1eef621c80f715778355b96734bc4a126b79f"},"cell_type":"code","source":"for i in range(0,len(df_group)):\n    if df_group.loc[i,'target'] == 0:\n        df_group.loc[i,'num_bounding_boxes'] = 0","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"68c647865b08838798628d65018dfdeea255b15c"},"cell_type":"markdown","source":"### Create the train set bounding boxes"},{"metadata":{"trusted":true,"_uuid":"b09da629d057a8a0f0f3218438e04e7fdb8c2a80"},"cell_type":"code","source":"# Source: https://www.kaggle.com/peterchang77/exploratory-data-analysis\n\ndef parse_data(df):\n    \"\"\"\n    Method to read a CSV file (Pandas dataframe) and parse the \n    data into the following nested dictionary:\n\n      parsed = {\n        \n        'patientId-00': {\n            'dicom': path/to/dicom/file,\n            'label': either 0 or 1 for normal or pnuemonia, \n            'boxes': list of box(es)\n        },\n        'patientId-01': {\n            'dicom': path/to/dicom/file,\n            'label': either 0 or 1 for normal or pnuemonia, \n            'boxes': list of box(es)\n        }, ...\n\n      }\n\n    \"\"\"\n    # --- Define lambda to extract coords in list [y, x, height, width]\n    extract_box = lambda row: [row['y'], row['x'], row['height'], row['width']]\n\n    parsed = {}\n    for n, row in df.iterrows():\n        # --- Initialize patient entry into parsed \n        pid = row['patientId']\n        if pid not in parsed:\n            parsed[pid] = {\n                'dicom': '../input/stage_1_train_images/%s.dcm' % pid,\n                'label': row['Target'],\n                'boxes': []}\n\n        # --- Add box if opacity is present\n        if parsed[pid]['label'] == 1:\n            parsed[pid]['boxes'].append(extract_box(row))\n\n    return parsed\n\n\n# run the function\n\ndf_boxes = pd.read_csv('../input/stage_1_train_labels.csv')\n\nparsed = parse_data(df_boxes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6cf9a9b2080d96a0ba92009e7e4c0d8746adda77"},"cell_type":"code","source":"# check the bounding boxes for one patientId\nbox = parsed['00436515-870c-4b36-a041-de91049b9ab4']['boxes']\n\nbox","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cb3d68fa201a5f35ab5971972f7a150231c63d8f"},"cell_type":"code","source":"# extract the boxes for each patient\n\ndf_group['bounding_boxes'] = df_group['patientId']\n\ndef bounding_boxes(x):\n    # get the dictionary value\n    box = parsed[x]['boxes']\n    \n    return box\n\ndf_group['bounding_boxes'] = df_group['bounding_boxes'].apply(bounding_boxes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"496475704448e531e67de15a246f8c96d38301de"},"cell_type":"code","source":"# drop the PredictionString col from df_test\ndf_test = df_test.drop('PredictionString', axis=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d181e659daa449ecc261d45a658d7fd24cf2a34d"},"cell_type":"markdown","source":"### TAIL CHECKS"},{"metadata":{"_uuid":"380fe846bb643b372891080b76f11d5da4084398"},"cell_type":"markdown","source":"Make sure that NaN's have not been introduced during pre processing."},{"metadata":{"trusted":true,"_uuid":"348491bfe2384da4fd56e654182b3e5028b0d4a6"},"cell_type":"code","source":"df_group.isnull().sum()        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"02e94b437be5bda5c2b6cf45e805e881d3a61f6b"},"cell_type":"code","source":"df_test.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"51b8c1178d31c8d41f39b01068685def1f4512da"},"cell_type":"markdown","source":"### Save the dataframes"},{"metadata":{"trusted":true,"_uuid":"9ff24bd11cd99e8603f9e1e9469d4465d830cc05"},"cell_type":"code","source":"# note: we are saving df_group with the name dftrain for easy reference later\npickle.dump(df_group,open('dftrain.pickle','wb'))\npickle.dump(df_test,open('dftest.pickle','wb'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e4443fcf220ef657c729a6fdaebfb0e3b0a9211e"},"cell_type":"code","source":"# check if the pickled files exist\n!ls","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7f5b840897ed2a159092e734d23f451e91fa7e52"},"cell_type":"markdown","source":"<hr>\n**Continued in Part 2...**"},{"metadata":{"trusted":true,"_uuid":"e358d2596584b9fb07bc1c65f349b81a50e82d4f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}