{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"## Importing the necessary libraries\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n## for converting images to arrays\nimport os, cv2","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"ls ../input/rsna-pneumonia-detection-challenge","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cb0c129207f06ef65bb8b3f05be169c8cb25e183"},"cell_type":"markdown","source":"## Merge dataframes containing patientID and class labels"},{"metadata":{"trusted":true,"_uuid":"a95d54890c82187c5fec8f31044177e5008165df"},"cell_type":"code","source":"# Importing the training label file\ntrain_df = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b0a3eef5680b06f2a4ee8444e2d970ad4da7b5e5"},"cell_type":"code","source":"## Importing the class labels file\nclass_info_df = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv')\nclass_info_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"023c67ff0f55dcf4c2f704d3ef3b20bd8e359cfb"},"cell_type":"code","source":"## Check the shape of dataframes\nprint(train_df.shape)\nprint(class_info_df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"72da1bd515b59af6b58358cd9eb7cc047d43e86c"},"cell_type":"code","source":"## Merging dataframes just to check the values\nmerged_df = pd.merge(train_df,class_info_df, \n                     left_index = True, \n                     right_index = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bd7f52151a58ae27282341a0bdf69573d9e30ff7"},"cell_type":"code","source":"## check the columns\nmerged_df.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b4b23d904eec26a61c7d4a715a6f8e38c4ebc39"},"cell_type":"code","source":"## delete repeated columns\nmerged_df.drop('patientId_y', axis=1, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4d52a6386986aed3f5a8181b0eec03ead1daac2c"},"cell_type":"code","source":"## Changing the column names into meaningful names\nmerged_df.rename(columns={'patientId_x':'patientId', 'Target':'target',\n                         'class':'target_class_desc'}, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"37a46d9f35e91dbf7dbf7625a8707aeacfc37e69"},"cell_type":"code","source":"## Check the final merge table\nmerged_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"30cabe7a28d58640ff082a5321d746952f83e59a","scrolled":true},"cell_type":"code","source":"ls ../input/rsna-stage-2-png-converted-files/stage_2_png_converted_files/","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"20afdb2271022f009fd0fbbab4755d303d58e19c"},"cell_type":"code","source":"train_path = os.path.join('..','input','rsna-stage-2-png-converted-files',\n                          'stage_2_png_converted_files','stage_2_png_converted_files')\ntrain_path","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ad9ffe4011ec52f896b324534a6f0eda2fa22adf"},"cell_type":"code","source":"test_path = os.path.join('..','input','rsna-pneumonia-detection-challenge','stage_2_test_images')\ntest_path","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f08f8c081b44e1e0966fbcb4c317aa59e33590c6"},"cell_type":"markdown","source":"## Convert dicom files to png"},{"metadata":{"_uuid":"99757de2cea01f5422887b87fb3d276a1def98d9"},"cell_type":"markdown","source":"### Note: \nUse this code to convert dicom to png in your own system.\nUnfortunately, I do not think we can create folder inside Kaggle system. \nSo, use the following code to create a png_converted_files. \nAlternatively, you could also use png_converted_files which i have uploaded\nin the Kaggle dataset for working on Stage 2 files. I have used png but we can work on jpg as well. The name of the dataset is \"RSNA stage 2 png converted files\".\n\nWe can work on dicom files directly as well converting them to arrays. However, we might just as well use images itself in keras for image processing. In that case, we would either require jpg or png. \n\nThe code for the function below was extracted from a source as well. My sincere apologies that I could not put source on it. I will reference the author after I will find it again."},{"metadata":{"trusted":true,"_uuid":"ca130a99c63be662a253b88118eb62df9aa73ddd"},"cell_type":"code","source":"'''\ndef convert_dicom_to_png():\n    # make it True if you want in PNG format\n    PNG = True\n    # Specify the .dcm folder path\n    folder_path = train_path\n    # Specify the output jpg/png folder path\n    png_folder_path = \"../datasets/png_converted_files\"\n    images_path = os.listdir(folder_path)\n    for n, image in enumerate(images_path):\n        ds = pydicom.dcmread(os.path.join(folder_path, image), force=True)\n        pixel_array_numpy = ds.pixel_array\n        if PNG == False:\n            image = image.replace('.dcm', '.jpg')\n        else:\n            image = image.replace('.dcm', '.png')\n        cv2.imwrite(os.path.join(png_folder_path, image), pixel_array_numpy)\n        if n % 1000 == 0:\n            print('{} image converted'.format(n))\n'''","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b51c192fd00076a494c5f8fdd590371a454b1418"},"cell_type":"markdown","source":"## Convert Images to arrays"},{"metadata":{"trusted":true,"_uuid":"d814d969a1f488fd47aba9785472a288de498d52"},"cell_type":"code","source":"## Function to convert training images to arrays\ndef train_images_to_arrays():\n    \"\"\"\n    Returns two arrays: \n        x is an array of resized images\n        y is an array of labels\n        Source: Chris Crawford: https://www.kaggle.com/crawford/resize-and-save-images-as-numpy-arrays-128x128\n    \"\"\"\n\n    x = [] # images as arrays\n    y = [] # labels \n    WIDTH = 224 # for VGG-16\n    HEIGHT = 224 # for VGG-16\n\n    for image in enumerate(merged_df.patientId):    \n        \n        img_name = image[1]\n        image_path = train_path + '/' + img_name + '.png'\n        \n        # Read and resize image\n        full_size_image = cv2.imread(image_path)\n        \n        x.append(cv2.resize(full_size_image, (WIDTH,HEIGHT), interpolation=cv2.INTER_CUBIC))\n        \n        # Labels\n        index_of_image = image[0]\n        target_value = merged_df.target.loc[index_of_image]\n        #print(target_value)\n        y.append(target_value)\n\n    return x,y","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"21d1c5fc3d515e78beaf5039f6b8fa7e4dbc6fc0"},"cell_type":"code","source":"## Obtain X and y as arrays...\nX,y = train_images_to_arrays()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"749d551855f4ac39436e0b3d7dc702e49a258b1d"},"cell_type":"code","source":"## Saving arrays for future use\n## Can't save...Kaggle is read-only.. need to import data locally!!!!\n\n## --- Remove the comment in the code below to save in your local machine for code reuse ----\n# np.savez_compressed(\"../input/rsna-stage-2-png-converted-files/x_images\", X)\n# np.savez_compressed(\"../input/rsna-stage-2-png-converted-files/y_pneumonia\", y)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}