{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1 style=\"border-style: outset;border-color: red;text-align: center;\">SIIM-FISABIO-RSNA COVID-19 Dataset Preparation</h1>\n\n<img src=\"https://content.presspage.com/uploads/2110/1920_gettyimages-1216304354.jpg\" height=\"500\" width=\"500\" style=\"display: block;margin-left: auto;margin-right: auto;\"> \n\n<h2 style=\"text-align: center;border-style: double;text-align: center;border-color: red; \">About SIIM</h2>\n<img src=\"https://siim.org/resource/resmgr/SIIM_logo-600x315.png\" width=\"200\" style=\"display: block;margin-left: auto;margin-right: auto;\">\n<p> <b>Society for Imaging Informatics in Medicine</b> (<a href=\"https://siim.org/\">SIIM</a>) is the leading healthcare professional organization for those interested in the current and future use of informatics in medical imaging. The society's mission is to advance medical imaging informatics across the enterprise through education, research, and innovation in a multi-disciplinary community.</p>\n\n<a href = \"https://www.kaggle.com/shanmukh05/siim-covid19-dataset-256px-jpg\" style=\"font-weight:'bold'; color:blue; font-family:monospace; \"> <h3>My Dataset</h3></a>\n<a href = \"https://www.kaggle.com/shanmukh05/siim-covid-19-detection-detectron2-training\" style=\"font-weight:'bold'; color:blue; font-family:monospace; \"> <h3>My Training Notebook</h3></a> To be updated\n<a href = \"\" style=\"font-weight:'bold'; color:blue; font-family:monospace; \"> <h3>My Data Visualization Notebook</h3></a> Will be created soon\n<a href = \"\" style=\"font-weight:'bold'; color:blue; font-family:monospace; \"> <h3>My Inference Notebook</h3></a> Will be created soon.","metadata":{}},{"cell_type":"code","source":"##---------------------------------\n# installing dependency for pydicom\n##---------------------------------\n\n!conda install gdcm -c conda-forge -y","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-05-23T02:07:28.845788Z","iopub.execute_input":"2021-05-23T02:07:28.846294Z","iopub.status.idle":"2021-05-23T02:08:39.446984Z","shell.execute_reply.started":"2021-05-23T02:07:28.846191Z","shell.execute_reply":"2021-05-23T02:08:39.445554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##--------------------------\n#importing required lbraries\n##--------------------------\n\nimport numpy as np\nimport pandas as pd\n\nimport os\nimport ast\n\nimport PIL\nfrom PIL import Image\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\n\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"execution":{"iopub.status.busy":"2021-05-23T02:08:39.449106Z","iopub.execute_input":"2021-05-23T02:08:39.449589Z","iopub.status.idle":"2021-05-23T02:08:45.779713Z","shell.execute_reply.started":"2021-05-23T02:08:39.449539Z","shell.execute_reply":"2021-05-23T02:08:45.7788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_PATH = \"../input/siim-covid19-detection/train\"\nTEST_PATH = \"../input/siim-covid19-detection/test\"\nFINAL_TRAIN_PATH = \"./dataset/train\"\nFINAL_TEST_PATH = \"./dataset/test\"\n\nHEIGHT,WIDTH = 640,640\n\nTRAIN_FILES = tf.io.gfile.glob(TRAIN_PATH+\"/*/*/*.dcm\")\nTEST_FILES = tf.io.gfile.glob(TEST_PATH+\"/*/*/*.dcm\")\n\ntrain_id,test_id = [], []\ntrain_h, test_h = [], []\ntrain_w, test_w = [], []","metadata":{"execution":{"iopub.status.busy":"2021-05-23T02:08:45.781717Z","iopub.execute_input":"2021-05-23T02:08:45.782406Z","iopub.status.idle":"2021-05-23T02:08:53.033529Z","shell.execute_reply.started":"2021-05-23T02:08:45.782353Z","shell.execute_reply":"2021-05-23T02:08:53.03259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style = \"font-family:'Courier New';font-weight: bold;margin-top: 0px;margin-bottom: 1px;text-align: center;\">Preparing Image Data</h1>","metadata":{}},{"cell_type":"markdown","source":"<h2 style=\"font-weight:'bold'; color:red; font-family:verdana; text-align: center;\">What is DICOM format</h2>\n<p> <b>Digital Imaging and Communications in Medicine</b> (DICOM) is the standard for the communication and management of medical imaging information and related data. DICOM is most commonly used for storing and transmitting medical images enabling the integration of medical imaging devices such as scanners, servers, workstations, printers, network hardware.</p>\n\n<a href = \"https://en.wikipedia.org/wiki/DICOM\" style=\"font-weight:'bold'; color:blue; font-family:monospace; text-align: center;\"> Read more about DICOM here</a>\n\n<h2 style=\"font-weight:'bold'; color:red; font-family:verdana; text-align: center;\">Why is DICOM used</h2>\n\n<p>Following are some of applications of using DICOM format in medical imaging</p>\n\n- Collaboration With Existing IT Systems\n- High-Performance Review\n- Complete Scanning and Image Reviewing\n\n<a href = \"https://www.covetus.com/blog/how-is-dicom-important-beneficial-for-the-healthcare-industry\" style=\"font-weight:'bold'; color:blue; font-family:monospace; text-align: center;\"> Read more about DICOM applications here</a>","metadata":{}},{"cell_type":"code","source":"##---------------------------------------\n#dicom to pixel array converting function\n##---------------------------------------\n\n# Ref : https://www.kaggle.com/raddar/convert-dicom-to-np-array-the-correct-way\ndef dicom2arr(path, voi_lut = True, fix_monochrome = True):\n    dicom = pydicom.read_file(path)\n    \n    if voi_lut:\n        arr = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        arr = dicom.pixel_array\n               \n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        arr = np.amax(arr) - arr\n        \n    arr = arr - np.min(arr)\n    arr = arr / np.max(arr)\n    arr = (arr * 255).astype(np.uint8)\n        \n    return arr\n\n##-----------------------\n#resizing the pixel array\n##-----------------------\n\ndef resizeArr(arr,is_train=True):\n    im = Image.fromarray(arr)\n    if is_train:\n        train_w.append(im.size[0])\n        train_h.append(im.size[1])\n    else:\n        test_w.append(im.size[0])\n        test_h.append(im.size[1]) \n    im = im.resize((HEIGHT,WIDTH),resample= Image.LANCZOS)\n    return im\n\n##------------------------------\n#Filename for resized jpg images \n##------------------------------\n\ndef getFilename(filepath,is_train=True):\n    '''\n        Fromat = '{STUDY-ID}_{SUB-STUDY-ID}_{IMAGE-ID}.jpg'\n    '''\n    ls = filepath.split(\"/\")\n    filename = ls[-3]+'_'+ls[-2]+'_'+ls[-1].split(\".\")[0]+\".jpg\"\n    \n    if is_train:\n        train_id.append(ls[-1].split(\".\")[0])\n    else:\n        test_id.append(ls[-1].split(\".\")[0])\n    return filename\n\n##-------------------\n#Finally saving image\n##-------------------\n\ndef saveImage(filepath,mainpath=FINAL_TRAIN_PATH,train=True):\n    arr = dicom2arr(filepath)\n    arr = resizeArr(arr,train)\n    path = os.path.join(mainpath,getFilename(filepath,train))\n    \n    arr.save(path)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T02:08:53.0346Z","iopub.execute_input":"2021-05-23T02:08:53.03485Z","iopub.status.idle":"2021-05-23T02:08:53.04679Z","shell.execute_reply.started":"2021-05-23T02:08:53.034826Z","shell.execute_reply":"2021-05-23T02:08:53.045611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.makedirs(FINAL_TRAIN_PATH,exist_ok=True)\nos.makedirs(FINAL_TEST_PATH,exist_ok=True)\n\n#Preparing Training Image Data\nfor filepath in TRAIN_FILES:\n    saveImage(filepath)\n\n#Preparing Test Image Data\nfor filepath in TEST_FILES:\n    saveImage(filepath,mainpath=FINAL_TEST_PATH,train=False)\n    \n# Ref : https://www.kaggle.com/xhlulu/siim-covid-19-convert-to-jpg-256px\n!tar -zcf train.tar.gz -C \"./dataset/train/\" .\n!tar -zcf test.tar.gz -C \"./dataset/test/\" .","metadata":{"execution":{"iopub.status.busy":"2021-05-23T02:08:53.048363Z","iopub.execute_input":"2021-05-23T02:08:53.049045Z","iopub.status.idle":"2021-05-23T02:09:00.338909Z","shell.execute_reply.started":"2021-05-23T02:08:53.048982Z","shell.execute_reply":"2021-05-23T02:09:00.33776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style = \"font-family:'Courier New';font-weight: bold;margin-top: 0px;margin-bottom: 1px;text-align: center;\">Saving metadata of Images</h1>","metadata":{}},{"cell_type":"code","source":"# Saving metadata of images into csv files\nmeta_train = pd.DataFrame.from_dict({\n    \"ImageInstanceUID\" : train_id,\n    \"width\" : train_w,\n    \"height\" : train_h\n})\nmeta_train.to_csv(\"./meta_train.csv\",index=False)\n\nmeta_test = pd.DataFrame.from_dict({\n    \"ImageInstanceUID\" : test_id,\n    \"width\" : test_w,\n    \"height\" : test_h\n})\nmeta_test.to_csv(\"./meta_test.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2021-05-22T17:18:40.859553Z","iopub.execute_input":"2021-05-22T17:18:40.859965Z","iopub.status.idle":"2021-05-22T17:18:40.871741Z","shell.execute_reply.started":"2021-05-22T17:18:40.85992Z","shell.execute_reply":"2021-05-22T17:18:40.871025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sample metadata of training set\nmeta_train.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-22T17:18:40.873544Z","iopub.execute_input":"2021-05-22T17:18:40.874163Z","iopub.status.idle":"2021-05-22T17:18:40.890921Z","shell.execute_reply.started":"2021-05-22T17:18:40.874104Z","shell.execute_reply":"2021-05-22T17:18:40.889662Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sample metadata of test set\nmeta_test.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-22T17:18:40.892496Z","iopub.execute_input":"2021-05-22T17:18:40.892825Z","iopub.status.idle":"2021-05-22T17:18:40.908119Z","shell.execute_reply.started":"2021-05-22T17:18:40.892795Z","shell.execute_reply":"2021-05-22T17:18:40.907083Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classes_dict = {\n    0 : \"Negative for Pneumonia\",\n    1  : \"Typical Appearance\",\n    2  : \"Indeterminate Appearance\",\n    3  : \"Atypical Appearance\"\n}\n\n\n# Format of dictionary : {\"image_id\" : \"study_id\"}\nid_dict = {}\nfor file in TRAIN_FILES:\n    ls = file.split(\"/\")\n    id_dict[ls[-1].split(\".\")[0]] = ls[-3]\n    \n##-----------------------------------------\n#getting filepath from study_id or image_id\n##-----------------------------------------\n\ndef get_path(file_id,main_path,id_type):\n    if id_type == \"study\":\n        path = tf.io.gfile.glob(main_path+f\"/{file_id}/*/*.dcm\")[0]\n    else:\n        path = tf.io.gfile.glob(main_path+f\"/*/*/{file_id}.dcm\")[0]\n    return path","metadata":{"execution":{"iopub.status.busy":"2021-05-22T16:13:00.745089Z","iopub.execute_input":"2021-05-22T16:13:00.745726Z","iopub.status.idle":"2021-05-22T16:13:00.753219Z","shell.execute_reply.started":"2021-05-22T16:13:00.745687Z","shell.execute_reply":"2021-05-22T16:13:00.751906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style = \"font-family:'Courier New';font-weight: bold;margin-top: 0px;margin-bottom: 1px;text-align: center;\">Preparing CSV file for Training</h1>","metadata":{}},{"cell_type":"code","source":"# Read CSV files\nimage_df = pd.read_csv(\"../input/siim-covid19-detection/train_image_level.csv\")\nstudy_df = pd.read_csv(\"../input/siim-covid19-detection/train_study_level.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-05-22T17:01:04.404729Z","iopub.execute_input":"2021-05-22T17:01:04.405346Z","iopub.status.idle":"2021-05-22T17:01:04.451627Z","shell.execute_reply.started":"2021-05-22T17:01:04.405306Z","shell.execute_reply":"2021-05-22T17:01:04.450624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print sample data of image_level csv file\nimage_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2021-05-22T17:01:06.067122Z","iopub.execute_input":"2021-05-22T17:01:06.067679Z","iopub.status.idle":"2021-05-22T17:01:06.080103Z","shell.execute_reply.started":"2021-05-22T17:01:06.067626Z","shell.execute_reply":"2021-05-22T17:01:06.079287Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print sample data of study_level csv file\nstudy_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2021-05-22T17:01:14.996707Z","iopub.execute_input":"2021-05-22T17:01:14.997079Z","iopub.status.idle":"2021-05-22T17:01:15.012243Z","shell.execute_reply.started":"2021-05-22T17:01:14.997048Z","shell.execute_reply":"2021-05-22T17:01:15.011093Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Making one-hot of study_level labels and removing other 4 class columns\nstudy_df[\"one_hot\"] = study_df.apply(lambda x : np.array([x[\"Negative for Pneumonia\"],\n                                                        x[\"Typical Appearance\"],\n                                                        x[\"Indeterminate Appearance\"],\n                                                        x[\"Atypical Appearance\"]]),axis=1)\n\nstudy_df[\"label_id\"] = study_df[\"one_hot\"].map(lambda x : classes_dict[np.argmax(x)])\nstudy_df[\"study_label\"] = study_df[\"one_hot\"].map(lambda x : np.argmax(x))\nstudy_df = study_df.drop([\"Negative for Pneumonia\",\"Typical Appearance\",\"Indeterminate Appearance\",\"Atypical Appearance\",\"one_hot\"],axis=1)\nstudy_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2021-05-22T17:03:18.404116Z","iopub.execute_input":"2021-05-22T17:03:18.404554Z","iopub.status.idle":"2021-05-22T17:03:18.670253Z","shell.execute_reply.started":"2021-05-22T17:03:18.404519Z","shell.execute_reply":"2021-05-22T17:03:18.669091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Renaming \"id\" to \"ImageInstanceUID\" and \"id\" to \"StudyInstanceUID\" for better understanding\nimage_df[\"id\"] = image_df[\"id\"].map(lambda x : x.replace(\"_image\",\"\"))\nimage_df.rename(columns={'id':\"ImageInstanceUID\",'label':\"image_label\"},inplace=True)\n\nstudy_df[\"id\"] = study_df[\"id\"].map(lambda x : x.replace(\"_study\",\"\"))\nstudy_df.rename(columns={\"id\" : \"StudyInstanceUID\"},inplace=True)","metadata":{"execution":{"iopub.status.busy":"2021-05-22T17:04:52.259774Z","iopub.execute_input":"2021-05-22T17:04:52.260248Z","iopub.status.idle":"2021-05-22T17:04:52.277173Z","shell.execute_reply.started":"2021-05-22T17:04:52.260208Z","shell.execute_reply":"2021-05-22T17:04:52.276259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3>In the below cell I used <i>ast</i> library to convert <b>string of boxes columns</b>  into <b>dictionary type</b></h3>\n\n<h2 style=\"font-weight:'bold'; color:red; font-family:verdana; text-align: center;\">What is ast library</h2>\n\n<h4>The ast (<b>Abstract Syntax Tree</b>) module helps Python applications to process trees of the Python abstract syntax grammar.</h4>\n\n<a href = \"https://docs.python.org/3/library/ast.html\" style=\"font-weight:'bold'; color:blue; font-family:monospace; text-align: center;\"> Read more about AST here</a>\n\n<img src=\"https://libcst.readthedocs.io/en/latest/_images/graphviz-d27e3495fa9bb130d76879db599060e8039a9fc5.png\" height=\"500\" width=\"500\" style=\"display: block;margin-left: auto;margin-right: auto;\"> ","metadata":{}},{"cell_type":"code","source":"train_df = pd.merge(image_df,study_df,on = \"StudyInstanceUID\") # Merging study_df and image_df\n\ntry :\n    train_df = pd.merge(train_df,meta_train,on = \"ImageInstanceUID\") # Merging to meta_train for height,width\nexcept:\n    pass\n\n# Filling NaN values \ntrain_df[\"boxes\"].fillna(\"[{'x':0,'y':0,'width':1,'height':1}]\",inplace=True)\ntemp = train_df # for going through the data\ntrain_df[\"boxes\"] = train_df[\"boxes\"].map(lambda x : ast.literal_eval(x))\n\ntry:\n    columns = [\"ImageInstanceUID\",\"StudyInstanceUID\",\"label_id\",\"study_label\",\"height\",\"width\",\"boxes\",\"image_label\"] # for proper order\n    train_df = train_df[columns]\nexcept:\n    columns = [\"ImageInstanceUID\",\"StudyInstanceUID\",\"label_id\",\"study_label\",\"boxes\",\"image_label\"] # for proper order\n    train_df = train_df[columns]\n\ntrain_df.to_csv(\"./train.csv\",index=False)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-22T17:05:23.866546Z","iopub.execute_input":"2021-05-22T17:05:23.866909Z","iopub.status.idle":"2021-05-22T17:05:24.12265Z","shell.execute_reply.started":"2021-05-22T17:05:23.866878Z","shell.execute_reply":"2021-05-22T17:05:24.121203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\nshutil.rmtree(\"./dataset\",ignore_errors=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"font-weight:'bold'; color:blue; font-family:verdana; text-align: center; background-image: url(https://image.freepik.com/free-vector/gradient-background-green-shades_23-2148363157.jpg);\">Do <b>UPVOTE</b> if you found this Notebook useful😊</h1>\n\n<img src=\"https://external-preview.redd.it/vWYcdynWxuFy6bKYpZwuw6KiNgYuBPM6daHCwWRs4mo.png?auto=webp&s=fe3bf857b3c9f1369aed1ea0bc5e1acd8ae39449\" height=\"200\" width=\"200\" style=\"display: block;margin-left: auto;margin-right: auto;\"> ","metadata":{}}]}