{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"DICOM is a standard file format used in medical imaaging.<br>\nWe need to extract pixel values and save them as jpg/png image in order to visualize and move forward with training.<br>\nAlong with pixel values, there is more data about patient in the file which might be useful for training.","metadata":{}},{"cell_type":"markdown","source":"The following code extracts png image and csv file from DICOM images.","metadata":{}},{"cell_type":"code","source":"input_dir=\"/kaggle/input/siim-covid19-detection\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-26T18:40:24.041847Z","iopub.execute_input":"2021-06-26T18:40:24.042170Z","iopub.status.idle":"2021-06-26T18:40:24.046102Z","shell.execute_reply.started":"2021-06-26T18:40:24.042143Z","shell.execute_reply":"2021-06-26T18:40:24.045360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%file dicom_image_description.csv\nDescription,Code\nSpecificCharacterSet,CS\nSOPClassUID,UI\nSOPInstanceUID,UI\nStudyDate,DA\nStudyTime,TM\nAccessionNumber,SH\nModality,CS\nPatientName,PN\nPatientID,LO\nPatientSex,CS\nBodyPartExamined,CS\nStudyInstanceUID,UI\nSeriesInstanceUID,UI\nStudyID,SH\nSeriesNumber,IS\nInstanceNumber,IS\nSamplesperPixel,US\nPhotometricInterpretation,CS\nRows,US\nColumns,US\nBitsAllocated,US\nBitsStored,US\nHighBit,US\nPixelRepresentation,US\nPixelData,OB","metadata":{"execution":{"iopub.status.busy":"2021-06-26T18:40:26.496568Z","iopub.execute_input":"2021-06-26T18:40:26.496901Z","iopub.status.idle":"2021-06-26T18:40:26.501951Z","shell.execute_reply.started":"2021-06-26T18:40:26.496876Z","shell.execute_reply":"2021-06-26T18:40:26.501112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### offline install gdcm handler for pydicom","metadata":{}},{"cell_type":"code","source":"!cp ../input/gdcm-conda-install/gdcm.tar .\n!tar -xvzf gdcm.tar\n!conda install --offline ./gdcm/gdcm-2.8.9-py37h71b2a6d_0.tar.bz2\nprint(\"done\")","metadata":{"execution":{"iopub.status.busy":"2021-06-26T18:40:29.802037Z","iopub.execute_input":"2021-06-26T18:40:29.802463Z","iopub.status.idle":"2021-06-26T18:40:40.450492Z","shell.execute_reply.started":"2021-06-26T18:40:29.802425Z","shell.execute_reply":"2021-06-26T18:40:40.449388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Extract Images","metadata":{}},{"cell_type":"code","source":"import pydicom as dicom\nimport matplotlib.pyplot as plt\nimport os\nimport glob\nimport cv2\nimport pandas as pd\nimport csv\n\n# Specify the .dcm folder path\nimages_path = glob.glob(input_dir+\"/train/**/*.dcm\", recursive=True)[:5]\n# Specify dest folder path\ndest_folder_path = \"/kaggle/tmp\"\n# list of attributes available in dicom image\n# download this file from the given link # https://github.com/vivek8981/DICOM-to-JPG\ndicom_image_description = pd.read_csv(\"dicom_image_description.csv\")\n\n# make dirs\nos.makedirs('/kaggle/tmp/', exist_ok=True)\nos.makedirs('/kaggle/tmp/train_images/', exist_ok=True)\n\nwith open(dest_folder_path+'/train_dcom_details.csv', 'w', newline ='') as csvfile:\n    fieldnames = list(dicom_image_description[\"Description\"])\n    writer = csv.writer(csvfile, delimiter=',')\n    # write header\n    writer.writerow(fieldnames)\n    for n, image_path in enumerate(images_path):\n        ds = dicom.dcmread(image_path)\n        rows = []\n        pixel_array_numpy = ds.pixel_array\n        \n        image = os.path.basename(image_path).replace('.dcm', '.png')\n        cv2.imwrite(os.path.join(dest_folder_path, 'train_images', image), pixel_array_numpy)\n        if n % 100 == 0:\n            print('{} image converted'.format(n))\n        for field in fieldnames:\n            if ds.data_element(field) is None:\n                rows.append('')\n            else:\n                x = str(ds.data_element(field)).replace(\"'\", \"\")\n                y = x.find(\":\")\n                x = x[y+2:]\n                rows.append(x)\n        writer.writerow(rows)","metadata":{"execution":{"iopub.status.busy":"2021-06-26T18:41:26.413225Z","iopub.execute_input":"2021-06-26T18:41:26.413655Z","iopub.status.idle":"2021-06-26T18:41:39.433719Z","shell.execute_reply.started":"2021-06-26T18:41:26.413624Z","shell.execute_reply":"2021-06-26T18:41:39.432809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(cv2.imread('/kaggle/tmp/train_images/d54f6204b044.png',0), cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2021-06-26T18:43:16.258090Z","iopub.execute_input":"2021-06-26T18:43:16.258655Z","iopub.status.idle":"2021-06-26T18:43:17.199344Z","shell.execute_reply.started":"2021-06-26T18:43:16.258623Z","shell.execute_reply":"2021-06-26T18:43:17.198231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv(dest_folder_path+'/train_dcom_details.csv').head()","metadata":{"execution":{"iopub.status.busy":"2021-06-26T18:44:10.195197Z","iopub.execute_input":"2021-06-26T18:44:10.195851Z","iopub.status.idle":"2021-06-26T18:44:10.226843Z","shell.execute_reply.started":"2021-06-26T18:44:10.195813Z","shell.execute_reply":"2021-06-26T18:44:10.226189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}