{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Convert and Export to pngs","metadata":{}},{"cell_type":"code","source":"!conda install gdcm -c conda-forge -y","metadata":{"execution":{"iopub.status.busy":"2023-03-09T14:48:46.515145Z","iopub.execute_input":"2023-03-09T14:48:46.515605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pylibjpeg ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-08-25T13:40:52.560654Z","iopub.execute_input":"2022-08-25T13:40:52.561022Z","iopub.status.idle":"2022-08-25T13:40:53.707070Z","shell.execute_reply.started":"2022-08-25T13:40:52.560983Z","shell.execute_reply":"2022-08-25T13:40:53.705699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert DICOM to standard image format\n# this will take quite some time to complete\n# https://www.kaggle.com/code/xhlulu/siim-covid-19-convert-to-jpg-256px\nfrom IPython.display import clear_output\n\nimport os, pydicom, gdcm, pylibjpeg, shutil\nfrom PIL import Image\nimport numpy as np\nimport pandas as pd\n\n\nos.mkdir(\"png_train\")\nos.mkdir(\"png_train/COVID\")\nos.mkdir(\"png_train/NON\")\nos.mkdir(\"png_train/COVID/Typical\")\nos.mkdir(\"png_train/COVID/Indeterminate\")\nos.mkdir(\"png_train/COVID/Atypical\")\n\ntrain_folder_path = \"../input/siim-covid19-detection/train/\"\n# load in training dataframes \ntrain_study_df = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_study_level.csv\")\n# train_image_df = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_image_level.csv\")\n# add labels\ntrain_study_df['Counts'] = train_study_df[['Negative for Pneumonia','Typical Appearance','Indeterminate Appearance','Atypical Appearance']].idxmax(axis=1)\ntrain_study_df['width-orig'] = 0\ntrain_study_df['height-orig'] = 0\ntrain_study_df['scale'] = 1\ntrain_study_df['Filepath'] = \"\"\n# train_image_df['StudyLabel'] = train_image_df['StudyInstanceUID'].apply(lambda x: train_study_df[train_study_df['id']==x+\"_study\"]['Counts'].values[0])\n\n# to prevent memory issues, split train image processing into 4 chunks\nsize_of_batches = int(len(os.listdir(train_folder_path))/4)\n\nbatch_1 = os.listdir(train_folder_path)[0:size_of_batches]\nbatch_2 = os.listdir(train_folder_path)[size_of_batches+1:2*size_of_batches]\nbatch_4 = os.listdir(train_folder_path)[(2*size_of_batches)+1:3*size_of_batches]\nbatch_3 = os.listdir(train_folder_path)[(3*size_of_batches)+1:]\n\n# batch 1\ni = 1\nfor study_uid in os.listdir(train_folder_path):#batch_1:\n    clear_output(wait=True)\n    print(\"Processing image {}/{}\".format(i, len(os.listdir(train_folder_path))))\n#     print(\"Processing image {}/{}\".format(i, len(batch_1)))\n    i+=1\n    img_dir = os.listdir(train_folder_path+study_uid)\n    filename = os.listdir(train_folder_path+study_uid+\"/\"+img_dir[0])[0]\n    img = pydicom.read_file(train_folder_path+study_uid+\"/\"+img_dir[0]+\"/\"+filename)\n    try:\n        px = img.pixel_array\n    #     some images have different colourmaps, translate images to conistent map\n        if img.PhotometricInterpretation == \"MONOCHROME1\":\n            px = np.amax(px) - px\n    #     convert to rgb\n        new_px = px - np.min(px)\n        new_px = new_px / np.max(new_px)\n        new_px = (new_px * 255).astype(np.uint8)\n    #     check type of image\n        label = train_study_df[train_study_df['id']==study_uid+\"_study\"]['Counts'].values[0]\n        if (label=='Negative for Pneumonia'):\n            save_path = \"png_train/NON/\"\n        elif (label=='Typical Appearance'):\n            save_path = \"png_train/COVID/Typical/\"\n        elif (label=='Indeterminate Appearance'):\n            save_path = \"png_train/COVID/Indeterminate/\"\n        else:\n            save_path = save_path = \"png_train/COVID/Atypical/\"\n        img_out = Image.fromarray(new_px)\n        # resize to be no bigger than 1000x1000 and maintin aspect ratio\n        if(img_out.height > 1000 and img_out.height >=img_out.width):\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = img_out.height/1000\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n            img_out = img_out.resize([int(img_out.width/(img_out.height/1000)),1000], Image.LANCZOS)\n        elif(img_out.width >1000 and img_out.width > img_out.height):\n            train_study_df.loc[train_study_df.index[train_st51udy_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = img_out.width/1000\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n            img_out = img_out.resize([1000,int(img_out.height/(img_out.width/1000))], Image.LANCZOS)\n        else:\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = 1\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n        img_out.save(save_path+filename.replace(\"dcm\",\"png\"))\n    except Exception as e:\n        print(\"Image \\\"{}\\\" failed to convert with the following error: \". format(train_folder_path+study_uid+\"/\"+img_dir[0]+\"/\"+filename))\n        print(e)\n\nshutil.make_archive(\"siim_1000x1000_png\", 'tar', \"png_train/\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import FileLink\nFileLink(r'siim_1000x1000_png.tar')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_study_df.to_csv(\"train_study_1000x1000.csv\",index=False)\nFileLink(r'train_study_1000x1000.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert DICOM to standard image format\n# this will take quite some time to complete\n# https://www.kaggle.com/code/xhlulu/siim-covid-19-convert-to-jpg-256px\nfrom IPython.display import clear_output\n\nimport os, pydicom, gdcm, pylibjpeg, shutil\nfrom PIL import Image\nimport numpy as np\nimport pandas as pd\n\n\nos.mkdir(\"png_train\")\nos.mkdir(\"png_train/COVID\")\nos.mkdir(\"png_train/NON\")\nos.mkdir(\"png_train/COVID/Typical\")\nos.mkdir(\"png_train/COVID/Indeterminate\")\nos.mkdir(\"png_train/COVID/Atypical\")\n\ntrain_folder_path = \"../input/siim-covid19-detection/train/\"\n# load in training dataframes \ntrain_study_df = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_study_level.csv\")\n# train_image_df = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_image_level.csv\")\n# add labels\ntrain_study_df['Counts'] = train_study_df[['Negative for Pneumonia','Typical Appearance','Indeterminate Appearance','Atypical Appearance']].idxmax(axis=1)\ntrain_study_df['width-orig'] = 0\ntrain_study_df['height-orig'] = 0\ntrain_study_df['scale'] = 1\ntrain_study_df['Filepath'] = \"\"\n# train_image_df['StudyLabel'] = train_image_df['StudyInstanceUID'].apply(lambda x: train_study_df[train_study_df['id']==x+\"_study\"]['Counts'].values[0])\n\n# to prevent memory issues, split train image processing into 4 chunks\nsize_of_batches = int(len(os.listdir(train_folder_path))/4)\n\nbatch_1 = os.listdir(train_folder_path)[0:size_of_batches]\nbatch_2 = os.listdir(train_folder_path)[size_of_batches+1:2*size_of_batches]\nbatch_4 = os.listdir(train_folder_path)[(2*size_of_batches)+1:3*size_of_batches]\nbatch_3 = os.listdir(train_folder_path)[(3*size_of_batches)+1:]\n\n# batch 1\ni = 1\nfor study_uid in batch_1:\n    clear_output(wait=True)\n    print(\"Processing image {}/{}\".format(i, len(batch_1)))\n    i+=1\n    img_dir = os.listdir(train_folder_path+study_uid)\n    filename = os.listdir(train_folder_path+study_uid+\"/\"+img_dir[0])[0]\n    img = pydicom.read_file(train_folder_path+study_uid+\"/\"+img_dir[0]+\"/\"+filename)\n    try:\n        px = img.pixel_array\n    #     some images have different colourmaps, translate images to conistent map\n        if img.PhotometricInterpretation == \"MONOCHROME1\":\n            px = np.amax(px) - px\n    #     convert to rgb\n        new_px = px - np.min(px)\n        new_px = new_px / np.max(new_px)\n        new_px = (new_px * 255).astype(np.uint8)\n    #     check type of image\n        label = train_study_df[train_study_df['id']==study_uid+\"_study\"]['Counts'].values[0]\n        if (label=='Negative for Pneumonia'):\n            save_path = \"png_train/NON/\"\n        elif (label=='Typical Appearance'):\n            save_path = \"png_train/COVID/Typical/\"\n        elif (label=='Indeterminate Appearance'):\n            save_path = \"png_train/COVID/Indeterminate/\"\n        else:\n            save_path = save_path = \"png_train/COVID/Atypical/\"\n        img_out = Image.fromarray(new_px)\n        # resize to be no bigger than 500x500 and maintin aspect ratio\n        if(img_out.height > 1000 and img_out.height >=img_out.width):\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = img_out.height/1000\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n            img_out = img_out.resize([int(img_out.width/(img_out.height/1000)),1000], Image.LANCZOS)\n        elif(img_out.width >1000 and img_out.width > img_out.height):\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = img_out.width/1000\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n            img_out = img_out.resize([1000,int(img_out.height/(img_out.width/1000))], Image.LANCZOS)\n        else:\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = 1\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n        img_out.save(save_path+filename.replace(\"dcm\",\"png\"))\n    except Exception as e:\n        print(\"Image \\\"{}\\\" failed to convert with the following error: \". format(train_folder_path+study_uid+\"/\"+img_dir[0]+\"/\"+filename))\n        print(e)\n\nshutil.make_archive(\"batch_1000x1000_1\", 'tar', \"png_train/\")","metadata":{"execution":{"iopub.status.busy":"2022-08-24T17:14:10.948233Z","iopub.execute_input":"2022-08-24T17:14:10.948575Z","iopub.status.idle":"2022-08-24T17:30:36.873477Z","shell.execute_reply.started":"2022-08-24T17:14:10.948540Z","shell.execute_reply":"2022-08-24T17:30:36.872602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import FileLink\nFileLink(r'batch_1000x1000_1.tar')","metadata":{"execution":{"iopub.status.busy":"2022-08-24T17:30:36.875139Z","iopub.execute_input":"2022-08-24T17:30:36.875524Z","iopub.status.idle":"2022-08-24T17:30:36.882110Z","shell.execute_reply.started":"2022-08-24T17:30:36.875493Z","shell.execute_reply":"2022-08-24T17:30:36.881261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf png_train","metadata":{"execution":{"iopub.status.busy":"2022-08-24T17:30:36.884556Z","iopub.execute_input":"2022-08-24T17:30:36.884797Z","iopub.status.idle":"2022-08-24T17:30:37.309479Z","shell.execute_reply.started":"2022-08-24T17:30:36.884772Z","shell.execute_reply":"2022-08-24T17:30:37.308496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.mkdir(\"png_train\")\nos.mkdir(\"png_train/COVID\")\nos.mkdir(\"png_train/NON\")\nos.mkdir(\"png_train/COVID/Typical\")\nos.mkdir(\"png_train/COVID/Indeterminate\")\nos.mkdir(\"png_train/COVID/Atypical\")\n\n# batch 2\ni=1\nfor study_uid in batch_2:\n    clear_output(wait=True)\n    print(\"Processing image {}/{}\".format(i, len(batch_2)))\n    i+=1\n    img_dir = os.listdir(train_folder_path+study_uid)\n    filename = os.listdir(train_folder_path+study_uid+\"/\"+img_dir[0])[0]\n    img = pydicom.read_file(train_folder_path+study_uid+\"/\"+img_dir[0]+\"/\"+filename)\n    try:\n        px = img.pixel_array\n    #     some images have different colourmaps, translate images to conistent map\n        if img.PhotometricInterpretation == \"MONOCHROME1\":\n            px = np.amax(px) - px\n    #     convert to rgb\n        new_px = px - np.min(px)\n        new_px = new_px / np.max(new_px)\n        new_px = (new_px * 255).astype(np.uint8)\n    #     check type of image\n        label = train_study_df[train_study_df['id']==study_uid+\"_study\"]['Counts'].values[0]\n        if (label=='Negative for Pneumonia'):\n            save_path = \"png_train/NON/\"\n        elif (label=='Typical Appearance'):\n            save_path = \"png_train/COVID/Typical/\"\n        elif (label=='Indeterminate Appearance'):\n            save_path = \"png_train/COVID/Indeterminate/\"\n        else:\n            save_path = save_path = \"png_train/COVID/Atypical/\"\n        img_out = Image.fromarray(new_px)\n        # resize to be no bigger than 1000x1000 and maintin aspect ratio\n        if(img_out.height > 1000 and img_out.height >=img_out.width):\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = img_out.height/1000\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n            img_out = img_out.resize([int(img_out.width/(img_out.height/1000)),1000], Image.LANCZOS)\n        elif(img_out.width >1000 and img_out.width > img_out.height):\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = img_out.width/1000\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n            img_out = img_out.resize([1000,int(img_out.height/(img_out.width/1000))], Image.LANCZOS)\n        else:\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = 1\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n        img_out.save(save_path+filename.replace(\"dcm\",\"png\"))\n    except Exception as e:\n        print(\"Image \\\"{}\\\" failed to convert with the following error: \". format(train_folder_path+study_uid+\"/\"+img_dir[0]+\"/\"+filename))\n        print(e)\n\nshutil.make_archive(\"batch_1000x1000_2\", 'tar', \"png_train/\")","metadata":{"execution":{"iopub.status.busy":"2022-08-24T17:30:37.311098Z","iopub.execute_input":"2022-08-24T17:30:37.311336Z","iopub.status.idle":"2022-08-24T17:47:52.931263Z","shell.execute_reply.started":"2022-08-24T17:30:37.311307Z","shell.execute_reply":"2022-08-24T17:47:52.930313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf png_train","metadata":{"execution":{"iopub.status.busy":"2022-08-24T17:47:52.932675Z","iopub.execute_input":"2022-08-24T17:47:52.932928Z","iopub.status.idle":"2022-08-24T17:47:53.364300Z","shell.execute_reply.started":"2022-08-24T17:47:52.932901Z","shell.execute_reply":"2022-08-24T17:47:53.363067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FileLink(r'batch_1000x1000_2.tar')","metadata":{"execution":{"iopub.status.busy":"2022-08-24T17:47:53.366145Z","iopub.execute_input":"2022-08-24T17:47:53.366524Z","iopub.status.idle":"2022-08-24T17:47:53.373323Z","shell.execute_reply.started":"2022-08-24T17:47:53.366478Z","shell.execute_reply":"2022-08-24T17:47:53.372370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.mkdir(\"png_train\")\nos.mkdir(\"png_train/COVID\")\nos.mkdir(\"png_train/NON\")\nos.mkdir(\"png_train/COVID/Typical\")\nos.mkdir(\"png_train/COVID/Indeterminate\")\nos.mkdir(\"png_train/COVID/Atypical\")\n\n# batch 3\ni=1\nfor study_uid in batch_3:\n    clear_output(wait=True)\n    print(\"Processing image {}/{}\".format(i, len(batch_3)))\n    i+=1\n    img_dir = os.listdir(train_folder_path+study_uid)\n    filename = os.listdir(train_folder_path+study_uid+\"/\"+img_dir[0])[0]\n    img = pydicom.read_file(train_folder_path+study_uid+\"/\"+img_dir[0]+\"/\"+filename)\n    try:\n        px = img.pixel_array\n    #     some images have different colourmaps, translate images to conistent map\n        if img.PhotometricInterpretation == \"MONOCHROME1\":\n            px = np.amax(px) - px\n    #     convert to rgb\n        new_px = px - np.min(px)\n        new_px = new_px / np.max(new_px)\n        new_px = (new_px * 255).astype(np.uint8)\n    #     check type of image\n        label = train_study_df[train_study_df['id']==study_uid+\"_study\"]['Counts'].values[0]\n        if (label=='Negative for Pneumonia'):\n            save_path = \"png_train/NON/\"\n        elif (label=='Typical Appearance'):\n            save_path = \"png_train/COVID/Typical/\"\n        elif (label=='Indeterminate Appearance'):\n            save_path = \"png_train/COVID/Indeterminate/\"\n        else:\n            save_path = save_path = \"png_train/COVID/Atypical/\"\n        img_out = Image.fromarray(new_px)\n        # resize to be no bigger than 500x500 and maintin aspect ratio\n        if(img_out.height > 1000 and img_out.height >=img_out.width):\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = img_out.height/1000\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n            img_out = img_out.resize([int(img_out.width/(img_out.height/1000)),1000], Image.LANCZOS)\n        elif(img_out.width >1000 and img_out.width > img_out.height):\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = img_out.width/1000\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n            img_out = img_out.resize([1000,int(img_out.height/(img_out.width/1000))], Image.LANCZOS)\n        else:\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = 1\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n        img_out.save(save_path+filename.replace(\"dcm\",\"png\"))\n    except Exception as e:\n        print(\"Image \\\"{}\\\" failed to convert with the following error: \". format(train_folder_path+study_uid+\"/\"+img_dir[0]+\"/\"+filename))\n        print(e)\n\nshutil.make_archive(\"batch_1000x1000_3\", 'tar', \"png_train/\")","metadata":{"execution":{"iopub.status.busy":"2022-08-24T17:47:53.374651Z","iopub.execute_input":"2022-08-24T17:47:53.374977Z","iopub.status.idle":"2022-08-24T18:04:18.040824Z","shell.execute_reply.started":"2022-08-24T17:47:53.374943Z","shell.execute_reply":"2022-08-24T18:04:18.039887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf png_train","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:04:18.042288Z","iopub.execute_input":"2022-08-24T18:04:18.042611Z","iopub.status.idle":"2022-08-24T18:04:18.463679Z","shell.execute_reply.started":"2022-08-24T18:04:18.042578Z","shell.execute_reply":"2022-08-24T18:04:18.462487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FileLink(r'./batch_1000x1000_3.tar')","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:04:18.465258Z","iopub.execute_input":"2022-08-24T18:04:18.465549Z","iopub.status.idle":"2022-08-24T18:04:18.471470Z","shell.execute_reply.started":"2022-08-24T18:04:18.465516Z","shell.execute_reply":"2022-08-24T18:04:18.470623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.mkdir(\"png_train\")\nos.mkdir(\"png_train/COVID\")\nos.mkdir(\"png_train/NON\")\nos.mkdir(\"png_train/COVID/Typical\")\nos.mkdir(\"png_train/COVID/Indeterminate\")\nos.mkdir(\"png_train/COVID/Atypical\")\n\n# batch 4\ni=1\nfor study_uid in batch_4:\n    clear_output(wait=True)\n    print(\"Processing image {}/{}\".format(i, len(batch_4)))\n    i+=1\n    img_dir = os.listdir(train_folder_path+study_uid)\n    filename = os.listdir(train_folder_path+study_uid+\"/\"+img_dir[0])[0]\n    img = pydicom.read_file(train_folder_path+study_uid+\"/\"+img_dir[0]+\"/\"+filename)\n    try:\n        px = img.pixel_array\n    #     some images have different colourmaps, translate images to conistent map\n        if img.PhotometricInterpretation == \"MONOCHROME1\":\n            px = np.amax(px) - px\n    #     convert to rgb\n        new_px = px - np.min(px)\n        new_px = new_px / np.max(new_px)\n        new_px = (new_px * 255).astype(np.uint8)\n    #     check type of image\n        label = train_study_df[train_study_df['id']==study_uid+\"_study\"]['Counts'].values[0]\n        if (label=='Negative for Pneumonia'):\n            save_path = \"png_train/NON/\"\n        elif (label=='Typical Appearance'):\n            save_path = \"png_train/COVID/Typical/\"\n        elif (label=='Indeterminate Appearance'):\n            save_path = \"png_train/COVID/Indeterminate/\"\n        else:\n            save_path = save_path = \"png_train/COVID/Atypical/\"\n        img_out = Image.fromarray(new_px)\n        # resize to be no bigger than 500x500 and maintin aspect ratio\n        if(img_out.height > 1000 and img_out.height >= img_out.width):\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = img_out.height/1000\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n            img_out = img_out.resize([int(img_out.width/(img_out.height/1000)),1000], Image.LANCZOS)\n        elif(img_out.width >1000 and img_out.width > img_out.height):\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = img_out.width/1000\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n            img_out = img_out.resize([1000,int(img_out.height/(img_out.width/1000))], Image.LANCZOS)\n        else:\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = 1\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n\n        img_out.save(save_path+filename.replace(\"dcm\",\"png\"))\n    except Exception as e:\n        print(\"Image \\\"{}\\\" failed to convert with the following error: \". format(train_folder_path+study_uid+\"/\"+img_dir[0]+\"/\"+filename))\n        print(e)\n\nshutil.make_archive(\"batch_1000x1000_4\", 'tar', \"png_train/\")","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:04:18.472785Z","iopub.execute_input":"2022-08-24T18:04:18.473031Z","iopub.status.idle":"2022-08-24T18:21:04.783494Z","shell.execute_reply.started":"2022-08-24T18:04:18.473005Z","shell.execute_reply":"2022-08-24T18:21:04.782439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FileLink(r'./batch_1000x1000_4.tar')","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:21:04.784946Z","iopub.execute_input":"2022-08-24T18:21:04.785270Z","iopub.status.idle":"2022-08-24T18:21:04.791803Z","shell.execute_reply.started":"2022-08-24T18:21:04.785235Z","shell.execute_reply":"2022-08-24T18:21:04.790740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculate scaling factor for each image\n\nfrom IPython.display import clear_output\n\nimport os, pydicom, gdcm, pylibjpeg, shutil\nfrom PIL import Image\nimport numpy as np\nimport pandas as pd\n\ntrain_folder_path = \"../input/siim-covid19-detection/train/\"\n# load in training dataframes \ntrain_study_df = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_study_level.csv\")\n\n# add labels\ntrain_study_df['Counts'] = train_study_df[['Negative for Pneumonia','Typical Appearance','Indeterminate Appearance','Atypical Appearance']].idxmax(axis=1)\ntrain_study_df['width-orig'] = 0\ntrain_study_df['height-orig'] = 0\ntrain_study_df['scale'] = 1\ntrain_study_df['Filepath'] = \"\"\n\n\n\n# batch 1\ni = 1\nfor study_uid in os.listdir(train_folder_path):\n    clear_output(wait=True)\n    print(\"Processing image {}/{}\".format(i, len(os.listdir(train_folder_path))))\n    i+=1\n    img_dir = os.listdir(train_folder_path+study_uid)\n    filename = os.listdir(train_folder_path+study_uid+\"/\"+img_dir[0])[0]\n    img = pydicom.read_file(train_folder_path+study_uid+\"/\"+img_dir[0]+\"/\"+filename)\n    try:\n        px = img.pixel_array\n    #     convert to rgb\n        new_px = px - np.min(px)\n        new_px = new_px / np.max(new_px)\n        new_px = (new_px * 255).astype(np.uint8)\n        \n#       label = train_study_df[train_study_df['id']==study_uid+\"_study\"]['Counts'].values[0]\n        if (label=='Negative for Pneumonia'):\n            save_path = \"png_train/NON/\"\n        elif (label=='Typical Appearance'):\n            save_path = \"png_train/COVID/Typical/\"\n        elif (label=='Indeterminate Appearance'):\n            save_path = \"png_train/COVID/Indeterminate/\"\n        else:\n            save_path = save_path = \"png_train/COVID/Atypical/\"\n        \n        img_out = Image.fromarray(new_px)\n        # resize to be no bigger than 1000x1000 and maintin aspect ratio\n        if(img_out.height > 1000 and img_out.height >img_out.width):\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = img_out.height/1000\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n\n        elif(img_out.width >1000 and img_out.width > img_out.height):\n#             img_out = img_out.resize([1000,int(img_out.height/(img_out.width/1000))], Image.LANCZOS)\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = img_out.width/1000\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n        else:\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'height-orig'] = img_out.height\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'width-orig'] = img_out.width\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'scale'] = 1\n            train_study_df.loc[train_study_df.index[train_study_df.id==study_uid+\"_study\"][0],'Filepath'] = save_path+filename.replace(\"dcm\",\"png\")\n        \n    except Exception as e:\n        print(\"Image \\\"{}\\\" failed to convert with the following error: \". format(train_folder_path+study_uid+\"/\"+img_dir[0]+\"/\"+filename))\n        print(e)\ntrain_study_df.to_csv(\"train_study_1000x1000_dimensions.csv\",index=False)\n# shutil.make_archive(\"batch_1\", 'tar', \"png_train/\")","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:53:27.017569Z","iopub.execute_input":"2022-08-24T18:53:27.018074Z","iopub.status.idle":"2022-08-24T19:19:18.436116Z","shell.execute_reply.started":"2022-08-24T18:53:27.018025Z","shell.execute_reply":"2022-08-24T19:19:18.434836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_out.width/1000","metadata":{"execution":{"iopub.status.busy":"2022-08-24T18:52:48.368682Z","iopub.execute_input":"2022-08-24T18:52:48.369138Z","iopub.status.idle":"2022-08-24T18:52:48.375552Z","shell.execute_reply.started":"2022-08-24T18:52:48.369091Z","shell.execute_reply":"2022-08-24T18:52:48.374877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FileLink(r'./train_study_1000x1000_dimensions.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-24T19:19:24.592388Z","iopub.execute_input":"2022-08-24T19:19:24.592936Z","iopub.status.idle":"2022-08-24T19:19:24.602019Z","shell.execute_reply.started":"2022-08-24T19:19:24.592896Z","shell.execute_reply":"2022-08-24T19:19:24.601148Z"},"trusted":true},"execution_count":null,"outputs":[]}]}