{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# WELCOME\n\n**Here in this notebook we will take a look on how to operate with DICOM files format.**","metadata":{}},{"cell_type":"code","source":"# First, let's import useful pyhton libraries\nimport pydicom\nfrom tqdm import tqdm\nimport pandas as pd\nfrom matplotlib import pyplot as plt\nimport os\nimport time\nimport numpy as np\nimport cv2\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:33.051287Z","iopub.execute_input":"2022-07-21T05:20:33.051538Z","iopub.status.idle":"2022-07-21T05:20:33.056516Z","shell.execute_reply.started":"2022-07-21T05:20:33.051510Z","shell.execute_reply":"2022-07-21T05:20:33.055775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## What is Dicom file\n\n[<iframe width=\"853\" height=\"480\" src=\"https://www.youtube.com/embed/eLS9nDVJx5Y\" title=\"What is a DICOM file? Lecture 001\" frameborder=\"0\" allow=\"accelerometer; autoplay; clipboard-write; encrypted-media; gyroscope; picture-in-picture\" allowfullscreen></iframe>](http://)","metadata":{}},{"cell_type":"code","source":"from IPython.display import YouTubeVideo\nYouTubeVideo('eLS9nDVJx5Y', width=800, height=450)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:33.070483Z","iopub.execute_input":"2022-07-21T05:20:33.071121Z","iopub.status.idle":"2022-07-21T05:20:33.161687Z","shell.execute_reply.started":"2022-07-21T05:20:33.071084Z","shell.execute_reply":"2022-07-21T05:20:33.161043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's take a look on the training csv file.\n\ntrain_df = pd.read_csv('../input/unifesp-xray-bodypart-classification/train.csv')\ntest_df = pd.read_csv('../input/unifesp-x-ray-body-part-classifier/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:33.163532Z","iopub.execute_input":"2022-07-21T05:20:33.164162Z","iopub.status.idle":"2022-07-21T05:20:33.190280Z","shell.execute_reply.started":"2022-07-21T05:20:33.164051Z","shell.execute_reply":"2022-07-21T05:20:33.189638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(8)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:33.191457Z","iopub.execute_input":"2022-07-21T05:20:33.191862Z","iopub.status.idle":"2022-07-21T05:20:33.204314Z","shell.execute_reply.started":"2022-07-21T05:20:33.191826Z","shell.execute_reply":"2022-07-21T05:20:33.203259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Let's  see our Target distribution\n\nbodyparts = {\n0 : 'Abdomen' ,\n1 :'Ankle' ,\n2 :'Cervical Spine',\n3 : 'Chest' ,\n4 :'Clavicles' ,\n5 :'Elbow' ,\n6 :'Feet' ,\n7 : 'Finger' ,\n8 : 'Forearm' ,\n9 : 'Hand' ,\n10 : 'Hip' ,\n11 : 'Knee' ,\n12 : 'Lower Leg' ,\n13 : 'Lumbar Spine' ,\n14 : 'Others' ,\n15 :'Pelvis',\n16 :'Shoulder' ,\n17 :'Sinus' ,\n18 : 'Skull' ,\n19 : 'Thigh' ,\n20 :'Thoracic Spine',\n21: 'Wrist',\n}\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:33.206245Z","iopub.execute_input":"2022-07-21T05:20:33.206622Z","iopub.status.idle":"2022-07-21T05:20:33.213436Z","shell.execute_reply.started":"2022-07-21T05:20:33.206587Z","shell.execute_reply":"2022-07-21T05:20:33.212697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#DICOM file formats contains meta-data that can be useful for deep learning model preprocessing\n#I'll use a funciton created by Felipe Kitamura to list the DICOM Tags from the files\n\ndef dcmtag2table(folder, list_of_tags):\n    \"\"\"\n    # Create a Pandas DataFrame with the <list_of_tags> DICOM tags\n    # from the DICOM files in <folder>\n    # Parameters:\n    #    folder (str): folder to be recursively walked looking for DICOM files.\n    #    list_of_tags (list of strings): list of DICOM tags with no whitespaces.\n    # Returns:\n    #    df (DataFrame): table of DICOM tags from the files in folder.\n    \"\"\"\n    list_of_tags = list_of_tags.copy()\n    items = []\n    table = []\n    filelist = []\n    print(\"Listing all files...\")\n    start = time.time()\n    for root, dirs, files in os.walk(folder, topdown=False):\n        for name in files:\n            filelist.append(os.path.join(root, name))\n    print(\"Time: \" + str(time.time() - start))\n    print(\"Reading files...\")\n    time.sleep(2)\n    for _f in tqdm(filelist):\n        try:\n            ds = pydicom.dcmread(_f, stop_before_pixels=True)\n            items = []\n            items.append(_f)\n\n            for _tag in list_of_tags:\n                if _tag in ds:\n                    items.append(ds.data_element(_tag).value)\n                else:\n                    items.append(\"Not found\")\n\n            table.append((items))\n        except:\n            print(\"Skipping non-DICOM: \" + _f)\n\n            \n    list_of_tags.insert(0, \"Filename\")\n    test = list(map(list, zip(*table)))\n    dictone = {}\n\n    for i, _tag in enumerate (list_of_tags):\n        dictone[_tag] = test[i]\n\n    df = pd.DataFrame(dictone)\n    time.sleep(2)\n    print(\"Finished.\")\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:33.214880Z","iopub.execute_input":"2022-07-21T05:20:33.215400Z","iopub.status.idle":"2022-07-21T05:20:33.237635Z","shell.execute_reply.started":"2022-07-21T05:20:33.215359Z","shell.execute_reply":"2022-07-21T05:20:33.236738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## How to read Images from Dicom using Pydicom","metadata":{}},{"cell_type":"code","source":"from IPython.display import YouTubeVideo\nYouTubeVideo('hWwAFNmPZFQ', width=800, height=450)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:33.239343Z","iopub.execute_input":"2022-07-21T05:20:33.239861Z","iopub.status.idle":"2022-07-21T05:20:33.334177Z","shell.execute_reply.started":"2022-07-21T05:20:33.239825Z","shell.execute_reply":"2022-07-21T05:20:33.333478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tags = ['PhotometricInterpretation','BitsAllocated', 'SOPInstanceUID' ]\ndicom_tags_train =  dcmtag2table('../input/unifesp-xray-bodypart-classification/train', tags)\ndicom_tags_test = dcmtag2table('../input/unifesp-xray-bodypart-classification/test', tags)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:33.335376Z","iopub.execute_input":"2022-07-21T05:20:33.336885Z","iopub.status.idle":"2022-07-21T05:20:47.902198Z","shell.execute_reply.started":"2022-07-21T05:20:33.336847Z","shell.execute_reply":"2022-07-21T05:20:47.901411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicom_tags_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:47.903216Z","iopub.execute_input":"2022-07-21T05:20:47.903792Z","iopub.status.idle":"2022-07-21T05:20:47.918133Z","shell.execute_reply.started":"2022-07-21T05:20:47.903749Z","shell.execute_reply":"2022-07-21T05:20:47.917186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Let's see the differente type of Photometric Interpreation and how it affects the display of an image\n\nprint(dicom_tags_train.PhotometricInterpretation.value_counts())\n\nprint('The following images are with Photometric Interpretation MONOCHROME1')\nn = 0\nfor idx,row in dicom_tags_train[dicom_tags_train.PhotometricInterpretation == 'MONOCHROME1'].iterrows():\n    dicom = pydicom.dcmread(row.Filename)\n    img = dicom.pixel_array\n    plt.imshow(img, cmap = 'gray')\n    plt.show()\n    n += 1\n    if n == 3:\n        break\n\nprint('The following images are with Photometric Interpretation MONOCHROME2')\n\nn = 0\nfor idx, row in dicom_tags_train[dicom_tags_train.PhotometricInterpretation == 'MONOCHROME2'].iterrows():\n    dicom = pydicom.dcmread(row.Filename)\n    img = dicom.pixel_array\n    plt.imshow(img, cmap = 'gray')\n    plt.show()\n    n += 1\n    if n == 3:\n        break","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:47.919445Z","iopub.execute_input":"2022-07-21T05:20:47.920266Z","iopub.status.idle":"2022-07-21T05:20:49.380584Z","shell.execute_reply.started":"2022-07-21T05:20:47.920225Z","shell.execute_reply":"2022-07-21T05:20:49.379947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As you can see, the Photometric Interpretation changes hwo the density of the elements presents in the body will be shown. \nIn MONOCHROME1 the element in the image with less density (air) is brighter, the reverse happens in MONCHROME2","metadata":{}},{"cell_type":"code","source":"#You can use numpy invert() funciton to change the display of the Photometric Interpratation between two different MONCHROMEs\n\nprint('The following images are with Photometric Interpretation MONOCHROME1 but will be displayed as MONCHROME2')\nn = 0\nfor idx,row in dicom_tags_train[dicom_tags_train.PhotometricInterpretation == 'MONOCHROME1'].iterrows():\n    dicom = pydicom.dcmread(row.Filename)\n    img = dicom.pixel_array\n    plt.imshow(np.invert(img), cmap = 'gray')\n    plt.show()\n    n += 1\n    if n == 3:\n        break","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:49.383603Z","iopub.execute_input":"2022-07-21T05:20:49.383975Z","iopub.status.idle":"2022-07-21T05:20:50.135092Z","shell.execute_reply.started":"2022-07-21T05:20:49.383939Z","shell.execute_reply":"2022-07-21T05:20:50.133596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = dicom_tags_train.merge(train_df, on =  'SOPInstanceUID')\ntest = dicom_tags_test.merge(test_df,on =  'SOPInstanceUID')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:50.136408Z","iopub.execute_input":"2022-07-21T05:20:50.137123Z","iopub.status.idle":"2022-07-21T05:20:50.153624Z","shell.execute_reply.started":"2022-07-21T05:20:50.137085Z","shell.execute_reply":"2022-07-21T05:20:50.152849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = ['Abdomen', 'Ankle', 'Cervical Spine',\n       'Chest', 'Clavicles', 'Elbow', 'Feet', 'Finger', 'Forearm', 'Hand',\n       'Hip', 'Knee', 'Lower Leg', 'Lumbar Spine', 'Others', 'Pelvis',\n       'Shoulder', 'Sinus', 'Skull', 'Thigh', 'Thoracic Spine', 'Wrist']","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:50.155039Z","iopub.execute_input":"2022-07-21T05:20:50.155422Z","iopub.status.idle":"2022-07-21T05:20:50.160775Z","shell.execute_reply.started":"2022-07-21T05:20:50.155384Z","shell.execute_reply":"2022-07-21T05:20:50.160148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create Label from target\ndef no_to_label(label):\n    label_list_string = []\n    trimed_label = label.rstrip()\n    label_list = trimed_label.split(\" \")\n    label_list = [int(i) for i in label_list]\n    for label in label_list:\n        label_list_string.append(bodyparts[label])\n    label_string = ' and '.join(label_list_string)\n    return label_string\n\n# Create a new column with label\ntarget_list = train['Target'].tolist()\nlabel_column = []\n\nfor label in tqdm(target_list):\n    label_string = no_to_label(label)\n    label_column.append(label_string)\ntrain['Label'] = label_column\n\n#Now we can see the distribution\ntrain['Label'].unique()\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:50.162343Z","iopub.execute_input":"2022-07-21T05:20:50.163101Z","iopub.status.idle":"2022-07-21T05:20:50.189951Z","shell.execute_reply.started":"2022-07-21T05:20:50.163065Z","shell.execute_reply":"2022-07-21T05:20:50.189173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Grouped_target = train.groupby(by='Label').size()\n%matplotlib inline\nplt.rcParams[\"figure.figsize\"] = (13,8)                              \nGrouped_target.plot.bar()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:50.191557Z","iopub.execute_input":"2022-07-21T05:20:50.192167Z","iopub.status.idle":"2022-07-21T05:20:50.781838Z","shell.execute_reply.started":"2022-07-21T05:20:50.192132Z","shell.execute_reply":"2022-07-21T05:20:50.781120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.Label.value_counts(normalize = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:50.782902Z","iopub.execute_input":"2022-07-21T05:20:50.783260Z","iopub.status.idle":"2022-07-21T05:20:50.793311Z","shell.execute_reply.started":"2022-07-21T05:20:50.783226Z","shell.execute_reply":"2022-07-21T05:20:50.792570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.Label.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:50.794463Z","iopub.execute_input":"2022-07-21T05:20:50.795043Z","iopub.status.idle":"2022-07-21T05:20:50.803934Z","shell.execute_reply.started":"2022-07-21T05:20:50.795006Z","shell.execute_reply":"2022-07-21T05:20:50.803083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We will drop rows from training data when the rows is lesser than 10\nv = train.Label.value_counts()\ntrain = train[train.Label.isin(v.index[v.gt(9)])]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:50.805172Z","iopub.execute_input":"2022-07-21T05:20:50.805902Z","iopub.status.idle":"2022-07-21T05:20:50.813806Z","shell.execute_reply.started":"2022-07-21T05:20:50.805862Z","shell.execute_reply":"2022-07-21T05:20:50.812918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.optimizers import SGD\nfrom tensorflow.keras.utils import to_categorical\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport argparse\nimport cv2\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:50.815027Z","iopub.execute_input":"2022-07-21T05:20:50.815675Z","iopub.status.idle":"2022-07-21T05:20:50.821550Z","shell.execute_reply.started":"2022-07-21T05:20:50.815623Z","shell.execute_reply":"2022-07-21T05:20:50.820925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nle = LabelEncoder()\ntrain['LabelCat'] = le.fit_transform(train['Target'])\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:50.822756Z","iopub.execute_input":"2022-07-21T05:20:50.823342Z","iopub.status.idle":"2022-07-21T05:20:50.833027Z","shell.execute_reply.started":"2022-07-21T05:20:50.823306Z","shell.execute_reply":"2022-07-21T05:20:50.832286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating Images for Training\n!rm -r  ./TrainingImages\n!mkdir ./TrainingImages","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:50.834496Z","iopub.execute_input":"2022-07-21T05:20:50.835060Z","iopub.status.idle":"2022-07-21T05:20:52.321007Z","shell.execute_reply.started":"2022-07-21T05:20:50.835020Z","shell.execute_reply":"2022-07-21T05:20:52.320078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom as dicom\nfrom PIL import Image, ImageOps\nimport os\n\nimages_path = \"./TrainingImages\"\nImagePath = []\nfile_name = train.Filename.to_list()\nsop = train.SOPInstanceUID.to_list()\nfor file,sopId in tqdm(zip(file_name,sop)):\n    ds = dicom.dcmread(file)\n    normalized = ( ds.pixel_array - np.mean(ds.pixel_array) ) / np.std(ds.pixel_array)\n    mat  = ( normalized + 1 ) /2\n    img = Image.fromarray(np.uint8(mat * 255) , 'L')\n    file_path = os.path.join(images_path,sopId+\".png\")\n    img.save(file_path)\n    ImagePath.append(file_path)\n    \ntrain['Imagepath'] = ImagePath","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:20:52.324723Z","iopub.execute_input":"2022-07-21T05:20:52.325040Z","iopub.status.idle":"2022-07-21T05:21:11.039779Z","shell.execute_reply.started":"2022-07-21T05:20:52.325004Z","shell.execute_reply":"2022-07-21T05:21:11.038924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Splitting dataframe from training and validation","metadata":{}},{"cell_type":"code","source":"#We are reserving \ntrain_data = train.sample(frac=0.9)\nval_data = train.loc[~train['Filename'].isin(train_data['Filename'])].copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:21:11.041450Z","iopub.execute_input":"2022-07-21T05:21:11.041739Z","iopub.status.idle":"2022-07-21T05:21:11.052176Z","shell.execute_reply.started":"2022-07-21T05:21:11.041699Z","shell.execute_reply":"2022-07-21T05:21:11.051406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data augmentation we will not be using all augmentation since it x-Ray it will not make sense to do augmentations lise vertical flipping","metadata":{}},{"cell_type":"code","source":"class DataAugmentation:\n    def __init__(self,train,val,batch_size):\n        self.train = train\n        self.val = val\n        self.test = test\n        self.batch_size = batch_size\n        \n    def train_augment(self):\n        train_datagen = ImageDataGenerator(\n            brightness_range=[0.4,1.5],# brightness\n        width_shift_range=0.2, # horizontal shift\n        height_shift_range=0.2, # vertical shift\n        zoom_range=0.2, # zoom\n        horizontal_flip=True)\n        \n        \n        train_generator_df = train_datagen.flow_from_dataframe(dataframe=self.train, \n                                              directory=None,\n                                              class_mode='raw',\n                                              x_col=\"Imagepath\", \n                                              y_col=\"LabelCat\", \n                                              target_size=(224, 224), \n                                              batch_size=self.batch_size,\n                                              rescale=1.0/255,\n                                              seed=2020)\n        \n        return train_generator_df\n    \n    \n    def valid_augment(self):\n        val_datagen = ImageDataGenerator()\n        \n        val_generator_df = val_datagen.flow_from_dataframe(dataframe=self.val, \n                                              directory=None,\n                                              class_mode='raw',            \n                                              x_col=\"Imagepath\", \n                                              y_col=\"LabelCat\", \n                                              target_size=(224, 224), \n                                              batch_size=self.batch_size,\n                                              rescale=1.0/255,\n                                              seed=2021)\n        \n        return val_generator_df\n        \n    \n        \n        \n        \n    \n        \n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:21:11.053344Z","iopub.execute_input":"2022-07-21T05:21:11.054781Z","iopub.status.idle":"2022-07-21T05:21:11.067866Z","shell.execute_reply.started":"2022-07-21T05:21:11.054741Z","shell.execute_reply":"2022-07-21T05:21:11.066967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Augment image for training\ndat_aug = DataAugmentation(train_data,val_data,64)\ntrain_gen = dat_aug.train_augment()\nval_gen = dat_aug.valid_augment()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:21:11.069701Z","iopub.execute_input":"2022-07-21T05:21:11.070397Z","iopub.status.idle":"2022-07-21T05:21:11.111147Z","shell.execute_reply.started":"2022-07-21T05:21:11.070362Z","shell.execute_reply":"2022-07-21T05:21:11.110498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Resnet model and train model ","metadata":{}},{"cell_type":"code","source":"from keras.applications.resnet_v2 import ResNet50V2\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:21:11.112613Z","iopub.execute_input":"2022-07-21T05:21:11.113200Z","iopub.status.idle":"2022-07-21T05:21:11.117266Z","shell.execute_reply.started":"2022-07-21T05:21:11.113154Z","shell.execute_reply":"2022-07-21T05:21:11.116439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resNet50 = ResNet50V2(weights=\"imagenet\",input_shape=(224,224,3),include_top=False)\nresNet50.trainable = False","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:21:11.119062Z","iopub.execute_input":"2022-07-21T05:21:11.119813Z","iopub.status.idle":"2022-07-21T05:21:12.422298Z","shell.execute_reply.started":"2022-07-21T05:21:11.119775Z","shell.execute_reply":"2022-07-21T05:21:12.421557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Freezing resNet50 layers for training and making last few layers ready for training","metadata":{}},{"cell_type":"code","source":"# deep learning libraries\nimport tensorflow as tf\nfrom keras import Sequential\nfrom keras.layers import Flatten,Dense,BatchNormalization,Dropout,LeakyReLU,GlobalAveragePooling2D\n\nmodel = tf.keras.Sequential([\n        resNet50,  \n        tf.keras.layers.BatchNormalization(renorm=True),\n        tf.keras.layers.GlobalAveragePooling2D(),\n        tf.keras.layers.Dense(512, activation='relu'),\n        tf.keras.layers.Dense(256, activation='relu'),\n        tf.keras.layers.Dropout(0.5),\n        tf.keras.layers.Dense(128, activation='relu'),\n    # We are using last layer which is equal to number label catagories in training data\n        tf.keras.layers.Dense(len(train['LabelCat'].unique()), activation='softmax')\n    ])\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:21:12.424008Z","iopub.execute_input":"2022-07-21T05:21:12.424266Z","iopub.status.idle":"2022-07-21T05:21:12.825286Z","shell.execute_reply.started":"2022-07-21T05:21:12.424233Z","shell.execute_reply":"2022-07-21T05:21:12.824575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training the model with F1 as metrics","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras import backend as K\n\ndef f1(y_true, y_pred):    \n    def recall_m(y_true, y_pred):\n        TP = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        Positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n        \n        recall = TP / (Positives+K.epsilon())    \n        return recall \n    \n    \n    def precision_m(y_true, y_pred):\n        TP = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        Pred_Positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n    \n        precision = TP / (Pred_Positives+K.epsilon())\n        return precision \n    \n    precision, recall = precision_m(y_true, y_pred), recall_m(y_true, y_pred)\n    \n    return 2*((precision*recall)/(precision+recall+K.epsilon()))","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:21:12.830463Z","iopub.execute_input":"2022-07-21T05:21:12.830750Z","iopub.status.idle":"2022-07-21T05:21:12.845183Z","shell.execute_reply.started":"2022-07-21T05:21:12.830711Z","shell.execute_reply":"2022-07-21T05:21:12.844374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_addons as tfa\n\nmodel.compile(optimizer='Adam',loss='sparse_categorical_crossentropy',metrics=['accuracy',f1])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:21:12.846359Z","iopub.execute_input":"2022-07-21T05:21:12.847180Z","iopub.status.idle":"2022-07-21T05:21:12.863726Z","shell.execute_reply.started":"2022-07-21T05:21:12.847143Z","shell.execute_reply":"2022-07-21T05:21:12.862880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:21:12.864863Z","iopub.execute_input":"2022-07-21T05:21:12.865328Z","iopub.status.idle":"2022-07-21T05:21:12.891207Z","shell.execute_reply.started":"2022-07-21T05:21:12.865292Z","shell.execute_reply":"2022-07-21T05:21:12.890433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nearly = tf.keras.callbacks.EarlyStopping( patience=20,\n                                          min_delta=0.0001,\n                                          restore_best_weights=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:34:23.555938Z","iopub.execute_input":"2022-07-21T05:34:23.556196Z","iopub.status.idle":"2022-07-21T05:34:23.560450Z","shell.execute_reply.started":"2022-07-21T05:34:23.556168Z","shell.execute_reply":"2022-07-21T05:34:23.559769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fit model\nhistory = model.fit(train_gen,\n                    validation_data=val_gen,\n                    epochs=100,\n                    callbacks=[early])","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:34:26.028457Z","iopub.execute_input":"2022-07-21T05:34:26.028738Z","iopub.status.idle":"2022-07-21T05:48:41.845820Z","shell.execute_reply.started":"2022-07-21T05:34:26.028704Z","shell.execute_reply":"2022-07-21T05:48:41.844984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'val'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:49:41.620347Z","iopub.execute_input":"2022-07-21T05:49:41.620717Z","iopub.status.idle":"2022-07-21T05:49:41.851239Z","shell.execute_reply.started":"2022-07-21T05:49:41.620681Z","shell.execute_reply":"2022-07-21T05:49:41.850572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Let us try to predict test data","metadata":{}},{"cell_type":"code","source":"!rm -r  testDir\n!mkdir testDir\n!mkdir testDir/Prediction","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:49:59.812002Z","iopub.execute_input":"2022-07-21T05:49:59.812296Z","iopub.status.idle":"2022-07-21T05:50:01.973184Z","shell.execute_reply.started":"2022-07-21T05:49:59.812263Z","shell.execute_reply":"2022-07-21T05:50:01.972115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_path = \"./testDir/Prediction\"\nImagePath = []\nfile_name = test.Filename.to_list()\nsop = test.SOPInstanceUID.to_list()\nfor file,sopId in tqdm(zip(file_name,sop)):\n    ds = dicom.dcmread(file)\n    normalized = ( ds.pixel_array - np.mean(ds.pixel_array) ) / np.std(ds.pixel_array)\n    mat  = ( normalized + 1 ) /2\n    img = Image.fromarray(np.uint8(mat * 255) , 'L')\n    file_path = os.path.join(images_path,sopId+\".png\")\n    img.save(file_path)\n    ImagePath.append(file_path)\n    \ntest['Imagepath'] = ImagePath","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:50:04.733304Z","iopub.execute_input":"2022-07-21T05:50:04.734037Z","iopub.status.idle":"2022-07-21T05:50:13.417905Z","shell.execute_reply.started":"2022-07-21T05:50:04.733987Z","shell.execute_reply":"2022-07-21T05:50:13.417184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# folder path\ndir_path = './testDir/Prediction'\n\n# list to store files\nos.listdir(dir_path)[:3]","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:50:17.430133Z","iopub.execute_input":"2022-07-21T05:50:17.430384Z","iopub.status.idle":"2022-07-21T05:50:17.439635Z","shell.execute_reply.started":"2022-07-21T05:50:17.430357Z","shell.execute_reply":"2022-07-21T05:50:17.438693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a new column with label\ntarget_list = test['Target'].tolist()\nlabel_column = []\n\nfor label in tqdm(target_list):\n    label_string = no_to_label(label)\n    label_column.append(label_string)\ntest['Label'] = label_column\n\n#Now we can see the distribution\ntest['Label'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:50:20.428533Z","iopub.execute_input":"2022-07-21T05:50:20.429340Z","iopub.status.idle":"2022-07-21T05:50:20.456671Z","shell.execute_reply.started":"2022-07-21T05:50:20.429287Z","shell.execute_reply":"2022-07-21T05:50:20.455962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generate Test generator for prediction","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\n\ntest_datagen = ImageDataGenerator(rescale=1. / 255)\ntest_path = Path('./testDir')\ntest_generator = test_datagen.flow_from_directory(\n    directory= test_path,\n    target_size=(224, 224),\n    batch_size=32,\n    class_mode=None,\n    shuffle=False\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:50:26.079247Z","iopub.execute_input":"2022-07-21T05:50:26.079787Z","iopub.status.idle":"2022-07-21T05:50:26.190109Z","shell.execute_reply.started":"2022-07-21T05:50:26.079748Z","shell.execute_reply":"2022-07-21T05:50:26.189389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predicting the test \nprediction = model.predict(test_generator)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:50:28.950542Z","iopub.execute_input":"2022-07-21T05:50:28.951239Z","iopub.status.idle":"2022-07-21T05:50:31.290169Z","shell.execute_reply.started":"2022-07-21T05:50:28.951182Z","shell.execute_reply":"2022-07-21T05:50:31.289397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_path = './testDir/Prediction'\nfirst_ten_prediction = prediction[:10]\nfirst_ten_images = os.listdir(test_path)[:10]\ndata = []\n\nfor file,prediction in zip(first_ten_images,first_ten_prediction): \n    \n    img = cv2.imread(os.path.join(test_path,file))\n    data.append(img) \n    plt.figure() \n    label = prediction.argmax(axis=-1)\n    title = train.loc[train['LabelCat'] == label, 'Label'].iloc[0]\n    Target = train.loc[train['LabelCat'] == label, 'Target'].iloc[0]\n    plt.title(title)\n    plt.imshow(img) \n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:50:35.212169Z","iopub.execute_input":"2022-07-21T05:50:35.212561Z","iopub.status.idle":"2022-07-21T05:50:38.211178Z","shell.execute_reply.started":"2022-07-21T05:50:35.212525Z","shell.execute_reply":"2022-07-21T05:50:38.210510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Not very correct could be improved","metadata":{}},{"cell_type":"markdown","source":"## Generating submission file","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv('../input/unifesp-xray-bodypart-classification/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:50:54.459437Z","iopub.execute_input":"2022-07-21T05:50:54.459731Z","iopub.status.idle":"2022-07-21T05:50:54.469812Z","shell.execute_reply.started":"2022-07-21T05:50:54.459696Z","shell.execute_reply":"2022-07-21T05:50:54.468923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = model.predict(test_generator)","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:52:27.530941Z","iopub.execute_input":"2022-07-21T05:52:27.531204Z","iopub.status.idle":"2022-07-21T05:52:29.872171Z","shell.execute_reply.started":"2022-07-21T05:52:27.531173Z","shell.execute_reply":"2022-07-21T05:52:29.871389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Generating Target through model predictions\nTarget = []\nfor predict in prediction:\n    label = predict.argmax(axis=-1)\n    Target.append(train.loc[train['LabelCat'] == label, 'Target'].iloc[0])\n    \nsubmission['Target'] = Target","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:52:51.172132Z","iopub.execute_input":"2022-07-21T05:52:51.172397Z","iopub.status.idle":"2022-07-21T05:52:51.374895Z","shell.execute_reply.started":"2022-07-21T05:52:51.172368Z","shell.execute_reply":"2022-07-21T05:52:51.374102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('./submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-21T05:52:53.451422Z","iopub.execute_input":"2022-07-21T05:52:53.451982Z","iopub.status.idle":"2022-07-21T05:52:53.467107Z","shell.execute_reply.started":"2022-07-21T05:52:53.451939Z","shell.execute_reply":"2022-07-21T05:52:53.466252Z"},"trusted":true},"execution_count":null,"outputs":[]}]}