{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"}],"dockerImageVersionId":30823,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport math\nimport cv2\n\nfrom matplotlib import pyplot\nimport matplotlib.patches as patches\n\nfrom skimage import measure\nfrom skimage.transform import resize\n\nimport seaborn as sns\nimport pydicom as dcm\n\nfrom sklearn import metrics\nfrom sklearn.metrics import classification_report\n\nimport keras\nkeras.config.disable_traceback_filtering()\n","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2024-12-28T12:25:56.889021Z","iopub.execute_input":"2024-12-28T12:25:56.889312Z","iopub.status.idle":"2024-12-28T12:26:06.297373Z","shell.execute_reply.started":"2024-12-28T12:25:56.889289Z","shell.execute_reply":"2024-12-28T12:26:06.296627Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !rm -rf /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:06.298335Z","iopub.execute_input":"2024-12-28T12:26:06.298843Z","iopub.status.idle":"2024-12-28T12:26:06.302256Z","shell.execute_reply.started":"2024-12-28T12:26:06.298817Z","shell.execute_reply":"2024-12-28T12:26:06.301514Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')\nimport os\n\n# Disable preallocation of GPU memory\nos.environ[\"XLA_PYTHON_CLIENT_PREALLOCATE\"] = \"false\"\n\n# Configure XLA flags for GPU algorithms and force single-thread compilation\nos.environ[\"XLA_FLAGS\"] = \"--xla_gpu_strict_conv_algorithm_picker=false --xla_gpu_force_compilation_parallelism=1\"\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:06.304081Z","iopub.execute_input":"2024-12-28T12:26:06.304306Z","iopub.status.idle":"2024-12-28T12:26:06.322322Z","shell.execute_reply.started":"2024-12-28T12:26:06.304286Z","shell.execute_reply":"2024-12-28T12:26:06.321662Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow.keras.utils as pltUtil\nfrom tensorflow.keras.utils import Sequence\n\nfrom tensorflow.keras.layers import Layer, Concatenate, UpSampling2D, Conv2D, Reshape, BatchNormalization, Activation\nfrom tensorflow.keras.models import Model\nfrom keras import backend as K\n\n\n#For MobileNet\nfrom tensorflow.keras.applications.mobilenet import MobileNet\nfrom tensorflow.keras.applications.mobilenet import preprocess_input\n\n#For ResNet50\nfrom tensorflow.keras.applications.resnet import ResNet50\n\n#For tunning of model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau, LearningRateScheduler","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:06.323761Z","iopub.execute_input":"2024-12-28T12:26:06.324094Z","iopub.status.idle":"2024-12-28T12:26:06.377301Z","shell.execute_reply.started":"2024-12-28T12:26:06.324058Z","shell.execute_reply":"2024-12-28T12:26:06.376657Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\nprint(\"TensorFlow version:\", tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:06.378026Z","iopub.execute_input":"2024-12-28T12:26:06.37828Z","iopub.status.idle":"2024-12-28T12:26:06.382597Z","shell.execute_reply.started":"2024-12-28T12:26:06.378259Z","shell.execute_reply":"2024-12-28T12:26:06.381955Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dir = '/kaggle/input/rsna-pneumonia-detection-challenge/'\n\ntrain_images = data_dir + 'stage_2_train_images'","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:06.383329Z","iopub.execute_input":"2024-12-28T12:26:06.38356Z","iopub.status.idle":"2024-12-28T12:26:06.394732Z","shell.execute_reply.started":"2024-12-28T12:26:06.38354Z","shell.execute_reply":"2024-12-28T12:26:06.394007Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class image_metadata():\n    \"\"\"\n    Arguments:\n        setName = name of the dataset\n        file = filename\n    \"\"\"\n    \n    def __init__(self, setName, file):\n        self.setName = setName\n        self.file = file\n    \n    def __repr__(self):\n        #a special method used to represent a class's objects as a string\n        return self.imagePath()\n    \n    def imagePath(self):\n        return os.path.join(self.setName, self.file)\n\n\n#this function will create the full path to the image in given dataset","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:06.395522Z","iopub.execute_input":"2024-12-28T12:26:06.395738Z","iopub.status.idle":"2024-12-28T12:26:06.408062Z","shell.execute_reply.started":"2024-12-28T12:26:06.395719Z","shell.execute_reply":"2024-12-28T12:26:06.407353Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#function to load image metadata\ndef loadimagemetadata(dataSetName):\n    \"\"\"\n    Arguments:\n        dataSetName: path of the data set folder\n    \"\"\"\n    \n    imageMetadata = []\n    for f in os.listdir(dataSetName):\n        ext = os.path.splitext(f)[1]\n        if ext == '.dcm':\n            imageMetadata.append(image_metadata(dataSetName, f))\n    return np.array(imageMetadata)\n\n\n## it will load image metadata","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:06.410301Z","iopub.execute_input":"2024-12-28T12:26:06.410493Z","iopub.status.idle":"2024-12-28T12:26:06.424887Z","shell.execute_reply.started":"2024-12-28T12:26:06.410476Z","shell.execute_reply":"2024-12-28T12:26:06.423996Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#function to load image and patientId\ndef loadImage(path):\n    \"\"\"\n    Arguments:\n        path: path of the image\n    \"\"\"\n    \n    img = dcm.dcmread(path)\n    return img\n\ndef getImgId(imgPath):\n    \"\"\"\n    Arguments:\n        imgPath: path of the image\n    \"\"\"\n    \n    return str(imgPath).split(\".dcm\")[0].split(\"/\")[5]","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:06.426379Z","iopub.execute_input":"2024-12-28T12:26:06.426598Z","iopub.status.idle":"2024-12-28T12:26:06.438619Z","shell.execute_reply.started":"2024-12-28T12:26:06.426579Z","shell.execute_reply":"2024-12-28T12:26:06.437985Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trainingSetImageMetadata = loadimagemetadata(train_images)\nprint('Shape of training set image metadata: ', trainingSetImageMetadata.shape)\nprint(\"Sample image path: \", trainingSetImageMetadata[0])","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:06.439296Z","iopub.execute_input":"2024-12-28T12:26:06.439479Z","iopub.status.idle":"2024-12-28T12:26:07.731764Z","shell.execute_reply.started":"2024-12-28T12:26:06.439462Z","shell.execute_reply":"2024-12-28T12:26:07.731031Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imgIndex = 2600\nimgPath = trainingSetImageMetadata[imgIndex]\nimgPath = imgPath.imagePath()\nimgData = loadImage(imgPath)\n\npyplot.imshow(imgData.pixel_array, cmap = pyplot.cm.bone)","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:07.7325Z","iopub.execute_input":"2024-12-28T12:26:07.732721Z","iopub.status.idle":"2024-12-28T12:26:08.171012Z","shell.execute_reply.started":"2024-12-28T12:26:07.732701Z","shell.execute_reply":"2024-12-28T12:26:08.170088Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trainSetImageMetadata_df = pd.DataFrame(trainingSetImageMetadata, columns = ['Path'])\n\nimageIdpaths = pd.DataFrame(columns = ['patientId', 'imgPath'])\n\nimageIdpaths['patientId'] = trainSetImageMetadata_df['Path'].apply(getImgId)\nimageIdpaths['imgPath'] = trainSetImageMetadata_df['Path']\n\nprint(\"Shape of the concerned dataset: \", imageIdpaths.shape)\nprint('The dataset looks as:\\n')\nimageIdpaths.head()\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:08.171853Z","iopub.execute_input":"2024-12-28T12:26:08.172108Z","iopub.status.idle":"2024-12-28T12:26:08.245776Z","shell.execute_reply.started":"2024-12-28T12:26:08.172086Z","shell.execute_reply":"2024-12-28T12:26:08.245142Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"detailclass_df = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv')\ntrainlabels_df = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:08.246458Z","iopub.execute_input":"2024-12-28T12:26:08.24671Z","iopub.status.idle":"2024-12-28T12:26:08.342133Z","shell.execute_reply.started":"2024-12-28T12:26:08.246688Z","shell.execute_reply":"2024-12-28T12:26:08.341167Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"print('The detailed class dataframe has {} rows and {} columns and looks like:'.format(detailclass_df.shape[0], detailclass_df.shape[1]))\ndetailclass_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:08.343074Z","iopub.execute_input":"2024-12-28T12:26:08.343362Z","iopub.status.idle":"2024-12-28T12:26:08.351247Z","shell.execute_reply.started":"2024-12-28T12:26:08.34333Z","shell.execute_reply":"2024-12-28T12:26:08.350524Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('The training label dataframe has {} rows and {} columns and looks like:'.format(trainlabels_df.shape[0], trainlabels_df.shape[1]))\ntrainlabels_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:08.352058Z","iopub.execute_input":"2024-12-28T12:26:08.352359Z","iopub.status.idle":"2024-12-28T12:26:08.372236Z","shell.execute_reply.started":"2024-12-28T12:26:08.352329Z","shell.execute_reply":"2024-12-28T12:26:08.371579Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From our visualization task we know the following things:\n\n* We have a total of 30227 records (in both of the files) and out of that 26684 are unique records, thus some patients having multiple entries in the file.\n* There are two values in the \"Target\" column of the labels file: 0 suggesting No Pneumonia and 1 suggesting that the concerned patient does have Pneumonia.\n* There are 20672 entries in the labels file where bounding box cordinates are not available, thus these belongs to the Patients not having Pneumonia.\n*  There are 9555 entries where the coordinates are provided.\n*  There are 3 Unique classes as observed from the class file: No Lung Opacity/Not Normal, Normal and Lung Opacity\nacity","metadata":{}},{"cell_type":"code","source":"#sorting both the datasets based on patientId\ntrainlabels_df.sort_values('patientId', inplace = True)\ndetailclass_df.sort_values('patientId', inplace = True)\n\n#concatenating the data\nmerge_data_df = pd.concat([trainlabels_df, detailclass_df['class']], axis = 1, sort = False)\nprint('The merged dataset has {} rows and {} columns and looks like:'.format(merge_data_df.shape[0], merge_data_df.shape[1]))\nmerge_data_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:08.373046Z","iopub.execute_input":"2024-12-28T12:26:08.373261Z","iopub.status.idle":"2024-12-28T12:26:08.432577Z","shell.execute_reply.started":"2024-12-28T12:26:08.373241Z","shell.execute_reply":"2024-12-28T12:26:08.431693Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Both the dataframe has been merged to form a new dataset to have a much more detailed view. From this it is clear that Target with 0 corresponds either to \"No Lung Opacity / Not Normal\" or \"Normal\" and 1 with \"Lung Opacity\".","metadata":{}},{"cell_type":"markdown","source":"**Prepare data for Training**\n\n* \nWe will convert the dataset to have just two classes for our ease of doing\n*  Going forward Target 0 will correspond to Normal class\n*  whereas Target 1 will corresponds with Lung Opacity.","metadata":{}},{"cell_type":"code","source":"# Identify rows with 'No Lung Opacity / Not Normal'\nno_opacity_rows = merge_data_df[merge_data_df['class'] == 'No Lung Opacity / Not Normal']\n\n# Randomly sample 40% of these rows\nrows_to_remove = no_opacity_rows.sample(frac=0.95, random_state=42)\n\n# Remove the sampled rows from the dataset\nmerge_data_df = merge_data_df.drop(rows_to_remove.index)\n\nprint('The merge dataset now looks like:')\nmerge_data_df.head()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:08.433611Z","iopub.execute_input":"2024-12-28T12:26:08.433859Z","iopub.status.idle":"2024-12-28T12:26:08.457979Z","shell.execute_reply.started":"2024-12-28T12:26:08.433838Z","shell.execute_reply":"2024-12-28T12:26:08.457131Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Splitting of the dataset**\n\nWe will split the merge_data_df to have three different datasets: train, validation and test.","metadata":{}},{"cell_type":"code","source":"print(f'No of entries which have Pneumonia: {merge_data_df[merge_data_df.Target == 1].shape[0]} i.e., {round(merge_data_df[merge_data_df.Target == 1].shape[0] / merge_data_df.shape[0] * 100, 0)}%')\nprint(f'No of entries which don\\'t have Pneumonia: {merge_data_df[merge_data_df.Target == 0].shape[0]} i.e., {round(merge_data_df[merge_data_df.Target == 0].shape[0] / merge_data_df.shape[0] * 100, 0)}%')\n\n# Plot the pie chart\n_ = merge_data_df['Target'].value_counts().plot(\n    kind='pie', \n    autopct='%.0f%%', \n    labels=['Negative', 'Positive'], \n    figsize=(10, 6)\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:08.458688Z","iopub.execute_input":"2024-12-28T12:26:08.45888Z","iopub.status.idle":"2024-12-28T12:26:08.591667Z","shell.execute_reply.started":"2024-12-28T12:26:08.458862Z","shell.execute_reply":"2024-12-28T12:26:08.590388Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\n# Filter imageIdpaths to include only patientIds that exist in merge_data_df\nimageIdpaths = imageIdpaths[imageIdpaths['patientId'].isin(merge_data_df['patientId'])]\n\n# Ensure the dataframes are sorted by 'patientId' before splitting\nmerge_data_df.sort_values(\"patientId\", inplace=True)\nimageIdpaths.sort_values(\"patientId\", inplace=True)\n\n# Define split ratios\ntrain_ratio = 0.7\nvalidate_ratio = 0.2\ntest_ratio = 0.1\n\n# Step 1: Get unique patient IDs and shuffle them\nunique_patients = merge_data_df['patientId'].unique()\nnp.random.shuffle(unique_patients)\n\n# Step 2: Determine split indices\nn_total = len(unique_patients)\nn_train = int(n_total * train_ratio)\nn_validate = int(n_total * validate_ratio)\n\ntrain_patientIds = unique_patients[:n_train]\nvalidate_patientIds = unique_patients[n_train:n_train + n_validate]\ntest_patientIds = unique_patients[n_train + n_validate:]\n\n# Step 3: Split `merge_data_df` based on patient IDs\ntrain_mergedata = merge_data_df[merge_data_df['patientId'].isin(train_patientIds)]\nvalidate_mergedata = merge_data_df[merge_data_df['patientId'].isin(validate_patientIds)]\ntest_mergedata = merge_data_df[merge_data_df['patientId'].isin(test_patientIds)]\n\n# Step 4: Align `imageIdpaths` with these splits\ntrain_imageIdpaths = imageIdpaths[imageIdpaths['patientId'].isin(train_patientIds)]\nvalidate_imageIdpaths = imageIdpaths[imageIdpaths['patientId'].isin(validate_patientIds)]\ntest_imageIdpaths = imageIdpaths[imageIdpaths['patientId'].isin(test_patientIds)]\n\n# Verify results\nprint(\"Shapes of data splits:\")\nprint(\"Training data: \", train_mergedata.shape)\nprint(\"Validation data: \", validate_mergedata.shape)\nprint(\"Test data: \", test_mergedata.shape)\n\nprint(\"Number of unique patients in training dataset: \", train_mergedata['patientId'].nunique())\nprint(\"Number of unique patients in validation dataset: \", validate_mergedata['patientId'].nunique())\nprint(\"Number of unique patients in test dataset: \", test_mergedata['patientId'].nunique())\n\nprint(\"Shapes of imageIdpaths splits:\")\nprint(\"Training imageIdpaths: \", train_imageIdpaths.shape)\nprint(\"Validation imageIdpaths: \", validate_imageIdpaths.shape)\nprint(\"Test imageIdpaths: \", test_imageIdpaths.shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:08.593046Z","iopub.execute_input":"2024-12-28T12:26:08.593461Z","iopub.status.idle":"2024-12-28T12:26:08.70531Z","shell.execute_reply.started":"2024-12-28T12:26:08.59342Z","shell.execute_reply":"2024-12-28T12:26:08.7046Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_mergedata.fillna(0, inplace = True)\nvalidate_mergedata.fillna(0, inplace = True)\ntest_mergedata.fillna(0,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:08.70596Z","iopub.execute_input":"2024-12-28T12:26:08.706185Z","iopub.status.idle":"2024-12-28T12:26:08.713603Z","shell.execute_reply.started":"2024-12-28T12:26:08.706164Z","shell.execute_reply":"2024-12-28T12:26:08.712698Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Number of unique patients in training dataset: \", train_mergedata[\"patientId\"].nunique())\nprint(\"Number of unique patients in validation dataset: \", validate_mergedata[\"patientId\"].nunique())\nprint(\"Number of unique patients in test dataset: \", test_mergedata[\"patientId\"].nunique())","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:08.714511Z","iopub.execute_input":"2024-12-28T12:26:08.714795Z","iopub.status.idle":"2024-12-28T12:26:08.727551Z","shell.execute_reply.started":"2024-12-28T12:26:08.714768Z","shell.execute_reply":"2024-12-28T12:26:08.726585Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"No of entries which has Pneumonia: {train_mergedata[train_mergedata.Target == 1].shape[0]} i.e., {round(train_mergedata[train_mergedata.Target == 1].shape[0] / train_mergedata.shape[0] * 100, 0)}%\")\nprint(f\"No of entries which don't have Pneumonia: {train_mergedata[train_mergedata.Target == 0].shape[0]} i.e., {round(train_mergedata[train_mergedata.Target == 0].shape[0] / train_mergedata.shape[0] * 100, 0)}%\")\n\n_ = train_mergedata['Target'].value_counts().plot(\n    kind='pie', \n    autopct='%.0f%%', \n    labels=['Negative', 'Positive'], \n    figsize=(10, 6)\n)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:08.728543Z","iopub.execute_input":"2024-12-28T12:26:08.728874Z","iopub.status.idle":"2024-12-28T12:26:08.820224Z","shell.execute_reply.started":"2024-12-28T12:26:08.728842Z","shell.execute_reply":"2024-12-28T12:26:08.819273Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"No of entries which has Pneumonia: {validate_mergedata[validate_mergedata.Target == 1].shape[0]} i.e., {round(validate_mergedata[validate_mergedata.Target == 1].shape[0] / validate_mergedata.shape[0] * 100, 0)}%\")\nprint(f\"No of entries which don't have Pneumonia: {validate_mergedata[validate_mergedata.Target == 0].shape[0]} i.e., {round(validate_mergedata[validate_mergedata.Target == 0].shape[0] / validate_mergedata.shape[0] * 100, 0)}%\")\n\n_ = validate_mergedata['Target'].value_counts().plot(\n    kind='pie', \n    autopct='%.0f%%', \n    labels=['Negative', 'Positive'], \n    figsize=(10, 6)\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:08.824111Z","iopub.execute_input":"2024-12-28T12:26:08.824347Z","iopub.status.idle":"2024-12-28T12:26:08.913534Z","shell.execute_reply.started":"2024-12-28T12:26:08.824325Z","shell.execute_reply":"2024-12-28T12:26:08.912618Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"No of entries which has Pneumonia: {test_mergedata[test_mergedata.Target == 1].shape[0]} i.e., {round(test_mergedata[test_mergedata.Target == 1].shape[0] / test_mergedata.shape[0] * 100, 0)}%\")\nprint(f\"No of entries which don't have Pneumonia: {test_mergedata[test_mergedata.Target == 0].shape[0]} i.e., {round(test_mergedata[test_mergedata.Target == 0].shape[0] / test_mergedata.shape[0] * 100, 0)}%\")\n\n_ = test_mergedata['Target'].value_counts().plot(\n    kind='pie', \n    autopct='%.0f%%', \n    labels=['Negative', 'Positive'], \n    figsize=(10, 6)\n)\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:08.915496Z","iopub.execute_input":"2024-12-28T12:26:08.915736Z","iopub.status.idle":"2024-12-28T12:26:09.004546Z","shell.execute_reply.started":"2024-12-28T12:26:08.915714Z","shell.execute_reply":"2024-12-28T12:26:09.003572Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Functions and Class Definitions\nHere we will define some of the functions that will help us in creating a model\n\n * **Loss and Metric Functions**\n   We will be using \"Jaccard Loss Function\" as the loss function and \"Mean IOU\" to measure the efficiency of the model.\ndel.","metadata":{}},{"cell_type":"code","source":"# define iou or jaccard loss function\ndef iou_loss(y_true, y_pred):\n    \"\"\"\n    Arguments:\n        y_true -- ground truth mask \n        y_pred -- predicted mask\n    \"\"\"\n    \n    y_true = tf.reshape(y_true, [-1])\n    y_pred = tf.reshape(y_pred, [-1])\n    intersection = tf.reduce_sum(y_true * y_pred)\n    score = (intersection + 1.) / (tf.reduce_sum(y_true) + tf.reduce_sum(y_pred) - intersection + 1.)\n    return 1 - score\n\n\ndef mean_iou(y_true, y_pred):\n    \"\"\"\n    Arguments:\n        y_true -- ground truth mask\n        y_pred -- predicted mask\n    \"\"\"\n    # Ensure y_true has the same shape and type as y_pred\n    y_true = tf.cast(y_true, tf.float32)  # Convert y_true to float32 for consistency\n    y_pred = tf.cast(y_pred, tf.float32)  # Convert y_pred to float32 for consistency\n    \n    # Expand dimensions of y_true to match the shape of y_pred (if needed)\n    if len(y_true.shape) < len(y_pred.shape):\n        y_true = tf.expand_dims(y_true, axis=-1)  # Add the channel dimension to y_true if missing\n    \n    # Round the predictions to get binary values\n    y_pred = tf.round(y_pred)\n    \n    # Calculate intersection and union\n    intersect = tf.reduce_sum(y_true * y_pred, axis=[1, 2])  # Sum over spatial dimensions (height and width)\n    union = tf.reduce_sum(y_true, axis=[1, 2]) + tf.reduce_sum(y_pred, axis=[1, 2])\n\n    # Add a smoothing term to avoid division by zero\n    smooth = tf.ones_like(intersect, dtype=tf.float32)  # Ensure smooth is float32 to match other tensors\n    \n    # Return the mean IoU\n    return tf.reduce_mean((intersect + smooth) / (union - intersect + smooth))\n\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:09.005564Z","iopub.execute_input":"2024-12-28T12:26:09.005881Z","iopub.status.idle":"2024-12-28T12:26:09.013285Z","shell.execute_reply.started":"2024-12-28T12:26:09.005844Z","shell.execute_reply":"2024-12-28T12:26:09.012015Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* Function to obtain Intersection Over Union (IoU) ratio from Ground Truth and Predicted Box Coordinates","metadata":{}},{"cell_type":"code","source":"def iouFromCoords(boxA, boxB) :\n    \"\"\"\n    Arguments:\n        boxA -- ground truth mask\n        boxB -- predicted mask\n    \"\"\"\n    \n    # determine the (x, y)-coordinates of the intersection rectangle\n    xA = max(boxA[0], boxB[0])\n    yA = max(boxA[1], boxB[1])\n    xB = min(boxA[2], boxB[2])\n    yB = min(boxA[3], boxB[3])\n\n    # compute the area of intersection rectangle\n    intersectionArea = abs(max((xB - xA, 0)) * max((yB - yA), 0))\n    if intersectionArea == 0:\n        return 0\n    \n    # compute the area of both the prediction and ground-truth rectangles\n    boxAArea = abs((boxA[2] - boxA[0]) * (boxA[3] - boxA[1]))\n    boxBArea = abs((boxB[2] - boxB[0]) * (boxB[3] - boxB[1]))\n\n    # compute the intersection over union\n    iou = intersectionArea / float(boxAArea + boxBArea - intersectionArea)\n\n    # return the intersection over union value\n    return iou","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:09.014104Z","iopub.execute_input":"2024-12-28T12:26:09.014382Z","iopub.status.idle":"2024-12-28T12:26:09.030683Z","shell.execute_reply.started":"2024-12-28T12:26:09.01436Z","shell.execute_reply":"2024-12-28T12:26:09.029723Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* Function to display the image with an imposing mask","metadata":{}},{"cell_type":"code","source":"def showMaskedImage(_imageSet, _maskSet, _index) :\n    \"\"\"\n    Arguments:\n        _imageSet -- set of images \n        _maskSet -- set of masks\n        _index -- index of a set/collection\n    \"\"\"\n    \n    maskImage = _imageSet[_index]\n\n    maskImage[:,:,0] = _maskSet[_index] * _imageSet[_index][:,:,0]\n    maskImage[:,:,1] = _maskSet[_index] * _imageSet[_index][:,:,1]\n    maskImage[:,:,2] = _maskSet[_index] * _imageSet[_index][:,:,2]\n\n    pyplot.imshow(maskImage[:,:,0], cmap=pyplot.cm.bone)","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:09.03154Z","iopub.execute_input":"2024-12-28T12:26:09.032123Z","iopub.status.idle":"2024-12-28T12:26:09.047633Z","shell.execute_reply.started":"2024-12-28T12:26:09.032091Z","shell.execute_reply":"2024-12-28T12:26:09.046722Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* Some basic variables are defined which will be used","metadata":{}},{"cell_type":"code","source":"image_size = 224\nimg_width = 1024\nimg_height = 1024\n\ntrain_batch_size = 32\ntest_batch_size = 32","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:09.04873Z","iopub.execute_input":"2024-12-28T12:26:09.049063Z","iopub.status.idle":"2024-12-28T12:26:09.062843Z","shell.execute_reply.started":"2024-12-28T12:26:09.049033Z","shell.execute_reply":"2024-12-28T12:26:09.061997Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* A class is being defined to get data batch for training the model\n","metadata":{}},{"cell_type":"code","source":"#Generator class to get data batch for training the model. It extends the Sequence class\nclass UNetTrainGenerator(Sequence):\n    \"\"\"\n    Arguments:\n        _imgaeIdpaths: dataframe having patientIds and image paths to load the image\n        _mergedata: dataframe having patientId, bounding box coordinates, target and class\n        idx: index of the batch\n    \"\"\"\n    \n    def __init__(self, _imageIdPaths, _mergedata):\n        self.pids = _mergedata['patientId'].to_numpy()\n        self.imgIdPaths = _imageIdPaths\n        self.coords = _mergedata[[\"x\", \"y\", \"width\", \"height\"]].to_numpy()\n        self.coords = self.coords * image_size / img_width\n        \n    def __len__(self):\n        return math.ceil(len(self.coords) / train_batch_size)\n    \n    def __getitem__(self, idx):\n        batch_coords = self.coords[idx * train_batch_size:(idx + 1) * train_batch_size] #image coords\n        batch_pids = self.pids[idx * train_batch_size:(idx + 1) * train_batch_size] #image pids\n        \n        batch_images = np.zeros((len(batch_pids), image_size, image_size, 3), dtype = np.float32)\n        batch_masks = np.zeros((len(batch_pids), image_size, image_size))\n        \n        for _indx, _pid in enumerate(batch_pids):\n            _path = self.imgIdPaths[self.imgIdPaths[\"patientId\"] == _pid]['imgPath'].array[0]\n            _imgData = loadImage(str(_path))\n            img = _imgData.pixel_array\n            \n            resized_image = cv2.resize(img, (image_size, image_size), interpolation = cv2.INTER_AREA)\n            \n            #preprocess image\n            batch_images[_indx][:,:,0] = preprocess_input(np.array(resized_image[:,:], dtype = np.float32))\n            batch_images[_indx][:,:,1] = preprocess_input(np.array(resized_image[:,:], dtype = np.float32))\n            batch_images[_indx][:,:,2] = preprocess_input(np.array(resized_image[:,:], dtype = np.float32))\n            \n            x = int(batch_coords[_indx, 0])\n            y = int(batch_coords[_indx, 1])\n            width = int(batch_coords[_indx, 2]) \n            height = int(batch_coords[_indx, 3])\n            \n            #batch_masks[_indx][y:y+height, x:x+height] = 1\n            batch_masks[_indx][y:y+height, x:x+width] = 1\n        \n        return batch_images, batch_masks\n\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:09.063799Z","iopub.execute_input":"2024-12-28T12:26:09.064153Z","iopub.status.idle":"2024-12-28T12:26:09.078716Z","shell.execute_reply.started":"2024-12-28T12:26:09.064122Z","shell.execute_reply":"2024-12-28T12:26:09.077771Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_generator = UNetTrainGenerator(_imageIdPaths=imageIdpaths, _mergedata=train_mergedata)\n# Fetching a single batch for testing\nbatch_images, batch_masks = train_generator[1]\n\n# Check the outputs\nprint(\"Batch Images Shape:\", batch_images.shape)\nprint(\"Batch Masks Shape:\", batch_masks.shape)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:09.079605Z","iopub.execute_input":"2024-12-28T12:26:09.079907Z","iopub.status.idle":"2024-12-28T12:26:09.679503Z","shell.execute_reply.started":"2024-12-28T12:26:09.079876Z","shell.execute_reply":"2024-12-28T12:26:09.678833Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* This class **UNetTrainGenerator** will help to load the training and validation data into the memory.","metadata":{}},{"cell_type":"markdown","source":"* A class is being defined to get data batch for testing","metadata":{}},{"cell_type":"code","source":"#Generator to Predict the model\nclass UNetTestGenerator(Sequence):\n    \"\"\"        \n    Arguments:\n        _imageIdPaths: dataframe having patientId and image paths to load image\n        _mergedata: dataframe having patientId, bounding box coordinates, target and class      \n        idx -- index of a batch\n    \"\"\"\n    \n    def __init__(self, _imageIdPaths, _mergedata):       \n        self.pids = _mergedata[\"patientId\"].to_numpy()\n        self.imgIdPaths = _imageIdPaths\n        self.coords = _mergedata[[\"x\", \"y\", \"width\", \"height\", \"Target\"]].to_numpy()\n        self.classes = _mergedata[\"class\"]\n        # Resize Bounding box\n        self.coordsOrig = self.coords\n        self.coords = self.coords * image_size / img_width           \n\n    def __len__(self):\n        # Returns total number of batches\n        return math.ceil(len(self.coords) / test_batch_size)\n    \n\n    def __getitem__(self, idx):\n        batch_coords = self.coords[idx * test_batch_size:(idx + 1) * test_batch_size]\n        batch_coordsOrig = self.coordsOrig[idx * test_batch_size:(idx + 1) * test_batch_size]\n        batch_pids = self.pids[idx * test_batch_size:(idx + 1) * test_batch_size]    \n        batch_classes = self.classes[idx * test_batch_size:(idx + 1) * test_batch_size]           \n        if len(batch_pids) == 0:\n            raise StopIteration     \n        batch_images = np.zeros((len(batch_pids), image_size, image_size, 3), dtype=np.float32)\n        batch_masks = np.zeros((len(batch_pids), image_size, image_size))\n        for _indx, _pid in enumerate(batch_pids):\n            _path = self.imgIdPaths[self.imgIdPaths[\"patientId\"] == _pid][\"imgPath\"].array[0]\n            _imgData = loadImage(str(_path))\n            img = _imgData.pixel_array \n            \n            # Resize image\n            resized_img = cv2.resize(img, (image_size, image_size), interpolation=cv2.INTER_AREA)\n            # preprocess image for the batch\n            batch_images[_indx][:,:,0] = preprocess_input(np.array(resized_img[:,:], dtype=np.float32))\n            batch_images[_indx][:,:,1] = preprocess_input(np.array(resized_img[:,:], dtype=np.float32))\n            batch_images[_indx][:,:,2] = preprocess_input(np.array(resized_img[:,:], dtype=np.float32))  \n            \n            x = int(batch_coords[_indx, 0])\n            y = int(batch_coords[_indx, 1])\n            width = int(batch_coords[_indx, 2])\n            height = int(batch_coords[_indx, 3])\n            target = int(batch_coords[_indx, 4])\n            \n            batch_coords[_indx, 0] = x\n            batch_coords[_indx, 1] = y \n            batch_coords[_indx, 2] = width \n            batch_coords[_indx, 3] = height    \n            batch_coords[_indx, 4] = target \n            \n            batch_masks[_indx][y:y+height, x:x+width] = 1\n\n        # Returns images, ground truth masks, patientIds, resized-coordinates, class targets and ground truth coordinates/lables.   \n        return batch_images, batch_masks, batch_pids, batch_coords, batch_classes, batch_coordsOrig\n\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:09.680265Z","iopub.execute_input":"2024-12-28T12:26:09.680474Z","iopub.status.idle":"2024-12-28T12:26:09.68971Z","shell.execute_reply.started":"2024-12-28T12:26:09.680455Z","shell.execute_reply":"2024-12-28T12:26:09.689032Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_generator = UNetTestGenerator(_imageIdPaths=imageIdpaths, _mergedata=train_mergedata\n                                  )\nbatch_images1, batch_masks1, _, _, _, _ = test_generator[1]\n\n\n# Check the outputs\nprint(\"Batch Images Shape:\", batch_images1.shape)\nprint(\"Batch Masks Shape:\", batch_masks1.shape)\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:09.690448Z","iopub.execute_input":"2024-12-28T12:26:09.690667Z","iopub.status.idle":"2024-12-28T12:26:10.036606Z","shell.execute_reply.started":"2024-12-28T12:26:09.690648Z","shell.execute_reply":"2024-12-28T12:26:10.035701Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* The above defined class UNetTestGenerator will help to load the test data.","metadata":{}},{"cell_type":"markdown","source":"# Creating functions for Model Architecture\n\nResNet50","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.layers import Input, UpSampling2D, Concatenate, Conv2D, BatchNormalization, Dropout\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.regularizers import l2\n\ndef ResNet_model(input_shape=(224, 224, 3), dropout_rate=0.1, l2_reg=1e-4):\n    # Load ResNet50 backbone\n    base_model = ResNet50(weights='imagenet', include_top=False, input_tensor=Input(shape=input_shape))\n    base_model.trainable = True  # Fine-tune the entire model\n\n    # Extract intermediate layers\n    block1 = base_model.get_layer(\"conv1_relu\").output  # 112x112x64\n    block2 = base_model.get_layer(\"conv2_block3_out\").output  # 56x56x256\n    block3 = base_model.get_layer(\"conv3_block4_out\").output  # 28x28x512\n    block4 = base_model.get_layer(\"conv4_block6_out\").output  # 14x14x1024\n    block5 = base_model.get_layer(\"conv5_block3_out\").output  # 7x7x2048\n\n    # Decoder path\n    x = Concatenate()([UpSampling2D()(block5), block4])  # Combine block5 and block4\n    x = Conv2D(256, (3, 3), padding=\"same\", activation=\"relu\", kernel_regularizer=l2(l2_reg))(x)\n    x = BatchNormalization()(x)\n    x = Dropout(dropout_rate)(x)\n\n    x = Concatenate()([UpSampling2D()(x), block3])  # Combine with block3\n    x = Conv2D(128, (3, 3), padding=\"same\", activation=\"relu\", kernel_regularizer=l2(l2_reg))(x)\n    x = BatchNormalization()(x)\n    x = Dropout(dropout_rate)(x)\n\n    x = Concatenate()([UpSampling2D()(x), block2])  # Combine with block2\n    x = Conv2D(64, (3, 3), padding=\"same\", activation=\"relu\", kernel_regularizer=l2(l2_reg))(x)\n    x = BatchNormalization()(x)\n    x = Dropout(dropout_rate)(x)\n\n    x = Concatenate()([UpSampling2D()(x), block1])  # Combine with block1\n    x = Conv2D(32, (3, 3), padding=\"same\", activation=\"relu\", kernel_regularizer=l2(l2_reg))(x)\n    x = BatchNormalization()(x)\n    x = Dropout(dropout_rate)(x)\n\n    # Final upsampling and output layer\n    x = UpSampling2D()(x)\n    x = Conv2D(1, (1, 1), activation=\"sigmoid\")(x)  # Single-channel output for binary segmentation\n\n    # Create the model\n    model = Model(inputs=base_model.input, outputs=x)\n    return model\n\n\n\n\n\n\n# from tensorflow.keras.applications.resnet50 import ResNet50\n# from tensorflow.keras.layers import Concatenate, UpSampling2D,Conv2D,Reshape\n# from tensorflow.keras.models import Model\n# from tensorflow.keras.layers import Dense\n# from tensorflow.keras import Sequential\n\n# model=ResNet50()\n# def ResNet_model():\n#   model = Sequential()\n#   model = ResNet50()\n#   for layer in model.layers[:-10]:\n#       layer.trainable = True\n\n\n#   block1 = model.get_layer(\"conv1_relu\").output\n#   block2 = model.get_layer(\"conv2_block3_out\").output\n#   block3 = model.get_layer(\"conv3_block4_out\").output\n#   block4 = model.get_layer(\"conv4_block6_out\").output\n#   block5 = model.get_layer(\"conv5_block3_out\").output\n  \n  \n#   x = Concatenate()([UpSampling2D()(block5), block4])\n#   x = Conv2D(100, (1, 1), activation='relu') (x)\n#   x = Concatenate()([UpSampling2D()(x), block3])\n#   x = Conv2D(100, (1, 1), activation='relu') (x)\n#   x = Concatenate()([UpSampling2D()(x), block2])\n#   x = Conv2D(100, (1, 1), activation='relu') (x)\n#   x = Concatenate()([UpSampling2D()(x), block1])\n#   x = Conv2D(100, (1, 1), activation='relu') (x)\n#   x = UpSampling2D()(x)\n#   x = Conv2D(1, kernel_size=1,strides=1, activation=\"sigmoid\")(x)\n#   x = Reshape((224, 224,1))(x)\n\n#   return Model(inputs=model.input, outputs=x)","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:10.037498Z","iopub.execute_input":"2024-12-28T12:26:10.037783Z","iopub.status.idle":"2024-12-28T12:26:10.047414Z","shell.execute_reply.started":"2024-12-28T12:26:10.037759Z","shell.execute_reply":"2024-12-28T12:26:10.046712Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Some more necessary functions\nFunction to create Confusion Matrix and Classification Report","metadata":{}},{"cell_type":"code","source":"def showConfusionMatrix(IOU_report) :\n    \"\"\"   \n    Arguments:\n        IOU_report: dataframe having target and prediction columns.\n    \"\"\"\n    \n    IOU_report.fillna(0, inplace=True)\n    \n    # Get Targets and Predictions\n    y_IOU_test = IOU_report[\"Target\"]\n    y_IOU_predicted = IOU_report[\"predTarget\"]\n    print(\"Predictions in terms of IOU :\\n\")\n    print(\"Confusion Matrix:\\n\", metrics.confusion_matrix(y_IOU_test, y_IOU_predicted))\n    print(\"\\nClassification Report:\\n\", metrics.classification_report(y_IOU_test, y_IOU_predicted))","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:10.048314Z","iopub.execute_input":"2024-12-28T12:26:10.048613Z","iopub.status.idle":"2024-12-28T12:26:10.062265Z","shell.execute_reply.started":"2024-12-28T12:26:10.048583Z","shell.execute_reply":"2024-12-28T12:26:10.061402Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* Function to predict test data set and save the submission report into a csv file","metadata":{}},{"cell_type":"code","source":"def predictBatches(_test_mergedata, _test_imageIdpaths, _UNetModel):\n    \"\"\"        \n    Arguments:\n        _test_mergedata: DataFrame containing patientId, bounding box coordinates, target, and class for test samples.\n        _test_imageIdpaths: DataFrame containing patientId and image paths for loading test images.\n        _UNetModel: Pre-trained UNet model used for predicting test data.\n    \"\"\"\n    \n    print('Number of Test Samples:', _test_mergedata[\"patientId\"].nunique())\n    \n    # Create test generator instance\n    testUNetDataGen = UNetTestGenerator(_test_imageIdpaths, _test_mergedata)\n    \n    # Initialize submission DataFrame\n    submissionDF = pd.DataFrame(columns=[\n        'patientId', 'x', 'y', 'width', 'height', 'Target', 'class', \n        'x_pred', 'y_pred', 'width_pred', 'height_pred', 'predTarget', 'iou', 'class_pred'\n    ])\n    dfIndex = 0\n    iouThreshold = 0.3  # IoU threshold of 30%\n\n    print(\"Predicting Batches \", end='')\n    for batchImages, gtBatchMasks, batchPids, batchCoords, batchClasses, batchCoordsOrig in testUNetDataGen:\n        try:\n            print(\"Batch images shape:\", batchImages.shape)\n            # Predict batch of images\n            batchPreds = _UNetModel.predict(batchImages, verbose=0)\n\n            prevPid = \"\"\n            # Loop through batch\n            for pred, gtMask, pid, coords, gtClass, coordsOrig in zip(\n                batchPreds, gtBatchMasks, batchPids, batchCoords, batchClasses, batchCoordsOrig\n            ):\n                if prevPid != pid:\n                    prevPid = pid\n\n                    # Resize predicted mask\n                    pred = resize(pred, (1024, 1024), mode='reflect')\n                    coords = coordsOrig  # Recompute coords for resized prediction\n\n                    # Threshold predicted mask\n                    strongPred = pred[:, :] > 0.5\n                    strongPred = measure.label(strongPred)\n\n                    iouCoordsDF = pd.DataFrame(columns=['iou', 'x', 'y', 'width', 'height'])\n                    for region in measure.regionprops(strongPred):\n                        y, x, _, y2, x2, _ = region.bbox\n                        height = y2 - y\n                        width = x2 - x\n\n                        # Get IoU\n                        coordsXYs = np.array([coords[0], coords[1], coords[2] + coords[0], coords[3] + coords[1]])\n                        regionXYs = np.array([x, y, x2, y2])\n                        IOU = iouFromCoords(coordsXYs, regionXYs)\n                        iouCoordsDF.loc[len(iouCoordsDF)] = [IOU, x, y, width, height]\n\n                    GTDFRow = [pid, coords[0], coords[1], coords[2], coords[3], coords[4], gtClass]\n                    prevGTDFRow = []\n\n                    # Get top 2 predictions based on IoU\n                    iouCoordsDF.sort_values(\"iou\", ascending=False, inplace=True)\n                    predIOUCoordCount = 0\n\n                    if len(iouCoordsDF) > 0:\n                        for predIOUCoordIdx in range(len(iouCoordsDF)):\n                            if iouCoordsDF.iloc[predIOUCoordIdx][\"iou\"] > iouThreshold:\n                                submissionDFRow = [\n                                    pid, coords[0], coords[1], coords[2], coords[3], coords[4], \n                                    gtClass, int(iouCoordsDF.iloc[predIOUCoordIdx][\"x\"]), \n                                    int(iouCoordsDF.iloc[predIOUCoordIdx][\"y\"]), \n                                    int(iouCoordsDF.iloc[predIOUCoordIdx][\"width\"]), \n                                    int(iouCoordsDF.iloc[predIOUCoordIdx][\"height\"]), \n                                    1, iouCoordsDF.iloc[predIOUCoordIdx][\"iou\"], \"Lung Opacity\"\n                                ]\n                                if predIOUCoordCount < 2 and GTDFRow != prevGTDFRow:\n                                    submissionDF.loc[dfIndex] = submissionDFRow\n                                    dfIndex += 1\n                                    predIOUCoordCount += 1\n                                    prevGTDFRow = GTDFRow\n                            else:\n                                if GTDFRow != prevGTDFRow:\n                                    submissionDFRow = [\n                                        pid, coords[0], coords[1], coords[2], coords[3], coords[4], \n                                        gtClass, 0, 0, 0, 0, 0, \n                                        iouCoordsDF.iloc[predIOUCoordIdx][\"iou\"], \"Normal\"\n                                    ]\n                                    submissionDF.loc[dfIndex] = submissionDFRow\n                                    dfIndex += 1\n                                    prevGTDFRow = GTDFRow\n                                    break\n                    else:\n                        submissionDFRow = [\n                            pid, coords[0], coords[1], coords[2], coords[3], coords[4], \n                            gtClass, 0, 0, 0, 0, 0, 'NA', \"Normal\"\n                        ]\n                        submissionDF.loc[dfIndex] = submissionDFRow\n                        dfIndex += 1\n\n        except Exception as e:\n            print(f\"Error encountered in batch: {e}\")\n            continue\n\n    # Save submission data to CSV\n    submissionDF.to_csv('submission_res.csv', index=False)\n    print(\"Prediction Complete!\")\n    \n    test_y = submissionDF[\"Target\"]\n    predicted_y = submissionDF[\"predTarget\"]\n    \n    return test_y.apply(int), predicted_y.apply(int)\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:10.062992Z","iopub.execute_input":"2024-12-28T12:26:10.063216Z","iopub.status.idle":"2024-12-28T12:26:10.078214Z","shell.execute_reply.started":"2024-12-28T12:26:10.063196Z","shell.execute_reply":"2024-12-28T12:26:10.077456Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* Function to visualize predictions by displaying the ground truth and predicted bounding box","metadata":{}},{"cell_type":"code","source":"def visualizePredictions(_predReportDF, _topNum) :\n    \"\"\"        \n    Arguments:\n        _predReportDF: dataframe having patientId, IoUs, target and prediction coordinate columns.\n        _topNum -- number indicating the count of top predictions to be visualized.\n    \"\"\"\n    # Sort on IOU to get higher IOUs on top\n    _predReportDF.sort_values(\"iou\", ascending=False, inplace=True)\n    # Get patientIds\n    topPids = _predReportDF[\"patientId\"].head(_topNum)\n    topPidsAry = np.array(topPids)\n    # Get IOUs\n    topIOUs = _predReportDF[\"iou\"].head(_topNum)\n    topIOUsAry = np.array(topIOUs)\n\n    # To get ground truth images for top IOU scored pids\n    imageCollc = np.zeros((_topNum, img_width, img_height), np.float32)\n\n    # Get ground truth coordinates for top IOU scored rows and prepare masks\n    gtCoordCollc = _predReportDF[[\"x\", \"y\", \"width\", \"height\"]].to_numpy()\n    # To get ground truth masks\n    gtMaskCollc  = np.zeros((_topNum, img_width, img_height), int)\n\n    # Get ground truth coordinates for top IOU scored rows and prepare masks\n    predCoordCollc = _predReportDF[[\"x_pred\", \"y_pred\", \"width_pred\", \"height_pred\"]].to_numpy()  # (1024, 1024)\n    # To get ground truth masks\n    predMaskCollc  = np.zeros((_topNum, img_width, img_height), int)\n\n    # Get ground truth and prediction masks\n    for indx in range(0, _topNum) :\n        # Get images\n        path = test_imageIdpaths[test_imageIdpaths[\"patientId\"] == topPidsAry[indx]][\"imgPath\"].array[0]\n        imgData = loadImage(str(path)) # Read image\n        img = imgData.pixel_array\n        imageCollc[indx][:,:] = preprocess_input(np.array(img[:,:], dtype=np.float32)) # Convert to float32 array\n\n        # prepare ground truth masks\n        x = int(gtCoordCollc[indx, 0])\n        y = int(gtCoordCollc[indx, 1])\n        width = int(gtCoordCollc[indx, 2])\n        height = int(gtCoordCollc[indx, 3])\n        gtMaskCollc[indx][y:y+height, x:x+width] = 1   # (1024, 1024)\n\n        # prepare predicted masks\n        x_pred = int(predCoordCollc[indx, 0])\n        y_pred = int(predCoordCollc[indx, 1])\n        width_pred = int(predCoordCollc[indx, 2])\n        height_pred = int(predCoordCollc[indx, 3])\n        predMaskCollc[indx][y_pred:y_pred+height_pred, x_pred:x_pred+width_pred] = 1   # (1024, 1024)\n        \n    # Show images and bounding boxes\n    imageArea, axesArry = pyplot.subplots(int(_topNum/2), 2, figsize=(18,18))\n    axesArry = axesArry.ravel()\n    for axidx in range(0, _topNum) :\n        axesArry[axidx].imshow(imageCollc[axidx][:, :], cmap=pyplot.cm.bone)\n\n        gtComp = gtMaskCollc[axidx][:, :] > 0.5\n        # apply connected components\n        gtComp = measure.label(gtComp)\n        # apply ground truth bounding boxes\n        for region in measure.regionprops(gtComp):\n            # retrieve x, y, height and width\n            y1, x1, y2, x2 = region.bbox\n            heightReg = y2 - y1\n            widthReg = x2 - x1\n            axesArry[axidx].add_patch(patches.Rectangle((x1, y1), widthReg, heightReg, linewidth=1, edgecolor='r', \n                                                        facecolor='none'))\n\n        predComp = predMaskCollc[axidx][:, :] > 0.5\n        # apply connected components\n        predComp = measure.label(predComp)\n        # apply predicted bounding boxes\n        for region_pred in measure.regionprops(predComp):\n            # retrieve x, y, height and width\n            y1_pred, x1_pred, y2_pred, x2_pred = region_pred.bbox\n            heightReg_pred = y2_pred - y1_pred\n            widthReg_pred = x2_pred - x1_pred\n            axesArry[axidx].add_patch(patches.Rectangle((x1_pred, y1_pred), widthReg_pred, heightReg_pred, linewidth=1, edgecolor='b', \n                                                        facecolor='none'))\n            axesArry[axidx].set_title('IOU : '+str(topIOUsAry[axidx]))\n    # Show subplots\n    pyplot.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:10.079017Z","iopub.execute_input":"2024-12-28T12:26:10.07925Z","iopub.status.idle":"2024-12-28T12:26:10.096765Z","shell.execute_reply.started":"2024-12-28T12:26:10.079225Z","shell.execute_reply":"2024-12-28T12:26:10.096007Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Create Generator instances for Train and Validation datasets**","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"trainUNet = UNetTrainGenerator(train_imageIdpaths, train_mergedata)\nvalidateUNet = UNetTrainGenerator(validate_imageIdpaths, validate_mergedata)","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:10.097632Z","iopub.execute_input":"2024-12-28T12:26:10.097986Z","iopub.status.idle":"2024-12-28T12:26:10.11588Z","shell.execute_reply.started":"2024-12-28T12:26:10.097896Z","shell.execute_reply":"2024-12-28T12:26:10.115138Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training model\nResNet50","metadata":{}},{"cell_type":"code","source":"model=ResNet_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:10.1167Z","iopub.execute_input":"2024-12-28T12:26:10.116891Z","iopub.status.idle":"2024-12-28T12:26:12.731165Z","shell.execute_reply.started":"2024-12-28T12:26:10.116872Z","shell.execute_reply":"2024-12-28T12:26:12.7303Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.callbacks import LearningRateScheduler, EarlyStopping, ModelCheckpoint\n\n# Cosine annealing learning rate schedule function\ndef cosine_annealing(epoch, lr_max=0.001, lr_min=1e-6, total_epochs=50):\n    \"\"\"\n    Cosine annealing learning rate schedule.\n    Args:\n    - epoch: Current epoch number.\n    - lr_max: Maximum learning rate (initial).\n    - lr_min: Minimum learning rate (final).\n    - total_epochs: Total number of epochs.\n    \n    Returns:\n    - New learning rate for the current epoch.\n    \"\"\"\n    cosine_decay = 0.5 * (1 + np.cos(np.pi * epoch / total_epochs))\n    lr = lr_min + (lr_max - lr_min) * cosine_decay\n    return lr\n\n# Define combined loss function (Binary Crossentropy + IoU loss)\ndef combined_loss(y_true, y_pred):\n    bce_loss = tf.keras.losses.BinaryCrossentropy()(y_true, y_pred)\n    iou_loss_value = iou_loss(y_true, y_pred)\n    return bce_loss+0.5*iou_loss_value\n\n# Define the Adam optimizer\nadamopt_r = tf.keras.optimizers.Adam(learning_rate=0.001)  # Initial learning rate\n\n# Compile the model with the combined loss function\nmodel.compile(optimizer=adamopt_r, loss=combined_loss, metrics=[mean_iou])\n\n# Create the learning rate scheduler\nlearning_rate_scheduler = LearningRateScheduler(lambda epoch: cosine_annealing(epoch, lr_max=0.001, lr_min=1e-6, total_epochs=30))\n\n# Define the EarlyStopping callback\nstop = EarlyStopping(monitor='loss', mode='min', patience=3)\n\n# Define ModelCheckpoint to save the best model based on validation loss\ncheckpoint = ModelCheckpoint(\"onlyres-{epoch:02d}-{val_mean_iou:.2f}.weights.h5\", \n                              monitor='loss', \n                              verbose=1, \n                              mode='min', \n                              save_best_only=True, \n                              save_weights_only=True)\n\n# Train the model with the callbacks\nhistory_ures = model.fit(\n    trainUNet,\n    epochs=50,\n    validation_data=validateUNet,\n    callbacks=[learning_rate_scheduler, checkpoint, stop],\n    shuffle=True,\n    verbose=1\n)\n\nmodel.save(\"final_res_model.h5\")\nnp.save(\"History_ResNet.npy\", history_ures.history, allow_pickle=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T12:26:12.732081Z","iopub.execute_input":"2024-12-28T12:26:12.732366Z","iopub.status.idle":"2024-12-28T14:11:09.488817Z","shell.execute_reply.started":"2024-12-28T12:26:12.732343Z","shell.execute_reply":"2024-12-28T14:11:09.487881Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.load_weights('onlyres-34-0.61.weights.h5')","metadata":{"execution":{"iopub.status.busy":"2024-12-28T14:15:18.977415Z","iopub.execute_input":"2024-12-28T14:15:18.97773Z","iopub.status.idle":"2024-12-28T14:15:19.870381Z","shell.execute_reply.started":"2024-12-28T14:15:18.977697Z","shell.execute_reply":"2024-12-28T14:15:19.869654Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test, y_predicted = predictBatches(test_mergedata, test_imageIdpaths, model)","metadata":{"execution":{"iopub.status.busy":"2024-12-28T14:15:23.000161Z","iopub.execute_input":"2024-12-28T14:15:23.000462Z","iopub.status.idle":"2024-12-28T14:18:34.128078Z","shell.execute_reply.started":"2024-12-28T14:15:23.000438Z","shell.execute_reply":"2024-12-28T14:18:34.1273Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"report_file_res = pd.read_csv('./submission_res.csv')\nshowConfusionMatrix(report_file_res)","metadata":{"execution":{"iopub.status.busy":"2024-12-28T14:18:44.140939Z","iopub.execute_input":"2024-12-28T14:18:44.141228Z","iopub.status.idle":"2024-12-28T14:18:44.162529Z","shell.execute_reply.started":"2024-12-28T14:18:44.141207Z","shell.execute_reply":"2024-12-28T14:18:44.16174Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Analysis from the Confusion Matrix:\n\n* True Positives(TP) - 148 cases which is a point of concern. Thus, our model failed in successfully predicting the cases of Pneumonia pr cases with Lung Opacity\n* True Negatives(TN) - 4214 patients who have actually 'No Lung Opacity'/'Normal' are correctly predicted as 'Normal' or cases with no Pneumonia\n* False Positives(FP) - 0 cases who have actually 'No Lung Opacity'/'Normal' incorrectly predicted as having 'Lung Opacity'. This means there are no model mistakes in identifying 'Normal' as 'Lung Opacity'. Type-I error is zero. This is a big positive point for a medical test\n* (FN) are - 559 cases where the patients who have actually 'Lung Opacity' or 'Pneumonia' are incorrectly predicted as'Normal'. These are model mistakes.Thus, we need to seek some other architecture.\ncture.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# Load the CSV to inspect its structure\ndf = pd.read_csv('./submission_res.csv')\n\n# Display the column names\nprint(\"Column names:\", df.columns)\n\n# Display the first few rows to check the data\nprint(df.head())\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T14:18:52.339617Z","iopub.execute_input":"2024-12-28T14:18:52.339907Z","iopub.status.idle":"2024-12-28T14:18:52.355788Z","shell.execute_reply.started":"2024-12-28T14:18:52.339884Z","shell.execute_reply":"2024-12-28T14:18:52.355032Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Load the CSV file\ndf = pd.read_csv('./submission_res.csv')\n\n# Extract the actual labels (Target) and predicted labels (predTarget)\ny_true = df['Target']  # Actual labels\ny_pred = df['predTarget']  # Predicted labels\n\n# Calculate confusion matrix\ncm = confusion_matrix(y_true, y_pred)\n\n# Plotting the confusion matrix using seaborn's heatmap\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, fmt='g', cmap='Blues', xticklabels=['Predicted Normal', 'Predicted Lung Opacity'], \n            yticklabels=['Actual Normal', 'Actual Lung Opacity'])\n\n# Adding labels and title to the plot\nplt.title(\"Confusion Matrix\")\nplt.xlabel(\"Predicted Labels\")\nplt.ylabel(\"True Labels\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T14:18:55.396556Z","iopub.execute_input":"2024-12-28T14:18:55.396864Z","iopub.status.idle":"2024-12-28T14:18:55.647813Z","shell.execute_reply.started":"2024-12-28T14:18:55.396839Z","shell.execute_reply":"2024-12-28T14:18:55.647018Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"top = 10\nvisualizePredictions(report_file_res, top)","metadata":{"execution":{"iopub.status.busy":"2024-12-28T14:18:59.530722Z","iopub.execute_input":"2024-12-28T14:18:59.531071Z","iopub.status.idle":"2024-12-28T14:19:02.360957Z","shell.execute_reply.started":"2024-12-28T14:18:59.531042Z","shell.execute_reply":"2024-12-28T14:19:02.35995Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom matplotlib import pyplot as plt\n\ndef plotHistory(history_file):\n    \"\"\"\n    Plots training and validation metrics from history.\n    Arguments:\n        history_file: File path to the saved training history (.npy).\n    \"\"\"\n    # Load the saved history\n    history = np.load(history_file, allow_pickle=True).item()\n    history_df = pd.DataFrame(history)\n    \n    # Plot loss\n    plt.figure(figsize=(10, 6))\n    plt.plot(history_df['loss'], label='Train Loss')\n    plt.plot(history_df['val_loss'], label='Validation Loss')\n    plt.title('Model Loss')\n    plt.xlabel('Epochs')\n    plt.ylabel('Loss')\n    plt.legend(loc='best')\n    plt.grid(True)\n    plt.show()\n    \n    # Plot mean IoU\n    plt.figure(figsize=(10, 6))\n    plt.plot(history_df['mean_iou'], label='Train Mean IoU')\n    plt.plot(history_df['val_mean_iou'], label='Validation Mean IoU')\n    plt.title('Model Mean IoU')\n    plt.xlabel('Epochs')\n    plt.ylabel('IoU')\n    plt.legend(loc='best')\n    plt.grid(True)\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T14:19:06.801542Z","iopub.execute_input":"2024-12-28T14:19:06.801839Z","iopub.status.idle":"2024-12-28T14:19:06.808063Z","shell.execute_reply.started":"2024-12-28T14:19:06.801816Z","shell.execute_reply":"2024-12-28T14:19:06.807201Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot history for UNet ResNet\nplotHistory(\"History_ResNet.npy\")\n\n","metadata":{"execution":{"iopub.status.busy":"2024-12-28T14:19:11.970552Z","iopub.execute_input":"2024-12-28T14:19:11.970923Z","iopub.status.idle":"2024-12-28T14:19:12.536339Z","shell.execute_reply.started":"2024-12-28T14:19:11.97088Z","shell.execute_reply":"2024-12-28T14:19:12.535285Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}