{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":23812,"sourceType":"datasetVersion","datasetId":17810}],"dockerImageVersionId":30498,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Pneumonia Detection**\n","metadata":{}},{"cell_type":"markdown","source":"**Libraries**","metadata":{}},{"cell_type":"code","source":"#Importing libraries\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt \nimport matplotlib.image as mpimg\nimport seaborn as sns\nimport sklearn\nimport os\nimport shutil\nimport cv2\nimport random\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, classification_report, accuracy_score, precision_score, recall_score, f1_score\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.preprocessing import image_dataset_from_directory","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:11:10.781899Z","iopub.execute_input":"2024-03-11T05:11:10.782884Z","iopub.status.idle":"2024-03-11T05:11:10.789256Z","shell.execute_reply.started":"2024-03-11T05:11:10.782841Z","shell.execute_reply":"2024-03-11T05:11:10.788263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        os.path.join(dirname, filename)\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-11T05:11:10.790844Z","iopub.execute_input":"2024-03-11T05:11:10.791094Z","iopub.status.idle":"2024-03-11T05:11:19.849331Z","shell.execute_reply.started":"2024-03-11T05:11:10.791072Z","shell.execute_reply":"2024-03-11T05:11:19.848246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data Cleaning and Preprocessing**","metadata":{}},{"cell_type":"markdown","source":"**Creating a new dataset by joining all the data of the 3 groups together and then dividing to train and test set using the 80:20 ratio**","metadata":{}},{"cell_type":"code","source":"#Function to parse through the folders and extract the images from their folders\nlabels = ['PNEUMONIA', 'NORMAL']\nimg_size = 224\ndef get_training_data(data_dir):\n    data = [] \n    for label in labels: \n        path = os.path.join(data_dir, label)\n        class_num = labels.index(label)\n        for img in os.listdir(path):\n            try:\n                img_arr = cv2.imread(os.path.join(path, img), cv2.IMREAD_COLOR)\n                resized_arr = cv2.resize(img_arr, (img_size, img_size)) # Reshaping images to preferred size\n                resized_arr_rgb_format = cv2.cvtColor(resized_arr, cv2.COLOR_BGR2RGB)\n                data.append([resized_arr_rgb_format, class_num])\n            except Exception as e:\n                print(e)\n    return np.array(data)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:11:19.851350Z","iopub.execute_input":"2024-03-11T05:11:19.851784Z","iopub.status.idle":"2024-03-11T05:11:19.861184Z","shell.execute_reply.started":"2024-03-11T05:11:19.851742Z","shell.execute_reply":"2024-03-11T05:11:19.860211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Getting the image datasets from paths of the training, test and validation dataset.\ntrain = get_training_data('/kaggle/input/chest-xray-pneumonia/chest_xray/train')\ntest = get_training_data('/kaggle/input/chest-xray-pneumonia/chest_xray/test')\nval = get_training_data('/kaggle/input/chest-xray-pneumonia/chest_xray/val')","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:44:25.953241Z","iopub.status.idle":"2024-03-11T05:44:25.953642Z","shell.execute_reply.started":"2024-03-11T05:44:25.953402Z","shell.execute_reply":"2024-03-11T05:44:25.953418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Joining the datasets to enable splitting the dataset using the 80:20 ratio\ndataset = np.concatenate((train, val, test), axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:33.249423Z","iopub.execute_input":"2024-03-11T05:13:33.249769Z","iopub.status.idle":"2024-03-11T05:13:33.254420Z","shell.execute_reply.started":"2024-03-11T05:13:33.249741Z","shell.execute_reply":"2024-03-11T05:13:33.253514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(dataset)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:33.256021Z","iopub.execute_input":"2024-03-11T05:13:33.256337Z","iopub.status.idle":"2024-03-11T05:13:33.277528Z","shell.execute_reply.started":"2024-03-11T05:13:33.256304Z","shell.execute_reply":"2024-03-11T05:13:33.276521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(dataset.shape)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:33.278808Z","iopub.execute_input":"2024-03-11T05:13:33.279092Z","iopub.status.idle":"2024-03-11T05:13:33.289224Z","shell.execute_reply.started":"2024-03-11T05:13:33.279067Z","shell.execute_reply":"2024-03-11T05:13:33.288197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Visualization & Preprocessing","metadata":{}},{"cell_type":"code","source":"#Function to display images in the folder in the new dataset\ndef plot_images_from_folder(dataset):\n    # Generate 8 random indices within the range of your dataset\n    random_indices = random.sample(range(0, len(dataset)), 8)\n    plt.figure(figsize=[14, 24])\n    for i in range(8):\n        plt.subplot(6, 4, i+1)\n        plt.imshow(dataset[random_indices[i]][0], cmap='gray')\n        plt.axis('off')\n        plt.title(dataset[random_indices[i]][1])  ","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:33.290612Z","iopub.execute_input":"2024-03-11T05:13:33.290888Z","iopub.status.idle":"2024-03-11T05:13:33.300005Z","shell.execute_reply.started":"2024-03-11T05:13:33.290864Z","shell.execute_reply":"2024-03-11T05:13:33.299115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Show random samples of images from the dataset\nplot_images_from_folder(dataset)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:33.301289Z","iopub.execute_input":"2024-03-11T05:13:33.305054Z","iopub.status.idle":"2024-03-11T05:13:34.328702Z","shell.execute_reply.started":"2024-03-11T05:13:33.305021Z","shell.execute_reply":"2024-03-11T05:13:34.327742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Split the dataset into the training and test dataset\ninitial_train_df, test_df = train_test_split(dataset, test_size = 0.20, random_state = 30)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:34.329898Z","iopub.execute_input":"2024-03-11T05:13:34.330186Z","iopub.status.idle":"2024-03-11T05:13:34.336328Z","shell.execute_reply.started":"2024-03-11T05:13:34.330161Z","shell.execute_reply":"2024-03-11T05:13:34.335439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Split the training dataset into the training and val dataset\ntrain_df, val_df = train_test_split(initial_train_df, test_size = 0.20, random_state = 30)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:34.340534Z","iopub.execute_input":"2024-03-11T05:13:34.340880Z","iopub.status.idle":"2024-03-11T05:13:34.349937Z","shell.execute_reply.started":"2024-03-11T05:13:34.340846Z","shell.execute_reply":"2024-03-11T05:13:34.348994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Function to display the count for the label for each class\ndef count_labels(labels):\n    extracted_labels = [data[1] for data in labels]\n    print(\"Number of labels\", len(extracted_labels))\n    # Count the occurrences of each class label\n    class_counts = np.bincount(extracted_labels)\n\n    # Print the counts\n    count_0 = class_counts[0]\n    count_1 = class_counts[1]\n    print(\"Count of 0:\", count_0)\n    print(\"Count of 1:\", count_1)\n    sns.set_style('darkgrid')\n    sns.countplot(x=extracted_labels).set(title=\"training data\")","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:34.351396Z","iopub.execute_input":"2024-03-11T05:13:34.351775Z","iopub.status.idle":"2024-03-11T05:13:34.362234Z","shell.execute_reply.started":"2024-03-11T05:13:34.351749Z","shell.execute_reply":"2024-03-11T05:13:34.361332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Class distribution for training dataset\ncount_labels(train_df)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:34.363597Z","iopub.execute_input":"2024-03-11T05:13:34.363892Z","iopub.status.idle":"2024-03-11T05:13:34.628751Z","shell.execute_reply.started":"2024-03-11T05:13:34.363869Z","shell.execute_reply":"2024-03-11T05:13:34.627725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Class distribution for validation dataset\ncount_labels(val_df)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:34.629907Z","iopub.execute_input":"2024-03-11T05:13:34.630176Z","iopub.status.idle":"2024-03-11T05:13:34.943571Z","shell.execute_reply.started":"2024-03-11T05:13:34.630153Z","shell.execute_reply":"2024-03-11T05:13:34.942531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Class distribution for test dataset\ncount_labels(test_df)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:34.945140Z","iopub.execute_input":"2024-03-11T05:13:34.945840Z","iopub.status.idle":"2024-03-11T05:13:35.284506Z","shell.execute_reply.started":"2024-03-11T05:13:34.945800Z","shell.execute_reply":"2024-03-11T05:13:35.283561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Shape of training dataset\ntrain_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:35.285580Z","iopub.execute_input":"2024-03-11T05:13:35.285849Z","iopub.status.idle":"2024-03-11T05:13:35.291367Z","shell.execute_reply.started":"2024-03-11T05:13:35.285816Z","shell.execute_reply":"2024-03-11T05:13:35.290514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Shape of val dataset\nval_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:35.292808Z","iopub.execute_input":"2024-03-11T05:13:35.293197Z","iopub.status.idle":"2024-03-11T05:13:35.305321Z","shell.execute_reply.started":"2024-03-11T05:13:35.293161Z","shell.execute_reply":"2024-03-11T05:13:35.304314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Shape of test dataset\ntest_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:35.306651Z","iopub.execute_input":"2024-03-11T05:13:35.307026Z","iopub.status.idle":"2024-03-11T05:13:35.319212Z","shell.execute_reply.started":"2024-03-11T05:13:35.306999Z","shell.execute_reply":"2024-03-11T05:13:35.318185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Seperate the images and labels\nx_train = []\ny_train = []\n\nx_val = []\ny_val = []\n\nx_test = []\ny_test = []\n\nfor feature, label in train_df:\n    x_train.append(feature)\n    y_train.append(label)\n\nfor feature, label in test_df:\n    x_test.append(feature)\n    y_test.append(label)\n    \nfor feature, label in val_df:\n    x_val.append(feature)\n    y_val.append(label)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:35.320537Z","iopub.execute_input":"2024-03-11T05:13:35.320834Z","iopub.status.idle":"2024-03-11T05:13:35.337949Z","shell.execute_reply.started":"2024-03-11T05:13:35.320808Z","shell.execute_reply":"2024-03-11T05:13:35.337132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Normalize the data\nx_train = np.array(x_train) / 255\nx_val = np.array(x_val) / 255\nx_test = np.array(x_test) / 255","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:35.339229Z","iopub.execute_input":"2024-03-11T05:13:35.339672Z","iopub.status.idle":"2024-03-11T05:13:37.946709Z","shell.execute_reply.started":"2024-03-11T05:13:35.339620Z","shell.execute_reply":"2024-03-11T05:13:37.945525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# reshape data for deep learning \nx_train = x_train.reshape(-1, img_size, img_size, 3)\ny_train = np.array(y_train)\n\nx_val = x_val.reshape(-1, img_size, img_size, 3)\ny_val = np.array(y_val)\n\nx_test = x_test.reshape(-1, img_size, img_size, 3)\ny_test = np.array(y_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:37.948041Z","iopub.execute_input":"2024-03-11T05:13:37.948402Z","iopub.status.idle":"2024-03-11T05:13:37.955677Z","shell.execute_reply.started":"2024-03-11T05:13:37.948371Z","shell.execute_reply":"2024-03-11T05:13:37.954550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# With data augmentation to prevent overfitting and handling the imbalance in dataset\n\ndatagen = ImageDataGenerator(\n        featurewise_center=False,  # set input mean to 0 over the dataset\n        samplewise_center=False,  # set each sample mean to 0\n        featurewise_std_normalization=False,  # divide inputs by std of the dataset\n        samplewise_std_normalization=False,  # divide each input by its std\n        zca_whitening=False,  # apply ZCA whitening\n        rotation_range = 30,  # randomly rotate images in the range (degrees, 0 to 180)\n        zoom_range = 0.2, # Randomly zoom image \n        width_shift_range=0.1,  # randomly shift images horizontally (fraction of total width)\n        height_shift_range=0.1,  # randomly shift images vertically (fraction of total height)\n        horizontal_flip = True,  # randomly flip images\n        vertical_flip=False)  # randomly flip images\n\n\ndatagen.fit(x_train)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:37.957066Z","iopub.execute_input":"2024-03-11T05:13:37.957439Z","iopub.status.idle":"2024-03-11T05:13:39.623944Z","shell.execute_reply.started":"2024-03-11T05:13:37.957410Z","shell.execute_reply":"2024-03-11T05:13:39.622991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.utils.class_weight import compute_class_weight\n# class_weights = compute_class_weight(class_weight = \"balanced\", classes= np.unique(train_generator.classes), y= train_generator.classes)\n# class_weights = dict(zip(np.unique(train_generator.classes), class_weights))","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:39.625406Z","iopub.execute_input":"2024-03-11T05:13:39.626038Z","iopub.status.idle":"2024-03-11T05:13:39.630623Z","shell.execute_reply.started":"2024-03-11T05:13:39.626000Z","shell.execute_reply":"2024-03-11T05:13:39.629703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class_weights","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:39.631927Z","iopub.execute_input":"2024-03-11T05:13:39.632204Z","iopub.status.idle":"2024-03-11T05:13:39.641082Z","shell.execute_reply.started":"2024-03-11T05:13:39.632180Z","shell.execute_reply":"2024-03-11T05:13:39.640232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the early stopping and learning rate reduction callback\nearly_stopping = tf.keras.callbacks.EarlyStopping(monitor='val_loss',patience=5, restore_best_weights=True)\nlearning_rate_reduction = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', patience = 3)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:39.642238Z","iopub.execute_input":"2024-03-11T05:13:39.642585Z","iopub.status.idle":"2024-03-11T05:13:39.653126Z","shell.execute_reply.started":"2024-03-11T05:13:39.642557Z","shell.execute_reply":"2024-03-11T05:13:39.652191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Modeling** ","metadata":{}},{"cell_type":"markdown","source":"# VGG 16 Model","metadata":{}},{"cell_type":"code","source":"#Loading the model\nfrom tensorflow.keras.applications.vgg16 import VGG16\n\nvgg16_base_model = VGG16(\n    include_top=False,\n    weights=\"imagenet\",\n    input_shape=(224, 224, 3),\n)\n\n#Making sure the layers of the VGG16 model are not retrained \nfor layer in vgg16_base_model.layers:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:39.654414Z","iopub.execute_input":"2024-03-11T05:13:39.654753Z","iopub.status.idle":"2024-03-11T05:13:46.102635Z","shell.execute_reply.started":"2024-03-11T05:13:39.654728Z","shell.execute_reply":"2024-03-11T05:13:46.101605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vgg16_model = tf.keras.models.Sequential()\nvgg16_model.add(vgg16_base_model)\nvgg16_model.add(tf.keras.layers.Flatten())\nvgg16_model.add(tf.keras.layers.BatchNormalization())\nvgg16_model.add(tf.keras.layers.Dense(128,activation='relu'))\nvgg16_model.add(tf.keras.layers.Dropout(0.5))\nvgg16_model.add(tf.keras.layers.Dense(1,activation='sigmoid'))\nvgg16_model.summary()\n\nvgg16_model.compile(\n    loss='binary_crossentropy',\n    optimizer=tf.keras.optimizers.Adam(),\n    metrics=['accuracy']\n)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:46.104024Z","iopub.execute_input":"2024-03-11T05:13:46.104350Z","iopub.status.idle":"2024-03-11T05:13:46.273532Z","shell.execute_reply.started":"2024-03-11T05:13:46.104323Z","shell.execute_reply":"2024-03-11T05:13:46.272415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the vgg16 model with early stopping\nvgg16_model_history = vgg16_model.fit(\n    datagen.flow(x_train,y_train, batch_size = 32),\n    epochs=30,\n    validation_data=datagen.flow(x_val, y_val),\n    callbacks=[early_stopping, learning_rate_reduction])","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:13:46.274943Z","iopub.execute_input":"2024-03-11T05:13:46.275241Z","iopub.status.idle":"2024-03-11T05:25:11.341531Z","shell.execute_reply.started":"2024-03-11T05:13:46.275216Z","shell.execute_reply":"2024-03-11T05:25:11.340402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Plotting the VGG16 model results\n\n#Getting the accuracy\nacc = vgg16_model_history.history['accuracy']\nval_acc = vgg16_model_history.history['val_accuracy']\n\n#Getting the losses\nloss = vgg16_model_history.history['loss']\nval_loss = vgg16_model_history.history['val_loss']\n\n#No of epochs it trained\nepochs_range = vgg16_model_history.epoch\n\n#Plotting Training and Validation accuracy\nplt.figure(figsize=(16, 6))\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label='Training Accuracy')\nplt.plot(epochs_range, val_acc, label='Validation Accuracy')\nplt.legend(loc='lower right')\nplt.title('Training and Validation Accuracy')\n\n#Plotting Training and Validation Loss\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Training Loss')\nplt.plot(epochs_range, val_loss, label='Validation Loss')\nplt.legend(loc='upper right')\nplt.title('Training and Validation Loss')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:11.351299Z","iopub.execute_input":"2024-03-11T05:25:11.351652Z","iopub.status.idle":"2024-03-11T05:25:12.117526Z","shell.execute_reply.started":"2024-03-11T05:25:11.351623Z","shell.execute_reply":"2024-03-11T05:25:12.116426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Vgg16 Performance Evaluation","metadata":{}},{"cell_type":"code","source":"evaluation_result=vgg16_model.evaluate(x_test,y_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:12.118777Z","iopub.execute_input":"2024-03-11T05:25:12.119081Z","iopub.status.idle":"2024-03-11T05:25:22.229012Z","shell.execute_reply.started":"2024-03-11T05:25:12.119054Z","shell.execute_reply":"2024-03-11T05:25:22.227842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Loss of the model is - \" , evaluation_result[0])\nprint(\"Accuracy of the model is - \" , evaluation_result[1]*100 , \"%\")","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:22.230599Z","iopub.execute_input":"2024-03-11T05:25:22.230919Z","iopub.status.idle":"2024-03-11T05:25:22.236544Z","shell.execute_reply.started":"2024-03-11T05:25:22.230892Z","shell.execute_reply":"2024-03-11T05:25:22.235525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vgg16_predictions = vgg16_model.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:22.237927Z","iopub.execute_input":"2024-03-11T05:25:22.238295Z","iopub.status.idle":"2024-03-11T05:25:29.426754Z","shell.execute_reply.started":"2024-03-11T05:25:22.238245Z","shell.execute_reply":"2024-03-11T05:25:29.425834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = (vgg16_predictions> 0.5).astype(\"int32\").flatten()","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:29.428284Z","iopub.execute_input":"2024-03-11T05:25:29.428674Z","iopub.status.idle":"2024-03-11T05:25:29.434023Z","shell.execute_reply.started":"2024-03-11T05:25:29.428638Z","shell.execute_reply":"2024-03-11T05:25:29.432987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:29.435375Z","iopub.execute_input":"2024-03-11T05:25:29.435690Z","iopub.status.idle":"2024-03-11T05:25:29.448503Z","shell.execute_reply.started":"2024-03-11T05:25:29.435664Z","shell.execute_reply":"2024-03-11T05:25:29.447500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:29.449857Z","iopub.execute_input":"2024-03-11T05:25:29.450175Z","iopub.status.idle":"2024-03-11T05:25:29.459418Z","shell.execute_reply.started":"2024-03-11T05:25:29.450147Z","shell.execute_reply":"2024-03-11T05:25:29.458339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Confusion matrix\ncm = confusion_matrix(y_test, y_pred)\nprint(cm)\n\nimport seaborn as sns\n\n#Setting the labels\nlabels = ['pneumonia', 'normal']\n\n#Plot the Confusion matrix graph\nfig= plt.figure(figsize=(8, 5))\nax = plt.subplot()\nsns.heatmap(cm, annot=True, ax = ax, fmt='g')\nax.set_xlabel('Predicted Labels', fontsize=10)\nax.xaxis.set_label_position('bottom')\nplt.xticks(rotation=90)\nax.xaxis.set_ticklabels(labels, fontsize = 5)\nax.xaxis.tick_bottom()\n\nax.set_ylabel('True Labels', fontsize=10)\nax.yaxis.set_ticklabels(labels, fontsize = 10)\nplt.yticks(rotation=0)\n\nplt.title('Confusion Matrix', fontsize=15)\n\nplt.savefig('VGG16_ConMat24.png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:29.460931Z","iopub.execute_input":"2024-03-11T05:25:29.461257Z","iopub.status.idle":"2024-03-11T05:25:29.909700Z","shell.execute_reply.started":"2024-03-11T05:25:29.461221Z","shell.execute_reply":"2024-03-11T05:25:29.908800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Classification report\nprint(classification_report(y_test, y_pred, target_names = ['Pneumonia (0)','Normal (1)']))","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:29.910839Z","iopub.execute_input":"2024-03-11T05:25:29.911124Z","iopub.status.idle":"2024-03-11T05:25:29.927436Z","shell.execute_reply.started":"2024-03-11T05:25:29.911099Z","shell.execute_reply":"2024-03-11T05:25:29.926585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Accuracy\naccuracy = accuracy_score(y_test, y_pred)\nprint('Accuracy: %f' % (accuracy*100))","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:29.928551Z","iopub.execute_input":"2024-03-11T05:25:29.928819Z","iopub.status.idle":"2024-03-11T05:25:29.935571Z","shell.execute_reply.started":"2024-03-11T05:25:29.928795Z","shell.execute_reply":"2024-03-11T05:25:29.934596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Precision\nprecision = precision_score(y_test, y_pred)\nprint('Precision: %f' % (precision*100))","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:29.936766Z","iopub.execute_input":"2024-03-11T05:25:29.937668Z","iopub.status.idle":"2024-03-11T05:25:29.958198Z","shell.execute_reply.started":"2024-03-11T05:25:29.937641Z","shell.execute_reply":"2024-03-11T05:25:29.957192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Recall\nrecall = recall_score(y_test, y_pred, pos_label=1)\nprint('Recall: %f' % (recall*100))","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:29.959391Z","iopub.execute_input":"2024-03-11T05:25:29.959712Z","iopub.status.idle":"2024-03-11T05:25:29.970035Z","shell.execute_reply.started":"2024-03-11T05:25:29.959686Z","shell.execute_reply":"2024-03-11T05:25:29.969004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#F1-score\nF1_score = f1_score(y_test, y_pred)\nprint('F1_score: %f' % (F1_score*100))","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:29.971046Z","iopub.execute_input":"2024-03-11T05:25:29.971366Z","iopub.status.idle":"2024-03-11T05:25:29.979742Z","shell.execute_reply.started":"2024-03-11T05:25:29.971341Z","shell.execute_reply":"2024-03-11T05:25:29.978704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Specificity \nspecificity = recall_score(y_test, y_pred, pos_label=0)\nprint('Specificity: %f' % (specificity*100))","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:29.981167Z","iopub.execute_input":"2024-03-11T05:25:29.981550Z","iopub.status.idle":"2024-03-11T05:25:29.991583Z","shell.execute_reply.started":"2024-03-11T05:25:29.981521Z","shell.execute_reply":"2024-03-11T05:25:29.990511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Save the model\nvgg16_model.save(\"VGG16_pneumonia_model.h5\")","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:29.992994Z","iopub.execute_input":"2024-03-11T05:25:29.993301Z","iopub.status.idle":"2024-03-11T05:25:30.167127Z","shell.execute_reply.started":"2024-03-11T05:25:29.993276Z","shell.execute_reply":"2024-03-11T05:25:30.166253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Convert the model to a tensorflow lite model\nfrom tensorflow import lite\n\nconverter = lite.TFLiteConverter.from_keras_model(vgg16_model)\ntfmodel = converter.convert()\nopen(\"vgg16_final_model.tflite\", \"wb\").write(tfmodel)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:30.168192Z","iopub.execute_input":"2024-03-11T05:25:30.168448Z","iopub.status.idle":"2024-03-11T05:25:38.516103Z","shell.execute_reply.started":"2024-03-11T05:25:30.168426Z","shell.execute_reply":"2024-03-11T05:25:38.515147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# RESNET50 Modelling","metadata":{}},{"cell_type":"code","source":"#Loading the model\nfrom tensorflow.keras.applications.resnet50 import ResNet50\n\nresnet50_base_model = ResNet50(\n    include_top=False,\n    weights=\"imagenet\",\n    input_shape=(224, 224, 3),\n)\n\n#Making sure the layers of the VGG16 model are not retrained \nfor layer in resnet50_base_model.layers:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:38.518402Z","iopub.execute_input":"2024-03-11T05:25:38.518842Z","iopub.status.idle":"2024-03-11T05:25:40.894909Z","shell.execute_reply.started":"2024-03-11T05:25:38.518801Z","shell.execute_reply":"2024-03-11T05:25:40.894044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resnet50_model = tf.keras.models.Sequential()\nresnet50_model.add(resnet50_base_model)\nresnet50_model.add(tf.keras.layers.Flatten())\nresnet50_model.add(tf.keras.layers.BatchNormalization())\nresnet50_model.add(tf.keras.layers.Dense(128,activation='relu'))\nresnet50_model.add(tf.keras.layers.Dropout(0.5))\nresnet50_model.add(tf.keras.layers.Dense(1,activation='sigmoid'))\nresnet50_model.summary()\n\nresnet50_model.compile(\n    loss='binary_crossentropy',\n    optimizer=tf.keras.optimizers.Adam(),\n    metrics=['accuracy']\n)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:40.896137Z","iopub.execute_input":"2024-03-11T05:25:40.896421Z","iopub.status.idle":"2024-03-11T05:25:41.497445Z","shell.execute_reply.started":"2024-03-11T05:25:40.896395Z","shell.execute_reply":"2024-03-11T05:25:41.496498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the resnet50 model with early stopping\nresnet50_model_history = resnet50_model.fit(\n    datagen.flow(x_train,y_train, batch_size = 32),\n    epochs=30,\n    validation_data=datagen.flow(x_val, y_val),\n    callbacks=[early_stopping, learning_rate_reduction])","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:25:41.498769Z","iopub.execute_input":"2024-03-11T05:25:41.499085Z","iopub.status.idle":"2024-03-11T05:44:08.818548Z","shell.execute_reply.started":"2024-03-11T05:25:41.499057Z","shell.execute_reply":"2024-03-11T05:44:08.817511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Plotting the resnet50 model results\n\n#Getting the accuracy\nacc = resnet50_model_history.history['accuracy']\nval_acc = resnet50_model_history.history['val_accuracy']\n\n#Getting the losses\nloss = resnet50_model_history.history['loss']\nval_loss = resnet50_model_history.history['val_loss']\n\n#No of epochs it trained\nepochs_range = resnet50_model_history.epoch\n\n#Plotting Training and Validation accuracy\nplt.figure(figsize=(16, 6))\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label='Training Accuracy')\nplt.plot(epochs_range, val_acc, label='Validation Accuracy')\nplt.legend(loc='lower right')\nplt.title('Training and Validation Accuracy')\n\n#Plotting Training and Validation Loss\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Training Loss')\nplt.plot(epochs_range, val_loss, label='Validation Loss')\nplt.legend(loc='upper right')\nplt.title('Training and Validation Loss')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:44:08.820115Z","iopub.execute_input":"2024-03-11T05:44:08.820506Z","iopub.status.idle":"2024-03-11T05:44:09.565542Z","shell.execute_reply.started":"2024-03-11T05:44:08.820453Z","shell.execute_reply":"2024-03-11T05:44:09.564586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# RESNET50 Performance Evaluation","metadata":{}},{"cell_type":"code","source":"evaluation_result=resnet50_model.evaluate(x_test,np.array(y_test))","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:44:09.566644Z","iopub.execute_input":"2024-03-11T05:44:09.566936Z","iopub.status.idle":"2024-03-11T05:44:16.364330Z","shell.execute_reply.started":"2024-03-11T05:44:09.566912Z","shell.execute_reply":"2024-03-11T05:44:16.363225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Loss of the model is - \" , evaluation_result[0])\nprint(\"Accuracy of the model is - \" , evaluation_result[1]*100 , \"%\")","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:44:16.365850Z","iopub.execute_input":"2024-03-11T05:44:16.366191Z","iopub.status.idle":"2024-03-11T05:44:16.372403Z","shell.execute_reply.started":"2024-03-11T05:44:16.366161Z","shell.execute_reply":"2024-03-11T05:44:16.371019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"resnet50_predictions = resnet50_model.predict(x_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:44:16.373866Z","iopub.execute_input":"2024-03-11T05:44:16.374260Z","iopub.status.idle":"2024-03-11T05:44:22.730593Z","shell.execute_reply.started":"2024-03-11T05:44:16.374223Z","shell.execute_reply":"2024-03-11T05:44:22.729665Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = (resnet50_predictions> 0.5).astype(\"int32\").flatten()","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:44:22.731892Z","iopub.execute_input":"2024-03-11T05:44:22.732170Z","iopub.status.idle":"2024-03-11T05:44:22.737057Z","shell.execute_reply.started":"2024-03-11T05:44:22.732145Z","shell.execute_reply":"2024-03-11T05:44:22.736027Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:44:22.738514Z","iopub.execute_input":"2024-03-11T05:44:22.738833Z","iopub.status.idle":"2024-03-11T05:44:22.750414Z","shell.execute_reply.started":"2024-03-11T05:44:22.738806Z","shell.execute_reply":"2024-03-11T05:44:22.749396Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:44:22.751727Z","iopub.execute_input":"2024-03-11T05:44:22.752727Z","iopub.status.idle":"2024-03-11T05:44:22.764177Z","shell.execute_reply.started":"2024-03-11T05:44:22.752697Z","shell.execute_reply":"2024-03-11T05:44:22.763046Z"},"jupyter":{"source_hidden":true,"outputs_hidden":true},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Confusion matrix\ncm = confusion_matrix(y_test, y_pred)\nprint(cm)\n\nimport seaborn as sns\n\n#Setting the labels\nlabels = ['pneumonia', 'normal']\n\n#Plot the Confusion matrix graph\nfig= plt.figure(figsize=(8, 5))\nax = plt.subplot()\nsns.heatmap(cm, annot=True, ax = ax, fmt='g')\nax.set_xlabel('Predicted Labels', fontsize=10)\nax.xaxis.set_label_position('bottom')\nplt.xticks(rotation=90)\nax.xaxis.set_ticklabels(labels, fontsize = 5)\nax.xaxis.tick_bottom()\n\nax.set_ylabel('True Labels', fontsize=10)\nax.yaxis.set_ticklabels(labels, fontsize = 10)\nplt.yticks(rotation=0)\n\nplt.title('Confusion Matrix', fontsize=15)\n\nplt.savefig('resnet50_ConMat24.png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:44:22.765789Z","iopub.execute_input":"2024-03-11T05:44:22.766630Z","iopub.status.idle":"2024-03-11T05:44:23.173407Z","shell.execute_reply.started":"2024-03-11T05:44:22.766594Z","shell.execute_reply":"2024-03-11T05:44:23.172505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Classification report\nprint(classification_report(y_test, y_pred, target_names = ['Pneumonia (0)','Normal (1)']))","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:44:23.174659Z","iopub.execute_input":"2024-03-11T05:44:23.174956Z","iopub.status.idle":"2024-03-11T05:44:23.193235Z","shell.execute_reply.started":"2024-03-11T05:44:23.174928Z","shell.execute_reply":"2024-03-11T05:44:23.192312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Accuracy\naccuracy = accuracy_score(y_test, y_pred)\nprint('Accuracy: %f' % (accuracy*100))","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:44:23.194498Z","iopub.execute_input":"2024-03-11T05:44:23.194808Z","iopub.status.idle":"2024-03-11T05:44:23.203431Z","shell.execute_reply.started":"2024-03-11T05:44:23.194780Z","shell.execute_reply":"2024-03-11T05:44:23.202342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Precision\nprecision = precision_score(y_test, y_pred)\nprint('Precision: %f' % (precision*100))","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:44:23.204967Z","iopub.execute_input":"2024-03-11T05:44:23.205228Z","iopub.status.idle":"2024-03-11T05:44:23.220316Z","shell.execute_reply.started":"2024-03-11T05:44:23.205204Z","shell.execute_reply":"2024-03-11T05:44:23.219054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Specificity \nspecificity = recall_score(y_test, y_pred, pos_label=0)\nprint('Specificity: %f' % (specificity*100))","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:44:23.244335Z","iopub.execute_input":"2024-03-11T05:44:23.244705Z","iopub.status.idle":"2024-03-11T05:44:23.253413Z","shell.execute_reply.started":"2024-03-11T05:44:23.244676Z","shell.execute_reply":"2024-03-11T05:44:23.252278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Save the model\nresnet50_model.save(\"resnet50_pneumonia_model.h5\")","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:44:23.254960Z","iopub.execute_input":"2024-03-11T05:44:23.255317Z","iopub.status.idle":"2024-03-11T05:44:23.886375Z","shell.execute_reply.started":"2024-03-11T05:44:23.255289Z","shell.execute_reply":"2024-03-11T05:44:23.885150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# user prediction","metadata":{}},{"cell_type":"code","source":"from keras.models import load_model\nfrom keras.preprocessing import image\nimport numpy as np\nfrom tensorflow.keras.preprocessing.image import load_img\n\nfrom tensorflow.keras.applications.vgg16 import preprocess_input\n\n# Load the saved model\nfrom tensorflow import keras\nmodel = keras.models.load_model('/kaggle/working/VGG16_pneumonia_model.h5')\nimport numpy as np\n\n# Assuming 'model' is already defined\n\nimage_path = r'/kaggle/input/chest-xray-pneumonia/chest_xray/test/NORMAL/IM-0035-0001.jpeg'\nimage = load_img(image_path, target_size=(224, 224))\nimg = np.array(image) / 255.0\nimg = img.reshape(1, 224, 224, 3)\n\nprediction = model.predict(img)\npredicted_class = int(round(prediction[0][0]))  # Assuming 1 is for \"Normal\" and 0 is for \"Pneumonia\"\n\nlabel_dict = {0: \"Pneumonia\", 1: \"Normal\"}\npredicted_label = label_dict[predicted_class]\n\nprint(f\"The predicted label is: {predicted_label}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-03-11T05:44:23.887948Z","iopub.execute_input":"2024-03-11T05:44:23.888290Z","iopub.status.idle":"2024-03-11T05:44:25.241664Z","shell.execute_reply.started":"2024-03-11T05:44:23.888260Z","shell.execute_reply":"2024-03-11T05:44:25.240435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}