{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":1199870,"sourceType":"datasetVersion","datasetId":615374}],"dockerImageVersionId":29928,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport cv2\nimport os\nfrom tqdm import tqdm\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import train_test_split\nfrom keras.utils.np_utils import to_categorical\nfrom keras.models import Model,Sequential, Input, load_model\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPool2D, BatchNormalization, AveragePooling2D, GlobalAveragePooling2D\nfrom keras.optimizers import Adam\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.callbacks import ModelCheckpoint, ReduceLROnPlateau\nfrom keras.applications import InceptionV3","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2024-05-14T03:10:25.772152Z","iopub.execute_input":"2024-05-14T03:10:25.772426Z","iopub.status.idle":"2024-05-14T03:10:30.821747Z","shell.execute_reply.started":"2024-05-14T03:10:25.772398Z","shell.execute_reply":"2024-05-14T03:10:30.820855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data","metadata":{}},{"cell_type":"code","source":"disease_types=['COVID', 'non-COVID']\ndata_dir = '../input/sarscov2-ctscan-dataset/'\ntrain_dir = os.path.join(data_dir)","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:10:30.823578Z","iopub.execute_input":"2024-05-14T03:10:30.823869Z","iopub.status.idle":"2024-05-14T03:10:30.827750Z","shell.execute_reply.started":"2024-05-14T03:10:30.823835Z","shell.execute_reply":"2024-05-14T03:10:30.826830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = []\nfor defects_id, sp in enumerate(disease_types):\n    for file in os.listdir(os.path.join(train_dir, sp)):\n        train_data.append(['{}/{}'.format(sp, file), defects_id, sp])\n        \ntrain = pd.DataFrame(train_data, columns=['File', 'DiseaseID','Disease Type'])\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:10:30.829287Z","iopub.execute_input":"2024-05-14T03:10:30.829770Z","iopub.status.idle":"2024-05-14T03:10:31.310233Z","shell.execute_reply.started":"2024-05-14T03:10:30.829691Z","shell.execute_reply":"2024-05-14T03:10:31.309568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Randomize the order of training set","metadata":{}},{"cell_type":"code","source":"SEED = 42\ntrain = train.sample(frac=1, random_state=SEED) \ntrain.index = np.arange(len(train)) # Reset indices\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:10:31.311338Z","iopub.execute_input":"2024-05-14T03:10:31.311630Z","iopub.status.idle":"2024-05-14T03:10:31.325703Z","shell.execute_reply.started":"2024-05-14T03:10:31.311603Z","shell.execute_reply":"2024-05-14T03:10:31.324805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Plot a histogram","metadata":{}},{"cell_type":"code","source":"plt.hist(train['DiseaseID'])\nplt.title('Frequency Histogram of Species')\nplt.figure(figsize=(12, 12))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:10:31.329365Z","iopub.execute_input":"2024-05-14T03:10:31.329757Z","iopub.status.idle":"2024-05-14T03:10:31.549060Z","shell.execute_reply.started":"2024-05-14T03:10:31.329720Z","shell.execute_reply":"2024-05-14T03:10:31.548171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Display images of COVID","metadata":{}},{"cell_type":"code","source":"\ndef plot_defects(defect_types, rows, cols):\n    fig, ax = plt.subplots(rows, cols, figsize=(12, 12))\n    defect_files = train['File'][train['Disease Type'] == defect_types].values\n    n = 0\n    for i in range(rows):\n        for j in range(cols):\n            image_path = os.path.join(data_dir, defect_files[n])\n            ax[i, j].set_xticks([])\n            ax[i, j].set_yticks([])\n            ax[i, j].imshow(cv2.imread(image_path))\n            n += 1\n# Displays first n images of class from training set\nplot_defects('COVID', 5, 5)","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:10:31.550743Z","iopub.execute_input":"2024-05-14T03:10:31.551040Z","iopub.status.idle":"2024-05-14T03:10:32.939107Z","shell.execute_reply.started":"2024-05-14T03:10:31.551010Z","shell.execute_reply":"2024-05-14T03:10:32.938159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Display images of non-COVID","metadata":{}},{"cell_type":"code","source":"# Displays first n images of class from training set\nplot_defects('non-COVID', 5, 5)","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:10:32.940482Z","iopub.execute_input":"2024-05-14T03:10:32.940950Z","iopub.status.idle":"2024-05-14T03:10:34.438796Z","shell.execute_reply.started":"2024-05-14T03:10:32.940909Z","shell.execute_reply":"2024-05-14T03:10:34.437881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Image Read and Resize Function","metadata":{}},{"cell_type":"code","source":"IMAGE_SIZE = 120\ndef read_image(filepath):\n    return cv2.imread(os.path.join(data_dir, filepath)) # Loading a color image is the default flag\n# Resize image to target size\ndef resize_image(image, image_size):\n    return cv2.resize(image.copy(), image_size, interpolation=cv2.INTER_AREA)","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:10:34.440078Z","iopub.execute_input":"2024-05-14T03:10:34.440492Z","iopub.status.idle":"2024-05-14T03:10:34.447662Z","shell.execute_reply.started":"2024-05-14T03:10:34.440440Z","shell.execute_reply":"2024-05-14T03:10:34.446814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training Images","metadata":{}},{"cell_type":"code","source":"X_train = np.zeros((train.shape[0], IMAGE_SIZE, IMAGE_SIZE, 3))\nfor i, file in tqdm(enumerate(train['File'].values)):\n    image = read_image(file)\n    if image is not None:\n        X_train[i] = resize_image(image, (IMAGE_SIZE, IMAGE_SIZE))\n# Normalize the data\nX_Train = X_train / 255.\nprint('Train Shape: {}'.format(X_Train.shape))","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:10:34.449019Z","iopub.execute_input":"2024-05-14T03:10:34.449358Z","iopub.status.idle":"2024-05-14T03:11:02.517768Z","shell.execute_reply.started":"2024-05-14T03:10:34.449324Z","shell.execute_reply":"2024-05-14T03:11:02.516930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Converting Labels to Categorical","metadata":{}},{"cell_type":"code","source":"Y_train = train['DiseaseID'].values\nY_train = to_categorical(Y_train, num_classes=2)","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:11:02.518874Z","iopub.execute_input":"2024-05-14T03:11:02.519140Z","iopub.status.idle":"2024-05-14T03:11:02.523391Z","shell.execute_reply.started":"2024-05-14T03:11:02.519113Z","shell.execute_reply":"2024-05-14T03:11:02.522662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train Test Splitting","metadata":{}},{"cell_type":"code","source":"BATCH_SIZE = 256\n\n# Split the train and validation sets \nX_train, X_val, Y_train, Y_val = train_test_split(X_Train, Y_train, test_size=0.2, random_state=SEED)","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:11:02.524776Z","iopub.execute_input":"2024-05-14T03:11:02.525099Z","iopub.status.idle":"2024-05-14T03:11:02.818794Z","shell.execute_reply.started":"2024-05-14T03:11:02.525070Z","shell.execute_reply":"2024-05-14T03:11:02.817732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 64*64 training images","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 3, figsize=(15, 15))\nfor i in range(3):\n    ax[i].set_axis_off()\n    ax[i].imshow(X_train[i])\n    ax[i].set_title(disease_types[np.argmax(Y_train[i])])","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:11:02.819942Z","iopub.execute_input":"2024-05-14T03:11:02.820227Z","iopub.status.idle":"2024-05-14T03:11:03.031042Z","shell.execute_reply.started":"2024-05-14T03:11:02.820200Z","shell.execute_reply":"2024-05-14T03:11:03.030264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EPOCHS = 200\nSIZE=120\nN_ch=3","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:11:03.032246Z","iopub.execute_input":"2024-05-14T03:11:03.032628Z","iopub.status.idle":"2024-05-14T03:11:03.036827Z","shell.execute_reply.started":"2024-05-14T03:11:03.032596Z","shell.execute_reply":"2024-05-14T03:11:03.035784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## InceptionV3","metadata":{}},{"cell_type":"code","source":"def build_in():\n    inception = InceptionV3(weights='imagenet', include_top=False)\n\n    input = Input(shape=(SIZE, SIZE, N_ch))\n    x = Conv2D(3, (3, 3), padding='same')(input)\n    \n    x = inception(x)\n    \n    x = GlobalAveragePooling2D()(x)\n    x = BatchNormalization()(x)\n    x = Dropout(0.5)(x)\n    x = Dense(256, activation='relu')(x)\n    x = BatchNormalization()(x)\n    x = Dropout(0.5)(x)\n\n    # multi output\n    output = Dense(2,activation = 'softmax', name='root')(x)\n \n\n    # model\n    model = Model(input,output)\n    \n    optimizer = Adam(lr=0.002, beta_1=0.9, beta_2=0.999, epsilon=0.1, decay=0.0)\n    model.compile(loss='categorical_crossentropy', optimizer=optimizer, metrics=['accuracy'])\n    model.summary()\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:11:03.038102Z","iopub.execute_input":"2024-05-14T03:11:03.038390Z","iopub.status.idle":"2024-05-14T03:11:03.048878Z","shell.execute_reply.started":"2024-05-14T03:11:03.038363Z","shell.execute_reply":"2024-05-14T03:11:03.048026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Augmentation and Fitting Model ","metadata":{}},{"cell_type":"code","source":"model = build_in()\nannealer = ReduceLROnPlateau(monitor='val_accuracy', factor=0.5, patience=5, verbose=1, min_lr=1e-3)\ncheckpoint = ModelCheckpoint('InceptionV3.h5', verbose=1, save_best_only=True)\n# Generates batches of image data with data augmentation\ndatagen = ImageDataGenerator(rotation_range=360, # Degree range for random rotations\n                        width_shift_range=0.2, # Range for random horizontal shifts\n                        height_shift_range=0.2, # Range for random vertical shifts\n                        zoom_range=0.2, # Range for random zoom\n                        horizontal_flip=True, # Randomly flip inputs horizontally\n                        vertical_flip=True) # Randomly flip inputs vertically\n\ndatagen.fit(X_train)\n# Fits the model on batches with real-time data augmentation\nhist = model.fit_generator(datagen.flow(X_train, Y_train, batch_size=BATCH_SIZE),\n               steps_per_epoch=X_train.shape[0] // BATCH_SIZE,\n               epochs=EPOCHS,\n               verbose=2,\n               callbacks=[annealer, checkpoint],\n               validation_data=(X_val, Y_val))","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:11:03.049994Z","iopub.execute_input":"2024-05-14T03:11:03.050326Z","iopub.status.idle":"2024-05-14T03:33:38.163597Z","shell.execute_reply.started":"2024-05-14T03:11:03.050296Z","shell.execute_reply":"2024-05-14T03:33:38.162147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Final Loss and Accuracy","metadata":{}},{"cell_type":"code","source":"#model = load_model('../output/kaggle/working/model.h5')\nfinal_loss, final_accuracy = model.evaluate(X_val, Y_val)\nprint('Final Loss: {}, Final Accuracy: {}'.format(final_loss, final_accuracy))","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:33:38.165699Z","iopub.execute_input":"2024-05-14T03:33:38.166031Z","iopub.status.idle":"2024-05-14T03:33:40.311964Z","shell.execute_reply.started":"2024-05-14T03:33:38.165995Z","shell.execute_reply":"2024-05-14T03:33:40.311156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Confusion Matrix","metadata":{}},{"cell_type":"code","source":"Y_pred = model.predict(X_val)\n\nY_pred = np.argmax(Y_pred, axis=1)\nY_true = np.argmax(Y_val, axis=1)\n\ncm = confusion_matrix(Y_true, Y_pred)\nplt.figure(figsize=(12, 12))\nax = sns.heatmap(cm, cmap=plt.cm.Greens, annot=True, square=True, xticklabels=disease_types, yticklabels=disease_types)\nax.set_ylabel('Actual', fontsize=40)\nax.set_xlabel('Predicted', fontsize=40)","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:33:40.313140Z","iopub.execute_input":"2024-05-14T03:33:40.313406Z","iopub.status.idle":"2024-05-14T03:33:45.510298Z","shell.execute_reply.started":"2024-05-14T03:33:40.313380Z","shell.execute_reply":"2024-05-14T03:33:45.509284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''FP = confusion_matrix.sum(axis=0) - np.diag(confusion_matrix)  \nFN = confusion_matrix.sum(axis=1) - np.diag(confusion_matrix)\nTP = np.diag(confusion_matrix)\nTN = confusion_matrix.values.sum() - (FP + FN + TP)'''\n\nTN = cm[0][0]\n#print(TN)\nFN = cm[1][0]\n#print(FN)\nTP = cm[1][1]\n#print(TP)\nFP = cm[0][1]\n#print(FP)\n\n# Sensitivity, hit rate, recall, or true positive rate\nTPR = TP/(TP+FN)\nprint(TPR)\n# Specificity or true negative rate\nTNR = TN/(TN+FP)\nprint(TNR)\n# Precision or positive predictive value\nPPV = TP/(TP+FP)\n# Negative predictive value\nNPV = TN/(TN+FN)\n# Fall out or false positive rate\nFPR = FP/(FP+TN)\n# False negative rate\nFNR = FN/(TP+FN)\nprint(FNR)\n# False discovery rate\nFDR = FP/(TP+FP)\n\n# Overall accuracy\nACC = (TP+TN)/(TP+FP+FN+TN)\nprint(ACC)","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:33:45.512016Z","iopub.execute_input":"2024-05-14T03:33:45.512458Z","iopub.status.idle":"2024-05-14T03:33:45.524665Z","shell.execute_reply.started":"2024-05-14T03:33:45.512411Z","shell.execute_reply":"2024-05-14T03:33:45.523612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Accuracy and Loss Curve","metadata":{}},{"cell_type":"code","source":"# accuracy plot \nplt.plot(hist.history['accuracy'])\nplt.plot(hist.history['val_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()\n\n# loss plot\nplt.plot(hist.history['loss'])\nplt.plot(hist.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:33:45.526078Z","iopub.execute_input":"2024-05-14T03:33:45.526479Z","iopub.status.idle":"2024-05-14T03:33:45.857509Z","shell.execute_reply.started":"2024-05-14T03:33:45.526437Z","shell.execute_reply":"2024-05-14T03:33:45.856591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prediction from Image","metadata":{}},{"cell_type":"code","source":"from skimage import io\nfrom keras.preprocessing import image\n#path='imbalanced/Scratch/Scratch_400.jpg'\nimg = image.load_img('/kaggle/input/sarscov2-ctscan-dataset/COVID/Covid (1).png', grayscale=False, target_size=(120, 120))\nshow_img=image.load_img('/kaggle/input/sarscov2-ctscan-dataset/COVID/Covid (1).png', grayscale=False, target_size=(200, 200))\ndisease_class=['Covid-19','Non Covid-19']\nx = image.img_to_array(img)\nx = np.expand_dims(x, axis = 0)\nx /= 255\n\ncustom = model.predict(x)\nprint(custom[0])\n\nplt.imshow(show_img)\nplt.show()\n\na=custom[0]\nind=np.argmax(a)\n        \nprint('Prediction:',disease_class[ind])","metadata":{"execution":{"iopub.status.busy":"2024-05-14T03:39:46.553844Z","iopub.execute_input":"2024-05-14T03:39:46.554243Z","iopub.status.idle":"2024-05-14T03:39:46.756823Z","shell.execute_reply.started":"2024-05-14T03:39:46.554204Z","shell.execute_reply":"2024-05-14T03:39:46.756018Z"},"trusted":true},"execution_count":null,"outputs":[]}]}