{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nimport tensorflow_io as tfio\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nsns.set(color_codes=True)\nfrom tqdm.keras import TqdmCallback","metadata":{"execution":{"iopub.status.busy":"2023-06-22T13:38:39.782387Z","iopub.execute_input":"2023-06-22T13:38:39.782706Z","iopub.status.idle":"2023-06-22T13:38:49.308064Z","shell.execute_reply.started":"2023-06-22T13:38:39.782674Z","shell.execute_reply":"2023-06-22T13:38:49.306889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CLASS_INFO_PATH = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv'\nTRAIN_LABELS_PATH = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv'\nTRAIN_IMAGES_FOLDER = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images'\nTEST_IMAGES_FOLDER = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_test_images'","metadata":{"execution":{"iopub.status.busy":"2023-06-22T13:38:56.806564Z","iopub.execute_input":"2023-06-22T13:38:56.807564Z","iopub.status.idle":"2023-06-22T13:38:56.815685Z","shell.execute_reply.started":"2023-06-22T13:38:56.807521Z","shell.execute_reply":"2023-06-22T13:38:56.812036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_info = pd.read_csv(CLASS_INFO_PATH)\nprint(f\"Shape Detailed class information: {class_info.shape}\")\ntrain_labels = pd.read_csv(TRAIN_LABELS_PATH)\nprint(f\"Shape Train labels information: {train_labels.shape}\")","metadata":{"execution":{"iopub.status.busy":"2023-06-22T13:40:21.564086Z","iopub.execute_input":"2023-06-22T13:40:21.564522Z","iopub.status.idle":"2023-06-22T13:40:21.695823Z","shell.execute_reply.started":"2023-06-22T13:40:21.564487Z","shell.execute_reply":"2023-06-22T13:40:21.694657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels['bbox'] = train_labels[['x', 'y', 'width', 'height']].values.tolist()\ntrain_labels.drop(['x', 'y', 'width', 'height'], axis=1, inplace=True)\nclass_info['class_codes'] = class_info['class'].astype('category').cat.codes\ndf = pd.concat([train_labels, class_info[['class','class_codes']]], axis=1)\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2023-06-22T13:40:24.407517Z","iopub.execute_input":"2023-06-22T13:40:24.408006Z","iopub.status.idle":"2023-06-22T13:40:24.682398Z","shell.execute_reply.started":"2023-06-22T13:40:24.407963Z","shell.execute_reply":"2023-06-22T13:40:24.681215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a new column with the filepaths for the chest x-ray images.\ndef get_image_path(patientId, use_test_images=False):\n    folder = TEST_IMAGES_FOLDER if use_test_images else TRAIN_IMAGES_FOLDER\n    return f'{folder}/{patientId}.dcm'\n\ndf.loc[:, 'image_path'] = df['patientId'].apply(get_image_path)","metadata":{"execution":{"iopub.status.busy":"2023-06-22T13:40:30.748231Z","iopub.execute_input":"2023-06-22T13:40:30.749204Z","iopub.status.idle":"2023-06-22T13:40:30.77321Z","shell.execute_reply.started":"2023-06-22T13:40:30.749165Z","shell.execute_reply":"2023-06-22T13:40:30.772203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 32\nIMG_INPUT_SIZE = 256\n\ndef read_image_tensor(filename):\n    raw = tf.io.read_file(filename)\n    im = tfio.image.decode_dicom_image(raw, color_dim=False, scale='auto', on_error='lossy', dtype=tf.float32)\n    im = tf.reshape(im, [1024, 1024, 1])\n    return tf.image.resize(im, [IMG_INPUT_SIZE, IMG_INPUT_SIZE])\n\ndef create_dataset(filenames, labels):\n    image_ds = tf.data.Dataset.from_tensor_slices(filenames).map(\n        lambda filename: read_image_tensor(filename),\n        num_parallel_calls=tf.data.AUTOTUNE\n    )\n    label_ds = tf.data.Dataset.from_tensor_slices(labels).map(\n        lambda label: tf.one_hot(label, 3)\n    )\n    return tf.data.Dataset.zip((image_ds, label_ds)).cache().shuffle(1000, seed=24).batch(BATCH_SIZE).prefetch(4)\n\ntrain_df, val_df = train_test_split(df, test_size=0.2, random_state=42)\ntrain_ds = create_dataset(train_df['image_path'].tolist(), train_df['class_codes'].tolist())\nval_ds = create_dataset(val_df['image_path'].tolist(), val_df['class_codes'].tolist())","metadata":{"execution":{"iopub.status.busy":"2023-06-22T14:06:49.909984Z","iopub.execute_input":"2023-06-22T14:06:49.910813Z","iopub.status.idle":"2023-06-22T14:06:50.274374Z","shell.execute_reply.started":"2023-06-22T14:06:49.910767Z","shell.execute_reply":"2023-06-22T14:06:50.273277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EPOCHS = 30\n\nmodel = tf.keras.models.Sequential([\n    tf.keras.layers.Resizing(IMG_INPUT_SIZE, IMG_INPUT_SIZE),\n    tf.keras.layers.Conv2D(filters=32, kernel_size=3, activation='relu'),\n    tf.keras.layers.MaxPooling2D(pool_size=2),\n    tf.keras.layers.Dropout(0.4),\n    tf.keras.layers.Flatten(), # flatten out the layers\n    tf.keras.layers.Dense(32, activation='relu'),\n    tf.keras.layers.Dense(3, activation='softmax')\n])\n\ncheckpoint = tf.keras.callbacks.ModelCheckpoint('base_cnn.h5', save_best_only=True, monitor='val_acc', mode='max', verbose=0)\nearly_stopping = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=10)\ntqdm = TqdmCallback()\nmodel.compile(loss='categorical_crossentropy', metrics=['acc'], optimizer=\"adam\")\nhistory = model.fit(train_ds, epochs=EPOCHS, validation_data=val_ds, callbacks=[early_stopping,checkpoint,tqdm], verbose=0)\ndisplay(model.summary())","metadata":{"execution":{"iopub.status.busy":"2023-06-22T14:06:57.403607Z","iopub.execute_input":"2023-06-22T14:06:57.404016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(model, \"base_cnn.png\", show_shapes=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Plottting the accuracy vs loss graph\nacc = history.history['acc']\nval_acc = history.history['val_acc']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nepochs_range = range(EPOCHS)\n\nfig=plt.figure(figsize=(10, 5))\ncolumns = 2; rows = 1\nax1 = fig.add_subplot(rows, columns, 1)\nax1.plot(epochs_range, acc, label='Training Accuracy')\nax1.plot(epochs_range, val_acc, label='Validation Accuracy')\nax1.set_title('Training and Validation Accuracy')\nax1.legend(loc='lower right')\n\nax2 = fig.add_subplot(rows, columns, 2)\nax2.plot(epochs_range, loss, label='Training Loss')\nax2.plot(epochs_range, val_loss, label='Validation Loss')\nax2.set_title('Training and Validation Loss')\nax2.legend(loc='upper right')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}