{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10338,"databundleVersionId":862042,"sourceType":"competition"}],"dockerImageVersionId":30498,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import warnings\nfrom sklearn.exceptions import ConvergenceWarning\nwarnings.filterwarnings(\"ignore\", category=ConvergenceWarning)\nwarnings.simplefilter(action='ignore', category=FutureWarning)\nwarnings.simplefilter(action='ignore', category=UserWarning)","metadata":{"execution":{"iopub.status.busy":"2023-06-19T07:11:56.481157Z","iopub.execute_input":"2023-06-19T07:11:56.481701Z","iopub.status.idle":"2023-06-19T07:11:57.547828Z","shell.execute_reply.started":"2023-06-19T07:11:56.481671Z","shell.execute_reply":"2023-06-19T07:11:57.546849Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport pydicom as dcm\nfrom pathlib import Path\nimport os\nfrom tqdm.notebook import tqdm\nimport cv2\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report\nfrom tensorflow.keras.layers import Layer, Convolution2D, Dense, RandomRotation, RandomFlip, Resizing, Rescaling\nfrom tensorflow.keras.layers import Concatenate, UpSampling2D, Conv2D, Reshape, GlobalAveragePooling2D, GlobalMaxPooling2D\nfrom tensorflow.keras.layers import Dense, Activation, Flatten, Dropout, MaxPooling2D, BatchNormalization\nfrom tensorflow.keras.layers import Input, ReLU, AveragePooling2D, Activation, Flatten\nfrom tensorflow.keras.models import Model, Sequential\nfrom tensorflow.keras.optimizers import Adam, AdamW\nfrom tensorflow.keras import losses, optimizers\nfrom keras import backend as K\nfrom tensorflow.keras.callbacks import Callback, EarlyStopping, ModelCheckpoint, ReduceLROnPlateau","metadata":{"execution":{"iopub.status.busy":"2023-06-19T07:43:37.911270Z","iopub.execute_input":"2023-06-19T07:43:37.911643Z","iopub.status.idle":"2023-06-19T07:43:37.924961Z","shell.execute_reply.started":"2023-06-19T07:43:37.911613Z","shell.execute_reply":"2023-06-19T07:43:37.923957Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_class = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv')\ntrain_labels = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')\n\ntrain_path = Path('../input/rsna-pneumonia-detection-challenge/stage_2_train_images')\ntest_path = Path('../input/rsna-pneumonia-detection-challenge/stage_2_test_images')\n\ntrain_meta = pd.concat([train_labels, train_class.drop(columns=['patientId'])], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-06-19T07:12:05.978019Z","iopub.execute_input":"2023-06-19T07:12:05.979127Z","iopub.status.idle":"2023-06-19T07:12:06.097177Z","shell.execute_reply.started":"2023-06-19T07:12:05.979091Z","shell.execute_reply":"2023-06-19T07:12:06.096157Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def save_img_from_dcm(dcm_dir, img_dir, patient_id):\n    img_fp = os.path.join(img_dir, \"{}.jpg\".format(patient_id))\n    if os.path.exists(img_fp):\n        return\n    dcm_fp = os.path.join(dcm_dir, \"{}.dcm\".format(patient_id))\n    img_1ch = pydicom.read_file(dcm_fp).pixel_array\n    img_3ch = np.stack([img_1ch]*3, -1)\n\n    img_fp = os.path.join(img_dir, \"{}.jpg\".format(patient_id))\n    cv2.imwrite(img_fp, img_3ch)","metadata":{"execution":{"iopub.status.busy":"2023-06-19T07:12:06.099696Z","iopub.execute_input":"2023-06-19T07:12:06.100077Z","iopub.status.idle":"2023-06-19T07:12:06.106802Z","shell.execute_reply.started":"2023-06-19T07:12:06.100043Z","shell.execute_reply":"2023-06-19T07:12:06.105891Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\ndef get_image(dcm_file):\n    ADJUSTED_IMAGE_SIZE = 128\n    dcm_data = dcm.read_file(dcm_file)\n    img = dcm_data.pixel_array\n    img = np.stack((img,) * 3, -1)\n\n    img = np.array(img).astype(np.uint8)\n    res = cv2.resize(img,(ADJUSTED_IMAGE_SIZE,ADJUSTED_IMAGE_SIZE), interpolation = cv2.INTER_LINEAR)\n    return res\n\ndef read_train(rowData):\n    imageList = []\n    for index, row in tqdm(rowData.iterrows()):\n        patientId = row.patientId\n        dcm_file = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_train_images/'+'{}.dcm'.format(patientId)\n        imageList.append(get_image(dcm_file))\n\n    return np.array(imageList)\n\ndef read_test(path):\n    imageList = []\n    for file_name in tqdm(os.listdir(path)):\n        dcm_file = dcm_file = os.sep.join([path, file_name])\n        imageList.append(get_image(dcm_file))\n    return np.array(imageList)\n        \ntest_images_path = '/kaggle/input/rsna-pneumonia-detection-challenge/stage_2_test_images'\ntest_images = read_test(test_images_path)\n\ntrain_images = read_train(train_labels)\nprint(train_images.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-19T07:12:06.108304Z","iopub.execute_input":"2023-06-19T07:12:06.108933Z","iopub.status.idle":"2023-06-19T07:25:02.445876Z","shell.execute_reply.started":"2023-06-19T07:12:06.108902Z","shell.execute_reply":"2023-06-19T07:25:02.444899Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(25,25)) # \nfor i, image in enumerate(train_images[:9]):\n\n    plt.subplot(3,3,i+1)\n    plt.imshow(image)\n    if train_labels.loc[i][\"Target\"]:\n        plt.title(\"Pneumonia\", color=\"red\", fontsize=25)\n    else:\n        plt.title(\"No Pneumonia\", color=\"blue\", fontsize=25)\n    plt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-19T07:25:11.570775Z","iopub.execute_input":"2023-06-19T07:25:11.571154Z","iopub.status.idle":"2023-06-19T07:25:12.584691Z","shell.execute_reply.started":"2023-06-19T07:25:11.571122Z","shell.execute_reply":"2023-06-19T07:25:12.583664Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = pd.get_dummies(train_labels[\"Target\"]).values\nrandom_state = 42\n\nX_train, X_test, y_train, y_test = train_test_split(train_images, y, test_size=0.2, random_state=random_state)","metadata":{"execution":{"iopub.status.busy":"2023-06-19T07:25:24.771207Z","iopub.execute_input":"2023-06-19T07:25:24.771573Z","iopub.status.idle":"2023-06-19T07:25:25.211065Z","shell.execute_reply.started":"2023-06-19T07:25:24.771543Z","shell.execute_reply":"2023-06-19T07:25:25.210105Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Building CNN Model","metadata":{}},{"cell_type":"code","source":"\ndef recall_m(y_true, y_pred):\n    true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n    possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n    recall = true_positives / (possible_positives + K.epsilon())\n    return recall\n\ndef precision_m(y_true, y_pred):\n    true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n    predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n    precision = true_positives / (predicted_positives + K.epsilon())\n    return precision\n\ndef f1_m(y_true, y_pred):\n    precision = precision_m(y_true, y_pred)\n    recall = recall_m(y_true, y_pred)\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))","metadata":{"execution":{"iopub.status.busy":"2023-06-19T07:26:36.348830Z","iopub.execute_input":"2023-06-19T07:26:36.349584Z","iopub.status.idle":"2023-06-19T07:26:36.359666Z","shell.execute_reply.started":"2023-06-19T07:26:36.349547Z","shell.execute_reply":"2023-06-19T07:26:36.358631Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ADJUSTED_IMAGE_SIZE = 128\ninput_shape = (ADJUSTED_IMAGE_SIZE, ADJUSTED_IMAGE_SIZE, 3)\nnum_classes = y_train.shape[1]\n\nmodel = Sequential()\nmodel.add(RandomRotation(factor=0.15))\nmodel.add(Rescaling(1./255))\nmodel.add(Conv2D(32, (3, 3), input_shape=input_shape)) # (3, 3) - conv kernel\n\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.2))\n\nmodel.add(Conv2D(32, (3, 3)))\n\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.3))\n\nmodel.add(Conv2D(64, (3, 3)))\n\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.4))\n\nmodel.add(Flatten())\nmodel.add(Dense(64))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(num_classes))\nmodel.add(Activation('softmax'))\n\nmodel.compile(loss='categorical_crossentropy',\n              optimizer=AdamW(learning_rate=0.0005),\n              metrics=['accuracy'] # ,f1_m\n             )\n# model.summary()\n\nhistory = model.fit(X_train, \n                  y_train, \n                  epochs = 50, \n                  validation_data = (X_test,y_test),\n                  batch_size = 16,\n                    callbacks=[\n                        EarlyStopping(monitor = \"val_loss\", patience = 10, restore_best_weights = True),\n                        ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=2, mode='min')\n                    ]\n                   )\n\nfcl_loss, fcl_accuracy = model.evaluate(X_test, y_test, verbose=1) # , fcl_f1\nprint('Test loss:', fcl_loss)\nprint('Test accuracy:', fcl_accuracy)\n# print('Test F1:', fcl_f1)\n\ndf = pd.DataFrame({\"pred\": np.argmax(model.predict(X_test), axis=1), \"true\": np.argmax(y_test, axis=1)})\nprint(classification_report(df[\"true\"], df[\"pred\"]))","metadata":{"execution":{"iopub.status.busy":"2023-06-19T07:26:46.942392Z","iopub.execute_input":"2023-06-19T07:26:46.942813Z","iopub.status.idle":"2023-06-19T07:33:30.921281Z","shell.execute_reply.started":"2023-06-19T07:26:46.942778Z","shell.execute_reply":"2023-06-19T07:33:30.920070Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nplt.figure(figsize=(16, 8))\n\nplt.plot(history.history['accuracy'], label='Train')\nplt.plot(history.history['val_accuracy'], label='Validation')\nplt.ylabel('Cross Entropy accuracy')\nplt.xlabel('Epoch')\nplt.title('Train accuracy', pad=13, fontsize=25)\nplt.legend(loc='upper right')\nplt.grid(000.1)\n\nplt.show()\n\n\nplt.figure(figsize=(16, 8))\n\nplt.plot(history.history['loss'], label='Train')\nplt.plot(history.history['val_loss'], label='Validation')\nplt.ylabel('Cross Entropy Loss')\nplt.xlabel('Epoch')\nplt.title('Train Loss', pad=13, fontsize=25)\nplt.legend(loc='upper right')\nplt.grid(000.1)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-19T07:42:00.023163Z","iopub.execute_input":"2023-06-19T07:42:00.023535Z","iopub.status.idle":"2023-06-19T07:42:00.486961Z","shell.execute_reply.started":"2023-06-19T07:42:00.023504Z","shell.execute_reply":"2023-06-19T07:42:00.485757Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Build transfer-learning model","metadata":{}},{"cell_type":"code","source":"# potential models: Xception ResNet50V2 InceptionV3 EfficientNetB3 ResNet152 VGG16\nbase_model = tf.keras.applications.VGG16(weights = 'imagenet', include_top = False)\nfor layer in base_model.layers:\n      layer.trainable = False\n\n# Data Augmentation Step\naugment = Sequential([\n#   RandomFlip(\"horizontal\"),\n#     Rescaling(1./255), # pretrained model already include rescaling\n    RandomRotation(0.15)\n], name='AugmentationLayer')\n\ninputs = Input(shape = input_shape, name='inputLayer')\nx = augment(inputs)\npretrain_out = base_model(x, training = False)\nx = Flatten()(pretrain_out)\nx = Dense(64, activation='relu')(x)\nx = Dense(y_train.shape[1], name='outputLayer')(x)\noutputs = Activation(activation=\"softmax\", dtype=tf.float32, name='activationLayer')(x)\nmodel = Model(inputs=inputs, outputs=outputs)\n\nmodel.compile(optimizer=AdamW(learning_rate=0.0005), # \n                   loss=losses.categorical_crossentropy, \n                   metrics=['accuracy']) # ,f1_m\n\nhistory = model.fit(X_train, \n                         y_train, \n                         batch_size=16, \n                         epochs=60, \n                         validation_data=(X_test, y_test),\n                         callbacks=[\n                             EarlyStopping(monitor = \"val_loss\", patience = 15, \n                                           restore_best_weights = True, mode='min'),\n                             ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=2, mode='min')\n                    ]\n                        )\nfcl_loss, fcl_accuracy  = model.evaluate(X_test, y_test, verbose=1) # fcl_f1\nprint('Test loss:', fcl_loss)\nprint('Test accuracy:', fcl_accuracy)\n# print('Test F1:', fcl_f1)\n\n\ndf = pd.DataFrame({\"pred\": np.argmax(model.predict(X_test), axis=1), \"true\": np.argmax(y_test, axis=1)})\nprint(classification_report(df[\"true\"], df[\"pred\"]))","metadata":{"execution":{"iopub.status.busy":"2023-06-19T08:40:41.662533Z","iopub.execute_input":"2023-06-19T08:40:41.662900Z","iopub.status.idle":"2023-06-19T08:44:09.794639Z","shell.execute_reply.started":"2023-06-19T08:40:41.662870Z","shell.execute_reply":"2023-06-19T08:44:09.793635Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nplt.figure(figsize=(16, 8))\n\nplt.plot(history.history['accuracy'], label='Train')\nplt.plot(history.history['val_accuracy'], label='Validation')\nplt.ylabel('Cross Entropy accuracy')\nplt.xlabel('Epoch')\nplt.title('Train accuracy', pad=13, fontsize=25)\nplt.legend(loc='upper right')\nplt.grid(000.1)\n\nplt.show()\n\n\nplt.figure(figsize=(16, 8))\n\nplt.plot(history.history['loss'], label='Train')\nplt.plot(history.history['val_loss'], label='Validation')\nplt.ylabel('Cross Entropy Loss')\nplt.xlabel('Epoch')\nplt.title('Train Loss', pad=13, fontsize=25)\nplt.legend(loc='upper right')\nplt.grid(000.1)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-19T09:02:35.741531Z","iopub.execute_input":"2023-06-19T09:02:35.741906Z","iopub.status.idle":"2023-06-19T09:03:07.753908Z","shell.execute_reply.started":"2023-06-19T09:02:35.741878Z","shell.execute_reply":"2023-06-19T09:03:07.752916Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Fine tuning model","metadata":{}},{"cell_type":"code","source":"base_model.trainable = True\nfor layer in base_model.layers:\n    if isinstance(layer, BatchNormalization): # set BatchNorm layers as not trainable\n        layer.trainable = False\n\nmodel.compile(optimizer=AdamW(learning_rate=0.00001), # \n                   loss=losses.categorical_crossentropy, \n                   metrics=['accuracy']) # ,f1_m\n\nhistory = model.fit(X_train, \n                         y_train, \n                         batch_size=16, \n                         epochs=50, \n                         validation_data=(X_test, y_test),\n                         callbacks=[\n                             EarlyStopping(monitor = \"val_loss\", patience = 15, \n                                           restore_best_weights = True, mode='min'),\n                             ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=2, mode='min')\n                    ]\n                        )\nfcl_loss, fcl_accuracy  = model.evaluate(X_test, y_test, verbose=1) # fcl_f1\nprint('Test loss:', fcl_loss)\nprint('Test accuracy:', fcl_accuracy)\n# print('Test F1:', fcl_f1)\n\ndf = pd.DataFrame({\"pred\": np.argmax(model.predict(X_test), axis=1), \"true\": np.argmax(y_test, axis=1)})\nprint(classification_report(df[\"true\"], df[\"pred\"]))","metadata":{"execution":{"iopub.status.busy":"2023-06-19T09:04:05.274951Z","iopub.execute_input":"2023-06-19T09:04:05.275936Z","iopub.status.idle":"2023-06-19T09:05:33.120718Z","shell.execute_reply.started":"2023-06-19T09:04:05.275899Z","shell.execute_reply":"2023-06-19T09:05:33.116605Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nplt.figure(figsize=(16, 8))\n\nplt.plot(history.history['accuracy'], label='Train')\nplt.plot(history.history['val_accuracy'], label='Validation')\nplt.ylabel('Cross Entropy accuracy')\nplt.xlabel('Epoch')\nplt.title('Train accuracy', pad=13, fontsize=25)\nplt.legend(loc='upper right')\nplt.grid(000.1)\n\nplt.show()\n\n\nplt.figure(figsize=(16, 8))\n\nplt.plot(history.history['loss'], label='Train')\nplt.plot(history.history['val_loss'], label='Validation')\nplt.ylabel('Cross Entropy Loss')\nplt.xlabel('Epoch')\nplt.title('Train Loss', pad=13, fontsize=25)\nplt.legend(loc='upper right')\nplt.grid(000.1)\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-13T18:16:39.321992Z","iopub.execute_input":"2023-06-13T18:16:39.323057Z","iopub.status.idle":"2023-06-13T18:17:01.154764Z","shell.execute_reply.started":"2023-06-13T18:16:39.323011Z","shell.execute_reply":"2023-06-13T18:17:01.153621Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}