{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n#import numpy as np # linear algebra\n#import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n#import os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-04T16:44:34.368925Z","iopub.execute_input":"2022-05-04T16:44:34.369396Z","iopub.status.idle":"2022-05-04T16:44:34.394708Z","shell.execute_reply.started":"2022-05-04T16:44:34.369298Z","shell.execute_reply":"2022-05-04T16:44:34.394055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nimport pydicom as dicom\nimport pandas as pd\n\n#pixel_data = []\ndf=pd.DataFrame()\n\n#image analysis. Take first 100 images alone\npaths = glob.glob(\"../input/rsna-pneumonia-detection-challenge/stage_2_train_images/*.dcm\")\n\nrsna_label=pd.read_csv(\"../input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv\")\n\n#lets find out label for each image and prepare the dataset for the CNN.\nfor i in range(1,1000):\n    dataset = dicom.dcmread(paths[i])\n    df_value = pd.DataFrame(dataset.values())\n    df_value[0] = df_value[0].apply(lambda x: dicom.dataelem.DataElement_from_raw(x) if isinstance(x, dicom.dataelem.RawDataElement) else x)\n    patientID=\"unknown\"\n    x= df_value[0][11]\n    #print(x)\n    if(x.name == \"Patient ID\"):\n        patientID = x.value\n    Target=rsna_label.loc[rsna_label['patientId'] == patientID].Target.to_list()[0]\n    df=df.append({'PatientID':patientID,'PixelData':dataset.pixel_array,'Label':Target},ignore_index=True)\n    #lookup the patient id in the label\n    #pixel_data.append(dataset.pixel_array)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-05-08T14:10:56.945263Z","iopub.execute_input":"2022-05-08T14:10:56.946042Z","iopub.status.idle":"2022-05-08T14:11:26.645679Z","shell.execute_reply.started":"2022-05-08T14:10:56.945931Z","shell.execute_reply":"2022-05-08T14:11:26.644634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:00:20.779771Z","iopub.execute_input":"2022-05-05T06:00:20.780027Z","iopub.status.idle":"2022-05-05T06:00:21.359237Z","shell.execute_reply.started":"2022-05-05T06:00:20.779997Z","shell.execute_reply":"2022-05-05T06:00:21.358452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import regularizers\nfrom tensorflow.keras.optimizers import Adam, RMSprop\nfrom tensorflow.keras.utils import plot_model\nfrom tensorflow.keras.models import Model, Sequential\nfrom tensorflow.keras.layers import Dense, Flatten, Conv2D, MaxPooling2D, Dropout\nfrom tensorflow.keras.layers import GlobalAveragePooling2D,  BatchNormalization, Activation\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, ModelCheckpoint, EarlyStopping\nfrom tensorflow.keras.applications.vgg16 import  VGG16\nfrom tensorflow.keras.applications import DenseNet121","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:00:30.4493Z","iopub.execute_input":"2022-05-05T06:00:30.450058Z","iopub.status.idle":"2022-05-05T06:00:35.203423Z","shell.execute_reply.started":"2022-05-05T06:00:30.450019Z","shell.execute_reply":"2022-05-05T06:00:35.202703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_size=512\n","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:00:42.051161Z","iopub.execute_input":"2022-05-05T06:00:42.05142Z","iopub.status.idle":"2022-05-05T06:00:42.057501Z","shell.execute_reply.started":"2022-05-05T06:00:42.05139Z","shell.execute_reply":"2022-05-05T06:00:42.056638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfc=df.copy()","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:00:44.966208Z","iopub.execute_input":"2022-05-05T06:00:44.966955Z","iopub.status.idle":"2022-05-05T06:00:44.971099Z","shell.execute_reply.started":"2022-05-05T06:00:44.966913Z","shell.execute_reply":"2022-05-05T06:00:44.969936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=dfc.PixelData\ny=dfc.Label.to_numpy()","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:00:52.216705Z","iopub.execute_input":"2022-05-05T06:00:52.216983Z","iopub.status.idle":"2022-05-05T06:00:52.221366Z","shell.execute_reply.started":"2022-05-05T06:00:52.216952Z","shell.execute_reply":"2022-05-05T06:00:52.220664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\ny=tf.keras.utils.to_categorical(y,num_classes=2)","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:00:54.906048Z","iopub.execute_input":"2022-05-05T06:00:54.906632Z","iopub.status.idle":"2022-05-05T06:00:54.91106Z","shell.execute_reply.started":"2022-05-05T06:00:54.906591Z","shell.execute_reply":"2022-05-05T06:00:54.910164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#converting the images to be accepted by densenet121\nimport numpy as np\nimport cv2\nimage_array = np.ndarray(shape=(X.shape[0], img_size, img_size, 3), dtype= np.uint8)\nfor i in range(X.size):\n    img=cv2.resize(X[i],(img_size,img_size),interpolation = cv2.INTER_AREA)\n    img.shape\n    img=np.reshape(img,(img_size, img_size, 1))\n    image_array[i, :, :, 0] = img[ :, :, 0]\n    image_array[i, :, :, 1] = img[ :, :, 0]\n    image_array[i, :, :, 2] = img[ :, :, 0]","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:01:00.944917Z","iopub.execute_input":"2022-05-05T06:01:00.945514Z","iopub.status.idle":"2022-05-05T06:01:02.66613Z","shell.execute_reply.started":"2022-05-05T06:01:00.945469Z","shell.execute_reply":"2022-05-05T06:01:02.665393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(image_array.shape)","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:01:12.247762Z","iopub.execute_input":"2022-05-05T06:01:12.248014Z","iopub.status.idle":"2022-05-05T06:01:12.253336Z","shell.execute_reply.started":"2022-05-05T06:01:12.247985Z","shell.execute_reply":"2022-05-05T06:01:12.252367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train,X_test,y_train,y_test=train_test_split(image_array,y,test_size=0.1,random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:01:15.495377Z","iopub.execute_input":"2022-05-05T06:01:15.496077Z","iopub.status.idle":"2022-05-05T06:01:16.2919Z","shell.execute_reply.started":"2022-05-05T06:01:15.496039Z","shell.execute_reply":"2022-05-05T06:01:16.291112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = X_train.astype('float32')/255\nX_test=X_test.astype('float32')/255\ny_train = tf.convert_to_tensor(y_train, dtype=tf.float32) \ny_test = tf.convert_to_tensor(y_test, dtype=tf.float32)","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:01:24.798312Z","iopub.execute_input":"2022-05-05T06:01:24.799104Z","iopub.status.idle":"2022-05-05T06:01:28.069304Z","shell.execute_reply.started":"2022-05-05T06:01:24.799053Z","shell.execute_reply":"2022-05-05T06:01:28.068121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_train = X_train.reshape(X_train.shape[0], img_size, img_size, 1)\n#X_test = X_test.reshape(X_test.shape[0], img_size, img_size, 1)","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:01:28.148532Z","iopub.execute_input":"2022-05-05T06:01:28.148832Z","iopub.status.idle":"2022-05-05T06:01:28.152253Z","shell.execute_reply.started":"2022-05-05T06:01:28.148801Z","shell.execute_reply":"2022-05-05T06:01:28.151371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_chextnet_model_v1():\n    \"\"\"\n    v1 - uses densenet + chextnet weights\n    \"\"\"\n    # load model design\n    pre_model = DenseNet121(weights=None,\n                        include_top=False,\n                        input_shape=(img_size,img_size,3)\n                        )\n    out = Dense(14, activation='sigmoid')(pre_model.output)\n    pre_model = Model(inputs=pre_model.input, outputs=out) \n    \n    # load model wieghts\n    chex_weights_path = '../input/chexnet-keras-weights/brucechou1983_CheXNet_Keras_0.3.0_weights.h5'\n    pre_model.load_weights(chex_weights_path)\n\n    # make layers trainable?\n    pre_model.trainable = False\n    #pre_model.trainable = True\n\n    # get summary/print\n    pre_model.summary()\n    \n    \n    # add future layers.\n    # last_layer = pre_model.get_layer('conv5_block16_concat')\n    last_layer = pre_model.layers[-2]\n\n    print('last layer output shape: ', last_layer.output_shape)\n    last_output = last_layer.output\n#     last_layer\n\n    # Flatten the output layer to 1 dimension\n    # x = Flatten()(last_output)\n    x = GlobalAveragePooling2D()(last_output)\n\n    # Add a fully connected layer with 512 hidden units and ReLU activation\n    # x = Dense(512, activation='relu')(x)\n    # Add a dropout rate of 0.2\n    # x = Dropout(0.2)(x)                  \n\n\n    # # Add a fully connected layer with 128 hidden units and ReLU activation\n    # x = Dense(128, activation='relu')(x)\n\n\n    # Add final classification layer\n    x = Dense(2, activation='softmax')(x)\n\n    # final model\n    model = Model( pre_model.input, x) \n\n    # model.summary()\n    # plot_model(model)\n\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:01:40.004515Z","iopub.execute_input":"2022-05-05T06:01:40.00483Z","iopub.status.idle":"2022-05-05T06:01:40.013248Z","shell.execute_reply.started":"2022-05-05T06:01:40.004794Z","shell.execute_reply":"2022-05-05T06:01:40.012218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_vanilla_cnn_model_v1():\n    in1 = tf.keras.layers.Input(shape=(img_size, img_size, 3))\n    \n#     out1 = tf.keras.layers.Conv2D(4,(3,3),activation=\"relu\")(in1)\n    out1 = tf.keras.layers.Conv2D(32,(3,3),\n                                  activation=\"relu\",\n                                  padding='same')(in1)\n    out1 = tf.keras.layers.MaxPooling2D((2,2))(out1)\n    \n    out1 = tf.keras.layers.Conv2D(32,(3,3),\n                                  activation=\"relu\",\n                                  padding='same')(out1)\n    out1 = tf.keras.layers.MaxPooling2D((2,2))(out1)\n\n    out1 = tf.keras.layers.Flatten()(out1)\n    \n    out2 = tf.keras.layers.Dense(20,activation=\"relu\")(out1)\n    out2 = tf.keras.layers.Dense(10,activation=\"relu\")(out2)\n    \n    \n    out2 = Dense(2, \n                 activation='softmax',\n                 name='class_out',\n                 kernel_regularizer=regularizers.l2(0.01))(out2)\n\n\n    model = tf.keras.Model(inputs=in1,\n                           outputs=out2)\n\n    model.summary()\n    return model\n","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:01:47.201335Z","iopub.execute_input":"2022-05-05T06:01:47.202102Z","iopub.status.idle":"2022-05-05T06:01:47.209998Z","shell.execute_reply.started":"2022-05-05T06:01:47.20206Z","shell.execute_reply":"2022-05-05T06:01:47.209269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_vgg16_model_v1():\n    \"\"\"\n    v1 - https://www.kaggle.com/saeedniksaz/transferlearning-with-vgg16\n    \"\"\"\n    base_model = VGG16(input_shape=(img_size,img_size,3), \n                         include_top=False,\n                         weights=\"imagenet\")\n    # get summary/print\n#     base_model.summary()\n\n    for layer in base_model.layers:\n        layer.trainable = False\n        \n    model = Sequential()\n    model.add(base_model)\n    model.add(Dropout(0.5))\n    model.add(Flatten())\n    model.add(BatchNormalization())\n    \n    model.add(Dense(256,kernel_initializer='he_uniform'))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n    model.add(Dropout(0.5))\n    \n    model.add(Dense(32,kernel_initializer='he_uniform'))\n    model.add(BatchNormalization())\n    model.add(Activation('relu'))\n    model.add(Dropout(0.5))\n\n    model.add(Dense(2,activation='softmax'))\n        \n    model.summary()\n    plot_model(model)\n\n    return model\n","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:01:53.720631Z","iopub.execute_input":"2022-05-05T06:01:53.721193Z","iopub.status.idle":"2022-05-05T06:01:53.729031Z","shell.execute_reply.started":"2022-05-05T06:01:53.721152Z","shell.execute_reply":"2022-05-05T06:01:53.728222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = build_chextnet_model_v1()\n# model = build_densenet_coursera_model_v1()\n#model = build_vgg16_model_v1()\n#model = build_vanilla_cnn_model_v1()","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:02:01.107995Z","iopub.execute_input":"2022-05-05T06:02:01.108272Z","iopub.status.idle":"2022-05-05T06:02:04.973525Z","shell.execute_reply.started":"2022-05-05T06:02:01.108241Z","shell.execute_reply":"2022-05-05T06:02:04.970963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# metrics = ['categorical_accuracy', 'accuracy']\n# metrics = [tf.keras.metrics.AUC(), 'accuracy']\n#metrics = [ 'accuracy', tf.keras.metrics.AUC()]\nmetrics = [ 'accuracy']","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:02:05.266574Z","iopub.execute_input":"2022-05-05T06:02:05.266818Z","iopub.status.idle":"2022-05-05T06:02:05.270365Z","shell.execute_reply.started":"2022-05-05T06:02:05.266791Z","shell.execute_reply":"2022-05-05T06:02:05.269499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(Adam(learning_rate=1e-3),\n              loss='categorical_crossentropy',\n              metrics=metrics)","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:02:20.420217Z","iopub.execute_input":"2022-05-05T06:02:20.420917Z","iopub.status.idle":"2022-05-05T06:02:20.440494Z","shell.execute_reply.started":"2022-05-05T06:02:20.42088Z","shell.execute_reply":"2022-05-05T06:02:20.439851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"es = EarlyStopping(monitor = 'val_loss', \n                   min_delta = 1e-4, \n                   patience = 3, \n                   mode = 'min', \n                   restore_best_weights = True, \n                   verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:02:23.488689Z","iopub.execute_input":"2022-05-05T06:02:23.488945Z","iopub.status.idle":"2022-05-05T06:02:23.492938Z","shell.execute_reply.started":"2022-05-05T06:02:23.488917Z","shell.execute_reply":"2022-05-05T06:02:23.492226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_name = 'chextnet_model_512_july11.h5'","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:03:08.78027Z","iopub.execute_input":"2022-05-05T06:03:08.780534Z","iopub.status.idle":"2022-05-05T06:03:08.784473Z","shell.execute_reply.started":"2022-05-05T06:03:08.780504Z","shell.execute_reply":"2022-05-05T06:03:08.783633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ckp = ModelCheckpoint(model_name,\n                      monitor = 'val_loss',\n                      verbose = 0, \n                      save_best_only = True, \n                      mode = 'min')","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:03:11.328825Z","iopub.execute_input":"2022-05-05T06:03:11.32941Z","iopub.status.idle":"2022-05-05T06:03:11.334245Z","shell.execute_reply.started":"2022-05-05T06:03:11.329368Z","shell.execute_reply":"2022-05-05T06:03:11.33358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs=10","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:02:40.579452Z","iopub.execute_input":"2022-05-05T06:02:40.582608Z","iopub.status.idle":"2022-05-05T06:02:40.588573Z","shell.execute_reply.started":"2022-05-05T06:02:40.582557Z","shell.execute_reply":"2022-05-05T06:02:40.587546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n      X_train,y_train,\n      validation_data=(X_test,y_test),\n      epochs=epochs,\n      callbacks=[es, ckp]\n#     class_weight=class_weight\n)","metadata":{"execution":{"iopub.status.busy":"2022-05-05T06:03:53.458329Z","iopub.execute_input":"2022-05-05T06:03:53.458655Z","iopub.status.idle":"2022-05-05T06:06:25.228028Z","shell.execute_reply.started":"2022-05-05T06:03:53.458619Z","shell.execute_reply":"2022-05-05T06:06:25.227203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}