{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install livelossplot --quiet","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:39.524982Z","iopub.execute_input":"2021-11-27T21:47:39.525242Z","iopub.status.idle":"2021-11-27T21:47:41.248631Z","shell.execute_reply.started":"2021-11-27T21:47:39.525212Z","shell.execute_reply":"2021-11-27T21:47:41.247783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing the libraries\nimport os\nimport shutil\nimport glob\nfrom tqdm.notebook import tqdm\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nfrom tensorflow import keras\nimport cv2\nfrom PIL import Image\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport random\nfrom random import seed\nfrom livelossplot import PlotLossesKeras\nimport math","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:41.251064Z","iopub.execute_input":"2021-11-27T21:47:41.251359Z","iopub.status.idle":"2021-11-27T21:47:41.256998Z","shell.execute_reply.started":"2021-11-27T21:47:41.251319Z","shell.execute_reply":"2021-11-27T21:47:41.256277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing the Training Dataset\ndf_train = pd.read_csv(\"../input/jpeg-melanoma-384x384/train.csv\")\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:41.258236Z","iopub.execute_input":"2021-11-27T21:47:41.258631Z","iopub.status.idle":"2021-11-27T21:47:41.336593Z","shell.execute_reply.started":"2021-11-27T21:47:41.258597Z","shell.execute_reply":"2021-11-27T21:47:41.335929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:41.337642Z","iopub.execute_input":"2021-11-27T21:47:41.337850Z","iopub.status.idle":"2021-11-27T21:47:41.370463Z","shell.execute_reply.started":"2021-11-27T21:47:41.337825Z","shell.execute_reply":"2021-11-27T21:47:41.369777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing the Test Dataset\ndf_test = pd.read_csv(\"../input/jpeg-melanoma-384x384/test.csv\")\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:41.372971Z","iopub.execute_input":"2021-11-27T21:47:41.373490Z","iopub.status.idle":"2021-11-27T21:47:41.398824Z","shell.execute_reply.started":"2021-11-27T21:47:41.373451Z","shell.execute_reply":"2021-11-27T21:47:41.398138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a config class to store all the configurations\nclass config:\n    \n    # Image and Tabular data paths\n    DIRECTORY_PATH = \"../input/jpeg-melanoma-384x384/\"\n    TRAINING_SAMPLES_FOLDER = DIRECTORY_PATH + \"train/\"\n    TESTING_SAMPLES_FOLDER = DIRECTORY_PATH + \"test/\"\n    TRAIN_FULL_DATA = DIRECTORY_PATH + \"train.csv\"\n    TEST_FULL_DATA = DIRECTORY_PATH + \"test.csv\"\n    \n    # New directory path for image data\n    WORK_DIRECTORY = \"dataset/\"\n    TRAIN_IMAGES_FOLDER = WORK_DIRECTORY + \"training_set/\"\n    TEST_IMAGES_FOLDER = WORK_DIRECTORY + \"test_set/\"\n    VALIDATION_IMAGES_FOLDER = WORK_DIRECTORY + \"validation_set/\"\n    \n    # Input parameters for data preprocessing\n    TARGET_NAME = \"target\"\n    TRAIN_SIZE = 0.80\n    VALIDATION_SIZE = 0.10\n    TEST_SIZE = 0.10\n    SEED = 42\n    \n    # Tensorflow settings for model training\n    IMAGE_HEIGHT = 299\n    IMAGE_WIDTH = 299\n    NO_CHANNELS = 3\n    BATCH_SIZE = 64\n    EPOCHS = 20\n    DROPOUT = 0.5\n    LEARNING_RATE = 0.01\n    PATIENCE = 5","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:41.400183Z","iopub.execute_input":"2021-11-27T21:47:41.400656Z","iopub.status.idle":"2021-11-27T21:47:41.408220Z","shell.execute_reply.started":"2021-11-27T21:47:41.400610Z","shell.execute_reply":"2021-11-27T21:47:41.407464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the files along with the number of samples\nprint(os.listdir(config.DIRECTORY_PATH))\nprint(len(os.listdir(config.TRAINING_SAMPLES_FOLDER)), \"Training Samples\")\nprint(len(os.listdir(config.TESTING_SAMPLES_FOLDER)), \"Testing Samples\")","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:41.409639Z","iopub.execute_input":"2021-11-27T21:47:41.410138Z","iopub.status.idle":"2021-11-27T21:47:41.439542Z","shell.execute_reply.started":"2021-11-27T21:47:41.410101Z","shell.execute_reply":"2021-11-27T21:47:41.438858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating folders for training and validation data\ndataset_home = \"./dataset/\"\nsubdirs = [\"training_set/\", \"test_set/\", \"validation_set/\"]\nfor subdir in subdirs:\n    labeldirs = [\"benign\", \"malignant\"]\n    for labeldir in labeldirs:\n        newdir = dataset_home + subdir + labeldir\n        os.makedirs(newdir, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:41.441836Z","iopub.execute_input":"2021-11-27T21:47:41.442605Z","iopub.status.idle":"2021-11-27T21:47:41.447991Z","shell.execute_reply.started":"2021-11-27T21:47:41.442571Z","shell.execute_reply":"2021-11-27T21:47:41.447274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(os.listdir(\"./dataset\"))","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:41.451038Z","iopub.execute_input":"2021-11-27T21:47:41.451267Z","iopub.status.idle":"2021-11-27T21:47:41.461928Z","shell.execute_reply.started":"2021-11-27T21:47:41.451236Z","shell.execute_reply":"2021-11-27T21:47:41.461031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Splitting the dataset into train, test and validation set\n\ntest_examples = train_examples = validation_examples = 0\nseed(config.SEED)\n\nfor record in open(config.TRAIN_FULL_DATA).readlines()[1:]:\n    split_record = record.split(\",\")\n    image_name = split_record[0]\n    target = split_record[7]\n    \n    random_num = random.random()\n    \n    if random_num < config.TRAIN_SIZE:\n        destination = config.TRAIN_IMAGES_FOLDER\n        train_examples += 1\n        \n    elif random_num < 0.9:\n        destination = config.VALIDATION_IMAGES_FOLDER\n        validation_examples += 1\n        \n    else:\n        destination = config.TEST_IMAGES_FOLDER\n        test_examples += 1\n        \n    if target == \"0\":\n        shutil.copy(\n            config.TRAINING_SAMPLES_FOLDER + image_name + \".jpg\",\n            destination + \"benign/\" + image_name + \".jpg\"\n        )\n    \n    elif target == \"1\":\n        shutil.copy(\n            config.TRAINING_SAMPLES_FOLDER + image_name + \".jpg\",\n            destination + \"malignant/\" + image_name + \".jpg\"\n        )\n\nprint(f\"Number of training examples: {train_examples}\")\nprint(f\"Number of test examples: {test_examples}\")\nprint(f\"Number of validation examples: {validation_examples}\")","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:41.463270Z","iopub.execute_input":"2021-11-27T21:47:41.464483Z","iopub.status.idle":"2021-11-27T21:47:50.335743Z","shell.execute_reply.started":"2021-11-27T21:47:41.464449Z","shell.execute_reply":"2021-11-27T21:47:50.333845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preparing the data and performing Data Augmentation\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    shear_range=0.2,\n    zoom_range=(0.95, 0.95),\n    rotation_range=15,\n    horizontal_flip=True,\n    vertical_flip=True,\n    data_format=\"channels_last\",\n    dtype=tf.float32\n)\n\nvalidation_datagen = ImageDataGenerator(\n    rescale=1./255,\n    dtype=tf.float32\n)\n\ntrain_generator = train_datagen.flow_from_directory(\n    directory=config.TRAIN_IMAGES_FOLDER,\n    target_size=(config.IMAGE_HEIGHT, config.IMAGE_WIDTH),\n    color_mode=\"rgb\",\n    batch_size=config.BATCH_SIZE,\n    class_mode=\"binary\",\n    shuffle=True\n)\n\nvalidation_generator = validation_datagen.flow_from_directory(\n    directory=config.VALIDATION_IMAGES_FOLDER,\n    target_size=(config.IMAGE_HEIGHT, config.IMAGE_WIDTH),\n    color_mode=\"rgb\",\n    batch_size=config.BATCH_SIZE,\n    class_mode=\"binary\",\n    shuffle=True\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.336579Z","iopub.status.idle":"2021-11-27T21:47:50.336874Z","shell.execute_reply.started":"2021-11-27T21:47:50.336719Z","shell.execute_reply":"2021-11-27T21:47:50.336739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating an exponential learning rate scheduler\ndef exponential_decay(lr0, s):\n    def exponential_decay_fn(epoch):\n        return lr0 * 0.1**(epoch / s)\n    return exponential_decay_fn\n\nexponential_decay_fn = exponential_decay(lr0=config.LEARNING_RATE, s=20)\nlr_scheduler = keras.callbacks.LearningRateScheduler(exponential_decay_fn)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.338242Z","iopub.status.idle":"2021-11-27T21:47:50.338661Z","shell.execute_reply.started":"2021-11-27T21:47:50.338429Z","shell.execute_reply":"2021-11-27T21:47:50.338452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using callbacks to save model parameters and perform early stopping\ncheckpoint_cb = keras.callbacks.ModelCheckpoint(\"melanoma.h5\", save_best_only=True)\nearly_stopping_cb = keras.callbacks.EarlyStopping(patience=config.PATIENCE, restore_best_weights=True)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.341859Z","iopub.status.idle":"2021-11-27T21:47:50.342510Z","shell.execute_reply.started":"2021-11-27T21:47:50.342269Z","shell.execute_reply":"2021-11-27T21:47:50.342293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculating the different step size for the model while training\nSTEP_SIZE_TRAIN = train_generator.n // train_generator.batch_size\nSTEP_SIZE_VALIDATION = validation_generator.n // validation_generator.batch_size","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.343765Z","iopub.status.idle":"2021-11-27T21:47:50.344525Z","shell.execute_reply.started":"2021-11-27T21:47:50.344200Z","shell.execute_reply":"2021-11-27T21:47:50.344225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Metrics to use for compiling the model\nMETRICS = [keras.metrics.AUC(name=\"auc\")]","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.345796Z","iopub.status.idle":"2021-11-27T21:47:50.346476Z","shell.execute_reply.started":"2021-11-27T21:47:50.346225Z","shell.execute_reply":"2021-11-27T21:47:50.346250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculating the number of benign/malignant images\ntotal_images = train_examples + validation_examples\nbenign_images = len(os.listdir(config.TRAIN_IMAGES_FOLDER + \"benign\")) + len(os.listdir(config.VALIDATION_IMAGES_FOLDER + \"benign\"))\nmalignant_images = len(os.listdir(config.TRAIN_IMAGES_FOLDER + \"malignant\")) + len(os.listdir(config.VALIDATION_IMAGES_FOLDER + \"malignant\"))","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.347770Z","iopub.status.idle":"2021-11-27T21:47:50.348434Z","shell.execute_reply.started":"2021-11-27T21:47:50.348198Z","shell.execute_reply":"2021-11-27T21:47:50.348224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Building a transfer learning model using Keras\nxception = keras.applications.xception.Xception(input_shape=(config.IMAGE_HEIGHT, config.IMAGE_WIDTH, config.NO_CHANNELS),\n                                       weights=\"imagenet\", include_top=False)\nxception.trainable = False\n\nxcept_model = keras.models.Sequential([\n    xception,\n    keras.layers.GlobalAveragePooling2D(),\n    keras.layers.Dense(8, activation=\"relu\"),\n    keras.layers.Dense(1, activation=\"sigmoid\")\n])","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.349688Z","iopub.status.idle":"2021-11-27T21:47:50.350357Z","shell.execute_reply.started":"2021-11-27T21:47:50.350120Z","shell.execute_reply":"2021-11-27T21:47:50.350144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compiling the transfer learning model\nxcept_model.compile(optimizer=keras.optimizers.Adam(), \n              loss=keras.losses.BinaryCrossentropy(),\n              metrics=METRICS\n             )","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.351592Z","iopub.status.idle":"2021-11-27T21:47:50.352274Z","shell.execute_reply.started":"2021-11-27T21:47:50.352010Z","shell.execute_reply":"2021-11-27T21:47:50.352035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training the CNN model\nxcept_history = xcept_model.fit(train_generator, steps_per_epoch=STEP_SIZE_TRAIN, epochs=config.EPOCHS,\n                  validation_data=validation_generator, validation_steps=STEP_SIZE_VALIDATION,\n                  callbacks=[early_stopping_cb, checkpoint_cb, lr_scheduler, PlotLossesKeras()]\n                 )","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.353482Z","iopub.status.idle":"2021-11-27T21:47:50.354174Z","shell.execute_reply.started":"2021-11-27T21:47:50.353935Z","shell.execute_reply":"2021-11-27T21:47:50.353960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluating the model on Validation Dataset\nxcept_model.evaluate(validation_generator, steps=STEP_SIZE_VALIDATION)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.355390Z","iopub.status.idle":"2021-11-27T21:47:50.356050Z","shell.execute_reply.started":"2021-11-27T21:47:50.355796Z","shell.execute_reply":"2021-11-27T21:47:50.355820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a test generator for test data\ntest_datagen = ImageDataGenerator(\n    rescale=1./255,\n    dtype=tf.float32\n)\n\ntest_generator = test_datagen.flow_from_directory(\n    directory=config.TEST_IMAGES_FOLDER,\n    target_size=(config.IMAGE_HEIGHT, config.IMAGE_WIDTH),\n    color_mode=\"rgb\",\n    batch_size=config.BATCH_SIZE,\n    class_mode=\"binary\",\n    shuffle=False\n)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.357405Z","iopub.status.idle":"2021-11-27T21:47:50.358068Z","shell.execute_reply.started":"2021-11-27T21:47:50.357812Z","shell.execute_reply":"2021-11-27T21:47:50.357836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TEST = test_generator.n // test_generator.batch_size","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.359589Z","iopub.status.idle":"2021-11-27T21:47:50.360281Z","shell.execute_reply.started":"2021-11-27T21:47:50.360042Z","shell.execute_reply":"2021-11-27T21:47:50.360067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting the actual classes of the test dataset\ny_test = np.array([])\nnum_batches = 0\nfor _, y in test_generator:\n    y_test = np.append(y_test, y)\n    num_batches += 1\n    if num_batches == math.ceil(test_examples / config.BATCH_SIZE):\n        break\ny_test","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.361510Z","iopub.status.idle":"2021-11-27T21:47:50.362192Z","shell.execute_reply.started":"2021-11-27T21:47:50.361950Z","shell.execute_reply":"2021-11-27T21:47:50.361975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predicting output on the test dataset\ny_pred = xcept_model.predict(test_generator)\ny_pred","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.363443Z","iopub.status.idle":"2021-11-27T21:47:50.364139Z","shell.execute_reply.started":"2021-11-27T21:47:50.363849Z","shell.execute_reply":"2021-11-27T21:47:50.363874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Computing the TPR and FPR values from the roc curve\nfrom sklearn.metrics import roc_curve\nfpr, tpr, thresholds = roc_curve(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.365343Z","iopub.status.idle":"2021-11-27T21:47:50.366029Z","shell.execute_reply.started":"2021-11-27T21:47:50.365770Z","shell.execute_reply":"2021-11-27T21:47:50.365794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting the ROC curve\ndef plot_roc_curve (fpr, tpr, label = None):\n    plt.plot(fpr, tpr, linewidth = 2, label = label)\n    plt.plot([0,1], [0,1], 'k--') # Dashed diagonal\n    plt.xlabel(\"False Positive Rate\")\n    plt.ylabel(\"True Positive Rate (Recall)\")\n    plt.grid()\n    \nplot_roc_curve(fpr, tpr)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.367249Z","iopub.status.idle":"2021-11-27T21:47:50.367940Z","shell.execute_reply.started":"2021-11-27T21:47:50.367643Z","shell.execute_reply":"2021-11-27T21:47:50.367679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluating the model on test dataset\nxcept_model.evaluate(test_generator, steps=STEP_SIZE_TEST)","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.369193Z","iopub.status.idle":"2021-11-27T21:47:50.369836Z","shell.execute_reply.started":"2021-11-27T21:47:50.369577Z","shell.execute_reply":"2021-11-27T21:47:50.369602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving the best model after training\nxcept_model.save(\"final_melanoma_model.h5\")","metadata":{"execution":{"iopub.status.busy":"2021-11-27T21:47:50.371105Z","iopub.status.idle":"2021-11-27T21:47:50.371741Z","shell.execute_reply.started":"2021-11-27T21:47:50.371491Z","shell.execute_reply":"2021-11-27T21:47:50.371516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}