{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install livelossplot --quiet","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:51:33.281900Z","iopub.execute_input":"2023-04-02T16:51:33.282600Z","iopub.status.idle":"2023-04-02T16:51:44.446414Z","shell.execute_reply.started":"2023-04-02T16:51:33.282512Z","shell.execute_reply":"2023-04-02T16:51:44.445532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing the libraries\nimport os\nimport shutil\nimport glob\nfrom tqdm.notebook import tqdm\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nfrom tensorflow import keras\nimport cv2\nfrom PIL import Image\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport random\nfrom random import seed\nfrom livelossplot import PlotLossesKeras\nimport math","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:51:44.448582Z","iopub.execute_input":"2023-04-02T16:51:44.448879Z","iopub.status.idle":"2023-04-02T16:51:55.768776Z","shell.execute_reply.started":"2023-04-02T16:51:44.448829Z","shell.execute_reply":"2023-04-02T16:51:55.767916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing the Training Dataset\ndf_train = pd.read_csv(\"../input/jpeg-melanoma-384x384/train.csv\")\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:51:55.770249Z","iopub.execute_input":"2023-04-02T16:51:55.770494Z","iopub.status.idle":"2023-04-02T16:51:55.914417Z","shell.execute_reply.started":"2023-04-02T16:51:55.770459Z","shell.execute_reply":"2023-04-02T16:51:55.913672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:51:55.916317Z","iopub.execute_input":"2023-04-02T16:51:55.916522Z","iopub.status.idle":"2023-04-02T16:51:55.956775Z","shell.execute_reply.started":"2023-04-02T16:51:55.916497Z","shell.execute_reply":"2023-04-02T16:51:55.955231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing the Test Dataset\ndf_test = pd.read_csv(\"../input/jpeg-melanoma-384x384/test.csv\")\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:51:55.958552Z","iopub.execute_input":"2023-04-02T16:51:55.958876Z","iopub.status.idle":"2023-04-02T16:51:56.007622Z","shell.execute_reply.started":"2023-04-02T16:51:55.958820Z","shell.execute_reply":"2023-04-02T16:51:56.006758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a config class to store all the configurations\nclass config:\n    \n    # Image and Tabular data paths\n    DIRECTORY_PATH = \"../input/jpeg-melanoma-384x384/\"\n    TRAINING_SAMPLES_FOLDER = DIRECTORY_PATH + \"train/\"\n    TESTING_SAMPLES_FOLDER = DIRECTORY_PATH + \"test/\"\n    TRAIN_FULL_DATA = DIRECTORY_PATH + \"train.csv\"\n    TEST_FULL_DATA = DIRECTORY_PATH + \"test.csv\"\n    \n    # New directory path for image data\n    WORK_DIRECTORY = \"dataset/\"\n    TRAIN_IMAGES_FOLDER = WORK_DIRECTORY + \"training_set/\"\n    TEST_IMAGES_FOLDER = WORK_DIRECTORY + \"test_set/\"\n    VALIDATION_IMAGES_FOLDER = WORK_DIRECTORY + \"validation_set/\"\n    \n    # Input parameters for data preprocessing\n    TARGET_NAME = \"target\"\n    TRAIN_SIZE = 0.80\n    VALIDATION_SIZE = 0.10\n    TEST_SIZE = 0.10\n    SEED = 42\n    \n    # Tensorflow settings for model training\n    IMAGE_HEIGHT = 299\n    IMAGE_WIDTH = 299\n    NO_CHANNELS = 3\n    BATCH_SIZE = 64\n    EPOCHS = 20\n    DROPOUT = 0.5\n    LEARNING_RATE = 0.01\n    PATIENCE = 5","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:51:56.009333Z","iopub.execute_input":"2023-04-02T16:51:56.009601Z","iopub.status.idle":"2023-04-02T16:51:56.018009Z","shell.execute_reply.started":"2023-04-02T16:51:56.009566Z","shell.execute_reply":"2023-04-02T16:51:56.016828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the files along with the number of samples\nprint(os.listdir(config.DIRECTORY_PATH))\nprint(len(os.listdir(config.TRAINING_SAMPLES_FOLDER)), \"Training Samples\")\nprint(len(os.listdir(config.TESTING_SAMPLES_FOLDER)), \"Testing Samples\")","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:51:56.020106Z","iopub.execute_input":"2023-04-02T16:51:56.020385Z","iopub.status.idle":"2023-04-02T16:51:57.808325Z","shell.execute_reply.started":"2023-04-02T16:51:56.020349Z","shell.execute_reply":"2023-04-02T16:51:57.807489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating folders for training and validation data\ndataset_home = \"./dataset/\"\nsubdirs = [\"training_set/\", \"test_set/\", \"validation_set/\"]\nfor subdir in subdirs:\n    labeldirs = [\"benign\", \"malignant\"]\n    for labeldir in labeldirs:\n        newdir = dataset_home + subdir + labeldir\n        os.makedirs(newdir, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:51:57.809892Z","iopub.execute_input":"2023-04-02T16:51:57.810315Z","iopub.status.idle":"2023-04-02T16:51:57.820881Z","shell.execute_reply.started":"2023-04-02T16:51:57.810279Z","shell.execute_reply":"2023-04-02T16:51:57.819823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(os.listdir(\"./dataset\"))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:51:57.829637Z","iopub.execute_input":"2023-04-02T16:51:57.830391Z","iopub.status.idle":"2023-04-02T16:51:58.042654Z","shell.execute_reply.started":"2023-04-02T16:51:57.830351Z","shell.execute_reply":"2023-04-02T16:51:58.041818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Splitting the dataset into train, test and validation set\n\ntest_examples = train_examples = validation_examples = 0\nseed(config.SEED)\n\nfor record in open(config.TRAIN_FULL_DATA).readlines()[1:]:\n    split_record = record.split(\",\")\n    image_name = split_record[0]\n    target = split_record[7]\n    \n    random_num = random.random()\n    \n    if random_num < config.TRAIN_SIZE:\n        destination = config.TRAIN_IMAGES_FOLDER\n        train_examples += 1\n        \n    elif random_num < 0.9:\n        destination = config.VALIDATION_IMAGES_FOLDER\n        validation_examples += 1\n        \n    else:\n        destination = config.TEST_IMAGES_FOLDER\n        test_examples += 1\n        \n    if target == \"0\":\n        shutil.copy(\n            config.TRAINING_SAMPLES_FOLDER + image_name + \".jpg\",\n            destination + \"benign/\" + image_name + \".jpg\"\n        )\n    \n    elif target == \"1\":\n        shutil.copy(\n            config.TRAINING_SAMPLES_FOLDER + image_name + \".jpg\",\n            destination + \"malignant/\" + image_name + \".jpg\"\n        )\n\nprint(f\"Number of training examples: {train_examples}\")\nprint(f\"Number of test examples: {test_examples}\")\nprint(f\"Number of validation examples: {validation_examples}\")","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:51:58.046033Z","iopub.execute_input":"2023-04-02T16:51:58.047087Z","iopub.status.idle":"2023-04-02T16:58:38.220388Z","shell.execute_reply.started":"2023-04-02T16:51:58.047051Z","shell.execute_reply":"2023-04-02T16:58:38.219557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preparing the data and performing Data Augmentation\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    shear_range=0.2,\n    zoom_range=(0.95, 0.95),\n    rotation_range=15,\n    horizontal_flip=True,\n    vertical_flip=True,\n    data_format=\"channels_last\",\n    dtype=tf.float32\n)\n\nvalidation_datagen = ImageDataGenerator(\n    rescale=1./255,\n    dtype=tf.float32\n)\n\ntrain_generator = train_datagen.flow_from_directory(\n    directory=config.TRAIN_IMAGES_FOLDER,\n    target_size=(config.IMAGE_HEIGHT, config.IMAGE_WIDTH),\n    color_mode=\"rgb\",\n    batch_size=config.BATCH_SIZE,\n    class_mode=\"binary\",\n    shuffle=True\n)\n\nvalidation_generator = validation_datagen.flow_from_directory(\n    directory=config.VALIDATION_IMAGES_FOLDER,\n    target_size=(config.IMAGE_HEIGHT, config.IMAGE_WIDTH),\n    color_mode=\"rgb\",\n    batch_size=config.BATCH_SIZE,\n    class_mode=\"binary\",\n    shuffle=True\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:58:38.221827Z","iopub.execute_input":"2023-04-02T16:58:38.222289Z","iopub.status.idle":"2023-04-02T16:58:39.085577Z","shell.execute_reply.started":"2023-04-02T16:58:38.222250Z","shell.execute_reply":"2023-04-02T16:58:39.084882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating an exponential learning rate scheduler\n# def exponential_decay(lr0, s):\n#    def exponential_decay_fn(epoch):\n#        return lr0 * 0.1**(epoch / s)\n#    return exponential_decay_fn\n\n# exponential_decay_fn = exponential_decay(lr0=config.LEARNING_RATE, s=20)\n# lr_scheduler = keras.callbacks.LearningRateScheduler(exponential_decay_fn)\n\n#Creating a step decay learning rate scheduler\ndef step_decay(epoch):\n   initial_lrate = 0.1\n   drop = 0.5\n   epochs_drop = 10.0\n   lrate = initial_lrate * math.pow(drop,  \n           math.floor((1+epoch)/epochs_drop))\n   return lrate\nlrate = keras.callbacks.LearningRateScheduler(step_decay)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:58:39.086942Z","iopub.execute_input":"2023-04-02T16:58:39.087251Z","iopub.status.idle":"2023-04-02T16:58:39.095158Z","shell.execute_reply.started":"2023-04-02T16:58:39.087215Z","shell.execute_reply":"2023-04-02T16:58:39.094369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using callbacks to save model parameters and perform early stopping\ncheckpoint_cb = keras.callbacks.ModelCheckpoint(\"melanoma.h5\", save_best_only=True)\nearly_stopping_cb = keras.callbacks.EarlyStopping(patience=config.PATIENCE, restore_best_weights=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:58:39.096806Z","iopub.execute_input":"2023-04-02T16:58:39.097198Z","iopub.status.idle":"2023-04-02T16:58:39.106261Z","shell.execute_reply.started":"2023-04-02T16:58:39.097159Z","shell.execute_reply":"2023-04-02T16:58:39.105343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculating the different step size for the model while training\nSTEP_SIZE_TRAIN = train_generator.n // train_generator.batch_size\nSTEP_SIZE_VALIDATION = validation_generator.n // validation_generator.batch_size","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:58:39.108062Z","iopub.execute_input":"2023-04-02T16:58:39.108350Z","iopub.status.idle":"2023-04-02T16:58:39.117906Z","shell.execute_reply.started":"2023-04-02T16:58:39.108315Z","shell.execute_reply":"2023-04-02T16:58:39.117170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Metrics to use for compiling the model\nMETRICS = [keras.metrics.AUC(name=\"auc\")]","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:58:39.119407Z","iopub.execute_input":"2023-04-02T16:58:39.120131Z","iopub.status.idle":"2023-04-02T16:58:44.706864Z","shell.execute_reply.started":"2023-04-02T16:58:39.120094Z","shell.execute_reply":"2023-04-02T16:58:44.706024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculating the number of benign/malignant images\ntotal_images = train_examples + validation_examples\nbenign_images = len(os.listdir(config.TRAIN_IMAGES_FOLDER + \"benign\")) + len(os.listdir(config.VALIDATION_IMAGES_FOLDER + \"benign\"))\nmalignant_images = len(os.listdir(config.TRAIN_IMAGES_FOLDER + \"malignant\")) + len(os.listdir(config.VALIDATION_IMAGES_FOLDER + \"malignant\"))","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:58:44.708243Z","iopub.execute_input":"2023-04-02T16:58:44.708498Z","iopub.status.idle":"2023-04-02T16:58:44.731502Z","shell.execute_reply.started":"2023-04-02T16:58:44.708463Z","shell.execute_reply":"2023-04-02T16:58:44.730813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = keras.models.Sequential([\n    keras.layers.Conv2D(filters=96, kernel_size=(11,11), strides=(4,4), activation='relu', input_shape=(config.IMAGE_HEIGHT, config.IMAGE_WIDTH, config.NO_CHANNELS)),\n    keras.layers.BatchNormalization(),\n    keras.layers.MaxPool2D(pool_size=(3,3), strides=(2,2)),\n    keras.layers.Conv2D(filters=256, kernel_size=(5,5), strides=(1,1), activation='relu', padding=\"same\"),\n    keras.layers.BatchNormalization(),\n    keras.layers.MaxPool2D(pool_size=(3,3), strides=(2,2)),\n    keras.layers.Conv2D(filters=384, kernel_size=(3,3), strides=(1,1), activation='relu', padding=\"same\"),\n    keras.layers.BatchNormalization(),\n    keras.layers.Conv2D(filters=384, kernel_size=(3,3), strides=(1,1), activation='relu', padding=\"same\"),\n    keras.layers.BatchNormalization(),\n    keras.layers.Conv2D(filters=256, kernel_size=(3,3), strides=(1,1), activation='relu', padding=\"same\"),\n    keras.layers.BatchNormalization(),\n    keras.layers.MaxPool2D(pool_size=(3,3), strides=(2,2)),\n    keras.layers.Flatten(),\n    keras.layers.Dense(4096, activation='relu'),\n    keras.layers.Dropout(0.5),\n    keras.layers.Dense(4096, activation='relu'),\n    keras.layers.Dropout(0.5),\n    keras.layers.Dense(1, activation='sigmoid')\n])\n","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:58:44.732908Z","iopub.execute_input":"2023-04-02T16:58:44.733192Z","iopub.status.idle":"2023-04-02T16:58:45.013447Z","shell.execute_reply.started":"2023-04-02T16:58:44.733157Z","shell.execute_reply":"2023-04-02T16:58:45.012735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compiling the transfer learning model\nmodel.compile(optimizer=keras.optimizers.Adam(), \n              loss=keras.losses.BinaryCrossentropy(),\n              metrics=METRICS\n             )","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:58:45.017680Z","iopub.execute_input":"2023-04-02T16:58:45.020058Z","iopub.status.idle":"2023-04-02T16:58:45.036334Z","shell.execute_reply.started":"2023-04-02T16:58:45.020018Z","shell.execute_reply":"2023-04-02T16:58:45.035603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_generator, epochs=config.EPOCHS,\n          validation_data=validation_generator,\n          validation_freq=1)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T16:58:45.040639Z","iopub.execute_input":"2023-04-02T16:58:45.042901Z","iopub.status.idle":"2023-04-02T20:18:24.049738Z","shell.execute_reply.started":"2023-04-02T16:58:45.042865Z","shell.execute_reply":"2023-04-02T20:18:24.048976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.evaluate(validation_generator, steps=STEP_SIZE_VALIDATION)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:18:24.052708Z","iopub.execute_input":"2023-04-02T20:18:24.053437Z","iopub.status.idle":"2023-04-02T20:18:42.194743Z","shell.execute_reply.started":"2023-04-02T20:18:24.053402Z","shell.execute_reply":"2023-04-02T20:18:42.193925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a test generator for test data\ntest_datagen = ImageDataGenerator(\n    rescale=1./255,\n    dtype=tf.float32\n)\n\ntest_generator = test_datagen.flow_from_directory(\n    directory=config.TEST_IMAGES_FOLDER,\n    target_size=(config.IMAGE_HEIGHT, config.IMAGE_WIDTH),\n    color_mode=\"rgb\",\n    batch_size=config.BATCH_SIZE,\n    class_mode=\"binary\",\n    shuffle=False\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:18:42.199163Z","iopub.execute_input":"2023-04-02T20:18:42.201150Z","iopub.status.idle":"2023-04-02T20:18:42.415398Z","shell.execute_reply.started":"2023-04-02T20:18:42.201096Z","shell.execute_reply":"2023-04-02T20:18:42.414627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TEST = test_generator.n // test_generator.batch_size","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:18:42.416628Z","iopub.execute_input":"2023-04-02T20:18:42.417497Z","iopub.status.idle":"2023-04-02T20:18:42.422055Z","shell.execute_reply.started":"2023-04-02T20:18:42.417453Z","shell.execute_reply":"2023-04-02T20:18:42.420991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting the actual classes of the test dataset\ny_test = np.array([])\nnum_batches = 0\nfor _, y in test_generator:\n    y_test = np.append(y_test, y)\n    num_batches += 1\n    if num_batches == math.ceil(test_examples / config.BATCH_SIZE):\n        break\ny_test","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:18:42.423542Z","iopub.execute_input":"2023-04-02T20:18:42.424198Z","iopub.status.idle":"2023-04-02T20:18:53.684558Z","shell.execute_reply.started":"2023-04-02T20:18:42.424160Z","shell.execute_reply":"2023-04-02T20:18:53.683736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predicting output on the test dataset\ny_pred = model.predict(test_generator)\ny_pred","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:18:53.685839Z","iopub.execute_input":"2023-04-02T20:18:53.686261Z","iopub.status.idle":"2023-04-02T20:19:11.525570Z","shell.execute_reply.started":"2023-04-02T20:18:53.686219Z","shell.execute_reply":"2023-04-02T20:19:11.524802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Computing the TPR and FPR values from the roc curve\nfrom sklearn.metrics import roc_curve\nfpr, tpr, thresholds = roc_curve(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:19:11.526758Z","iopub.execute_input":"2023-04-02T20:19:11.527053Z","iopub.status.idle":"2023-04-02T20:19:11.830716Z","shell.execute_reply.started":"2023-04-02T20:19:11.527012Z","shell.execute_reply":"2023-04-02T20:19:11.829986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting the ROC curve\ndef plot_roc_curve (fpr, tpr, label = None):\n    plt.plot(fpr, tpr, linewidth = 2, label = label)\n    plt.plot([0,1], [0,1], 'k--') # Dashed diagonal\n    plt.xlabel(\"False Positive Rate\")\n    plt.ylabel(\"True Positive Rate (Recall)\")\n    plt.grid()\n    \nplot_roc_curve(fpr, tpr)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:19:11.831842Z","iopub.execute_input":"2023-04-02T20:19:11.832137Z","iopub.status.idle":"2023-04-02T20:19:12.085490Z","shell.execute_reply.started":"2023-04-02T20:19:11.832103Z","shell.execute_reply":"2023-04-02T20:19:12.084673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluating the model on test dataset\nmodel.evaluate(test_generator, steps=STEP_SIZE_TEST)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:19:12.086645Z","iopub.execute_input":"2023-04-02T20:19:12.086952Z","iopub.status.idle":"2023-04-02T20:19:29.885674Z","shell.execute_reply.started":"2023-04-02T20:19:12.086914Z","shell.execute_reply":"2023-04-02T20:19:29.884833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving the best model after training\nmodel.save(\"final_melanoma_model_20ep.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-04-02T20:25:04.365878Z","iopub.execute_input":"2023-04-02T20:25:04.366585Z","iopub.status.idle":"2023-04-02T20:25:07.919047Z","shell.execute_reply.started":"2023-04-02T20:25:04.366551Z","shell.execute_reply":"2023-04-02T20:25:07.918256Z"},"trusted":true},"execution_count":null,"outputs":[]}]}