{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install livelossplot --quiet","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:49:33.722223Z","iopub.execute_input":"2023-02-17T13:49:33.723113Z","iopub.status.idle":"2023-02-17T13:49:41.173816Z","shell.execute_reply.started":"2023-02-17T13:49:33.723066Z","shell.execute_reply":"2023-02-17T13:49:41.172874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing the libraries\nimport os\nimport shutil\nimport glob\nfrom tqdm.notebook import tqdm\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nimport cv2\nfrom PIL import Image\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport random\nfrom random import seed\nfrom livelossplot import PlotLossesKeras\nimport math","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:49:41.177609Z","iopub.execute_input":"2023-02-17T13:49:41.177847Z","iopub.status.idle":"2023-02-17T13:49:41.185706Z","shell.execute_reply.started":"2023-02-17T13:49:41.177818Z","shell.execute_reply":"2023-02-17T13:49:41.184994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:49:41.189029Z","iopub.execute_input":"2023-02-17T13:49:41.189230Z","iopub.status.idle":"2023-02-17T13:49:41.201419Z","shell.execute_reply.started":"2023-02-17T13:49:41.189205Z","shell.execute_reply":"2023-02-17T13:49:41.200776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing the Training Dataset\ndf_train = pd.read_csv(\"../input/jpeg-melanoma-384x384/train.csv\")\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:49:41.204691Z","iopub.execute_input":"2023-02-17T13:49:41.204894Z","iopub.status.idle":"2023-02-17T13:49:41.280743Z","shell.execute_reply.started":"2023-02-17T13:49:41.204871Z","shell.execute_reply":"2023-02-17T13:49:41.280053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:49:41.282005Z","iopub.execute_input":"2023-02-17T13:49:41.282249Z","iopub.status.idle":"2023-02-17T13:49:41.304080Z","shell.execute_reply.started":"2023-02-17T13:49:41.282216Z","shell.execute_reply":"2023-02-17T13:49:41.303192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing the Test Dataset\ndf_test = pd.read_csv(\"../input/jpeg-melanoma-384x384/test.csv\")\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:49:41.305662Z","iopub.execute_input":"2023-02-17T13:49:41.305923Z","iopub.status.idle":"2023-02-17T13:49:41.332310Z","shell.execute_reply.started":"2023-02-17T13:49:41.305890Z","shell.execute_reply":"2023-02-17T13:49:41.331487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a config class to store all the configurations\nclass config:\n    \n    # Image and Tabular data paths\n    DIRECTORY_PATH = \"../input/jpeg-melanoma-384x384/\"\n    TRAINING_SAMPLES_FOLDER = DIRECTORY_PATH + \"train/\"\n    TESTING_SAMPLES_FOLDER = DIRECTORY_PATH + \"test/\"\n    TRAIN_FULL_DATA = DIRECTORY_PATH + \"train.csv\"\n    TEST_FULL_DATA = DIRECTORY_PATH + \"test.csv\"\n    \n    # New directory path for image data\n    WORK_DIRECTORY = \"dataset/\"\n    TRAIN_IMAGES_FOLDER = WORK_DIRECTORY + \"training_set/\"\n    TEST_IMAGES_FOLDER = WORK_DIRECTORY + \"test_set/\"\n    VALIDATION_IMAGES_FOLDER = WORK_DIRECTORY + \"validation_set/\"\n    \n    # Input parameters for data preprocessing\n    TARGET_NAME = \"target\"\n    TRAIN_SIZE = 0.80\n    VALIDATION_SIZE = 0.10\n    TEST_SIZE = 0.10\n    SEED = 42\n    \n    # Tensorflow settings for model training\n    IMAGE_HEIGHT = 299\n    IMAGE_WIDTH = 299\n    NO_CHANNELS = 3\n    BATCH_SIZE = 64\n    EPOCHS = 30\n    DROPOUT = 0.5\n    LEARNING_RATE = 0.01\n    PATIENCE = 5","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:49:41.333809Z","iopub.execute_input":"2023-02-17T13:49:41.334090Z","iopub.status.idle":"2023-02-17T13:49:41.340871Z","shell.execute_reply.started":"2023-02-17T13:49:41.334052Z","shell.execute_reply":"2023-02-17T13:49:41.340011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the files along with the number of samples\nprint(os.listdir(config.DIRECTORY_PATH))\nprint(len(os.listdir(config.TRAINING_SAMPLES_FOLDER)), \"Training Samples\")\nprint(len(os.listdir(config.TESTING_SAMPLES_FOLDER)), \"Testing Samples\")","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:49:41.342324Z","iopub.execute_input":"2023-02-17T13:49:41.342859Z","iopub.status.idle":"2023-02-17T13:49:41.381542Z","shell.execute_reply.started":"2023-02-17T13:49:41.342820Z","shell.execute_reply":"2023-02-17T13:49:41.380878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating folders for training and validation data\ndataset_home = \"./dataset/\"\nsubdirs = [\"training_set/\", \"test_set/\", \"validation_set/\"]\nfor subdir in subdirs:\n    labeldirs = [\"benign\", \"malignant\"]\n    for labeldir in labeldirs:\n        newdir = dataset_home + subdir + labeldir\n        os.makedirs(newdir, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:49:41.382716Z","iopub.execute_input":"2023-02-17T13:49:41.382979Z","iopub.status.idle":"2023-02-17T13:49:41.388560Z","shell.execute_reply.started":"2023-02-17T13:49:41.382941Z","shell.execute_reply":"2023-02-17T13:49:41.387724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(os.listdir(\"./dataset\"))","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:49:41.391002Z","iopub.execute_input":"2023-02-17T13:49:41.391701Z","iopub.status.idle":"2023-02-17T13:49:41.399183Z","shell.execute_reply.started":"2023-02-17T13:49:41.391665Z","shell.execute_reply":"2023-02-17T13:49:41.398352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Splitting the dataset into train, test and validation set\n\ntest_examples = train_examples = validation_examples = 0\nseed(config.SEED)\n\nfor record in open(config.TRAIN_FULL_DATA).readlines()[1:]:\n    split_record = record.split(\",\")\n    image_name = split_record[0]\n    target = split_record[7]\n    \n    random_num = random.random()\n    \n    if random_num < config.TRAIN_SIZE:\n        destination = config.TRAIN_IMAGES_FOLDER\n        train_examples += 1\n        \n    elif random_num < 0.9:\n        destination = config.VALIDATION_IMAGES_FOLDER\n        validation_examples += 1\n        \n    else:\n        destination = config.TEST_IMAGES_FOLDER\n        test_examples += 1\n        \n    if target == \"0\":\n        shutil.copy(\n            config.TRAINING_SAMPLES_FOLDER + image_name + \".jpg\",\n            destination + \"benign/\" + image_name + \".jpg\"\n        )\n    \n    elif target == \"1\":\n        shutil.copy(\n            config.TRAINING_SAMPLES_FOLDER + image_name + \".jpg\",\n            destination + \"malignant/\" + image_name + \".jpg\"\n        )\n\nprint(f\"Number of training examples: {train_examples}\")\nprint(f\"Number of test examples: {test_examples}\")\nprint(f\"Number of validation examples: {validation_examples}\")","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:49:41.400382Z","iopub.execute_input":"2023-02-17T13:49:41.401018Z","iopub.status.idle":"2023-02-17T13:50:01.680077Z","shell.execute_reply.started":"2023-02-17T13:49:41.400981Z","shell.execute_reply":"2023-02-17T13:50:01.678063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preparing the data and performing Data Augmentation\ntrain_datagen = ImageDataGenerator(\n    #rescale=1./255,\n    shear_range=0.2,\n    zoom_range=(0.95, 0.95),\n    rotation_range=15,\n    horizontal_flip=True,\n    vertical_flip=True,\n    data_format=\"channels_last\",\n    dtype=tf.float32\n)\n\nvalidation_datagen = ImageDataGenerator(\n    #rescale=1./255,\n    dtype=tf.float32\n)\n\ntrain_generator = train_datagen.flow_from_directory(\n    directory=config.TRAIN_IMAGES_FOLDER,\n    target_size=(config.IMAGE_HEIGHT, config.IMAGE_WIDTH),\n    color_mode=\"rgb\",\n    batch_size=config.BATCH_SIZE,\n    class_mode=\"binary\",\n    shuffle=True\n)\n\nvalidation_generator = validation_datagen.flow_from_directory(\n    directory=config.VALIDATION_IMAGES_FOLDER,\n    target_size=(config.IMAGE_HEIGHT, config.IMAGE_WIDTH),\n    color_mode=\"rgb\",\n    batch_size=config.BATCH_SIZE,\n    class_mode=\"binary\",\n    shuffle=True\n)","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:50:01.680929Z","iopub.status.idle":"2023-02-17T13:50:01.681226Z","shell.execute_reply.started":"2023-02-17T13:50:01.681065Z","shell.execute_reply":"2023-02-17T13:50:01.681087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Metrics to use for compiling the model\nMETRICS = [keras.metrics.AUC(name=\"auc\")]","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:50:01.682353Z","iopub.status.idle":"2023-02-17T13:50:01.682859Z","shell.execute_reply.started":"2023-02-17T13:50:01.682650Z","shell.execute_reply":"2023-02-17T13:50:01.682673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculating the different step size for the model while training\nSTEP_SIZE_TRAIN = train_generator.n // train_generator.batch_size\nSTEP_SIZE_VALIDATION = validation_generator.n // validation_generator.batch_size","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:50:01.684257Z","iopub.status.idle":"2023-02-17T13:50:01.685058Z","shell.execute_reply.started":"2023-02-17T13:50:01.684826Z","shell.execute_reply":"2023-02-17T13:50:01.684850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculating the number of benign/malignant images\ntotal_images = train_examples + validation_examples\nbenign_images = len(os.listdir(config.TRAIN_IMAGES_FOLDER + \"benign\")) + len(os.listdir(config.VALIDATION_IMAGES_FOLDER + \"benign\"))\nmalignant_images = len(os.listdir(config.TRAIN_IMAGES_FOLDER + \"malignant\")) + len(os.listdir(config.VALIDATION_IMAGES_FOLDER + \"malignant\"))","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:50:01.686272Z","iopub.status.idle":"2023-02-17T13:50:01.686995Z","shell.execute_reply.started":"2023-02-17T13:50:01.686756Z","shell.execute_reply":"2023-02-17T13:50:01.686781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = tf.keras.applications.efficientnet.EfficientNetB0(include_top= False, classes =2)\n\n# 2. Freeze the base model(so the underlying pre-trained patterns aren't updated during training)\nbase_model.trainable = True\n\n# 3. Create Inputs into our model\ninputs = tf.keras.layers.Input(shape=(config.IMAGE_HEIGHT, config.IMAGE_WIDTH, config.NO_CHANNELS),name='input_layer')\n\n# 4. If using ResNet50V2, add this to speed up convergence, remove for EfficientNet\n# x = tf.keras.layers.experimental.preprocessing.Rescaling(1./255)(inputs)\n\n# 5. Pass the inputs to the base_model\nx = base_model(inputs)\nprint(f\"shape after passing inputs throught base modelL{x.shape}\")\n\n# 6. Average pool the outputs of the base model(aggreate all the most important information, reduce number of computation)\nx = tf.keras.layers.GlobalAveragePooling2D(name = \"global_average_pooling\") (x)\nprint(f\"Shape after GlobalAveragingPooling2D: {x.shape}\")\n\n# 7. Create output activation layer\noutputs = tf.keras.layers.Dense(1, activation ='sigmoid',name = 'output_layer')(x)\n\n# 8. Combine the inputs and outputs into a model\nmodel_0 = tf.keras.Model(inputs, outputs)\n\n#9. Compile the model\nmodel_0.compile(optimizer='sgd', \n              loss=keras.losses.BinaryCrossentropy(),\n              metrics=METRICS\n             )\n\n","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:50:01.688173Z","iopub.status.idle":"2023-02-17T13:50:01.689018Z","shell.execute_reply.started":"2023-02-17T13:50:01.688783Z","shell.execute_reply":"2023-02-17T13:50:01.688808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compiling the transfer learning model\n#model.compile(optimizer=keras.optimizers.Adam(), \n             # loss=keras.losses.BinaryCrossentropy(),\n              #metrics=METRICS\n             #)","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:50:01.690216Z","iopub.status.idle":"2023-02-17T13:50:01.690943Z","shell.execute_reply.started":"2023-02-17T13:50:01.690704Z","shell.execute_reply":"2023-02-17T13:50:01.690733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model_0.fit(train_generator, epochs=config.EPOCHS,\n          validation_data=validation_generator,\n          validation_freq=1)","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:50:01.692167Z","iopub.status.idle":"2023-02-17T13:50:01.692971Z","shell.execute_reply.started":"2023-02-17T13:50:01.692740Z","shell.execute_reply":"2023-02-17T13:50:01.692769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluating the model on Validation Dataset\nmodel_0.evaluate(validation_generator, steps=STEP_SIZE_VALIDATION)","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:19:46.236844Z","iopub.execute_input":"2023-02-17T13:19:46.237123Z","iopub.status.idle":"2023-02-17T13:20:05.195030Z","shell.execute_reply.started":"2023-02-17T13:19:46.237087Z","shell.execute_reply":"2023-02-17T13:20:05.194242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a test generator for test data\ntest_datagen = ImageDataGenerator(\n    rescale=1./255,\n    dtype=tf.float32\n)\n\ntest_generator = test_datagen.flow_from_directory(\n    directory=config.TEST_IMAGES_FOLDER,\n    target_size=(config.IMAGE_HEIGHT, config.IMAGE_WIDTH),\n    color_mode=\"rgb\",\n    batch_size=config.BATCH_SIZE,\n    class_mode=\"binary\",\n    shuffle=False\n)","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:23:30.687410Z","iopub.execute_input":"2023-02-17T13:23:30.687694Z","iopub.status.idle":"2023-02-17T13:23:30.799229Z","shell.execute_reply.started":"2023-02-17T13:23:30.687664Z","shell.execute_reply":"2023-02-17T13:23:30.798369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TEST = test_generator.n // test_generator.batch_size","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:23:33.011485Z","iopub.execute_input":"2023-02-17T13:23:33.012035Z","iopub.status.idle":"2023-02-17T13:23:33.016343Z","shell.execute_reply.started":"2023-02-17T13:23:33.011996Z","shell.execute_reply":"2023-02-17T13:23:33.015192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting the actual classes of the test dataset\ny_test = np.array([])\nnum_batches = 0\nfor _, y in test_generator:\n    y_test = np.append(y_test, y)\n    num_batches += 1\n    if num_batches == math.ceil(test_examples / config.BATCH_SIZE):\n        break\ny_test","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:23:37.743358Z","iopub.execute_input":"2023-02-17T13:23:37.744093Z","iopub.status.idle":"2023-02-17T13:23:49.475539Z","shell.execute_reply.started":"2023-02-17T13:23:37.744060Z","shell.execute_reply":"2023-02-17T13:23:49.474759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predicting output on the test dataset\ny_pred = model_0.predict(test_generator)\ny_pred","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:23:49.477241Z","iopub.execute_input":"2023-02-17T13:23:49.477513Z","iopub.status.idle":"2023-02-17T13:24:08.289637Z","shell.execute_reply.started":"2023-02-17T13:23:49.477479Z","shell.execute_reply":"2023-02-17T13:24:08.288872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Computing the TPR and FPR values from the roc curve\nfrom sklearn.metrics import roc_curve\nfpr, tpr, thresholds = roc_curve(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:25:42.227152Z","iopub.execute_input":"2023-02-17T13:25:42.227733Z","iopub.status.idle":"2023-02-17T13:25:42.528589Z","shell.execute_reply.started":"2023-02-17T13:25:42.227693Z","shell.execute_reply":"2023-02-17T13:25:42.527853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting the ROC curve\ndef plot_roc_curve (fpr, tpr, label = None):\n    plt.plot(fpr, tpr, linewidth = 2, label = label)\n    plt.plot([0,1], [0,1], 'k--') # Dashed diagonal\n    plt.xlabel(\"False Positive Rate\")\n    plt.ylabel(\"True Positive Rate (Recall)\")\n    plt.grid()\n    \nplot_roc_curve(fpr, tpr)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:25:47.012045Z","iopub.execute_input":"2023-02-17T13:25:47.012327Z","iopub.status.idle":"2023-02-17T13:25:47.262999Z","shell.execute_reply.started":"2023-02-17T13:25:47.012274Z","shell.execute_reply":"2023-02-17T13:25:47.262199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluating the model on test dataset\nmodel_0.evaluate(test_generator, steps=STEP_SIZE_TEST)","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:26:06.927164Z","iopub.execute_input":"2023-02-17T13:26:06.927879Z","iopub.status.idle":"2023-02-17T13:26:26.535179Z","shell.execute_reply.started":"2023-02-17T13:26:06.927842Z","shell.execute_reply":"2023-02-17T13:26:26.534493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving the best model after training\nmodel_0.save(\"final_melanoma_model_0.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-02-17T13:26:52.619121Z","iopub.execute_input":"2023-02-17T13:26:52.619411Z","iopub.status.idle":"2023-02-17T13:26:52.939335Z","shell.execute_reply.started":"2023-02-17T13:26:52.619380Z","shell.execute_reply":"2023-02-17T13:26:52.937833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}