{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install livelossplot --quiet","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:43:57.498087Z","iopub.execute_input":"2022-11-04T16:43:57.498351Z","iopub.status.idle":"2022-11-04T16:44:08.088517Z","shell.execute_reply.started":"2022-11-04T16:43:57.498320Z","shell.execute_reply":"2022-11-04T16:44:08.087643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing the libraries\nimport os\nimport shutil\nimport glob\nfrom tqdm.notebook import tqdm\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nimport cv2\nfrom PIL import Image\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport random\nfrom random import seed\nfrom livelossplot import PlotLossesKeras\nimport math","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:44:08.090815Z","iopub.execute_input":"2022-11-04T16:44:08.091108Z","iopub.status.idle":"2022-11-04T16:44:19.714310Z","shell.execute_reply.started":"2022-11-04T16:44:08.091071Z","shell.execute_reply":"2022-11-04T16:44:19.713438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:44:23.434293Z","iopub.execute_input":"2022-11-04T16:44:23.434848Z","iopub.status.idle":"2022-11-04T16:44:23.438192Z","shell.execute_reply.started":"2022-11-04T16:44:23.434808Z","shell.execute_reply":"2022-11-04T16:44:23.437505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing the Training Dataset\ndf_train = pd.read_csv(\"../input/jpeg-melanoma-384x384/train.csv\")\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:44:24.428942Z","iopub.execute_input":"2022-11-04T16:44:24.429209Z","iopub.status.idle":"2022-11-04T16:44:24.566656Z","shell.execute_reply.started":"2022-11-04T16:44:24.429181Z","shell.execute_reply":"2022-11-04T16:44:24.565810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:44:25.289684Z","iopub.execute_input":"2022-11-04T16:44:25.289949Z","iopub.status.idle":"2022-11-04T16:44:25.332042Z","shell.execute_reply.started":"2022-11-04T16:44:25.289920Z","shell.execute_reply":"2022-11-04T16:44:25.331142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing the Test Dataset\ndf_test = pd.read_csv(\"../input/jpeg-melanoma-384x384/test.csv\")\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:44:26.146180Z","iopub.execute_input":"2022-11-04T16:44:26.146525Z","iopub.status.idle":"2022-11-04T16:44:26.191174Z","shell.execute_reply.started":"2022-11-04T16:44:26.146475Z","shell.execute_reply":"2022-11-04T16:44:26.190398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a config class to store all the configurations\nclass config:\n    \n    # Image and Tabular data paths\n    DIRECTORY_PATH = \"../input/jpeg-melanoma-384x384/\"\n    TRAINING_SAMPLES_FOLDER = DIRECTORY_PATH + \"train/\"\n    TESTING_SAMPLES_FOLDER = DIRECTORY_PATH + \"test/\"\n    TRAIN_FULL_DATA = DIRECTORY_PATH + \"train.csv\"\n    TEST_FULL_DATA = DIRECTORY_PATH + \"test.csv\"\n    \n    # New directory path for image data\n    WORK_DIRECTORY = \"dataset/\"\n    TRAIN_IMAGES_FOLDER = WORK_DIRECTORY + \"training_set/\"\n    TEST_IMAGES_FOLDER = WORK_DIRECTORY + \"test_set/\"\n    VALIDATION_IMAGES_FOLDER = WORK_DIRECTORY + \"validation_set/\"\n    \n    # Input parameters for data preprocessing\n    TARGET_NAME = \"target\"\n    TRAIN_SIZE = 0.80\n    VALIDATION_SIZE = 0.10\n    TEST_SIZE = 0.10\n    SEED = 42\n    \n    # Tensorflow settings for model training\n    IMAGE_HEIGHT = 299\n    IMAGE_WIDTH = 299\n    NO_CHANNELS = 3\n    BATCH_SIZE = 64\n    EPOCHS = 20\n    DROPOUT = 0.5\n    LEARNING_RATE = 0.01\n    PATIENCE = 5","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:44:26.620247Z","iopub.execute_input":"2022-11-04T16:44:26.620529Z","iopub.status.idle":"2022-11-04T16:44:26.626916Z","shell.execute_reply.started":"2022-11-04T16:44:26.620498Z","shell.execute_reply":"2022-11-04T16:44:26.626066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the files along with the number of samples\nprint(os.listdir(config.DIRECTORY_PATH))\nprint(len(os.listdir(config.TRAINING_SAMPLES_FOLDER)), \"Training Samples\")\nprint(len(os.listdir(config.TESTING_SAMPLES_FOLDER)), \"Testing Samples\")","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:44:27.185155Z","iopub.execute_input":"2022-11-04T16:44:27.185400Z","iopub.status.idle":"2022-11-04T16:44:28.135175Z","shell.execute_reply.started":"2022-11-04T16:44:27.185372Z","shell.execute_reply":"2022-11-04T16:44:28.134462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating folders for training and validation data\ndataset_home = \"./dataset/\"\nsubdirs = [\"training_set/\", \"test_set/\", \"validation_set/\"]\nfor subdir in subdirs:\n    labeldirs = [\"benign\", \"malignant\"]\n    for labeldir in labeldirs:\n        newdir = dataset_home + subdir + labeldir\n        os.makedirs(newdir, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:44:28.138537Z","iopub.execute_input":"2022-11-04T16:44:28.140137Z","iopub.status.idle":"2022-11-04T16:44:28.146084Z","shell.execute_reply.started":"2022-11-04T16:44:28.140102Z","shell.execute_reply":"2022-11-04T16:44:28.145147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(os.listdir(\"./dataset\"))","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:44:28.149022Z","iopub.execute_input":"2022-11-04T16:44:28.149541Z","iopub.status.idle":"2022-11-04T16:44:28.157468Z","shell.execute_reply.started":"2022-11-04T16:44:28.149504Z","shell.execute_reply":"2022-11-04T16:44:28.156246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Splitting the dataset into train, test and validation set\n\ntest_examples = train_examples = validation_examples = 0\nseed(config.SEED)\n\nfor record in open(config.TRAIN_FULL_DATA).readlines()[1:]:\n    split_record = record.split(\",\")\n    image_name = split_record[0]\n    target = split_record[7]\n    \n    random_num = random.random()\n    \n    if random_num < config.TRAIN_SIZE:\n        destination = config.TRAIN_IMAGES_FOLDER\n        train_examples += 1\n        \n    elif random_num < 0.9:\n        destination = config.VALIDATION_IMAGES_FOLDER\n        validation_examples += 1\n        \n    else:\n        destination = config.TEST_IMAGES_FOLDER\n        test_examples += 1\n        \n    if target == \"0\":\n        shutil.copy(\n            config.TRAINING_SAMPLES_FOLDER + image_name + \".jpg\",\n            destination + \"benign/\" + image_name + \".jpg\"\n        )\n    \n    elif target == \"1\":\n        shutil.copy(\n            config.TRAINING_SAMPLES_FOLDER + image_name + \".jpg\",\n            destination + \"malignant/\" + image_name + \".jpg\"\n        )\n\nprint(f\"Number of training examples: {train_examples}\")\nprint(f\"Number of test examples: {test_examples}\")\nprint(f\"Number of validation examples: {validation_examples}\")","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:44:28.529759Z","iopub.execute_input":"2022-11-04T16:44:28.530022Z","iopub.status.idle":"2022-11-04T16:49:23.006980Z","shell.execute_reply.started":"2022-11-04T16:44:28.529992Z","shell.execute_reply":"2022-11-04T16:49:23.006121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preparing the data and performing Data Augmentation\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    shear_range=0.2,\n    zoom_range=(0.95, 0.95),\n    rotation_range=15,\n    horizontal_flip=True,\n    vertical_flip=True,\n    data_format=\"channels_last\",\n    dtype=tf.float32\n)\n\nvalidation_datagen = ImageDataGenerator(\n    rescale=1./255,\n    dtype=tf.float32\n)\n\ntrain_generator = train_datagen.flow_from_directory(\n    directory=config.TRAIN_IMAGES_FOLDER,\n    target_size=(config.IMAGE_HEIGHT, config.IMAGE_WIDTH),\n    color_mode=\"rgb\",\n    batch_size=config.BATCH_SIZE,\n    class_mode=\"binary\",\n    shuffle=True\n)\n\nvalidation_generator = validation_datagen.flow_from_directory(\n    directory=config.VALIDATION_IMAGES_FOLDER,\n    target_size=(config.IMAGE_HEIGHT, config.IMAGE_WIDTH),\n    color_mode=\"rgb\",\n    batch_size=config.BATCH_SIZE,\n    class_mode=\"binary\",\n    shuffle=True\n)","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:49:23.008818Z","iopub.execute_input":"2022-11-04T16:49:23.009082Z","iopub.status.idle":"2022-11-04T16:49:23.767995Z","shell.execute_reply.started":"2022-11-04T16:49:23.009043Z","shell.execute_reply":"2022-11-04T16:49:23.767239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Metrics to use for compiling the model\nMETRICS = [keras.metrics.AUC(name=\"auc\")]","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:49:23.769443Z","iopub.execute_input":"2022-11-04T16:49:23.769841Z","iopub.status.idle":"2022-11-04T16:49:29.674821Z","shell.execute_reply.started":"2022-11-04T16:49:23.769804Z","shell.execute_reply":"2022-11-04T16:49:29.673296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculating the different step size for the model while training\nSTEP_SIZE_TRAIN = train_generator.n // train_generator.batch_size\nSTEP_SIZE_VALIDATION = validation_generator.n // validation_generator.batch_size","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:49:29.676733Z","iopub.execute_input":"2022-11-04T16:49:29.677080Z","iopub.status.idle":"2022-11-04T16:49:29.681554Z","shell.execute_reply.started":"2022-11-04T16:49:29.677041Z","shell.execute_reply":"2022-11-04T16:49:29.680734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculating the number of benign/malignant images\ntotal_images = train_examples + validation_examples\nbenign_images = len(os.listdir(config.TRAIN_IMAGES_FOLDER + \"benign\")) + len(os.listdir(config.VALIDATION_IMAGES_FOLDER + \"benign\"))\nmalignant_images = len(os.listdir(config.TRAIN_IMAGES_FOLDER + \"malignant\")) + len(os.listdir(config.VALIDATION_IMAGES_FOLDER + \"malignant\"))","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:49:29.683149Z","iopub.execute_input":"2022-11-04T16:49:29.683737Z","iopub.status.idle":"2022-11-04T16:49:29.708429Z","shell.execute_reply.started":"2022-11-04T16:49:29.683703Z","shell.execute_reply":"2022-11-04T16:49:29.707748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = keras.models.Sequential([\n    keras.layers.Conv2D(filters=96, kernel_size=(11,11), strides=(4,4), activation='relu', input_shape=(config.IMAGE_HEIGHT, config.IMAGE_WIDTH, config.NO_CHANNELS)),\n    keras.layers.BatchNormalization(),\n    keras.layers.MaxPool2D(pool_size=(3,3), strides=(2,2)),\n    keras.layers.Conv2D(filters=256, kernel_size=(5,5), strides=(1,1), activation='relu', padding=\"same\"),\n    keras.layers.BatchNormalization(),\n    keras.layers.MaxPool2D(pool_size=(3,3), strides=(2,2)),\n    keras.layers.Conv2D(filters=384, kernel_size=(3,3), strides=(1,1), activation='relu', padding=\"same\"),\n    keras.layers.BatchNormalization(),\n    keras.layers.Conv2D(filters=384, kernel_size=(3,3), strides=(1,1), activation='relu', padding=\"same\"),\n    keras.layers.BatchNormalization(),\n    keras.layers.Conv2D(filters=256, kernel_size=(3,3), strides=(1,1), activation='relu', padding=\"same\"),\n    keras.layers.BatchNormalization(),\n    keras.layers.MaxPool2D(pool_size=(3,3), strides=(2,2)),\n    keras.layers.Flatten(),\n    keras.layers.Dense(4096, activation='relu'),\n    keras.layers.Dropout(0.5),\n    keras.layers.Dense(4096, activation='relu'),\n    keras.layers.Dropout(0.5),\n    keras.layers.Dense(1, activation='sigmoid')\n])","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:49:29.711003Z","iopub.execute_input":"2022-11-04T16:49:29.711193Z","iopub.status.idle":"2022-11-04T16:49:29.883845Z","shell.execute_reply.started":"2022-11-04T16:49:29.711166Z","shell.execute_reply":"2022-11-04T16:49:29.883208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compiling the transfer learning model\nmodel.compile(optimizer=keras.optimizers.Adam(), \n              loss=keras.losses.BinaryCrossentropy(),\n              metrics=METRICS\n             )","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:49:29.885002Z","iopub.execute_input":"2022-11-04T16:49:29.885240Z","iopub.status.idle":"2022-11-04T16:49:29.898246Z","shell.execute_reply.started":"2022-11-04T16:49:29.885208Z","shell.execute_reply":"2022-11-04T16:49:29.897414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_generator, epochs=config.EPOCHS,\n          validation_data=validation_generator,\n          validation_freq=1)","metadata":{"execution":{"iopub.status.busy":"2022-11-04T16:49:29.899524Z","iopub.execute_input":"2022-11-04T16:49:29.900096Z","iopub.status.idle":"2022-11-04T20:10:44.152632Z","shell.execute_reply.started":"2022-11-04T16:49:29.900058Z","shell.execute_reply":"2022-11-04T20:10:44.151700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluating the model on Validation Dataset\nmodel.evaluate(validation_generator, steps=STEP_SIZE_VALIDATION)","metadata":{"execution":{"iopub.status.busy":"2022-11-04T20:15:02.601976Z","iopub.execute_input":"2022-11-04T20:15:02.602256Z","iopub.status.idle":"2022-11-04T20:15:23.435896Z","shell.execute_reply.started":"2022-11-04T20:15:02.602224Z","shell.execute_reply":"2022-11-04T20:15:23.434875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a test generator for test data\ntest_datagen = ImageDataGenerator(\n    rescale=1./255,\n    dtype=tf.float32\n)\n\ntest_generator = test_datagen.flow_from_directory(\n    directory=config.TEST_IMAGES_FOLDER,\n    target_size=(config.IMAGE_HEIGHT, config.IMAGE_WIDTH),\n    color_mode=\"rgb\",\n    batch_size=config.BATCH_SIZE,\n    class_mode=\"binary\",\n    shuffle=False\n)","metadata":{"execution":{"iopub.status.busy":"2022-11-04T20:15:23.437717Z","iopub.execute_input":"2022-11-04T20:15:23.438062Z","iopub.status.idle":"2022-11-04T20:15:23.549995Z","shell.execute_reply.started":"2022-11-04T20:15:23.438025Z","shell.execute_reply":"2022-11-04T20:15:23.548847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TEST = test_generator.n // test_generator.batch_size","metadata":{"execution":{"iopub.status.busy":"2022-11-04T20:15:28.911168Z","iopub.execute_input":"2022-11-04T20:15:28.911469Z","iopub.status.idle":"2022-11-04T20:15:28.915143Z","shell.execute_reply.started":"2022-11-04T20:15:28.911415Z","shell.execute_reply":"2022-11-04T20:15:28.914458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting the actual classes of the test dataset\ny_test = np.array([])\nnum_batches = 0\nfor _, y in test_generator:\n    y_test = np.append(y_test, y)\n    num_batches += 1\n    if num_batches == math.ceil(test_examples / config.BATCH_SIZE):\n        break\ny_test","metadata":{"execution":{"iopub.status.busy":"2022-11-04T20:15:30.826162Z","iopub.execute_input":"2022-11-04T20:15:30.826460Z","iopub.status.idle":"2022-11-04T20:15:42.315966Z","shell.execute_reply.started":"2022-11-04T20:15:30.826405Z","shell.execute_reply":"2022-11-04T20:15:42.315278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predicting output on the test dataset\ny_pred = model.predict(test_generator)\ny_pred","metadata":{"execution":{"iopub.status.busy":"2022-11-04T20:15:47.282020Z","iopub.execute_input":"2022-11-04T20:15:47.282629Z","iopub.status.idle":"2022-11-04T20:16:05.611283Z","shell.execute_reply.started":"2022-11-04T20:15:47.282590Z","shell.execute_reply":"2022-11-04T20:16:05.610397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Computing the TPR and FPR values from the roc curve\nfrom sklearn.metrics import roc_curve\nfpr, tpr, thresholds = roc_curve(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-11-04T20:16:05.613164Z","iopub.execute_input":"2022-11-04T20:16:05.614596Z","iopub.status.idle":"2022-11-04T20:16:05.931399Z","shell.execute_reply.started":"2022-11-04T20:16:05.614542Z","shell.execute_reply":"2022-11-04T20:16:05.930658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting the ROC curve\ndef plot_roc_curve (fpr, tpr, label = None):\n    plt.plot(fpr, tpr, linewidth = 2, label = label)\n    plt.plot([0,1], [0,1], 'k--') # Dashed diagonal\n    plt.xlabel(\"False Positive Rate\")\n    plt.ylabel(\"True Positive Rate (Recall)\")\n    plt.grid()\n    \nplot_roc_curve(fpr, tpr)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-04T20:16:05.933001Z","iopub.execute_input":"2022-11-04T20:16:05.933317Z","iopub.status.idle":"2022-11-04T20:16:06.187352Z","shell.execute_reply.started":"2022-11-04T20:16:05.933276Z","shell.execute_reply":"2022-11-04T20:16:06.186623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluating the model on test dataset\nmodel.evaluate(test_generator, steps=STEP_SIZE_TEST)","metadata":{"execution":{"iopub.status.busy":"2022-11-04T20:16:06.189630Z","iopub.execute_input":"2022-11-04T20:16:06.190126Z","iopub.status.idle":"2022-11-04T20:16:27.033494Z","shell.execute_reply.started":"2022-11-04T20:16:06.190088Z","shell.execute_reply":"2022-11-04T20:16:27.032773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving the best model after training\nmodel.save(\"final_melanoma_model.h5\")","metadata":{"execution":{"iopub.status.busy":"2022-11-04T20:16:27.034744Z","iopub.execute_input":"2022-11-04T20:16:27.035084Z","iopub.status.idle":"2022-11-04T20:16:29.741929Z","shell.execute_reply.started":"2022-11-04T20:16:27.035047Z","shell.execute_reply":"2022-11-04T20:16:29.741117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}