{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div align=\"center\">\n<font size=\"6\"> SIIM-ISIC Melanoma Classification  </font>  \n</div> \n\n\n<div align=\"center\">\n<font size=\"4\"> Isac Lopes Silva - 18.1.8135  </font>  \n</div> ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\nif False:\n    for dirname, _, filenames in os.walk('/kaggle/input'):\n        for filename in filenames:\n            print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:10.048539Z","iopub.execute_input":"2023-07-12T11:41:10.049351Z","iopub.status.idle":"2023-07-12T11:41:10.061014Z","shell.execute_reply.started":"2023-07-12T11:41:10.049230Z","shell.execute_reply":"2023-07-12T11:41:10.059955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:LightSeaGreen; font-family:newtimeroman; font-size:200%; text-align:left;\">  Load Required Libraries </h1>","metadata":{}},{"cell_type":"code","source":"import os\nimport re\nimport glob\nimport pathlib\nimport time\nimport math\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nimport cv2\n\nimport PIL\nfrom PIL import Image\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils import class_weight\n\nfrom collections import Counter\n\nfrom warnings import filterwarnings\nfilterwarnings('ignore')\n\nSEED=123\nnp.random.seed(SEED)","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:10.062941Z","iopub.execute_input":"2023-07-12T11:41:10.063306Z","iopub.status.idle":"2023-07-12T11:41:10.973212Z","shell.execute_reply.started":"2023-07-12T11:41:10.063271Z","shell.execute_reply":"2023-07-12T11:41:10.972270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:LightSeaGreen; font-family:newtimeroman; font-size:200%; text-align:left;\"> Load TensorFlow </h1>","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.models import Model,Sequential\nfrom tensorflow.keras.optimizers import Adam, SGD, RMSprop\nfrom tensorflow.keras.layers import Dropout, BatchNormalization\nfrom tensorflow.keras.layers import (\n    Input, Dense, Conv2D, Flatten, Activation, \n    MaxPooling2D, AveragePooling2D, ZeroPadding2D, GlobalAveragePooling2D, GlobalMaxPooling2D, add\n)\n\nfrom tensorflow.python.keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom tensorflow.keras.utils import plot_model\n\nfrom tensorflow.keras.applications.vgg19 import VGG19\nfrom tensorflow.keras.applications.vgg19 import preprocess_input\nfrom tensorflow.keras.applications.resnet50 import ResNet50\nfrom tensorflow.keras.applications.resnet50 import preprocess_input\nfrom tensorflow.keras.applications.inception_v3 import InceptionV3\n#from tensorflow.keras.applications.inception_v3 import preprocess_input","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:10.974993Z","iopub.execute_input":"2023-07-12T11:41:10.975342Z","iopub.status.idle":"2023-07-12T11:41:15.967189Z","shell.execute_reply.started":"2023-07-12T11:41:10.975305Z","shell.execute_reply":"2023-07-12T11:41:15.966208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:LightSeaGreen; font-family:newtimeroman; font-size:200%; text-align:left;\">  Configs </h1>","metadata":{}},{"cell_type":"code","source":"CFG = dict(\n        batch_size        =  16,     # 8; 16; 32; 64; bigger batch size => moemry allocation issue\n        epochs            =  100,   # 5; 10; 20;\n        verbose           =   1,    # 0; 1\n        workers           =   4,    # 1; 2; 3\n\n        optimizer         = 'adam', # 'SGD', 'RMSprop'\n\n        RANDOM_STATE      =  123,   \n    \n        # Path to save a model\n        path_model        = '../working/',\n\n        # Images sizes\n        img_size          = 224, \n        img_height        = 224, \n        img_width         = 224, \n\n        # Images augs\n        ROTATION          = 180.0,\n        ZOOM              =  10.0,\n        ZOOM_RANGE        =  [0.9,1.1],\n        HZOOM             =  10.0,\n        WZOOM             =  10.0,\n        HSHIFT            =  10.0,\n        WSHIFT            =  10.0,\n        SHEAR             =   5.0,\n        HFLIP             = True,\n        VFLIP             = True,\n\n        # Postprocessing\n        label_smooth_fac  =  0.00,  # 0.01; 0.05; 0.1; 0.2;    \n)","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:15.969857Z","iopub.execute_input":"2023-07-12T11:41:15.970483Z","iopub.status.idle":"2023-07-12T11:41:15.977911Z","shell.execute_reply.started":"2023-07-12T11:41:15.970443Z","shell.execute_reply":"2023-07-12T11:41:15.976999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:LightSeaGreen; font-family:newtimeroman; font-size:200%; text-align:left;\">  Paths </h1>","metadata":{"execution":{"iopub.status.busy":"2021-06-05T23:43:09.018882Z","iopub.execute_input":"2021-06-05T23:43:09.019279Z","iopub.status.idle":"2021-06-05T23:43:09.761092Z","shell.execute_reply.started":"2021-06-05T23:43:09.019246Z","shell.execute_reply":"2021-06-05T23:43:09.759959Z"}}},{"cell_type":"code","source":"BASEPATH = \"../input/siim-isic-melanoma-classification\"\ndf_train_full = pd.read_csv(os.path.join(BASEPATH, 'train.csv'))\ndf_test  = pd.read_csv(os.path.join(BASEPATH, 'test.csv'))\ndf_sub   = pd.read_csv(os.path.join(BASEPATH, 'sample_submission.csv'))","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:15.979178Z","iopub.execute_input":"2023-07-12T11:41:15.979555Z","iopub.status.idle":"2023-07-12T11:41:16.142552Z","shell.execute_reply.started":"2023-07-12T11:41:15.979518Z","shell.execute_reply":"2023-07-12T11:41:16.141575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_path = '../input/siim-isic-melanoma-classification/jpeg/train'\n#test_path  = '../input/siim-isic-melanoma-classification/jpeg/test'\n\n# Dataset ready for Keras load from directories (structured with respect to classes)\ntrain_path = '../input/skin-cancer9-classesisic/Skin cancer ISIC The International Skin Imaging Collaboration/Train'\ntest_path  = '../input/skin-cancer9-classesisic/Skin cancer ISIC The International Skin Imaging Collaboration/Test'","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:16.146079Z","iopub.execute_input":"2023-07-12T11:41:16.146371Z","iopub.status.idle":"2023-07-12T11:41:16.151826Z","shell.execute_reply.started":"2023-07-12T11:41:16.146342Z","shell.execute_reply":"2023-07-12T11:41:16.150478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = pathlib.Path(train_path)\ntest_dir  = pathlib.Path(test_path)","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:16.154527Z","iopub.execute_input":"2023-07-12T11:41:16.154939Z","iopub.status.idle":"2023-07-12T11:41:16.163482Z","shell.execute_reply.started":"2023-07-12T11:41:16.154901Z","shell.execute_reply":"2023-07-12T11:41:16.162584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can load images by\n- `.flow_from_directory()` using information from subdirectories which has names from labels (we need to prepare data for that, 9 classes give 9 subdirs). Check [here](https://keras.io/api/preprocessing/image/#flowfromdataframe-method).\n- `.flow_from_dataframe()` using information about labels from dataframe. Check [here](https://keras.io/api/preprocessing/image/). ","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"background-color:LightSeaGreen; font-family:newtimeroman; font-size:200%; text-align:left;\">  Dataset </h1>\n<h3 style=\"background-color:LightSeaGreen; font-family:newtimeroman; font-size:200%; text-align:left;\">  Description </h3>","metadata":{}},{"cell_type":"markdown","source":"This set consists of **2357** images of **malignant** and **benign** oncological diseases, which were formed from [The International Skin Imaging Collaboration (ISIC)](https://www.isic-archive.com/).    \n   - All images were sorted according to the classification taken with ISIC, and all subsets were divided into the same number of images, with the exception of melanomas and moles, whose images are slightly dominant.\n\nThe data set contains the following diseases:  \n- actinic keratosis\n- basal cell carcinoma\n- dermatofibroma\n- melanoma\n- nevus\n- pigmented benign keratosis\n- seborrheic keratosis\n- squamous cell carcinoma\n- vascular lesion","metadata":{}},{"cell_type":"code","source":"classes=[\n    'pigmented benign keratosis',\n    'melanoma',\n    'vascular lesion',\n    'actinic keratosis',\n    'squamous cell carcinoma',\n    'basal cell carcinoma',\n    'seborrheic keratosis',\n    'dermatofibroma',\n    'nevus'\n]","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:16.167222Z","iopub.execute_input":"2023-07-12T11:41:16.167565Z","iopub.status.idle":"2023-07-12T11:41:16.175396Z","shell.execute_reply.started":"2023-07-12T11:41:16.167525Z","shell.execute_reply":"2023-07-12T11:41:16.174467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:LightSeaGreen; font-family:newtimeroman; font-size:200%; text-align:left;\"> 5.2 EDA </h1>","metadata":{}},{"cell_type":"code","source":"# Image size check.   \n# We plan to feed our network the images with size 224x224.\n\nimg = Image.open(train_path+'/melanoma/ISIC_0000139.jpg')\nprint(img.size) ","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:16.178641Z","iopub.execute_input":"2023-07-12T11:41:16.178912Z","iopub.status.idle":"2023-07-12T11:41:16.232480Z","shell.execute_reply.started":"2023-07-12T11:41:16.178887Z","shell.execute_reply":"2023-07-12T11:41:16.231576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count number of images in each set.\nimg_count_train = len(list(train_dir.glob('*/*.jpg')))\nimg_count_test  = len(list(test_dir.glob('*/*.jpg')))\nprint('{} train images'.format(img_count_train))\nprint('{} test  images'.format(img_count_test))","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:16.233794Z","iopub.execute_input":"2023-07-12T11:41:16.234145Z","iopub.status.idle":"2023-07-12T11:41:17.323085Z","shell.execute_reply.started":"2023-07-12T11:41:16.234107Z","shell.execute_reply":"2023-07-12T11:41:17.322131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can read train and validation sets from same directory using Keras.\n\ntrain_ds = tf.keras.preprocessing.image_dataset_from_directory(train_dir, validation_split=0.3, image_size=(224,224), subset=\"training\", seed=SEED)\nvalid_ds = tf.keras.preprocessing.image_dataset_from_directory(train_dir, validation_split=0.3, image_size=(224,224), subset=\"validation\",seed=SEED)\ntest_ds  = tf.keras.preprocessing.image_dataset_from_directory(test_dir, image_size=(224,224), seed=SEED)\n\nclass_names = train_ds.class_names\nnum_classes = len(class_names)\nprint('\\n{} classes:\\n{}'.format(num_classes,class_names))\n\nplt.figure(figsize=(23, 12))\nfor images, labels in train_ds.take(1):\n    for i in range(18):\n        ax = plt.subplot(3, 6, i + 1)\n        plt.imshow(images[i].numpy().astype(\"uint8\"))\n        plt.title(class_names[labels[i]])\n        plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:17.324461Z","iopub.execute_input":"2023-07-12T11:41:17.324842Z","iopub.status.idle":"2023-07-12T11:41:26.762944Z","shell.execute_reply.started":"2023-07-12T11:41:17.324804Z","shell.execute_reply":"2023-07-12T11:41:26.761991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:LightSeaGreen; font-family:newtimeroman; font-size:200%; text-align:left;\">  Keras image data processing </h1>","metadata":{}},{"cell_type":"code","source":"# https://keras.io/api/preprocessing/image/\n# Check data processing flow from directory\n# Check data augmentation \n\ntrain_datagen = ImageDataGenerator(rescale=1./255, validation_split=0.3,\n    rotation_range            = CFG['ROTATION'],\n    zoom_range                = CFG['ZOOM_RANGE'],\n    horizontal_flip           = CFG['HFLIP'],\n    vertical_flip             = CFG['VFLIP'],\n    height_shift_range        = CFG['HSHIFT'],\n    width_shift_range         = CFG['WSHIFT'],\n    shear_range               = CFG['SHEAR'],\n    channel_shift_range       = 0.0,\n    brightness_range          = None,\n    fill_mode                 = 'nearest',                          \n    )\n\nvalid_generator = ImageDataGenerator(rescale=1./255, validation_split=0.3)              # no aug for valid\ntest_generator  = ImageDataGenerator(rescale=1./255)                                    # no aug for test\n\n\n# Train data\ntrain_generator = train_datagen.flow_from_directory(train_dir,\n                                                    subset='training',                  # to read train/valid from same directory \n                                                    target_size=(CFG['img_size'], CFG['img_size']),\n                                                    batch_size = CFG['batch_size'],\n                                                    class_mode='categorical',\n                                                    )\n\n# Validation data\nvalid_generator = valid_generator.flow_from_directory(train_dir,\n                                                     subset='validation',               # to read train/valid from same directory \n                                                     target_size=(CFG['img_size'], CFG['img_size']),\n                                                     batch_size = CFG['batch_size'],\n                                                     class_mode='categorical'\n                                                     ) \n# Test data\ntest_generator  = test_generator.flow_from_directory(test_dir,\n                                                     target_size=(CFG['img_size'], CFG['img_size']),\n                                                     batch_size = 1,                    # using 1 to easily manage mapping between test_gen & pred\n                                                     class_mode='categorical'\n                                                     )","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:26.764060Z","iopub.execute_input":"2023-07-12T11:41:26.764392Z","iopub.status.idle":"2023-07-12T11:41:27.090465Z","shell.execute_reply.started":"2023-07-12T11:41:26.764350Z","shell.execute_reply":"2023-07-12T11:41:27.089594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:LightSeaGreen; font-family:newtimeroman; font-size:200%; text-align:left;\"> 7. Class weights </h1>","metadata":{}},{"cell_type":"code","source":"# Class weights\nclass_weights = class_weight.compute_class_weight('balanced',\n                                                  np.unique(train_generator.classes), \n                                                  train_generator.classes) \n\nunique_class_weights = np.unique(train_generator.classes)\nclass_weights_dict   = { unique_class_weights[i]: w for i,w in enumerate(class_weights) }\n\nprint('\\nCLASS WEIGHTS: {}\\n'.format(class_weights))\nprint(np.unique(train_generator.classes))\nprint(train_generator.classes)\nprint(unique_class_weights)\nprint(Counter(train_generator.classes).keys())   # equals to list(set(x))\nprint(Counter(train_generator.classes).values()) # counts the elements' frequency","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:27.093673Z","iopub.execute_input":"2023-07-12T11:41:27.093959Z","iopub.status.idle":"2023-07-12T11:41:27.107867Z","shell.execute_reply.started":"2023-07-12T11:41:27.093930Z","shell.execute_reply":"2023-07-12T11:41:27.106578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:LightSeaGreen; font-family:newtimeroman; font-size:200%; text-align:left;\"> Model </h1>","metadata":{"execution":{"iopub.status.busy":"2021-06-06T15:25:49.684302Z","iopub.execute_input":"2021-06-06T15:25:49.684653Z","iopub.status.idle":"2021-06-06T15:25:49.689686Z","shell.execute_reply.started":"2021-06-06T15:25:49.684615Z","shell.execute_reply":"2021-06-06T15:25:49.688632Z"}}},{"cell_type":"markdown","source":"<h1 style=\"background-color:LightSeaGreen; font-family:newtimeroman; font-size:200%; text-align:left;\">  Build model with ResNet50 </h1>","metadata":{}},{"cell_type":"code","source":"model_ResNet50 = tf.keras.Sequential([\n     tf.keras.applications.ResNet50(\n        input_shape=(224, 224, 3),\n        weights='imagenet',\n        include_top=False\n    ),\n    \n    GlobalAveragePooling2D(),\n    \n    #Dense(1024, activation = 'relu'), \n    #Dropout(0.5), \n    #BatchNormalization(),\n    \n    #Dense(256, activation='relu'), \n    #Dropout(0.3), \n    #BatchNormalization(),\n    \n    #Dense(64, activation='relu'), \n    #Dropout(0.2), \n    #BatchNormalization(),\n    \n    Dense(num_classes, activation='softmax') # num classes = 9\n    \n])\n    \nmodel_ResNet50.compile(\n    optimizer = CFG['optimizer'],\n    loss = tf.keras.losses.BinaryCrossentropy(label_smoothing = CFG['label_smooth_fac']),\n    #loss = 'binary_crossentropy',\n    metrics=['accuracy']\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:27.110439Z","iopub.execute_input":"2023-07-12T11:41:27.111051Z","iopub.status.idle":"2023-07-12T11:41:29.477479Z","shell.execute_reply.started":"2023-07-12T11:41:27.111011Z","shell.execute_reply":"2023-07-12T11:41:29.476586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Possible loss: focal loss\n# BCE -> focal loss (due to class imbalance)\n# Advance by youself!\n\nfrom keras import backend as K\n\ndef focal_loss(alpha=0.20,gamma=2.0):\n    def focal_crossentropy(y_true, y_pred):\n        bce = K.binary_crossentropy(y_true, y_pred)\n        \n        y_pred = K.clip(y_pred, K.epsilon(), 1.- K.epsilon())\n        p_t = (y_true*y_pred) + ((1-y_true)*(1-y_pred))\n        \n        alpha_factor = 1\n        modulating_factor = 1\n\n        alpha_factor = y_true*alpha + ((1-alpha)*(1-y_true))\n        modulating_factor = K.pow((1-p_t), gamma)\n\n        # compute the final loss and return\n        return K.mean(alpha_factor*modulating_factor*bce, axis=-1)\n    \n    return focal_crossentropy","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:29.478785Z","iopub.execute_input":"2023-07-12T11:41:29.479133Z","iopub.status.idle":"2023-07-12T11:41:29.524530Z","shell.execute_reply.started":"2023-07-12T11:41:29.479097Z","shell.execute_reply":"2023-07-12T11:41:29.523739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:LightSeaGreen; font-family:newtimeroman; font-size:200%; text-align:left;\">  Visualize model with ResNet50 </h1>","metadata":{}},{"cell_type":"code","source":"model_ResNet50.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:29.525810Z","iopub.execute_input":"2023-07-12T11:41:29.526173Z","iopub.status.idle":"2023-07-12T11:41:29.548143Z","shell.execute_reply.started":"2023-07-12T11:41:29.526137Z","shell.execute_reply":"2023-07-12T11:41:29.547180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We reduce significantly number of trainable parameters by freezing certain layers, excluding from training, i.e. their weights will never be updated\n\n# freeze the first 1 layer\n\nmodel_ResNet50.layers[0].trainable = False\n#for layer in model_ResNet50.layers[:1]:\n#    layer.trainable = False\nmodel_ResNet50.summary()","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:29.549351Z","iopub.execute_input":"2023-07-12T11:41:29.549892Z","iopub.status.idle":"2023-07-12T11:41:29.580589Z","shell.execute_reply.started":"2023-07-12T11:41:29.549854Z","shell.execute_reply":"2023-07-12T11:41:29.579642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot model scheme with TF/Keras plot_model function\nplot_model(model_ResNet50, to_file='model_plot.png', show_shapes=True, show_layer_names=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:29.581804Z","iopub.execute_input":"2023-07-12T11:41:29.582147Z","iopub.status.idle":"2023-07-12T11:41:30.068407Z","shell.execute_reply.started":"2023-07-12T11:41:29.582112Z","shell.execute_reply":"2023-07-12T11:41:30.067392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:LightSeaGreen; font-family:newtimeroman; font-size:200%; text-align:left;\"> 9 Fit model </h1>","metadata":{}},{"cell_type":"code","source":"#tf.function-decorated function tried to create variables on non-first call'. \ntf.config.run_functions_eagerly(True) # otherwise error\n\n# https://www.tensorflow.org/api_docs/python/tf/keras/callbacks/ModelCheckpoint\ncb_early_stopper = EarlyStopping(monitor = 'val_loss', patience = 10)\ncb_checkpointer  = ModelCheckpoint(#filepath=path_model,\n                                   #filepath=CFG['path_model']+'ResNet50.hdf5'\n                                   filepath = CFG['path_model']+'ResNet50-{epoch:02d}-{val_loss:.2f}.hdf5',\n                                   monitor  = 'val_loss', \n                                   verbose  = CFG['verbose'], \n                                   save_best_only=True, \n                                   mode='min'\n                                  )\n\ncallbacks_list = [cb_checkpointer, cb_early_stopper]\n\nhistory = model_ResNet50.fit(train_generator, \n                             epochs=CFG['epochs'], \n                             workers=CFG['workers'],\n                             #steps_per_epoch = train_generator.n // 2, # hide if you wish\n                             validation_data=valid_generator, \n                             #validation_steps=valid_generator.n // 2,  # hide if you wish\n                             callbacks = callbacks_list,\n                             class_weight = class_weights_dict\n                            )","metadata":{"execution":{"iopub.status.busy":"2023-07-12T11:41:30.072881Z","iopub.execute_input":"2023-07-12T11:41:30.073186Z","iopub.status.idle":"2023-07-12T12:02:48.613434Z","shell.execute_reply.started":"2023-07-12T11:41:30.073152Z","shell.execute_reply":"2023-07-12T12:02:48.612366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:LightSeaGreen; font-family:newtimeroman; font-size:200%; text-align:left;\"> 10. Visualize performance </h1>","metadata":{}},{"cell_type":"code","source":"acc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\n\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nmetrics = history.history['accuracy']\nepochs_range = range(1, len(metrics) + 1) \n\nplt.figure(figsize=(23, 8))\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label='Training Accuracy')\nplt.plot(epochs_range, val_acc, label='Validation Accuracy')\nplt.legend(loc='lower right')\nplt.title('Training and Validation Accuracy')\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Training Loss')\nplt.plot(epochs_range, val_loss, label='Validation Loss')\nplt.legend(loc='upper right')\nplt.title('Training and Validation Loss')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-12T12:02:48.615527Z","iopub.execute_input":"2023-07-12T12:02:48.615909Z","iopub.status.idle":"2023-07-12T12:02:48.959415Z","shell.execute_reply.started":"2023-07-12T12:02:48.615870Z","shell.execute_reply":"2023-07-12T12:02:48.958562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:LightSeaGreen; font-family:newtimeroman; font-size:200%; text-align:left;\"> 11. Evaluate on test </h1>","metadata":{}},{"cell_type":"code","source":"print('Computing predictions...')\ntest_images_ds = test_ds.map(lambda image, idnum: image)\nprobabilities = model_ResNet50.predict(test_images_ds) # check that images processed same size, default for image from dir is 256,256","metadata":{"execution":{"iopub.status.busy":"2023-07-12T12:02:48.960790Z","iopub.execute_input":"2023-07-12T12:02:48.961346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 118 images, 9 classes\nprobabilities.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# probabilities for each class for first test image\nprobabilities[0,:]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probabilities[0,:].shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds.class_names","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# probability for melanoma class\nprobabilities[:,4] ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probabilities[:,4].shape # N images and (N,) predictions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Issue with predictions to be fixed. ","metadata":{}},{"cell_type":"markdown","source":"<h1 style=\"background-color:LightSeaGreen; font-family:newtimeroman; font-size:200%; text-align:left;\"> 12. Submit predictions </h1>","metadata":{}},{"cell_type":"code","source":"file_paths = test_ds.file_paths\n\nk = 1\nwhile k<20:\n    print(file_paths[k])\n    k += 1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_paths[1].split(os.sep)[-1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Generating submission.csv file...')\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(img_count_test))).numpy().astype('U')\n\nimg_list = []\nimg_id_list = []\nimg_name_list = []\nfor i in range(len(file_paths)):\n    img_list.append(file_paths[i].split(os.sep)[-1])\n    img_id_list.append(i)\n    img_name_list.append(file_paths[i].split(os.sep)[-1][0:-4])\n\nimg_name_list_by_test_ids = []\nfor iid in list(test_ids):\n    print(int(iid),img_name_list[int(iid)],probabilities[:,4][int(iid)]) # here dummy iid got str not int, thus converted\n    img_name_list_by_test_ids.append(img_name_list[int(iid)])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#pred_df = pd.DataFrame({'image_name': img_name_list_by_test_ids, 'target': probabilities[:,1]}) # 'ids':test_ids\npred_df = pd.DataFrame({'image_name': img_name_list, 'target': probabilities[:,4]}) # 'ids':test_ids\npred_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df['target'].unique()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#del df_sub['target']\n#sub = df_sub.merge(pred_df, on='image_name')\n#sub.to_csv('submission.csv', index=False)\n#sub.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h1 style=\"background-color:LightSeaGreen; font-family:newtimeroman; font-size:200%; text-align:left;\"> References </h1>\n\n- [Data] [SIIM-ISIC Skin cancer 9 classes](https://www.kaggle.com/nodoubttome/skin-cancer9-classesisic)\n- [ResNet] [Keras ResNet](https://keras.io/api/applications/resnet/)\n- [ResNet] [Keras ResNet50 on Kaggle](https://www.kaggle.com/keras/resnet50)\n- [Article] [Kaiming He et al Deep Residual Learning for Image Recognition. (CVPR 2015)](https://arxiv.org/abs/1512.03385)\n- [Article] [Simonyan, K. and Zisserman, A. Very Deep Convolutional Networks for Large-Scale Image Recognition. (ICLR 2015)](https://arxiv.org/abs/1409.1556)\n- [VGG] [Keras VGG](https://keras.io/api/applications/vgg/)\n- [Keras] [Keras image data preprocessing](https://keras.io/api/preprocessing/image/)\n- [TF/Keras] [BinaryCrossentropy](https://www.tensorflow.org/api_docs/python/tf/keras/losses/BinaryCrossentropy)\n- [TF/Keras] [ModelCheckpoint](https://www.tensorflow.org/api_docs/python/tf/keras/callbacks/ModelCheckpoint), [see also](https://keras.io/api/callbacks/model_checkpoint/) \n- [Notebook] [SIIM-ISIC Melanoma Classification EfficientNet](https://www.kaggle.com/muhakabartay/siim-isic-melanoma-classification-efficientnet) ","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}