{"cells":[{"metadata":{},"cell_type":"markdown","source":"Introduction : \n    \n  Une collection de 21397 images étiquetées.\n  \n  l'objectif est de classer chaque image de manioc en quatre catégories de maladies ou une cinquième catégorie indiquant une feuille saine.\n  \n  \"0\": \"Cassava Bacterial Blight (CBB)\",\n  \n  \"1\": \"Cassava Brown Streak Disease (CBSD)\",\n  \n  \"2\": \"Cassava Green Mottle (CGM)\",\n  \n  \"3\": \"Cassava Mosaic Disease (CMD)\",\n  \n  \"4\": \"Healthy\"\n  \n  Evaluation : \n  Précision de classification(accuracy)\n"},{"metadata":{},"cell_type":"markdown","source":"Premiére étape : Importation des données "},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install efficientnet","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"# Les imports\nimport math\nimport numpy as np \nimport re\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n\nimport json # to read in the 'label_num_to_disease_map.json' file\nfrom tensorflow.keras.preprocessing.image import load_img\nfrom tensorflow.keras.preprocessing.image import img_to_array\nfrom tensorflow.keras.utils import plot_model\n\nimport matplotlib.pyplot as plt\nimport plotly.express as px\nimport seaborn as sns\nimport cv2 #OpenCV\nimport random\n\nimport tensorflow as tf\nprint(\"Using TensorFlow version %s\" % tf.__version__)\n\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# set the training and test directory paths\nTRAIN_DIR = '../input/cassava-leaf-disease-classification/train_images/'\nTEST_DIR = '../input/cassava-leaf-disease-classification/test_images/'\nROOT_DIR = '../input/cassava-leaf-disease-classification/'\nos.listdir(ROOT_DIR)\n\nPREPROC_PATH = KaggleDatasets().get_gcs_path('cassava-leaf-disease-classification')\nAUTOTUNE = tf.data.experimental.AUTOTUNE\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\"\"\" #set seed\nseed = 42\n\ndef seed_everything(seed):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n    \nseed_everything(seed)\"\"\"\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Deuxiéme étape : Exploration des données"},{"metadata":{"trusted":true},"cell_type":"code","source":" # returns a data frame 'with pandas'\ntrain_df = pd.read_csv(ROOT_DIR + 'train.csv') \nsample_df = pd.read_csv(ROOT_DIR + 'sample_submission.csv')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(train_df.shape, sample_df.shape)\ndisplay(train_df.head())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f = open(ROOT_DIR + 'label_num_to_disease_map.json')\ndata = json.load(f)\nprint(json.dumps(data, indent = 2))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"z = train_df.sample(20)\ndisplay(z)\nimages, labels = z['image_id'].tolist(), z['label'].tolist()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\"Visaulisation des données\"\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize = (20,20))\nfor i in range(20):\n    plt.subplot(4,5,i+1)\n    img = cv2.imread(TRAIN_DIR + images[i])\n    img = cv2.cvtColor(img,cv2.COLOR_BGR2RGB)\n    plt.imshow(img)\n    plt.title(data[str(labels[i])])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"> Preprocessing\n\nPreparation des donnees pour le modele"},{"metadata":{"trusted":true},"cell_type":"code","source":"from functools import partial\n\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import models\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model, Sequential\nfrom tensorflow.keras.layers import Flatten, Dense\nfrom tensorflow.keras.losses import categorical_crossentropy\nfrom tensorflow.keras.metrics import categorical_accuracy\nfrom tensorflow.keras.activations import softmax\nfrom tensorflow.keras.optimizers import SGD\nfrom tensorflow.keras import losses\nimport tensorflow.keras.backend as K","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Verification de utilisation du TPU, sinon affichage 1 alors activer le TPU pour avoir les 8 coeurs a utiliser\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Device:', tpu.master())\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept:\n    strategy = tf.distribute.get_strategy()\nprint('Number of replicas:', strategy.num_replicas_in_sync)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def read_tfrecord(example, labeled):\n    tfrecord_format = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"target\": tf.io.FixedLenFeature([], tf.int64)\n    } if labeled else {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"image_name\": tf.io.FixedLenFeature([], tf.string)\n    }\n    example = tf.io.parse_single_example(example, tfrecord_format)\n    image = decode_image(example['image'])\n    if labeled:\n        label = tf.cast(example['target'], tf.int32)\n        return image, label\n    idnum = example['image_name']\n    return image, idnum","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"In the code chunk below we'll set up a series of functions that allow us to convert our images into tensors so that we can utilize them in our model. We'll also normalize our data. Our images are using a \"Red, Blue, Green (RBG)\" scale that has a range of [0, 255], and by normalizing it we'll set each pixel's value to a number in the range of [0, 1]."},{"metadata":{"trusted":true},"cell_type":"code","source":"def decode_image(image):\n    image = tf.image.decode_jpeg(image, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0\n    image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    return image\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We'll use the following function to load our dataset. One of the advantages of a TPU is that we can run multiple files across the TPU at once, and this accounts for the speed advantages of using a TPU. To capitalize on that, we want to make sure that we're using data as soon as it streams in, rather than creating a data streaming bottleneck"},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False # disable order, increase speed\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTOTUNE) # automatically interleaves reads from multiple files\n    dataset = dataset.with_options(ignore_order) # uses data as soon as it streams in, rather than in its original order\n    dataset = dataset.map(partial(read_tfrecord, labeled=labeled), num_parallel_calls=AUTOTUNE)\n    return dataset","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def data_augment(image, label):\n    p_rotation = tf.random.uniform([], 0, 1.0, dtype=tf.float32)\n    p_spatial = tf.random.uniform([], 0, 1.0, dtype=tf.float32)\n    p_rotate = tf.random.uniform([], 0, 1.0, dtype=tf.float32)\n    p_pixel_1 = tf.random.uniform([], 0, 1.0, dtype=tf.float32)\n    p_pixel_2 = tf.random.uniform([], 0, 1.0, dtype=tf.float32)\n    p_pixel_3 = tf.random.uniform([], 0, 1.0, dtype=tf.float32)\n    p_shear = tf.random.uniform([], 0, 1.0, dtype=tf.float32)\n    p_crop = tf.random.uniform([], 0, 1.0, dtype=tf.float32)\n    \n    # Shear\n    if p_shear > .2:\n        if p_shear > .6:\n            image = transform_shear(image, HEIGHT, shear=20.)\n        else:\n            image = transform_shear(image, HEIGHT, shear=-20.)\n            \n    # Rotation\n    if p_rotation > .2:\n        if p_rotation > .6:\n            image = transform_rotation(image, HEIGHT, rotation=45.)\n        else:\n            image = transform_rotation(image, HEIGHT, rotation=-45.)\n            \n    # Flips\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_flip_up_down(image)\n    if p_spatial > .75:\n        image = tf.image.transpose(image)\n        \n    # Rotates\n    if p_rotate > .75:\n        image = tf.image.rot90(image, k=3) # rotate 270º\n    elif p_rotate > .5:\n        image = tf.image.rot90(image, k=2) # rotate 180º\n    elif p_rotate > .25:\n        image = tf.image.rot90(image, k=1) # rotate 90º\n        \n    # Pixel-level transforms\n    if p_pixel_1 >= .4:\n        image = tf.image.random_saturation(image, lower=.7, upper=1.3)\n    if p_pixel_2 >= .4:\n        image = tf.image.random_contrast(image, lower=.8, upper=1.2)\n    if p_pixel_3 >= .4:\n        image = tf.image.random_brightness(image, max_delta=.1)\n        \n    # Crops\n    if p_crop > .7:\n        if p_crop > .9:\n            image = tf.image.central_crop(image, central_fraction=.6)\n        elif p_crop > .8:\n            image = tf.image.central_crop(image, central_fraction=.7)\n        else:\n            image = tf.image.central_crop(image, central_fraction=.8)\n    elif p_crop > .4:\n        crop_size = tf.random.uniform([], int(HEIGHT*.6), HEIGHT, dtype=tf.int32)\n        image = tf.image.random_crop(image, size=[crop_size, crop_size, CHANNELS])\n            \n    image = tf.image.resize(image, size=[HEIGHT, WIDTH])\n\n    return image, label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# data augmentation @cdeotte kernel: https://www.kaggle.com/cdeotte/rotation-augmentation-gpu-tpu-0-96\ndef transform_rotation(image, height, rotation):\n    # input image - is one image of size [dim,dim,3] not a batch of [b,dim,dim,3]\n    # output - image randomly rotated\n    DIM = height\n    XDIM = DIM%2 #fix for size 331\n    \n    rotation = rotation * tf.random.uniform([1],dtype='float32')\n    # CONVERT DEGREES TO RADIANS\n    rotation = math.pi * rotation / 180.\n    \n    # ROTATION MATRIX\n    c1 = tf.math.cos(rotation)\n    s1 = tf.math.sin(rotation)\n    one = tf.constant([1],dtype='float32')\n    zero = tf.constant([0],dtype='float32')\n    rotation_matrix = tf.reshape(tf.concat([c1,s1,zero, -s1,c1,zero, zero,zero,one],axis=0),[3,3])\n\n    # LIST DESTINATION PIXEL INDICES\n    x = tf.repeat( tf.range(DIM//2,-DIM//2,-1), DIM )\n    y = tf.tile( tf.range(-DIM//2,DIM//2),[DIM] )\n    z = tf.ones([DIM*DIM],dtype='int32')\n    idx = tf.stack( [x,y,z] )\n    \n    # ROTATE DESTINATION PIXELS ONTO ORIGIN PIXELS\n    idx2 = K.dot(rotation_matrix,tf.cast(idx,dtype='float32'))\n    idx2 = K.cast(idx2,dtype='int32')\n    idx2 = K.clip(idx2,-DIM//2+XDIM+1,DIM//2)\n    \n    # FIND ORIGIN PIXEL VALUES \n    idx3 = tf.stack( [DIM//2-idx2[0,], DIM//2-1+idx2[1,]] )\n    d = tf.gather_nd(image, tf.transpose(idx3))\n        \n    return tf.reshape(d,[DIM,DIM,3])\n\ndef transform_shear(image, height, shear):\n    # input image - is one image of size [dim,dim,3] not a batch of [b,dim,dim,3]\n    # output - image randomly sheared\n    DIM = height\n    XDIM = DIM%2 #fix for size 331\n    \n    shear = shear * tf.random.uniform([1],dtype='float32')\n    shear = math.pi * shear / 180.\n        \n    # SHEAR MATRIX\n    one = tf.constant([1],dtype='float32')\n    zero = tf.constant([0],dtype='float32')\n    c2 = tf.math.cos(shear)\n    s2 = tf.math.sin(shear)\n    shear_matrix = tf.reshape(tf.concat([one,s2,zero, zero,c2,zero, zero,zero,one],axis=0),[3,3])    \n\n    # LIST DESTINATION PIXEL INDICES\n    x = tf.repeat( tf.range(DIM//2,-DIM//2,-1), DIM )\n    y = tf.tile( tf.range(-DIM//2,DIM//2),[DIM] )\n    z = tf.ones([DIM*DIM],dtype='int32')\n    idx = tf.stack( [x,y,z] )\n    \n    # ROTATE DESTINATION PIXELS ONTO ORIGIN PIXELS\n    idx2 = K.dot(shear_matrix,tf.cast(idx,dtype='float32'))\n    idx2 = K.cast(idx2,dtype='int32')\n    idx2 = K.clip(idx2,-DIM//2+XDIM+1,DIM//2)\n    \n    # FIND ORIGIN PIXEL VALUES \n    idx3 = tf.stack( [DIM//2-idx2[0,], DIM//2-1+idx2[1,]] )\n    d = tf.gather_nd(image, tf.transpose(idx3))\n        \n    return tf.reshape(d,[DIM,DIM,3])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TRAINING_FILENAMES, VALID_FILENAMES = train_test_split(\n    tf.io.gfile.glob(PREPROC_PATH + '/train_tfrecords/ld_train*.tfrec'),\n    test_size=0.35, random_state=5\n)\n\nTEST_FILENAMES = tf.io.gfile.glob(PREPROC_PATH + '/test_tfrecords/ld_test*.tfrec')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Get datasets\nThe following functions will be used to load our training, validation, and test datasets, as well as print out the number of images in each dataset."},{"metadata":{"trusted":true},"cell_type":"code","source":"#DATASET PARAMS\nIMAGE_SIZE = [512, 512]\nCLASSES = ['0', '1', '2', '3', '4']\nEPOCHS = 50\nHEIGHT = 512\nWIDTH = 512\nCHANNELS = 3\n\nBATCH_SIZE = 512\n# BATCH_SIZE = 64 * strategy.num_replicas_in_sync\nprint(BATCH_SIZE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_training_dataset():\n    dataset = load_dataset(TRAINING_FILENAMES, labeled=True)  \n    dataset = dataset.map(data_augment, num_parallel_calls=AUTOTUNE)\n#     dataset = dataset.map(transform_rotation(HEIGHT,0.2), num_parallel_calls=AUTOTUNE)\n#     dataset = dataset.map(transform_shear, num_parallel_calls=AUTOTUNE)\n    dataset = dataset.repeat()\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTOTUNE)\n    return dataset","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_validation_dataset(ordered=False):\n    dataset = load_dataset(VALID_FILENAMES, labeled=True, ordered=ordered) \n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(AUTOTUNE)\n    return dataset","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_test_dataset(ordered=False):\n    dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTOTUNE)\n    return dataset","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def count_data_items(filenames):\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) for filename in filenames]\n    return np.sum(n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"NUM_TRAINING_IMAGES = count_data_items(TRAINING_FILENAMES)\nNUM_VALIDATION_IMAGES = count_data_items(VALID_FILENAMES)\nNUM_TEST_IMAGES = count_data_items(TEST_FILENAMES)\n\nprint('Dataset: {} training images, {} validation images, {} (unlabeled) test images'.format(\n    NUM_TRAINING_IMAGES, NUM_VALIDATION_IMAGES, NUM_TEST_IMAGES))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Training data shapes:\")\nfor image, label in get_training_dataset().take(3):\n    print(image.numpy().shape, label.numpy().shape)\nprint(\"Training data label examples:\", label.numpy())\nprint(\"Validation data shapes:\")\nfor image, label in get_validation_dataset().take(3):\n    print(image.numpy().shape, label.numpy().shape)\nprint(\"Validation data label examples:\", label.numpy())\nprint(\"Test data shapes:\")\nfor image, idnum in get_test_dataset().take(3):\n    print(image.numpy().shape, idnum.numpy().shape)\nprint(\"Test data IDs:\", idnum.numpy().astype('U')) # U=unicode string\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Creating Model\n\nIn order to ensure that our model is trained on the TPU, we build it using with strategy.scope()."},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_base_model(add_custom_layer_func) -> Model:\n    img_adjust_layer = tf.keras.layers.Lambda(tf.keras.applications.resnet50.preprocess_input, input_shape=[*IMAGE_SIZE, 3])\n    \n    m = Sequential([\n        img_adjust_layer])\n    add_custom_layer_func(m)\n    \n    m.add(Flatten())\n    m.add(Dense(len(CLASSES), activation= softmax))\n    \n    m.compile(\n        optimizer=SGD(momentum=0.9, lr=0.01),\n        loss=losses.SparseCategoricalCrossentropy(),\n        metrics=[\"sparse_categorical_accuracy\"]\n    )\n\n    return m","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def add_linear_layers(_m: Sequential):\n    pass","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.layers import Input\ndef add_resnet50v2_layers(_m: Sequential):\n    img_adjust_layer = tf.keras.layers.Lambda(tf.keras.applications.resnet50.preprocess_input, input_shape=[*IMAGE_SIZE, 3])\n    subModel = tf.keras.applications.ResNet50(weights='imagenet', include_top=False)\n    m.add(subModel)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"STEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\nVALID_STEPS = NUM_VALIDATION_IMAGES // BATCH_SIZE\ndef train_models(m: Model, dataset_train_it, dataset_val_it):\n    history = m.fit(dataset_train_it, \n                    steps_per_epoch=STEPS_PER_EPOCH, \n                    epochs=EPOCHS,\n                    validation_data=dataset_val_it,\n                    validation_steps=VALID_STEPS)\n    return history","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import efficientnet.keras as efn","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"histories = []\nwith strategy.scope():\n    img_adjust_layer = tf.keras.layers.Lambda(tf.keras.applications.resnet50.preprocess_input, input_shape=[*IMAGE_SIZE, 3])\n#     inputs = layers.Input(shape=(IMAGE_SIZE, IMAGE_SIZE, 3))\n    model = tf.keras.Sequential([\n            img_adjust_layer,\n            efn.EfficientNetB0(include_top=False,weights='imagenet'),\n            layers.GlobalAveragePooling2D(name=\"avg_pool\"),\n            layers.BatchNormalization(),\n            layers.Dropout(0.2, name=\"top_dropout\"),\n            layers.Dense(5, activation=\"softmax\", name=\"pred\")\n        ])\n    \n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(lr = 0.001),\n        loss='sparse_categorical_crossentropy',  \n        metrics=['accuracy'])\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# models = []\n# with strategy.scope():\n#     for i in range(5):\n#         img_adjust_layer = tf.keras.layers.Lambda(tf.keras.applications.resnet50.preprocess_input, input_shape=[*IMAGE_SIZE, 3])\n        \n#         model = tf.keras.Sequential([\n#             img_adjust_layer,\n#             tf.keras.applications.ResNet50(weights='imagenet', include_top=False),\n#             tf.keras.layers.GlobalAveragePooling2D(name=\"avg_pool\"),\n#             tf.keras.layers.BatchNormalization(),\n#             tf.keras.layers.Dropout(0.2, name=\"top_dropout\"),\n#             tf.keras.layers.Dense(5, activation=\"softmax\", name=\"pred\"),\n#             tf.keras.layers.Dense(len(CLASSES), activation='softmax')  \n#         ])\n#         model.compile(loss=losses.SparseCategoricalCrossentropy(), optimizer=tf.optimizers.Adam(lr=0.0001), metrics=['accuracy'])\n#         models.append(model)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.callbacks import LearningRateScheduler, EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\n\ndef callback():\n    reduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.3, patience=3,\n                              verbose=1, mode='auto', min_delta=0.0001,\n                              cooldown=0, min_lr=0)\n\n    early_stopping = EarlyStopping(monitor = \"val_loss\",\n                                min_delta=0.001,\n                                patience=5,\n                                verbose=1,\n                                mode=\"min\",\n                                #baseline=None,\n                                restore_best_weights=False)\n    model_cp = ModelCheckpoint(F'EffNetB0_Tweeked_best.h5', \n                                 save_best_only = True, \n                                 save_weights_only = True,\n                                 monitor = 'val_loss', \n                                 mode = 'min', \n                                 verbose = 1)\n    \n    return [early_stopping, model_cp, reduce_lr]\n#     return [model_cp, reduce_lr]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"histories= []\ntrain_size  = int(21397 *0.8)\n#     train_dataset, validation_dataset = n_fold_dataset(augment = True, index=i)\ntrain_dataset = get_training_dataset()\nvalidation_dataset = get_validation_dataset()\n\nhistory = model.fit(train_dataset,\n      steps_per_epoch=train_size//BATCH_SIZE,\n      validation_data=validation_dataset,\n      batch_size=BATCH_SIZE, epochs=1, \n      callbacks=callback())\nhistories.append(history)\n# print out variables available to us\n# print(history.history.keys())\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# histories= []\n# train_size  = int(21397 *0.8)\n# for i, model in enumerate(models):\n#     print(F\"Training model: {i}\")\n# #     train_dataset, validation_dataset = n_fold_dataset(augment = True, index=i)\n#     train_dataset = get_training_dataset()\n#     validation_dataset = get_validation_dataset()\n\n#     history = model.fit(train_dataset,\n#           steps_per_epoch=train_size//BATCH_SIZE,\n#           validation_data=validation_dataset,\n#           batch_size=BATCH_SIZE, epochs=100, \n#           callbacks=callback(i))\n#     histories.append(history)\n# # print out variables available to us\n# # print(history.history.keys())\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# print(models[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, axes = plt.subplots(nrows=2, ncols=2, figsize=(16, 12))\n\nfor i, history in enumerate(histories):\n    history_df = pd.DataFrame(history.history)\n    history_df[['loss', 'val_loss']].plot(ax=axes[i,0])\n    history_df[['accuracy', 'val_accuracy']].plot(ax=axes[i,1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# model.evaluate(get_training_dataset(), steps = STEPS_PER_EPOCH)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# model.evaluate(get_validation_dataset(), steps=NUM_VALIDATION_IMAGES // BATCH_SIZE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# print(\"ResNet_Tweeked\")\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# history = train_models(model,get_training_dataset(), get_validation_dataset())\n# # histories.append(history)\n\n# # print out variables available to us\n# print(history.history.keys())\n# history_frame = pd.DataFrame(history.history)\n# history_frame.loc[:, ['loss', 'val_loss']].plot()\n# history_frame.loc[:, ['accuracy', 'val_accuracy']].plot();\n# model.evaluate(get_training_dataset(), steps = STEPS_PER_EPOCH)\n# model.evaluate(get_validation_dataset(), steps=NUM_VALIDATION_IMAGES // BATCH_SIZE)\n# print(\"EffNetB7-lr0001.keras\")\n# model.summary()\n# model.save(\"EffNetB7-lr0001+batchNorm.keras\")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def to_float32(image, label):\n    return tf.cast(image, tf.float32), label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_ds = get_test_dataset(ordered=True) \ntest_ds = test_ds.map(to_float32)\ntesting_dataset = get_test_dataset()\ntesting_dataset = testing_dataset.unbatch().batch(20)\ntest_batch = iter(testing_dataset)\n\nprint('Computing predictions...')\ntest_images_ds = testing_dataset\ntest_images_ds = test_ds.map(lambda image, idnum: image)\nprobabilities = model.predict(test_images_ds)\npredictions = np.argmax(probabilities, axis=-1)\nprint(predictions)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Generating submission.csv file...')\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(NUM_TEST_IMAGES))).numpy().astype('U') # all in one batch\nnp.savetxt('submission_effnet0.csv', np.rec.fromarrays([test_ids, predictions]), fmt=['%s', '%d'], delimiter=',', header='id,label', comments='')\n!head submission.csv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# # for i in models:\n# # from tensorflow import keras\n# # model = keras.models.load_model(\"./ResNet50_Tweeked_best1.h5\")\n# with strategy.scope():\n#     img_adjust_layer = tf.keras.layers.Lambda(tf.keras.applications.resnet50.preprocess_input, input_shape=[*IMAGE_SIZE, 3])\n\n#     model = tf.keras.Sequential([\n#         img_adjust_layer,\n#         tf.keras.applications.ResNet50(weights='imagenet', include_top=False),\n#         tf.keras.layers.GlobalAveragePooling2D(name=\"avg_pool\"),\n#         tf.keras.layers.BatchNormalization(),\n#         tf.keras.layers.Dropout(0.2, name=\"top_dropout\"),\n#         tf.keras.layers.Dense(5, activation=\"softmax\", name=\"pred\"),\n#         tf.keras.layers.Dense(len(CLASSES), activation='softmax')  \n#     ])\n#     model.compile(loss=losses.SparseCategoricalCrossentropy(), optimizer=tf.optimizers.Adam(lr=9.999999974752428e-08), metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# histories= []\n# train_size  = int(21397 *0.8)\n# print(F\"Training model: {i}\")\n# #     train_dataset, validation_dataset = n_fold_dataset(augment = True, index=i)\n# train_dataset = get_training_dataset()\n# validation_dataset = get_validation_dataset()\n\n# history = model.fit(train_dataset,\n#       steps_per_epoch=train_size//BATCH_SIZE,\n#       validation_data=validation_dataset,\n#       batch_size=BATCH_SIZE, epochs=200, \n#       callbacks=callback(0))\n# histories.append(history)\n# # print out variables available to us\n# # print(history.history.keys())\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# fig, axes = plt.subplots(nrows=5, ncols=2, figsize=(16, 12))\n\n# for i, history in enumerate(histories):\n#     history_df = pd.DataFrame(history.history)\n#     history_df[['loss', 'val_loss']].plot(ax=axes[i,0])\n#     history_df[['accuracy', 'val_accuracy']].plot(ax=axes[i,1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# model.save(\"resnet50-tweeked.keras\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# test_ds = get_test_dataset(ordered=True) \n# test_ds = test_ds.map(to_float32)\n# testing_dataset = get_test_dataset()\n# testing_dataset = testing_dataset.unbatch().batch(20)\n# test_batch = iter(testing_dataset)\n\n# print('Computing predictions...')\n# test_images_ds = testing_dataset\n# test_images_ds = test_ds.map(lambda image, idnum: image)\n# probabilities = model.predict(test_images_ds)\n# predictions = np.argmax(probabilities, axis=-1)\n# print(predictions)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}