{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"},{"sourceId":1668058,"sourceType":"datasetVersion","datasetId":987852},{"sourceId":1668110,"sourceType":"datasetVersion","datasetId":987886},{"sourceId":1833486,"sourceType":"datasetVersion","datasetId":1089841},{"sourceId":1853904,"sourceType":"datasetVersion","datasetId":1102747},{"sourceId":1854570,"sourceType":"datasetVersion","datasetId":1103226},{"sourceId":1856853,"sourceType":"datasetVersion","datasetId":1104699},{"sourceId":1857685,"sourceType":"datasetVersion","datasetId":1105270},{"sourceId":1868032,"sourceType":"datasetVersion","datasetId":1111995},{"sourceId":1869210,"sourceType":"datasetVersion","datasetId":1112749},{"sourceId":1872205,"sourceType":"datasetVersion","datasetId":1114609},{"sourceId":1872710,"sourceType":"datasetVersion","datasetId":1114815},{"sourceId":1874758,"sourceType":"datasetVersion","datasetId":1116049},{"sourceId":1877279,"sourceType":"datasetVersion","datasetId":1117701}],"dockerImageVersionId":30042,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n    #for filename in filenames:\n        #print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n!pip install --quiet /kaggle/input/kerasapplications\n!pip install --quiet /kaggle/input/efficientnet-git","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport json\n\nfrom PIL import Image\nimport cv2\nimport seaborn as sb\n\n# Seb 06-01-21\nimport tensorflow as tf\nfrom tensorflow import keras\nimport json\nimport math, re, os\nfrom math import sqrt\n\nfrom kaggle_datasets import KaggleDatasets\nfrom matplotlib import pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom functools import partial\n","metadata":{"_uuid":"aeec83f6-c83a-43f9-8f11-55a573f7836c","_cell_guid":"6e283b85-2177-4105-bc80-ecf576db7083","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"main_dir = '../input/cassava-leaf-disease-classification/'\nos.listdir(main_dir) \ntrain_img_path = '../input/cassava-leaf-disease-classification/train_images'","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Detect TPU\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Device:', tpu.master())\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept:\n    strategy = tf.distribute.get_strategy()\nprint('Number of replicas:', strategy.num_replicas_in_sync)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# set up variables\nAUTOTUNE = tf.data.experimental.AUTOTUNE\nGCS_PATH = '/kaggle/input/cassava-leaf-disease-classification'                                                \nIMAGE_SIZE = [512,512]\nCLASSES = ['0', '1', '2', '3', '4']\n\nt_1 = 0.6\nt_2 = 1.4\nSMOOTH_FRACTION = 0.1\nN_ITER = 5\n\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nEPOCHS = 30\ninput_shape = (512,512,3)\ndropout_rate = 0.2","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAINING_FILENAMES, VALID_FILENAMES = train_test_split(\n    tf.io.gfile.glob(GCS_PATH + '/train_tfrecords/ld_train*.tfrec'),\n    test_size=0.2, random_state=5\n)\n\nTEST_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/test_tfrecords/ld_test*.tfrec')\nTEST_FILENAMES","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Functions for loading the data","metadata":{}},{"cell_type":"code","source":"#Decode the data\n#turn the images into tensors\n#normalize the image (get every pixel to have a value between 0 and 1)\ndef decode_image(image):\n    image = tf.image.decode_jpeg(image, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0\n    image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    return image","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#setting up variables X and y; in this case image and prediction (for images with no label)\ndef read_tfrecord(example, labeled):\n    tfrecord_format = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"target\": tf.io.FixedLenFeature([], tf.int64)\n    } if labeled else {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"image_name\": tf.io.FixedLenFeature([], tf.string)\n    }\n    example = tf.io.parse_single_example(example, tfrecord_format)\n    image = decode_image(example['image'])\n    if labeled:\n        label = tf.cast(example['target'], tf.int32)\n        return image, label\n    idnum = example['image_name']\n    return image, idnum","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# the following code will load the dataset using the TPU\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False # disable order, increase speed\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTOTUNE) # automatically interleaves reads from multiple files\n    dataset = dataset.with_options(ignore_order) # uses data as soon as it streams in, rather than in its original order\n    dataset = dataset.map(partial(read_tfrecord, labeled=labeled), num_parallel_calls=AUTOTUNE)\n    return dataset","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#The following functions will be used to load our training, validation, and test datasets, as well as print out the number of images in each dataset.\ndef get_training_dataset():\n    dataset = load_dataset(TRAINING_FILENAMES, labeled=True)  \n#     dataset = dataset.map(data_augment, num_parallel_calls=AUTOTUNE)  \n    dataset = dataset.repeat()\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTOTUNE)\n    return dataset\n\n\ndef get_validation_dataset(ordered=False):\n    dataset = load_dataset(VALID_FILENAMES, labeled=True, ordered=ordered) \n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(AUTOTUNE)\n    return dataset\n\n\n\ndef get_test_dataset(ordered=False):\n    dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTOTUNE)\n    return dataset\n\n\ndef count_data_items(filenames):\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) for filename in filenames]\n    return np.sum(n)\n\n\n\nNUM_TRAINING_IMAGES = count_data_items(TRAINING_FILENAMES)\nNUM_VALIDATION_IMAGES = count_data_items(VALID_FILENAMES)\nNUM_TEST_IMAGES = count_data_items(TEST_FILENAMES)\n\n\nprint('Dataset: {} training images, {} validation images, {} (unlabeled) test images'.format(\n    NUM_TRAINING_IMAGES, NUM_VALIDATION_IMAGES, NUM_TEST_IMAGES))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loss Function (for loading the model)","metadata":{}},{"cell_type":"code","source":"# import module we'll need to import our custom module\nfrom shutil import copyfile\n\n# copy our file into the working directory (make sure it has .py suffix)\ncopyfile(src = \"../input/bitempered-logistic-loss-tensorflow-v2/bi_tempered_loss.py\", dst = \"../working/loss.py\")\n\n# import all our functions\nfrom loss import bi_tempered_logistic_loss","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":" class BiTemperedLogisticLoss(tf.keras.losses.Loss):\n    def __init__(self, t1, t2, lbl_smth, n_iter):\n      super(BiTemperedLogisticLoss, self).__init__()\n      self.t1 = t1\n      self.t2 = t2\n      self.lbl_smth = lbl_smth\n      self.n_iter = n_iter\n\n    def call(self, y_true, y_pred):\n      return bi_tempered_logistic_loss(y_pred, y_true, self.t1, self.t2, self.lbl_smth, self.n_iter)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loading the model","metadata":{}},{"cell_type":"markdown","source":"FOR ENSEMBLENCE","metadata":{}},{"cell_type":"code","source":"from tensorflow import keras\n\nimport efficientnet.keras as efn\nfrom keras import models  \nfrom keras.models import load_model","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# nets = 5 \n# model = [0] * nets\n\n# for index in range (nets):\n#     model[index] = load_model('/kaggle/input/bitemp-ensemblence-2-dense/EfficentNetB3_last_fold_{}_.h5'.format(index),\n#                    custom_objects={'loss': BiTemperedLogisticLoss(t1=t_1, \n#                                                                   t2=t_2, \n#                                                                   lbl_smth=SMOOTH_FRACTION, \n#                                                                   n_iter=N_ITER)}, \n#                    compile=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"FOR SINGLE MODEL","metadata":{}},{"cell_type":"code","source":"model = load_model('/kaggle/input/modelcomparisingbest-and-last/EfficentNetB4_best_fold_0_.h5',\n                   custom_objects={'loss': BiTemperedLogisticLoss(t1=t_1, \n                                                                  t2=t_2, \n                                                                  lbl_smth=SMOOTH_FRACTION, \n                                                                  n_iter=N_ITER)}, \n                   compile=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preparing predictions","metadata":{}},{"cell_type":"code","source":"def to_float32(image, label):\n    return tf.cast(image, tf.float32), label\n\ntest_dataset = get_test_dataset()\n\ntest_ds = get_test_dataset(ordered=True) \ntest_ds = test_ds.map(to_float32)\n\ntest_images_ds = test_dataset\ntest_images_ds = test_ds.map(lambda image, idnum: image)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Predictions","metadata":{}},{"cell_type":"markdown","source":"FOR ENSEMBLENCE","metadata":{}},{"cell_type":"code","source":"# #Result\n# results = np.zeros( (int(NUM_TEST_IMAGES),5) ) \n\n# for index in range(nets):\n#     #each model gives its predictions\n#     results = results + model[index].predict(test_images_ds)\n \n# #Finding the best prediction \n# predictions = np.argmax(results,axis = 1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"FOR SINGLE MODELS","metadata":{}},{"cell_type":"code","source":"probabilities = model.predict(test_images_ds)\npredictions = np.argmax(probabilities, axis=-1)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission File","metadata":{}},{"cell_type":"code","source":"print('Generating submission.csv file...')\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(NUM_TEST_IMAGES))).numpy().astype('U') # all in one batch\nnp.savetxt('submission.csv', np.rec.fromarrays([test_ids, predictions]), fmt=['%s', '%d'], delimiter=',', header='image_id,label', comments='')\n!head submission.csv","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}