{"cells":[{"metadata":{},"cell_type":"markdown","source":"## About this kernel\n\nCredited to @Nayu's kernel. \n\nI changed it with tf2.x codes and support TPU. For fast training, I resized the images into 64*64. <br>\nLet your TPU burn..."},{"metadata":{},"cell_type":"markdown","source":"I refered following kernels, thank you!\n\nhttps://www.kaggle.com/ateplyuk/inat2019-starter-keras-efficientnet/data\n\nhttps://www.kaggle.com/mobassir/keras-efficientnetb2-for-classifying-cloud\n\nhttps://www.kaggle.com/mgornergoogle/getting-started-with-100-flowers-on-tpu\n"},{"metadata":{},"cell_type":"markdown","source":"**Example of Fine-tuning from pretrained model using Keras  and Efficientnet (https://pypi.org/project/efficientnet/).**"},{"metadata":{"trusted":true},"cell_type":"code","source":"import os, glob\nimport random\nfrom sklearn.model_selection import train_test_split\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport multiprocessing\nfrom copy import deepcopy\nfrom sklearn.metrics import precision_recall_curve, auc\nfrom kaggle_datasets import KaggleDatasets\nimport tensorflow as tf\nimport tensorflow.keras.layers as L\nimport tensorflow.keras\nimport tensorflow.keras.backend as K\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.applications.densenet import DenseNet201\nfrom tensorflow.keras.layers import Dense, Flatten, Activation, Dropout, GlobalAveragePooling2D\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras import optimizers, applications\nfrom tensorflow.keras.models import Model, load_model\nfrom tensorflow.keras.callbacks import Callback, ModelCheckpoint, LearningRateScheduler, TensorBoard, EarlyStopping\nfrom tensorflow.keras.utils import Sequence\nimport matplotlib.pyplot as plt\nfrom IPython.display import Image\nfrom tqdm import tqdm_notebook as tqdm\nimport json\nimport os\nimport gc\nfrom numpy.random import seed\nseed(10)\n\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install -q efficientnet\nimport efficientnet.tfkeras as efn","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# TPU setup"},{"metadata":{"trusted":true},"cell_type":"code","source":"os.listdir('../input')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nAUTO = tf.data.experimental.AUTOTUNE\ntry:\n    # Create strategy from tpu\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\n    print('tpu:',tpu)\nexcept:\n    print('no tpu.....')\n    strategy=None\n    tpu=None\n\n\n\n# Data access\nif tpu:\n    GCS_DS_PATH = KaggleDatasets().get_gcs_path('iwildcam-2020-fgvc7')\n    tf_records_path= KaggleDatasets().get_gcs_path('iwildcam2020-64-tf-records')\n    tf_records_path2=KaggleDatasets().get_gcs_path('iwildcam2020-64-tf-records')\n    test_tf_records_path=KaggleDatasets().get_gcs_path('iwildcam2020-64-tf-records')\n    #If you want to use the best speed of TPU,then the batch_size should be times of 16. \n    #Since TPU V3-8 has 8 cores,so 16*8\n    BATCH_SIZE = 16 * strategy.num_replicas_in_sync\n\n# Configuration\nEPOCHS = 30#3\nimg_size = 64#96\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TRAINING_FILENAMES = tf.io.gfile.glob(tf_records_path + '/*train.rec')\nTRAINING_FILENAMES.extend(tf.io.gfile.glob(tf_records_path2 + '/*zero.rec'))\nTRAINING_FILENAMES","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TEST_FILENAMES=tf.io.gfile.glob(test_tf_records_path + '/*test.rec')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Train data"},{"metadata":{},"cell_type":"markdown","source":"## Data processing functions:"},{"metadata":{"trusted":true},"cell_type":"code","source":"\ndef decode_image(image):\n    image = tf.image.decode_jpeg(image, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0  # convert image to floats in [0, 1] range\n    image = tf.reshape(image, [img_size,img_size, 3]) # explicit size needed for TPU\n    return image\n\ndef data_augment(image, label=None):\n    image = tf.image.random_flip_left_right(image)\n#     image = tf.image.random_flip_up_down(image)\n    \n    if label is None:\n        return image\n    else:\n        return image, label\n    \n    \ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), # tf.string means bytestring\n        \"class\": tf.io.FixedLenFeature([], tf.int64),  # shape [] means single element\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label # returns a dataset of (image, label) pairs\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), # tf.string means bytestring\n        \"id\": tf.io.FixedLenFeature([], tf.string),  # shape [] means single element\n        # class is missing, this competitions's challenge is to predict flower classes for the test dataset\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum # returns a dataset of image(s)\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    # Read from TFRecords. For optimal performance, reading from multiple files at once and\n    # disregarding data order. Order does not matter since we will be shuffling the data anyway.\n\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False # disable order, increase speed\n\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO) # automatically interleaves reads from multiple files\n    dataset = dataset.with_options(ignore_order) # uses data as soon as it streams in, rather than in its original order\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, num_parallel_calls=AUTO)\n    # returns a dataset of (image, label) pairs if labeled=True or (image, id) pairs if labeled=False\n    return dataset\n\n\ndef get_training_dataset():\n    dataset = load_dataset(TRAINING_FILENAMES, labeled=True)\n    dataset = dataset.map(data_augment, num_parallel_calls=AUTO)\n    dataset = dataset.repeat() # the training dataset must repeat for several epochs\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO) # prefetch next batch while training (autotune prefetch buffer size)\n    return dataset\n\ndef get_validation_dataset(ordered=False):\n    dataset = load_dataset(VALIDATION_FILENAMES, labeled=True, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(AUTO) # prefetch next batch while training (autotune prefetch buffer size)\n    return dataset\n\ndef get_test_dataset(ordered=False):\n    dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO) # prefetch next batch while training (autotune prefetch buffer size)\n    return dataset","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_dataset=get_training_dataset()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub_df = pd.read_csv('../input/iwildcam-2020-fgvc7/sample_submission.csv')\nsub_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_dataset=get_test_dataset()\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Model"},{"metadata":{"trusted":true},"cell_type":"code","source":"with strategy.scope():\n    pretrained_model = efn.EfficientNetB7(\n        input_shape=(img_size, img_size, 3),\n        #weights='noisy-student',\n        weights='imagenet',\n        include_top=False\n    )\n    #pretrained_model.trainable = False\n    pretrained_model.trainable = True\n    model = tf.keras.Sequential([\n        pretrained_model,\n        L.GlobalAveragePooling2D(),\n        L.Dense(216, activation='softmax')#573\n    ])\n\n    model.compile(\n        optimizer='adam',\n         loss = 'sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n    model.summary()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"scheduler = tf.keras.callbacks.ReduceLROnPlateau(patience=3, verbose=1)\n\nLR_START = 0.00001\nLR_MAX = 0.00005 * strategy.num_replicas_in_sync\nLR_MIN = 0.00001\nLR_RAMPUP_EPOCHS = 5\nLR_SUSTAIN_EPOCHS = 0\nLR_EXP_DECAY = .8\n\ndef lrfn(epoch):\n    if epoch < LR_RAMPUP_EPOCHS:\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        lr = LR_MAX\n    else:\n        lr = (LR_MAX - LR_MIN) * LR_EXP_DECAY**(epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS) + LR_MIN\n    return lr\n    \nlr_callback = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=True)\n\nrng = [i for i in range(EPOCHS)]\ny = [lrfn(x) for x in rng]\nplt.plot(rng, y)\nprint(\"Learning rate schedule: {:.3g} to {:.3g} to {:.3g}\".format(y[0], max(y), y[-1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Callbacks\nearly = EarlyStopping(monitor='val_loss', min_delta=0, patience=3, verbose=1, mode='auto')\nbest_checkpoint='model.h5'\ncheckpoint = ModelCheckpoint(\n    best_checkpoint, \n    monitor='val_accuracy', \n    verbose=1, \n    #save_best_only=True, \n    save_weights_only=True,\n    mode='auto'\n)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Train"},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n# STEPS_PER_EPOCH = train_x.shape[0] // BATCH_SIZE\nSTEPS_PER_EPOCH=(143736+3709) // BATCH_SIZE\nhistory = model.fit(\n    train_dataset, \n    epochs=EPOCHS, \n    callbacks=[early,checkpoint],\n    steps_per_epoch=STEPS_PER_EPOCH,\n#     validation_data=valid_dataset\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import gc\n\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Test data"},{"metadata":{"trusted":true},"cell_type":"code","source":"sam_sub_df=sub_df.copy()\nsam_sub_df[\"file_name\"] = sam_sub_df[\"Id\"].map(lambda str : str + \".jpg\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Prediction"},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n#model = tf.keras.models.load_model(best_checkpoint)\nmodel.load_weights(best_checkpoint)\n\npredict=model.predict(test_dataset, verbose=1).astype(float)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(len(predict))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predicted_class_indices=np.argmax(predict,axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predicted_class_indices","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import pickle\nwith open('../input/iwildcam2020-classes-dict/cid_invert_dict.pkl', mode='rb') as fin:\n    cid_invert_dict=pickle.load(fin)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def transform(x):\n    return cid_invert_dict[str(x)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sam_sub_df[\"Category\"] = predicted_class_indices\nsam_sub_df[\"Category\"]=sam_sub_df[\"Category\"].apply(transform)\n\n\n         \nsam_sub_df = sam_sub_df.loc[:,[\"Id\", \"Category\"]]\nsam_sub_df.to_csv(\"submission.csv\",index=False)\nsam_sub_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.6"}},"nbformat":4,"nbformat_minor":4}