{"cells":[{"metadata":{"id":"arK14FWEx4jf","outputId":"930b73af-cde9-45fa-8e01-0d89e37a69cd","trusted":true},"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/content/data'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install efficientnet -q","execution_count":null,"outputs":[]},{"metadata":{"id":"HKczwJ-PxyC8","outputId":"ab18c737-0396-4ed2-9e2d-28b2a553d568","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport tensorflow as tf\nimport tensorflow_hub as hub\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport shutil\nimport glob\nfrom functools import partial\nimport IPython.display as display\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import efficientnet.keras","execution_count":null,"outputs":[]},{"metadata":{"id":"_615hszxDc-V","outputId":"901b4d19-2153-4005-a4f8-0636c77d4010","trusted":true},"cell_type":"code","source":"tf.__version__","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tpu_name = os.getenv('TPU_NAME')","execution_count":null,"outputs":[]},{"metadata":{"id":"vX8zLKUQxyC_","outputId":"fb9c6c1c-3486-4119-dc76-4dec53e90ab9","trusted":true},"cell_type":"code","source":"try: # detect TPUs\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() # TPU detection\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept ValueError: # no TPU found, detect GPUs\n    strategy = tf.distribute.MirroredStrategy() # for GPU or multi-GPU machines\n    #strategy = tf.distribute.get_strategy() # default strategy that works on CPU and single GPU\n    #strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy() # for clusters of multi-GPU machines\n\nprint(\"Number of accelerators: \", strategy.num_replicas_in_sync)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from kaggle_datasets import KaggleDatasets\nGCS_PATH = KaggleDatasets().get_gcs_path('cassava-leaf-disease-classification')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"GCS_PATH","execution_count":null,"outputs":[]},{"metadata":{"id":"727g5_PameAY","trusted":true},"cell_type":"code","source":"AUTOTUNE = tf.data.experimental.AUTOTUNE\nIMAGE_SIZE = [512,512]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"AUTOTUNE","execution_count":null,"outputs":[]},{"metadata":{"id":"knLpgOBBxyC_","outputId":"af08f21f-f025-4c91-f120-a95d64f6e623","trusted":true},"cell_type":"code","source":"train_records = pd.read_csv('../input/cassava-leaf-disease-classification/train.csv')\ntrain_records.head()","execution_count":null,"outputs":[]},{"metadata":{"id":"6KVDv85gKaet","trusted":true},"cell_type":"code","source":"data_value_c = pd.DataFrame(train_records['label'].value_counts())","execution_count":null,"outputs":[]},{"metadata":{"id":"9CwtqWbZldwq","outputId":"83697c89-c3bb-4055-f254-97f87352bd60","trusted":true},"cell_type":"code","source":"sns.barplot(data_value_c.index,data_value_c.label)","execution_count":null,"outputs":[]},{"metadata":{"id":"iPIhVmwszmrz","trusted":true},"cell_type":"code","source":"class_weights = train_records.groupby(['label']).count()\nclass_weights['image_id'] = class_weights.image_id/train_records.shape[0]","execution_count":null,"outputs":[]},{"metadata":{"id":"zE4Uc2s80Fil","trusted":true},"cell_type":"code","source":"cl_w = class_weights.image_id.to_dict()","execution_count":null,"outputs":[]},{"metadata":{"id":"Vjn98WujxyDA","trusted":true},"cell_type":"code","source":"def submit_gen(model,classes,dataset_path,col='image_id'):\n    test_data = os.listdir(dataset_path)\n    pred_array=[]\n    for i in test_data:\n        img = tf.keras.preprocessing.image.load_img(\n            dataset_path+i,target_size=(512,512))\n        img_array = tf.keras.preprocessing.image.img_to_array(img)\n        img_array = tf.expand_dims(img_array, 0)\n        prediction = np.squeeze(model.predict(img_array))\n        name_c = classes[np.argmax(prediction)]\n        pred_array.append(name_c)\n    return pred_array,test_data","execution_count":null,"outputs":[]},{"metadata":{"id":"TstP7vTC3MXS","outputId":"ad99e23d-209f-4ff9-f4c2-5ef3a9001f2a","trusted":true},"cell_type":"code","source":"gs_filenames = tf.io.gfile.glob(GCS_PATH + \"/train_tfrecords/*.tfrec\")\ngs_filenames","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"split_ind = int(0.9 * len(gs_filenames))\nTRAINING_FILENAMES, VALID_FILENAMES = gs_filenames[:split_ind], gs_filenames[split_ind:]","execution_count":null,"outputs":[]},{"metadata":{"id":"JfMwErjgEJmU","trusted":true},"cell_type":"code","source":"dataset = tf.data.TFRecordDataset(filenames=TRAINING_FILENAMES)\nval_dataset = tf.data.TFRecordDataset(filenames=VALID_FILENAMES)","execution_count":null,"outputs":[]},{"metadata":{"id":"xFR5Gg_ALmpE","trusted":true},"cell_type":"code","source":"def augment(image):\n    image = tf.io.decode_jpeg(image,channels=3)\n    image = tf.image.random_flip_left_right(image,seed=2)\n    image = tf.keras.preprocessing.image.img_to_array(image)\n    #image = tf.reshape(image,(3,512,512))\n    image = tf.keras.preprocessing.image.random_rotation(image,35,row_axis=1 ,col_axis=0, channel_axis=2)\n    image = tf.keras.preprocessing.image.random_shear(image,5,row_axis=1 ,col_axis=0, channel_axis=2)\n    image = tf.keras.preprocessing.image.random_shift(image,0.1,0.1,row_axis=1 ,col_axis=0, channel_axis=2)\n    #image = tf.reshape(image,(*IMAGE_SIZE,3))\n    #image = tf.keras.preprocessing.image.array_to_img(image)\n    return image","execution_count":null,"outputs":[]},{"metadata":{"id":"677I809zUioo","trusted":true},"cell_type":"code","source":"image_feature_description = {\n    'image': tf.io.FixedLenFeature([], tf.string),\n    'image_name': tf.io.FixedLenFeature([], tf.string),\n    'target': tf.io.FixedLenFeature([], tf.int64),\n    \n}\n\ndef _parse_image_function(example_proto):\n    # Parse the input tf.train.Example proto using the dictionary above.\n    example = tf.io.parse_single_example(example_proto, image_feature_description)\n    image = tf.image.decode_jpeg(example['image'],channels=3)\n    image = tf.cast(image, tf.float32)\n    image = tf.reshape(image,[*IMAGE_SIZE,3])\n    label = tf.cast(example[\"target\"], tf.int32)\n    return image,label","execution_count":null,"outputs":[]},{"metadata":{"id":"_QdqqDxOUwAi","trusted":true},"cell_type":"code","source":"parsed_image_dataset = dataset.map(partial( _parse_image_function),num_parallel_calls=AUTOTUNE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"val_image_dataset = val_dataset.map(partial( _parse_image_function),num_parallel_calls=AUTOTUNE)","execution_count":null,"outputs":[]},{"metadata":{"id":"X5aNog8bM0OW","outputId":"52e43ac1-a43f-4cbf-e86f-38c849490ef2","trusted":true},"cell_type":"code","source":"for i,x in parsed_image_dataset.take(1):\n    print(i.shape)","execution_count":null,"outputs":[]},{"metadata":{"id":"N9UsIQiYASdC","trusted":true},"cell_type":"code","source":"data_augmentation = tf.keras.Sequential([\n    tf.keras.layers.experimental.preprocessing.RandomFlip(\"horizontal\"),\n    tf.keras.layers.experimental.preprocessing.RandomRotation((0.2,0.3)),\n    #tf.keras.layers.experimental.preprocessing.RandomZoom(height_factor=0.2,),\n    #tf.keras.layers.experimental.preprocessing.Rescaling(scale=1./255)\n])","execution_count":null,"outputs":[]},{"metadata":{"id":"w_A7ezFTATIR","outputId":"a1954f73-b831-4dff-ea32-b0c5bce18010","trusted":true},"cell_type":"code","source":"train_aug_ds = parsed_image_dataset.filter(lambda x,y: tf.math.not_equal(y,3))\nfor i,x in train_aug_ds.take(1):\n    print(i.shape)\nparsed_image_dataset = parsed_image_dataset.concatenate(train_aug_ds)","execution_count":null,"outputs":[]},{"metadata":{"id":"r69P3MUDk-Ea","trusted":true},"cell_type":"code","source":"def create_dataset(dataset,aug=False):    \n    # Set the number of datapoints you want to load and shuffle \n    dataset = dataset.shuffle(4096)\n    dataset = dataset.prefetch(buffer_size=AUTOTUNE)\n    # Set the batchsize\n        \n    dataset = dataset.batch(96)\n    if(aug==True):\n        dataset=dataset.map(lambda x,y:(data_augmentation(x),y))\n\n    return dataset","execution_count":null,"outputs":[]},{"metadata":{"id":"ADqnIjkdeFkz","trusted":true},"cell_type":"code","source":"train_dataset = create_dataset(parsed_image_dataset,False)\n#train_dataset = train_dataset.map(lambda x,y:(data_augmentation(x),y))\nvalidation_dataset = create_dataset(val_image_dataset)","execution_count":null,"outputs":[]},{"metadata":{"id":"mP00kRXTxyDA"},"cell_type":"markdown","source":"# Preprocessing - Tensorflow"},{"metadata":{"id":"F0kvwDD1xyDB"},"cell_type":"markdown","source":"# Model - Tensorflow"},{"metadata":{"trusted":true},"cell_type":"code","source":"MODEL_WEIGHTS = '../input/efficientnet-keras-weights-b0b5/efficientnet-b3_imagenet_1000_notop.h5'","execution_count":null,"outputs":[]},{"metadata":{"id":"xHQRQbU_xyDB","outputId":"d4320637-a7dc-4f21-cd8d-89b2dae84ed3","trusted":true},"cell_type":"code","source":"with strategy.scope():\n    base = efficientnet.keras.EfficientNetB3(include_top=False,input_shape=(*IMAGE_SIZE,3),\n                                                weights=MODEL_WEIGHTS\n                                            )\n    base.trainable = False\n    inputs = tf.keras.layers.Input([*IMAGE_SIZE, 3])\n    pre_layer = efficientnet.keras.preprocess_input(inputs)#.get_layer(\n    base_model = base(pre_layer)\n    #base_model = tf.keras.layers.GlobalAveragePooling2D()(base_model)\n    base_model = tf.keras.layers.Flatten()(base_model)\n    base_model = tf.keras.layers.Dropout(0.3)(base_model)\n    base_model = tf.keras.layers.Dense(64,activation='relu')(base_model)\n    base_model = tf.keras.layers.Dense(32,activation='relu')(base_model)\n    base_model = tf.keras.layers.Dense(5,activation='softmax')(base_model)\n    \n    model_custom = tf.keras.Model(inputs=inputs,outputs=base_model)\n","execution_count":null,"outputs":[]},{"metadata":{"id":"SVDIFZy1xyDF","outputId":"8e69dc76-c6e9-4372-cdeb-a01185e511e8","trusted":true},"cell_type":"code","source":"model_custom.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"initial_learning_rate = 0.001\nlr_schedule = tf.keras.optimizers.schedules.ExponentialDecay(\n    initial_learning_rate, decay_steps=100, decay_rate=0.96, staircase=True\n)\n\ncheckpoint_cb = tf.keras.callbacks.ModelCheckpoint(\n    \"./b3_model.h5\", save_best_only=True\n)\n\nearly_stopping_cb = tf.keras.callbacks.EarlyStopping(\n    patience=10, restore_best_weights=True\n)","execution_count":null,"outputs":[]},{"metadata":{"id":"mbpJMLYbEwMn","trusted":true},"cell_type":"code","source":"model_custom.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=1e-5),\n              loss=tf.keras.losses.SparseCategoricalCrossentropy(),\n              metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"id":"MzyU31ngxyDF","outputId":"6e2edc6c-93f6-46e5-f315-eec1ac463ebe","trusted":true},"cell_type":"code","source":"epochs=50\nhistory = model_custom.fit(\n  train_dataset,\n  epochs=epochs,\n  validation_data = validation_dataset,\ncallbacks=[checkpoint_cb,early_stopping_cb ],\n \n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_custom.optimizer._decayed_lr(tf.float32).numpy()","execution_count":null,"outputs":[]},{"metadata":{"id":"w6sQPv5uwn9O","trusted":true,"collapsed":true},"cell_type":"code","source":"model_custom.save('./b3_model/')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Prediction"},{"metadata":{"trusted":true},"cell_type":"code","source":"model = tf.keras.models.load_model('./b4_model.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"id":"iH-4SG4fxyDF","trusted":true},"cell_type":"code","source":"prediction,test_data = submit_gen(model,\n                                  classes=sorted(train_records.label.unique()),\n                                  dataset_path='../input/cassava-leaf-disease-classification/test_images/')","execution_count":null,"outputs":[]},{"metadata":{"id":"vUWWuRp6xyDF","outputId":"bd271157-3c96-4d83-f11b-cfd21553f5ea","trusted":true},"cell_type":"code","source":"submission = pd.DataFrame({\"image_id\":np.squeeze(test_data),\"label\":prediction})\nsubmission.head()","execution_count":null,"outputs":[]},{"metadata":{"id":"noB1fgOuxyDF","trusted":true},"cell_type":"code","source":"submission.to_csv('./submission.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.read_csv('../input/cassava-leaf-disease-classification/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}