{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport tensorflow.keras.layers as layers\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.experimental import CosineDecay\nfrom tensorflow.keras.callbacks import ModelCheckpoint","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#ls ../input\ntraining_mode = False\npreviously_trained_model_file =  '../input/best-model/best_model.h5'\nanswer_to_life = 42\nfirst_time_train = False","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"project_folder = '../input/cassava-leaf-disease-classification'\ndf = pd.read_csv(f'{project_folder}/train.csv')\ndf['image_path'] = df['image_id'].apply(lambda x : f'{project_folder}/train_images/{x}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from PIL import Image\nimage1 = Image.open(df['image_path'].tolist()[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\ndef get_random_crops(img):\n    imgs = []\n    for i in range(5):\n        cropped_img = tf.image.random_crop(img, size = [512,512,3])\n        imgs.append(cropped_img)\n    return imgs\n        \nimages = get_random_crops(np.array(image1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_augmented_images(img):\n    img = np.array(img)\n    imgs = []\n    cropped_images = get_random_crops(img)\n    for cropped_img in cropped_images:\n        \n\n        horizontal_flip = np.random.rand() > 0.5\n        vertical_flip   = np.random.rand() > 0.5\n        \n        aug_img = tf.keras.preprocessing.image.random_shear(cropped_img, 0.20)\n        if horizontal_flip: aug_img = tf.image.flip_left_right(aug_img)\n        if vertical_flip: aug_img = tf.image.flip_up_down(aug_img)\n            \n            \n        imgs.append(aug_img)\n    return imgs\n        ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images = get_augmented_images(image1)\ntmp_img = image1.resize((512,512))\ntmp_img = np.array(tmp_img)\nimages.append(tmp_img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.array(images).shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nf, axarr = plt.subplots(1,6, figsize=(20,20)) \nfor i in range(6):\n    axarr[i].imshow(images[i])\n#plt.imshow(images[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"datagen = ImageDataGenerator(rescale = 1.0/255.0, horizontal_flip = True,shear_range = 0.2, rotation_range=25,\n                                channel_shift_range =0.2,zoom_range=0.2, height_shift_range =0.2, \n                                     vertical_flip = True, validation_split = 0.2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['label'] = df['label'].astype(str)\nimg_size = 512\n\ntrain_datagen = datagen.flow_from_dataframe(df, x_col = 'image_path', y_col = 'label', batch_size = 8, \n                                               class_mode = 'categorical', target_size = (img_size, img_size),\n                                                  seed = answer_to_life, subset = 'training')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"val_datagen = datagen.flow_from_dataframe(df, x_col = 'image_path', y_col = 'label', batch_size = 8, \n                                            class_mode = 'categorical', target_size = (img_size, img_size),\n                                              seed = answer_to_life, subset = 'validation')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"if training_mode:\n    def create_pretrained_model():\n\n        pretrained_model = tf.keras.applications.EfficientNetB3(weights = '../input/efficientnet-model-file/efficientnetb3_notop.h5', \n                                                                    include_top=False,drop_connect_rate=0.4,  \n                                                                        input_shape=(img_size, img_size, 3))\n        x = layers.GlobalAveragePooling2D()(pretrained_model.output)\n\n        x = layers.Dense(512, activation= 'relu')(x)\n        x = layers.Dropout(0.4)(x)\n        x = layers.Dense(5, activation='softmax')(x)\n\n\n        model = tf.keras.models.Model(inputs=pretrained_model.input, outputs=x)\n\n        decay_steps = int(round(17118.0/8.0))*3\n        cosine_decay = CosineDecay(initial_learning_rate=1e-4, decay_steps=decay_steps, alpha=0.3)\n\n        callbacks = [ModelCheckpoint(filepath='best_model.h5', monitor='val_loss', save_best_only=True)]\n\n        model.compile(optimizer=tf.keras.optimizers.Adam(cosine_decay), loss='categorical_crossentropy', metrics=['accuracy'])\n        return model\n\n\n    model     = create_pretrained_model()\n    callbacks = [ModelCheckpoint(filepath='best_model_final.h5', monitor='val_loss', save_best_only=True)]\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\n\nif not first_time_train:\n    model = tf.keras.models.load_model(previously_trained_model_file)\n    \nif training_mode:\n    #model = tf.keras.models.load_model(previously_trained_model_file)\n    history   = model.fit(train_datagen, epochs = 6, validation_data = val_datagen, callbacks = callbacks)  \n\n#model.save('best_model_final.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#model.save('last_model.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#history","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from PIL import Image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_image_path = df['image_path'].tolist()[4]\nimg = Image.open(test_image_path)\nimg = img.resize((512,512))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimg = np.array(img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img = img / 255.0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"imgs = get_random_crops(img)\nimgs = np.array(imgs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = model.predict(imgs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = np.argmax(np.sum(preds, axis=0))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_folder = '../input/cassava-leaf-disease-classification/test_images'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\ntest_files = os.listdir(test_folder)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_files","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df = pd.DataFrame()\nimage_names = []\npredictions = []\n\nfor img_name in test_files:\n\n    tmp_img = Image.open(f'{test_folder}/{img_name}')\n    \n    aug_imgs = get_augmented_images(tmp_img)\n    \n    tmp_img = tmp_img.resize((512,512))\n    tmp_img = np.array(tmp_img)\n    \n    aug_imgs.append(tmp_img)\n    imgs    = np.array(aug_imgs)\n    \n    imgs    = imgs/ 255.0\n    \n    \n    \n    preds   = model.predict(imgs)\n    prediction = np.argmax(np.sum(preds, axis=0))\n    image_names.append(img_name)\n    predictions.append(prediction)\n    \nsubmission_df = pd.DataFrame( data = {'image_id' : image_names, 'label' : predictions})\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"         ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_df.to_csv(\"submission.csv\", index=False) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}