{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow import keras\nfrom tensorflow.keras import models\nfrom tensorflow.keras import layers","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tf.config.list_physical_devices('GPU')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_path = '../input/cassava-leaf-disease-classification/'\ntrain_df = pd.read_csv(data_path + 'train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df['label'] = train_df['label'].map(lambda lbl: str(lbl))\ntrain_df['image_id'] = train_df['image_id'].map(lambda img: data_path + 'train_images/' + img)\ntrain_df.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_train, x_val, y_train, y_val  = train_test_split(\n    train_df['image_id'], train_df['label'], random_state = 0, shuffle = True\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data = pd.merge(x_train, y_train, right_index = True, left_index = True)\ntrain_data.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"val_data = pd.merge(x_val, y_val, right_index = True, left_index = True)\nval_data.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Size of training set: ' + str(train_data.size))\nprint('Size of validation set: ' + str(val_data.size))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_datagen = ImageDataGenerator(\n    rescale = 1./255,\n    rotation_range = 45,\n    shear_range = 0.2,\n    zoom_range = 0.2,\n    horizontal_flip = True,\n    width_shift_range = 0.2,\n    height_shift_range = 0.2,\n    fill_mode = 'nearest'\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"val_datagen = ImageDataGenerator(rescale = 1./255)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_generator = train_datagen.flow_from_dataframe(\n    train_data, directory = None, x_col = 'image_id', y_col = 'label',\n    batch_size = 32, class_mode = 'categorical', target_size = (224, 224)\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"val_generator = val_datagen.flow_from_dataframe(\n    val_data, directory = None, x_col = 'image_id', y_col = 'label',\n    batch_size = 32, class_mode = 'categorical', target_size = (224, 224)\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"base_model = keras.applications.ResNet50(\n    include_top = False, \n    weights = '../input/pretrained-resnet50-weights/resnet50_weights_tf_dim_ordering_tf_kernels_notop.h5', \n    input_shape = (224, 224, 3)\n)\n\n#base_model.trainable = False","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"base_model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = models.Sequential()\nmodel.add(base_model)\nmodel.add(layers.Flatten())\nmodel.add(layers.Dense(256, activation = 'relu'))\nmodel.add(layers.Dropout(0.2))\nmodel.add(layers.Dense(5, activation = 'softmax'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(loss = 'categorical_crossentropy',\n             optimizer = 'adam',\n             metrics = ['accuracy']\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"rlr_callback = keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=5)       \ncheckpoint = keras.callbacks.ModelCheckpoint('model.h5',monitor = 'val_accuracy',\n                      verbose = 0, save_best_only = True, mode = 'max')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"history =  model.fit(\n    train_generator, callbacks = [rlr_callback, checkpoint], \n    validation_data = val_generator,\n    #steps_per_epoch = 100,\n    epochs = 50,\n    #validation_steps = 50\n    batch_size = 256\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.save('./model.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\ntrain_accuracy = history.history['accuracy']\ntrain_loss = history.history['loss']\nval_accuracy = history.history['val_accuracy']\nval_loss = history.history['val_loss']\n\nepochs = range(len(train_accuracy))\n\nplt.plot(epochs, train_accuracy, 'r', label = 'Training Accuracy')\nplt.plot(epochs, val_accuracy, 'b', label = 'Validation Accuracy')\nplt.title('Training and Validation Accuracy')\nplt.figure()\n\nplt.plot(epochs, train_loss, 'r', label = 'Training Loss')\nplt.plot(epochs, val_loss, 'b', label = 'Validation Loss')\nplt.title('Training and Validation Loss')\nplt.figure()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''from PIL import Image\ntest_img_path = '../input/cassava-leaf-disease-classification/test_images/'\nfile_names = os.listdir(test_img_path)\nfor img in file_names:\n    image = Image.open(test_img_path + img)\n    image = image.resize((224, 224))\n    image = np.expand_dims(image, axis = 0)\n    prediction = model.predict(image)\n\nprediction = np.argmax(prediction)\nprediction'''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''result_df = pd.DataFrame(file_names, columns = ['image_id'])\nresult_df['label'] = prediction\nresult_df'''","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}