---
title: 'Cassava Disease - Python and R EDA'
author: 'Thomas Meli'
date: '`r Sys.Date()`'
output:
  html_document:
    fig_width: 6
    fig_height: 4
    fig_caption: true
    number_sections: true
    toc: true
    theme: cosmo
    highlight: tango
    code_folding: hide
---

```{r setup, include=FALSE, echo=FALSE}

set.seed(10)

```
# Compact Form - Click on Headers below to Reveal Code and Outputs {.tabset}

# Problem Description and Evaluation Metric {.tabset}

## Problem Description {.tabset}

**Identifying Cassava Leaf Disease Classification**

In plain English, this is an **image classification problem** where we have to train a model to classify the disease type of the cassava plant.


## Evaluation Metric - Accuracy {.tabset}

This competition is measured by the **accuracy metric** which may be biased towards the most prevalent category.  Therefore it will be very important to get a sense of what the distributions of the train, validation, and test sets to account for class imbalances.

# Environment and Modules {.tabset}

## About this Document {.tabset}

This file is a **RMarkdown Script**.  You can make one by clicking RMarkdown and Script when you create a new notebook.  

I am using the **Reticulate** package to work both with Python and R in the same script.

These tabs show the virtual environment and packages and versions imported for **both Python and R**. 

## R Modules {.tabset}

```{r load_r_packages}

library("reticulate")
Sys.which("python")
options(reticulate.traceback = TRUE)

py_install("pandas")
py_install("scikit-learn")
py_install("scikit-image")
py_install("matplotlib")
py_install("numpy")
py_install("seaborn")
#py_install("torch")
py_install("tensorflow")
# py_install("cv2")

library("tidyverse") 
library("reactable")
library("magick")
library("skimr")
library("caret")
#library("neuralnet")
#library("mxnet")
#library("keras")

#library("corrplot")
library("data.table")
library("readr")
library("vroom")

```


## Python Imports {.tabset}

```{python load_python_modules}

import pandas as pd
import numpy as np
import numpy as np
import matplotlib.pyplot as plt
import seaborn as sns

from skimage.io import imread, imshow

from sklearn.metrics import accuracy_score
from sklearn.model_selection import train_test_split

# import torch

import tensorflow as tf
from tensorflow.keras import models, layers
from tensorflow.keras.preprocessing.image import ImageDataGenerator
from keras.callbacks import ModelCheckpoint, TensorBoard

from keras.layers import Dense, BatchNormalization, Dropout, Embedding
#from tensorflow_addons.layers import WeightNormalization
from keras.regularizers import l2


```

## Environment Config with Reticulate {.tabset}

```{r python_config}

reticulate::py_config()
sessionInfo()
try(conda_list(conda="auto"))
try(virtualenv_list())
try(virtualenv_root())

```

# Dataset Exploration {.tabset}


## Train.csv Data {.tabset}

```{r R_read_train}

base_path <- "../input/cassava-leaf-disease-classification/"

train <- data.table::fread(paste0(base_path, "train.csv"))
reactable(train[1:5, ])
skimr::skim(train)


```

## Sample Submission {.tabset}
 
```{r R_read_submission}

sample_submission <- data.table::fread(paste0(base_path, "sample_submission.csv"))
reactable(sample_submission[1:5, ])
skimr::skim(sample_submission)


```

## Distribution of Classes {.tabset}

The classes are unbalanced.  

```{r class_counts}

ggplot(train, aes(x = label)) + geom_bar()

```

## No Duplicate Observations {.tabset}
```{r duplicates}

dupes_train <- duplicated(train)
table(dupes_train)
# which(dupes_train == "TRUE")

```

# Image Processing in Python {.tabset}


## Image Samples From Each Label {.tabset}

### Label 0 {.tabset}

```{python load_image_set0}

import os

train_df = pd.read_csv("../input/cassava-leaf-disease-classification/train.csv")
BASE_DIR = "../input/cassava-leaf-disease-classification/"

def visualize_four(label_value):
    df_of_vals = train_df[train_df['label'] == label_value]

    tmp_df = df_of_vals.sample(4)

    image_ids = tmp_df["image_id"].values
    labels = tmp_df["label"].values
    
    print(f"Total train images for class {label_value}: {df_of_vals.shape[0]}")

    plt.figure(figsize=(16, 12))
    
    for ind, (image_id, label) in enumerate(zip(image_ids, labels)):
        plt.subplot(2, 2, ind + 1)
        image = imread(os.path.join(BASE_DIR, "train_images", image_id))

        plt.imshow(image)
        plt.title(f"Class: {label} -  {image_id}", fontsize=12)
        plt.axis("off")
    
    plt.show()

visualize_four(0)

```

### Label 1 {.tabset}

```{python load_image_set1}
visualize_four(1)
```

### Label 2 {.tabset}

```{python load_image_set2}
visualize_four(2)
```

### Label 3 {.tabset}

```{python load_image_set3}
visualize_four(3)
```

### Label 4 {.tabset}

```{python load_image_set4}
visualize_four(4)
```

# Keras {.tabset}

## Keras Helpers and Model {.tabset}

```{python keras_model}
# Proximate Source: https://www.kaggle.com/maksymshkliarevskyi/cassava-leaf-disease-keras-cnn-baseline

train = train_df
train_labels = pd.read_csv(os.path.join(BASE_DIR, "train.csv"))

train_labels.label = train_labels.label.astype('str')
train, valid = train_test_split(train_labels, train_size = 0.8, shuffle = True,
                                random_state = 0)

BATCH_SIZE = 64
STEPS_PER_EPOCH = len(train) / BATCH_SIZE
VALIDATION_STEPS = len(valid) / BATCH_SIZE

EPOCHS = 15

train_generator = ImageDataGenerator(rescale=1./255, 
                                     rotation_range = 40,
                                     height_shift_range = 0.2,
                                     width_shift_range = 0.2,
                                     shear_range = 0.2,
                                     zoom_range = 0.2,
                                     horizontal_flip = True,
                                     fill_mode = 'nearest') \
    .flow_from_dataframe(train,
                         directory = os.path.join(BASE_DIR, "train_images"),
                         x_col = "image_id",
                         y_col = "label",
                         target_size = (150, 150),
                         batch_size = BATCH_SIZE,
                         class_mode = "categorical")

validation_generator = ImageDataGenerator(rescale=1./255) \
    .flow_from_dataframe(valid,
                         directory = os.path.join(BASE_DIR, "train_images"),
                         x_col = "image_id",
                         y_col = "label",
                         target_size = (150, 150),
                         batch_size = BATCH_SIZE,
                         class_mode = "categorical")

def create_model():
    model = models.Sequential()
    model.add(layers.Conv2D(64, (3, 3), activation = "relu", 
                            input_shape=(150, 150, 3)))
    model.add(layers.MaxPooling2D((2, 2)))

    model.add(layers.Conv2D(128, (3, 3), activation = "relu", padding = "same"))
    model.add(layers.Conv2D(128, (3, 3), activation = "relu", padding = "same"))
    model.add(layers.MaxPooling2D((2, 2)))

    model.add(layers.Conv2D(256, (3, 3), activation = "relu", padding = "same"))
    model.add(layers.Conv2D(256, (3, 3), activation = "relu", padding = "same"))
    
    model.add(layers.MaxPooling2D((2, 2)))
    model.add(layers.Flatten())

    model.add(layers.Dense(128, activation = "relu"))
    model.add(layers.Dropout(0.5))
    model.add(layers.Dense(64, activation = "relu"))
    model.add(layers.Dropout(0.5))

    model.add(layers.Dense(5, activation = "softmax"))

    model.compile(optimizer = 'rmsprop',
                  loss = "categorical_crossentropy",
                  metrics = ["acc"])
    return model

model = create_model()
model.summary()

model_save = ModelCheckpoint('./best_baseline_model.h5', 
                             save_best_only = True, 
                             monitor = 'val_loss', 
                             mode = 'min')

```
## Keras Fitting {.tabset}
```{python keras_fitting}
history = model.fit_generator(
    train_generator,
    steps_per_epoch = STEPS_PER_EPOCH,
    epochs = EPOCHS,
    validation_data = validation_generator,
    validation_steps = VALIDATION_STEPS,
    callbacks = [model_save]
)

acc = history.history['acc']
val_acc = history.history['val_acc']
loss = history.history['loss']
val_loss = history.history['val_loss']

epochs = range(1, len(acc) + 1)

fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(15, 5))
sns.set_style("white")
plt.suptitle('Train history', size = 15)

ax1.plot(epochs, acc, "bo", label = "Training acc")
ax1.plot(epochs, val_acc, "b", label = "Validation acc")
ax1.set_title("Training and validation acc")
ax1.legend()

ax2.plot(epochs, loss, "bo", label = "Training loss", color = 'red')
ax2.plot(epochs, val_loss, "b", label = "Validation loss", color = 'red')
ax2.set_title("Training and validation loss")
ax2.legend()

plt.show()

model.save('./baseline_model.h5')
model.load_weights("best_baseline_model.h5")

```
# Submission {.tabset}

```{python submission}
# Proximate Source: https://www.kaggle.com/maksymshkliarevskyi/cassava-leaf-disease-keras-cnn-baseline

ss = pd.read_csv(os.path.join(BASE_DIR, "sample_submission.csv"))
ss.label = ss.label.astype('str')
ss

test_generator = ImageDataGenerator(rescale=1./255) \
    .flow_from_dataframe(ss,
                         directory = os.path.join(BASE_DIR, "test_images"),
                         x_col = "image_id",
                         y_col = "label",
                         target_size = (150, 150),
                         batch_size = BATCH_SIZE,
                         class_mode = "categorical")

predict = model.predict_generator(test_generator)
ss['label'] = predict.argmax(axis=1)
ss.to_csv('submission.csv', index = False)

```