# Loading all the libraries

library(keras)
library(tensorflow)
library(tfdatasets)
library(stringr)
library(dplyr)



# We pick a random .wav file and decode it using tf$audio$decode_wav.
# This will give us access to two tensors: the samples themselves, and the sampling rate.


audio <- tf$contrib$framework$python$ops$audio_ops


files <- fs::dir_ls(path = "../input/train_curated/",
                    recursive = TRUE,
                    glob = "*.wav")
                    
cast_variables <- function(df, variable){
      x <- as.character(unique(df[[variable]]))
      x <- gsub(" ", "", toString(x)) #so it can split on strings like "A1,A2" and "A1, A2"
      x <- unlist(strsplit(x, ","))
      x <- as.character(x)
      new_columns <- unique(sort(x))
      for (i in seq_along(new_columns)){
            df$temp <- NA
            df$temp <- ifelse(grepl(new_columns[i], df[[variable]]), 1, 0)
            colnames(df)[colnames(df) == "temp"] <- new_columns[i]
      }
      return(df)
}

train <- read.csv('../input/train_curated.csv', stringsAsFactors = F)
new_df <- cast_variables(train, 'labels')
new_df$fname <- paste('../input/train_curated/',train$fname, sep = "")


# Sett
head(new_df)
labels <- c('Accelerating_and_revving_and_vroom','Accordion','Acoustic_guitar', 'Applause', 'Bark' ,'Bass_drum',
            'Bass_guitar' ,'Bathtub_(filling_or_washing)', 'Bicycle_bell' ,'Burping_and_eructation', 'Bus' ,'Buzz' ,
            'Car_passing_by' ,'Cheering' ,'Chewing_and_mastication', 'Child_speech_and_kid_speaking' ,
            'Chink_and_clink', 'Chirp_and_tweet', 'Church_bell', 'Clapping', 'Computer_keyboard', 'Crackle' ,'Cricket',
            'Crowd', 'Cupboard_open_or_close', 'Cutlery_and_silverware' ,'Dishes_and_pots_and_pans',
            'Drawer_open_or_close', 'Drip' ,'Electric_guitar', 'Fart' ,'Female_singing',
            'Female_speech_and_woman_speaking', 'Fill_(with_liquid)', 'Finger_snapping', 'Frying_(food)' ,'Gasp' ,
            'Glockenspiel', 'Gong', 'Gurgling', 'Harmonica', 'Hi-hat', 'Hiss' ,'Keys_jangling', 'Knock',
            'Male_singing', 'Male_speech_and_man_speaking', 'Marimba_and_xylophone', 'Mechanical_fan', 'Meow',
            'Microwave_oven', 'Motorcycle' ,'Printer' ,'Purr', 'Race_car_and_auto_racing' ,'Raindrop', 'Run' ,
            'Scissors', 'Screaming', 'Shatter', 'Sigh', 'Sink_(filling_or_washing)', 'Skateboard' ,'Slam' ,'Sneeze',
            'Squeak', 'Stream', 'Strum', 'Tap' ,'Tick-tock', 'Toilet_flush', 'Traffic_noise_and_roadway_noise',
            'Trickle_and_dribble', 'Walk_and_footsteps', 'Water_tap_and_faucet' ,'Waves_and_surf' ,'Whispering',
            'Writing' ,'Yell' ,'Zipper_(clothing)')
            
batch_size <- 256
buffer_size <- nrow(new_df)

window_size_ms <- 30
window_stride_ms <- 10

data_generator <- function(df, window_size_ms, window_stride_ms, shuffle = FALSE) {
      
      sampling_rate <- 16000
      window_size <- as.integer(sampling_rate*window_size_ms/1000)
      stride <- as.integer(sampling_rate*window_stride_ms/1000)
     
      
      ds <- tensor_slices_dataset(df)
      
      if(shuffle)
          ds <- ds %>% dataset_shuffle(buffer_size = 100)  
      
      ds <- ds %>%
            dataset_map(function(obs) {
                  
                  # decoding wav files
                  audio_binary <- tf$read_file(tf$reshape(obs$fname, shape = list()))
                  wav <- audio$decode_wav(audio_binary, desired_channels = 1)
                  
                  # create the spectrogram
                  spectrogram <- audio$audio_spectrogram(
                        wav$audio, 
                        window_size = window_size, 
                        stride = stride,
                        magnitude_squared = TRUE
                  )
                  
                  # Custom brightness
                
                  brightness <- 100L
                  mul = tf$multiply(spectrogram, brightness)
                  
                  # Normalize pixels
                  min_const = tf$constant(255)
                  minimum =  tf$minimum(mul, min_const)
                  
                  # Expand dims so we get the proper shape
                  expand_dims = tf$expand_dims(minimum, -1L)
                  
                  # Resize the spectrogram to input size of the model
                  resize = tf$image$resize_bilinear(expand_dims, c(128L, 128L))
                  
                  # Remove the trailing dimension
                  squeeze = tf$squeeze(resize, 0L)
                  
                  response <- as.matrix(obs[labels])
                  list(squeeze, response)
            }) %>%
            dataset_repeat()
      
      ds <- ds %>% 
            dataset_padded_batch(batch_size, list(shape(128, 128, 1), shape(80)))
      
      return(ds)            
}


id_train <- sample(nrow(new_df), size = 0.7 * nrow(new_df))
n_train <- length(id_train)
n_test <- nrow(new_df) - n_train

ds_train <- data_generator(new_df[id_train, ],
                           window_size_ms = window_size_ms,
                           window_stride_ms = window_stride_ms)
ds_val <- data_generator(new_df[-id_train, ],
                         window_size_ms = window_size_ms,
                         window_stride_ms = window_stride_ms)
                         
model <- keras_model_sequential()
model %>%  
      layer_conv_2d(input_shape = c(128, 128, 1), 
                    filters = 32, kernel_size = c(3,3), activation = 'relu') %>% 
      layer_max_pooling_2d(pool_size = c(2, 2)) %>%
      layer_conv_2d(filters = 64, kernel_size = c(3,3), activation = 'relu') %>% 
      layer_max_pooling_2d(pool_size = c(2, 2)) %>% 
      layer_conv_2d(filters = 128, kernel_size = c(3,3), activation = 'relu') %>% 
      layer_max_pooling_2d(pool_size = c(2, 2)) %>% 
      layer_conv_2d(filters = 256, kernel_size = c(3,3), activation = 'relu') %>% 
      layer_max_pooling_2d(pool_size = c(2, 2)) %>% 
      layer_dropout(rate = 0.25) %>% 
      layer_flatten() %>% 
      layer_dense(units = 256, activation = 'relu') %>% 
      layer_dropout(rate = 0.2) %>% 
      layer_dense(units = 80, activation = 'softmax')




model %>% compile(
      loss = loss_binary_crossentropy,
      optimizer = optimizer_adam(),
      metrics = c('accuracy')
)
model %>% summary()

early_stopping <- callback_early_stopping(monitor = 'val_loss', patience = 2)

model %>% fit_generator(
      generator = ds_train,
      steps_per_epoch = 0.7*nrow(new_df)/batch_size,
      epochs = 50,
      verbose = 1,
      validation_data = ds_val, 
      validation_steps = 0.3*nrow(new_df)/batch_size, callbacks = c(early_stopping)
)

test <- read.csv('../input/sample_submission.csv', stringsAsFactors = F)
cnames <- data.frame(fname = test$fname)
test$fname <- paste('../input/test/',test$fname, sep = "")
colnames(test) <- c('fname', labels)

ds_test <- data_generator(test,
                           window_size_ms = window_size_ms,
                           window_stride_ms = window_stride_ms)
                           
n_steps <- nrow(test)/batch_size +1
predictions <- predict_generator(object = model,
      ds_test, 
      steps = n_steps
)

predictions <- as.data.frame(predictions[1:1120,])

predictions <- cbind(cnames, predictions)

colnames(predictions) <- c('fname', labels)
head(predictions)

write.csv(predictions, 'submission.csv', row.names=FALSE)
