{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"#This is my first notebook on Kaggle which I used for the melanoma challenge. \n#I tried the VGG16 pre-trained network with a unfrozen top layer. In my training and validation \n#I had a score around 0.90, but unfortunately here on Kaggle it's a bit less. \n#nevertheless I'm happy to share my model. If you have any ideas how to improve it, please let me know!For now I'm gonan start with my xgbTree for the tabular prediction.  ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Creating our working directories here \nmaster_dataset_dir <- \"D:/Kaggle competitions\"\n\n#creating our labels \ntrain_csv_data <- train\nimage_names <- as.matrix(train_csv_data[, 1])\n\ntest_csv_data <- test_1_\ntest_image_names <- as.matrix(test_csv_data[, 1])\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# list of melanomas to modle\nmelanoma_list <- c(\"Yes\", \"No\")\n\n# number of output classes\noutput_n <- length(melanoma_list)\n\n# The images are of various sizes and they are scaled down to 299,299\nimg_width <- 299\nimg_height <- 299\ntarget_size <- c(img_width, img_height)\n\nchannels <- 3\n\n# Get the data that are +ve for melanoma\nmelanoma_data <- train_csv_data %>% filter(target == 1)\n# Get the data that are -ve for melanoma\nno_melanoma_data <- train_csv_data %>% filter(target == 0) \n# Get the image names when is it +ve\nmelanoma_names <- as.matrix(melanoma_data[, 1])\n# Get the image names when is it -ve\nno_melanoma_names <- as.matrix(no_melanoma_data[, 1])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"melanoma_dir <- \"D:/Kaggle competitions/UnderSamplingData/Yes\"\nno_melanoma_dir <- \"D:/Kaggle competitions/UnderSamplingData/No\"\ntrain_dir <- \"D:/Kaggle competitions/UnderSamplingData/\"\ntest_dir <- \"D:/Kaggle competitions/Data/\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#I sampled 1200 normal skin pictures versus te 583 melanoma pictures: \n\nno_melanoma_names <- as.matrix(sample(no_melanoma_names, 1200))\n\n#kopieer de YES : 1200 stuks \nfor (i in 1:length(melanoma_names)) {\n  #copy vector\n  #setwd(current_dir)\n  name <- melanoma_names[i, 1]\n  filename <- paste(sep=\"\",name,\".jpg\")\n  file.copy(file.path(current_dir, filename),\n            file.path(melanoma_dir))\n}\n\n\n\n# # kopieer NO: alles \nfor (i in 1:length(no_melanoma_names)) {\n  #copy vector\n  #setwd(current_dir)\n  name <- no_melanoma_names[i, 1]\n  filename <- paste(sep=\"\",name,\".jpg\")\n  file.copy(file.path(current_dir, filename),\n            file.path(no_melanoma_dir))\n}\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# ### Data loading and augmentation\n# \ntrain_data_gen = image_data_generator(\n  rescale = 1/255, #,\n  rotation_range = 40,\n  width_shift_range = 0.2,\n  height_shift_range = 0.2,\n  shear_range = 0.2,\n  zoom_range = 0.2,\n  horizontal_flip = TRUE\n)\n\n# # # Validation data shouldn't be augmented! But it should also be scaled.\nvalid_data_gen <- image_data_generator(\n  rescale = 1/255\n)\n\n# Validation data shouldn't be augmented! But it should also be scaled.\ntest_data_gen <- image_data_generator(\n  rescale = 1/255\n)\n# \n# # training images\ntrain_image_array_gen <- flow_images_from_directory(train_dir,\n                                                    train_data_gen,\n                                                \n                                                    target_size = target_size,\n                                                    class_mode = \"binary\",\n                                                   # classes = melanoma_list,\n                                                  #  subset = \"training\",\n                                                    batch_size = 15)\n\n\n# validation images\nvalid_image_array_gen <- flow_images_from_directory(train_dir,\n                                                    valid_data_gen,\n                                           \n                                                    target_size = target_size,\n                                                    class_mode = \"binary\",\n                                                #    classes = melanoma_list,\n                                                  #  subset = \"validation\",\n                                                    batch_size = 15)\n# \n# \ntest_image_array_gen <- flow_images_from_directory(test_dir,\n                                                   test_data_gen,\n                                                   shuffle = FALSE,\n                                                   target_size = target_size,\n                                                   class_mode = NULL, batch_size = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#my pretrained network \n\nconv_base <- application_vgg16(\n  \n  weights = \"imagenet\",\n  include_top = FALSE,\n  input_shape = c(299, 299, 3)\n  \n)\n\n\nmodel <- keras_model_sequential() %>%\nconv_base() %>% \n  layer_flatten() %>% \n  layer_dense(units = 256, activation = \"relu\") %>%\n  layer_dense(units = 1, activation = \"sigmoid\")\n\nmodel\n\n\nfreeze_weights(conv_base)\n\n#I Unfreeze my network from the last block to finetune my network\nunfreeze_weights(conv_base, from = \"block5_conv1\")\n############################################################\n\nmodel %>% compile(\n  loss = \"binary_crossentropy\",\n  optimizer = optimizer_rmsprop(lr = 1e-5),\n  metrics = c(\"acc\")\n)\n\n\n##################################################################################\nhist_CNN <- model %>% fit_generator(\n  # training data\n  train_image_array_gen,\n  \n  # epochs\n  steps_per_epoch = 100,\n  epochs = 100,\n  \n  # validation data\n  validation_data = valid_image_array_gen,\n  validation_steps = 50\n  \n)\n\n\n\n#Predict\n#Use predict_generator to predict the results on the test data set. \n#When I ran this on my machine, I got the public score of 0.75 % in my first attempt \n\n\npredictions <- model %>% predict_generator(test_image_array_gen, steps = 10982)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"3.6.3"}},"nbformat":4,"nbformat_minor":4}