# This R environment comes with all of CRAN preinstalled, as well as many other helpful packages
# The environment is defined by the kaggle/rstats docker image: https://github.com/kaggle/docker-rstats
# For example, here's several helpful packages to load in 

library(ggplot2) # Data visualization
library(readr) # CSV file I/O, e.g. the read_csv function
library(jpeg) # Read jpgs
library(ripa) #image processing library




# Input data files are available in the "../input/" directory.
# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory

#system("ls ../input/train/c0")
#list.files("../input/test")

#s1<-"../input/train/c0"
#list.files(s1)

#function to resize images
#if mean is true we create
#new values by averaging 
#nearby values
resize<-function(imMat, rows, columns, mean=TRUE)
{
  result<-matrix(nrow=rows, ncol=columns)
  rowStep<-nrow(imMat)/rows
  colStep<-ncol(imMat)/columns
  
  for(i in 1:rows)
  {
    for(j in 1:columns)
    {
      if(mean)
      {
        rstepP<-rowStep*(i-1)+1        
        cstepP<-colStep*(j-1)+1          
        result[i,j]<-mean(imMat[rstepP:(rowStep*i), 
                                cstepP:(colStep*j)])        
      }
      else
      {
        result[i,j]<-imMat[rowStep*i, colStep*j]
      }
    }
  }
  return(imagematrix(result))  
}

#create a dataframe where the 1st column 
#is the classification and the remaining columns
#are the values of the
#pixels in the image matrix
#we shrink the image to 10x10 to give 100 variables


createTrainDataFrame<-function(fileNames, classes)
{
  dataMatrix<-matrix(nrow=length(fileNames), ncol=100)
  for(i in 1:length(fileNames))
  {
    #print(fileNames[i])
    imMat<-imagematrix(readJPEG(toString(fileNames[i])))
    imMatg<-rgb2grey(imMat)
    
    imMatSmall<-resize(imMatg,10,10)
    
    dataMatrix[i,]<-as.numeric(imMatSmall)
    
  }
  return(data.frame(class=classes, dataMatrix))
}


#get file names for training data
#the data is stored in folders so we
#loop through a list of folders
#store the classification in column 1
#and the location in column 2
createTrainDataFileNames<-function(folders)
{
  
  string<-paste("../input/train/", folders[1], sep="")
  
  ans<-data.frame(class=folders[1], 
                  filenames=paste(string,"/",  list.files(string), sep=""))
                   
  for(i in 2:length(folders))
  {
    string2<-paste("../input/train/", folders[i], sep="")
    ans2<-data.frame(class=folders[i], 
                    filenames=paste(string2, "/", list.files(string2), sep=""))
    ans<-rbind(ans,ans2)
  }
  
  return(ans)  
}



#get file names for test data 
createTestDataFileNames<-function()
{
  ans<-data.frame(filenames=paste("../input/test/",list.files("../input/test"), sep=""))
  return(ans)
}

#get the folder names
folderNames<-list.files("../input/train")
#print(folderNames)
#get the train file names with classifications
trainFiles<-createTrainDataFileNames(folderNames)

#print(trainFiles[1,2])
#get the test files names
testFiles<-createTestDataFileNames()

#create dataframe for training
#we cannot create the full data frame
#within 10 minutes so we just create a sample
#of the first 100 rows 
trainDF<-createTrainDataFrame(trainFiles[1:2489,2], trainFiles[1:2489,1])

#create dataframe for testing
#list the classification as unknown
testDF<-createTrainDataFrame(testFiles[1:10000,1], "unknown")

#now we have converted the data into a bunch of 
#numeric variables we can apply standard machine
#learning algorithms

write.csv(trainDF, row.names=FALSE, file="trainData.csv")
write.csv(testDF, row.names=FALSE, file="testData.csv")
