## This IDE program is to know total of rows in the huge size training CSV
## and the distribution for each chunk of training data (3MM rows per chunk)
## R still has a way to handle a huge size of input file

library(bigtabulate)
library(bigmemory)
library(biganalytics)
library(data.table)
library(dplyr)
library(tidyverse) 

# Input data files are available in the "../input/" directory.
# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory

options(digits = 17)

## read in huge data in R
read_dt <- read.big.matrix("../input/train.csv", header = TRUE, backingfile = "s.bin", descriptorfile ="s.desc", extraCols =NULL) 

## Get the location of the pointer 
##dd <- describe(read_dt)

## outputs are in the comments below - 629MM rows
##str(dd)

 #Formal class 'big.matrix.descriptor' [package "bigmemory"] with 1 slot
  #..@ description:List of 13
  #.. ..$ sharedType: chr "FileBacked"
  #.. ..$ filename  : chr "school.bin"
  #.. ..$ dirname   : chr "C:/kaggle/Earthquake_prediction/"
  #.. ..$ totalRows : int 629145480
  #.. ..$ totalCols : int 2
  #.. ..$ rowOffset : num [1:2] 0.00 6.29e+08
  #.. ..$ colOffset : num [1:2] 0 2
  #.. ..$ nrow      : num 6.29e+08
  #.. ..$ ncol      : num 2
  #.. ..$ rowNames  : NULL
  #.. ..$ colNames  : chr [1:2] "acoustic_data" "time_to_failure"
  #.. ..$ type      : chr "double"
  #.. ..$ separated : logi FALSE
  
## histograms after knowing total number of rows   
##maxsize<-3000000
## we get 210 for the max number of loops by 210=630/3
##for (i in 1:210){
##traind  <- fread("../input/train.csv", nrows=maxsize)
##hist(traind$time_to_failure, freq=FALSE, main=paste("Histogram of", i))
#}

# Any results you write to the current directory are saved as output.