{"cells":[{"metadata":{"_uuid":"282b89d0dee6bb214e17ee61684e833ac5583c78","_execution_state":"idle","trusted":true},"cell_type":"code","source":"# carregando pacotes e verificando arquivos presentes\nlibrary(tidyverse)\nlibrary(fpp2)\nlist.files(path = \"../input/\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a2cdd7fa878afd97dbeacad9cb3a6a0068f603cb"},"cell_type":"code","source":"length(list.files(path = \"../input/test\")) # 2624\n# há 2624 segmentos de teste, sendo que, cada segmento possui 150000 linhas\n# e o arquivo de submissão deve informar o time do failure da última linha de cada segmento\ndim(read_csv(\"../input/test/seg_00c35b.csv\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"17d391c6f6b88e98eb4021c0e97f142c2983591e"},"cell_type":"code","source":"# o arquivo de treino é um unico segmento em sequência\noptions(digits = 15)\n# escolhendo aleatoriamente uma parte do segmento para analisar\nset.seed(8291003);sample(1:4194,1)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"857fc478e4c042cbb7ad9be0cdd207c1dee4bfe8"},"cell_type":"code","source":"nomes = c(\"acoustic_data\",\"time_to_failure\")\ndados = read_csv(\"../input/train.csv\",n_max =150000,col_names=nomes,skip=(1292*150000)+1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d62ff3945cd7ec1e07423430f31fe16983ba6395"},"cell_type":"code","source":"head(dados);dim(dados)\napply(dados,2,anyNA) # não há valores ausentes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"464d8f95114a6f965afad7f372feca4b11f57366"},"cell_type":"code","source":"cor.test(dados$acoustic_data,dados$time_to_failure)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0f8e886d1828402e44a0cc051b0d53943344848a"},"cell_type":"code","source":"ggAcf(dados$time_to_failure,main=\"\",lag.max=100)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"99bf8f74bbafd2bc6914ae2364459ae87103f040"},"cell_type":"code","source":"summary(dados$acoustic_data);summary(dados$time_to_failure)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"35f54d51247c3094d38ab71804ca7d55628e6d97"},"cell_type":"code","source":"ggplot(dados)+geom_histogram(mapping=aes(acoustic_data),fill=\"blue\",col=\"black\")\nggplot(dados)+geom_histogram(mapping=aes(time_to_failure),fill=\"blue\",col=\"black\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c31080a8f10aba0bde95c7dbd6e88e3f1d9c0ba8"},"cell_type":"code","source":"# analisando como time_to_failure varia de acordo com o acoustic data\nwith(dados[1:1000,], plot(acoustic_data ~ seq(1:1000), type = \"l\",xlab=\"\",ylab=\"\")) \npar(new = T) \nwith(dados[1:1000,], plot(time_to_failure ~ seq(1:1000), type = \"l\", axes = F, frame = T, \n     ann = F, col = 2)) \naxis(4, col.axis = 2, col = 2) \nlegend(\"topleft\",legend=c(\"acoustic data\",\"time to failure\"),\n  text.col=c(\"black\",\"red\"),lty=c(1,1),col=c(\"black\",\"red\"),box.lty=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f155708354d46f28f7b70e8452a182315b5cffe2"},"cell_type":"code","source":"with(dados[1:5000,], plot(acoustic_data ~ seq(1:5000), type = \"l\",xlab=\"\",ylab=\"\")) \npar(new = T) \nwith(dados[1:5000,], plot(time_to_failure ~ seq(1:5000), type = \"l\", axes = F, frame = T, \n     ann = F, col = 2)) \naxis(4, col.axis = 2, col = 2) \nlegend(\"topleft\",legend=c(\"acoustic data\",\"time to failure\"),\n  text.col=c(\"black\",\"red\"),lty=c(1,1),col=c(\"black\",\"red\"),box.lty=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c01a5fc977027949087d3247df1d20937ca441a3"},"cell_type":"code","source":"with(dados[1:5000,], plot(acoustic_data ~ seq(1:5000), type = \"l\",xlab=\"\",ylab=\"\")) \npar(new = T) \nwith(dados[1:5000,], plot(time_to_failure ~ seq(1:5000), type = \"p\", axes = F, frame = T, \n     ann = F, col = 2)) \naxis(4, col.axis = 2, col = 2) \nlegend(\"topleft\",legend=c(\"acoustic data\",\"time to failure\"),\n  text.col=c(\"black\",\"red\"),lty=c(1,1),col=c(\"black\",\"red\"),box.lty=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"3a689f8c7c9d0559585ef24d802be9bcea275a98"},"cell_type":"code","source":"with(dados[1:50000,], plot(acoustic_data ~ seq(1:50000), type = \"l\",xlab=\"\",ylab=\"\")) \npar(new = T) \nwith(dados[1:50000,], plot(time_to_failure ~ seq(1:50000), type = \"l\", axes = F, frame = T, \n     ann = F, col = 2)) \naxis(4, col.axis = 2, col = 2) \nlegend(\"topleft\",legend=c(\"acoustic data\",\"time to failure\"),\n  text.col=c(\"black\",\"red\"),lty=c(1,1),col=c(\"black\",\"red\"),box.lty=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"34b59562d08665621ee33e10a3b42479e674eb69"},"cell_type":"code","source":"with(dados[1:50000,], plot(acoustic_data ~ seq(1:50000), type = \"l\",xlab=\"\",ylab=\"\")) \npar(new = T) \nwith(dados[1:50000,], plot(time_to_failure ~ seq(1:50000), type = \"p\", axes = F, frame = T, \n     ann = F, col = 2)) \naxis(4, col.axis = 2, col = 2) \nlegend(\"topleft\",legend=c(\"acoustic data\",\"time to failure\"),\n  text.col=c(\"black\",\"red\"),lty=c(1,1),col=c(\"black\",\"red\"),box.lty=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"47aa5ca849dc4b8259e0a78f6f9064b7007ca4fd"},"cell_type":"code","source":"plot(dados$time_to_failure[1:5000],ylab=\"time to failure\",xlab=\"\",type=\"l\")\nplot(dados$acoustic_data[1:5000],ylab=\"acoustic data\",xlab=\"\",type=\"l\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1efcb5fc26f768bfc0bdd9acaa742c636a5683e5"},"cell_type":"code","source":"# entre as observações 2190 e 2215 há uma queda abrupta em time to failure\n# no restante dos dados tbm é percebido quedas abruptas em time to failure\nplot(2190:2215,dados$time_to_failure[2190:2215],ylab=\"time to failure\",xlab=\"\",type=\"p\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eb525528d40a9d10e905e373b88d8601eaaf5d19"},"cell_type":"code","source":"# vamos coletar 10 partes do segmento \n# abrangendo todo o segmento\nround(seq(0,4193,length=10))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f31bb837173c30437303cdeba565ec5b8f8b674c"},"cell_type":"code","source":"dados_1 = read_csv(\"../input/train.csv\",n_max=150000)\ndados_2 = read_csv(\"../input/train.csv\",n_max =150000,col_names=nomes,skip=(466*150000)+1)\ndados_3 = read_csv(\"../input/train.csv\",n_max =150000,col_names=nomes,skip=(932*150000)+1)\ndados_4 = read_csv(\"../input/train.csv\",n_max =150000,col_names=nomes,skip=(1398*150000)+1)\ndados_5 = read_csv(\"../input/train.csv\",n_max =150000,col_names=nomes,skip=(1864*150000)+1)\ndados_6 = read_csv(\"../input/train.csv\",n_max =150000,col_names=nomes,skip=(2329*150000)+1)\ndados_7 = read_csv(\"../input/train.csv\",n_max =150000,col_names=nomes,skip=(2795*150000)+1)\ndados_8 = read_csv(\"../input/train.csv\",n_max =150000,col_names=nomes,skip=(3261*150000)+1)\ndados_9 = read_csv(\"../input/train.csv\",n_max =150000,col_names=nomes,skip=(3727*150000)+1)\ndados_10 = read_csv(\"../input/train.csv\",n_max =150000,col_names=nomes,skip=(4193*150000)+1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d91f3c1a344c7e391ecb4fc610ec4bd8374a51a9"},"cell_type":"code","source":"time_to_failure = data.frame(time_to_failure_1 = dados_1$time_to_failure,\n                            time_to_failure_2 = dados_2$time_to_failure,\n                            time_to_failure_3 = dados_3$time_to_failure,\n                            time_to_failure_4 = dados_4$time_to_failure,\n                            time_to_failure_5 = dados_5$time_to_failure,\n                            time_to_failure_6 = dados_6$time_to_failure,\n                            time_to_failure_7 = dados_7$time_to_failure,\n                            time_to_failure_8 = dados_8$time_to_failure,\n                            time_to_failure_9 = dados_9$time_to_failure,\n                            time_to_failure_10 = dados_10$time_to_failure)\nacoustic_data = data.frame(acoustic_data_1 = dados_1$acoustic_data,\n                            acoustic_data_2 = dados_2$acoustic_data,\n                            acoustic_data_3 = dados_3$acoustic_data,\n                            acoustic_data_4 = dados_4$acoustic_data,\n                            acoustic_data_5 = dados_5$acoustic_data,\n                            acoustic_data_6 = dados_6$acoustic_data,\n                            acoustic_data_7 = dados_7$acoustic_data,\n                            acoustic_data_8 = dados_8$acoustic_data,\n                            acoustic_data_9 = dados_9$acoustic_data,\n                            acoustic_data_10 = dados_10$acoustic_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2dda9190845baa988855187e5a012af1829b7e39"},"cell_type":"code","source":"time_to_failure_mean = apply(time_to_failure,2,mean)\ntime_to_failure_median = apply(time_to_failure,2,median)\nacoustic_data_mean = apply(acoustic_data,2,mean)\nacoustic_data_median = apply(acoustic_data,2,median)\nacoustic_data_sd = apply(acoustic_data,2,sd)\nacoustic_data_var = apply(acoustic_data,2,var)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a6203cb70059c8eece3543dc0092578dc4057504"},"cell_type":"code","source":"analysis = data.frame(cbind(time_to_failure_mean,time_to_failure_median,acoustic_data_mean,\n     acoustic_data_sd,acoustic_data_var))\nhead(analysis)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ef96d15672d19ea721b0d3db5d5281929c501cd"},"cell_type":"code","source":"analysis[order(analysis$time_to_failure_mean),]","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"R","language":"R","name":"ir"},"language_info":{"mimetype":"text/x-r-source","name":"R","pygments_lexer":"r","version":"3.4.2","file_extension":".r","codemirror_mode":"r"}},"nbformat":4,"nbformat_minor":1}