{"metadata":{"kernelspec":{"display_name":"R","language":"R","name":"ir"},"language_info":{"mimetype":"text/x-r-source","name":"R","pygments_lexer":"r","version":"3.6.0","file_extension":".r","codemirror_mode":"r"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"## Importing packages\n\n# This R environment comes with all of CRAN and many other helpful packages preinstalled.\n# You can see which packages are installed by checking out the kaggle/rstats docker image: \n# https://github.com/kaggle/docker-rstats\n\nlibrary(tidyverse) # metapackage with lots of helpful functions\nlibrary(randomForest)\nlibrary(e1071)\nlibrary(caret)\n\n## Running code\n\n# In a notebook, you can run a single code cell by clicking in the cell and then hitting \n# the blue arrow to the left, or by clicking in the cell and pressing Shift+Enter. In a script, \n# you can run code by highlighting the code you want to run and then clicking the blue arrow\n# at the bottom of this window.\n\n## Reading in files\n\n# You can access files from datasets you've added to this kernel in the \"../input/\" directory.\n# You can see the files added to this kernel by running the code below. \n\nlist.files(path = \"../input/google-cloud-ncaa-march-madness-2020-division-1-mens-tournament/MDataFiles_Stage1\")\n\n## Saving data\n\n# If you save any files or images, these will be put in the \"output\" directory. You \n# can see the output directory by committing and running your kernel (using the \n# Commit & Run button) and then checking out the compiled version of your kernel.","metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","execution":{"iopub.status.busy":"2022-08-04T21:07:47.741511Z","iopub.execute_input":"2022-08-04T21:07:47.744454Z","iopub.status.idle":"2022-08-04T21:07:50.970215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#2003-2014 Data\nCity <- read.csv('../input/google-cloud-ncaa-march-madness-2020-division-1-mens-tournament/MDataFiles_Stage1/Cities.csv', header = TRUE) \nConference <- read.csv('../input/google-cloud-ncaa-march-madness-2020-division-1-mens-tournament/MDataFiles_Stage1/Conferences.csv', header = TRUE)\nRankingsD <- read.csv('../input/google-cloud-ncaa-march-madness-2020-division-1-mens-tournament/MDataFiles_Stage1/MMasseyOrdinals.csv', header = TRUE)\nTourneyD <- read.csv('../input/google-cloud-ncaa-march-madness-2020-division-1-mens-tournament/MDataFiles_Stage1/MNCAATourneyDetailedResults.csv', header = TRUE)\nTourneySeeds <- read.csv('../input/google-cloud-ncaa-march-madness-2020-division-1-mens-tournament/MDataFiles_Stage1/MNCAATourneySeeds.csv', header = TRUE)\nRegseasonD <- read.csv('../input/google-cloud-ncaa-march-madness-2020-division-1-mens-tournament/MDataFiles_Stage1/MRegularSeasonDetailedResults.csv', header = TRUE)\nTeam_conf <- read.csv('../input/google-cloud-ncaa-march-madness-2020-division-1-mens-tournament/MDataFiles_Stage1/MTeamConferences.csv', header = TRUE)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T21:07:50.973179Z","iopub.execute_input":"2022-08-04T21:07:51.001186Z","iopub.status.idle":"2022-08-04T21:07:59.857532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#2020 data\nConference20 <- read.csv('../input/google-cloud-ncaa-march-madness-2020-division-1-mens-tournament/MDataFiles_Stage2/Conferences.csv', header = TRUE, stringsAsFactors = FALSE)\nRankingsD20 <- read.csv('../input/google-cloud-ncaa-march-madness-2020-division-1-mens-tournament/MDataFiles_Stage2/MMasseyOrdinals.csv', header = TRUE, stringsAsFactors = FALSE)\nTourneyD20 <- read.csv('../input/google-cloud-ncaa-march-madness-2020-division-1-mens-tournament/MDataFiles_Stage2/MNCAATourneySeeds.csv', header = TRUE, stringsAsFactors = FALSE)\nRegseasonD20 <- read.csv('../input/google-cloud-ncaa-march-madness-2020-division-1-mens-tournament/MDataFiles_Stage2/MRegularSeasonDetailedResults.csv', header = TRUE, stringsAsFactors = FALSE)\nTeam_conf20 <- read.csv('../input/google-cloud-ncaa-march-madness-2020-division-1-mens-tournament/MDataFiles_Stage2/MTeamConferences.csv', header = TRUE, stringsAsFactors = FALSE)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T21:54:39.584279Z","iopub.execute_input":"2022-08-04T21:54:39.585149Z","iopub.status.idle":"2022-08-04T21:54:48.971844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Examine the top of each data set. \nhead(City)\nhead(Conference)\nhead(RankingsD)\nhead(TourneyD)\nhead(TourneySeeds)\nhead(RegseasonD)\nhead(Team_conf)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T21:57:19.063443Z","iopub.execute_input":"2022-08-04T21:57:19.065539Z","iopub.status.idle":"2022-08-04T21:57:19.162680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#############\n# Converts subset for a single year into a dataframe that has each row representing a single game. \n#############\nBuildWinLossData<-function(allYear){\ntimes <- unique(allYear$DayNum)\nfinalData <- data.frame()\n\nfor(i in 1:length(times)){\n  subset <- allYear[which(allYear$DayNum == times[i]),]\n  \n  Wteams_reg <- data.frame(subset$WTeamID, subset$DayNum, subset$WScore, subset$WFGM, subset$WFGA, subset$WFGM3, subset$WFGA3, subset$WFTM, subset$WFTA, subset$WOR, subset$WDR, \n                           subset$WAst, subset$WTO, subset$WStl, subset$WBlk, subset$WPF, rep(1, length(subset[,1])))\n\n  Lteams_reg <- data.frame(subset$LTeamID, subset$DayNum, subset$LScore, subset$LFGM, subset$LFGA, subset$LFGM3, subset$LFGA3, subset$LFTM, subset$LFTA, subset$LOR, subset$LDR, \n                           subset$LAst, subset$LTO, subset$LStl, subset$LBlk, subset$LPF,  rep(0, length(subset[,1])))\n\n\n\n  for(j in 1:length(subset[,1]))\n  if(j %%2 == 0){\n    colnames(Wteams_reg) <- c('TeamID1', 'Time', 'Score1', 'FGM1', 'FGA1', 'FGM31', 'FGA31', 'FTM1', 'FTA1', 'OR1', 'DR1', 'Ast1', 'TO1', 'Stl1', 'Blk1', 'PF1', 'Win1')\n    colnames(Lteams_reg) <- c('TeamID2', 'Time', 'Score2', 'FGM2', 'FGA2', 'FGM32', 'FGA32', 'FTM2', 'FTA2', 'OR2', 'DR2', 'Ast2', 'TO2', 'Stl2', 'Blk2', 'PF2', 'Win2')\n    finalData <- rbind(finalData, cbind(Wteams_reg[j,], Lteams_reg[j,]))\n    \n  }\n  else{\n    colnames(Lteams_reg) <- c('TeamID1', 'Time', 'Score1', 'FGM1', 'FGA1', 'FGM31', 'FGA31', 'FTM1', 'FTA1', 'OR1', 'DR1', 'Ast1', 'TO1', 'Stl1', 'Blk1', 'PF1', 'Win1')\n    colnames(Wteams_reg) <- c('TeamID2', 'Time', 'Score2', 'FGM2', 'FGA2', 'FGM32', 'FGA32', 'FTM2', 'FTA2', 'OR2', 'DR2', 'Ast2', 'TO2', 'Stl2', 'Blk2', 'PF2', 'Win2')\n\n    finalData <- rbind(finalData, cbind(Lteams_reg[j,], Wteams_reg[j,]))\n\n  }\n}\nfinalData$rowname <- seq.int(nrow(finalData))\nreturn(finalData)\n\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-04T21:57:19.165706Z","iopub.execute_input":"2022-08-04T21:57:19.168028Z","iopub.status.idle":"2022-08-04T21:57:19.180683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Log Loss\n\nlog_loss <- function(tourneydata, predictions){\n  n <- length(tourneydata[,1])\n  summand <- 0\n  for(i in 1:n){\n    wteam <- tourneydata[i,'WTeamID']\n    lteam <- tourneydata[i, 'LTeamID']\n    \n    pred_row <- predictions[which((predictions[,'Team1'] == wteam & predictions[,'Team2'] == lteam)|(predictions[,'Team2'] == wteam & predictions[,'Team1'] == lteam)),]\n    \n    yhat <- pred_row[,'Prob']\n    \n    \n    y <- if(pred_row$Team1 == wteam){1}else{0}\n    summand <- summand + y*log(yhat) + (1-y)*log(1-yhat)\n  }\n  \n  return(-summand/n)\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-04T21:57:19.183641Z","iopub.execute_input":"2022-08-04T21:57:19.186676Z","iopub.status.idle":"2022-08-04T21:57:19.197778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Creates every possible combination and returns a vector containing all unique matchups\n#Using rule from competition that teamID on the left is greater than the teamID on the right. \ngetMatchups <- function(Tourneydata){\n  \n  teams <- unique(c(Tourneydata$LTeamID, Tourneydata$WTeamID))\n  teams_in_tourney <- expand.grid(\"Team1\" = teams, \"Team2\" = teams)\n  u_teams_in_tourney <- teams_in_tourney[which(teams_in_tourney$Team1 != teams_in_tourney$Team2),]\n  u2_teams_in_tourney <- u_teams_in_tourney[which(u_teams_in_tourney$Team1 < u_teams_in_tourney$Team2),] \n  \n  return(u2_teams_in_tourney)\n  \n}\n\n#head(TourneyD19[which(TourneyD19$Season == '2019'),])\n\n#Tournament data for 2019 is slightly different than the rest because it doesn't have a win/loss team recorded like the other seasons. \ngetMatchups19 <- function(Tourneydata){\n    teams <- unique(c(Tourneydata$TeamID))\n    teams_in_tourney <- expand.grid(\"Team1\" = teams, \"Team2\" = teams)\n    u_teams_in_tourney <- teams_in_tourney[which(teams_in_tourney$Team1 != teams_in_tourney$Team2),]\n    u2_teams_in_tourney <- u_teams_in_tourney[which(u_teams_in_tourney$Team1 < u_teams_in_tourney$Team2),] \n    return(u2_teams_in_tourney)\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-04T21:57:19.200661Z","iopub.execute_input":"2022-08-04T21:57:19.202813Z","iopub.status.idle":"2022-08-04T21:57:19.216741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##########\n#finalData <- game data with winner and loser attributes\n#rankingdata <- season ranking data for a single season\n#confData <- conference data for a single season\n#year <- the last year of regular season data to use\n#########\nconstruct_Time_Data <- function(finalData, rankingdata, confData, year){\n\ntimes <- unique(finalData$Time)\nreturn_data <- data.frame()\nfor(i in 1:length(times)){\n  subset <- finalData[which(finalData$Time == times[i]),]\n  cols <- c('Score1', 'FGM1', 'FGA1', 'FGM31', 'FGA31', 'FTM1', 'FTA1', 'OR1', 'DR1', 'Ast1', 'TO1', 'Stl1', 'Blk1', 'PF1', 'Win1')\n  for(j in 1:length(subset[,1])){\n    final_stack <- data.frame()\n    rownums <- subset[j,]$rowname\n    team1 <- subset[j,]$TeamID1\n    team2 <- subset[j,]$TeamID2\n    season_meanw1 <- finalData[which(finalData$Time < times[i] & finalData$TeamID1 == team1),c('Score1', 'FGM1', 'FGA1', 'FGM31', 'FGA31', 'FTM1', 'FTA1', 'OR1', 'DR1', 'Ast1', 'TO1', 'Stl1', 'Blk1', 'PF1', 'Win1')]\n    season_meanw2 <- finalData[which(finalData$Time < times[i] & finalData$TeamID1 == team2),c('Score1', 'FGM1', 'FGA1', 'FGM31', 'FGA31', 'FTM1', 'FTA1', 'OR1', 'DR1', 'Ast1', 'TO1', 'Stl1', 'Blk1', 'PF1', 'Win1')]\n    season_meanl1 <- finalData[which(finalData$Time < times[i] & finalData$TeamID2 == team1),c('Score2', 'FGM2', 'FGA2', 'FGM32', 'FGA32', 'FTM2', 'FTA2', 'OR2', 'DR2', 'Ast2', 'TO2', 'Stl2', 'Blk2', 'PF2', 'Win2')]\n    season_meanl2 <- finalData[which(finalData$Time < times[i] & finalData$TeamID2 == team2),c('Score2', 'FGM2', 'FGA2', 'FGM32', 'FGA32', 'FTM2', 'FTA2', 'OR2', 'DR2', 'Ast2', 'TO2', 'Stl2', 'Blk2', 'PF2', 'Win2')]\n   \n    season_avw1 <- data.frame()\n    season_avl2 <- data.frame()\n    if(nrow(season_meanw1) !=0){\n      \n      \n    season_av1 <- as.data.frame(t(colMeans(season_meanw1[,c(1:(length(cols)-1))])))\n    #add new variables here\n    season_av1[,'Wins'] <- sum(season_meanw1[,'Win1'])\n    season_av1[,'Losses'] <- length(season_meanw1[,'Win1']) - sum(season_meanw1[,'Win1'])\n    \n    \n    }\n    else{\n      season_av1 <- rep(0, length(cols)+1)\n      dim(season_av1) <- c(1,length(cols)+1)\n      season_av1 <- as.data.frame(season_av1)\n    }\n    if(nrow(season_meanl1) !=0){\n      temp_df <- as.data.frame(t(colMeans(season_meanl1[,c(1:(length(cols)-1))])))\n      temp_df[,'Wins'] <- sum(season_meanl1[,'Win2'])\n      temp_df[,'Losses'] <- length(season_meanl1[,'Win2']) - sum(season_meanl1[,'Win2'])\n      #add new variables here\n      season_av1 <- season_av1 + temp_df\n      \n      \n    }\n    else{\n      season_av1 <- as.data.frame(season_av1)\n    }\n    colnames(season_av1) <- c('PPG1', 'AFGM1', 'AFGA1', 'AFGM31', 'AFGA31', 'AFTM1', 'AFTA1', 'AOR1', 'ADR1', 'AAst1', 'ATO1', 'AStl1', 'ABlk1', 'APF1', 'Tot_Win1', 'Tot_Loss1')\n    if(nrow(season_meanw2) !=0){\n      \n      season_av2 <- as.data.frame(t(colMeans(season_meanw2[,c(1:(length(cols)-1))])))\n      season_av2[,'Wins'] <- sum(season_meanw2[,'Win1'])\n      season_av2[,'Losses'] <- length(season_meanw2[,'Win1']) - sum(season_meanw2[,'Win1'])\n      \n    }\n    else{\n      season_av2 <- rep(0, length(cols)+1)\n      dim(season_av2) <- c(1,length(cols)+1)\n      season_av2 <- as.data.frame(season_av2)\n    }\n    if(nrow(season_meanl2) !=0){\n      temp_df <- as.data.frame(t(colMeans(season_meanl2[,c(1:(length(cols)-1))])))\n      temp_df[,'Wins'] <- sum(season_meanl2[,'Win2'])\n      temp_df[,'Losses'] <- length(season_meanl2[,'Win2']) - sum(season_meanl2[,'Win2'])\n      season_av2 <- season_av2 + temp_df\n    }\n    else{\n      season_av2 <- as.data.frame(season_av2)\n      \n    }\n    colnames(season_av2) <- c('PPG2', 'AFGM2', 'AFGA2', 'AFGM32', 'AFGA32', 'AFTM2', 'AFTA2', 'AOR2', 'ADR2', 'AAst2', 'ATO2', 'AStl2', 'ABlk2', 'APF2', 'Tot_Win2', 'Tot_Loss2')\n    final_rows <- finalData[rownums,]\n    final_stack <- cbind(final_rows, season_av1, season_av2)\n    top_25_1 <- rankingdata[which(rankingdata$RankingDayNum < times[i] & rankingdata$TeamID == team1),]\n    if(length(top_25_1$RankingDayNum) != 0){\n        top_25_recent_1 <- top_25_1[which(top_25_1$RankingDayNum == max(top_25_1$RankingDayNum)),]\n        final_stack[,'APTop1'] <- 1\n        }\n      else{\n         final_stack[,'APTop1'] <- 0 \n      }\n    conf_1 <- as.character(confData[which(confData$TeamID == team1 & confData$Season == year),'ConfAbbrev'])\n    final_stack[,'Conf1'] <- conf_1\n    \n    top_25_2 <- rankingdata[which(rankingdata$RankingDayNum < times[i] & rankingdata$TeamID == team2),]\n    if(length(top_25_2$RankingDayNum) != 0){\n        top_25_recent_2 <- top_25_2[which(top_25_2$RankingDayNum == max(top_25_2$RankingDayNum)),]\n        final_stack[,'APTop2'] <- 1\n    }\n    else{\n         final_stack[,'APTop2'] <- 0 \n      }\n    conf_2 <- as.character(confData[which(confData$TeamID == team2 & confData$Season == year),'ConfAbbrev'])\n    final_stack[,'Conf2'] <- conf_2\n    \n    final_stack[, 'TPT1'] <- final_stack[,'AFGM31']/final_stack[,'AFGA31']\n    final_stack[, 'TPT2'] <- final_stack[,'AFGM32']/final_stack[,'AFGA32']\n    return_data <- rbind(return_data, final_stack)\n    \n  }\n  \n\n\n}\nreturn_data$Win1 <- as.factor(return_data$Win1)\nreturn_data$Conf1 <- factor(return_data$Conf1)\nreturn_data$Conf2 <- factor(return_data$Conf2)\nreturn(as.data.frame(return_data))\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-04T21:57:19.221012Z","iopub.execute_input":"2022-08-04T21:57:19.222670Z","iopub.status.idle":"2022-08-04T21:57:19.233937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#######################\n#creates test data with each possible matchup and associated attributes\n#times <- unique times vector\n#timeData <- dataframe constrcuted with construct_Time_Data function\n#matchups <- vector containing all unique matchups\n#######################\n\nconstruct_Test_Data <- function(timeData, matchups){\n  return_data <- data.frame()\n  for(i in 1:length(matchups[,1])){\n  final_stack <- data.frame()\n  team1 <- matchups[i,]$Team1\n  team1_times <- timeData[which(timeData$TeamID1 == team1), 'Time']\n  team1_2times <- timeData[which(timeData$TeamID2 == team1), 'Time']\n  team2 <- matchups[i,]$Team2\n  team2_times <- timeData[which(timeData$TeamID1 == team2), 'Time']\n  team2_2times <- timeData[which(timeData$TeamID2 == team2), 'Time']\n      team1_max <- 0\n      team2_max <- 0\n    if(max(team1_times) > max(team1_2times)){\n      season_mean1 <- timeData[which(timeData$Time == max(team1_times) & timeData$TeamID1 == team1),c('PPG1', 'AFGM1', 'AFGA1', 'AFGM31', 'AFGA31', 'AFTM1', 'AFTA1', 'AOR1', 'ADR1', 'AAst1', 'ATO1', 'AStl1', 'ABlk1', 'APF1', 'Tot_Win1', 'Tot_Loss1','APTop1', 'Conf1', 'TPT1' )]\n      team1_max <- max(team1_times)\n    }\n      else{\n        season_mean1 <- timeData[which(timeData$Time == max(team1_2times)& timeData$TeamID2 == team1),c('PPG2', 'AFGM2', 'AFGA2', 'AFGM32', 'AFGA32', 'AFTM2', 'AFTA2', 'AOR2', 'ADR2', 'AAst2', 'ATO2', 'AStl2', 'ABlk2', 'APF2', 'Tot_Win2', 'Tot_Loss2', 'APTop2', 'Conf2', 'TPT2')]\n        team1_max <- max(team1_2times)\n        colnames(season_mean1) <- c('PPG1', 'AFGM1', 'AFGA1', 'AFGM31', 'AFGA31', 'AFTM1', 'AFTA1', 'AOR1', 'ADR1', 'AAst1', 'ATO1', 'AStl1', 'ABlk1', 'APF1', 'Tot_Win1', 'Tot_Loss1', 'APTop1', 'Conf1', 'TPT1')\n      }\n      if(max(team2_times) > max(team2_2times)){\n        season_mean2 <- timeData[which(timeData$Time == max(team2_times) & timeData$TeamID1 == team2),c('PPG1', 'AFGM1', 'AFGA1', 'AFGM31', 'AFGA31', 'AFTM1', 'AFTA1', 'AOR1', 'ADR1', 'AAst1', 'ATO1', 'AStl1', 'ABlk1', 'APF1', 'Tot_Win1', 'Tot_Loss1', 'APTop1', 'Conf1', 'TPT1')]\n        team2_max <- max(team2_times)\n        colnames(season_mean2) <- c('PPG2', 'AFGM2', 'AFGA2', 'AFGM32', 'AFGA32', 'AFTM2', 'AFTA2', 'AOR2', 'ADR2', 'AAst2', 'ATO2', 'AStl2', 'ABlk2', 'APF2', 'Tot_Win2', 'Tot_Loss2', 'APTop2', 'Conf2', 'TPT2')\n      }\n      else{\n        season_mean2 <- timeData[which(timeData$Time == max(team2_2times) & timeData$TeamID2 == team2),c('PPG2', 'AFGM2', 'AFGA2', 'AFGM32', 'AFGA32', 'AFTM2', 'AFTA2', 'AOR2', 'ADR2', 'AAst2', 'ATO2', 'AStl2', 'ABlk2', 'APF2', 'Tot_Win2', 'Tot_Loss2', 'APTop2', 'Conf2', 'TPT2')]\n        \n        team2_max <- max(team2_2times)\n      }\n    #colnames(season_mean1) <- c('PPG', 'AFGM2', 'AFGA2', 'AFGM32', 'AFGA32', 'AFTM2', 'AFTA2', 'AOR2', 'ADR2', 'AAst2', 'ATO2', 'AStl2', 'ABlk2', 'APF2', 'Tot_Win2')\n    final_stack <- cbind(matchups[i,], season_mean1, season_mean2)\n    \n\n      return_data <- rbind(return_data, final_stack)\n}\n\n\n\nreturn(na.omit(return_data))\n\n\n\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-04T21:57:19.236053Z","iopub.execute_input":"2022-08-04T21:57:19.237416Z","iopub.status.idle":"2022-08-04T21:57:19.248312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#####################\n#Construct column that shows what the result of a matchup resulted in a win or loss\n#####################\ngetActualWins <- function(matches, tourney){\n  matches_with_wins <- data.frame()\n  for(i in 1:length(matches[,1])){\n    win <- 0\n    w_matches <- tourney[which(tourney$WTeamID == matches[i,]$Team1 & tourney$LTeamID == matches[i,]$Team2),]\n    if(length(w_matches[,1]) != 0){\n      win <- 1\n    }\n    matches_with_wins <- rbind(matches_with_wins, cbind(matches[i,], win))\n  }\n  colnames(matches_with_wins) <- c(\"Team1\", \"Team2\", \"Win\")\n  return(matches_with_wins)\n}\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T21:57:19.251236Z","iopub.execute_input":"2022-08-04T21:57:19.252877Z","iopub.status.idle":"2022-08-04T21:57:19.261443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"#####################\n#Constructs all training data up to and including regular season for a given year\n#####################\nconstruct_all_training_data <- function(RegseasonD, Ranking,ConfData, year){\n  years <- unique(RegseasonD$Season)\n  all_data <- data.frame()\n  for(i in 1:length(years)){\n    print(years[i])\n    sub <- RegseasonD[which(RegseasonD$Season == as.character(years[i])),]\n    finaldata_sub <- BuildWinLossData(sub)\n    ranking <- RankingsD[which(RankingsD$Season == years[i]),]\n    ap_rankings <- ranking[which(ranking$SystemName == 'AP'),]\n    time_series <- construct_Time_Data(finaldata_sub, ap_rankings, ConfData, year)\n    all_data <- rbind(all_data, time_series)\n  }\n  return(all_data)\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-04T21:57:19.263511Z","iopub.execute_input":"2022-08-04T21:57:19.265492Z","iopub.status.idle":"2022-08-04T21:57:19.274656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##################\n#Constructs predictions for a given year\n#year - year of tournament to make predictions as 'YYYY'\n#regseason - regular season data\n#tourneyD - tournament data\n#rankD - season ranking data\n#forest - random forest corresponding to a certain year\n#type - type of model, logistic needs to be handled differently\n#ConfData - conference data\n##################\nconstruct_predictions_by_year <- function(year, regseason, tourneyD, rankD, forest, type, ConfData){\n  sub <- regseason[which(regseason$Season == year),]\n  finaldata_sub <- BuildWinLossData(sub)\n  times <- unique(sub$Time)\n  ranking <- rankD[which(rankD$Season == year),]\n  ap_rankings <- ranking[which(ranking$SystemName == 'AP'),]\n  time_series <- construct_Time_Data(finaldata_sub, ap_rankings, ConfData, year)\n  Tsub <- tourneyD[which(tourneyD$Season == year),]\n  matches <- getMatchups(Tsub)\n  test_data <- construct_Test_Data(time_series, matches)\n  actuals <- getActualWins(matches, Tsub)\n  if(type == \"logistic\"){\n    preds <- predict(forest, test_data, type = \"response\")\n      pred <- cbind(preds, actuals)\n  }\n  else{\n    preds <- predict(forest, test_data,type = \"prob\")\n    pred <- cbind(preds[,1], actuals)\n  }\n\n  \n  colnames(pred) <- c('Prob', 'Team1', 'Team2', 'Win')\n  return(pred)\n}\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T21:57:19.276749Z","iopub.execute_input":"2022-08-04T21:57:19.278636Z","iopub.status.idle":"2022-08-04T21:57:19.288627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#######################\n#Constructs predictions for year 2019\n#year\n#regseason - regular season data up through year 2019\n#tourneyD - tournament data for the year 2019\n#rankD - ranking data for year 2019\n#forest - model object\n#type - type of model object\n#ConfData - conference data for year 2019\n#######################\n\nconstruct_predictions_by_year20 <- function(regseason, tourneyD, rankD, forest, type, ConfData){\n  sub <- regseason[which(regseason$Season == '2020'),]\n  finaldata_sub <- BuildWinLossData(sub)\n  times <- unique(sub$Time)\n  ranking <- rankD[which(rankD$Season == '2020'),]\n  ap_rankings <- ranking[which(ranking$SystemName == 'AP'),]\n  time_series <- construct_Time_Data(finaldata_sub, ap_rankings, ConfData, '2020')\n  Tsub <- tourneyD[which(tourneyD$Season == '2020'),]\n  matches <- getMatchups19(Tsub)\n  test_data <- construct_Test_Data(time_series, matches)\n  if(type == \"logistic\"){\n    preds <- predict(forest, test_data, type = \"response\")\n    pred <- cbind(preds, matches)\n  }\n  else{\n    preds <- predict(forest, test_data, type = \"prob\")\n    pred <- cbind(preds[,1], matches)\n  }\n\n  colnames(pred) <- c('Prob', 'Team1', 'Team2')\n  return(pred)\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-04T21:57:19.290618Z","iopub.execute_input":"2022-08-04T21:57:19.291940Z","iopub.status.idle":"2022-08-04T21:57:19.302690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"buildSubmission <- function(data14, data15, data16, data17, data18, data19, data20, filename){\n  years <- c('2014', '2015', '2016', '2017', '2018', '2019', '2020')\n  submission <- data.frame()\n  data14 <- data14[order(data14$Team1),]\n  data15 <- data15[order(data15$Team1),]\n  data16 <- data16[order(data16$Team1),]\n  data17 <- data17[order(data17$Team1),]\n  data18 <- data18[order(data18$Team1),]\n  data19 <- data19[order(data19$Team1),]\n  \n  for(i in 1:length(data14[,1])){\n    id_string <- paste(years[1], '_', data14[i,]$Team1, '_', data14[i,]$Team2, sep = \"\")\n    row <- cbind(id_string, data14[i,]$Prob)\n    submission <- rbind(submission, row)\n  }\n  for(i in 1:length(data15[,1])){\n    id_string <- paste(years[2], '_', data15[i,]$Team1, '_', data15[i,]$Team2, sep = \"\")\n    row <- cbind(id_string, data15[i,]$Prob)\n    submission <- rbind(submission, row)\n  }\n  for(i in 1:length(data16[,1])){\n    id_string <- paste(years[3], '_', data16[i,]$Team1, '_', data16[i,]$Team2, sep = \"\")\n    row <- cbind(id_string, data16[i,]$Prob)\n    submission <- rbind(submission, row)\n  }\n  for(i in 1:length(data17[,1])){\n    id_string <- paste(years[4], '_', data17[i,]$Team1, '_', data17[i,]$Team2, sep = \"\")\n    row <- cbind(id_string, data17[i,]$Prob)\n    submission <- rbind(submission, row)\n  }\n  for(i in 1:length(data18[,1])){\n    id_string <- paste(years[5], '_', data18[i,]$Team1, '_', data18[i,]$Team2, sep = \"\")\n    row <- cbind(id_string, data18[i,]$Prob)\n    submission <- rbind(submission, row)\n  }\n      for(i in 1:length(data19[,1])){\n    id_string <- paste(years[6], '_', data19[i,]$Team1, '_', data19[i,]$Team2, sep = \"\")\n    row <- cbind(id_string, data19[i,]$Prob)\n    submission <- rbind(submission, row)\n  }\n  colnames(submission) <- c(\"ID\", \"Pred\")\n  write.csv(submission, file = filename, row.names=FALSE)\n}\n\n\nbuildSubmission19 <- function(data19, filename){\n  years <- c('2019')\n  submission <- data.frame()\n  data19 <- data19[order(data19$Team1),]\n  for(i in 1:length(data19[,1])){\n    id_string <- paste(years[1], '_', data19[i,]$Team1, '_', data19[i,]$Team2, sep = \"\")\n    row <- cbind(id_string, data19[i,]$Prob)\n    submission <- rbind(submission, row)\n  }\n  colnames(submission) <- c(\"ID\", \"Pred\")\n  write.csv(submission, file = filename, row.names=FALSE)\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-04T21:57:19.304737Z","iopub.execute_input":"2022-08-04T21:57:19.306147Z","iopub.status.idle":"2022-08-04T21:57:19.318523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"library(keras)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T21:57:19.320609Z","iopub.execute_input":"2022-08-04T21:57:19.322434Z","iopub.status.idle":"2022-08-04T21:57:19.331316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create training data with regular season featurees through 2018\nall_training_data20 <- construct_all_training_data(RegseasonD, RankingsD, Team_conf, '2020')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T21:57:19.333358Z","iopub.execute_input":"2022-08-04T21:57:19.335142Z","iopub.status.idle":"2022-08-04T22:24:25.495989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat <- c('PPG1','AFGM1','AFGA1','AFGM31','AFGA31','AFTM1','AFTA1','AOR1','ADR1','AAst1','ATO1','AStl1','ABlk1','APF1','Tot_Win1','Tot_Loss1', 'APTop1','PPG2','AFGM2','AFGA2','AFGM32','AFGA32','AFTM2'\n          ,'AFTA2','AOR2','ADR2',\n          'AAst2','ATO2','AStl2','ABlk2','APF2','Tot_Win2','Tot_Loss2', 'APTop2')\ny_col <- c('Win1')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:24:25.498218Z","iopub.execute_input":"2022-08-04T22:24:25.499009Z","iopub.status.idle":"2022-08-04T22:24:25.510356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"construct_all_training_data20 <- function(RegseasonD, Ranking,ConfData, year){\n  all_data <- data.frame()\n    sub <- RegseasonD[which(RegseasonD$Season == as.character(year)),]\n    finaldata_sub <- BuildWinLossData(sub)\n    ranking <- RankingsD[which(RankingsD$Season == year),]\n    ap_rankings <- ranking[which(ranking$SystemName == 'AP'),]\n    time_series <- construct_Time_Data(finaldata_sub, ap_rankings, ConfData, year)\n    all_data <- rbind(all_data, time_series)\n  \n  return(all_data)\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:24:25.512325Z","iopub.execute_input":"2022-08-04T22:24:25.514242Z","iopub.status.idle":"2022-08-04T22:24:25.523901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_data20 <- construct_all_training_data20(RegseasonD20, RankingsD20, Team_conf, '2020')\n#x_train['Conf1'] <- factor(x_train['Conf1'])\n#x_train['Conf2'] <- factor(x_train['Conf2'])","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:29:08.972852Z","iopub.execute_input":"2022-08-04T22:29:08.974874Z","iopub.status.idle":"2022-08-04T22:30:49.145895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_training_data20 <- na.omit(as.matrix(all_training_data20))\nval_data20 <- na.omit(as.matrix(val_data20))\n\nx_train <- all_training_data20[, feat]\ny_train <- all_training_data20[, y_col]\nx_val <- val_data20[, feat]\ny_val <- val_data20[, y_col]\n#y_train <- to_categorical(y_train[,1], 2)\n#y_val <- to_categorical(y_val[,1], 2)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:35:04.294035Z","iopub.execute_input":"2022-08-04T22:35:04.296037Z","iopub.status.idle":"2022-08-04T22:35:09.236537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model <- keras_model_sequential() %>% \n  layer_dense(units = 32, activation = \"relu\", input_shape = c(length(feat))) %>% \n  layer_dense(units = 16, activation = \"relu\") %>% \n  layer_dense(units = 1, activation = \"sigmoid\")","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:35:13.065544Z","iopub.execute_input":"2022-08-04T22:35:13.066382Z","iopub.status.idle":"2022-08-04T22:35:18.399368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model %>% compile(\n  optimizer = 'adam',\n  loss = \"binary_crossentropy\",\n  metrics = c(\"accuracy\")\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:35:18.402839Z","iopub.execute_input":"2022-08-04T22:35:18.406032Z","iopub.status.idle":"2022-08-04T22:35:18.472406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fit the model and record epochs in 'history'\nhistory <- model %>% fit(\n     x_train, y_train, \n     epochs = 15, \n     batch_size = 50, \n     validation_split = 0.2\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:35:23.556416Z","iopub.execute_input":"2022-08-04T22:35:23.559930Z","iopub.status.idle":"2022-08-04T22:36:12.032342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot(history)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:43:42.157303Z","iopub.execute_input":"2022-08-04T22:43:42.159212Z","iopub.status.idle":"2022-08-04T22:43:42.933960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Evaluate Model\nmodel %>% evaluate(x_val, y_val,verbose = 0)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T22:43:49.005737Z","iopub.execute_input":"2022-08-04T22:43:49.006857Z","iopub.status.idle":"2022-08-04T22:43:49.201115Z"},"trusted":true},"execution_count":null,"outputs":[]}]}