{"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.0.5"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Intro\nThis is a very quick submission based on just one feature - the y-position of the ball.\nWe use the .parquet files generously provided @Reybahl in [this notebook](https://www.kaggle.com/code/reymaster/compress-files-parquet-7x-loading-speedup). Please up-vote that notebook","metadata":{}},{"cell_type":"code","source":"library(tidyverse) # metapackage of all tidyverse packages\nlibrary(arrow)\nlibrary(ggplot2)\ndata_path <- \"../input/tps-oct-2022-compressed-parquet-files/\"","metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","execution":{"iopub.status.busy":"2022-10-01T07:55:41.314964Z","iopub.execute_input":"2022-10-01T07:55:41.316469Z","iopub.status.idle":"2022-10-01T07:55:41.331946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x <- 0 # we will use file 0 only for this model\ntraindata <- read_parquet(paste0(data_path, \"train_\", x, \".parquet.gzip\"))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see that the closer the ball is to the opponet's side, the greater the probability of scoring within the next 10 seconds. We will just use the ball's y position for now.","metadata":{}},{"cell_type":"code","source":"ggplot(traindata, aes(x = ball_pos_x, y = ball_pos_y, fill = team_A_scoring_within_10sec)) +\n  geom_tile() + ggtitle(\"probability of A scoring\")","metadata":{"execution":{"iopub.status.busy":"2022-10-01T07:59:06.981433Z","iopub.execute_input":"2022-10-01T07:59:06.982868Z","iopub.status.idle":"2022-10-01T07:59:17.386771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove the center spot from the model\n#   and divide the y dimension of the field into 1000 sections -- you can tune this\nNUM_BUCKETS = 1000\ntraindata <- traindata %>% mutate(isCenter = (!ball_pos_x & !ball_pos_y),\n          ball_pos_y_bucket = ntile(ball_pos_y, NUM_BUCKETS))\n\n# get the mean target value for that bucket\ntraindata_summary <- traindata %>% mutate(ball_pos_y_bucket = ifelse(isCenter, NA, ball_pos_y_bucket)) %>%\n    group_by(ball_pos_y_bucket) %>%\n    summarise(\n        bucket_center = mean(ball_pos_y),\n        pct_score_A = mean(team_A_scoring_within_10sec),\n        pct_score_B = mean(team_B_scoring_within_10sec)\n    )\n\n# fit to lowess model\ntrainset <- traindata_summary %>% filter(pct_score_A < 0.14, !is.na(pct_score_A))\nmodelx_A <- loess(pct_score_A ~ bucket_center, trainset)\n\ntrainset <- traindata_summary %>% filter(pct_score_B < 0.14, !is.na(pct_score_B))\nmodelx_B <- loess(pct_score_B ~ bucket_center, trainset)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T08:34:30.794092Z","iopub.execute_input":"2022-10-01T08:34:30.795494Z","iopub.status.idle":"2022-10-01T08:34:31.852212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# deal with the edge cases\ncenter_pctA <- traindata_summary$pct_score_A[is.na(traindata_summary$ball_pos_y_bucket)]\ncenter_pctB <- traindata_summary$pct_score_B[is.na(traindata_summary$ball_pos_y_bucket)]\nmax_model_range_A <- max(modelx_A$x)\nmin_model_range_A <- min(modelx_A$x)\nmax_model_range_B <- max(modelx_B$x)\nmin_model_range_B <- min(modelx_B$x)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T08:35:40.601341Z","iopub.execute_input":"2022-10-01T08:35:40.603702Z","iopub.status.idle":"2022-10-01T08:35:40.626723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_model_range_A","metadata":{"execution":{"iopub.status.busy":"2022-10-01T08:35:31.643439Z","iopub.execute_input":"2022-10-01T08:35:31.644725Z","iopub.status.idle":"2022-10-01T08:35:31.658623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## load test set\ntestdata <- read_parquet(paste0(data_path, \"test.parquet.gzip\")) %>%\n    mutate(isCenter = (!ball_pos_x & !ball_pos_y),\n          bucket_center = ball_pos_y)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T08:31:24.194231Z","iopub.execute_input":"2022-10-01T08:31:24.195615Z","iopub.status.idle":"2022-10-01T08:31:24.650403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculate predictions\nsubmission <- testdata %>%\n    transmute( id,\n                    team_A_scoring_within_10sec =\n                    ifelse(isCenter, center_pctA,\n                          ifelse(ball_pos_y > max_model_range_A, max(traindata_summary$pct_score_A),\n                                 ifelse(ball_pos_y < min_model_range_A, min(traindata_summary$pct_score_A),\n                                        predict(modelx_A, newdata = testdata)))),\n                    team_B_scoring_within_10sec =\n                    ifelse(isCenter, center_pctB,\n                          ifelse(ball_pos_y > max_model_range_B, min(traindata_summary$pct_score_B),\n                                 ifelse(ball_pos_y < min_model_range_B, max(traindata_summary$pct_score_B),\n                                        predict(modelx_B, newdata = testdata)))),\n    )\n\n# output results\nwrite.csv(submission, file = \"submission.csv\", row.names=F)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T08:36:35.808285Z","iopub.execute_input":"2022-10-01T08:36:35.811963Z","iopub.status.idle":"2022-10-01T08:36:38.562159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualize predictions\nhist(submission$team_A_scoring_within_10sec,100)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T07:46:47.853199Z","iopub.execute_input":"2022-10-01T07:46:47.855520Z","iopub.status.idle":"2022-10-01T07:46:47.969564Z"},"trusted":true},"execution_count":null,"outputs":[]}]}