{"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.0.5"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"library(tidyverse) # metapackage of all tidyverse packages\nlibrary(reticulate) # to access python libraries\nlibrary(ggthemes) # for ggplot themes\nlibrary(RColorBrewer) # for ggplot colors\nlibrary(tictoc) # for time measurement\nlist.files(path = '../input/predict-ai-model-runtime')","metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","execution":{"iopub.status.busy":"2023-09-26T23:04:51.627086Z","iopub.execute_input":"2023-09-26T23:04:51.628257Z","iopub.status.idle":"2023-09-26T23:04:51.664290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the list of .npzs for this competition\nnpz_list <- list.files(path = '../input/predict-ai-model-runtime/npz_all/npz', full.names = TRUE, recursive = TRUE)\n\n# Convert npz_list to dataframe.  \nnpz_df <- data.frame(path = npz_list)\n\n# 7868 .npz files!\nnrow(npz_df)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T22:48:05.211879Z","iopub.execute_input":"2023-09-26T22:48:05.311508Z","iopub.status.idle":"2023-09-26T22:48:07.007820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I want to parse some information out of the filepath, and from the information presented in the notebook in Reference 1 below, the file structure is different depending on the configuration type. Tile configurations have one fewer subdirectory, so we will need to account for that.  \nnpz_df <- npz_df %>% separate(path, into = c('throw1', 'throw2', 'throw3', 'throw4', 'throw5', 'configuration', 'graph_type', 'throw6', 'train_test_valid', 'throw7'),sep = '/', remove = FALSE) %>% \nmutate(graph_type = ifelse(configuration == 'layout', paste(graph_type,throw6,sep=\"_\"),graph_type), train_test_valid = ifelse(configuration == 'layout', train_test_valid, throw6)) %>% \nselect(-throw1,-throw2,-throw3,-throw4,-throw5,-throw6,-throw7)\n\n# For now, I have consolidated the model with default & random for the layout configuration\nnpz_df %>% head()\nnpz_df %>% nrow()","metadata":{"execution":{"iopub.status.busy":"2023-09-26T22:48:07.019972Z","iopub.execute_input":"2023-09-26T22:48:07.021421Z","iopub.status.idle":"2023-09-26T22:48:07.256973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"npz_df <- npz_df %>% mutate(configuration = factor(configuration), graph_type = factor(graph_type), train_test_valid = factor(train_test_valid))\nstr(npz_df)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T22:48:07.259816Z","iopub.execute_input":"2023-09-26T22:48:07.260899Z","iopub.status.idle":"2023-09-26T22:48:07.303669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# With the file paths & data parsing complete, we can look at some summary information such as file counts for layout vs. tile\n\nnpz_df %>% group_by(configuration) %>% summarize(totals = n()) %>% ggplot(aes(x=configuration, y=totals, fill = configuration)) + geom_col() + scale_fill_brewer(palette = \"Blues\") + theme_clean()  ","metadata":{"execution":{"iopub.status.busy":"2023-09-26T22:48:07.306624Z","iopub.execute_input":"2023-09-26T22:48:07.307769Z","iopub.status.idle":"2023-09-26T22:48:07.866633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can also look at the file counts by graph type\n\nnpz_df %>% group_by(graph_type) %>% summarize(totals = n()) %>% ggplot(aes(x=reorder(graph_type, -totals), y=totals, fill = graph_type)) + \ngeom_col() + scale_fill_brewer(palette = \"Blues\") + theme_clean()","metadata":{"execution":{"iopub.status.busy":"2023-09-26T22:48:07.869903Z","iopub.execute_input":"2023-09-26T22:48:07.870994Z","iopub.status.idle":"2023-09-26T22:48:08.243760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# And finally look at file counts for our train & test data\n\nnpz_df %>% group_by(train_test_valid) %>% summarize(totals = n()) %>% \nggplot(aes(x=reorder(train_test_valid, -totals), y=totals, fill = train_test_valid)) + geom_col() + scale_fill_brewer(palette = \"Blues\") + theme_clean()","metadata":{"execution":{"iopub.status.busy":"2023-09-26T22:48:08.247048Z","iopub.execute_input":"2023-09-26T22:48:08.248387Z","iopub.status.idle":"2023-09-26T22:48:08.524725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Choose the first .npz file in our npz_df and see what files are contained in the zipped archive:\nunzip(npz_df[1,1], list = TRUE)\n\n# All .npy files!","metadata":{"execution":{"iopub.status.busy":"2023-09-26T22:48:08.528016Z","iopub.execute_input":"2023-09-26T22:48:08.529149Z","iopub.status.idle":"2023-09-26T22:48:08.565851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the numpy package\nnp <- import('numpy', convert = TRUE)","metadata":{"execution":{"iopub.status.busy":"2023-09-26T22:48:08.567739Z","iopub.execute_input":"2023-09-26T22:48:08.568716Z","iopub.status.idle":"2023-09-26T22:48:10.530531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display a single file (graph_type == 'xla') from the test set using np$load\nnpz_df_single <- npz_df %>% filter(graph_type == 'xla' & train_test_valid == 'test')\n\nnpz_file <- np$load(npz_df_single[1,1]) # grab the first .npz file location\nnpz_file$files # list files contained in the .npz\ntest_df <- npz_file$f['config_feat'] # view 'node_feat' file ","metadata":{"execution":{"iopub.status.busy":"2023-09-26T22:48:10.533774Z","iopub.execute_input":"2023-09-26T22:48:10.535567Z","iopub.status.idle":"2023-09-26T22:48:10.589184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Understanding the relationship between some of the key components\n# Looking at the relationship between # of nodes & mean run time with a scatter plot\n\n# Filter the data to train only\nunique_configs_df <- npz_df %>% filter(train_test_valid == 'train') %>% select(configuration, graph_type, path) %>% unique()\n\n# Create the empty df used to store the graph_type, num_nodes, & mean_runtime\nperformance_df <- data.frame(matrix(ncol = 3, nrow = 0))\nx <- c('graph_type', 'num_nodes', 'mean_runtime')\ncolnames(performance_df) <- x\n\n# Iterate through the unique configurations, grab the number of nodes & mean run times\nfor(i in 1:nrow(unique_configs_df)){\n    npz_files <- np$load(unique_configs_df[i,3])\n    performance_df <- rbind(performance_df, data.frame(graph_type = unique_configs_df[i,2],num_nodes = nrow(npz_files$f['node_feat']), mean_runtime = mean(npz_files$f['config_runtime'])))\n}\n\n# A plot of the relationship between number of nodes & mean runtime.\nperformance_df %>% ggplot(aes(x=num_nodes, y=mean_runtime, col=graph_type)) + geom_point() + scale_color_brewer(palette = \"Set1\") + theme_clean()","metadata":{"execution":{"iopub.status.busy":"2023-09-26T22:48:10.592099Z","iopub.execute_input":"2023-09-26T22:48:10.593704Z","iopub.status.idle":"2023-09-26T22:49:52.684403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# There appears to be at least one candidate for removal as an outlier\n\nperformance_df[which.max(performance_df$mean_runtime),]","metadata":{"execution":{"iopub.status.busy":"2023-09-26T22:49:52.686517Z","iopub.execute_input":"2023-09-26T22:49:52.687634Z","iopub.status.idle":"2023-09-26T22:49:52.704364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Looking at the relationships between number of configurations & mean run time with a scatter plot\n# Focussing on tile, xla as that represents ~ 73% of training data\n\n# Filter the data to train only\nunique_configs_df <- npz_df %>% filter(train_test_valid == 'train', configuration == 'tile', graph_type == 'xla') %>% select(configuration, graph_type, path) %>% unique()\n\n# Create the empty df used to store the graph_type, num_nodes, & mean_runtime\nconfigs_df <- data.frame(matrix(ncol = 4, nrow = 0))\nx <- c('graph_type', 'num_configs', 'mean_runtime', 'id')\ncolnames(configs_df) <- x\n\n# Iterate through the unique configurations, grab the number of nodes & mean run times\nfor(i in 1:nrow(unique_configs_df)){\n    npz_files <- np$load(unique_configs_df[i,3])\n    configs_df <- rbind(configs_df, data.frame(graph_type = unique_configs_df[i,2], num_configs = nrow(npz_files$f['config_feat']),mean_runtime = mean(npz_files$f['config_runtime']), id=unique_configs_df[i,3]))\n}\n\n# Plot of the relationship between number of configs & mean runtime.\nconfigs_df %>% ggplot(aes(x=num_configs, y=mean_runtime, col = graph_type)) + geom_point() + scale_color_brewer(palette = \"Set1\") + theme_clean()","metadata":{"execution":{"iopub.status.busy":"2023-09-26T22:49:52.707024Z","iopub.execute_input":"2023-09-26T22:49:52.708382Z","iopub.status.idle":"2023-09-26T22:50:39.100494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tic('Meta building for train tile xla')\n\n# Filter the data to train only\nunique_configs_df <- npz_df %>% filter(train_test_valid == 'train', configuration == 'tile', graph_type == 'xla') %>% select(configuration, graph_type, path) %>% unique()\n\n# Create the empty df used to store the graph_type, num_nodes, & mean_runtime\ntile_meta_df <- data.frame(matrix(ncol = 26, nrow = 0))\nx <- c(paste('cf',1:24), 'runtime', 'runtime_normalizers')\ncolnames(tile_meta_df) <- x\n\n# Iterate through the unique configurations, grab the number of nodes & mean run times\nfor(i in 1:nrow(unique_configs_df)){\n    npz_file <- np$load(unique_configs_df[i,3]) # grab the first .npz file location\n    npz_file$files # list files contained in the .npz\n\n    # config_feat\n    cf_df <- data.frame(npz_file$f['config_feat'])\n    colnames(cf_df) <- c(paste('cf',1:24))\n\n    # config_runtime\n    cr_df <- data.frame(npz_file$f['config_runtime'])\n    colnames(cr_df) <- 'runtime'\n\n    # config_runtime_normalizers\n    crn_df <- data.frame(npz_file$f['config_runtime_normalizers'])\n    colnames(crn_df) <- 'runtime_normalizers'\n\n    # combine all into a single df\n    tmp_df <- cbind(cf_df, cr_df, crn_df)\n    \n    tile_meta_df <- rbind(tile_meta_df, tmp_df)\n    \n}\n\ntoc()","metadata":{"execution":{"iopub.status.busy":"2023-09-26T23:05:02.502826Z","iopub.execute_input":"2023-09-26T23:05:02.504166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# For saving the meta file we created\nwrite_csv(tile_meta_df, 'tile_meta.csv')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Exploring sample_submission.csv\nsample_submission <- read.csv('../input/predict-ai-model-runtime/sample_submission.csv')\n\nsample_submission %>% head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Submit the sample_submission to test!\nwrite_csv(sample_submission, 'submission.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I plan on adding more to this as time permits.  If it has been helpful, please upvote!  Thanks!","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# References -- thanks!\n# (1) https://www.kaggle.com/code/rishabh15virgo/first-impression-understand-data-eda-baseline-15 ","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}