From f4d52e8937a689250393759ee12d33080509a8f4 Mon Sep 17 00:00:00 2001 From: Michael Wagner Date: Thu, 20 Aug 2026 09:33:36 -0500 Subject: [PATCH 01/16] Add first draft of fix --- pipeline/01-train.R | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/pipeline/01-train.R b/pipeline/01-train.R index e6805a89..b2d2eb42 100644 --- a/pipeline/01-train.R +++ b/pipeline/01-train.R @@ -315,9 +315,14 @@ if (cv_enable) { ) # Save tuning results to file. This is a data.frame where each row is one - # CV iteration + # CV iteration. tune >= 2.0 adds a `trace` column (rlang call stack objects) + # to each nested .notes tibble, which arrow cannot serialize, so drop it + # before writing. any_of() keeps this a no-op on tune 1.x results lgbm_search %>% lightsnip::axe_tune_data() %>% + mutate( + .notes = purrr::map(.notes, ~ dplyr::select(.x, -dplyr::any_of("trace"))) + ) %>% arrow::write_parquet(paths$output$parameter_raw$local) # Save the parameter ranges searched while tuning From 6aba6f5081668d7d5c6062ed6053bbef2faa96da Mon Sep 17 00:00:00 2001 From: Michael Wagner Date: Thu, 20 Aug 2026 11:50:16 -0500 Subject: [PATCH 02/16] Swap to targeted removal instead of silent --- pipeline/01-train.R | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/pipeline/01-train.R b/pipeline/01-train.R index b2d2eb42..5a3946ba 100644 --- a/pipeline/01-train.R +++ b/pipeline/01-train.R @@ -317,12 +317,10 @@ if (cv_enable) { # Save tuning results to file. This is a data.frame where each row is one # CV iteration. tune >= 2.0 adds a `trace` column (rlang call stack objects) # to each nested .notes tibble, which arrow cannot serialize, so drop it - # before writing. any_of() keeps this a no-op on tune 1.x results + # before writing lgbm_search %>% lightsnip::axe_tune_data() %>% - mutate( - .notes = purrr::map(.notes, ~ dplyr::select(.x, -dplyr::any_of("trace"))) - ) %>% + mutate(.notes = purrr::map(.notes, ~ dplyr::select(.x, -trace))) %>% arrow::write_parquet(paths$output$parameter_raw$local) # Save the parameter ranges searched while tuning From 57438892e2b14ad9dc5e55ceee93c280fb6a9d77 Mon Sep 17 00:00:00 2001 From: Michael Wagner Date: Thu, 20 Aug 2026 11:51:37 -0500 Subject: [PATCH 03/16] Decrease CV computation --- params.yaml | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/params.yaml b/params.yaml index dacea015..1db3bdc9 100644 --- a/params.yaml +++ b/params.yaml @@ -20,7 +20,7 @@ run_note: Potential residential baseline with SHAP values, revert DVC path toggle: # Should the train stage run full cross-validation? Otherwise, the model # will be trained with the default hyperparameters specified below - cv_enable: false + cv_enable: true # Should SHAP values be calculated for this run in the interpret stage? Can be # desirable to save time when testing many models @@ -31,7 +31,7 @@ toggle: # Upload all modeling artifacts and results to S3 in the upload stage. Set # to false if you are not a CCAO employee - upload_enable: true + upload_enable: false # Should the feature report be run, which compares the inputs to the previous # year's model inputs? @@ -96,7 +96,7 @@ input: # by year, township, and class to maintain representation across categories subset: # Whether to use a subset of the training data instead of the full dataset - enable: false + enable: true # Fraction of the training data to keep (between 0 and 1) fraction: 0.25 @@ -119,7 +119,7 @@ cv: # Number of folds to use for cross-validation. For v-fold CV, the data will be # randomly split. For rolling-origin, the data will be split into V chunks by # time, with each chunk/period calculated automatically - num_folds: 10 + num_folds: 2 # The number of months time-based folds should overlap each other. Only # applicable to rolling-origin CV. See https://www.tmwr.org/resampling#rolling @@ -130,7 +130,7 @@ cv: initial_set: 20 # Max number of total search iterations - max_iterations: 50 + max_iterations: 1 # Max number of search iterations without improvement before stopping search no_improve: 15 From f9ffcd11875d4cbf33ed165c769ebf91f7bb36de Mon Sep 17 00:00:00 2001 From: Michael Wagner Date: Thu, 20 Aug 2026 14:04:54 -0500 Subject: [PATCH 04/16] Rework docs --- pipeline/01-train.R | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/pipeline/01-train.R b/pipeline/01-train.R index 5a3946ba..e32478d9 100644 --- a/pipeline/01-train.R +++ b/pipeline/01-train.R @@ -315,11 +315,11 @@ if (cv_enable) { ) # Save tuning results to file. This is a data.frame where each row is one - # CV iteration. tune >= 2.0 adds a `trace` column (rlang call stack objects) - # to each nested .notes tibble, which arrow cannot serialize, so drop it - # before writing + # CV iteration lgbm_search %>% lightsnip::axe_tune_data() %>% + # Tune added a trace column (rlang stack objects) to each nested .notes + # tibble which arrow cannot serialize mutate(.notes = purrr::map(.notes, ~ dplyr::select(.x, -trace))) %>% arrow::write_parquet(paths$output$parameter_raw$local) From 458ad7861b2f74081b0853547c57ac72917b760f Mon Sep 17 00:00:00 2001 From: Michael Wagner Date: Thu, 20 Aug 2026 14:06:45 -0500 Subject: [PATCH 05/16] Revert params to master --- params.yaml | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/params.yaml b/params.yaml index 1db3bdc9..dacea015 100644 --- a/params.yaml +++ b/params.yaml @@ -20,7 +20,7 @@ run_note: Potential residential baseline with SHAP values, revert DVC path toggle: # Should the train stage run full cross-validation? Otherwise, the model # will be trained with the default hyperparameters specified below - cv_enable: true + cv_enable: false # Should SHAP values be calculated for this run in the interpret stage? Can be # desirable to save time when testing many models @@ -31,7 +31,7 @@ toggle: # Upload all modeling artifacts and results to S3 in the upload stage. Set # to false if you are not a CCAO employee - upload_enable: false + upload_enable: true # Should the feature report be run, which compares the inputs to the previous # year's model inputs? @@ -96,7 +96,7 @@ input: # by year, township, and class to maintain representation across categories subset: # Whether to use a subset of the training data instead of the full dataset - enable: true + enable: false # Fraction of the training data to keep (between 0 and 1) fraction: 0.25 @@ -119,7 +119,7 @@ cv: # Number of folds to use for cross-validation. For v-fold CV, the data will be # randomly split. For rolling-origin, the data will be split into V chunks by # time, with each chunk/period calculated automatically - num_folds: 2 + num_folds: 10 # The number of months time-based folds should overlap each other. Only # applicable to rolling-origin CV. See https://www.tmwr.org/resampling#rolling @@ -130,7 +130,7 @@ cv: initial_set: 20 # Max number of total search iterations - max_iterations: 1 + max_iterations: 50 # Max number of search iterations without improvement before stopping search no_improve: 15 From f42f259e36c27a9f949c8fd13fa2836a5bfd0047 Mon Sep 17 00:00:00 2001 From: Michael Wagner Date: Tue, 25 Aug 2026 09:40:41 -0500 Subject: [PATCH 06/16] Upload full proposed fix --- pipeline/01-train.R | 11 +++++++++-- pipeline/06-upload.R | 36 ++++++++++++++++++++++++++++-------- 2 files changed, 37 insertions(+), 10 deletions(-) diff --git a/pipeline/01-train.R b/pipeline/01-train.R index e32478d9..d93900f9 100644 --- a/pipeline/01-train.R +++ b/pipeline/01-train.R @@ -319,8 +319,15 @@ if (cv_enable) { lgbm_search %>% lightsnip::axe_tune_data() %>% # Tune added a trace column (rlang stack objects) to each nested .notes - # tibble which arrow cannot serialize - mutate(.notes = purrr::map(.notes, ~ dplyr::select(.x, -trace))) %>% + # tibble which arrow cannot serialize. Flatten each trace to plain text so + # the diagnostic content survives into the parquet (a few KB as text, vs + # gigabytes of captured environments if left as rlang objects) + mutate(.notes = purrr::map( + .notes, + ~ dplyr::mutate(.x, trace = purrr::map_chr(trace, function(tr) { + if (is.null(tr)) NA_character_ else paste(format(tr), collapse = "\n") + })) + )) %>% arrow::write_parquet(paths$output$parameter_raw$local) # Save the parameter ranges searched while tuning diff --git a/pipeline/06-upload.R b/pipeline/06-upload.R index 5f7e1761..0ed743cd 100644 --- a/pipeline/06-upload.R +++ b/pipeline/06-upload.R @@ -91,15 +91,32 @@ if (upload_enable) { write_parquet(paths$output$parameter_range$s3) # Clean and unnest the raw parameters data, then write the results to S3 + parameter_raw <- read_parquet(paths$output$parameter_raw$local) + + # Collapse the nested notes to one row per (fold, iteration) before + # joining them onto the metrics. A single key can hold many notes (e.g. + # one warning per candidate model in the initial grid), and joining those + # directly would fan out the metrics rows, double-counting them in any + # downstream aggregation. Identical messages/traces are deduplicated + # before pasting; n_notes preserves the raw count + notes_collapsed <- parameter_raw %>% + select(id, .iter, .notes) %>% + tidyr::unnest(cols = .notes) %>% + group_by(id, .iter) %>% + summarize( + n_notes = n(), + location = paste(unique(location), collapse = "; "), + type = paste(sort(unique(type)), collapse = "; "), + notes = paste(unique(note), collapse = "\n---\n"), + trace = paste(unique(trace[!is.na(trace)]), collapse = "\n=====\n"), + .groups = "drop" + ) + bind_cols( - read_parquet(paths$output$parameter_raw$local) %>% + parameter_raw %>% tidyr::unnest(cols = .metrics) %>% mutate(run_id = !!run_id) %>% - left_join( - rename(., notes = .notes) %>% - tidyr::unnest(cols = notes) %>% - rename(notes = note) - ) %>% + left_join(notes_collapsed, by = c("id", ".iter")) %>% select(-.notes) %>% rename_with(~ gsub("^\\.", "", .x)) %>% tidyr::pivot_wider(names_from = "metric", values_from = "estimate") %>% @@ -110,8 +127,11 @@ if (upload_enable) { "configuration" = "config", "fold_id" = "id" )) ) %>% - relocate(c(location, type, notes), .after = everything()), - read_parquet(paths$output$parameter_raw$local) %>% + relocate( + c(n_notes, location, type, notes, trace), + .after = everything() + ), + parameter_raw %>% tidyr::unnest(cols = .extracts) %>% tidyr::unnest(cols = .extracts) %>% dplyr::select(num_iterations = .extracts) From d2af7616a41eec030a8bb5fefdfd3189830555a2 Mon Sep 17 00:00:00 2001 From: Michael Wagner Date: Tue, 25 Aug 2026 10:14:40 -0500 Subject: [PATCH 07/16] Update with location based note logging --- pipeline/06-upload.R | 18 +++++++++++------- 1 file changed, 11 insertions(+), 7 deletions(-) diff --git a/pipeline/06-upload.R b/pipeline/06-upload.R index 0ed743cd..6b1ead6c 100644 --- a/pipeline/06-upload.R +++ b/pipeline/06-upload.R @@ -93,21 +93,25 @@ if (upload_enable) { # Clean and unnest the raw parameters data, then write the results to S3 parameter_raw <- read_parquet(paths$output$parameter_raw$local) - # Collapse the nested notes to one row per (fold, iteration) before - # joining them onto the metrics. A single key can hold many notes (e.g. - # one warning per candidate model in the initial grid), and joining those - # directly would fan out the metrics rows, double-counting them in any - # downstream aggregation. Identical messages/traces are deduplicated - # before pasting; n_notes preserves the raw count + # One (fold, iteration) key can hold many notes. The initial grid logs + # every candidate model under iteration 0, so one warning per candidate + # becomes dozens of notes on that key. Later iterations can also log more + # than one warning. Joining notes onto the metrics directly would repeat + # each metric row once per note. So first collapse the notes to one row + # per key. Each message keeps its location prefix, so a warning can + # still be traced back to the candidate model that threw it. Duplicate + # messages are dropped, and n_notes records the original count notes_collapsed <- parameter_raw %>% select(id, .iter, .notes) %>% tidyr::unnest(cols = .notes) %>% group_by(id, .iter) %>% summarize( n_notes = n(), + # notes must be built before location is overwritten below, since + # summarize() makes each new column visible to the expressions after it + notes = paste(unique(paste0(location, ": ", note)), collapse = "\n---\n"), location = paste(unique(location), collapse = "; "), type = paste(sort(unique(type)), collapse = "; "), - notes = paste(unique(note), collapse = "\n---\n"), trace = paste(unique(trace[!is.na(trace)]), collapse = "\n=====\n"), .groups = "drop" ) From 7b6bc90279fe72c4ccd839c7d706e07f7b70fc0e Mon Sep 17 00:00:00 2001 From: Michael Wagner Date: Tue, 25 Aug 2026 11:09:01 -0500 Subject: [PATCH 08/16] remove n_notes --- pipeline/06-upload.R | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/pipeline/06-upload.R b/pipeline/06-upload.R index 6b1ead6c..40edf085 100644 --- a/pipeline/06-upload.R +++ b/pipeline/06-upload.R @@ -100,13 +100,12 @@ if (upload_enable) { # each metric row once per note. So first collapse the notes to one row # per key. Each message keeps its location prefix, so a warning can # still be traced back to the candidate model that threw it. Duplicate - # messages are dropped, and n_notes records the original count + # messages are dropped notes_collapsed <- parameter_raw %>% select(id, .iter, .notes) %>% tidyr::unnest(cols = .notes) %>% group_by(id, .iter) %>% summarize( - n_notes = n(), # notes must be built before location is overwritten below, since # summarize() makes each new column visible to the expressions after it notes = paste(unique(paste0(location, ": ", note)), collapse = "\n---\n"), @@ -132,7 +131,7 @@ if (upload_enable) { )) ) %>% relocate( - c(n_notes, location, type, notes, trace), + c(location, type, notes, trace), .after = everything() ), parameter_raw %>% From 79009fe059ff9147e678d9b86ef4a1140470faa9 Mon Sep 17 00:00:00 2001 From: Michael Wagner Date: Tue, 25 Aug 2026 12:09:30 -0500 Subject: [PATCH 09/16] Simplofy --- pipeline/06-upload.R | 16 +++++----------- 1 file changed, 5 insertions(+), 11 deletions(-) diff --git a/pipeline/06-upload.R b/pipeline/06-upload.R index 40edf085..7290c8a5 100644 --- a/pipeline/06-upload.R +++ b/pipeline/06-upload.R @@ -93,22 +93,16 @@ if (upload_enable) { # Clean and unnest the raw parameters data, then write the results to S3 parameter_raw <- read_parquet(paths$output$parameter_raw$local) - # One (fold, iteration) key can hold many notes. The initial grid logs - # every candidate model under iteration 0, so one warning per candidate - # becomes dozens of notes on that key. Later iterations can also log more - # than one warning. Joining notes onto the metrics directly would repeat - # each metric row once per note. So first collapse the notes to one row - # per key. Each message keeps its location prefix, so a warning can - # still be traced back to the candidate model that threw it. Duplicate - # messages are dropped + # A single (fold, iteration) can hold many notes; iteration 0 alone + # logs one per initial candidate model. Collapse them to one row per + # key so the join below can't duplicate metric rows, which would fail + # the bind_cols against the extracts (mismatched row counts) notes_collapsed <- parameter_raw %>% select(id, .iter, .notes) %>% tidyr::unnest(cols = .notes) %>% group_by(id, .iter) %>% summarize( - # notes must be built before location is overwritten below, since - # summarize() makes each new column visible to the expressions after it - notes = paste(unique(paste0(location, ": ", note)), collapse = "\n---\n"), + notes = paste(unique(note), collapse = "\n---\n"), location = paste(unique(location), collapse = "; "), type = paste(sort(unique(type)), collapse = "; "), trace = paste(unique(trace[!is.na(trace)]), collapse = "\n=====\n"), From af9bdab58a9b345961cff2a653f21ffd792dee52 Mon Sep 17 00:00:00 2001 From: Michael Wagner Date: Tue, 25 Aug 2026 12:15:42 -0500 Subject: [PATCH 10/16] Simplify --- pipeline/06-upload.R | 12 +++++------- 1 file changed, 5 insertions(+), 7 deletions(-) diff --git a/pipeline/06-upload.R b/pipeline/06-upload.R index 7290c8a5..25ab5182 100644 --- a/pipeline/06-upload.R +++ b/pipeline/06-upload.R @@ -90,27 +90,25 @@ if (upload_enable) { relocate(run_id) %>% write_parquet(paths$output$parameter_range$s3) - # Clean and unnest the raw parameters data, then write the results to S3 - parameter_raw <- read_parquet(paths$output$parameter_raw$local) - # A single (fold, iteration) can hold many notes; iteration 0 alone # logs one per initial candidate model. Collapse them to one row per # key so the join below can't duplicate metric rows, which would fail # the bind_cols against the extracts (mismatched row counts) - notes_collapsed <- parameter_raw %>% + notes_collapsed <- read_parquet(paths$output$parameter_raw$local) %>% select(id, .iter, .notes) %>% tidyr::unnest(cols = .notes) %>% group_by(id, .iter) %>% summarize( notes = paste(unique(note), collapse = "\n---\n"), location = paste(unique(location), collapse = "; "), - type = paste(sort(unique(type)), collapse = "; "), + type = paste(unique(type), collapse = "; "), trace = paste(unique(trace[!is.na(trace)]), collapse = "\n=====\n"), .groups = "drop" ) + # Clean and unnest the raw parameters data, then write the results to S3 bind_cols( - parameter_raw %>% + read_parquet(paths$output$parameter_raw$local) %>% tidyr::unnest(cols = .metrics) %>% mutate(run_id = !!run_id) %>% left_join(notes_collapsed, by = c("id", ".iter")) %>% @@ -128,7 +126,7 @@ if (upload_enable) { c(location, type, notes, trace), .after = everything() ), - parameter_raw %>% + read_parquet(paths$output$parameter_raw$local) %>% tidyr::unnest(cols = .extracts) %>% tidyr::unnest(cols = .extracts) %>% dplyr::select(num_iterations = .extracts) From 3d89fb3196b9d43b13d1313e50f950bf98a4eb64 Mon Sep 17 00:00:00 2001 From: Michael Wagner Date: Wed, 26 Aug 2026 10:07:09 -0500 Subject: [PATCH 11/16] Add Billy suggestion --- pipeline/01-train.R | 26 +++++++++++++++++++------- pipeline/06-upload.R | 6 +----- 2 files changed, 20 insertions(+), 12 deletions(-) diff --git a/pipeline/01-train.R b/pipeline/01-train.R index d93900f9..495bde86 100644 --- a/pipeline/01-train.R +++ b/pipeline/01-train.R @@ -314,19 +314,31 @@ if (cv_enable) { ) ) + # Print any tuning notes (warnings/errors) and their backtraces so they're + # visible in CloudWatch logs. Deduplicate by message since iteration 0 logs + # one note per initial candidate model + tune::collect_notes(lgbm_search) %>% + dplyr::distinct(location, type, note, .keep_all = TRUE) %>% + purrr::pwalk(function(id, .iter, location, type, note, trace, ...) { + cat(sprintf( + "---- %s | %s | iter %s | %s ----\n", + toupper(type), id, .iter, location + )) + cat(note, "\n") + if (!is.null(trace)) cat(paste(format(trace), collapse = "\n"), "\n") + }) + # Save tuning results to file. This is a data.frame where each row is one # CV iteration lgbm_search %>% lightsnip::axe_tune_data() %>% - # Tune added a trace column (rlang stack objects) to each nested .notes - # tibble which arrow cannot serialize. Flatten each trace to plain text so - # the diagnostic content survives into the parquet (a few KB as text, vs - # gigabytes of captured environments if left as rlang objects) + # Drop the trace column tune attaches to each nested .notes tibble: it + # holds rlang stack objects that arrow cannot serialize. Traces are + # printed above rather than persisted, in part because a fatal error + # could take down the pipeline before this file is written or uploaded mutate(.notes = purrr::map( .notes, - ~ dplyr::mutate(.x, trace = purrr::map_chr(trace, function(tr) { - if (is.null(tr)) NA_character_ else paste(format(tr), collapse = "\n") - })) + ~ dplyr::select(.x, -dplyr::any_of("trace")) )) %>% arrow::write_parquet(paths$output$parameter_raw$local) diff --git a/pipeline/06-upload.R b/pipeline/06-upload.R index 25ab5182..6d4c9197 100644 --- a/pipeline/06-upload.R +++ b/pipeline/06-upload.R @@ -102,7 +102,6 @@ if (upload_enable) { notes = paste(unique(note), collapse = "\n---\n"), location = paste(unique(location), collapse = "; "), type = paste(unique(type), collapse = "; "), - trace = paste(unique(trace[!is.na(trace)]), collapse = "\n=====\n"), .groups = "drop" ) @@ -122,10 +121,7 @@ if (upload_enable) { "configuration" = "config", "fold_id" = "id" )) ) %>% - relocate( - c(location, type, notes, trace), - .after = everything() - ), + relocate(c(location, type, notes), .after = everything()), read_parquet(paths$output$parameter_raw$local) %>% tidyr::unnest(cols = .extracts) %>% tidyr::unnest(cols = .extracts) %>% From 28fbfd90789dde107e530a7c877cc6962e73f5df Mon Sep 17 00:00:00 2001 From: Michael Wagner Date: Wed, 26 Aug 2026 10:22:50 -0500 Subject: [PATCH 12/16] Standardize with condos solution --- pipeline/06-upload.R | 35 +++++++++++++++++++---------------- 1 file changed, 19 insertions(+), 16 deletions(-) diff --git a/pipeline/06-upload.R b/pipeline/06-upload.R index 6d4c9197..fcf3a881 100644 --- a/pipeline/06-upload.R +++ b/pipeline/06-upload.R @@ -90,27 +90,30 @@ if (upload_enable) { relocate(run_id) %>% write_parquet(paths$output$parameter_range$s3) - # A single (fold, iteration) can hold many notes; iteration 0 alone - # logs one per initial candidate model. Collapse them to one row per - # key so the join below can't duplicate metric rows, which would fail - # the bind_cols against the extracts (mismatched row counts) - notes_collapsed <- read_parquet(paths$output$parameter_raw$local) %>% - select(id, .iter, .notes) %>% - tidyr::unnest(cols = .notes) %>% - group_by(id, .iter) %>% - summarize( - notes = paste(unique(note), collapse = "\n---\n"), - location = paste(unique(location), collapse = "; "), - type = paste(unique(type), collapse = "; "), - .groups = "drop" - ) - # Clean and unnest the raw parameters data, then write the results to S3 bind_cols( read_parquet(paths$output$parameter_raw$local) %>% tidyr::unnest(cols = .metrics) %>% mutate(run_id = !!run_id) %>% - left_join(notes_collapsed, by = c("id", ".iter")) %>% + # Each model fit can leave a warning in .notes, but notes don't + # record which config produced them. At .iter = 0 each fold holds + # all initial configs, so joining the notes unnested would attach + # every warning to every metrics row for that fold. Collapse notes + # to one row per fold and iteration first so the join can't + # multiply rows + left_join( + read_parquet(paths$output$parameter_raw$local) %>% + select(id, .iter, .notes) %>% + tidyr::unnest(cols = .notes) %>% + group_by(id, .iter) %>% + summarize( + location = paste(unique(location), collapse = "; "), + type = paste(unique(type), collapse = "; "), + notes = paste(unique(note), collapse = "; "), + .groups = "drop" + ), + by = c("id", ".iter") + ) %>% select(-.notes) %>% rename_with(~ gsub("^\\.", "", .x)) %>% tidyr::pivot_wider(names_from = "metric", values_from = "estimate") %>% From 4f8afc1646eb05bf28559415a7f4a29e90071808 Mon Sep 17 00:00:00 2001 From: Michael Wagner Date: Wed, 26 Aug 2026 10:27:38 -0500 Subject: [PATCH 13/16] Temporarily change params for faster cv testing --- params.yaml | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/params.yaml b/params.yaml index dacea015..1db3bdc9 100644 --- a/params.yaml +++ b/params.yaml @@ -20,7 +20,7 @@ run_note: Potential residential baseline with SHAP values, revert DVC path toggle: # Should the train stage run full cross-validation? Otherwise, the model # will be trained with the default hyperparameters specified below - cv_enable: false + cv_enable: true # Should SHAP values be calculated for this run in the interpret stage? Can be # desirable to save time when testing many models @@ -31,7 +31,7 @@ toggle: # Upload all modeling artifacts and results to S3 in the upload stage. Set # to false if you are not a CCAO employee - upload_enable: true + upload_enable: false # Should the feature report be run, which compares the inputs to the previous # year's model inputs? @@ -96,7 +96,7 @@ input: # by year, township, and class to maintain representation across categories subset: # Whether to use a subset of the training data instead of the full dataset - enable: false + enable: true # Fraction of the training data to keep (between 0 and 1) fraction: 0.25 @@ -119,7 +119,7 @@ cv: # Number of folds to use for cross-validation. For v-fold CV, the data will be # randomly split. For rolling-origin, the data will be split into V chunks by # time, with each chunk/period calculated automatically - num_folds: 10 + num_folds: 2 # The number of months time-based folds should overlap each other. Only # applicable to rolling-origin CV. See https://www.tmwr.org/resampling#rolling @@ -130,7 +130,7 @@ cv: initial_set: 20 # Max number of total search iterations - max_iterations: 50 + max_iterations: 1 # Max number of search iterations without improvement before stopping search no_improve: 15 From 99553eb7c48c03c01744d5388645e3de35a7e898 Mon Sep 17 00:00:00 2001 From: Michael Wagner Date: Wed, 2 Sep 2026 13:43:11 -0500 Subject: [PATCH 14/16] Revert to trace column removal --- pipeline/01-train.R | 18 +----------------- 1 file changed, 1 insertion(+), 17 deletions(-) diff --git a/pipeline/01-train.R b/pipeline/01-train.R index 495bde86..7379c08c 100644 --- a/pipeline/01-train.R +++ b/pipeline/01-train.R @@ -314,28 +314,12 @@ if (cv_enable) { ) ) - # Print any tuning notes (warnings/errors) and their backtraces so they're - # visible in CloudWatch logs. Deduplicate by message since iteration 0 logs - # one note per initial candidate model - tune::collect_notes(lgbm_search) %>% - dplyr::distinct(location, type, note, .keep_all = TRUE) %>% - purrr::pwalk(function(id, .iter, location, type, note, trace, ...) { - cat(sprintf( - "---- %s | %s | iter %s | %s ----\n", - toupper(type), id, .iter, location - )) - cat(note, "\n") - if (!is.null(trace)) cat(paste(format(trace), collapse = "\n"), "\n") - }) - # Save tuning results to file. This is a data.frame where each row is one # CV iteration lgbm_search %>% lightsnip::axe_tune_data() %>% # Drop the trace column tune attaches to each nested .notes tibble: it - # holds rlang stack objects that arrow cannot serialize. Traces are - # printed above rather than persisted, in part because a fatal error - # could take down the pipeline before this file is written or uploaded + # holds rlang stack objects that arrow cannot serialize mutate(.notes = purrr::map( .notes, ~ dplyr::select(.x, -dplyr::any_of("trace")) From 8f04a43d8b3c16d7efc644ee0a18e9857b9de548 Mon Sep 17 00:00:00 2001 From: Michael Wagner Date: Wed, 2 Sep 2026 13:48:34 -0500 Subject: [PATCH 15/16] Revert to main params --- params.yaml | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/params.yaml b/params.yaml index 1db3bdc9..dacea015 100644 --- a/params.yaml +++ b/params.yaml @@ -20,7 +20,7 @@ run_note: Potential residential baseline with SHAP values, revert DVC path toggle: # Should the train stage run full cross-validation? Otherwise, the model # will be trained with the default hyperparameters specified below - cv_enable: true + cv_enable: false # Should SHAP values be calculated for this run in the interpret stage? Can be # desirable to save time when testing many models @@ -31,7 +31,7 @@ toggle: # Upload all modeling artifacts and results to S3 in the upload stage. Set # to false if you are not a CCAO employee - upload_enable: false + upload_enable: true # Should the feature report be run, which compares the inputs to the previous # year's model inputs? @@ -96,7 +96,7 @@ input: # by year, township, and class to maintain representation across categories subset: # Whether to use a subset of the training data instead of the full dataset - enable: true + enable: false # Fraction of the training data to keep (between 0 and 1) fraction: 0.25 @@ -119,7 +119,7 @@ cv: # Number of folds to use for cross-validation. For v-fold CV, the data will be # randomly split. For rolling-origin, the data will be split into V chunks by # time, with each chunk/period calculated automatically - num_folds: 2 + num_folds: 10 # The number of months time-based folds should overlap each other. Only # applicable to rolling-origin CV. See https://www.tmwr.org/resampling#rolling @@ -130,7 +130,7 @@ cv: initial_set: 20 # Max number of total search iterations - max_iterations: 1 + max_iterations: 50 # Max number of search iterations without improvement before stopping search no_improve: 15 From 6fdba7a59e85f54623e2d5fbc11f6f99ae48e369 Mon Sep 17 00:00:00 2001 From: Michael Wagner Date: Thu, 3 Sep 2026 14:20:55 -0500 Subject: [PATCH 16/16] revert to dot syntax --- pipeline/06-upload.R | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/pipeline/06-upload.R b/pipeline/06-upload.R index fcf3a881..ebcc2ff1 100644 --- a/pipeline/06-upload.R +++ b/pipeline/06-upload.R @@ -102,8 +102,7 @@ if (upload_enable) { # to one row per fold and iteration first so the join can't # multiply rows left_join( - read_parquet(paths$output$parameter_raw$local) %>% - select(id, .iter, .notes) %>% + select(., id, .iter, .notes) %>% tidyr::unnest(cols = .notes) %>% group_by(id, .iter) %>% summarize(