From b7ebb4cd8842f2303de688658f8773a5796edf50 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Wed, 29 Apr 2026 17:50:59 -0400 Subject: [PATCH] feat(rollup): drop unavailable-pop rows + provenance audit cog_geographic_rollup(per_capita = TRUE) now drops rows whose government has no observed population for that year (pop_source == 'unavailable'), matching the spec's exclusion rule. Records included/excluded govids in provenance$rollup. --- R/rollup.R | 34 +++++++++++++++++++++++++++------- man/cog_geographic_rollup.Rd | 20 +++++++++++++++----- 2 files changed, 42 insertions(+), 12 deletions(-) diff --git a/R/rollup.R b/R/rollup.R index 7f8e495..cc2adcc 100644 --- a/R/rollup.R +++ b/R/rollup.R @@ -8,28 +8,35 @@ #' "place portraits" that compare a city to the surrounding county and #' containing state on one set of axes. #' +#' When `per_capita = TRUE`, rows whose government has no observed +#' population in that year (`pop_source == "unavailable"`) are dropped from +#' the result. The dropped govids are recorded in +#' `provenance$rollup$excluded_govids`. This excludes special districts +#' (gov type 4) and school districts (gov type 5) from per-capita rollups +#' by design — see `vignette('population-denominators')`. +#' #' @param govids Named list with any non-empty subset of elements named #' `state`, `county`, `city`. Each element is a character vector of #' `canonical_govid` values. At least one layer required. #' @param category Single category name or character vector (passed through #' to [cog_spending()]). #' @param years Integer vector of years. -#' @param per_capita If `TRUE`, per-capita uses each layer's own population -#' from `canonical_fips_xwalk.population_acs`. +#' @param per_capita If `TRUE`, per-capita uses each gov's own per-year +#' population from `gov_population_yearly`. Govs with missing population +#' are excluded from the result. #' @param adjust_to_year Integer base year for CPI-U conversion, or `NULL`. #' @return Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`, #' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real` / -#' `amt_per_capita_nominal` / `amt_per_capita_real`, `codes_included`, -#' `aggregate_fallback`, `scope_note`, `notes`. Carries a `provenance` -#' attribute with `verb = "cog_geographic_rollup"` and `layers`. +#' `amt_per_capita_nominal` / `amt_per_capita_real`, optional `pop_source`, +#' `codes_included`, `aggregate_fallback`, `scope_note`, `notes`. Carries a +#' `provenance` attribute with `verb = "cog_geographic_rollup"`, `layers`, +#' and `rollup$included_govids` / `rollup$excluded_govids`. #' @export cog_geographic_rollup <- function(govids, category, years, per_capita = FALSE, adjust_to_year = NULL) { call <- match.call() .validate_rollup_layers(govids) - # Accept character vector OR a data.frame with canonical_govid per layer, - # so cog_gov_search() output can be piped into one of the layer slots. govids <- lapply(govids, .coerce_govid_input, arg = "govids[[layer]]") if (any(lengths(govids) == 0L)) { cli::cli_abort("Each layer in `govids` must be non-empty after coercion.") @@ -45,12 +52,25 @@ cog_geographic_rollup <- function(govids, category, years, r <- dplyr::left_join(r, layer_map, by = "canonical_govid", relationship = "many-to-many") r$scope_note <- .rollup_scope_note(r$layer) + + excluded <- character(0) + if (isTRUE(per_capita) && "pop_source" %in% names(r)) { + drop <- r$pop_source == "unavailable" + excluded <- unique(r$canonical_govid[drop]) + r <- r[!drop, , drop = FALSE] + } + included <- unique(r$canonical_govid) + r <- .reorder_rollup_cols(r) prov <- attr(r, "provenance") prov$verb <- "cog_geographic_rollup" prov$call <- paste(deparse(call), collapse = " ") prov$layers <- layer_names + prov$rollup <- list( + included_govids = included, + excluded_govids = excluded + ) attr(r, "provenance") <- prov r diff --git a/man/cog_geographic_rollup.Rd b/man/cog_geographic_rollup.Rd index ada1003..ffb2310 100644 --- a/man/cog_geographic_rollup.Rd +++ b/man/cog_geographic_rollup.Rd @@ -22,17 +22,19 @@ to [cog_spending()]).} \item{years}{Integer vector of years.} -\item{per_capita}{If `TRUE`, per-capita uses each layer's own population -from `canonical_fips_xwalk.population_acs`.} +\item{per_capita}{If `TRUE`, per-capita uses each gov's own per-year +population from `gov_population_yearly`. Govs with missing population +are excluded from the result.} \item{adjust_to_year}{Integer base year for CPI-U conversion, or `NULL`.} } \value{ Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`, `spend_subtype`, `category`, `amt_nominal`, optional `amt_real` / - `amt_per_capita_nominal` / `amt_per_capita_real`, `codes_included`, - `aggregate_fallback`, `scope_note`, `notes`. Carries a `provenance` - attribute with `verb = "cog_geographic_rollup"` and `layers`. + `amt_per_capita_nominal` / `amt_per_capita_real`, optional `pop_source`, + `codes_included`, `aggregate_fallback`, `scope_note`, `notes`. Carries a + `provenance` attribute with `verb = "cog_geographic_rollup"`, `layers`, + and `rollup$included_govids` / `rollup$excluded_govids`. } \description{ Wraps [cog_spending()], tags each row with its layer, and attaches a @@ -41,3 +43,11 @@ human-readable `scope_note` documenting geographic-scope caveats (e.g. "place portraits" that compare a city to the surrounding county and containing state on one set of axes. } +\details{ +When `per_capita = TRUE`, rows whose government has no observed +population in that year (`pop_source == "unavailable"`) are dropped from +the result. The dropped govids are recorded in +`provenance$rollup$excluded_govids`. This excludes special districts +(gov type 4) and school districts (gov type 5) from per-capita rollups +by design — see `vignette('population-denominators')`. +}