From 5e22e940e7c23938fffbafdec21c689b3e90be8b Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Wed, 5 Aug 2026 11:21:35 -0400 Subject: [PATCH 1/9] feat: all-categories mode in .build_verb_sql() The concept boundary in this package is subtype, not category, so a total is the existing query with the category dimension collapsed and no category predicate applied. subtype is deliberately kept in the grouping: subtype=operations plus all-categories is 'operating expenditure', which is the measure a fiscal comparison wants. Named 'All Categories' rather than 'Total' because category='Total' would sit one argument from expenditure_concept='total' and mean something different. --- R/spending.R | 38 ++++++++++++++++++++++----- tests/testthat/test-all-categories.R | 39 ++++++++++++++++++++++++++++ 2 files changed, 71 insertions(+), 6 deletions(-) create mode 100644 tests/testthat/test-all-categories.R diff --git a/R/spending.R b/R/spending.R index 29393c8..6aa82ec 100644 --- a/R/spending.R +++ b/R/spending.R @@ -19,6 +19,14 @@ .spend_subtypes_primary <- c("operations", "capital", "assistance") .spend_subtypes_direct <- c(.spend_subtypes_primary, "interest", "insurance_benefits") +# The reserved pseudo-category. Deliberately NOT "Total": `category = "Total"` +# would sit one argument away from `expenditure_concept = "total"` and mean +# something different -- the concept selects WHICH SUBTYPES are in scope, this +# selects whether the rows inside that scope are broken out by category or +# summed. "All Categories" states the operation and cannot be misread as the +# concept. +.ALL_CATEGORIES <- "All Categories" + #' @noRd .expenditure_concept_subtypes <- function(concept) { switch(concept, @@ -570,10 +578,16 @@ cog_spending <- function(govid, years, category = NULL, #' @noRd .build_verb_sql <- function(view, subtype_col, govid, years, category, - ig_view = NULL, subtype_scope = NULL) { + ig_view = NULL, subtype_scope = NULL, + all_categories = FALSE) { govid_lit <- .sql_lit_chr(govid) years_lit <- paste(as.integer(years), collapse = ",") - category_pred <- if (is.null(category)) { + # In all-categories mode there is no category filter: the sum is defined by + # the concept's SUBTYPE allowlist (subtype_pred below), which is the real + # concept boundary. Filtering by category as well would be a no-op at best + # and, if the crosswalk ever gained an uncategorized code, a silent + # under-count of the very total this mode exists to guarantee. + category_pred <- if (all_categories || is.null(category)) { "" } else { sprintf("AND category IN (%s)", .sql_lit_chr(category)) @@ -612,13 +626,24 @@ cog_spending <- function(govid, years, category = NULL, # though its dollars came entirely from an aggregate row, silently # suppressing the "Aggregate fallback applied" note on exactly the rows # this feature exists to surface. + + # Collapse the category dimension. subtype is deliberately KEPT: it is what + # makes `subtype = "operations"` + all-categories mean "operating + # expenditure", the measure a fiscal comparison actually wants. + category_select <- if (all_categories) { + sprintf("%s AS category", .sql_lit_chr(.ALL_CATEGORIES)) + } else { + "category" + } + category_group <- if (all_categories) "" else ", category" + sprintf( "SELECT year, canonical_govid, COALESCE(xwalk_gov_name, gov_name) AS gov_name, %1$s, - category, + %7$s, SUM(amt) * 1000.0 AS amt_nominal, string_agg(DISTINCT item_code, ',' ORDER BY item_code) AS codes_included, bool_or(is_aggregate) AS aggregate_fallback @@ -627,9 +652,10 @@ cog_spending <- function(govid, years, category = NULL, AND year IN (%4$s) %5$s %6$s - GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, %1$s, category - ORDER BY year, canonical_govid, %1$s, category", - subtype_col, source_expr, govid_lit, years_lit, category_pred, subtype_pred + GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, %1$s%8$s + ORDER BY year, canonical_govid, %1$s%8$s", + subtype_col, source_expr, govid_lit, years_lit, category_pred, subtype_pred, + category_select, category_group ) } diff --git a/tests/testthat/test-all-categories.R b/tests/testthat/test-all-categories.R new file mode 100644 index 0000000..d19dd3c --- /dev/null +++ b/tests/testthat/test-all-categories.R @@ -0,0 +1,39 @@ +# Baseline at branch point: 843 PASS / 0 FAIL / 0 SKIP / 0 WARN (2026-08-05, origin/main 2fc9e75) + +test_that(".build_verb_sql emits a literal category and no category filter in all-categories mode", { + sql <- uscogdata:::.build_verb_sql( + view = "spending_annotated", + subtype_col = "spend_subtype", + govid = "552025209777", + years = 2019L, + category = NULL, + subtype_scope = c("operations", "capital"), + all_categories = TRUE + ) + + expect_match(sql, "'All Categories' AS category", fixed = TRUE) + # no category filter of any kind + expect_false(grepl("AND category IN", sql, fixed = TRUE)) + # category is not a grouping key + expect_false(grepl("GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, spend_subtype, category", + sql, fixed = TRUE)) + # the subtype allowlist still applies -- this is what makes the sum a concept + expect_match(sql, "AND spend_subtype IN ('operations','capital')", fixed = TRUE) +}) + +test_that(".build_verb_sql is unchanged when all_categories is FALSE", { + args <- list( + view = "spending_annotated", subtype_col = "spend_subtype", + govid = "552025209777", years = 2019L, category = NULL, + subtype_scope = c("operations", "capital") + ) + old <- do.call(uscogdata:::.build_verb_sql, args) + new <- do.call(uscogdata:::.build_verb_sql, c(args, list(all_categories = FALSE))) + expect_identical(old, new) + expect_match(new, "GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, spend_subtype, category", + fixed = TRUE) +}) + +test_that(".ALL_CATEGORIES is the exact reserved string", { + expect_identical(uscogdata:::.ALL_CATEGORIES, "All Categories") +}) From 11ae99c382e3010d6a3bb4d22af0d059d3af3fad Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Wed, 5 Aug 2026 11:32:47 -0400 Subject: [PATCH 2/9] feat: accept category = 'All Categories' on cog_spending/cog_revenue Returns one summed row per (year, govid, subtype) across every category in the requested concept's subtype scope, so a caller never sums categories client-side and cannot sum the wrong scope. Combining it with other category names is an error rather than a silent partial sum. --- R/revenue.R | 10 +++++ R/spending.R | 27 +++++++++++-- man/cog_revenue.Rd | 10 ++++- man/cog_spending.Rd | 10 ++++- tests/testthat/test-all-categories.R | 59 ++++++++++++++++++++++++++++ 5 files changed, 111 insertions(+), 5 deletions(-) diff --git a/R/revenue.R b/R/revenue.R index 583752c..6e9aa5f 100644 --- a/R/revenue.R +++ b/R/revenue.R @@ -8,6 +8,16 @@ #' multiplies by 1000 and records the conversion in `provenance`). #' #' @inheritParams cog_spending +#' @param category Character vector of category names (from +#' `summary_categories.category`), or `NULL` for all categories broken out +#' one row each. The reserved value `"All Categories"` instead returns a +#' single summed row per `(year, canonical_govid, subtype)`, covering every +#' category inside the requested concept's subtype scope. It cannot be +#' combined with other category names, and it is not the same thing as +#' `revenue_concept = "total"`: the concept chooses which subtypes are in +#' scope, `"All Categories"` chooses whether rows inside that scope are +#' broken out or summed. Combine with `subtype = "operations"` for an +#' own-source revenue total. #' @param revenue_concept Which of Census's two published revenue concepts to #' return. Concepts are defined as sets of the crosswalk's `revenue_subtype` #' values -- never as item-code first letters, which cannot classify diff --git a/R/spending.R b/R/spending.R index 6aa82ec..9086e64 100644 --- a/R/spending.R +++ b/R/spending.R @@ -71,7 +71,15 @@ #' @param govid Character vector of `canonical_govid` values. #' @param years Integer vector of years. #' @param category Character vector of category names (from -#' `summary_categories.category`), or `NULL` for all categories. +#' `summary_categories.category`), or `NULL` for all categories broken out +#' one row each. The reserved value `"All Categories"` instead returns a +#' single summed row per `(year, canonical_govid, subtype)`, covering every +#' category inside the requested concept's subtype scope. It cannot be +#' combined with other category names, and it is not the same thing as +#' `expenditure_concept = "total"`: the concept chooses which subtypes are in +#' scope, `"All Categories"` chooses whether rows inside that scope are +#' broken out or summed. Combine with `subtype = "operations"` for an +#' operating-expenditure total. #' @param per_capita If `TRUE`, adds `amt_per_capita_nominal` (and #' `amt_per_capita_real` when `adjust_to_year` is set) using the per-year #' Census F-33 population from `gov_population_yearly`. Result also gains @@ -268,6 +276,17 @@ cog_spending <- function(govid, years, category = NULL, .validate_verb_inputs(govid, years, category, per_capita, adjust_to_year, recipe) + # Recognize the reserved pseudo-category. Detected after type validation so a + # non-character `category` still fails with the ordinary type error. + all_categories <- !is.null(category) && .ALL_CATEGORIES %in% category + if (all_categories && length(category) > 1L) { + cli::cli_abort(c( + "{.val {(.ALL_CATEGORIES)}} cannot be combined with other categories.", + "i" = "It already sums every category in the requested concept's scope.", + "*" = "Ask for it alone, or list the specific categories you want." + ), class = "uscogdata_all_categories_not_combinable") + } + if (!is.null(recipe) && identical(expenditure_concept, "total")) { cli::cli_abort(c( "`recipe` and `expenditure_concept = \"total\"` are mutually exclusive.", @@ -341,8 +360,10 @@ cog_spending <- function(govid, years, category = NULL, } else { NULL } - sql <- .build_verb_sql(view, subtype_col, govid, years, category, ig_view, - subtype_scope) + sql <- .build_verb_sql(view, subtype_col, govid, years, + if (all_categories) NULL else category, + ig_view, subtype_scope, + all_categories = all_categories) result <- tibble::as_tibble(DBI::dbGetQuery(con, sql)) } diff --git a/man/cog_revenue.Rd b/man/cog_revenue.Rd index 753dede..de677dd 100644 --- a/man/cog_revenue.Rd +++ b/man/cog_revenue.Rd @@ -22,7 +22,15 @@ cog_revenue( \item{years}{Integer vector of years.} \item{category}{Character vector of category names (from -`summary_categories.category`), or `NULL` for all categories.} +`summary_categories.category`), or `NULL` for all categories broken out +one row each. The reserved value `"All Categories"` instead returns a +single summed row per `(year, canonical_govid, subtype)`, covering every +category inside the requested concept's subtype scope. It cannot be +combined with other category names, and it is not the same thing as +`revenue_concept = "total"`: the concept chooses which subtypes are in +scope, `"All Categories"` chooses whether rows inside that scope are +broken out or summed. Combine with `subtype = "operations"` for an +own-source revenue total.} \item{per_capita}{If `TRUE`, adds `amt_per_capita_nominal` (and `amt_per_capita_real` when `adjust_to_year` is set) using the per-year diff --git a/man/cog_spending.Rd b/man/cog_spending.Rd index 7a958b9..e0dedd2 100644 --- a/man/cog_spending.Rd +++ b/man/cog_spending.Rd @@ -22,7 +22,15 @@ cog_spending( \item{years}{Integer vector of years.} \item{category}{Character vector of category names (from -`summary_categories.category`), or `NULL` for all categories.} +`summary_categories.category`), or `NULL` for all categories broken out +one row each. The reserved value `"All Categories"` instead returns a +single summed row per `(year, canonical_govid, subtype)`, covering every +category inside the requested concept's subtype scope. It cannot be +combined with other category names, and it is not the same thing as +`expenditure_concept = "total"`: the concept chooses which subtypes are in +scope, `"All Categories"` chooses whether rows inside that scope are +broken out or summed. Combine with `subtype = "operations"` for an +operating-expenditure total.} \item{per_capita}{If `TRUE`, adds `amt_per_capita_nominal` (and `amt_per_capita_real` when `adjust_to_year` is set) using the per-year diff --git a/tests/testthat/test-all-categories.R b/tests/testthat/test-all-categories.R index d19dd3c..6b9f6ec 100644 --- a/tests/testthat/test-all-categories.R +++ b/tests/testthat/test-all-categories.R @@ -37,3 +37,62 @@ test_that(".build_verb_sql is unchanged when all_categories is FALSE", { test_that(".ALL_CATEGORIES is the exact reserved string", { expect_identical(uscogdata:::.ALL_CATEGORIES, "All Categories") }) + +test_that('cog_spending(category = "All Categories") sums to the per-category total', { + gov <- "552025209777" + by_cat <- cog_spending(gov, 2019L) + total <- cog_spending(gov, 2019L, category = "All Categories") + + expect_true(nrow(total) > 0L) + expect_setequal(unique(total$category), "All Categories") + # one row per subtype present in the by-category result + expect_setequal(unique(total$spend_subtype), unique(by_cat$spend_subtype)) + expect_equal(nrow(total), length(unique(by_cat$spend_subtype))) + + # the dollars agree, per subtype + lhs <- tapply(by_cat$amt_nominal, by_cat$spend_subtype, sum) + rhs <- tapply(total$amt_nominal, total$spend_subtype, sum) + expect_equal(as.numeric(rhs[names(lhs)]), as.numeric(lhs), tolerance = 1e-8) +}) + +test_that('"All Categories" respects expenditure_concept', { + gov <- "552025209777" + prim <- cog_spending(gov, 2019L, category = "All Categories", + expenditure_concept = "primary") + dir <- cog_spending(gov, 2019L, category = "All Categories", + expenditure_concept = "direct") + # direct = primary plus interest and insurance benefits, so it is never smaller + expect_gte(sum(dir$amt_nominal), sum(prim$amt_nominal)) +}) + +test_that('"All Categories" works on revenue and respects revenue_concept', { + gov <- "552025209777" + gen <- cog_revenue(gov, 2019L, category = "All Categories", + revenue_concept = "general") + tot <- cog_revenue(gov, 2019L, category = "All Categories", + revenue_concept = "total") + expect_setequal(unique(gen$category), "All Categories") + expect_gte(sum(tot$amt_nominal), sum(gen$amt_nominal)) +}) + +test_that('"All Categories" cannot be combined with another category', { + expect_error( + cog_spending("552025209777", 2019L, category = c("All Categories", "Police")), + class = "uscogdata_all_categories_not_combinable" + ) +}) + +test_that('"All Categories" is recorded in provenance', { + r <- cog_spending("552025209777", 2019L, category = "All Categories") + expect_identical(cog_explain(r, format = "list")$category, "All Categories") +}) + +test_that('"All Categories" combines with subtype to give operating totals', { + gov <- "552025209777" + ops_by_cat <- cog_spending(gov, 2019L) + ops_by_cat <- ops_by_cat[ops_by_cat$spend_subtype == "operations", ] + ops_total <- cog_spending(gov, 2019L, category = "All Categories") + ops_total <- ops_total[ops_total$spend_subtype == "operations", ] + expect_equal(sum(ops_total$amt_nominal), sum(ops_by_cat$amt_nominal), + tolerance = 1e-8) +}) From 503fa6562f931457790d9801308aa68728c1d72b Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Wed, 5 Aug 2026 11:35:47 -0400 Subject: [PATCH 3/9] docs: fix false subtype= claim in All Categories roxygen (F1) The @param category text on cog_spending()/cog_revenue() told users to "Combine with subtype = ..." but neither verb has a subtype argument. Replace with accurate guidance: filter the returned frame's spend_subtype/revenue_subtype column. --- R/revenue.R | 5 +++-- R/spending.R | 5 +++-- man/cog_revenue.Rd | 5 +++-- man/cog_spending.Rd | 5 +++-- 4 files changed, 12 insertions(+), 8 deletions(-) diff --git a/R/revenue.R b/R/revenue.R index 6e9aa5f..c5d99b6 100644 --- a/R/revenue.R +++ b/R/revenue.R @@ -16,8 +16,9 @@ #' combined with other category names, and it is not the same thing as #' `revenue_concept = "total"`: the concept chooses which subtypes are in #' scope, `"All Categories"` chooses whether rows inside that scope are -#' broken out or summed. Combine with `subtype = "operations"` for an -#' own-source revenue total. +#' broken out or summed. Because the result keeps one row per +#' `revenue_subtype`, filtering the returned frame to +#' `revenue_subtype == "own_source"` gives an own-source revenue total. #' @param revenue_concept Which of Census's two published revenue concepts to #' return. Concepts are defined as sets of the crosswalk's `revenue_subtype` #' values -- never as item-code first letters, which cannot classify diff --git a/R/spending.R b/R/spending.R index 9086e64..c30fa26 100644 --- a/R/spending.R +++ b/R/spending.R @@ -78,8 +78,9 @@ #' combined with other category names, and it is not the same thing as #' `expenditure_concept = "total"`: the concept chooses which subtypes are in #' scope, `"All Categories"` chooses whether rows inside that scope are -#' broken out or summed. Combine with `subtype = "operations"` for an -#' operating-expenditure total. +#' broken out or summed. Because the result keeps one row per +#' `spend_subtype`, filtering the returned frame to +#' `spend_subtype == "operations"` gives an operating-expenditure total. #' @param per_capita If `TRUE`, adds `amt_per_capita_nominal` (and #' `amt_per_capita_real` when `adjust_to_year` is set) using the per-year #' Census F-33 population from `gov_population_yearly`. Result also gains diff --git a/man/cog_revenue.Rd b/man/cog_revenue.Rd index de677dd..ab978a1 100644 --- a/man/cog_revenue.Rd +++ b/man/cog_revenue.Rd @@ -29,8 +29,9 @@ category inside the requested concept's subtype scope. It cannot be combined with other category names, and it is not the same thing as `revenue_concept = "total"`: the concept chooses which subtypes are in scope, `"All Categories"` chooses whether rows inside that scope are -broken out or summed. Combine with `subtype = "operations"` for an -own-source revenue total.} +broken out or summed. Because the result keeps one row per +`revenue_subtype`, filtering the returned frame to +`revenue_subtype == "own_source"` gives an own-source revenue total.} \item{per_capita}{If `TRUE`, adds `amt_per_capita_nominal` (and `amt_per_capita_real` when `adjust_to_year` is set) using the per-year diff --git a/man/cog_spending.Rd b/man/cog_spending.Rd index e0dedd2..1fee7d1 100644 --- a/man/cog_spending.Rd +++ b/man/cog_spending.Rd @@ -29,8 +29,9 @@ category inside the requested concept's subtype scope. It cannot be combined with other category names, and it is not the same thing as `expenditure_concept = "total"`: the concept chooses which subtypes are in scope, `"All Categories"` chooses whether rows inside that scope are -broken out or summed. Combine with `subtype = "operations"` for an -operating-expenditure total.} +broken out or summed. Because the result keeps one row per +`spend_subtype`, filtering the returned frame to +`spend_subtype == "operations"` gives an operating-expenditure total.} \item{per_capita}{If `TRUE`, adds `amt_per_capita_nominal` (and `amt_per_capita_real` when `adjust_to_year` is set) using the per-year From 12a9be110ffd34564a82b889d81c49c8766b4a39 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Wed, 5 Aug 2026 11:43:02 -0400 Subject: [PATCH 4/9] feat: advertise 'All Categories' from cog_categories() A reserved value nobody can discover is a trap, and this is the view the API's /categories endpoint is built from. Emitted for the two flow vocabularies only -- cog_balances() returns a stock and has no concept to sum within. Also fix test-categories.R to exclude pseudo-category rows from crosswalk-specific assertions (one row per (category, subtype) pair, non-empty item_codes, valid subtypes). Co-Authored-By: Claude Opus 5 (1M context) --- R/categories.R | 30 ++++++++++++++++++++++++++-- man/cog_categories.Rd | 5 ++++- tests/testthat/test-all-categories.R | 27 +++++++++++++++++++++++++ tests/testthat/test-categories.R | 12 +++++++++-- 4 files changed, 69 insertions(+), 5 deletions(-) diff --git a/R/categories.R b/R/categories.R index 593d9c1..5450396 100644 --- a/R/categories.R +++ b/R/categories.R @@ -22,7 +22,10 @@ #' `category` column (e.g. `"Police"` or `"Tax"`). #' @return Tibble with columns `category`, `category_type`, `subtype`, #' `n_codes`, `item_codes` (comma-separated, alphabetical). Sorted by -#' `category_type`, `category`, `subtype`. +#' `category_type`, `category`, `subtype`. Includes one row per flow for the +#' reserved pseudo-category `"All Categories"`, which carries `NA` for +#' `subtype`, `n_codes` and `item_codes` because it is a query mode rather +#' than a crosswalk entry — see [cog_spending()]'s `category` argument. #' @export cog_categories <- function(type = NULL, pattern = NULL) { if (!is.null(type)) { @@ -63,5 +66,28 @@ cog_categories <- function(type = NULL, pattern = NULL) { "GROUP BY category, category_type, subtype ORDER BY category_type, category, subtype" ) - tibble::as_tibble(DBI::dbGetQuery(con, sql)) + out <- tibble::as_tibble(DBI::dbGetQuery(con, sql)) + + # The reserved pseudo-category is a query mode, not a crosswalk row, so it + # has no item codes to report -- hence NA rather than 0 for n_codes. It is + # emitted for the two FLOW vocabularies only: cog_balances() returns a stock + # and has no concept argument to sum within. + pseudo <- tibble::tibble( + category = .ALL_CATEGORIES, + category_type = c("expenditure", "revenue"), + subtype = NA_character_, + n_codes = NA_integer_, + item_codes = NA_character_ + ) + if (!is.null(type)) { + db_type <- if (type == "spending") "expenditure" else type + pseudo <- pseudo[pseudo$category_type == db_type, , drop = FALSE] + } + if (!is.null(pattern) && nrow(pseudo) > 0L) { + keep <- grepl(pattern, pseudo$category, ignore.case = TRUE) + pseudo <- pseudo[keep, , drop = FALSE] + } + if (nrow(pseudo) == 0L) return(out) + out <- rbind(out, pseudo) + out[order(out$category_type, out$category, out$subtype), , drop = FALSE] } diff --git a/man/cog_categories.Rd b/man/cog_categories.Rd index fa33cfb..8e72212 100644 --- a/man/cog_categories.Rd +++ b/man/cog_categories.Rd @@ -16,7 +16,10 @@ balance), `"spending"`, `"revenue"`, or `"balance"`.} \value{ Tibble with columns `category`, `category_type`, `subtype`, `n_codes`, `item_codes` (comma-separated, alphabetical). Sorted by - `category_type`, `category`, `subtype`. + `category_type`, `category`, `subtype`. Includes one row per flow for the + reserved pseudo-category `"All Categories"`, which carries `NA` for + `subtype`, `n_codes` and `item_codes` because it is a query mode rather + than a crosswalk entry — see [cog_spending()]'s `category` argument. } \description{ Returns the category taxonomy exposed by the corpus's diff --git a/tests/testthat/test-all-categories.R b/tests/testthat/test-all-categories.R index 6b9f6ec..c181fad 100644 --- a/tests/testthat/test-all-categories.R +++ b/tests/testthat/test-all-categories.R @@ -96,3 +96,30 @@ test_that('"All Categories" combines with subtype to give operating totals', { expect_equal(sum(ops_total$amt_nominal), sum(ops_by_cat$amt_nominal), tolerance = 1e-8) }) + +test_that('cog_categories() advertises "All Categories" for both flows', { + all <- cog_categories() + rows <- all[all$category == "All Categories", ] + expect_setequal(rows$category_type, c("expenditure", "revenue")) + expect_true(all(is.na(rows$subtype))) + expect_true(all(is.na(rows$n_codes))) +}) + +test_that('cog_categories(type=) still scopes, including the pseudo-category', { + sp <- cog_categories(type = "spending") + expect_setequal(unique(sp$category_type), "expenditure") + expect_true("All Categories" %in% sp$category) + + rev <- cog_categories(type = "revenue") + expect_setequal(unique(rev$category_type), "revenue") + expect_true("All Categories" %in% rev$category) + + # balances have no concept vocabulary, so no pseudo-category + bal <- cog_categories(type = "balance") + expect_false("All Categories" %in% bal$category) +}) + +test_that('cog_categories(pattern=) matches the pseudo-category', { + hit <- cog_categories(pattern = "^All Categories$") + expect_equal(nrow(hit), 2L) +}) diff --git a/tests/testthat/test-categories.R b/tests/testthat/test-categories.R index 11db517..5becf89 100644 --- a/tests/testthat/test-categories.R +++ b/tests/testthat/test-categories.R @@ -29,7 +29,9 @@ test_that("cog_categories(type = 'spending') returns only expenditure rows", { # joined with the I/Q/Y flow batch -- the last two characters of Census's # expenditure taxonomy. `interest` is what makes the three-concept model # computable: primary = direct minus debt service. - expect_true(all(r$subtype %in% + # Exclude pseudo-category which has NA for subtype + r_crosswalk <- r[r$category != "All Categories", ] + expect_true(all(r_crosswalk$subtype %in% c("operations", "capital", "intergovernmental", "assistance", "interest", "insurance_benefits"))) }) @@ -54,7 +56,9 @@ test_that("cog_categories(type = 'revenue') returns only revenue rows", { # plus the employee-retirement X codes), utility (A91-A94) and liquor store # (A90) revenue by definition, which is what makes both of its published # revenue concepts computable -- see `revenue_concept` in `?cog_revenue`. - expect_true(all(r$subtype %in% + # Exclude pseudo-category which has NA for subtype + r_crosswalk <- r[r$category != "All Categories", ] + expect_true(all(r_crosswalk$subtype %in% c("own_source", "federal", "state", "local_aid", "insurance_trust", "utility", "liquor_store"))) }) @@ -69,6 +73,8 @@ test_that("cog_categories(pattern = ...) filters case-insensitively", { test_that("cog_categories has one row per (category, subtype)", { skip_if_no_corpus() r <- cog_categories() + # Exclude pseudo-category which is not a crosswalk entry + r <- r[r$category != "All Categories", ] key <- paste(r$category, r$subtype, sep = "|") expect_equal(length(key), length(unique(key))) }) @@ -76,6 +82,8 @@ test_that("cog_categories has one row per (category, subtype)", { test_that("cog_categories item_codes is non-empty comma-separated string", { skip_if_no_corpus() r <- cog_categories() + # Exclude pseudo-category which has NA for n_codes and item_codes + r <- r[r$category != "All Categories", ] expect_true(all(nzchar(r$item_codes))) expect_true(all(r$n_codes >= 1L)) # n_codes should equal count of commas + 1 From f1e9aa383a0daa42ff2711e8aca93bf2d8d6ef1e Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Wed, 5 Aug 2026 11:53:27 -0400 Subject: [PATCH 5/9] test: prove 'All Categories' passes through the geographic rollup Geographic totals are the expensive case cog-api#37 was filed about -- without this a caller issues one rollup per category and sums them. The pass-through was expected to work by construction; this asserts it rather than assuming it, including under per_capita and inflation adjustment. --- R/rollup.R | 6 +++- man/cog_geographic_rollup.Rd | 6 +++- tests/testthat/test-all-categories-rollup.R | 40 +++++++++++++++++++++ 3 files changed, 50 insertions(+), 2 deletions(-) create mode 100644 tests/testthat/test-all-categories-rollup.R diff --git a/R/rollup.R b/R/rollup.R index 3b7d5c1..834fc8b 100644 --- a/R/rollup.R +++ b/R/rollup.R @@ -19,7 +19,11 @@ #' `state`, `county`, `city`. Each element is a character vector of #' `canonical_govid` values. At least one layer required. #' @param category Single category name or character vector (passed through -#' to [cog_spending()]). +#' to [cog_spending()]), or the reserved `"All Categories"` for one summed +#' row per `(year, canonical_govid, subtype)` covering every category in the +#' concept's scope. `"All Categories"` is the efficient way to build a +#' geographic total: without it a caller must issue one rollup per category +#' and sum the results themselves. #' @param years Integer vector of years. #' @param per_capita If `TRUE`, per-capita uses each gov's own per-year #' population from `gov_population_yearly`. Govs with missing population diff --git a/man/cog_geographic_rollup.Rd b/man/cog_geographic_rollup.Rd index 497fde8..d157b5b 100644 --- a/man/cog_geographic_rollup.Rd +++ b/man/cog_geographic_rollup.Rd @@ -20,7 +20,11 @@ cog_geographic_rollup( `canonical_govid` values. At least one layer required.} \item{category}{Single category name or character vector (passed through -to [cog_spending()]).} +to [cog_spending()]), or the reserved `"All Categories"` for one summed +row per `(year, canonical_govid, subtype)` covering every category in the +concept's scope. `"All Categories"` is the efficient way to build a +geographic total: without it a caller must issue one rollup per category +and sum the results themselves.} \item{years}{Integer vector of years.} diff --git a/tests/testthat/test-all-categories-rollup.R b/tests/testthat/test-all-categories-rollup.R new file mode 100644 index 0000000..5e43f6e --- /dev/null +++ b/tests/testthat/test-all-categories-rollup.R @@ -0,0 +1,40 @@ +test_that('cog_geographic_rollup() accepts "All Categories" and agrees with per-category sums', { + skip_if_no_corpus() + govs <- cog_gov_search(name = NULL, state = "WI", type = 2L) + expect_gt(nrow(govs), 1L) + ids <- list(city = utils::head(govs$canonical_govid, 25L)) + + by_cat <- cog_geographic_rollup(ids, category = NULL, years = 2019L) + total <- cog_geographic_rollup(ids, category = "All Categories", years = 2019L) + + expect_setequal(unique(total$category), "All Categories") + # one row per (govid, subtype) that appears in the per-category result + key_by_cat <- unique(paste(by_cat$canonical_govid, by_cat$spend_subtype)) + key_total <- paste(total$canonical_govid, total$spend_subtype) + expect_setequal(key_total, key_by_cat) + + lhs <- tapply(by_cat$amt_nominal, paste(by_cat$canonical_govid, by_cat$spend_subtype), sum) + rhs <- tapply(total$amt_nominal, key_total, sum) + expect_equal(as.numeric(rhs[names(lhs)]), as.numeric(lhs), tolerance = 1e-8) +}) + +test_that('"All Categories" survives per_capita and inflation adjustment through the rollup', { + skip_if_no_corpus() + govs <- cog_gov_search(name = NULL, state = "WI", type = 2L) + ids <- list(city = utils::head(govs$canonical_govid, 10L)) + r <- cog_geographic_rollup(ids, category = "All Categories", years = 2019L, + per_capita = TRUE, adjust_to_year = 2020L) + expect_true(all(c("amt_per_capita_nominal", "amt_real", "amt_per_capita_real") %in% names(r))) + expect_setequal(unique(r$category), "All Categories") + expect_true(all(is.finite(r$amt_real))) +}) + +test_that('cog_geographic_rollup() still refuses expenditure_concept = "total" with "All Categories"', { + skip_if_no_corpus() + govs <- cog_gov_search(name = NULL, state = "WI", type = 2L) + ids <- list(city = utils::head(govs$canonical_govid, 5L)) + expect_error( + cog_geographic_rollup(ids, category = "All Categories", years = 2019L, + expenditure_concept = "total") + ) +}) From 44e9b40b86411e504f533640598b26967dca05db Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Wed, 5 Aug 2026 11:58:25 -0400 Subject: [PATCH 6/9] docs: n_units_reporting is category-conditional, not a response rate Closes uscogdata#36. It counts governments with rows for the requested category, so a surveyed government that genuinely spends nothing there is indistinguishable from one never surveyed. In FY2022, a complete census year, Georgia reports 393 of 567 cities for Police -- the gap is cities that contract to the sheriff. Documents the comparison that IS valid: same category, census year vs sample year. --- R/peers.R | 17 +++++++++++++++++ R/rollup.R | 17 +++++++++++++++++ man/cog_geographic_rollup.Rd | 20 ++++++++++++++++++++ man/cog_peer_compare.Rd | 20 ++++++++++++++++++++ tests/testthat/test-all-categories-rollup.R | 18 ++++++++++++++++++ 5 files changed, 92 insertions(+) diff --git a/R/peers.R b/R/peers.R index 292453e..1919e35 100644 --- a/R/peers.R +++ b/R/peers.R @@ -240,6 +240,23 @@ cog_find_peers <- function(target_govid, #' group_by(year) |> #' summarise(p50 = quantile(total, 0.5, na.rm = TRUE)) #' ``` +#' @section Reading `coverage`: +#' `provenance$coverage` reports `n_units_reporting` against +#' `n_units_expected` per year. **`n_units_reporting` is category-conditional: +#' it counts cohort members with rows for the category you asked for, not +#' cohort members collected that year.** A government that was surveyed and +#' genuinely spends nothing in that category is indistinguishable here from one +#' that was never surveyed. +#' +#' The ratio is therefore **not a response rate** and must not be used as one. +#' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for +#' `category = "Police"`; the 174-city gap is overwhelmingly cities that +#' contract policing to the county sheriff, not non-response. +#' +#' The comparison that *is* valid is the same category across a census year +#' (ending in 2 or 7) and a sample year, where the real-zero component is +#' roughly constant and the difference reflects the survey cycle. `is_census_year` +#' marks which is which. #' @export cog_peer_compare <- function(target_govid, peers, category, years, per_capita = TRUE, adjust_to_year = NULL, diff --git a/R/rollup.R b/R/rollup.R index 834fc8b..8edac2d 100644 --- a/R/rollup.R +++ b/R/rollup.R @@ -60,6 +60,23 @@ #' `codes_included`, `aggregate_fallback`, `scope_note`, `notes`. Carries a #' `provenance` attribute with `verb = "cog_geographic_rollup"`, `layers`, #' and `rollup$included_govids` / `rollup$excluded_govids`. +#' @section Reading `coverage`: +#' `provenance$coverage` reports `n_units_reporting` against +#' `n_units_expected` per year. **`n_units_reporting` is category-conditional: +#' it counts governments with rows for the category you asked for, not +#' governments collected that year.** A government that was surveyed and +#' genuinely spends nothing in that category is indistinguishable here from one +#' that was never surveyed. +#' +#' The ratio is therefore **not a response rate** and must not be used as one. +#' In FY2022 — a complete census year — Georgia reports 393 of 567 cities for +#' `category = "Police"`; the 174-city gap is overwhelmingly cities that +#' contract policing to the county sheriff, not non-response. +#' +#' The comparison that *is* valid is the same category across a census year +#' (ending in 2 or 7) and a sample year, where the real-zero component is +#' roughly constant and the difference reflects the survey cycle. `is_census_year` +#' marks which is which. #' @export cog_geographic_rollup <- function(govids, category, years, per_capita = FALSE, adjust_to_year = NULL, diff --git a/man/cog_geographic_rollup.Rd b/man/cog_geographic_rollup.Rd index d157b5b..15245a6 100644 --- a/man/cog_geographic_rollup.Rd +++ b/man/cog_geographic_rollup.Rd @@ -84,3 +84,23 @@ the result. The dropped govids are recorded in (gov type 4) and school districts (gov type 5) from per-capita rollups by design — see `vignette('population-denominators')`. } +\section{Reading `coverage`}{ + +`provenance$coverage` reports `n_units_reporting` against +`n_units_expected` per year. **`n_units_reporting` is category-conditional: +it counts governments with rows for the category you asked for, not +governments collected that year.** A government that was surveyed and +genuinely spends nothing in that category is indistinguishable here from one +that was never surveyed. + +The ratio is therefore **not a response rate** and must not be used as one. +In FY2022 — a complete census year — Georgia reports 393 of 567 cities for +`category = "Police"`; the 174-city gap is overwhelmingly cities that +contract policing to the county sheriff, not non-response. + +The comparison that *is* valid is the same category across a census year +(ending in 2 or 7) and a sample year, where the real-zero component is +roughly constant and the difference reflects the survey cycle. `is_census_year` +marks which is which. +} + diff --git a/man/cog_peer_compare.Rd b/man/cog_peer_compare.Rd index bd8e21a..c918195 100644 --- a/man/cog_peer_compare.Rd +++ b/man/cog_peer_compare.Rd @@ -107,3 +107,23 @@ call. Those summary rows are quantiles **within each category**, not quantiles of each peer's total — see the `@return` section before summing them. } +\section{Reading `coverage`}{ + +`provenance$coverage` reports `n_units_reporting` against +`n_units_expected` per year. **`n_units_reporting` is category-conditional: +it counts cohort members with rows for the category you asked for, not +cohort members collected that year.** A government that was surveyed and +genuinely spends nothing in that category is indistinguishable here from one +that was never surveyed. + +The ratio is therefore **not a response rate** and must not be used as one. +In FY2022 — a complete census year — Georgia reports 393 of 567 cities for +`category = "Police"`; the 174-city gap is overwhelmingly cities that +contract policing to the county sheriff, not non-response. + +The comparison that *is* valid is the same category across a census year +(ending in 2 or 7) and a sample year, where the real-zero component is +roughly constant and the difference reflects the survey cycle. `is_census_year` +marks which is which. +} + diff --git a/tests/testthat/test-all-categories-rollup.R b/tests/testthat/test-all-categories-rollup.R index 5e43f6e..5b979fe 100644 --- a/tests/testthat/test-all-categories-rollup.R +++ b/tests/testthat/test-all-categories-rollup.R @@ -38,3 +38,21 @@ test_that('cog_geographic_rollup() still refuses expenditure_concept = "total" w expenditure_concept = "total") ) }) + +test_that("n_units_reporting is category-conditional, not a response rate", { + skip_if_no_corpus() + govs <- cog_gov_search(name = NULL, state = "WI", type = 2L) + ids <- list(city = govs$canonical_govid) + + police <- cog_geographic_rollup(ids, category = "Police", years = 2012L) + allcat <- cog_geographic_rollup(ids, category = "All Categories", years = 2012L) + + cov_police <- cog_explain(police, format = "list")$coverage + cov_all <- cog_explain(allcat, format = "list")$coverage + + # Same year, same requested govids, same collection -- yet a single category + # reports fewer units than the all-categories query. That gap is real zeros, + # not non-response, which is exactly why the ratio is not a response rate. + expect_lte(cov_police$n_units_reporting, cov_all$n_units_reporting) + expect_identical(cov_police$n_units_expected, cov_all$n_units_expected) +}) From 61b9c95731c8e7719b9ecff0a6b6f2cef4f56918 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Wed, 5 Aug 2026 12:04:24 -0400 Subject: [PATCH 7/9] chore: release 0.2.0 Bumps the minor version because 'All Categories' adds public surface without breaking any existing call. The bump is load-bearing, not cosmetic: cog-api installs this package with install_local(), which no-ops when the version already matches. Without it, Phase 1 would silently test against the 0.1.0 reader and pass while proving nothing. --- DESCRIPTION | 2 +- NEWS.md | 28 ++++++++++++++++++++++++++++ 2 files changed, 29 insertions(+), 1 deletion(-) diff --git a/DESCRIPTION b/DESCRIPTION index 7e248c8..8ccdcc3 100644 --- a/DESCRIPTION +++ b/DESCRIPTION @@ -1,7 +1,7 @@ Package: uscogdata Type: Package Title: Curated Reader for the Civilytics US Census of Governments Finance Corpus -Version: 0.1.0 +Version: 0.2.0 Authors@R: person("Civilytics", , , "jknowles@gmail.com", role = c("aut", "cre")) Description: Curated R verbs over the Civilytics US Census of Governments diff --git a/NEWS.md b/NEWS.md index 6e3b47d..31492fc 100644 --- a/NEWS.md +++ b/NEWS.md @@ -1,3 +1,31 @@ +# uscogdata 0.2.0 + +## New features + +* `cog_spending()` and `cog_revenue()` accept the reserved category + `"All Categories"`, returning one summed row per + `(year, canonical_govid, subtype)` across every category inside the + requested concept's subtype scope. Combine with `subtype = "operations"` + for an operating-expenditure total. `cog_geographic_rollup()` inherits it, + which is the efficient way to build a geographic total — previously a + caller had to issue one rollup per category and sum the results + (cog-api#37). + + `"All Categories"` is not the same thing as `expenditure_concept = "total"`. + The concept chooses which subtypes are in scope; `"All Categories"` chooses + whether the rows inside that scope are broken out or summed. + +* `cog_categories()` advertises `"All Categories"` for the expenditure and + revenue vocabularies, so the reserved value is discoverable. + +## Documentation + +* `cog_geographic_rollup()` and `cog_peer_compare()` now document that + `provenance$coverage`'s `n_units_reporting` is **category-conditional** and + is not a response rate: a government that was surveyed and genuinely spends + nothing in the requested category is indistinguishable from one never + surveyed (uscogdata#36). + # uscogdata 0.1.0 (development) ## Signposting now catches partially-suppressed categories From a5f86d87b3d46929430c462c6d9f7c73d438b4f6 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Wed, 5 Aug 2026 12:30:23 -0400 Subject: [PATCH 8/9] fix: close five final-review gaps in all-categories mode - .detect_direct_suppressed() keys on (year, canonical_govid, category); all-categories mode collapses category to one literal value, so the key collides and the detector silently reports FALSE instead of "unknown". Report NA there instead, and stop isTRUE() in .build_provenance() from collapsing that NA back to FALSE. Schema widened to allow null. - Refuse complete = TRUE + category = "All Categories": the completion grid has no per-category cells left to fill once categories are collapsed, so the prior silent 0-rows-filled result was never actually checked. - cog_balances(category = "All Categories") returned zero rows with no error. .validate_verb_inputs() gains allow_all_categories (default FALSE); .verb_spendrev() passes TRUE, cog_balances() does not, so the three verbs share one place to reject it instead of drifting again. - Fix the false `subtype = "operations"` argument claim (no such argument exists) in NEWS.md and an internal spending.R comment. Adds three covering tests to test-all-categories.R for the three behaviour changes above. --- NEWS.md | 5 +- R/balances.R | 14 ++++- R/provenance.R | 10 +++- R/spending.R | 85 +++++++++++++++++++++++++--- inst/schemas/provenance-v1.json | 4 +- man/cog_balances.Rd | 7 ++- man/cog_spending.Rd | 7 ++- tests/testthat/test-all-categories.R | 66 +++++++++++++++++++++ 8 files changed, 182 insertions(+), 16 deletions(-) diff --git a/NEWS.md b/NEWS.md index 31492fc..cd93936 100644 --- a/NEWS.md +++ b/NEWS.md @@ -5,8 +5,9 @@ * `cog_spending()` and `cog_revenue()` accept the reserved category `"All Categories"`, returning one summed row per `(year, canonical_govid, subtype)` across every category inside the - requested concept's subtype scope. Combine with `subtype = "operations"` - for an operating-expenditure total. `cog_geographic_rollup()` inherits it, + requested concept's subtype scope. Filtering the result to + `spend_subtype == "operations"` gives an operating-expenditure total. + `cog_geographic_rollup()` inherits it, which is the efficient way to build a geographic total — previously a caller had to issue one rollup per category and sum the results (cog-api#37). diff --git a/R/balances.R b/R/balances.R index 4a9567d..994e0b2 100644 --- a/R/balances.R +++ b/R/balances.R @@ -28,7 +28,12 @@ #' every combination would be either redundant or empty. #' `category = "Fund Balances"` is exactly the `general` family #' (`W01`/`W31`/`W61`). `balance_subtype` is returned, so a finer split is -#' one `dplyr::filter()` away. +#' one `dplyr::filter()` away. The reserved pseudo-category +#' `"All Categories"` (see [cog_spending()]) is **not** supported here and +#' errors with class `uscogdata_all_categories_unsupported`: it sums a +#' concept's subtype scope, and holdings are a stock with no concept +#' vocabulary to sum across. Omit `category` to get every category broken +#' out instead. #' @param per_capita Divide holdings by population. Note this is a **stock per #' resident** (reserves per person), which is *not* comparable to #' [cog_spending()]'s per-capita figures -- those are a flow per person. @@ -74,6 +79,13 @@ cog_balances <- function(govid, years, category = NULL, # helper reuse as .build_verb_sql()/.attach_per_capita() below; it does NOT # route the verb through .verb_spendrev(), which stays deliberately unused # here because its flow vocabulary is meaningless for a stock. + # + # allow_all_categories is left at its FALSE default (contrast + # .verb_spendrev(), which passes TRUE): the all-categories mode's "sum" + # only means something in terms of a concept's subtype scope, and holdings + # have no concept vocabulary. The reuse above is exactly why this can be a + # one-line default rather than a second bespoke check -- see the + # validator's own doc comment for the incident that made that matter. .validate_verb_inputs(govid, years, category, per_capita, adjust_to_year, recipe) years <- as.integer(years) diff --git a/R/provenance.R b/R/provenance.R index 77cf2d1..55126fc 100644 --- a/R/provenance.R +++ b/R/provenance.R @@ -67,7 +67,15 @@ basis_note = basis_note, expenditure_concept = expenditure_concept, expenditure_concept_note = expenditure_concept_note, - expenditure_concept_direct_suppressed = isTRUE(expenditure_concept_direct_suppressed), + # isTRUE() alone would collapse a deliberate NA (all-categories mode, + # where suppression detection cannot run -- see .verb_spendrev()) down to + # FALSE, turning "we don't know" back into the false claim this field + # exists to avoid. Preserve NA; otherwise normalize to a strict logical. + expenditure_concept_direct_suppressed = if (isTRUE(is.na(expenditure_concept_direct_suppressed))) { + NA + } else { + isTRUE(expenditure_concept_direct_suppressed) + }, revenue_concept = revenue_concept, harmonization = harmonization %||% list( applied = FALSE, na_rows_excluded = 0L, na_amount_excluded = 0, diff --git a/R/spending.R b/R/spending.R index c30fa26..6a692d3 100644 --- a/R/spending.R +++ b/R/spending.R @@ -150,7 +150,12 @@ #' component (when one exists), and #' `provenance$expenditure_concept_direct_suppressed` is `TRUE` -- the #' figure in those rows is the intergovernmental leg alone, not Direct + -#' IG. +#' IG. When `category = "All Categories"` is combined with +#' `expenditure_concept = "total"`, this detection cannot run (it keys on +#' per-category rows, which all-categories mode collapses to one literal +#' value), so `expenditure_concept_direct_suppressed` is `NA` rather than a +#' possibly-false `FALSE`; query an explicit `category` to get a real +#' answer. #' @param complete If `TRUE`, fill the requested grid so that a cell the #' corpus does not carry still appears, labelled with **why** it is #' missing, and add a `value_source` column to every row: @@ -274,8 +279,13 @@ cog_spending <- function(govid, years, category = NULL, } govid <- .coerce_govid_input(govid, arg = "govid") + # allow_all_categories = TRUE: cog_spending()/cog_revenue() are the two + # verbs the reserved pseudo-category is defined for. cog_balances() shares + # this validator but leaves the argument at its FALSE default, so it + # rejects "All Categories" instead of silently returning zero rows + # (finding 3, all-categories review). .validate_verb_inputs(govid, years, category, per_capita, adjust_to_year, - recipe) + recipe, allow_all_categories = TRUE) # Recognize the reserved pseudo-category. Detected after type validation so a # non-character `category` still fails with the ordinary type error. @@ -327,6 +337,12 @@ cog_spending <- function(govid, years, category = NULL, "Use `expenditure_concept = \"direct\"` with `complete = TRUE`, or drop `complete`." ) } + if (complete && all_categories) { + .abort_complete_unsupported( + "`category = \"All Categories\"` collapses the category dimension that `code_set` grids over (see `.completion_grid_sql()`), so there is no per-category grid left to fill -- filling a summed row has no defined semantics.", + "Drop `complete`, or use `complete = TRUE` with an explicit `category` (or `category = NULL` for every category)." + ) + } years <- as.integer(years) if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year) @@ -438,13 +454,32 @@ cog_spending <- function(govid, years, category = NULL, # direct spending in that category, which is correct, ordinary data). When # a covering recipe is found, both the row-level notes and the provenance # say so rather than pass silently as a plausible Total. - direct_suppressed_info <- if (identical(expenditure_concept, "total")) { + # + # In all-categories mode this cannot run at all: .detect_direct_suppressed() + # keys on (year, canonical_govid, category), and every row shares the same + # literal "All Categories" value, so the key collides across every real + # category for that (year, govid) -- an IG-only row for a suppressed + # category becomes indistinguishable from one sharing a key with an + # unrelated category's ordinary Direct row. `has_direct` would then read + # TRUE whenever the government has ANY direct spending at all, and the + # detector could never fire. Rather than run it and report a false FALSE, + # skip it and record NA -- the provenance must stop making a claim it + # cannot support (finding 1, all-categories review). + suppression_unavailable <- all_categories && + identical(expenditure_concept, "total") + direct_suppressed_info <- if (suppression_unavailable) { + list(flag = rep(NA, nrow(result)), notes = rep(NA_character_, nrow(result))) + } else if (identical(expenditure_concept, "total")) { .detect_direct_suppressed(con, result, subtype_col) } else { list(flag = rep(FALSE, nrow(result)), notes = rep(NA_character_, nrow(result))) } direct_suppressed <- direct_suppressed_info$flag - direct_suppressed_flag <- isTRUE(any(direct_suppressed)) + direct_suppressed_flag <- if (suppression_unavailable) { + NA + } else { + isTRUE(any(direct_suppressed)) + } result$notes <- .notes_column(result, direct_suppressed_info$notes) @@ -453,9 +488,20 @@ cog_spending <- function(govid, years, category = NULL, # leg is suppressed for at least one requested (year, category), append an # explicit warning rather than let the base note's "Total = Direct + IG" # framing stand unqualified for rows where that arithmetic didn't happen. + # When suppression detection itself is unavailable (all-categories mode), + # say so instead of silently reusing the unqualified base note. expenditure_concept_note_for_prov <- if (identical(expenditure_concept, "total")) { base_note <- "Total = Direct + intergovernmental (M to local govts + L to state govts). Legacy-era IG is assembled from aggregate-flagged rows, which are year-disjoint from their modern leaf components; the L-- family total is excluded." - if (direct_suppressed_flag) { + if (suppression_unavailable) { + paste0( + base_note, + " NOTE: direct-leg-suppression detection is unavailable when ", + "`category = \"All Categories\"` -- it keys on per-category rows, ", + "which this mode collapses. `expenditure_concept_direct_suppressed` ", + "is NA here rather than a possibly-false FALSE; query an explicit ", + "`category` (or `category = NULL`) to get a real answer." + ) + } else if (isTRUE(direct_suppressed_flag)) { paste0( base_note, " NOTE: for at least one requested (year, category) the Direct leg ", @@ -503,9 +549,22 @@ cog_spending <- function(govid, years, category = NULL, result } +#' Shared input validation for the money/holdings verbs. +#' +#' `allow_all_categories` gates the reserved pseudo-category +#' `.ALL_CATEGORIES` ("All Categories"). It is meaningful only where a +#' concept's subtype scope defines what "all" sums over -- +#' `cog_spending()`/`cog_revenue()`, via `.verb_spendrev()`, pass `TRUE`. +#' `cog_balances()` leaves it at the `FALSE` default: holdings are a stock +#' with no concept vocabulary to sum across (see R/balances.R), and before +#' this guard existed `cog_balances(category = "All Categories")` silently +#' matched zero crosswalk rows and returned an empty result with no error +#' (finding 3, all-categories review). This validator is shared specifically +#' so the three verbs cannot drift apart on this again. #' @noRd .validate_verb_inputs <- function(govid, years, category, - per_capita, adjust_to_year, recipe = NULL) { + per_capita, adjust_to_year, recipe = NULL, + allow_all_categories = FALSE) { if (!is.character(govid) || length(govid) == 0L) { cli::cli_abort("`govid` must be a non-empty character vector.") } @@ -515,6 +574,14 @@ cog_spending <- function(govid, years, category = NULL, if (!is.null(category) && !is.character(category)) { cli::cli_abort("`category` must be character or NULL.") } + if (!allow_all_categories && !is.null(category) && + .ALL_CATEGORIES %in% category) { + cli::cli_abort(c( + "{.val {(.ALL_CATEGORIES)}} is not supported here.", + i = "It sums a spending or revenue concept's subtype scope; this verb has no concept vocabulary to sum across.", + i = "Use {.fn cog_spending} or {.fn cog_revenue} for an all-categories total." + ), class = "uscogdata_all_categories_unsupported") + } if (!is.logical(per_capita) || length(per_capita) != 1L) { cli::cli_abort("`per_capita` must be a length-1 logical.") } @@ -650,8 +717,10 @@ cog_spending <- function(govid, years, category = NULL, # this feature exists to surface. # Collapse the category dimension. subtype is deliberately KEPT: it is what - # makes `subtype = "operations"` + all-categories mean "operating - # expenditure", the measure a fiscal comparison actually wants. + # lets a caller filter the result to `spend_subtype == "operations"` and + # get an operating-expenditure total, the measure a fiscal comparison + # actually wants. (There is no `subtype` argument -- this is a post-hoc + # filter on the returned column, not a query parameter.) category_select <- if (all_categories) { sprintf("%s AS category", .sql_lit_chr(.ALL_CATEGORIES)) } else { diff --git a/inst/schemas/provenance-v1.json b/inst/schemas/provenance-v1.json index a47945d..597fcd4 100644 --- a/inst/schemas/provenance-v1.json +++ b/inst/schemas/provenance-v1.json @@ -22,8 +22,8 @@ "description": "How the intergovernmental leg was assembled; null for 'primary' and 'direct'." }, "expenditure_concept_direct_suppressed": { - "type": "boolean", - "description": "TRUE when expenditure_concept = 'total' and at least one requested (year, category) has intergovernmental rows but NO Direct rows in this corpus (typically a legacy aggregate-only family) -- those result rows report the intergovernmental leg alone, not Direct + IG. Always FALSE for expenditure_concept = 'primary' or 'direct'. See the affected rows' `notes` for the recovering recipe, if any." + "type": ["boolean", "null"], + "description": "TRUE when expenditure_concept = 'total' and at least one requested (year, category) has intergovernmental rows but NO Direct rows in this corpus (typically a legacy aggregate-only family) -- those result rows report the intergovernmental leg alone, not Direct + IG. Always FALSE for expenditure_concept = 'primary' or 'direct'. null (NA) when expenditure_concept = 'total' AND category = 'All Categories': the detector keys on per-category rows, which that mode collapses, so suppression cannot be computed -- see `expenditure_concept_note`. See the affected rows' `notes` for the recovering recipe, if any." }, "revenue_concept": { "type": "string", diff --git a/man/cog_balances.Rd b/man/cog_balances.Rd index f594eba..4dab4a7 100644 --- a/man/cog_balances.Rd +++ b/man/cog_balances.Rd @@ -28,7 +28,12 @@ argument: for holdings, `category` is a strict coarsening of every combination would be either redundant or empty. `category = "Fund Balances"` is exactly the `general` family (`W01`/`W31`/`W61`). `balance_subtype` is returned, so a finer split is -one `dplyr::filter()` away.} +one `dplyr::filter()` away. The reserved pseudo-category +`"All Categories"` (see [cog_spending()]) is **not** supported here and +errors with class `uscogdata_all_categories_unsupported`: it sums a +concept's subtype scope, and holdings are a stock with no concept +vocabulary to sum across. Omit `category` to get every category broken +out instead.} \item{per_capita}{Divide holdings by population. Note this is a **stock per resident** (reserves per person), which is *not* comparable to diff --git a/man/cog_spending.Rd b/man/cog_spending.Rd index 1fee7d1..4a9a0e4 100644 --- a/man/cog_spending.Rd +++ b/man/cog_spending.Rd @@ -106,7 +106,12 @@ possibly-misleading `"harmonized"`/`"raw"` value.} component (when one exists), and `provenance$expenditure_concept_direct_suppressed` is `TRUE` -- the figure in those rows is the intergovernmental leg alone, not Direct + - IG.} + IG. When `category = "All Categories"` is combined with + `expenditure_concept = "total"`, this detection cannot run (it keys on + per-category rows, which all-categories mode collapses to one literal + value), so `expenditure_concept_direct_suppressed` is `NA` rather than a + possibly-false `FALSE`; query an explicit `category` to get a real + answer.} \item{complete}{If `TRUE`, fill the requested grid so that a cell the corpus does not carry still appears, labelled with **why** it is diff --git a/tests/testthat/test-all-categories.R b/tests/testthat/test-all-categories.R index c181fad..641fea6 100644 --- a/tests/testthat/test-all-categories.R +++ b/tests/testthat/test-all-categories.R @@ -123,3 +123,69 @@ test_that('cog_categories(pattern=) matches the pseudo-category', { hit <- cog_categories(pattern = "^All Categories$") expect_equal(nrow(hit), 2L) }) + +# --- final whole-branch review fixes --------------------------------------- + +test_that('complete = TRUE is refused when combined with "All Categories"', { + # .completion_grid_sql() would emit `AND c.category IN ('All Categories')`, + # match zero crosswalk rows, and the early return in .complete_result() + # would stamp completion$applied = TRUE, rows_filled = 0 -- reading as "the + # grid was checked and nothing was missing" when nothing was actually + # checked. Filling a summed row has no defined semantics, so the verb must + # refuse the combination outright (finding 2). + expect_error( + cog_spending("552025209777", 2019L, category = "All Categories", + complete = TRUE), + class = "uscogdata_complete_unsupported" + ) + expect_error( + cog_revenue("552025209777", 2019L, category = "All Categories", + complete = TRUE), + class = "uscogdata_complete_unsupported" + ) +}) + +test_that('cog_balances() rejects "All Categories" instead of silently returning zero rows', { + # cog_balances() reuses .validate_verb_inputs() but did not pass + # allow_all_categories = TRUE, so "All Categories" used to become + # `AND category IN ('All Categories')` against balance_annotated -- 0 + # matching crosswalk rows, 0 rows back, no error (finding 3). Holdings are + # a stock with no concept vocabulary to sum across, so the honest answer is + # to refuse, the same way cog_spending()/cog_revenue() refuse other + # nonsensical combinations. + expect_error( + cog_balances("552025209777", 2019L, category = "All Categories"), + class = "uscogdata_all_categories_unsupported" + ) + # An ordinary category still works -- this is not a blanket regression. + r <- suppressMessages( + cog_balances("552025209777", 2019L, category = "Fund Balances") + ) + expect_gt(nrow(r), 0L) +}) + +test_that('expenditure_concept_direct_suppressed is NA, not FALSE, when categories are collapsed', { + # .detect_direct_suppressed() keys on + # paste(year, canonical_govid, category, sep = "\r"). In all-categories + # mode every row carries the literal "All Categories" value, so an IG-only + # row's key collides with any ordinary Direct row for the same + # (year, govid) -- has_direct reads TRUE whenever the government has ANY + # direct spending at all, candidate is always empty, and the detector can + # never fire. Before the fix this silently reported FALSE, an affirmative + # claim the code did not actually compute (finding 1). NA is the honest + # answer: cog_explain(x, format = "list") is required here, since without + # format = "list" it returns the result tibble, not the provenance list. + gov <- "552025209777" + t <- cog_spending(gov, 2019L, category = "All Categories", + expenditure_concept = "total") + prov <- cog_explain(t, format = "list") + expect_true(is.na(prov$expenditure_concept_direct_suppressed)) + expect_false(isTRUE(prov$expenditure_concept_direct_suppressed)) + expect_match(prov$expenditure_concept_note, "unavailable", fixed = TRUE) + + # A per-category "total" query on the same government/year is unaffected -- + # the detector can still key correctly and reports a strict logical. + t_by_cat <- cog_spending(gov, 2019L, expenditure_concept = "total") + prov_by_cat <- cog_explain(t_by_cat, format = "list") + expect_false(is.na(prov_by_cat$expenditure_concept_direct_suppressed)) +}) From 498950afa634196c6d6de0434e667ad2c80ebbc6 Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Wed, 5 Aug 2026 12:38:12 -0400 Subject: [PATCH 9/9] fix: scope all-categories suggestion candidates by subtype, not category (finding 6) .build_suggestions()'s recipe-candidate sub-select was keyed on `WHERE category IN ()`. The reserved pseudo-category "All Categories" is never itself a row in summary_categories.category, so in all-categories mode `candidates` always came back empty and coverage signposting (uscogdata#9) was structurally impossible for the one mode whose entire premise is "you cannot sum the wrong scope" -- measured on Los Angeles County FY2011: category = "Public Welfare" reports 2 suggestions (incl. $271,589,000 excluded E68), category = "All Categories" reported 0, silently losing that same signal. Apply the branch's own design principle: the concept boundary is subtype, not category. .build_suggestions() now accepts all_categories/subtype_col/ subtype_scope (all optional, default off, so no other caller's behaviour changes) and, when all-categories mode is active, scopes the candidate sub-select by ` IN ()` instead -- symmetric with .build_verb_sql()'s own WHERE predicate. The M/L recipe exclusion and the is.null(category) early return are unchanged. After the fix, LA County FY2011 "All Categories" reports 5 suggestions, including welfare_cash_e68_wide for the exact $271,589,000 gap. Adds two covering tests to test-all-categories.R using the bundled fixture (AL state gov, FY2011, "Corrections"): one end-to-end (per-category and all-categories both signpost the same recipe) and one direct on .build_suggestions() proving the subtype-vs-category branch is what changes the query. Updates the 0.2.0 NEWS entry. --- NEWS.md | 13 +++++ R/spending.R | 5 +- R/suggestions.R | 49 ++++++++++++++++-- tests/testthat/test-all-categories.R | 76 ++++++++++++++++++++++++++++ 4 files changed, 139 insertions(+), 4 deletions(-) diff --git a/NEWS.md b/NEWS.md index cd93936..0b43b3f 100644 --- a/NEWS.md +++ b/NEWS.md @@ -19,6 +19,19 @@ * `cog_categories()` advertises `"All Categories"` for the expenditure and revenue vocabularies, so the reserved value is discoverable. +* Coverage signposting (see "Signposting now catches partially-suppressed + categories" below) now also works in `category = "All Categories"` mode. + The recipe-suggestion candidate query used to be scoped by `category`, + which is never a match for the reserved `"All Categories"` value, so + `provenance$suggestions` always came back empty there — the one mode whose + whole point is "you cannot sum the wrong scope" was silently unable to + signal a wrong scope. The candidate query is now scoped by the concept's + subtype allowlist instead, symmetric with how `.build_verb_sql()` itself + scopes the summed total: Los Angeles County FY2011, `category = "All + Categories"` still excludes $271,589,000 of aggregate-published Public + Welfare (`E68`), but now names `recipe = "welfare_cash_e68_wide"` to + recover it instead of reporting zero suggestions. + ## Documentation * `cog_geographic_rollup()` and `cog_peer_compare()` now document that diff --git a/R/spending.R b/R/spending.R index 6a692d3..d636ad4 100644 --- a/R/spending.R +++ b/R/spending.R @@ -442,7 +442,10 @@ cog_spending <- function(govid, years, category = NULL, suggestions <- .build_suggestions(con, govid, years, category, direct_leg_result, resolved$basis, flow_prefixes, - .select_long_view(view_base, resolved$basis)) + .select_long_view(view_base, resolved$basis), + all_categories = all_categories, + subtype_col = subtype_col, + subtype_scope = subtype_scope) } # C1(b): when expenditure_concept = "total", flag any row where the IG diff --git a/R/suggestions.R b/R/suggestions.R index d0d45af..b2b0ce4 100644 --- a/R/suggestions.R +++ b/R/suggestions.R @@ -59,12 +59,36 @@ #' @param long_view Name of the verb's own long view (from #' `.select_long_view()`), passed through to `.suppressed_components()` to #' measure the second qualifying path (uscogdata#9). +#' @param all_categories `TRUE` when the caller's `category` is the reserved +#' pseudo-category (`.ALL_CATEGORIES`). Defaults to `FALSE` so no other +#' caller's behaviour changes. When `TRUE`, the candidate-recipe sub-select +#' is scoped by `subtype_col`/`subtype_scope` instead of by `category` -- +#' symmetric with `.build_verb_sql()`'s own all-categories branch (see +#' R/spending.R): the concept's subtype allowlist is the real scope +#' boundary, not any literal category value, and +#' `.ALL_CATEGORIES` ("All Categories") is never itself a row in +#' `summary_categories.category`, so leaving the category-keyed sub-select +#' in place here always returned zero candidates and silently disabled +#' signposting in all-categories mode (final whole-branch review, finding +#' 6). +#' @param subtype_col Name of the `summary_categories` subtype column to +#' scope by when `all_categories = TRUE` (`"spend_subtype"` or +#' `"revenue_subtype"` -- the same value `.build_verb_sql()` already +#' receives as its own `subtype_col`). Ignored when `all_categories = +#' FALSE`. `NULL` by default. +#' @param subtype_scope Character vector of subtype values to scope by when +#' `all_categories = TRUE` (the same value `.build_verb_sql()` already +#' receives as its own `subtype_scope` -- the concept's subtype allowlist, +#' e.g. `.expenditure_concept_subtypes(expenditure_concept)`). Ignored when +#' `all_categories = FALSE`. `NULL` by default. #' @return List of `list(recipe_id, label, available_years, hint, #' ig_recipe_id, trigger, suppressed_amount, suppressed_years, #' suppressed_codes)`, possibly empty. #' @noRd .build_suggestions <- function(con, govid, years, category, result, basis, - flow_prefixes, long_view) { + flow_prefixes, long_view, + all_categories = FALSE, + subtype_col = NULL, subtype_scope = NULL) { if (!identical(basis, "harmonized") || is.null(category)) return(list()) # Exclude any recipe that is ITSELF an intergovernmental (M/L) recipe -- @@ -79,16 +103,35 @@ # flow-prefix gate below/in `.attach_ig_counterparts()`: an M/L recipe # should never be suggested as a coverage-gap filler for EITHER verb, not # just kept from being named as the *counterpart* of another suggestion. + # + # The inner sub-select is the concept boundary (finding 6, final + # whole-branch review): in all-categories mode it is scoped by + # `subtype_col`/`subtype_scope` -- the same allowlist `.build_verb_sql()` + # applies as a WHERE predicate to make the summed result a *concept*, not + # by `category` (`.ALL_CATEGORIES` is never a row in + # `summary_categories.category`, so a category-keyed sub-select always + # came back empty here). The M/L exclusion below is unchanged either way. + candidate_scope_sql <- if (isTRUE(all_categories)) { + sprintf( + "SELECT DISTINCT item_code FROM summary_categories WHERE %s IN (%s)", + subtype_col, .sql_lit_chr(subtype_scope) + ) + } else { + sprintf( + "SELECT DISTINCT item_code FROM summary_categories WHERE category IN (%s)", + .sql_lit_chr(category) + ) + } candidates <- DBI::dbGetQuery(con, sprintf( "SELECT DISTINCT recipe_id FROM harmonization_recipes WHERE component_code IN ( - SELECT DISTINCT item_code FROM summary_categories WHERE category IN (%s) + %s ) AND recipe_id NOT IN ( SELECT DISTINCT recipe_id FROM harmonization_recipes WHERE LEFT(component_code, 1) IN ('M', 'L') )", - .sql_lit_chr(category) + candidate_scope_sql ))$recipe_id if (length(candidates) == 0L) return(list()) diff --git a/tests/testthat/test-all-categories.R b/tests/testthat/test-all-categories.R index 641fea6..a8ceb08 100644 --- a/tests/testthat/test-all-categories.R +++ b/tests/testthat/test-all-categories.R @@ -189,3 +189,79 @@ test_that('expenditure_concept_direct_suppressed is NA, not FALSE, when categori prov_by_cat <- cog_explain(t_by_cat, format = "list") expect_false(is.na(prov_by_cat$expenditure_concept_direct_suppressed)) }) + +test_that('"All Categories" still signposts coverage gaps (finding 6, final whole-branch review)', { + # .build_suggestions()'s candidate sub-select used to be keyed on + # `category`, e.g. `WHERE category IN ('All Categories')`. Since + # .ALL_CATEGORIES is never itself a row in summary_categories.category, + # that sub-select always came back empty in all-categories mode, so + # `candidates` was empty and .build_suggestions() short-circuited to + # list() -- coverage signposting was structurally impossible for the one + # mode whose whole selling point is "you cannot sum the wrong scope" + # (uscogdata#9's entire point, silently defeated). + # + # AL state government, FY2011, category = "Corrections": this category has + # no legacy leaf rows in FY2011 (aggregate-flagged E04/E05 family), so the + # per-category query returns 0 rows and 3 recipe-hint suggestions fire + # (empty_year path). All-categories mode does not have an empty year -- + # the government has other primary spending in FY2011 -- but the same + # suppressed Corrections dollars are still excluded from the summed total, + # so the fix (scoping the candidate sub-select by subtype_col/subtype_scope + # instead of by category, symmetric with .build_verb_sql()) must still + # surface them via the suppressed_component path. + gov <- "010000226085" + + by_cat <- suppressMessages(cog_spending(gov, 2011L, category = "Corrections")) + sugg_by_cat <- cog_explain(by_cat, format = "list")$suggestions + expect_gt(length(sugg_by_cat), 0L) + + all_cat <- suppressMessages(cog_spending(gov, 2011L, category = "All Categories")) + sugg_all_cat <- cog_explain(all_cat, format = "list")$suggestions + expect_gt(length(sugg_all_cat), 0L) + + # The same Corrections recipe that fired per-category must also fire in + # all-categories mode -- not just some unrelated recipe. + ids_by_cat <- vapply(sugg_by_cat, function(s) s$recipe_id %||% "", character(1)) + ids_all_cat <- vapply(sugg_all_cat, function(s) s$recipe_id %||% "", character(1)) + expect_true("corrections_combined" %in% ids_by_cat) + expect_true("corrections_combined" %in% ids_all_cat) + + # In all-categories mode the government DOES have other primary spending + # in FY2011 (the year itself is not a gap), so the suggestion can only have + # fired via the suppressed_component path, not empty_year. + corr_all <- sugg_all_cat[[which(ids_all_cat == "corrections_combined")]] + expect_identical(corr_all$trigger, "suppressed_component") + expect_gt(corr_all$suppressed_amount, 0) +}) + +test_that('"All Categories" candidate scoping is symmetric with .build_verb_sql() -- subtype, not category', { + # Direct assertion on the mechanism itself (finding 6): in all-categories + # mode .build_suggestions() must scope its candidate recipe sub-select by + # subtype_col/subtype_scope, not by the literal "All Categories" value. + # Passing all_categories = FALSE with the identical category value proves + # the branch -- not merely the subtype_col/subtype_scope arguments' mere + # presence -- is what changes the query. + con <- uscogdata:::.ensure_session() + + none <- uscogdata:::.build_suggestions( + con, govid = "010000226085", years = 2011L, + category = "All Categories", result = NULL, basis = "harmonized", + flow_prefixes = c("E", "F", "G"), + long_view = "spending_long_harmonized", + all_categories = FALSE, + subtype_col = "spend_subtype", + subtype_scope = c("operations", "capital", "assistance") + ) + expect_length(none, 0L) + + scoped <- uscogdata:::.build_suggestions( + con, govid = "010000226085", years = 2011L, + category = "All Categories", result = NULL, basis = "harmonized", + flow_prefixes = c("E", "F", "G"), + long_view = "spending_long_harmonized", + all_categories = TRUE, + subtype_col = "spend_subtype", + subtype_scope = c("operations", "capital", "assistance") + ) + expect_gt(length(scoped), 0L) +})