fix: cog_categories() surfaces balance subtypes and accepts type = "balance" #29

Merged
jared merged 1 commits from fix/cog-categories-balance-subtype into main 2026-08-03 12:20:11 -04:00
3 changed files with 61 additions and 9 deletions
+13 -6
View File
@@ -5,12 +5,19 @@
#' Returns the category taxonomy exposed by the corpus's
#' `summary_categories` view, grouped to one row per
#' `(category, subtype)` pair. Use this to discover valid `category`
#' values for [cog_spending()] / [cog_revenue()] /
#' values for [cog_spending()] / [cog_revenue()] / [cog_balances()] /
#' [cog_geographic_rollup()] and to audit which Census item codes feed
#' each category.
#'
#' @param type Either `NULL` (default, return both spending and revenue
#' rows), `"spending"`, or `"revenue"`.
#' `subtype` COALESCEs the crosswalk's three subtype columns, so it carries
#' `spend_subtype` on expenditure rows, `revenue_subtype` on revenue rows and
#' `balance_subtype` on balance rows. Note that [cog_balances()] itself takes
#' no `subtype` argument — for holdings, `category` is a strict coarsening of
#' `balance_subtype` — but the value is surfaced here because it is the
#' discovery surface downstream consumers build their vocabulary from.
#'
#' @param type Either `NULL` (default, every row: expenditure, revenue and
#' balance), `"spending"`, `"revenue"`, or `"balance"`.
#' @param pattern Optional regex matched case-insensitively against the
#' `category` column (e.g. `"Police"` or `"Tax"`).
#' @return Tibble with columns `category`, `category_type`, `subtype`,
@@ -20,8 +27,8 @@
cog_categories <- function(type = NULL, pattern = NULL) {
if (!is.null(type)) {
if (!is.character(type) || length(type) != 1L ||
!type %in% c("spending", "revenue")) {
cli::cli_abort('`type` must be NULL, "spending", or "revenue".')
!type %in% c("spending", "revenue", "balance")) {
cli::cli_abort('`type` must be NULL, "spending", "revenue", or "balance".')
}
}
if (!is.null(pattern) &&
@@ -48,7 +55,7 @@ cog_categories <- function(type = NULL, pattern = NULL) {
sql <- paste(
"SELECT category, category_type,
COALESCE(spend_subtype, revenue_subtype) AS subtype,
COALESCE(spend_subtype, revenue_subtype, balance_subtype) AS subtype,
COUNT(DISTINCT item_code) AS n_codes,
string_agg(DISTINCT item_code, ',' ORDER BY item_code) AS item_codes
FROM summary_categories",
+11 -3
View File
@@ -7,8 +7,8 @@
cog_categories(type = NULL, pattern = NULL)
}
\arguments{
\item{type}{Either `NULL` (default, return both spending and revenue
rows), `"spending"`, or `"revenue"`.}
\item{type}{Either `NULL` (default, every row: expenditure, revenue and
balance), `"spending"`, `"revenue"`, or `"balance"`.}
\item{pattern}{Optional regex matched case-insensitively against the
`category` column (e.g. `"Police"` or `"Tax"`).}
@@ -22,7 +22,15 @@ Tibble with columns `category`, `category_type`, `subtype`,
Returns the category taxonomy exposed by the corpus's
`summary_categories` view, grouped to one row per
`(category, subtype)` pair. Use this to discover valid `category`
values for [cog_spending()] / [cog_revenue()] /
values for [cog_spending()] / [cog_revenue()] / [cog_balances()] /
[cog_geographic_rollup()] and to audit which Census item codes feed
each category.
}
\details{
`subtype` COALESCEs the crosswalk's three subtype columns, so it carries
`spend_subtype` on expenditure rows, `revenue_subtype` on revenue rows and
`balance_subtype` on balance rows. Note that [cog_balances()] itself takes
no `subtype` argument — for holdings, `category` is a strict coarsening of
`balance_subtype` — but the value is surfaced here because it is the
discovery surface downstream consumers build their vocabulary from.
}
+37
View File
@@ -93,3 +93,40 @@ test_that("cog_categories sorted by category_type, category, subtype", {
test_that("cog_categories rejects invalid type", {
expect_error(cog_categories(type = "both"), "type")
})
test_that("cog_categories() surfaces balance subtypes", {
skip_if_no_corpus()
with_fixture_corpus({
cc <- cog_categories()
b <- cc[cc$category_type == "balance", ]
expect_true(nrow(b) > 0L)
# Every balance row must carry its subtype. Before the COALESCE included
# balance_subtype these were all NA, which silently made the balance
# taxonomy undiscoverable -- cog-api derives its subtype vocabulary from
# this function, so an NA here becomes an unusable API parameter.
expect_false(any(is.na(b$subtype)))
# The exact set, read independently from the crosswalk rather than from
# the function under test.
con2 <- DBI::dbConnect(duckdb::duckdb())
on.exit(DBI::dbDisconnect(con2, shutdown = TRUE), add = TRUE)
p <- file.path(fixture_corpus_path(), "data", "summary_categories.parquet")
want <- DBI::dbGetQuery(con2, sprintf(
"SELECT DISTINCT balance_subtype FROM read_parquet(%s)
WHERE category_type = 'balance' AND balance_subtype IS NOT NULL
ORDER BY 1", uscogdata:::.sql_lit_chr(p)))$balance_subtype
expect_true(length(want) > 1L)
expect_identical(sort(unique(b$subtype)), sort(want))
})
})
test_that('cog_categories(type = "balance") filters to holdings', {
skip_if_no_corpus()
with_fixture_corpus({
b <- cog_categories(type = "balance")
expect_true(nrow(b) > 0L)
expect_identical(unique(b$category_type), "balance")
expect_false(any(is.na(b$subtype)))
})
})