fix: cog_categories() surfaces balance subtypes and accepts type = "balance" #29

Merged
jared merged 1 commits from fix/cog-categories-balance-subtype into main 2026-08-03 12:20:11 -04:00
3 changed files with 61 additions and 9 deletions
+13 -6
View File
@@ -5,12 +5,19 @@
#' Returns the category taxonomy exposed by the corpus's #' Returns the category taxonomy exposed by the corpus's
#' `summary_categories` view, grouped to one row per #' `summary_categories` view, grouped to one row per
#' `(category, subtype)` pair. Use this to discover valid `category` #' `(category, subtype)` pair. Use this to discover valid `category`
#' values for [cog_spending()] / [cog_revenue()] / #' values for [cog_spending()] / [cog_revenue()] / [cog_balances()] /
#' [cog_geographic_rollup()] and to audit which Census item codes feed #' [cog_geographic_rollup()] and to audit which Census item codes feed
#' each category. #' each category.
#' #'
#' @param type Either `NULL` (default, return both spending and revenue #' `subtype` COALESCEs the crosswalk's three subtype columns, so it carries
#' rows), `"spending"`, or `"revenue"`. #' `spend_subtype` on expenditure rows, `revenue_subtype` on revenue rows and
#' `balance_subtype` on balance rows. Note that [cog_balances()] itself takes
#' no `subtype` argument — for holdings, `category` is a strict coarsening of
#' `balance_subtype` — but the value is surfaced here because it is the
#' discovery surface downstream consumers build their vocabulary from.
#'
#' @param type Either `NULL` (default, every row: expenditure, revenue and
#' balance), `"spending"`, `"revenue"`, or `"balance"`.
#' @param pattern Optional regex matched case-insensitively against the #' @param pattern Optional regex matched case-insensitively against the
#' `category` column (e.g. `"Police"` or `"Tax"`). #' `category` column (e.g. `"Police"` or `"Tax"`).
#' @return Tibble with columns `category`, `category_type`, `subtype`, #' @return Tibble with columns `category`, `category_type`, `subtype`,
@@ -20,8 +27,8 @@
cog_categories <- function(type = NULL, pattern = NULL) { cog_categories <- function(type = NULL, pattern = NULL) {
if (!is.null(type)) { if (!is.null(type)) {
if (!is.character(type) || length(type) != 1L || if (!is.character(type) || length(type) != 1L ||
!type %in% c("spending", "revenue")) { !type %in% c("spending", "revenue", "balance")) {
cli::cli_abort('`type` must be NULL, "spending", or "revenue".') cli::cli_abort('`type` must be NULL, "spending", "revenue", or "balance".')
} }
} }
if (!is.null(pattern) && if (!is.null(pattern) &&
@@ -48,7 +55,7 @@ cog_categories <- function(type = NULL, pattern = NULL) {
sql <- paste( sql <- paste(
"SELECT category, category_type, "SELECT category, category_type,
COALESCE(spend_subtype, revenue_subtype) AS subtype, COALESCE(spend_subtype, revenue_subtype, balance_subtype) AS subtype,
COUNT(DISTINCT item_code) AS n_codes, COUNT(DISTINCT item_code) AS n_codes,
string_agg(DISTINCT item_code, ',' ORDER BY item_code) AS item_codes string_agg(DISTINCT item_code, ',' ORDER BY item_code) AS item_codes
FROM summary_categories", FROM summary_categories",
+11 -3
View File
@@ -7,8 +7,8 @@
cog_categories(type = NULL, pattern = NULL) cog_categories(type = NULL, pattern = NULL)
} }
\arguments{ \arguments{
\item{type}{Either `NULL` (default, return both spending and revenue \item{type}{Either `NULL` (default, every row: expenditure, revenue and
rows), `"spending"`, or `"revenue"`.} balance), `"spending"`, `"revenue"`, or `"balance"`.}
\item{pattern}{Optional regex matched case-insensitively against the \item{pattern}{Optional regex matched case-insensitively against the
`category` column (e.g. `"Police"` or `"Tax"`).} `category` column (e.g. `"Police"` or `"Tax"`).}
@@ -22,7 +22,15 @@ Tibble with columns `category`, `category_type`, `subtype`,
Returns the category taxonomy exposed by the corpus's Returns the category taxonomy exposed by the corpus's
`summary_categories` view, grouped to one row per `summary_categories` view, grouped to one row per
`(category, subtype)` pair. Use this to discover valid `category` `(category, subtype)` pair. Use this to discover valid `category`
values for [cog_spending()] / [cog_revenue()] / values for [cog_spending()] / [cog_revenue()] / [cog_balances()] /
[cog_geographic_rollup()] and to audit which Census item codes feed [cog_geographic_rollup()] and to audit which Census item codes feed
each category. each category.
} }
\details{
`subtype` COALESCEs the crosswalk's three subtype columns, so it carries
`spend_subtype` on expenditure rows, `revenue_subtype` on revenue rows and
`balance_subtype` on balance rows. Note that [cog_balances()] itself takes
no `subtype` argument — for holdings, `category` is a strict coarsening of
`balance_subtype` — but the value is surfaced here because it is the
discovery surface downstream consumers build their vocabulary from.
}
+37
View File
@@ -93,3 +93,40 @@ test_that("cog_categories sorted by category_type, category, subtype", {
test_that("cog_categories rejects invalid type", { test_that("cog_categories rejects invalid type", {
expect_error(cog_categories(type = "both"), "type") expect_error(cog_categories(type = "both"), "type")
}) })
test_that("cog_categories() surfaces balance subtypes", {
skip_if_no_corpus()
with_fixture_corpus({
cc <- cog_categories()
b <- cc[cc$category_type == "balance", ]
expect_true(nrow(b) > 0L)
# Every balance row must carry its subtype. Before the COALESCE included
# balance_subtype these were all NA, which silently made the balance
# taxonomy undiscoverable -- cog-api derives its subtype vocabulary from
# this function, so an NA here becomes an unusable API parameter.
expect_false(any(is.na(b$subtype)))
# The exact set, read independently from the crosswalk rather than from
# the function under test.
con2 <- DBI::dbConnect(duckdb::duckdb())
on.exit(DBI::dbDisconnect(con2, shutdown = TRUE), add = TRUE)
p <- file.path(fixture_corpus_path(), "data", "summary_categories.parquet")
want <- DBI::dbGetQuery(con2, sprintf(
"SELECT DISTINCT balance_subtype FROM read_parquet(%s)
WHERE category_type = 'balance' AND balance_subtype IS NOT NULL
ORDER BY 1", uscogdata:::.sql_lit_chr(p)))$balance_subtype
expect_true(length(want) > 1L)
expect_identical(sort(unique(b$subtype)), sort(want))
})
})
test_that('cog_categories(type = "balance") filters to holdings', {
skip_if_no_corpus()
with_fixture_corpus({
b <- cog_categories(type = "balance")
expect_true(nrow(b) > 0L)
expect_identical(unique(b$category_type), "balance")
expect_false(any(is.na(b$subtype)))
})
})