From da726a61f6521d42305ef1a2806c1674f5a2a57f Mon Sep 17 00:00:00 2001 From: Jared Knowles Date: Mon, 3 Aug 2026 12:02:26 -0400 Subject: [PATCH] fix: cog_categories() surfaces balance subtypes and accepts type = "balance" The balance work added category_type = "balance" rows to the corpus and cog_balances() to read them, but left cog_categories() -- the discovery surface -- unable to describe them: - subtype COALESCEd only spend_subtype and revenue_subtype, so every balance row came back with subtype = NA - type rejected "balance", so there was no way to ask for the holdings taxonomy at all Both matter downstream: cog-api derives its subtype vocabulary from cog_categories(), so an NA subtype becomes an unusable API parameter. Found while implementing cog-api#26. Note cog_balances() itself still takes no subtype argument -- for holdings category is a strict coarsening of balance_subtype -- but the value belongs in the discovery surface regardless. Tests read the expected subtype set independently from the crosswalk parquet rather than from the function under test. --- R/categories.R | 19 ++++++++++------ man/cog_categories.Rd | 14 +++++++++--- tests/testthat/test-categories.R | 37 ++++++++++++++++++++++++++++++++ 3 files changed, 61 insertions(+), 9 deletions(-) diff --git a/R/categories.R b/R/categories.R index d1903e1..593d9c1 100644 --- a/R/categories.R +++ b/R/categories.R @@ -5,12 +5,19 @@ #' Returns the category taxonomy exposed by the corpus's #' `summary_categories` view, grouped to one row per #' `(category, subtype)` pair. Use this to discover valid `category` -#' values for [cog_spending()] / [cog_revenue()] / +#' values for [cog_spending()] / [cog_revenue()] / [cog_balances()] / #' [cog_geographic_rollup()] and to audit which Census item codes feed #' each category. #' -#' @param type Either `NULL` (default, return both spending and revenue -#' rows), `"spending"`, or `"revenue"`. +#' `subtype` COALESCEs the crosswalk's three subtype columns, so it carries +#' `spend_subtype` on expenditure rows, `revenue_subtype` on revenue rows and +#' `balance_subtype` on balance rows. Note that [cog_balances()] itself takes +#' no `subtype` argument — for holdings, `category` is a strict coarsening of +#' `balance_subtype` — but the value is surfaced here because it is the +#' discovery surface downstream consumers build their vocabulary from. +#' +#' @param type Either `NULL` (default, every row: expenditure, revenue and +#' balance), `"spending"`, `"revenue"`, or `"balance"`. #' @param pattern Optional regex matched case-insensitively against the #' `category` column (e.g. `"Police"` or `"Tax"`). #' @return Tibble with columns `category`, `category_type`, `subtype`, @@ -20,8 +27,8 @@ cog_categories <- function(type = NULL, pattern = NULL) { if (!is.null(type)) { if (!is.character(type) || length(type) != 1L || - !type %in% c("spending", "revenue")) { - cli::cli_abort('`type` must be NULL, "spending", or "revenue".') + !type %in% c("spending", "revenue", "balance")) { + cli::cli_abort('`type` must be NULL, "spending", "revenue", or "balance".') } } if (!is.null(pattern) && @@ -48,7 +55,7 @@ cog_categories <- function(type = NULL, pattern = NULL) { sql <- paste( "SELECT category, category_type, - COALESCE(spend_subtype, revenue_subtype) AS subtype, + COALESCE(spend_subtype, revenue_subtype, balance_subtype) AS subtype, COUNT(DISTINCT item_code) AS n_codes, string_agg(DISTINCT item_code, ',' ORDER BY item_code) AS item_codes FROM summary_categories", diff --git a/man/cog_categories.Rd b/man/cog_categories.Rd index 6cef86a..fa33cfb 100644 --- a/man/cog_categories.Rd +++ b/man/cog_categories.Rd @@ -7,8 +7,8 @@ cog_categories(type = NULL, pattern = NULL) } \arguments{ -\item{type}{Either `NULL` (default, return both spending and revenue -rows), `"spending"`, or `"revenue"`.} +\item{type}{Either `NULL` (default, every row: expenditure, revenue and +balance), `"spending"`, `"revenue"`, or `"balance"`.} \item{pattern}{Optional regex matched case-insensitively against the `category` column (e.g. `"Police"` or `"Tax"`).} @@ -22,7 +22,15 @@ Tibble with columns `category`, `category_type`, `subtype`, Returns the category taxonomy exposed by the corpus's `summary_categories` view, grouped to one row per `(category, subtype)` pair. Use this to discover valid `category` -values for [cog_spending()] / [cog_revenue()] / +values for [cog_spending()] / [cog_revenue()] / [cog_balances()] / [cog_geographic_rollup()] and to audit which Census item codes feed each category. } +\details{ +`subtype` COALESCEs the crosswalk's three subtype columns, so it carries +`spend_subtype` on expenditure rows, `revenue_subtype` on revenue rows and +`balance_subtype` on balance rows. Note that [cog_balances()] itself takes +no `subtype` argument — for holdings, `category` is a strict coarsening of +`balance_subtype` — but the value is surfaced here because it is the +discovery surface downstream consumers build their vocabulary from. +} diff --git a/tests/testthat/test-categories.R b/tests/testthat/test-categories.R index bed3100..11db517 100644 --- a/tests/testthat/test-categories.R +++ b/tests/testthat/test-categories.R @@ -93,3 +93,40 @@ test_that("cog_categories sorted by category_type, category, subtype", { test_that("cog_categories rejects invalid type", { expect_error(cog_categories(type = "both"), "type") }) + +test_that("cog_categories() surfaces balance subtypes", { + skip_if_no_corpus() + with_fixture_corpus({ + cc <- cog_categories() + b <- cc[cc$category_type == "balance", ] + expect_true(nrow(b) > 0L) + + # Every balance row must carry its subtype. Before the COALESCE included + # balance_subtype these were all NA, which silently made the balance + # taxonomy undiscoverable -- cog-api derives its subtype vocabulary from + # this function, so an NA here becomes an unusable API parameter. + expect_false(any(is.na(b$subtype))) + + # The exact set, read independently from the crosswalk rather than from + # the function under test. + con2 <- DBI::dbConnect(duckdb::duckdb()) + on.exit(DBI::dbDisconnect(con2, shutdown = TRUE), add = TRUE) + p <- file.path(fixture_corpus_path(), "data", "summary_categories.parquet") + want <- DBI::dbGetQuery(con2, sprintf( + "SELECT DISTINCT balance_subtype FROM read_parquet(%s) + WHERE category_type = 'balance' AND balance_subtype IS NOT NULL + ORDER BY 1", uscogdata:::.sql_lit_chr(p)))$balance_subtype + expect_true(length(want) > 1L) + expect_identical(sort(unique(b$subtype)), sort(want)) + }) +}) + +test_that('cog_categories(type = "balance") filters to holdings', { + skip_if_no_corpus() + with_fixture_corpus({ + b <- cog_categories(type = "balance") + expect_true(nrow(b) > 0L) + expect_identical(unique(b$category_type), "balance") + expect_false(any(is.na(b$subtype))) + }) +}) -- 2.54.0