fix: cog_categories() surfaces balance subtypes and accepts type = "balance"
The balance work added category_type = "balance" rows to the corpus and cog_balances() to read them, but left cog_categories() -- the discovery surface -- unable to describe them: - subtype COALESCEd only spend_subtype and revenue_subtype, so every balance row came back with subtype = NA - type rejected "balance", so there was no way to ask for the holdings taxonomy at all Both matter downstream: cog-api derives its subtype vocabulary from cog_categories(), so an NA subtype becomes an unusable API parameter. Found while implementing cog-api#26. Note cog_balances() itself still takes no subtype argument -- for holdings category is a strict coarsening of balance_subtype -- but the value belongs in the discovery surface regardless. Tests read the expected subtype set independently from the crosswalk parquet rather than from the function under test.
This commit is contained in:
+13
-6
@@ -5,12 +5,19 @@
|
|||||||
#' Returns the category taxonomy exposed by the corpus's
|
#' Returns the category taxonomy exposed by the corpus's
|
||||||
#' `summary_categories` view, grouped to one row per
|
#' `summary_categories` view, grouped to one row per
|
||||||
#' `(category, subtype)` pair. Use this to discover valid `category`
|
#' `(category, subtype)` pair. Use this to discover valid `category`
|
||||||
#' values for [cog_spending()] / [cog_revenue()] /
|
#' values for [cog_spending()] / [cog_revenue()] / [cog_balances()] /
|
||||||
#' [cog_geographic_rollup()] and to audit which Census item codes feed
|
#' [cog_geographic_rollup()] and to audit which Census item codes feed
|
||||||
#' each category.
|
#' each category.
|
||||||
#'
|
#'
|
||||||
#' @param type Either `NULL` (default, return both spending and revenue
|
#' `subtype` COALESCEs the crosswalk's three subtype columns, so it carries
|
||||||
#' rows), `"spending"`, or `"revenue"`.
|
#' `spend_subtype` on expenditure rows, `revenue_subtype` on revenue rows and
|
||||||
|
#' `balance_subtype` on balance rows. Note that [cog_balances()] itself takes
|
||||||
|
#' no `subtype` argument — for holdings, `category` is a strict coarsening of
|
||||||
|
#' `balance_subtype` — but the value is surfaced here because it is the
|
||||||
|
#' discovery surface downstream consumers build their vocabulary from.
|
||||||
|
#'
|
||||||
|
#' @param type Either `NULL` (default, every row: expenditure, revenue and
|
||||||
|
#' balance), `"spending"`, `"revenue"`, or `"balance"`.
|
||||||
#' @param pattern Optional regex matched case-insensitively against the
|
#' @param pattern Optional regex matched case-insensitively against the
|
||||||
#' `category` column (e.g. `"Police"` or `"Tax"`).
|
#' `category` column (e.g. `"Police"` or `"Tax"`).
|
||||||
#' @return Tibble with columns `category`, `category_type`, `subtype`,
|
#' @return Tibble with columns `category`, `category_type`, `subtype`,
|
||||||
@@ -20,8 +27,8 @@
|
|||||||
cog_categories <- function(type = NULL, pattern = NULL) {
|
cog_categories <- function(type = NULL, pattern = NULL) {
|
||||||
if (!is.null(type)) {
|
if (!is.null(type)) {
|
||||||
if (!is.character(type) || length(type) != 1L ||
|
if (!is.character(type) || length(type) != 1L ||
|
||||||
!type %in% c("spending", "revenue")) {
|
!type %in% c("spending", "revenue", "balance")) {
|
||||||
cli::cli_abort('`type` must be NULL, "spending", or "revenue".')
|
cli::cli_abort('`type` must be NULL, "spending", "revenue", or "balance".')
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (!is.null(pattern) &&
|
if (!is.null(pattern) &&
|
||||||
@@ -48,7 +55,7 @@ cog_categories <- function(type = NULL, pattern = NULL) {
|
|||||||
|
|
||||||
sql <- paste(
|
sql <- paste(
|
||||||
"SELECT category, category_type,
|
"SELECT category, category_type,
|
||||||
COALESCE(spend_subtype, revenue_subtype) AS subtype,
|
COALESCE(spend_subtype, revenue_subtype, balance_subtype) AS subtype,
|
||||||
COUNT(DISTINCT item_code) AS n_codes,
|
COUNT(DISTINCT item_code) AS n_codes,
|
||||||
string_agg(DISTINCT item_code, ',' ORDER BY item_code) AS item_codes
|
string_agg(DISTINCT item_code, ',' ORDER BY item_code) AS item_codes
|
||||||
FROM summary_categories",
|
FROM summary_categories",
|
||||||
|
|||||||
+11
-3
@@ -7,8 +7,8 @@
|
|||||||
cog_categories(type = NULL, pattern = NULL)
|
cog_categories(type = NULL, pattern = NULL)
|
||||||
}
|
}
|
||||||
\arguments{
|
\arguments{
|
||||||
\item{type}{Either `NULL` (default, return both spending and revenue
|
\item{type}{Either `NULL` (default, every row: expenditure, revenue and
|
||||||
rows), `"spending"`, or `"revenue"`.}
|
balance), `"spending"`, `"revenue"`, or `"balance"`.}
|
||||||
|
|
||||||
\item{pattern}{Optional regex matched case-insensitively against the
|
\item{pattern}{Optional regex matched case-insensitively against the
|
||||||
`category` column (e.g. `"Police"` or `"Tax"`).}
|
`category` column (e.g. `"Police"` or `"Tax"`).}
|
||||||
@@ -22,7 +22,15 @@ Tibble with columns `category`, `category_type`, `subtype`,
|
|||||||
Returns the category taxonomy exposed by the corpus's
|
Returns the category taxonomy exposed by the corpus's
|
||||||
`summary_categories` view, grouped to one row per
|
`summary_categories` view, grouped to one row per
|
||||||
`(category, subtype)` pair. Use this to discover valid `category`
|
`(category, subtype)` pair. Use this to discover valid `category`
|
||||||
values for [cog_spending()] / [cog_revenue()] /
|
values for [cog_spending()] / [cog_revenue()] / [cog_balances()] /
|
||||||
[cog_geographic_rollup()] and to audit which Census item codes feed
|
[cog_geographic_rollup()] and to audit which Census item codes feed
|
||||||
each category.
|
each category.
|
||||||
}
|
}
|
||||||
|
\details{
|
||||||
|
`subtype` COALESCEs the crosswalk's three subtype columns, so it carries
|
||||||
|
`spend_subtype` on expenditure rows, `revenue_subtype` on revenue rows and
|
||||||
|
`balance_subtype` on balance rows. Note that [cog_balances()] itself takes
|
||||||
|
no `subtype` argument — for holdings, `category` is a strict coarsening of
|
||||||
|
`balance_subtype` — but the value is surfaced here because it is the
|
||||||
|
discovery surface downstream consumers build their vocabulary from.
|
||||||
|
}
|
||||||
|
|||||||
@@ -93,3 +93,40 @@ test_that("cog_categories sorted by category_type, category, subtype", {
|
|||||||
test_that("cog_categories rejects invalid type", {
|
test_that("cog_categories rejects invalid type", {
|
||||||
expect_error(cog_categories(type = "both"), "type")
|
expect_error(cog_categories(type = "both"), "type")
|
||||||
})
|
})
|
||||||
|
|
||||||
|
test_that("cog_categories() surfaces balance subtypes", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
cc <- cog_categories()
|
||||||
|
b <- cc[cc$category_type == "balance", ]
|
||||||
|
expect_true(nrow(b) > 0L)
|
||||||
|
|
||||||
|
# Every balance row must carry its subtype. Before the COALESCE included
|
||||||
|
# balance_subtype these were all NA, which silently made the balance
|
||||||
|
# taxonomy undiscoverable -- cog-api derives its subtype vocabulary from
|
||||||
|
# this function, so an NA here becomes an unusable API parameter.
|
||||||
|
expect_false(any(is.na(b$subtype)))
|
||||||
|
|
||||||
|
# The exact set, read independently from the crosswalk rather than from
|
||||||
|
# the function under test.
|
||||||
|
con2 <- DBI::dbConnect(duckdb::duckdb())
|
||||||
|
on.exit(DBI::dbDisconnect(con2, shutdown = TRUE), add = TRUE)
|
||||||
|
p <- file.path(fixture_corpus_path(), "data", "summary_categories.parquet")
|
||||||
|
want <- DBI::dbGetQuery(con2, sprintf(
|
||||||
|
"SELECT DISTINCT balance_subtype FROM read_parquet(%s)
|
||||||
|
WHERE category_type = 'balance' AND balance_subtype IS NOT NULL
|
||||||
|
ORDER BY 1", uscogdata:::.sql_lit_chr(p)))$balance_subtype
|
||||||
|
expect_true(length(want) > 1L)
|
||||||
|
expect_identical(sort(unique(b$subtype)), sort(want))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that('cog_categories(type = "balance") filters to holdings', {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
b <- cog_categories(type = "balance")
|
||||||
|
expect_true(nrow(b) > 0L)
|
||||||
|
expect_identical(unique(b$category_type), "balance")
|
||||||
|
expect_false(any(is.na(b$subtype)))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|||||||
Reference in New Issue
Block a user