Compare commits

..
Author SHA1 Message Date
jared 5668d6b102 fix: accept corpus schema_version 7 (#80)
R-CMD-check / check (pull_request) Has been cancelled
R-CMD-check / check (push) Has been cancelled
2026-08-04 11:22:28 -04:00
jared 0a6d878a36 Merge pull request 'fix: cog_categories() surfaces balance subtypes and accepts type = "balance"' (#29) from fix/cog-categories-balance-subtype into main
R-CMD-check / check (push) Successful in 3m28s
Reviewed-on: #29
2026-08-03 12:20:10 -04:00
jared da726a61f6 fix: cog_categories() surfaces balance subtypes and accepts type = "balance"
R-CMD-check / check (pull_request) Successful in 3m39s
R-CMD-check / check (push) Successful in 3m39s
The balance work added category_type = "balance" rows to the corpus and
cog_balances() to read them, but left cog_categories() -- the discovery
surface -- unable to describe them:

- subtype COALESCEd only spend_subtype and revenue_subtype, so every balance
  row came back with subtype = NA
- type rejected "balance", so there was no way to ask for the holdings
  taxonomy at all

Both matter downstream: cog-api derives its subtype vocabulary from
cog_categories(), so an NA subtype becomes an unusable API parameter. Found
while implementing cog-api#26.

Note cog_balances() itself still takes no subtype argument -- for holdings
category is a strict coarsening of balance_subtype -- but the value belongs
in the discovery surface regardless.

Tests read the expected subtype set independently from the crosswalk parquet
rather than from the function under test.
2026-08-03 12:02:26 -04:00
jared 03c313b46d Merge pull request 'feat: cog_balances(), a reader surface for cash and security holdings (#25)' (#28) from feat/cog-balances-25 into main
R-CMD-check / check (push) Successful in 3m18s
Reviewed-on: #28
2026-08-03 11:52:13 -04:00
5 changed files with 63 additions and 11 deletions
+13 -6
View File
@@ -5,12 +5,19 @@
#' Returns the category taxonomy exposed by the corpus's #' Returns the category taxonomy exposed by the corpus's
#' `summary_categories` view, grouped to one row per #' `summary_categories` view, grouped to one row per
#' `(category, subtype)` pair. Use this to discover valid `category` #' `(category, subtype)` pair. Use this to discover valid `category`
#' values for [cog_spending()] / [cog_revenue()] / #' values for [cog_spending()] / [cog_revenue()] / [cog_balances()] /
#' [cog_geographic_rollup()] and to audit which Census item codes feed #' [cog_geographic_rollup()] and to audit which Census item codes feed
#' each category. #' each category.
#' #'
#' @param type Either `NULL` (default, return both spending and revenue #' `subtype` COALESCEs the crosswalk's three subtype columns, so it carries
#' rows), `"spending"`, or `"revenue"`. #' `spend_subtype` on expenditure rows, `revenue_subtype` on revenue rows and
#' `balance_subtype` on balance rows. Note that [cog_balances()] itself takes
#' no `subtype` argument — for holdings, `category` is a strict coarsening of
#' `balance_subtype` — but the value is surfaced here because it is the
#' discovery surface downstream consumers build their vocabulary from.
#'
#' @param type Either `NULL` (default, every row: expenditure, revenue and
#' balance), `"spending"`, `"revenue"`, or `"balance"`.
#' @param pattern Optional regex matched case-insensitively against the #' @param pattern Optional regex matched case-insensitively against the
#' `category` column (e.g. `"Police"` or `"Tax"`). #' `category` column (e.g. `"Police"` or `"Tax"`).
#' @return Tibble with columns `category`, `category_type`, `subtype`, #' @return Tibble with columns `category`, `category_type`, `subtype`,
@@ -20,8 +27,8 @@
cog_categories <- function(type = NULL, pattern = NULL) { cog_categories <- function(type = NULL, pattern = NULL) {
if (!is.null(type)) { if (!is.null(type)) {
if (!is.character(type) || length(type) != 1L || if (!is.character(type) || length(type) != 1L ||
!type %in% c("spending", "revenue")) { !type %in% c("spending", "revenue", "balance")) {
cli::cli_abort('`type` must be NULL, "spending", or "revenue".') cli::cli_abort('`type` must be NULL, "spending", "revenue", or "balance".')
} }
} }
if (!is.null(pattern) && if (!is.null(pattern) &&
@@ -48,7 +55,7 @@ cog_categories <- function(type = NULL, pattern = NULL) {
sql <- paste( sql <- paste(
"SELECT category, category_type, "SELECT category, category_type,
COALESCE(spend_subtype, revenue_subtype) AS subtype, COALESCE(spend_subtype, revenue_subtype, balance_subtype) AS subtype,
COUNT(DISTINCT item_code) AS n_codes, COUNT(DISTINCT item_code) AS n_codes,
string_agg(DISTINCT item_code, ',' ORDER BY item_code) AS item_codes string_agg(DISTINCT item_code, ',' ORDER BY item_code) AS item_codes
FROM summary_categories", FROM summary_categories",
+1 -1
View File
@@ -138,7 +138,7 @@
#' year, matching canonical_fips_xwalk) rather than as-of-year; as-of-year #' year, matching canonical_fips_xwalk) rather than as-of-year; as-of-year
#' moved to the *_asof columns. This package's own geography always came from #' moved to the *_asof columns. This package's own geography always came from
#' the xwalk (already present-based), so behaviour is unchanged. #' the xwalk (already present-based), so behaviour is unchanged.
.validate_schema <- function(manifest, supported = c(4L, 5L, 6L)) { .validate_schema <- function(manifest, supported = c(4L, 5L, 6L, 7L)) {
if (!manifest$schema_version %in% supported) { if (!manifest$schema_version %in% supported) {
cli::cli_abort(c( cli::cli_abort(c(
"Corpus schema version mismatch.", "Corpus schema version mismatch.",
+1 -1
View File
@@ -12,7 +12,7 @@ cog_open <- function(url = .resolve_url(),
DBI::dbExecute(con, "INSTALL httpfs; LOAD httpfs;") DBI::dbExecute(con, "INSTALL httpfs; LOAD httpfs;")
manifest <- .fetch_or_cache_manifest(url, cache_dir) manifest <- .fetch_or_cache_manifest(url, cache_dir)
.validate_schema(manifest, supported = c(4L, 5L, 6L)) .validate_schema(manifest, supported = c(4L, 5L, 6L, 7L))
.validate_scope(manifest) .validate_scope(manifest)
.register_views(con, url, manifest) .register_views(con, url, manifest)
+11 -3
View File
@@ -7,8 +7,8 @@
cog_categories(type = NULL, pattern = NULL) cog_categories(type = NULL, pattern = NULL)
} }
\arguments{ \arguments{
\item{type}{Either `NULL` (default, return both spending and revenue \item{type}{Either `NULL` (default, every row: expenditure, revenue and
rows), `"spending"`, or `"revenue"`.} balance), `"spending"`, `"revenue"`, or `"balance"`.}
\item{pattern}{Optional regex matched case-insensitively against the \item{pattern}{Optional regex matched case-insensitively against the
`category` column (e.g. `"Police"` or `"Tax"`).} `category` column (e.g. `"Police"` or `"Tax"`).}
@@ -22,7 +22,15 @@ Tibble with columns `category`, `category_type`, `subtype`,
Returns the category taxonomy exposed by the corpus's Returns the category taxonomy exposed by the corpus's
`summary_categories` view, grouped to one row per `summary_categories` view, grouped to one row per
`(category, subtype)` pair. Use this to discover valid `category` `(category, subtype)` pair. Use this to discover valid `category`
values for [cog_spending()] / [cog_revenue()] / values for [cog_spending()] / [cog_revenue()] / [cog_balances()] /
[cog_geographic_rollup()] and to audit which Census item codes feed [cog_geographic_rollup()] and to audit which Census item codes feed
each category. each category.
} }
\details{
`subtype` COALESCEs the crosswalk's three subtype columns, so it carries
`spend_subtype` on expenditure rows, `revenue_subtype` on revenue rows and
`balance_subtype` on balance rows. Note that [cog_balances()] itself takes
no `subtype` argument — for holdings, `category` is a strict coarsening of
`balance_subtype` — but the value is surfaced here because it is the
discovery surface downstream consumers build their vocabulary from.
}
+37
View File
@@ -93,3 +93,40 @@ test_that("cog_categories sorted by category_type, category, subtype", {
test_that("cog_categories rejects invalid type", { test_that("cog_categories rejects invalid type", {
expect_error(cog_categories(type = "both"), "type") expect_error(cog_categories(type = "both"), "type")
}) })
test_that("cog_categories() surfaces balance subtypes", {
skip_if_no_corpus()
with_fixture_corpus({
cc <- cog_categories()
b <- cc[cc$category_type == "balance", ]
expect_true(nrow(b) > 0L)
# Every balance row must carry its subtype. Before the COALESCE included
# balance_subtype these were all NA, which silently made the balance
# taxonomy undiscoverable -- cog-api derives its subtype vocabulary from
# this function, so an NA here becomes an unusable API parameter.
expect_false(any(is.na(b$subtype)))
# The exact set, read independently from the crosswalk rather than from
# the function under test.
con2 <- DBI::dbConnect(duckdb::duckdb())
on.exit(DBI::dbDisconnect(con2, shutdown = TRUE), add = TRUE)
p <- file.path(fixture_corpus_path(), "data", "summary_categories.parquet")
want <- DBI::dbGetQuery(con2, sprintf(
"SELECT DISTINCT balance_subtype FROM read_parquet(%s)
WHERE category_type = 'balance' AND balance_subtype IS NOT NULL
ORDER BY 1", uscogdata:::.sql_lit_chr(p)))$balance_subtype
expect_true(length(want) > 1L)
expect_identical(sort(unique(b$subtype)), sort(want))
})
})
test_that('cog_categories(type = "balance") filters to holdings', {
skip_if_no_corpus()
with_fixture_corpus({
b <- cog_categories(type = "balance")
expect_true(nrow(b) > 0L)
expect_identical(unique(b$category_type), "balance")
expect_false(any(is.na(b$subtype)))
})
})