feat: cog_spending + cog_revenue + cog_explain

Three core query verbs over the spending_annotated / revenue_annotated
DuckDB views. Each verb accepts vector govid, vector years, optional
category filter, per_capita flag, and adjust_to_year for CPI-U
real-dollar conversion (bundled index).

Amounts are returned in full USD (SUM(amt) * 1000) so callers can
freely rescale to millions/billions. The $1,000s -> $USD conversion
is recorded in provenance$transformations$units_conversion.

Every result carries an attr(., "provenance") list matching
inst/schemas/provenance-v1.json. cog_explain() prints the structured
form via cli or returns the raw list for MCP/JSON consumers.

Also: .fetch_or_cache_manifest() now handles local fixture paths so
tests can point USCOGDATA_FIXTURE_URL at the pipeline publish_cache/
without a working HTTP server.

Tests: 80 pass / 0 fail. devtools::check() 0E/0W/2N (both notes
pre-existing / environmental).
This commit is contained in:
2026-04-24 09:49:09 -04:00
parent cea7a241c5
commit c682e6547d
12 changed files with 672 additions and 1 deletions
+32
View File
@@ -0,0 +1,32 @@
test_that("cog_explain prints verb header and target", {
skip_if_no_corpus()
r <- cog_spending("101006006", 2020L, "Corrections")
# cli writes to stderr; capture both stdout and message streams.
txt <- paste(c(
capture.output(cog_explain(r)),
capture.output(cog_explain(r), type = "message")
), collapse = "\n")
expect_true(grepl("cog_spending", txt))
expect_true(grepl("Corrections", txt))
expect_true(grepl("101006006", txt))
})
test_that("cog_explain format='list' returns structured provenance", {
skip_if_no_corpus()
r <- cog_spending("101006006", 2020L, "Corrections")
prov <- cog_explain(r, format = "list")
expect_identical(prov, attr(r, "provenance"))
})
test_that("cog_explain returns result invisibly for chaining", {
skip_if_no_corpus()
r <- cog_spending("101006006", 2020L, "Corrections")
res <- withVisible(cog_explain(r))
expect_false(res$visible)
expect_identical(res$value, r)
})
test_that("cog_explain errors on non-verb input", {
df <- tibble::tibble(a = 1)
expect_error(cog_explain(df), "provenance")
})
+38
View File
@@ -0,0 +1,38 @@
test_that("cog_revenue returns expected shape for Broward Property Tax 2020", {
skip_if_no_corpus()
r <- cog_revenue("101006006", years = 2020L, category = "Property Tax")
expect_s3_class(r, "tbl_df")
expected_cols <- c("year", "canonical_govid", "gov_name", "revenue_subtype",
"category", "amt_nominal", "codes_included",
"aggregate_fallback", "notes")
expect_true(all(expected_cols %in% names(r)))
expect_equal(unique(r$canonical_govid), "101006006")
expect_equal(unique(r$year), 2020L)
})
test_that("cog_revenue with no category filter returns multiple categories", {
skip_if_no_corpus()
r <- cog_revenue("101006006", years = 2020L)
expect_gt(length(unique(r$category)), 1L)
})
test_that("cog_revenue with per_capita + adjust_to_year adds all columns", {
skip_if_no_corpus()
r <- cog_revenue("101006006", 2020L,
per_capita = TRUE, adjust_to_year = 2022L)
expect_true(all(c("amt_nominal", "amt_real",
"amt_per_capita_nominal", "amt_per_capita_real") %in%
names(r)))
})
test_that("cog_revenue result has provenance attribute", {
skip_if_no_corpus()
r <- cog_revenue("101006006", 2020L)
prov <- attr(r, "provenance")
expect_equal(prov$verb, "cog_revenue")
expect_true(grepl("revenue_annotated", prov$sql_query))
})
test_that("cog_revenue rejects invalid inputs", {
expect_error(cog_revenue(123, 2020L), "character")
})
+82
View File
@@ -0,0 +1,82 @@
test_that("cog_spending returns expected shape for Broward Corrections 2020", {
skip_if_no_corpus()
r <- cog_spending("101006006", years = 2020L, category = "Corrections")
expect_s3_class(r, "tbl_df")
expected_cols <- c("year", "canonical_govid", "gov_name", "spend_subtype",
"category", "amt_nominal", "codes_included",
"aggregate_fallback", "notes")
expect_true(all(expected_cols %in% names(r)))
expect_equal(unique(r$canonical_govid), "101006006")
expect_equal(unique(r$year), 2020L)
expect_equal(unique(r$category), "Corrections")
expect_true(all(r$spend_subtype %in% c("operations", "capital")))
expect_true(all(r$amt_nominal > 0))
})
test_that("cog_spending vectorised years + categories", {
skip_if_no_corpus()
r <- cog_spending("101006006", 2019:2020,
category = c("Corrections", "Police"))
expect_true(all(r$year %in% 2019:2020))
expect_true(all(r$category %in% c("Corrections", "Police")))
expect_gte(nrow(r), 4L)
})
test_that("cog_spending with per_capita adds per-capita nominal column", {
skip_if_no_corpus()
r <- cog_spending("101006006", 2020L, "Corrections", per_capita = TRUE)
expect_true("amt_per_capita_nominal" %in% names(r))
expect_false("amt_real" %in% names(r))
expect_false("amt_per_capita_real" %in% names(r))
expect_true(all(is.finite(r$amt_per_capita_nominal)))
expect_true(all(r$amt_per_capita_nominal < r$amt_nominal))
})
test_that("cog_spending with adjust_to_year adds real column", {
skip_if_no_corpus()
r <- cog_spending("101006006", 2015:2020, "Corrections",
adjust_to_year = 2022L)
expect_true("amt_real" %in% names(r))
r2015 <- dplyr::filter(r, year == 2015L)
expect_true(any(r2015$amt_nominal != r2015$amt_real))
})
test_that("cog_spending with per_capita + adjust_to_year adds all columns", {
skip_if_no_corpus()
r <- cog_spending("101006006", 2020L, "Corrections",
per_capita = TRUE, adjust_to_year = 2022L)
expect_true(all(c("amt_nominal", "amt_real",
"amt_per_capita_nominal", "amt_per_capita_real") %in%
names(r)))
})
test_that("cog_spending for unknown govid returns empty tibble", {
skip_if_no_corpus()
r <- cog_spending("XXXINVALID", 2020L, "Corrections")
expect_s3_class(r, "tbl_df")
expect_equal(nrow(r), 0L)
expect_true("notes" %in% names(r))
# provenance still attached
expect_false(is.null(attr(r, "provenance")))
})
test_that("cog_spending result has provenance attribute matching schema", {
skip_if_no_corpus()
r <- cog_spending("101006006", 2020L, "Corrections")
prov <- attr(r, "provenance")
expect_type(prov, "list")
expect_equal(prov$verb, "cog_spending")
required <- c("verb", "target", "years", "scope", "manifest", "sql_query")
expect_true(all(required %in% names(prov)))
expect_equal(prov$years, 2020L)
expect_equal(prov$category, "Corrections")
expect_type(prov$sql_query, "character")
expect_true(grepl("spending_annotated", prov$sql_query))
expect_type(prov$codes_summed$observed, "character")
expect_true(all(c("E04") %in% prov$codes_summed$observed))
})
test_that("cog_spending rejects invalid inputs", {
expect_error(cog_spending(123, 2020L), "character")
expect_error(cog_spending("101006006", "2020"), "years")
})