provenance$coverage's n_units_reporting is category-conditional: it counts governments with rows for the SPECIFIC requested category, which conflates two different things -- a government never collected that year (sampling), and one collected but genuinely spending nothing in that category (a real zero). FY2012 Georgia Police is the motivating case from the issue: a complete census year reads as a 69% "response rate" because most of the gap is cities that contract policing to the county sheriff, not non-response. Adds a second counter, n_units_collected: how many of the caller's expected cohort appear in the corpus that year for ANY category. n_units_collected / n_units_expected is the true collection rate; n_units_reporting / n_units_collected is category participation among collected units. cog_geographic_rollup() and cog_peer_compare() both carry it; cog_explain() prints it alongside n_units_reporting. Two real bugs caught and fixed while finishing this (both against the already-written, previously-uncommitted draft): - .coverage_table()'s candidate list for the collection query was derived from the category-filtered result rows, not the caller's full expected cohort. A government with zero rows in the requested category across every requested year never appears in that result, so it was silently excluded from n_units_collected too -- collapsing the new counter back to the old, broken one for exactly the governments it exists to count. Fixed by threading an explicit `expected_ids` (all_govids / peer_govids) through instead. - The collection query hardcoded long_view = "spending_long_harmonized", which does not exist on a corpus with schema_version < 5 (R/basis.R resolves basis = "raw" there; R/views.R only registers the harmonized views on v5+). cog_geographic_rollup()/cog_peer_compare() would hard-error on a corpus vintage the package otherwise explicitly supports. Fixed by deriving long_view from the basis cog_spending() actually resolved (prov$basis) via the existing .select_long_view() helper, matching how every other basis-aware query in the package already does this. Also: cog_explain()'s general "complete census only in years ending in 2 or 7" footnote was gated on the OLD counter's absence, making it permanently unreachable now that both callers always supply the new one -- ungated it, since the explanation is orthogonal to which counter set is present. Dropped a dead conditional branch, fixed two stale roxygen blocks in R/peers.R/R/rollup.R still describing the old two-counter model, fixed the same staleness in README.md, and switched two `uscogdata:::` self-references to the package's own convention of calling internal helpers unqualified. 1101 tests pass (2 skipped live-corpus), including new direct regression tests for both bugs above (one exercising a government collected-but-absent from a category result, one running the full rollup/peer-compare path against a doctored schema_version 4 corpus). Reviewed by an independent code-reviewer pass (1 HIGH, 1 MEDIUM, 3 LOW -- all addressed above). Closes #36. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
135 lines
4.8 KiB
R
135 lines
4.8 KiB
R
test_that("cog_explain prints verb header and target", {
|
|
skip_if_no_corpus()
|
|
r <- cog_spending("121011212191", 2020L, "Corrections")
|
|
# cli writes to stderr; capture both stdout and message streams.
|
|
txt <- paste(c(
|
|
capture.output(cog_explain(r)),
|
|
capture.output(cog_explain(r), type = "message")
|
|
), collapse = "\n")
|
|
expect_true(grepl("cog_spending", txt))
|
|
expect_true(grepl("Corrections", txt))
|
|
expect_true(grepl("121011212191", txt))
|
|
})
|
|
|
|
test_that("cog_explain format='list' returns structured provenance", {
|
|
skip_if_no_corpus()
|
|
r <- cog_spending("121011212191", 2020L, "Corrections")
|
|
prov <- cog_explain(r, format = "list")
|
|
expect_identical(prov, attr(r, "provenance"))
|
|
})
|
|
|
|
test_that("cog_explain returns result invisibly for chaining", {
|
|
skip_if_no_corpus()
|
|
r <- cog_spending("121011212191", 2020L, "Corrections")
|
|
res <- withVisible(cog_explain(r))
|
|
expect_false(res$visible)
|
|
expect_identical(res$value, r)
|
|
})
|
|
|
|
test_that("cog_explain errors on non-verb input", {
|
|
df <- tibble::tibble(a = 1)
|
|
expect_error(cog_explain(df), "provenance")
|
|
})
|
|
|
|
test_that("cog_explain prints basis + harmonization block", {
|
|
skip_if_no_corpus()
|
|
r <- cog_spending("121011212191", 2020L, "Corrections")
|
|
txt <- paste(c(
|
|
capture.output(cog_explain(r)),
|
|
capture.output(cog_explain(r), type = "message")
|
|
), collapse = "\n")
|
|
expect_true(grepl("Basis: harmonized", txt))
|
|
expect_true(grepl("Harmonization", txt))
|
|
expect_true(grepl("Excluded 0 row", txt))
|
|
})
|
|
|
|
test_that("cog_explain prints a Recipe section for recipe = results", {
|
|
skip_if_no_corpus()
|
|
r <- cog_spending("121011212191", c(2011L, 2012L), recipe = "corrections_combined")
|
|
txt <- paste(c(
|
|
capture.output(cog_explain(r)),
|
|
capture.output(cog_explain(r), type = "message")
|
|
), collapse = "\n")
|
|
expect_true(grepl("Recipe", txt))
|
|
expect_true(grepl("corrections_combined", txt))
|
|
expect_true(grepl("E04", txt))
|
|
expect_true(grepl("E05", txt))
|
|
})
|
|
|
|
test_that("cog_explain prints a Suggestions section when the provenance has one", {
|
|
skip_if_no_corpus()
|
|
r <- suppressMessages(
|
|
cog_spending("121011212191", c(2011L, 2012L), category = "Corrections")
|
|
)
|
|
txt <- paste(c(
|
|
capture.output(cog_explain(r)),
|
|
capture.output(cog_explain(r), type = "message")
|
|
), collapse = "\n")
|
|
expect_true(grepl("Suggestions", txt))
|
|
expect_true(grepl("corrections_combined", txt))
|
|
expect_true(grepl("re-run with recipe", txt))
|
|
})
|
|
|
|
test_that("cog_explain prints the expenditure concept (I1)", {
|
|
skip_if_no_corpus()
|
|
d <- cog_spending("010000226085", years = 2019, category = "Police")
|
|
t <- cog_spending("010000226085", years = 2019, category = "Police",
|
|
expenditure_concept = "total")
|
|
txt_d <- paste(c(
|
|
capture.output(cog_explain(d)),
|
|
capture.output(cog_explain(d), type = "message")
|
|
), collapse = "\n")
|
|
txt_t <- paste(c(
|
|
capture.output(cog_explain(t)),
|
|
capture.output(cog_explain(t), type = "message")
|
|
), collapse = "\n")
|
|
expect_true(grepl("Concept: primary", txt_d))
|
|
expect_true(grepl("Concept: total", txt_t))
|
|
})
|
|
|
|
test_that("cog_explain surfaces the C1(b) direct-suppressed flag as a warning", {
|
|
skip_if_no_corpus()
|
|
t <- suppressMessages(cog_spending(
|
|
"010000226085", years = 2011, category = "Corrections",
|
|
expenditure_concept = "total"
|
|
))
|
|
expect_true(attr(t, "provenance")$expenditure_concept_direct_suppressed)
|
|
txt <- paste(c(
|
|
capture.output(cog_explain(t)),
|
|
capture.output(cog_explain(t), type = "message")
|
|
), collapse = "\n")
|
|
expect_true(grepl("Direct leg unavailable", txt))
|
|
})
|
|
|
|
test_that("cog_explain prints denominator + popyear_range + counts", {
|
|
skip_if_no_corpus()
|
|
with_fixture_corpus({
|
|
r <- cog_spending("121011212191", years = 2019:2020,
|
|
category = "Police", per_capita = TRUE)
|
|
out <- paste(c(
|
|
capture.output(cog_explain(r)),
|
|
capture.output(cog_explain(r), type = "message")
|
|
), collapse = "\n")
|
|
expect_true(grepl("Census F-33", out))
|
|
expect_true(grepl("popyear", out, ignore.case = TRUE))
|
|
expect_true(grepl("census_f33", out))
|
|
# popyear_range should render as 4-digit calendar years, not raw 2-digit
|
|
expect_true(grepl("2019-2020", out))
|
|
expect_false(grepl("popyear range: 19-20", out, fixed = TRUE))
|
|
})
|
|
})
|
|
|
|
test_that("cog_explain reports units collected alongside units reporting (uscogdata#36)", {
|
|
skip_if_no_corpus()
|
|
wi <- cog_gov_search(name = NULL, state = "WI", type = "city")
|
|
roll <- suppressMessages(cog_geographic_rollup(
|
|
govids = list(city = wi$canonical_govid), category = "Police",
|
|
years = 2012L))
|
|
out <- paste(c(
|
|
capture.output(cog_explain(roll)),
|
|
capture.output(cog_explain(roll), type = "message")
|
|
), collapse = "\n")
|
|
expect_true(grepl("597 of 608 units collected", out, fixed = TRUE))
|
|
expect_true(grepl("485 reporting in this category", out, fixed = TRUE))
|
|
})
|