.recipe_coverage()'s covered_years were computed once per recipe as a
union across ALL of its components (aggregate rows included), without
excluding the component currently being tested for a gap. So a code
whose only representation in a year was its own wide-era aggregate row
satisfied its own "covered" check -- self-coverage, not the "other
components" review-doc 0.3's criterion actually specifies ("...has no
rows ... but other components do").
.recipe_coverage() now returns (recipe_id, component_code, year)
triples instead of collapsing across components, and
.recipe_component_gapped() excludes the component under test before
checking coverage, so a gap only fires when a genuinely different
sibling component has data in that year.
Adds the boundary test this gap in coverage let slip through untested:
Broward FY2011 alone, where E05/F05/G05 each report solely as their own
wide-era aggregate row and E04/F04/G04 don't exist as codes before 2012
corpus-wide, so none of the three Corrections recipes have any OTHER
component to cover them -- must produce zero suggestions. The existing
2011-2012 combined test still passes, now firing because of the 2012
E05-gapped/E04-covers pair rather than 2011's self-coverage. Updates the
header comment to state the other-component requirement explicitly.
313 lines
15 KiB
R
313 lines
15 KiB
R
# tests/testthat/test-recipes.R
|
|
#
|
|
# cog_recipes(), recipe = in cog_spending()/cog_revenue(), and the
|
|
# recipe-component-driven signposting in prov$suggestions (Phase R2 /
|
|
# Task 11, schema_version 5).
|
|
|
|
test_that("cog_recipes lists the curated catalog including corrections_combined", {
|
|
skip_if_no_corpus()
|
|
r <- cog_recipes()
|
|
expect_s3_class(r, "tbl_df")
|
|
expect_equal(names(r), c("recipe_id", "label", "n_components", "year_min", "year_max"))
|
|
expect_equal(nrow(r), 24L)
|
|
expect_true("corrections_combined" %in% r$recipe_id)
|
|
expect_true("t19_selective_sales_wide" %in% r$recipe_id)
|
|
expect_true("ig_federal_b89_wide" %in% r$recipe_id)
|
|
expect_true("rents_royalties_u4_wide" %in% r$recipe_id)
|
|
expect_true("higher_ed_e18_wide" %in% r$recipe_id)
|
|
expect_true("cash_securities_z77_wide" %in% r$recipe_id)
|
|
# Superseded id from the pre-curation brief text must NOT be present.
|
|
expect_false("corrections_judicial_combined" %in% r$recipe_id)
|
|
})
|
|
|
|
test_that("cog_recipes(pattern=) filters by recipe_id or label", {
|
|
skip_if_no_corpus()
|
|
r <- cog_recipes("corrections")
|
|
expect_true(nrow(r) >= 1L)
|
|
expect_true(all(grepl("corrections", r$recipe_id, ignore.case = TRUE) |
|
|
grepl("corrections", r$label, ignore.case = TRUE)))
|
|
})
|
|
|
|
test_that("cog_recipes requires schema_version >= 5", {
|
|
skip_if_no_corpus()
|
|
with_doctored_schema_version(4L, {
|
|
expect_error(cog_recipes(), class = "uscogdata_schema_unsupported")
|
|
})
|
|
})
|
|
|
|
# --- recipe = : generic join, no is_aggregate filter -----------------------
|
|
|
|
test_that("recipe = 'corrections_combined' is continuous across the 2011->2012 seam", {
|
|
skip_if_no_corpus()
|
|
r <- cog_spending("121011212191", years = c(2011L, 2012L),
|
|
recipe = "corrections_combined")
|
|
expect_equal(nrow(r), 2L)
|
|
expect_true(all(c("year", "canonical_govid", "gov_name", "spend_subtype",
|
|
"category", "amt_nominal", "codes_included",
|
|
"aggregate_fallback", "notes") %in% names(r)))
|
|
expect_equal(unique(r$spend_subtype), "recipe")
|
|
expect_equal(unique(r$category), "Corrections (functions 04+05 combined)")
|
|
expect_false(any(r$aggregate_fallback))
|
|
|
|
r2011 <- r$amt_nominal[r$year == 2011L]
|
|
r2012 <- r$amt_nominal[r$year == 2012L]
|
|
# 2011: E05 only exists as a wide-era AGGREGATE row (is_aggregate = TRUE)
|
|
# for Broward -- data-verified $216,088,000. Since .run_recipe() does NOT
|
|
# filter is_aggregate (amendment: the recipe join must not, because these
|
|
# families exist ONLY as aggregate rows in the wide era), the recipe
|
|
# correctly picks this up.
|
|
expect_equal(r2011, 216088000)
|
|
# 2012: modern E04 leaf ($213,056,000); Broward reports no E05 leaf that
|
|
# year, so the recipe total equals E04 alone -- still continuous with the
|
|
# 2011 aggregate, proving the wide-aggregate -> modern-leaf handoff.
|
|
expect_equal(r2012, 213056000)
|
|
expect_true(all(grepl("E04|E05", r$codes_included)))
|
|
})
|
|
|
|
test_that("recipe result carries a recipe provenance block with component rows", {
|
|
skip_if_no_corpus()
|
|
r <- cog_spending("121011212191", years = c(2011L, 2012L),
|
|
recipe = "corrections_combined")
|
|
prov <- attr(r, "provenance")
|
|
expect_equal(prov$basis, "recipe")
|
|
expect_equal(prov$category, "Corrections (functions 04+05 combined)")
|
|
expect_type(prov$recipe, "list")
|
|
expect_equal(prov$recipe$recipe_id, "corrections_combined")
|
|
expect_equal(prov$recipe$label, "Corrections (functions 04+05 combined)")
|
|
expect_length(prov$recipe$components, 2L)
|
|
comp_codes <- vapply(prov$recipe$components, function(x) x$component_code, character(1))
|
|
expect_setequal(comp_codes, c("E04", "E05"))
|
|
# A recipe query resolves its own coverage; it should never also carry
|
|
# suggestions for itself.
|
|
expect_length(prov$suggestions, 0L)
|
|
})
|
|
|
|
test_that("recipe results report an unambiguous basis/harmonization, ignoring basis=", {
|
|
skip_if_no_corpus()
|
|
# A recipe query bypasses spending_annotated(_harmonized) entirely --
|
|
# .run_recipe() joins `long` directly -- so `basis` must never read
|
|
# "harmonized"/"raw" (which would describe a code path this query never
|
|
# took) regardless of what the caller passed for `basis`. Task 12
|
|
# consumes provenance verbatim, so this needs to be unambiguous.
|
|
r_default <- cog_spending("121011212191", years = c(2011L, 2012L),
|
|
recipe = "corrections_combined")
|
|
r_raw <- cog_spending("121011212191", years = c(2011L, 2012L),
|
|
recipe = "corrections_combined", basis = "raw")
|
|
r_harm <- cog_spending("121011212191", years = c(2011L, 2012L),
|
|
recipe = "corrections_combined", basis = "harmonized")
|
|
|
|
for (r in list(r_default, r_raw, r_harm)) {
|
|
prov <- attr(r, "provenance")
|
|
expect_equal(prov$basis, "recipe")
|
|
expect_true(is.na(prov$basis_note))
|
|
expect_false(prov$harmonization$applied)
|
|
expect_equal(prov$harmonization$na_rows_excluded, 0L)
|
|
expect_match(prov$harmonization$note, "recipe", ignore.case = TRUE)
|
|
}
|
|
|
|
# basis= truly has zero effect on a recipe query's actual numbers.
|
|
expect_equal(r_raw$amt_nominal, r_harm$amt_nominal)
|
|
expect_equal(r_default$amt_nominal, r_raw$amt_nominal)
|
|
})
|
|
|
|
test_that("recipe = 't19_selective_sales_wide' sums the local T11/T14 legs when present", {
|
|
skip_if_no_corpus()
|
|
# Westminster City, CA (canonical_govid 082001211654): T11 = 0 in 2011,
|
|
# T11 = 568 (T14 = 0/absent) in 2012 -- a real, data-verified equality/
|
|
# inequality pair inside the amended fixture window (2011-2012), standing
|
|
# in for the brief's original 2004/2005 example (out of scope per the
|
|
# amended fixture years; the underlying local-tax-split boundary is
|
|
# nationally FY2005, but this government's own T11 reporting activates
|
|
# within our 2011-2012 window).
|
|
r <- cog_revenue("082001211654", years = c(2011L, 2012L),
|
|
recipe = "t19_selective_sales_wide")
|
|
|
|
# Raw, single-code T19 total (not the "Other Taxes" category total, which
|
|
# would also sum in T11/T14/T21/T23/T27/T29/T53/T99 -- queried directly to
|
|
# isolate exactly the code the brief's equality/inequality check is about).
|
|
con <- uscogdata:::.ensure_session()
|
|
raw_t19 <- DBI::dbGetQuery(con, "
|
|
SELECT year, SUM(amt) * 1000.0 AS amt
|
|
FROM revenue_long
|
|
WHERE canonical_govid = '082001211654' AND item_code = 'T19'
|
|
AND year IN (2011, 2012)
|
|
GROUP BY year ORDER BY year
|
|
")
|
|
raw_t19_2011 <- raw_t19$amt[raw_t19$year == 2011L]
|
|
raw_t19_2012 <- raw_t19$amt[raw_t19$year == 2012L]
|
|
expect_equal(raw_t19_2011, 2231000)
|
|
expect_equal(raw_t19_2012, 2365000)
|
|
|
|
recipe_2011 <- r$amt_nominal[r$year == 2011L]
|
|
recipe_2012 <- r$amt_nominal[r$year == 2012L]
|
|
|
|
expect_equal(recipe_2011, raw_t19_2011) # equality: no local T11/T14 yet
|
|
expect_gt(recipe_2012, raw_t19_2012) # inequality: local T11 joins in
|
|
expect_equal(recipe_2012, raw_t19_2012 + 568000)
|
|
})
|
|
|
|
test_that("recipe = and category = together aborts", {
|
|
skip_if_no_corpus()
|
|
expect_error(
|
|
cog_spending("121011212191", 2020L, category = "Corrections",
|
|
recipe = "corrections_combined"),
|
|
class = "uscogdata_recipe_category_conflict"
|
|
)
|
|
})
|
|
|
|
test_that("unknown recipe id aborts and lists valid ids", {
|
|
skip_if_no_corpus()
|
|
err <- tryCatch(
|
|
cog_spending("121011212191", 2020L, recipe = "does_not_exist"),
|
|
error = identity
|
|
)
|
|
expect_s3_class(err, "uscogdata_unknown_recipe")
|
|
expect_match(conditionMessage(err), "corrections_combined")
|
|
})
|
|
|
|
test_that("recipe = requires schema_version >= 5", {
|
|
skip_if_no_corpus()
|
|
with_doctored_schema_version(4L, {
|
|
expect_error(
|
|
cog_spending("121011212191", 2020L, recipe = "corrections_combined"),
|
|
class = "uscogdata_schema_unsupported"
|
|
)
|
|
})
|
|
})
|
|
|
|
# --- signposting -------------------------------------------------------
|
|
#
|
|
# Phase R3 / Task 19c: .build_suggestions() was narrowed from a whole-result
|
|
# gap check (R2: does the ENTIRE category result have zero rows in a
|
|
# requested year) to per-code gap detection (does a specific recipe
|
|
# component -- itself a member of the requested category -- have zero rows
|
|
# in a year the recipe's own generic join otherwise covers). See
|
|
# R/suggestions.R's header comment and docs/phase_r_harmonization_review.md
|
|
# § 0.3. The R2 test below ("...across the 2011->2012 gap") is unaffected
|
|
# by the refinement (it already passed under both the coarse and per-code
|
|
# rule). The next few pin cases the coarse rule specifically could NOT see.
|
|
|
|
test_that("signposting suggests corrections_combined across the 2011->2012 gap", {
|
|
skip_if_no_corpus()
|
|
expect_message(
|
|
r <- cog_spending("121011212191", years = c(2011L, 2012L),
|
|
category = "Corrections"),
|
|
"recipe"
|
|
)
|
|
prov <- attr(r, "provenance")
|
|
expect_true(length(prov$suggestions) >= 1L)
|
|
ids <- vapply(prov$suggestions, function(s) s$recipe_id, character(1))
|
|
expect_true("corrections_combined" %in% ids)
|
|
hit <- prov$suggestions[[which(ids == "corrections_combined")]]
|
|
expect_equal(hit$hint, "re-run with recipe = 'corrections_combined'")
|
|
expect_equal(hit$available_years, c(1967L, 2023L))
|
|
})
|
|
|
|
test_that("per-code gap does NOT fire when a code's only coverage is its own aggregate row (self-coverage is not \"other components\")", {
|
|
skip_if_no_corpus()
|
|
# Broward, FY2011 ONLY (isolating the 2011 half of the query above): E05,
|
|
# F05, and G05 each report SOLELY as a wide-era AGGREGATE row that year
|
|
# (216088, 1453, 270 respectively); E04/F04/G04 -- their modern-only
|
|
# siblings -- don't exist as codes at all before 2012, corpus-wide (zero
|
|
# rows for any government). Each component's own aggregate row would
|
|
# trivially satisfy a same-component "covered" check, but review-doc
|
|
# § 0.3's criterion is explicit that a gap must be covered by "OTHER
|
|
# components", not the gapped component's own aggregate form. With no
|
|
# OTHER component present for any of the three Corrections recipes in
|
|
# 2011, none of them should fire -- this is what the combined
|
|
# 2011-2012 test above actually relies on 2012 (E05 gapped, E04 -- a
|
|
# genuinely different component -- covers) to fire, not 2011.
|
|
r <- cog_spending("121011212191", years = 2011L, category = "Corrections")
|
|
prov <- attr(r, "provenance")
|
|
expect_length(prov$suggestions, 0L)
|
|
})
|
|
|
|
test_that("per-code gap fires even when a sibling code masks the whole-result check (Cleburne County, FY2012)", {
|
|
skip_if_no_corpus()
|
|
# Cleburne County, AL (canonical_govid 011029122489), FY2012: E04 ($854)
|
|
# and E05 ($1) both report ("operations" subtype), and G04 ($14,000,
|
|
# corrections_other_capital_combined's modern-only leg) also reports
|
|
# ("capital" subtype) -- so the WHOLE category result is non-empty for
|
|
# 2012 (2 rows) and the R2 whole-result check would never look further.
|
|
# But G05 -- G04's OWN recipe sibling, the 1967-2023 wide leg -- has
|
|
# ZERO rows at all that year: a genuine, per-code gap the recipe exists
|
|
# to bridge, invisible at the category-result grain because it's masked
|
|
# by G04's own data, let alone the unrelated E04/E05 pair.
|
|
r <- cog_spending("011029122489", years = 2012L, category = "Corrections")
|
|
expect_equal(nrow(r), 2L) # operations + capital rows: a non-empty result
|
|
|
|
prov <- attr(r, "provenance")
|
|
ids <- vapply(prov$suggestions, function(s) s$recipe_id, character(1))
|
|
expect_true("corrections_other_capital_combined" %in% ids)
|
|
hit <- prov$suggestions[[which(ids == "corrections_other_capital_combined")]]
|
|
expect_equal(hit$hint, "re-run with recipe = 'corrections_other_capital_combined'")
|
|
expect_equal(hit$available_years, c(1967L, 2023L))
|
|
|
|
# corrections_combined must NOT fire: E04 AND E05 both have real 2012
|
|
# data for this government, so neither of ITS OWN components is gapped.
|
|
expect_false("corrections_combined" %in% ids)
|
|
})
|
|
|
|
test_that("per-code gap does not fire when no recipe component has any data at all (ordinary reporting variance, not a format-boundary gap)", {
|
|
skip_if_no_corpus()
|
|
# Same government/year as above: F04 and F05 (corrections_capital_combined)
|
|
# are BOTH completely absent -- Cleburne simply never reported capital
|
|
# corrections spending under that code family in 2012, wide-era or
|
|
# modern. The recipe's own generic join (aggregate-inclusive, either
|
|
# component) has nothing to offer either, so this must stay silent --
|
|
# the per-government `covered` guard the header comment describes is
|
|
# unchanged and still does this filtering.
|
|
r <- cog_spending("011029122489", years = 2012L, category = "Corrections")
|
|
prov <- attr(r, "provenance")
|
|
ids <- vapply(prov$suggestions, function(s) s$recipe_id, character(1))
|
|
expect_false("corrections_capital_combined" %in% ids)
|
|
})
|
|
|
|
test_that("per-code gap fires for Broward 2019-2020 even though the category result looks complete", {
|
|
skip_if_no_corpus()
|
|
# Broward reports E04 + G04 (modern leaf codes) in BOTH 2019 and 2020 but
|
|
# never reports E05 or G05 (their own recipe siblings) in either year --
|
|
# a real per-code gap in two of the three Corrections recipes, invisible
|
|
# under the R2 coarse check because the category *result* is non-empty
|
|
# both years (this replaces the old R2-era "full year coverage" test,
|
|
# whose premise -- that a non-empty result implies nothing to signpost --
|
|
# is exactly what this refinement narrows; see data-raw/
|
|
# measure_signposting_rate.R for the measured rate change this causes).
|
|
# corrections_capital_combined correctly stays silent: Broward reports
|
|
# neither F04 nor F05 in 2019 or 2020, so that recipe's own join has
|
|
# nothing to offer either (ordinary non-reporting, not a format-boundary
|
|
# gap) -- the per-government `covered` guard still does its job here too.
|
|
r <- cog_spending("121011212191", years = 2019:2020, category = "Corrections")
|
|
prov <- attr(r, "provenance")
|
|
ids <- vapply(prov$suggestions, function(s) s$recipe_id, character(1))
|
|
expect_true("corrections_combined" %in% ids)
|
|
expect_true("corrections_other_capital_combined" %in% ids)
|
|
expect_false("corrections_capital_combined" %in% ids)
|
|
})
|
|
|
|
test_that("no signposting when every recipe component genuinely has data (true full per-code coverage)", {
|
|
skip_if_no_corpus()
|
|
# Maricopa County, AZ (canonical_govid 041013160815): all six Corrections
|
|
# codes (E04, E05, F04, F05, G04, G05) report real, nonzero, non-aggregate
|
|
# amounts in BOTH 2019 and 2020 -- genuinely nothing for any recipe to
|
|
# fill, even at the finer per-code grain this refinement now checks.
|
|
r <- cog_spending("041013160815", years = 2019:2020, category = "Corrections")
|
|
prov <- attr(r, "provenance")
|
|
expect_length(prov$suggestions, 0L)
|
|
})
|
|
|
|
test_that("no signposting when category is NULL (unscoped query)", {
|
|
skip_if_no_corpus()
|
|
r <- cog_spending("121011212191", years = c(2011L, 2012L))
|
|
prov <- attr(r, "provenance")
|
|
expect_length(prov$suggestions, 0L)
|
|
})
|
|
|
|
test_that("no signposting under basis = 'raw'", {
|
|
skip_if_no_corpus()
|
|
r <- cog_spending("121011212191", years = c(2011L, 2012L),
|
|
category = "Corrections", basis = "raw")
|
|
prov <- attr(r, "provenance")
|
|
expect_length(prov$suggestions, 0L)
|
|
})
|