The coarse and per-code signposting checks are partly DISJOINT, not nested: coarse fires on queries per-code does not, so the coarse -> percode move both adds and removes signposting. Every `*_delta_pp` the harness reports is therefore a NET that can mask a coverage loss in either direction. The staged-corpus headline (+1.875 pp, coarse 1/640 -> percode 13/640) sits on top of Corrections losing coverage outright (0.05 -> 0.00, -5 pp). The cause is structural, not sampling: coarse's coverage test is at recipe grain and self-coverage-permissive, while per-code requires a DIFFERENT component of the same recipe. When a whole category is empty in a year -- coarse's own trigger -- and the only covering evidence is the gapped component's own wide-era aggregate row, per-code cannot fire by construction. That case is already pinned as intended behaviour in test-recipes.R; this change measures what it costs, it does not change it. Measurement and disclosure only. R/suggestions.R is untouched -- which arm ships is the human ruling at Checkpoint R3. - header: replace the "noise trade" framing with an explicit statement that the checks are partly disjoint and every delta is a net - .measure_subset_relation(): split the disagreement into violations (coarse fired, per-code silent -- coverage LOST) and additions, returning the offending rows, not just counts. No assertion: the violation set is genuinely non-empty and a stopifnot() would only break the harness that is supposed to surface it - .measure_format_subset_report(): prominent HOLDS / *** VIOLATED *** section naming each offending (category, government, year) - detail gains coarse_gap_years / coarse_recipes / percode_recipes; by_category gains n_coarse_only / n_percode_only so the two netted flows are visible per category - new test-signposting-harness.R pins the reporting, including inversion guards and an end-to-end case (Broward FY2011 Corrections) where coarse fires and per-code does not Tests: 518 PASS / 0 FAIL / 0 WARN / 0 SKIP (was 476). Mutation-checked: inverting the violation direction fails 18 assertions, removing the violation reporting fails 9.
165 lines
7.0 KiB
R
165 lines
7.0 KiB
R
# tests/testthat/test-signposting-harness.R
|
|
#
|
|
# Pins the subset-relation REPORTING in data-raw/measure_signposting_rate.R.
|
|
#
|
|
# Phase R3 Task 19c narrowed signposting from a coarse whole-result gap
|
|
# check to per-code gap detection. Those two checks are partly DISJOINT,
|
|
# not nested: a query can fire under coarse and stay silent under per-code,
|
|
# so the harness's `*_delta_pp` figures are NETS that can hide a coverage
|
|
# loss. `.measure_subset_relation()` is what separates the two flows, and
|
|
# `.measure_format_subset_report()` is what puts the loss in front of a
|
|
# human. Both are load-bearing for the Checkpoint R3 ruling, so both are
|
|
# pinned here: if the violation detection is deleted, inverted, or quietly
|
|
# downgraded to a count with no identities, these tests fail.
|
|
#
|
|
# These tests do NOT assert that the violation set is empty -- it is
|
|
# genuinely non-empty, and asserting otherwise would be pinning a bug as a
|
|
# contract. They assert only that a real violation is DETECTED and NAMED.
|
|
|
|
# The harness lives in data-raw/, which is .Rbuildignore'd, so it is absent
|
|
# from an installed/checked tarball. Source it into an env parented on the
|
|
# namespace so it resolves the package internals it calls (.build_verb_sql,
|
|
# .build_suggestions) exactly as it does when run for real.
|
|
harness_env <- function() {
|
|
path <- testthat::test_path("..", "..", "data-raw", "measure_signposting_rate.R")
|
|
skip_if_not(file.exists(path),
|
|
"data-raw/ is .Rbuildignore'd; harness not present in this tree")
|
|
env <- new.env(parent = asNamespace("uscogdata"))
|
|
source(path, local = env)
|
|
env
|
|
}
|
|
|
|
# A detail frame in exactly the shape .measure_one_query() emits, covering
|
|
# all four quadrants of the coarse x percode cross-tab.
|
|
fake_detail <- function() {
|
|
data.frame(
|
|
category = c("Corrections", "Other Taxes", "Police", "Fire"),
|
|
category_type = c("expenditure", "revenue", "expenditure", "expenditure"),
|
|
canonical_govid = c("121011212191", "472155175824", "011029122489",
|
|
"041013160815"),
|
|
n_result_rows = c(0L, 1L, 2L, 6L),
|
|
coarse_gap_years = c("2011", "", "", ""),
|
|
fired_coarse = c(TRUE, FALSE, TRUE, FALSE),
|
|
fired_percode = c(FALSE, TRUE, TRUE, FALSE),
|
|
coarse_recipes = c("corrections_combined", "", "police_combined", ""),
|
|
percode_recipes = c("", "t29_license_wide", "police_combined", ""),
|
|
stringsAsFactors = FALSE
|
|
)
|
|
}
|
|
|
|
test_that(".measure_subset_relation() separates coverage LOST from coverage ADDED", {
|
|
e <- harness_env()
|
|
rel <- e$.measure_subset_relation(fake_detail())
|
|
|
|
# Row 1 (coarse fired, per-code silent) is the violation; row 2 is the
|
|
# addition; row 3 agrees; row 4 is silent.
|
|
expect_false(rel$holds)
|
|
expect_equal(rel$n_violations, 1L)
|
|
expect_equal(rel$n_additions, 1L)
|
|
expect_equal(rel$n_coarse_fired, 2L)
|
|
expect_equal(rel$n_percode_fired, 2L)
|
|
expect_equal(rel$n_both, 1L)
|
|
expect_equal(rel$n_queries, 4L)
|
|
|
|
# The violation must be NAMED down to (category, government, year), not
|
|
# merely counted -- that is what makes it inspectable at Checkpoint R3.
|
|
expect_equal(rel$violations$category, "Corrections")
|
|
expect_equal(rel$violations$canonical_govid, "121011212191")
|
|
expect_equal(rel$violations$coarse_gap_years, "2011")
|
|
expect_equal(rel$violations$coarse_recipes, "corrections_combined")
|
|
|
|
# Inversion guard: an implementation that swapped the two directions
|
|
# would report the addition as a violation and vice versa.
|
|
expect_false("Other Taxes" %in% rel$violations$category)
|
|
expect_equal(rel$additions$category, "Other Taxes")
|
|
expect_false("Corrections" %in% rel$additions$category)
|
|
|
|
# Agreeing and silent queries belong to neither set.
|
|
expect_false("Police" %in% c(rel$violations$category, rel$additions$category))
|
|
expect_false("Fire" %in% c(rel$violations$category, rel$additions$category))
|
|
})
|
|
|
|
test_that(".measure_subset_relation() reports holds = TRUE only when nothing fires coarse-only", {
|
|
e <- harness_env()
|
|
# Drop the violating row: coarse is now genuinely a subset of per-code.
|
|
clean <- fake_detail()[-1L, , drop = FALSE]
|
|
rel <- e$.measure_subset_relation(clean)
|
|
|
|
expect_true(rel$holds)
|
|
expect_equal(rel$n_violations, 0L)
|
|
expect_equal(nrow(rel$violations), 0L)
|
|
expect_equal(rel$n_additions, 1L)
|
|
})
|
|
|
|
test_that(".measure_subset_relation() validates its input rather than silently mis-reporting", {
|
|
e <- harness_env()
|
|
expect_error(e$.measure_subset_relation("not a data frame"), "must be a data frame")
|
|
expect_error(e$.measure_subset_relation(fake_detail()[, c("category", "canonical_govid")]),
|
|
"fired_coarse")
|
|
bad <- fake_detail()
|
|
bad$fired_percode[1] <- NA
|
|
expect_error(e$.measure_subset_relation(bad), "non-NA logicals")
|
|
})
|
|
|
|
test_that("the subset report NAMES a coarse-only firing as a violation", {
|
|
e <- harness_env()
|
|
txt <- paste(e$.measure_format_subset_report(e$.measure_subset_relation(fake_detail())),
|
|
collapse = "\n")
|
|
|
|
# Stated plainly as a violation, not buried.
|
|
expect_match(txt, "VIOLATED")
|
|
expect_match(txt, "COVERAGE LOST")
|
|
expect_no_match(txt, "HOLDS")
|
|
# ...and the offending query named, so a human can go look at it.
|
|
expect_match(txt, "Corrections")
|
|
expect_match(txt, "121011212191")
|
|
expect_match(txt, "2011")
|
|
# ...and the delta explicitly flagged as a net of both directions.
|
|
expect_match(txt, "NET")
|
|
})
|
|
|
|
test_that("the subset report says HOLDS when coarse really is a subset", {
|
|
e <- harness_env()
|
|
rel <- e$.measure_subset_relation(fake_detail()[-1L, , drop = FALSE])
|
|
txt <- paste(e$.measure_format_subset_report(rel), collapse = "\n")
|
|
|
|
expect_match(txt, "HOLDS")
|
|
expect_no_match(txt, "VIOLATED")
|
|
expect_no_match(txt, "COVERAGE LOST")
|
|
})
|
|
|
|
test_that("a REAL coarse-fires/per-code-silent query is measured and reported as a violation", {
|
|
skip_if_no_corpus()
|
|
# Broward County FY2011, Corrections: E05/F05/G05 report SOLELY as
|
|
# wide-era aggregate rows, which basis = "harmonized" excludes, so the
|
|
# whole category result is empty -- coarse's trigger. Their modern-only
|
|
# siblings E04/F04/G04 do not exist as codes at all before 2012, so no
|
|
# OTHER component can supply per-code's covering evidence and per-code
|
|
# is structurally unable to fire. This is the disjointness the harness
|
|
# exists to surface, measured end-to-end through the real git-loaded
|
|
# coarse arm and the live per-code arm (not a hand-built frame).
|
|
e <- harness_env()
|
|
con <- uscogdata:::cog_open()
|
|
row <- e$.measure_one_query(
|
|
con,
|
|
coarse_env = e$.measure_load_git_impl("b0df1ec"),
|
|
selfcov_env = e$.measure_load_git_impl("da72bf3"),
|
|
category = "Corrections", category_type = "expenditure",
|
|
govid = "121011212191", years = 2011L
|
|
)
|
|
|
|
expect_true(row$fired_coarse)
|
|
expect_false(row$fired_percode)
|
|
expect_equal(row$n_result_rows, 0L)
|
|
expect_equal(row$coarse_gap_years, "2011")
|
|
|
|
rel <- e$.measure_subset_relation(row)
|
|
expect_false(rel$holds)
|
|
expect_equal(rel$n_violations, 1L)
|
|
expect_equal(rel$violations$canonical_govid, "121011212191")
|
|
|
|
txt <- paste(e$.measure_format_subset_report(rel), collapse = "\n")
|
|
expect_match(txt, "VIOLATED")
|
|
expect_match(txt, "121011212191")
|
|
})
|