test: add failing tests for Madison walkthrough findings
Six skipped tests, one per issue opened from the Madison walkthrough audit (docs/walkthroughs/FINDINGS.md in cog_explorer). Each asserts the desired behaviour, so it fails today and goes green when the fix lands; each is guarded by a single skip() naming its issue and finding IDs, so the suite stays green and activating a test is a one-line deletion. test-expenditure-concepts.R #11 F-012, F-017, F-018 test-revenue-concept-insurance-trust.R #12 F-014 test-coverage-disclosure.R #13 F-020, F-023 test-peer-summary-scope.R #14 F-021 test-amount-units-documented.R #15 F-004 test-gov-search-literal-match.R #16 F-025 helper-walkthrough-raw.R reads the corpus's long parquet partitions directly, bypassing uscogdata's SQL views. Every expected amount comes from there rather than from the verb under test - verifying an absence through the filter that creates it proves nothing, which was the most common defect in the audit itself. Verified: with the skips removed all six fail (or error) against the bundled fixture; with them in place the full suite is 576 pass / 0 fail / 6 skip.
This commit is contained in:
@@ -0,0 +1,44 @@
|
||||
# Helper for the Madison-walkthrough finding tests (uscogdata #11-#16).
|
||||
#
|
||||
# Those tests all assert something about what a `cog_*` verb includes or
|
||||
# excludes. The expected amounts must therefore come from the RAW corpus, never
|
||||
# from the verb under test: verifying an absence through the filter that creates
|
||||
# it proves nothing. `wt_raw_*()` opens its own DuckDB connection straight onto
|
||||
# the corpus's `long` parquet partitions, bypassing uscogdata's SQL views (and
|
||||
# therefore its `flow_prefixes` filtering) entirely.
|
||||
|
||||
wt_corpus_glob <- function() {
|
||||
url <- Sys.getenv("USCOGDATA_URL")
|
||||
if (!nzchar(url)) testthat::skip("USCOGDATA_URL is not set")
|
||||
paste0(sub("/$", "", url), "/data/long/**/*.parquet")
|
||||
}
|
||||
|
||||
wt_raw_query <- function(sql) {
|
||||
con <- DBI::dbConnect(duckdb::duckdb())
|
||||
on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE)
|
||||
DBI::dbGetQuery(con, sql)
|
||||
}
|
||||
|
||||
# Sum of `amt` (in $1,000s, as the corpus stores it) for one government-year,
|
||||
# restricted either to an explicit set of item codes or to a set of first-letter
|
||||
# prefixes. Aggregate rows are excluded, matching every published verb.
|
||||
wt_raw_amt <- function(govid, year, codes = NULL, prefixes = NULL) {
|
||||
stopifnot(xor(is.null(codes), is.null(prefixes)))
|
||||
filter_sql <- if (!is.null(codes)) {
|
||||
paste0("item_code IN (", paste0("'", codes, "'", collapse = ", "), ")")
|
||||
} else {
|
||||
paste0("LEFT(item_code, 1) IN (", paste0("'", prefixes, "'", collapse = ", "), ")")
|
||||
}
|
||||
out <- wt_raw_query(paste0(
|
||||
"SELECT COALESCE(SUM(amt), 0) AS amt FROM read_parquet('", wt_corpus_glob(), "') ",
|
||||
"WHERE canonical_govid = '", govid, "' AND year = ", year,
|
||||
" AND NOT is_aggregate AND ", filter_sql
|
||||
))
|
||||
out$amt[[1]]
|
||||
}
|
||||
|
||||
# The item codes a verb reports having summed, flattened out of the
|
||||
# comma-separated `codes_included` column.
|
||||
wt_codes_included <- function(df) {
|
||||
sort(unique(trimws(unlist(strsplit(stats::na.omit(df$codes_included), ",")))))
|
||||
}
|
||||
Reference in New Issue
Block a user