C1: spending_long/spending_long_harmonized filter NOT is_aggregate but
ig_long deliberately doesn't (legacy IG lives on aggregate rows), so a
legacy aggregate-only family (e.g. Corrections pre-2012) can survive on
the IG leg while Direct is suppressed. expenditure_concept = "total"
then UNIONs an IG-only figure that reads as a plausible Total, and the
coverage-gap suggestion machinery -- fed the UNION'd result -- saw the
surviving IG row as coverage and stayed silent.
(a) .build_suggestions() is now fed a Direct-leg-only view of the
result (IG rows filtered out before the gap-years computation),
so the recipe hints fire for "total" exactly as they do for
"direct".
(b) Any row where IG has dollars but Direct has none for the same
(year, canonical_govid, category) is now flagged: the row's
`notes` name the recovering recipe (drawn from the Direct-leg
suggestions), and provenance gains an explicit
`expenditure_concept_direct_suppressed` boolean plus an appended
warning on `expenditure_concept_note` -- both cheap for a
downstream consumer (cog-api passes provenance through verbatim)
to test, rather than silently asserting Direct + IG when that
arithmetic didn't happen.
Measured before/after on AL state government, Corrections, 2011:
"total" already correctly returns the corpus's actual IG-only figure
($31,358,000, vs. true Direct of $521,651,000 via recipe =
"corrections_combined"), but before this fix it did so with 0
suggestions and an unqualified "Total = Direct + IG" note; after, it
fires 3 recipe hints and both the row notes and provenance say plainly
that Direct is unavailable through this basis.
C2: the 66 M/L summary_categories rows arrived via cog_pipeline PR #59
with no schema_version bump, so schema_version can't gate "total" --
a pre-#59 corpus can report any supported schema_version and still
have zero M/L category rows, in which case ig_annotated's LEFT JOIN
silently produces NA category/spend_subtype (0 rows for a specific
category, or one invisible NA-subtype group for category = NULL). New
.require_ig_categories() checks summary_categories directly and aborts
with class uscogdata_ig_categories_unsupported, naming PR #59 and
directing the user to a newer corpus.
Reconciles tests/testthat/test-views.R's v4-shaped-corpus test (whose
synthetic summary_categories carries only one E36 row) by asserting
the new guard fires against that same connection, rather than leaving
the two silently contradictory.
94 lines
4.0 KiB
R
94 lines
4.0 KiB
R
# tests/testthat/helper-fixture.R
|
|
|
|
# Returns the path to the bundled fixture corpus (trailing slash for DuckDB globs).
|
|
fixture_corpus_path <- function() {
|
|
p <- system.file("extdata/fixture_corpus", package = "uscogdata")
|
|
if (nzchar(p)) paste0(p, "/") else ""
|
|
}
|
|
|
|
# Skip a test if no corpus is reachable (bundled fixture or explicit remote URL).
|
|
skip_if_no_corpus <- function() {
|
|
p <- fixture_corpus_path()
|
|
has_fixture <- nzchar(p) && file.exists(sub("/$", "/manifest.json", p))
|
|
has_remote <- nzchar(Sys.getenv("USCOGDATA_FIXTURE_URL", ""))
|
|
testthat::skip_if(!has_fixture && !has_remote, "No fixture corpus available")
|
|
}
|
|
|
|
# Run a block against the fixture corpus with a clean session.
|
|
# Restores the previous URL and closes the DuckDB connection when done.
|
|
with_fixture_corpus <- function(code) {
|
|
old_url <- Sys.getenv("USCOGDATA_URL", unset = NA)
|
|
uscogdata:::cog_close()
|
|
Sys.setenv(USCOGDATA_URL = fixture_corpus_path())
|
|
on.exit({
|
|
uscogdata:::cog_close()
|
|
if (is.na(old_url)) Sys.unsetenv("USCOGDATA_URL") else Sys.setenv(USCOGDATA_URL = old_url)
|
|
}, add = TRUE)
|
|
force(code)
|
|
}
|
|
|
|
# Copy the bundled fixture to a temp dir with manifest.json's schema_version
|
|
# patched to `version`, then run `code` against it with a clean session
|
|
# (mirrors with_fixture_corpus()). Used to exercise the v4/v5 dual-accept
|
|
# path without a second physical fixture tree: a real v4 corpus has no
|
|
# harmonization_map/harmonization_recipes/series_breaks parquet files, but
|
|
# .register_views() only *reads* those when schema_version >= 5 (see
|
|
# R/views.R), so a doctored copy of the (v5) bundled fixture with the
|
|
# manifest's schema_version knocked down to 4 is a faithful stand-in.
|
|
with_doctored_schema_version <- function(version, code) {
|
|
src <- fixture_corpus_path()
|
|
tmp <- withr::local_tempdir(.local_envir = parent.frame())
|
|
file.copy(list.files(src, full.names = TRUE), tmp, recursive = TRUE)
|
|
|
|
manifest_path <- file.path(tmp, "manifest.json")
|
|
m <- jsonlite::fromJSON(manifest_path, simplifyVector = FALSE)
|
|
m$schema_version <- as.integer(version)
|
|
writeLines(
|
|
jsonlite::toJSON(m, auto_unbox = TRUE, pretty = TRUE, null = "null"),
|
|
manifest_path
|
|
)
|
|
|
|
old_url <- Sys.getenv("USCOGDATA_URL", unset = NA)
|
|
uscogdata:::cog_close()
|
|
Sys.setenv(USCOGDATA_URL = paste0(tmp, "/"))
|
|
on.exit({
|
|
uscogdata:::cog_close()
|
|
if (is.na(old_url)) Sys.unsetenv("USCOGDATA_URL") else Sys.setenv(USCOGDATA_URL = old_url)
|
|
}, add = TRUE)
|
|
force(code)
|
|
}
|
|
|
|
# Copy the bundled fixture to a temp dir with summary_categories.parquet
|
|
# rewritten to drop every M/L (intergovernmental) row, then run `code`
|
|
# against it with a clean session (mirrors with_fixture_corpus()/
|
|
# with_doctored_schema_version()). Models a real pre-cog_pipeline-PR#59
|
|
# corpus: the 66 M/L category rows shipped with NO schema_version bump (see
|
|
# C2 in the expenditure-concept review), so schema_version is left
|
|
# untouched here -- only the category data itself is rolled back.
|
|
with_corpus_missing_ig_categories <- function(code) {
|
|
src <- fixture_corpus_path()
|
|
tmp <- withr::local_tempdir(.local_envir = parent.frame())
|
|
file.copy(list.files(src, full.names = TRUE), tmp, recursive = TRUE)
|
|
|
|
cats_path <- file.path(tmp, "data", "summary_categories.parquet")
|
|
filtered_path <- file.path(tmp, "data", "summary_categories_filtered.parquet")
|
|
write_con <- DBI::dbConnect(duckdb::duckdb())
|
|
on.exit(DBI::dbDisconnect(write_con, shutdown = TRUE), add = TRUE)
|
|
DBI::dbExecute(write_con, sprintf(
|
|
"COPY (SELECT * FROM read_parquet(%s) WHERE LEFT(item_code, 1) NOT IN ('M', 'L'))
|
|
TO %s (FORMAT PARQUET)",
|
|
uscogdata:::.sql_lit_chr(cats_path), uscogdata:::.sql_lit_chr(filtered_path)
|
|
))
|
|
file.remove(cats_path)
|
|
file.rename(filtered_path, cats_path)
|
|
|
|
old_url <- Sys.getenv("USCOGDATA_URL", unset = NA)
|
|
uscogdata:::cog_close()
|
|
Sys.setenv(USCOGDATA_URL = paste0(tmp, "/"))
|
|
on.exit({
|
|
uscogdata:::cog_close()
|
|
if (is.na(old_url)) Sys.unsetenv("USCOGDATA_URL") else Sys.setenv(USCOGDATA_URL = old_url)
|
|
}, add = TRUE)
|
|
force(code)
|
|
}
|