fix: exercise real harmonized-view SQL in tests; unambiguous recipe provenance
test-views.R's harmonized-view test previously ran a hand-rolled REPLACE
query with no WHERE clause, so a regression in any of
inst/sql/22-spending_long_harmonized.sql / 23-revenue_long_harmonized.sql's
three predicates (NOT is_aggregate, harmonized_code IS NOT NULL, the
E/F/G/K or T/A/U/B/C/D prefix filter) would go uncaught. Replaced it with a
test that reads the real SQL files off disk, substitutes {url} exactly as
.register_views() does, and executes them (plus their 10-long.sql
dependency) against a synthetic hive-partitioned parquet tree written via
DuckDB's own COPY ... TO (FORMAT PARQUET) (no arrow dependency, matching
this package's existing convention). Ten rows are crafted so each predicate
is independently falsifiable by a specific row; manually broke each
predicate in turn to confirm the test fails exactly as expected, then
restored the SQL files (see the task report for the RED-phase transcript).
Also fixes a provenance ambiguity: a recipe= query bypasses
spending_annotated(_harmonized)/revenue_annotated(_harmonized) entirely
(.run_recipe() joins `long` directly), so basis= has no effect on it, but
provenance was still reporting basis = "harmonized"/"raw" (whatever the
argument resolved to) with harmonization$applied = FALSE alongside it --
misleading, since it looks like harmonization was evaluated and found
nothing to exclude rather than "not applicable here." Recipe results now
report basis = "recipe" with an inert harmonization block carrying an
explicit note, regardless of what basis= was passed.
This commit is contained in:
@@ -69,7 +69,7 @@ test_that("recipe result carries a recipe provenance block with component rows",
|
||||
r <- cog_spending("121011212191", years = c(2011L, 2012L),
|
||||
recipe = "corrections_combined")
|
||||
prov <- attr(r, "provenance")
|
||||
expect_equal(prov$basis, "harmonized")
|
||||
expect_equal(prov$basis, "recipe")
|
||||
expect_equal(prov$category, "Corrections (functions 04+05 combined)")
|
||||
expect_type(prov$recipe, "list")
|
||||
expect_equal(prov$recipe$recipe_id, "corrections_combined")
|
||||
@@ -82,6 +82,34 @@ test_that("recipe result carries a recipe provenance block with component rows",
|
||||
expect_length(prov$suggestions, 0L)
|
||||
})
|
||||
|
||||
test_that("recipe results report an unambiguous basis/harmonization, ignoring basis=", {
|
||||
skip_if_no_corpus()
|
||||
# A recipe query bypasses spending_annotated(_harmonized) entirely --
|
||||
# .run_recipe() joins `long` directly -- so `basis` must never read
|
||||
# "harmonized"/"raw" (which would describe a code path this query never
|
||||
# took) regardless of what the caller passed for `basis`. Task 12
|
||||
# consumes provenance verbatim, so this needs to be unambiguous.
|
||||
r_default <- cog_spending("121011212191", years = c(2011L, 2012L),
|
||||
recipe = "corrections_combined")
|
||||
r_raw <- cog_spending("121011212191", years = c(2011L, 2012L),
|
||||
recipe = "corrections_combined", basis = "raw")
|
||||
r_harm <- cog_spending("121011212191", years = c(2011L, 2012L),
|
||||
recipe = "corrections_combined", basis = "harmonized")
|
||||
|
||||
for (r in list(r_default, r_raw, r_harm)) {
|
||||
prov <- attr(r, "provenance")
|
||||
expect_equal(prov$basis, "recipe")
|
||||
expect_true(is.na(prov$basis_note))
|
||||
expect_false(prov$harmonization$applied)
|
||||
expect_equal(prov$harmonization$na_rows_excluded, 0L)
|
||||
expect_match(prov$harmonization$note, "recipe", ignore.case = TRUE)
|
||||
}
|
||||
|
||||
# basis= truly has zero effect on a recipe query's actual numbers.
|
||||
expect_equal(r_raw$amt_nominal, r_harm$amt_nominal)
|
||||
expect_equal(r_default$amt_nominal, r_raw$amt_nominal)
|
||||
})
|
||||
|
||||
test_that("recipe = 't19_selective_sales_wide' sums the local T11/T14 legs when present", {
|
||||
skip_if_no_corpus()
|
||||
# Westminster City, CA (canonical_govid 082001211654): T11 = 0 in 2011,
|
||||
|
||||
Reference in New Issue
Block a user