Adds schema_version 5 support alongside the existing v4 corpus:
.validate_schema() now accepts a supported set (4, 5) instead of a single
expected version, and cog_spending()/cog_revenue() gain basis =
c("harmonized", "raw"). Harmonized basis routes to new
spending_annotated_harmonized / revenue_annotated_harmonized views built on
spending_long_harmonized / revenue_long_harmonized (REPLACE(harmonized_code
AS item_code), excluding aggregate and NA-harmonized rows); raw basis is
byte-identical to the pre-Phase-R2 behavior. On a v4 corpus, an unspecified
basis silently resolves to "raw" with a provenance note; an explicit
basis = "harmonized" aborts with an actionable message.
Provenance gains basis, basis_note, and a harmonization block
(applied/na_rows_excluded/na_amount_excluded). The five new schema-v5-only
SQL views (harmonized long/annotated views, harmonization_map,
harmonization_recipes, series_breaks_pq) are registered conditionally on
manifest$schema_version >= 5, since DuckDB's read_parquet() errors eagerly
at CREATE VIEW time when the backing file doesn't exist on a v4 corpus.
Fixture corpus regenerated to schema_version 5 / years 2011, 2012, 2019,
2020 (2011->2012 spans the wide-aggregate -> modern-leaf format boundary
needed for the harmonization/recipe work), with the harmonization_map /
harmonization_recipes / series_breaks parquet tables bundled alongside the
existing metadata registries.
56 lines
2.1 KiB
R
56 lines
2.1 KiB
R
test_that("cog_revenue returns expected shape for Broward Property Tax 2020", {
|
|
skip_if_no_corpus()
|
|
r <- cog_revenue("121011212191", years = 2020L, category = "Property Tax")
|
|
expect_s3_class(r, "tbl_df")
|
|
expected_cols <- c("year", "canonical_govid", "gov_name", "revenue_subtype",
|
|
"category", "amt_nominal", "codes_included",
|
|
"aggregate_fallback", "notes")
|
|
expect_true(all(expected_cols %in% names(r)))
|
|
expect_equal(unique(r$canonical_govid), "121011212191")
|
|
expect_equal(unique(r$year), 2020L)
|
|
})
|
|
|
|
test_that("cog_revenue with no category filter returns multiple categories", {
|
|
skip_if_no_corpus()
|
|
r <- cog_revenue("121011212191", years = 2020L)
|
|
expect_gt(length(unique(r$category)), 1L)
|
|
})
|
|
|
|
test_that("cog_revenue with per_capita + adjust_to_year adds all columns", {
|
|
skip_if_no_corpus()
|
|
r <- cog_revenue("121011212191", 2020L,
|
|
per_capita = TRUE, adjust_to_year = 2022L)
|
|
expect_true(all(c("amt_nominal", "amt_real",
|
|
"amt_per_capita_nominal", "amt_per_capita_real") %in%
|
|
names(r)))
|
|
})
|
|
|
|
test_that("cog_revenue result has provenance attribute", {
|
|
skip_if_no_corpus()
|
|
r <- cog_revenue("121011212191", 2020L)
|
|
prov <- attr(r, "provenance")
|
|
expect_equal(prov$verb, "cog_revenue")
|
|
expect_true(grepl("revenue_annotated", prov$sql_query))
|
|
})
|
|
|
|
test_that("cog_revenue rejects invalid inputs", {
|
|
expect_error(cog_revenue(list(), 2020L), "character|data frame")
|
|
})
|
|
|
|
test_that("cog_revenue basis = 'harmonized' (default) matches 'raw' in this fixture window", {
|
|
skip_if_no_corpus()
|
|
r_raw <- cog_revenue("121011212191", 2019:2020, basis = "raw")
|
|
r_harm <- cog_revenue("121011212191", 2019:2020, basis = "harmonized")
|
|
expect_equal(attr(r_raw, "provenance")$basis, "raw")
|
|
expect_equal(attr(r_harm, "provenance")$basis, "harmonized")
|
|
expect_equal(sum(r_raw$amt_nominal), sum(r_harm$amt_nominal))
|
|
})
|
|
|
|
test_that("cog_revenue provenance carries the harmonization block", {
|
|
skip_if_no_corpus()
|
|
r <- cog_revenue("121011212191", 2020L)
|
|
h <- attr(r, "provenance")$harmonization
|
|
expect_true(h$applied)
|
|
expect_true(h$na_rows_excluded >= 0L)
|
|
})
|