diff --git a/NEWS.md b/NEWS.md index 2f171b0..6def2c7 100644 --- a/NEWS.md +++ b/NEWS.md @@ -1,5 +1,31 @@ # uscogdata 0.1.0 (development) +## Bundled fixture regenerated against the sparsified corpus + +* `inst/extdata/fixture_corpus/` now tracks the corpus published on + 2026-07-29 (`pipeline_commit 83f9715`, schema v6). The wide era no longer + stores explicit zeros: FY2011 fell from 2,864,212 rows to 496,004, of + which none are `$0`. **Absence now means two different things** — in a + `dense_source` year (≤ FY2011) an absent cell means Census published `$0`; + in a `sparse_source` year (≥ FY2012) it means not reported. The corpus + carries that rule in two new tables the fixture now ships, + `representation.parquet` and `code_set.parquet`, alongside + `census_collection_coverage.parquet` and `lineage_events.parquet` + (all ten publish-tree metadata tables, up from six). Catalogued upstream + as series break `SB194`. +* `cog_categories()` gains an `assistance` spending subtype: the J-prefix + aid/benefit codes (`J19`, `J67`, `J68`, `J85`) are categorised now that + the upstream crosswalk covers every flow code carrying dollars. +* Two consequences worth knowing about, both visible in provenance rather + than in returned dollars. The harmonization block's `na_rows_excluded` + counts only rows that exist, so wide-era codes that were zero-padded no + longer appear there. Coverage-gap `suggestions` are presence-based for the + same reason, so a recipe whose component codes were all `$0` for a given + government-year is no longer suggested for it. +* `tests/testthat/test-fixture-vintage.R` pins these structural facts, so a + fixture left behind by a future publish fails loudly instead of letting the + suite pass against a corpus that no longer exists. + ## Breaking: corpus schema_version 4 (Phase P canonical ids) * The package now requires corpus `schema_version = 4` (`MinCorpusSchema` / diff --git a/data-raw/regenerate_fixture_corpus.R b/data-raw/regenerate_fixture_corpus.R index 086b633..417427b 100644 --- a/data-raw/regenerate_fixture_corpus.R +++ b/data-raw/regenerate_fixture_corpus.R @@ -11,12 +11,14 @@ # Each partition is a full year (all states/govs) as published, so # Broward County FL and every other previously-pinned government stay # covered without any per-gov slicing logic. -# 2. Copies the full canonical_fips_xwalk.parquet, canonical_alias.parquet, -# summary_categories.parquet, harmonization_map.parquet, -# harmonization_recipes.parquet, and series_breaks.parquet metadata -# tables as-is (these are small cross-vintage registries, not -# partitioned by year, so the fixture ships the complete tables rather -# than a year-scoped subset). +# 2. Copies every metadata parquet the publish tree ships (see +# .FIXTURE_METADATA_FILES) as-is. These are small cross-vintage +# registries, not partitioned by year, so the fixture ships the complete +# tables rather than a year-scoped subset. representation.parquet and +# code_set.parquet are what make the sparse wide era interpretable -- +# absence means "$0" in a dense_source year and "not reported" in a +# sparse_source one -- so a fixture without them cannot represent the +# published corpus. # 3. Resyncs the four reference docs (data_dictionary.md, # reader-specification.md, README.md, series_breaks.md) from the # publish tree's docs/. @@ -38,6 +40,22 @@ # source("data-raw/regenerate_fixture_corpus.R") # regenerate_fixture_corpus(publish_cache_dir = "/path/to/publish_cache") +# Every metadata parquet the publish tree ships, in the order they appear in +# the corpus manifest. Single source of truth for both the copy step and the +# fixture manifest, so the two can never drift apart. +.FIXTURE_METADATA_FILES <- c( + "canonical_alias.parquet", + "canonical_fips_xwalk.parquet", + "census_collection_coverage.parquet", + "code_set.parquet", + "harmonization_map.parquet", + "harmonization_recipes.parquet", + "lineage_events.parquet", + "representation.parquet", + "series_breaks.parquet", + "summary_categories.parquet" +) + regenerate_fixture_corpus <- function( publish_cache_dir = file.path( "..", "cog_pipeline", "_targets", "publish_cache" @@ -100,20 +118,11 @@ regenerate_fixture_corpus <- function( invisible(NULL) } -# Copy the full (not year-scoped) canonical_fips_xwalk, canonical_alias, -# summary_categories, and (schema v5+) harmonization_map/ -# harmonization_recipes/series_breaks parquet tables. +# Copy the full (not year-scoped) metadata tables listed in +# .FIXTURE_METADATA_FILES. #' @noRd .copy_metadata_parquets <- function(publish_cache_dir, fixture_dir) { - files <- c( - "canonical_fips_xwalk.parquet", - "canonical_alias.parquet", - "summary_categories.parquet", - "harmonization_map.parquet", - "harmonization_recipes.parquet", - "series_breaks.parquet" - ) - for (f in files) { + for (f in .FIXTURE_METADATA_FILES) { src <- file.path(publish_cache_dir, "data", f) dst <- file.path(fixture_dir, "data", f) if (!file.exists(src)) { @@ -179,15 +188,7 @@ regenerate_fixture_corpus <- function( ) }) - metadata_files <- c( - "canonical_alias.parquet", - "canonical_fips_xwalk.parquet", - "summary_categories.parquet", - "harmonization_map.parquet", - "harmonization_recipes.parquet", - "series_breaks.parquet" - ) - metadata <- lapply(metadata_files, function(f) { + metadata <- lapply(.FIXTURE_METADATA_FILES, function(f) { rel <- file.path("data", f) path <- file.path(fixture_dir, rel) list( @@ -203,13 +204,16 @@ regenerate_fixture_corpus <- function( pipeline_commit = source_manifest$pipeline_commit, fixture_note = paste( "Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full", - "corpus available via USCOGDATA_URL. Regenerated for Phase R2", - "(schema_version 5, harmonization_map/harmonization_recipes/", - "series_breaks parquet tables added). 2011/2012 straddle the", - "wide-aggregate -> modern-leaf format boundary exercised by basis=", - "\"harmonized\" and recipe= queries; 2019/2020 retain the prior", - "per-capita/CPI regression anchors. Full canonical_fips_xwalk master", - "and canonical_alias lookup table included via", + "corpus available via USCOGDATA_URL. Regenerated from the sparsified", + "schema-v6 corpus: the wide era (<= FY2011) no longer stores explicit", + "zeros, so FY2011 absence means Census published $0 while FY2012+", + "absence means not reported. representation.parquet and", + "code_set.parquet carry that rule and ship in full, as do every other", + "metadata table in the publish tree. 2011/2012 straddle both the", + "wide-aggregate -> modern-leaf format boundary (exercised by", + "basis=\"harmonized\" and recipe= queries) and the dense -> sparse", + "representation boundary (SB194); 2019/2020 retain the prior", + "per-capita/CPI regression anchors. Regenerated via", "data-raw/regenerate_fixture_corpus.R." ), data_vintage = source_manifest$data_vintage, diff --git a/inst/extdata/fixture_corpus/data/census_collection_coverage.parquet b/inst/extdata/fixture_corpus/data/census_collection_coverage.parquet new file mode 100644 index 0000000..e878d32 Binary files /dev/null and b/inst/extdata/fixture_corpus/data/census_collection_coverage.parquet differ diff --git a/inst/extdata/fixture_corpus/data/code_set.parquet b/inst/extdata/fixture_corpus/data/code_set.parquet new file mode 100644 index 0000000..487edda Binary files /dev/null and b/inst/extdata/fixture_corpus/data/code_set.parquet differ diff --git a/inst/extdata/fixture_corpus/data/lineage_events.parquet b/inst/extdata/fixture_corpus/data/lineage_events.parquet new file mode 100644 index 0000000..0a25d95 Binary files /dev/null and b/inst/extdata/fixture_corpus/data/lineage_events.parquet differ diff --git a/inst/extdata/fixture_corpus/data/long/year=2011/part-0.parquet b/inst/extdata/fixture_corpus/data/long/year=2011/part-0.parquet index 79db86c..7bbb8ae 100644 Binary files a/inst/extdata/fixture_corpus/data/long/year=2011/part-0.parquet and b/inst/extdata/fixture_corpus/data/long/year=2011/part-0.parquet differ diff --git a/inst/extdata/fixture_corpus/data/representation.parquet b/inst/extdata/fixture_corpus/data/representation.parquet new file mode 100644 index 0000000..0350fb5 Binary files /dev/null and b/inst/extdata/fixture_corpus/data/representation.parquet differ diff --git a/inst/extdata/fixture_corpus/data/series_breaks.parquet b/inst/extdata/fixture_corpus/data/series_breaks.parquet index 65c21bd..0c5e42e 100644 Binary files a/inst/extdata/fixture_corpus/data/series_breaks.parquet and b/inst/extdata/fixture_corpus/data/series_breaks.parquet differ diff --git a/inst/extdata/fixture_corpus/data/summary_categories.parquet b/inst/extdata/fixture_corpus/data/summary_categories.parquet index d87566b..a55d04d 100644 Binary files a/inst/extdata/fixture_corpus/data/summary_categories.parquet and b/inst/extdata/fixture_corpus/data/summary_categories.parquet differ diff --git a/inst/extdata/fixture_corpus/manifest.json b/inst/extdata/fixture_corpus/manifest.json index d86af01..90f5d17 100644 --- a/inst/extdata/fixture_corpus/manifest.json +++ b/inst/extdata/fixture_corpus/manifest.json @@ -1,8 +1,8 @@ { "schema_version": 6, - "built_at": "2026-07-27T13:04:05Z", - "pipeline_commit": "6098baf", - "fixture_note": "Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated for Phase R2 (schema_version 5, harmonization_map/harmonization_recipes/ series_breaks parquet tables added). 2011/2012 straddle the wide-aggregate -> modern-leaf format boundary exercised by basis= \"harmonized\" and recipe= queries; 2019/2020 retain the prior per-capita/CPI regression anchors. Full canonical_fips_xwalk master and canonical_alias lookup table included via data-raw/regenerate_fixture_corpus.R.", + "built_at": "2026-07-30T14:07:36Z", + "pipeline_commit": "83f9715", + "fixture_note": "Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated from the sparsified schema-v6 corpus: the wide era (<= FY2011) no longer stores explicit zeros, so FY2011 absence means Census published $0 while FY2012+ absence means not reported. representation.parquet and code_set.parquet carry that rule and ship in full, as do every other metadata table in the publish tree. 2011/2012 straddle both the wide-aggregate -> modern-leaf format boundary (exercised by basis=\"harmonized\" and recipe= queries) and the dense -> sparse representation boundary (SB194); 2019/2020 retain the prior per-capita/CPI regression anchors. Regenerated via data-raw/regenerate_fixture_corpus.R.", "data_vintage": { "source_vintages": { "2012": "10162019", @@ -36,9 +36,9 @@ { "year": 2011, "path": "data/long/year=2011/part-0.parquet", - "sha256": "84302ab364dc9fc3b3fbbc3c3f8b826e3508b4d73ff7c42d094d3863cd1e37b5", - "row_count": 2864212, - "size_bytes": 3845911 + "sha256": "7848e18497080c8980a4f89c5b386205b2c5bc90db6773827ea01ab3943d16b1", + "row_count": 496004, + "size_bytes": 2202455 }, { "year": 2012, @@ -74,9 +74,14 @@ "description": "canonical_fips_xwalk.parquet" }, { - "path": "data/summary_categories.parquet", - "sha256": "0985b607f3f35a8dff62c0561261ab6922423b81d11c07b03bcb3e3461f85e33", - "description": "summary_categories.parquet" + "path": "data/census_collection_coverage.parquet", + "sha256": "143e025616cde684da7c4442bc00d07fbd1556fabb0ea96223931b737e5d10a4", + "description": "census_collection_coverage.parquet" + }, + { + "path": "data/code_set.parquet", + "sha256": "4cffcb0198dd51e4ff2b694050bb371a5f9965cdac12f25521cb628fb8e118a9", + "description": "code_set.parquet" }, { "path": "data/harmonization_map.parquet", @@ -88,10 +93,25 @@ "sha256": "1133e9a0b02f8f34f5f936e55c5ecd596bb8a55d8425dcce76767f0f3203581c", "description": "harmonization_recipes.parquet" }, + { + "path": "data/lineage_events.parquet", + "sha256": "36c16acfbe621d61010984767f1c566993b8a5f481a2c1e134c4c0a600e4502f", + "description": "lineage_events.parquet" + }, + { + "path": "data/representation.parquet", + "sha256": "31ec328a7dd505a321b45f97aafff12e53d68a1a986f63509863035b22a4360d", + "description": "representation.parquet" + }, { "path": "data/series_breaks.parquet", - "sha256": "b0b6794b6887a4f300079adfa10029c2a77109faa4952fbff1c5a270793cc02b", + "sha256": "5ae050dd7a76c4d25e5f99e7c2e81c1896482e3504e0443b47ab5d78ba148953", "description": "series_breaks.parquet" + }, + { + "path": "data/summary_categories.parquet", + "sha256": "e71d6d70d767c26c983fe56213baf204355f879582aa94841e62d9aea1877f83", + "description": "summary_categories.parquet" } ] }, diff --git a/tests/testthat/test-categories.R b/tests/testthat/test-categories.R index 74b6aa1..d9f8574 100644 --- a/tests/testthat/test-categories.R +++ b/tests/testthat/test-categories.R @@ -15,7 +15,11 @@ test_that("cog_categories(type = 'spending') returns only expenditure rows", { skip_if_no_corpus() r <- cog_categories(type = "spending") expect_true(all(r$category_type == "expenditure")) - expect_true(all(r$subtype %in% c("operations", "capital", "intergovernmental"))) + # "assistance" (the J-prefix aid/benefit codes) joined the vocabulary with + # the crosswalk completion in cog_pipeline#60/#65 -- every flow code + # carrying dollars now maps to a category. + expect_true(all(r$subtype %in% + c("operations", "capital", "intergovernmental", "assistance"))) }) test_that("cog_categories surfaces the intergovernmental spending subtype", { diff --git a/tests/testthat/test-expenditure-concept.R b/tests/testthat/test-expenditure-concept.R index 95954ff..4151921 100644 --- a/tests/testthat/test-expenditure-concept.R +++ b/tests/testthat/test-expenditure-concept.R @@ -298,8 +298,16 @@ test_that("a mis-scoped cog_spending() call never attaches an M/L counterpart to # (M47/M94, same suffixes) -- a coincidence of reused digits, not a real # Direct/Total pairing. The flow-family gate in # .attach_ig_counterparts() must keep ig_recipe_id NULL here. + # + # Anchored on FL state government, not AL. Coverage is presence-based: a + # recipe is only suggested when its component codes have rows for the + # requested government-year. AL state's only FY2011 B47 cell was an + # explicit zero, which the corpus no longer stores after sparsification + # (SB194, cog_pipeline#64), so the recipe stopped being a candidate there. + # FL state carries a real FY2011 B47 amount, so this exercises the guard + # against a suggestion that genuinely fires. r <- suppressMessages( - cog_spending("010000226085", years = c(2005, 2011), category = "IG Federal") + cog_spending("120000226351", years = c(2005, 2011), category = "IG Federal") ) sugg <- attr(r, "provenance")$suggestions expect_gt(length(sugg), 0L) diff --git a/tests/testthat/test-fixture-vintage.R b/tests/testthat/test-fixture-vintage.R new file mode 100644 index 0000000..092d0bf --- /dev/null +++ b/tests/testthat/test-fixture-vintage.R @@ -0,0 +1,112 @@ +# tests/testthat/test-fixture-vintage.R +# +# The bundled fixture is a slice of a real cog_pipeline publish tree, and +# every test in this package -- plus the whole cog-api suite -- runs against +# it. When the published corpus changes shape and the fixture does not, both +# suites stay green against a corpus that no longer exists (uscogdata#18). +# +# These tests pin the structural facts that distinguish the current published +# vintage from its predecessor, so a stale fixture fails loudly instead of +# passing quietly. They assert shape, never dollar values: re-running +# data-raw/regenerate_fixture_corpus.R against a newer publish tree should +# keep them green. + +# Open a bare DuckDB connection on the fixture's parquet files. Deliberately +# not the package session: these assertions are about what the fixture +# CONTAINS, and routing them through the reader's own views would let a +# filter hide the very absence being checked. +fixture_query <- function(sql, ...) { + con <- DBI::dbConnect(duckdb::duckdb()) + on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE) + path <- function(rel) { + sprintf("read_parquet(%s)", + DBI::dbQuoteString(con, file.path(fixture_corpus_path(), rel))) + } + DBI::dbGetQuery(con, do.call(sprintf, c(list(sql), lapply(c(...), path)))) +} + +test_that("fixture ships every metadata table the publish tree does", { + skip_if_no_corpus() + # representation/code_set are what make a sparse corpus interpretable; a + # fixture without them predates sparsification (cog_pipeline#64). + expected <- c( + "canonical_alias.parquet", "canonical_fips_xwalk.parquet", + "census_collection_coverage.parquet", "code_set.parquet", + "harmonization_map.parquet", "harmonization_recipes.parquet", + "lineage_events.parquet", "representation.parquet", + "series_breaks.parquet", "summary_categories.parquet" + ) + on_disk <- basename(list.files( + file.path(fixture_corpus_path(), "data"), pattern = "\\.parquet$" + )) + expect_true(all(expected %in% on_disk)) + + # The manifest must list them too -- consumers read the manifest, not ls(). + in_manifest <- with_fixture_corpus( + basename(vapply(cog_manifest()$files$metadata, function(f) f$path, character(1))) + ) + expect_true(all(expected %in% in_manifest)) +}) + +test_that("fixture carries the dense/sparse representation contract", { + skip_if_no_corpus() + rep <- fixture_query( + "SELECT year, representation, absence_means FROM %s + WHERE year IN (2011, 2012, 2019, 2020) ORDER BY year", + "data/representation.parquet" + ) + expect_equal(nrow(rep), 4L) + expect_equal(rep$representation, c("dense_source", rep("sparse_source", 3L))) + expect_equal(rep$absence_means, c("census_zero", rep("not_reported", 3L))) +}) + +test_that("the fixture's wide era is sparse, not zero-padded", { + skip_if_no_corpus() + # FY2011 is a dense_source year: the corpus publishes only the cells Census + # reported non-zero, and an absent cell means Census published $0. Before + # sparsification this partition was 2,864,212 rows, ~83% of them explicit + # zeros. A single explicit zero here means the fixture predates the change. + zeros_2011 <- fixture_query( + "SELECT COUNT(*) AS n FROM %s WHERE amt = 0", + "data/long/year=2011/part-0.parquet" + )$n + expect_equal(zeros_2011, 0L) + + # The modern era is a different regime: a reported zero there is real data + # (the government filed $0), so zeros legitimately survive and must not be + # asserted away. + expect_gt( + fixture_query("SELECT COUNT(*) AS n FROM %s", "data/long/year=2012/part-0.parquet")$n, + 0L + ) +}) + +test_that("code_set covers every fixture year with the reader-spec columns", { + skip_if_no_corpus() + cs <- fixture_query( + "SELECT * FROM %s WHERE year IN (2011, 2012, 2019, 2020)", + "data/code_set.parquet" + ) + expect_true(all( + c("code_set_id", "year", "type", "item_code", "is_aggregate", "n_units") + %in% names(cs) + )) + expect_setequal(unique(cs$year), c(2011L, 2012L, 2019L, 2020L)) +}) + +test_that("every flow code carrying dollars has a category, J-prefix included", { + skip_if_no_corpus() + # The J (assistance/benefit) codes were uncategorised until the crosswalk + # completion shipped (cog_pipeline#60/#65, J19 held back until #64's + # duplication fix landed). Their absence is how a pre-crosswalk fixture + # gives itself away. + j <- fixture_query( + "SELECT item_code, category, category_type, spend_subtype FROM %s + WHERE LEFT(item_code, 1) = 'J' ORDER BY item_code", + "data/summary_categories.parquet" + ) + expect_true("J19" %in% j$item_code) + expect_true(all(j$category_type == "expenditure")) + expect_true(all(j$spend_subtype == "assistance")) + expect_false(any(is.na(j$category))) +}) diff --git a/tests/testthat/test-spending.R b/tests/testthat/test-spending.R index 63fc01c..8897c50 100644 --- a/tests/testthat/test-spending.R +++ b/tests/testthat/test-spending.R @@ -244,24 +244,42 @@ test_that("basis defaults to 'harmonized' when not passed", { test_that("provenance carries basis + harmonization block with na_rows_excluded", { skip_if_no_corpus() with_fixture_corpus({ - r <- cog_spending("121011212191", 2011:2012, "Corrections") + # FL state government. The harmonization block is scoped by government, + # year and flow prefix -- NOT by category -- so a Corrections query still + # counts every E/F/G-prefixed row the harmonized basis drops for having + # no harmonized_code. The three that apply here are E21/F21/G21 + # (Education NEC, SB184-186, "discontinued_na", wide-era window ending + # FY2011); the other discontinued_na rulings live outside E/F/G. + # See docs/phase_r_harmonization_review.md § 1.3/1.4 and cog_pipeline + # data/harmonization_map.csv. + r <- cog_spending("120000226351", 2011:2012, "Corrections") prov <- attr(r, "provenance") expect_equal(prov$basis, "harmonized") expect_true(prov$harmonization$applied) - expect_true(prov$harmonization$na_rows_excluded >= 0L) - expect_true(prov$harmonization$na_amount_excluded >= 0) - # Data-verified for the v6 fixture (corpus 2026-07-22). The Task 18 map - # extension added E/F/G-prefix discontinued_na rulings the earlier pin's - # comment predated: E21/F21/G21 (Education NEC local, SB184-186, - # "trivial; explicit-NA, full wide-era window"). Broward's 2011 legacy - # partition zero-pads exactly those three codes, so this query now - # excludes 3 NA-harmonized rows -- all with amt = 0, hence the excluded - # AMOUNT stays exactly zero. (The other discontinued_na rulings -- S74, - # Z61, X04, X06, the debt-detail family, L24 -- remain outside the - # E/F/G/K prefixes.) See docs/phase_r_harmonization_review.md § 1.3/1.4 - # and cog_pipeline data/harmonization_map.csv E21/F21/G21 rows. expect_equal(prov$harmonization$na_rows_excluded, 3L) - expect_equal(prov$harmonization$na_amount_excluded, 0) + # $2,825,439 thousands of FY2011 E21 + F21 + G21, reported in full USD. + # Pinning a non-zero amount is the point: the earlier Broward anchor's + # three rows were all explicit zeros, so the AMOUNT accounting was + # asserted only against 0 and could not have caught a bug. + expect_equal(prov$harmonization$na_amount_excluded, 2825439 * 1000) + }) +}) + +test_that("sparsification removed the wide era's zero-pads from the exclusion count", { + skip_if_no_corpus() + with_fixture_corpus({ + # Broward County FY2011 used to carry E21/F21/G21 rows of exactly $0 -- + # the wide era stored every government x every code, zeros included. The + # published corpus no longer does (SB194, cog_pipeline#64), so there is + # now nothing for the harmonized basis to exclude. Absence in a + # dense_source year means Census published $0; it does not mean the + # exclusion machinery stopped working, which the FL state anchor above + # proves independently. + r <- cog_spending("121011212191", 2011:2012, "Corrections") + h <- attr(r, "provenance")$harmonization + expect_true(h$applied) + expect_equal(h$na_rows_excluded, 0L) + expect_equal(h$na_amount_excluded, 0) }) })