diff --git a/inst/extdata/fixture_corpus/data/series_breaks.parquet b/inst/extdata/fixture_corpus/data/series_breaks.parquet index 0c5e42e..eba2c94 100644 Binary files a/inst/extdata/fixture_corpus/data/series_breaks.parquet and b/inst/extdata/fixture_corpus/data/series_breaks.parquet differ diff --git a/inst/extdata/fixture_corpus/data/summary_categories.parquet b/inst/extdata/fixture_corpus/data/summary_categories.parquet index a55d04d..130944e 100644 Binary files a/inst/extdata/fixture_corpus/data/summary_categories.parquet and b/inst/extdata/fixture_corpus/data/summary_categories.parquet differ diff --git a/inst/extdata/fixture_corpus/manifest.json b/inst/extdata/fixture_corpus/manifest.json index 90f5d17..a8cfd11 100644 --- a/inst/extdata/fixture_corpus/manifest.json +++ b/inst/extdata/fixture_corpus/manifest.json @@ -1,7 +1,7 @@ { "schema_version": 6, - "built_at": "2026-07-30T14:07:36Z", - "pipeline_commit": "83f9715", + "built_at": "2026-07-30T20:01:56Z", + "pipeline_commit": "e64a046", "fixture_note": "Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated from the sparsified schema-v6 corpus: the wide era (<= FY2011) no longer stores explicit zeros, so FY2011 absence means Census published $0 while FY2012+ absence means not reported. representation.parquet and code_set.parquet carry that rule and ship in full, as do every other metadata table in the publish tree. 2011/2012 straddle both the wide-aggregate -> modern-leaf format boundary (exercised by basis=\"harmonized\" and recipe= queries) and the dense -> sparse representation boundary (SB194); 2019/2020 retain the prior per-capita/CPI regression anchors. Regenerated via data-raw/regenerate_fixture_corpus.R.", "data_vintage": { "source_vintages": { @@ -105,12 +105,12 @@ }, { "path": "data/series_breaks.parquet", - "sha256": "5ae050dd7a76c4d25e5f99e7c2e81c1896482e3504e0443b47ab5d78ba148953", + "sha256": "381090660c8b8a1bee852e7f63d29b9ecaf10f71870f41c63de92017c83b6f1f", "description": "series_breaks.parquet" }, { "path": "data/summary_categories.parquet", - "sha256": "e71d6d70d767c26c983fe56213baf204355f879582aa94841e62d9aea1877f83", + "sha256": "e4918abf8e9e6d1372d7ddc255dc199f9c303e250f4106449241b48c68abee67", "description": "summary_categories.parquet" } ] diff --git a/tests/testthat/test-categories.R b/tests/testthat/test-categories.R index d9f8574..4f89219 100644 --- a/tests/testthat/test-categories.R +++ b/tests/testthat/test-categories.R @@ -8,7 +8,14 @@ test_that("cog_categories returns all categories grouped by subtype", { expect_gt(nrow(r), 10L) # corpus preserves Census-native "expenditure" vocabulary; the API takes # "spending" as a friendlier alias. - expect_setequal(unique(r$category_type), c("expenditure", "revenue")) + # + # `balance` joined as a third category_type with the cash-and-security + # holding codes (pipeline#76). `cog_categories()` is a CATALOGUE verb, not a + # money verb, so it surfaces every category_type the corpus carries -- the + # stock/flow guard belongs on cog_spending()/cog_revenue(), which must never + # return a balance row. + expect_setequal(unique(r$category_type), + c("expenditure", "revenue", "balance")) }) test_that("cog_categories(type = 'spending') returns only expenditure rows", { @@ -18,8 +25,13 @@ test_that("cog_categories(type = 'spending') returns only expenditure rows", { # "assistance" (the J-prefix aid/benefit codes) joined the vocabulary with # the crosswalk completion in cog_pipeline#60/#65 -- every flow code # carrying dollars now maps to a category. + # `interest` (I89, I91-I94) and `insurance_benefits` (Y05/Y06/Y14/Y53) + # joined with the I/Q/Y flow batch -- the last two characters of Census's + # expenditure taxonomy. `interest` is what makes the three-concept model + # computable: primary = direct minus debt service. expect_true(all(r$subtype %in% - c("operations", "capital", "intergovernmental", "assistance"))) + c("operations", "capital", "intergovernmental", "assistance", + "interest", "insurance_benefits"))) }) test_that("cog_categories surfaces the intergovernmental spending subtype", { @@ -37,8 +49,12 @@ test_that("cog_categories(type = 'revenue') returns only revenue rows", { skip_if_no_corpus() r <- cog_categories(type = "revenue") expect_true(all(r$category_type == "revenue")) + # `insurance_trust` (Y01/Y02/Y04/Y11/Y12/Y51/Y52) is deliberately NOT + # own_source: Census's "General Revenue" excludes insurance trust revenue, + # and Y01 alone is $1.31T corpus-wide. expect_true(all(r$subtype %in% - c("own_source", "federal", "state", "local_aid"))) + c("own_source", "federal", "state", "local_aid", + "insurance_trust"))) }) test_that("cog_categories(pattern = ...) filters case-insensitively", {