The fixture predated three shipped corpus changes at once: no J rows in summary_categories (it was built before the crosswalk completion), no representation.parquet or code_set.parquet, and a still-dense wide era. Every test in this package and in cog-api runs against it, so both suites were green against a corpus that no longer exists. This is #18's stated prerequisite; it proves nothing about production until it lands. Regenerated from the publish tree at pipeline_commit 83f9715 (schema v6, built 2026-07-29). FY2011 goes from 2,864,212 rows to 496,004 -- 82.7% of the old partition was explicit zeros -- and the fixture now ships all ten publish-tree metadata tables rather than six. The generator's file list is a single constant now, so the copy step and the manifest step cannot drift. Three test repairs, each a real consequence of sparsification rather than a number to bump: test-categories.R "assistance" joined the spending subtype vocabulary with the J-prefix codes. test-spending.R The harmonization block counts rows that exist. Broward's E21/F21/G21 were zero-pads and are gone, so the anchor moves to FL state, whose three NA-mapped rows carry $2.83B -- the amount accounting was previously asserted only against 0 and could not have caught a bug. Broward keeps a test of its own, now asserting the zero-pads are absent. test-expenditure-concept.R Coverage-gap suggestions are presence-based. AL state's only FY2011 B47 cell was an explicit zero, so ig_federal_b47_wide stopped being a candidate there; FL state carries a real amount, so the counterpart guard is exercised against a suggestion that fires. test-fixture-vintage.R pins the structural facts that separate this vintage from its predecessor -- the ten metadata tables, the dense/sparse representation contract, zero explicit zeros in FY2011, code_set coverage, and J19's category. Checked against the old fixture: FY2011 carried 2,368,208 explicit zeros, so the assertion discriminates rather than merely passing. Suites: uscogdata 594 pass / 0 fail / 6 skip (was 576/0/6). cog-api 357 pass / 0 fail / 8 skip against the regenerated fixture, unchanged from its baseline.
78 lines
2.9 KiB
R
78 lines
2.9 KiB
R
test_that("cog_categories returns all categories grouped by subtype", {
|
|
skip_if_no_corpus()
|
|
r <- cog_categories()
|
|
expect_s3_class(r, "tbl_df")
|
|
expected <- c("category", "category_type", "subtype",
|
|
"n_codes", "item_codes")
|
|
expect_true(all(expected %in% names(r)))
|
|
expect_gt(nrow(r), 10L)
|
|
# corpus preserves Census-native "expenditure" vocabulary; the API takes
|
|
# "spending" as a friendlier alias.
|
|
expect_setequal(unique(r$category_type), c("expenditure", "revenue"))
|
|
})
|
|
|
|
test_that("cog_categories(type = 'spending') returns only expenditure rows", {
|
|
skip_if_no_corpus()
|
|
r <- cog_categories(type = "spending")
|
|
expect_true(all(r$category_type == "expenditure"))
|
|
# "assistance" (the J-prefix aid/benefit codes) joined the vocabulary with
|
|
# the crosswalk completion in cog_pipeline#60/#65 -- every flow code
|
|
# carrying dollars now maps to a category.
|
|
expect_true(all(r$subtype %in%
|
|
c("operations", "capital", "intergovernmental", "assistance")))
|
|
})
|
|
|
|
test_that("cog_categories surfaces the intergovernmental spending subtype", {
|
|
skip_if_no_corpus()
|
|
r <- cog_categories(type = "spending")
|
|
expect_true("intergovernmental" %in% r$subtype)
|
|
# IG rows reuse the existing functional categories -- they add a subtype,
|
|
# not new category values.
|
|
ig_cats <- sort(unique(r$category[r$subtype == "intergovernmental"]))
|
|
direct_cats <- sort(unique(r$category[r$subtype != "intergovernmental"]))
|
|
expect_true(all(ig_cats %in% c(direct_cats, "Other Education")))
|
|
})
|
|
|
|
test_that("cog_categories(type = 'revenue') returns only revenue rows", {
|
|
skip_if_no_corpus()
|
|
r <- cog_categories(type = "revenue")
|
|
expect_true(all(r$category_type == "revenue"))
|
|
expect_true(all(r$subtype %in%
|
|
c("own_source", "federal", "state", "local_aid")))
|
|
})
|
|
|
|
test_that("cog_categories(pattern = ...) filters case-insensitively", {
|
|
skip_if_no_corpus()
|
|
r <- cog_categories(pattern = "police")
|
|
expect_gt(nrow(r), 0L)
|
|
expect_true(all(grepl("Police", r$category, ignore.case = TRUE)))
|
|
})
|
|
|
|
test_that("cog_categories has one row per (category, subtype)", {
|
|
skip_if_no_corpus()
|
|
r <- cog_categories()
|
|
key <- paste(r$category, r$subtype, sep = "|")
|
|
expect_equal(length(key), length(unique(key)))
|
|
})
|
|
|
|
test_that("cog_categories item_codes is non-empty comma-separated string", {
|
|
skip_if_no_corpus()
|
|
r <- cog_categories()
|
|
expect_true(all(nzchar(r$item_codes)))
|
|
expect_true(all(r$n_codes >= 1L))
|
|
# n_codes should equal count of commas + 1
|
|
expect_equal(r$n_codes,
|
|
vapply(strsplit(r$item_codes, ","), length, integer(1)))
|
|
})
|
|
|
|
test_that("cog_categories sorted by category_type, category, subtype", {
|
|
skip_if_no_corpus()
|
|
r <- cog_categories()
|
|
sorted <- r[order(r$category_type, r$category, r$subtype), ]
|
|
expect_identical(r, sorted)
|
|
})
|
|
|
|
test_that("cog_categories rejects invalid type", {
|
|
expect_error(cog_categories(type = "both"), "type")
|
|
})
|