Compare commits
4
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c375c55da7
|
||
|
|
82acda6f93 | ||
|
|
9233c3d18e
|
||
|
|
1f257812b6 |
@@ -1,5 +1,31 @@
|
|||||||
# uscogdata 0.1.0 (development)
|
# uscogdata 0.1.0 (development)
|
||||||
|
|
||||||
|
## Bundled fixture regenerated against the sparsified corpus
|
||||||
|
|
||||||
|
* `inst/extdata/fixture_corpus/` now tracks the corpus published on
|
||||||
|
2026-07-29 (`pipeline_commit 83f9715`, schema v6). The wide era no longer
|
||||||
|
stores explicit zeros: FY2011 fell from 2,864,212 rows to 496,004, of
|
||||||
|
which none are `$0`. **Absence now means two different things** — in a
|
||||||
|
`dense_source` year (≤ FY2011) an absent cell means Census published `$0`;
|
||||||
|
in a `sparse_source` year (≥ FY2012) it means not reported. The corpus
|
||||||
|
carries that rule in two new tables the fixture now ships,
|
||||||
|
`representation.parquet` and `code_set.parquet`, alongside
|
||||||
|
`census_collection_coverage.parquet` and `lineage_events.parquet`
|
||||||
|
(all ten publish-tree metadata tables, up from six). Catalogued upstream
|
||||||
|
as series break `SB194`.
|
||||||
|
* `cog_categories()` gains an `assistance` spending subtype: the J-prefix
|
||||||
|
aid/benefit codes (`J19`, `J67`, `J68`, `J85`) are categorised now that
|
||||||
|
the upstream crosswalk covers every flow code carrying dollars.
|
||||||
|
* Two consequences worth knowing about, both visible in provenance rather
|
||||||
|
than in returned dollars. The harmonization block's `na_rows_excluded`
|
||||||
|
counts only rows that exist, so wide-era codes that were zero-padded no
|
||||||
|
longer appear there. Coverage-gap `suggestions` are presence-based for the
|
||||||
|
same reason, so a recipe whose component codes were all `$0` for a given
|
||||||
|
government-year is no longer suggested for it.
|
||||||
|
* `tests/testthat/test-fixture-vintage.R` pins these structural facts, so a
|
||||||
|
fixture left behind by a future publish fails loudly instead of letting the
|
||||||
|
suite pass against a corpus that no longer exists.
|
||||||
|
|
||||||
## Breaking: corpus schema_version 4 (Phase P canonical ids)
|
## Breaking: corpus schema_version 4 (Phase P canonical ids)
|
||||||
|
|
||||||
* The package now requires corpus `schema_version = 4` (`MinCorpusSchema` /
|
* The package now requires corpus `schema_version = 4` (`MinCorpusSchema` /
|
||||||
|
|||||||
@@ -11,12 +11,14 @@
|
|||||||
# Each partition is a full year (all states/govs) as published, so
|
# Each partition is a full year (all states/govs) as published, so
|
||||||
# Broward County FL and every other previously-pinned government stay
|
# Broward County FL and every other previously-pinned government stay
|
||||||
# covered without any per-gov slicing logic.
|
# covered without any per-gov slicing logic.
|
||||||
# 2. Copies the full canonical_fips_xwalk.parquet, canonical_alias.parquet,
|
# 2. Copies every metadata parquet the publish tree ships (see
|
||||||
# summary_categories.parquet, harmonization_map.parquet,
|
# .FIXTURE_METADATA_FILES) as-is. These are small cross-vintage
|
||||||
# harmonization_recipes.parquet, and series_breaks.parquet metadata
|
# registries, not partitioned by year, so the fixture ships the complete
|
||||||
# tables as-is (these are small cross-vintage registries, not
|
# tables rather than a year-scoped subset. representation.parquet and
|
||||||
# partitioned by year, so the fixture ships the complete tables rather
|
# code_set.parquet are what make the sparse wide era interpretable --
|
||||||
# than a year-scoped subset).
|
# absence means "$0" in a dense_source year and "not reported" in a
|
||||||
|
# sparse_source one -- so a fixture without them cannot represent the
|
||||||
|
# published corpus.
|
||||||
# 3. Resyncs the four reference docs (data_dictionary.md,
|
# 3. Resyncs the four reference docs (data_dictionary.md,
|
||||||
# reader-specification.md, README.md, series_breaks.md) from the
|
# reader-specification.md, README.md, series_breaks.md) from the
|
||||||
# publish tree's docs/.
|
# publish tree's docs/.
|
||||||
@@ -38,6 +40,22 @@
|
|||||||
# source("data-raw/regenerate_fixture_corpus.R")
|
# source("data-raw/regenerate_fixture_corpus.R")
|
||||||
# regenerate_fixture_corpus(publish_cache_dir = "/path/to/publish_cache")
|
# regenerate_fixture_corpus(publish_cache_dir = "/path/to/publish_cache")
|
||||||
|
|
||||||
|
# Every metadata parquet the publish tree ships, in the order they appear in
|
||||||
|
# the corpus manifest. Single source of truth for both the copy step and the
|
||||||
|
# fixture manifest, so the two can never drift apart.
|
||||||
|
.FIXTURE_METADATA_FILES <- c(
|
||||||
|
"canonical_alias.parquet",
|
||||||
|
"canonical_fips_xwalk.parquet",
|
||||||
|
"census_collection_coverage.parquet",
|
||||||
|
"code_set.parquet",
|
||||||
|
"harmonization_map.parquet",
|
||||||
|
"harmonization_recipes.parquet",
|
||||||
|
"lineage_events.parquet",
|
||||||
|
"representation.parquet",
|
||||||
|
"series_breaks.parquet",
|
||||||
|
"summary_categories.parquet"
|
||||||
|
)
|
||||||
|
|
||||||
regenerate_fixture_corpus <- function(
|
regenerate_fixture_corpus <- function(
|
||||||
publish_cache_dir = file.path(
|
publish_cache_dir = file.path(
|
||||||
"..", "cog_pipeline", "_targets", "publish_cache"
|
"..", "cog_pipeline", "_targets", "publish_cache"
|
||||||
@@ -100,20 +118,11 @@ regenerate_fixture_corpus <- function(
|
|||||||
invisible(NULL)
|
invisible(NULL)
|
||||||
}
|
}
|
||||||
|
|
||||||
# Copy the full (not year-scoped) canonical_fips_xwalk, canonical_alias,
|
# Copy the full (not year-scoped) metadata tables listed in
|
||||||
# summary_categories, and (schema v5+) harmonization_map/
|
# .FIXTURE_METADATA_FILES.
|
||||||
# harmonization_recipes/series_breaks parquet tables.
|
|
||||||
#' @noRd
|
#' @noRd
|
||||||
.copy_metadata_parquets <- function(publish_cache_dir, fixture_dir) {
|
.copy_metadata_parquets <- function(publish_cache_dir, fixture_dir) {
|
||||||
files <- c(
|
for (f in .FIXTURE_METADATA_FILES) {
|
||||||
"canonical_fips_xwalk.parquet",
|
|
||||||
"canonical_alias.parquet",
|
|
||||||
"summary_categories.parquet",
|
|
||||||
"harmonization_map.parquet",
|
|
||||||
"harmonization_recipes.parquet",
|
|
||||||
"series_breaks.parquet"
|
|
||||||
)
|
|
||||||
for (f in files) {
|
|
||||||
src <- file.path(publish_cache_dir, "data", f)
|
src <- file.path(publish_cache_dir, "data", f)
|
||||||
dst <- file.path(fixture_dir, "data", f)
|
dst <- file.path(fixture_dir, "data", f)
|
||||||
if (!file.exists(src)) {
|
if (!file.exists(src)) {
|
||||||
@@ -179,15 +188,7 @@ regenerate_fixture_corpus <- function(
|
|||||||
)
|
)
|
||||||
})
|
})
|
||||||
|
|
||||||
metadata_files <- c(
|
metadata <- lapply(.FIXTURE_METADATA_FILES, function(f) {
|
||||||
"canonical_alias.parquet",
|
|
||||||
"canonical_fips_xwalk.parquet",
|
|
||||||
"summary_categories.parquet",
|
|
||||||
"harmonization_map.parquet",
|
|
||||||
"harmonization_recipes.parquet",
|
|
||||||
"series_breaks.parquet"
|
|
||||||
)
|
|
||||||
metadata <- lapply(metadata_files, function(f) {
|
|
||||||
rel <- file.path("data", f)
|
rel <- file.path("data", f)
|
||||||
path <- file.path(fixture_dir, rel)
|
path <- file.path(fixture_dir, rel)
|
||||||
list(
|
list(
|
||||||
@@ -203,13 +204,16 @@ regenerate_fixture_corpus <- function(
|
|||||||
pipeline_commit = source_manifest$pipeline_commit,
|
pipeline_commit = source_manifest$pipeline_commit,
|
||||||
fixture_note = paste(
|
fixture_note = paste(
|
||||||
"Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full",
|
"Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full",
|
||||||
"corpus available via USCOGDATA_URL. Regenerated for Phase R2",
|
"corpus available via USCOGDATA_URL. Regenerated from the sparsified",
|
||||||
"(schema_version 5, harmonization_map/harmonization_recipes/",
|
"schema-v6 corpus: the wide era (<= FY2011) no longer stores explicit",
|
||||||
"series_breaks parquet tables added). 2011/2012 straddle the",
|
"zeros, so FY2011 absence means Census published $0 while FY2012+",
|
||||||
"wide-aggregate -> modern-leaf format boundary exercised by basis=",
|
"absence means not reported. representation.parquet and",
|
||||||
"\"harmonized\" and recipe= queries; 2019/2020 retain the prior",
|
"code_set.parquet carry that rule and ship in full, as do every other",
|
||||||
"per-capita/CPI regression anchors. Full canonical_fips_xwalk master",
|
"metadata table in the publish tree. 2011/2012 straddle both the",
|
||||||
"and canonical_alias lookup table included via",
|
"wide-aggregate -> modern-leaf format boundary (exercised by",
|
||||||
|
"basis=\"harmonized\" and recipe= queries) and the dense -> sparse",
|
||||||
|
"representation boundary (SB194); 2019/2020 retain the prior",
|
||||||
|
"per-capita/CPI regression anchors. Regenerated via",
|
||||||
"data-raw/regenerate_fixture_corpus.R."
|
"data-raw/regenerate_fixture_corpus.R."
|
||||||
),
|
),
|
||||||
data_vintage = source_manifest$data_vintage,
|
data_vintage = source_manifest$data_vintage,
|
||||||
|
|||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+30
-10
@@ -1,8 +1,8 @@
|
|||||||
{
|
{
|
||||||
"schema_version": 6,
|
"schema_version": 6,
|
||||||
"built_at": "2026-07-27T13:04:05Z",
|
"built_at": "2026-07-30T14:07:36Z",
|
||||||
"pipeline_commit": "6098baf",
|
"pipeline_commit": "83f9715",
|
||||||
"fixture_note": "Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated for Phase R2 (schema_version 5, harmonization_map/harmonization_recipes/ series_breaks parquet tables added). 2011/2012 straddle the wide-aggregate -> modern-leaf format boundary exercised by basis= \"harmonized\" and recipe= queries; 2019/2020 retain the prior per-capita/CPI regression anchors. Full canonical_fips_xwalk master and canonical_alias lookup table included via data-raw/regenerate_fixture_corpus.R.",
|
"fixture_note": "Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated from the sparsified schema-v6 corpus: the wide era (<= FY2011) no longer stores explicit zeros, so FY2011 absence means Census published $0 while FY2012+ absence means not reported. representation.parquet and code_set.parquet carry that rule and ship in full, as do every other metadata table in the publish tree. 2011/2012 straddle both the wide-aggregate -> modern-leaf format boundary (exercised by basis=\"harmonized\" and recipe= queries) and the dense -> sparse representation boundary (SB194); 2019/2020 retain the prior per-capita/CPI regression anchors. Regenerated via data-raw/regenerate_fixture_corpus.R.",
|
||||||
"data_vintage": {
|
"data_vintage": {
|
||||||
"source_vintages": {
|
"source_vintages": {
|
||||||
"2012": "10162019",
|
"2012": "10162019",
|
||||||
@@ -36,9 +36,9 @@
|
|||||||
{
|
{
|
||||||
"year": 2011,
|
"year": 2011,
|
||||||
"path": "data/long/year=2011/part-0.parquet",
|
"path": "data/long/year=2011/part-0.parquet",
|
||||||
"sha256": "84302ab364dc9fc3b3fbbc3c3f8b826e3508b4d73ff7c42d094d3863cd1e37b5",
|
"sha256": "7848e18497080c8980a4f89c5b386205b2c5bc90db6773827ea01ab3943d16b1",
|
||||||
"row_count": 2864212,
|
"row_count": 496004,
|
||||||
"size_bytes": 3845911
|
"size_bytes": 2202455
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"year": 2012,
|
"year": 2012,
|
||||||
@@ -74,9 +74,14 @@
|
|||||||
"description": "canonical_fips_xwalk.parquet"
|
"description": "canonical_fips_xwalk.parquet"
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"path": "data/summary_categories.parquet",
|
"path": "data/census_collection_coverage.parquet",
|
||||||
"sha256": "0985b607f3f35a8dff62c0561261ab6922423b81d11c07b03bcb3e3461f85e33",
|
"sha256": "143e025616cde684da7c4442bc00d07fbd1556fabb0ea96223931b737e5d10a4",
|
||||||
"description": "summary_categories.parquet"
|
"description": "census_collection_coverage.parquet"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"path": "data/code_set.parquet",
|
||||||
|
"sha256": "4cffcb0198dd51e4ff2b694050bb371a5f9965cdac12f25521cb628fb8e118a9",
|
||||||
|
"description": "code_set.parquet"
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"path": "data/harmonization_map.parquet",
|
"path": "data/harmonization_map.parquet",
|
||||||
@@ -88,10 +93,25 @@
|
|||||||
"sha256": "1133e9a0b02f8f34f5f936e55c5ecd596bb8a55d8425dcce76767f0f3203581c",
|
"sha256": "1133e9a0b02f8f34f5f936e55c5ecd596bb8a55d8425dcce76767f0f3203581c",
|
||||||
"description": "harmonization_recipes.parquet"
|
"description": "harmonization_recipes.parquet"
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
"path": "data/lineage_events.parquet",
|
||||||
|
"sha256": "36c16acfbe621d61010984767f1c566993b8a5f481a2c1e134c4c0a600e4502f",
|
||||||
|
"description": "lineage_events.parquet"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"path": "data/representation.parquet",
|
||||||
|
"sha256": "31ec328a7dd505a321b45f97aafff12e53d68a1a986f63509863035b22a4360d",
|
||||||
|
"description": "representation.parquet"
|
||||||
|
},
|
||||||
{
|
{
|
||||||
"path": "data/series_breaks.parquet",
|
"path": "data/series_breaks.parquet",
|
||||||
"sha256": "b0b6794b6887a4f300079adfa10029c2a77109faa4952fbff1c5a270793cc02b",
|
"sha256": "5ae050dd7a76c4d25e5f99e7c2e81c1896482e3504e0443b47ab5d78ba148953",
|
||||||
"description": "series_breaks.parquet"
|
"description": "series_breaks.parquet"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"path": "data/summary_categories.parquet",
|
||||||
|
"sha256": "e71d6d70d767c26c983fe56213baf204355f879582aa94841e62d9aea1877f83",
|
||||||
|
"description": "summary_categories.parquet"
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
|
|||||||
@@ -0,0 +1,44 @@
|
|||||||
|
# Helper for the Madison-walkthrough finding tests (uscogdata #11-#16).
|
||||||
|
#
|
||||||
|
# Those tests all assert something about what a `cog_*` verb includes or
|
||||||
|
# excludes. The expected amounts must therefore come from the RAW corpus, never
|
||||||
|
# from the verb under test: verifying an absence through the filter that creates
|
||||||
|
# it proves nothing. `wt_raw_*()` opens its own DuckDB connection straight onto
|
||||||
|
# the corpus's `long` parquet partitions, bypassing uscogdata's SQL views (and
|
||||||
|
# therefore its `flow_prefixes` filtering) entirely.
|
||||||
|
|
||||||
|
wt_corpus_glob <- function() {
|
||||||
|
url <- Sys.getenv("USCOGDATA_URL")
|
||||||
|
if (!nzchar(url)) testthat::skip("USCOGDATA_URL is not set")
|
||||||
|
paste0(sub("/$", "", url), "/data/long/**/*.parquet")
|
||||||
|
}
|
||||||
|
|
||||||
|
wt_raw_query <- function(sql) {
|
||||||
|
con <- DBI::dbConnect(duckdb::duckdb())
|
||||||
|
on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE)
|
||||||
|
DBI::dbGetQuery(con, sql)
|
||||||
|
}
|
||||||
|
|
||||||
|
# Sum of `amt` (in $1,000s, as the corpus stores it) for one government-year,
|
||||||
|
# restricted either to an explicit set of item codes or to a set of first-letter
|
||||||
|
# prefixes. Aggregate rows are excluded, matching every published verb.
|
||||||
|
wt_raw_amt <- function(govid, year, codes = NULL, prefixes = NULL) {
|
||||||
|
stopifnot(xor(is.null(codes), is.null(prefixes)))
|
||||||
|
filter_sql <- if (!is.null(codes)) {
|
||||||
|
paste0("item_code IN (", paste0("'", codes, "'", collapse = ", "), ")")
|
||||||
|
} else {
|
||||||
|
paste0("LEFT(item_code, 1) IN (", paste0("'", prefixes, "'", collapse = ", "), ")")
|
||||||
|
}
|
||||||
|
out <- wt_raw_query(paste0(
|
||||||
|
"SELECT COALESCE(SUM(amt), 0) AS amt FROM read_parquet('", wt_corpus_glob(), "') ",
|
||||||
|
"WHERE canonical_govid = '", govid, "' AND year = ", year,
|
||||||
|
" AND NOT is_aggregate AND ", filter_sql
|
||||||
|
))
|
||||||
|
out$amt[[1]]
|
||||||
|
}
|
||||||
|
|
||||||
|
# The item codes a verb reports having summed, flattened out of the
|
||||||
|
# comma-separated `codes_included` column.
|
||||||
|
wt_codes_included <- function(df) {
|
||||||
|
sort(unique(trimws(unlist(strsplit(stats::na.omit(df$codes_included), ",")))))
|
||||||
|
}
|
||||||
@@ -0,0 +1,46 @@
|
|||||||
|
# Madison walkthrough audit -- finding F-004. Tracked as uscogdata#15.
|
||||||
|
# See docs/walkthroughs/FINDINGS.md in cog_explorer.
|
||||||
|
#
|
||||||
|
# The raw Census files report thousands of dollars; this package multiplies by
|
||||||
|
# 1000 and returns full US dollars. That is the friendlier choice and is not
|
||||||
|
# wrong -- but cog_explorer's CLAUDE.md states "All raw `amt` values are in
|
||||||
|
# $1,000s", so a reader who applies that rule to amt_nominal overstates every
|
||||||
|
# figure by 1000x, and gets a plausible-looking number rather than an obvious
|
||||||
|
# error. The audit rates this the highest-consequence definitional gap it found.
|
||||||
|
#
|
||||||
|
# Deliberately NOT asserted here: man/cog_spending.Rd and man/cog_revenue.Rd,
|
||||||
|
# which ALREADY carry the statement in their @return sections (verified
|
||||||
|
# 2026-07-29), as does cog-api's data-dictionary.md (since 2b71b41). The gap is
|
||||||
|
# in the surfaces a reader meets first and in cog_explorer's own conventions
|
||||||
|
# doc -- see uscogdata#15 for the full surface-by-surface table and for the two
|
||||||
|
# secondary tasks (cog_explorer/CLAUDE.md, which has no git remote, and
|
||||||
|
# cog-api's llms.txt, which is silent on units).
|
||||||
|
|
||||||
|
test_that("returned amounts are documented as full US dollars where readers meet the package", {
|
||||||
|
testthat::skip("Blocked on uscogdata#15 (finding F-004)")
|
||||||
|
|
||||||
|
says_units <- function(path) {
|
||||||
|
txt <- paste(readLines(path, warn = FALSE), collapse = " ")
|
||||||
|
grepl("full US dollars|full U\\.S\\. dollars", txt, ignore.case = TRUE) &&
|
||||||
|
grepl("\\$1,000s|thousands of dollars", txt, ignore.case = TRUE)
|
||||||
|
}
|
||||||
|
|
||||||
|
expect_true(says_units(testthat::test_path("..", "..", "README.md")))
|
||||||
|
expect_true(says_units(testthat::test_path("..", "..", "vignettes", "total-spending.Rmd")))
|
||||||
|
expect_true(says_units(testthat::test_path("..", "..", "vignettes",
|
||||||
|
"population-denominators.Rmd")))
|
||||||
|
|
||||||
|
# Pin the documented claim to the actual behaviour, so the two cannot drift.
|
||||||
|
# The expected raw amount is read straight from the corpus's parquet
|
||||||
|
# partitions -- never through cog_spending(), which is the thing being
|
||||||
|
# described. Madison FY2020: E/F/G = 623,347 ($1,000s) -> $623,347,000.
|
||||||
|
raw_thousands <- wt_raw_amt("552025209777", 2020L, prefixes = c("E", "F", "G"))
|
||||||
|
expect_equal(raw_thousands, 623347)
|
||||||
|
|
||||||
|
returned <- cog_spending(govid = "552025209777", years = 2020L)
|
||||||
|
expect_equal(sum(returned$amt_nominal), raw_thousands * 1000)
|
||||||
|
|
||||||
|
units <- attr(returned, "provenance")$transformations$units_conversion
|
||||||
|
expect_true(units$applied)
|
||||||
|
expect_equal(units$multiplier, 1000)
|
||||||
|
})
|
||||||
@@ -15,7 +15,11 @@ test_that("cog_categories(type = 'spending') returns only expenditure rows", {
|
|||||||
skip_if_no_corpus()
|
skip_if_no_corpus()
|
||||||
r <- cog_categories(type = "spending")
|
r <- cog_categories(type = "spending")
|
||||||
expect_true(all(r$category_type == "expenditure"))
|
expect_true(all(r$category_type == "expenditure"))
|
||||||
expect_true(all(r$subtype %in% c("operations", "capital", "intergovernmental")))
|
# "assistance" (the J-prefix aid/benefit codes) joined the vocabulary with
|
||||||
|
# the crosswalk completion in cog_pipeline#60/#65 -- every flow code
|
||||||
|
# carrying dollars now maps to a category.
|
||||||
|
expect_true(all(r$subtype %in%
|
||||||
|
c("operations", "capital", "intergovernmental", "assistance")))
|
||||||
})
|
})
|
||||||
|
|
||||||
test_that("cog_categories surfaces the intergovernmental spending subtype", {
|
test_that("cog_categories surfaces the intergovernmental spending subtype", {
|
||||||
|
|||||||
@@ -0,0 +1,92 @@
|
|||||||
|
# Madison walkthrough audit -- findings F-020 and F-023. Tracked as uscogdata#13.
|
||||||
|
# See docs/walkthroughs/FINDINGS.md in cog_explorer.
|
||||||
|
#
|
||||||
|
# The owner's settled design (2026-07-28): a `coverage` argument on
|
||||||
|
# cog_geographic_rollup(), cog_find_peers()/cog_peer_compare() and their
|
||||||
|
# cog-api equivalents --
|
||||||
|
# "all" every unit that reported that year (today's behaviour, DEFAULT)
|
||||||
|
# "census" census years only (years ending 2 or 7)
|
||||||
|
# "consistent" only units reporting in every requested year (balanced panel)
|
||||||
|
# -- PLUS always-on coverage metadata on every result regardless of mode:
|
||||||
|
# n_units_reporting, n_units_expected, is_census_year.
|
||||||
|
#
|
||||||
|
# Motivating principle: using these verbs correctly must not require the user to
|
||||||
|
# know that the Census of Governments is a complete census only in years ending
|
||||||
|
# in 2 and 7.
|
||||||
|
#
|
||||||
|
# The helper below accepts that metadata either as columns on the returned
|
||||||
|
# tibble or as a per-year table in provenance$coverage -- the design fixes the
|
||||||
|
# three field names and that they reach the caller, not the container.
|
||||||
|
|
||||||
|
wt_coverage <- function(x) {
|
||||||
|
prov <- attr(x, "provenance")
|
||||||
|
cov <- prov$coverage
|
||||||
|
if (is.null(cov)) {
|
||||||
|
needed <- c("year", "n_units_reporting", "n_units_expected", "is_census_year")
|
||||||
|
expect_true(all(needed %in% names(x)))
|
||||||
|
cov <- unique(x[, needed])
|
||||||
|
}
|
||||||
|
cov[order(cov$year), ]
|
||||||
|
}
|
||||||
|
|
||||||
|
test_that("multi-government aggregates disclose reporting coverage on every result", {
|
||||||
|
testthat::skip("Blocked on uscogdata#13 (findings F-020, F-023)")
|
||||||
|
|
||||||
|
# -- F-020: geographic rollups -------------------------------------------
|
||||||
|
# Wisconsin's city/village universe is 608 governments. On the bundled
|
||||||
|
# fixture, FY2012 (a census year) has 597 of them reporting while FY2019 and
|
||||||
|
# FY2020 (sample years) have 112 and 114 -- an 18%-98% swing that today's
|
||||||
|
# return value says nothing about. Counts cross-checked against the raw
|
||||||
|
# corpus, not through cog_geographic_rollup(), which is under test.
|
||||||
|
wi <- cog_gov_search(name = NULL, state = "WI", type = "city")
|
||||||
|
expect_equal(nrow(wi), 608L)
|
||||||
|
|
||||||
|
roll <- cog_geographic_rollup(govids = list(city = wi$canonical_govid),
|
||||||
|
category = NULL, years = c(2011L, 2012L, 2019L, 2020L))
|
||||||
|
cov <- wt_coverage(roll)
|
||||||
|
|
||||||
|
expect_equal(cov$n_units_expected, rep(608L, 4L))
|
||||||
|
expect_equal(cov$n_units_reporting, c(152L, 597L, 112L, 114L))
|
||||||
|
expect_equal(cov$is_census_year, c(FALSE, TRUE, FALSE, FALSE))
|
||||||
|
|
||||||
|
raw_2012 <- wt_raw_query(paste0(
|
||||||
|
"SELECT COUNT(DISTINCT canonical_govid) n FROM read_parquet('", wt_corpus_glob(), "') ",
|
||||||
|
"WHERE type = 2 AND fips_state = 55 AND year = 2012 ",
|
||||||
|
"AND LEFT(item_code, 1) IN ('E','F','G') AND NOT is_aggregate"))
|
||||||
|
expect_equal(cov$n_units_reporting[cov$year == 2012], as.integer(raw_2012$n[[1]]))
|
||||||
|
|
||||||
|
# -- F-023: peer cohorts --------------------------------------------------
|
||||||
|
# CHILTON CITY, WI (ACS population 4,017): a 15-peer cohort fixed at FY2012
|
||||||
|
# reports 15 of 15 in FY2012 and only 3 of 15 in FY2019 and FY2020. Nothing
|
||||||
|
# in cog_peer_compare()'s return distinguishes those years today.
|
||||||
|
chilton <- "552015177095"
|
||||||
|
peers <- cog_find_peers(chilton, year = 2012L, max_peers = 15L)
|
||||||
|
expect_equal(nrow(peers), 15L)
|
||||||
|
|
||||||
|
cmp <- cog_peer_compare(target_govid = chilton, peers = peers, category = NULL,
|
||||||
|
years = c(2012L, 2019L, 2020L), per_capita = TRUE)
|
||||||
|
cov_peers <- wt_coverage(cmp)
|
||||||
|
expect_equal(cov_peers$n_units_expected, rep(15L, 3L))
|
||||||
|
expect_equal(cov_peers$n_units_reporting, c(15L, 3L, 3L))
|
||||||
|
expect_equal(cov_peers$is_census_year, c(TRUE, FALSE, FALSE))
|
||||||
|
|
||||||
|
# -- the three coverage modes --------------------------------------------
|
||||||
|
expect_equal(attr(cog_peer_compare(target_govid = chilton, peers = peers,
|
||||||
|
category = NULL, years = c(2012L, 2019L, 2020L),
|
||||||
|
per_capita = TRUE),
|
||||||
|
"provenance")$coverage_mode, "all") # unchanged default
|
||||||
|
|
||||||
|
consistent <- cog_peer_compare(target_govid = chilton, peers = peers,
|
||||||
|
category = NULL, years = c(2012L, 2019L, 2020L),
|
||||||
|
per_capita = TRUE, coverage = "consistent")
|
||||||
|
n_by_year <- tapply(consistent$canonical_govid[consistent$role == "peer"],
|
||||||
|
consistent$year[consistent$role == "peer"],
|
||||||
|
function(g) length(unique(g)))
|
||||||
|
expect_equal(unname(as.integer(n_by_year)), c(3L, 3L, 3L)) # balanced panel
|
||||||
|
|
||||||
|
census_only <- cog_geographic_rollup(govids = list(city = wi$canonical_govid),
|
||||||
|
category = NULL,
|
||||||
|
years = c(2011L, 2012L, 2019L, 2020L),
|
||||||
|
coverage = "census")
|
||||||
|
expect_equal(sort(unique(census_only$year)), 2012)
|
||||||
|
})
|
||||||
@@ -298,8 +298,16 @@ test_that("a mis-scoped cog_spending() call never attaches an M/L counterpart to
|
|||||||
# (M47/M94, same suffixes) -- a coincidence of reused digits, not a real
|
# (M47/M94, same suffixes) -- a coincidence of reused digits, not a real
|
||||||
# Direct/Total pairing. The flow-family gate in
|
# Direct/Total pairing. The flow-family gate in
|
||||||
# .attach_ig_counterparts() must keep ig_recipe_id NULL here.
|
# .attach_ig_counterparts() must keep ig_recipe_id NULL here.
|
||||||
|
#
|
||||||
|
# Anchored on FL state government, not AL. Coverage is presence-based: a
|
||||||
|
# recipe is only suggested when its component codes have rows for the
|
||||||
|
# requested government-year. AL state's only FY2011 B47 cell was an
|
||||||
|
# explicit zero, which the corpus no longer stores after sparsification
|
||||||
|
# (SB194, cog_pipeline#64), so the recipe stopped being a candidate there.
|
||||||
|
# FL state carries a real FY2011 B47 amount, so this exercises the guard
|
||||||
|
# against a suggestion that genuinely fires.
|
||||||
r <- suppressMessages(
|
r <- suppressMessages(
|
||||||
cog_spending("010000226085", years = c(2005, 2011), category = "IG Federal")
|
cog_spending("120000226351", years = c(2005, 2011), category = "IG Federal")
|
||||||
)
|
)
|
||||||
sugg <- attr(r, "provenance")$suggestions
|
sugg <- attr(r, "provenance")$suggestions
|
||||||
expect_gt(length(sugg), 0L)
|
expect_gt(length(sugg), 0L)
|
||||||
|
|||||||
@@ -0,0 +1,75 @@
|
|||||||
|
# Madison walkthrough audit -- findings F-012, F-017, F-018.
|
||||||
|
# Tracked as uscogdata#11. See docs/walkthroughs/FINDINGS.md in cog_explorer.
|
||||||
|
#
|
||||||
|
# The owner's settled three-concept model (2026-07-28):
|
||||||
|
# total = primary + interest + intergovernmental transfers
|
||||||
|
# direct = primary + interest (Census's published Direct Expenditure)
|
||||||
|
# primary = direct minus debt service (the NEW DEFAULT)
|
||||||
|
# implemented by reclassifying on the crosswalk's `spend_type` column, NOT on
|
||||||
|
# item-code first letters -- F-018 shows prefix `Y` carries both revenue
|
||||||
|
# (Y01/Y02) and expenditure (Y05/Y06) codes, so no first-letter allowlist can
|
||||||
|
# route them correctly.
|
||||||
|
#
|
||||||
|
# Fixture reproducibility: the finding's headline reconciliation is Madison
|
||||||
|
# FY2022, where the corpus carries I89 = 46,609 (thousands) and Census's
|
||||||
|
# published Direct Expenditure is $654,893,000 against cog_spending()'s
|
||||||
|
# $608,284,000 (-7.1%). FY2022 is outside the bundled fixture's year window
|
||||||
|
# (2011/2012/2019/2020), so the same invariant is asserted on FY2020, where the
|
||||||
|
# fixture carries I89 = 27,704. Anyone running against the full corpus should
|
||||||
|
# also check the FY2022 numbers above.
|
||||||
|
|
||||||
|
test_that("expenditure concepts classify on spend_type, not item-code prefix", {
|
||||||
|
testthat::skip("Blocked on uscogdata#11 (findings F-012, F-017, F-018)")
|
||||||
|
|
||||||
|
mad <- "552025209777" # MADISON CITY, WI
|
||||||
|
wi_state <- "550000227544" # WISCONSIN (state government)
|
||||||
|
|
||||||
|
# -- F-012: `primary` is the new default, and equals today's E/F/G figure ---
|
||||||
|
primary <- cog_spending(govid = mad, years = 2020L)
|
||||||
|
expect_equal(attr(primary, "provenance")$expenditure_concept, "primary")
|
||||||
|
expect_equal(sum(primary$amt_nominal), 623347000)
|
||||||
|
|
||||||
|
# -- F-012: `direct` adds interest on long-term debt ------------------------
|
||||||
|
# Expected interest read from the RAW corpus, never through cog_spending(),
|
||||||
|
# which is the filter under test.
|
||||||
|
interest <- wt_raw_amt(mad, 2020L, prefixes = "I")
|
||||||
|
expect_equal(interest, 27704) # I89, in $1,000s
|
||||||
|
|
||||||
|
direct <- cog_spending(govid = mad, years = 2020L, expenditure_concept = "direct")
|
||||||
|
expect_equal(sum(direct$amt_nominal), 651051000) # 623,347 + 27,704 thousands
|
||||||
|
expect_equal(sum(direct$amt_nominal) - sum(primary$amt_nominal), interest * 1000)
|
||||||
|
expect_true("I89" %in% wt_codes_included(direct))
|
||||||
|
|
||||||
|
# -- F-017: `total` carries Q12/Q18, state IG transfers to school districts --
|
||||||
|
# Wisconsin FY2019: Q12 = 6,431,530 and Q18 = 533,391 (thousands). Today
|
||||||
|
# neither verb's flow_prefixes contains "Q", so both are dropped from the one
|
||||||
|
# concept that is supposed to include intergovernmental transfers.
|
||||||
|
ig_expected <- wt_raw_amt(wi_state, 2019L, prefixes = c("M", "L", "Q"))
|
||||||
|
expect_equal(ig_expected, 11609814) # M 4,644,893 + Q 6,964,921
|
||||||
|
|
||||||
|
wi_direct <- cog_spending(govid = wi_state, years = 2019L,
|
||||||
|
expenditure_concept = "direct")
|
||||||
|
wi_total <- cog_spending(govid = wi_state, years = 2019L,
|
||||||
|
expenditure_concept = "total")
|
||||||
|
|
||||||
|
# total - direct is exactly the intergovernmental component. Asserted as a
|
||||||
|
# delta rather than a grand total so this stays correct however the J and Y
|
||||||
|
# families land inside `primary`.
|
||||||
|
expect_equal(sum(wi_total$amt_nominal) - sum(wi_direct$amt_nominal),
|
||||||
|
ig_expected * 1000)
|
||||||
|
expect_true(all(c("Q12", "Q18") %in% wt_codes_included(wi_total)))
|
||||||
|
|
||||||
|
# -- F-018: prefix Y splits revenue from expenditure, by spend_type ---------
|
||||||
|
# Y01/Y02 are Insurance Trust revenue; Y05/Y06 are Insurance Trust benefit
|
||||||
|
# payments. All four share the first letter `Y` and the spend_type
|
||||||
|
# "Insurance Trust", so this pair of assertions is the concrete proof that
|
||||||
|
# classification is no longer keyed on the first letter.
|
||||||
|
wi_revenue <- cog_revenue(govid = wi_state, years = 2019L)
|
||||||
|
spend_codes <- wt_codes_included(wi_total)
|
||||||
|
rev_codes <- wt_codes_included(wi_revenue)
|
||||||
|
|
||||||
|
expect_true("Y05" %in% spend_codes)
|
||||||
|
expect_false("Y05" %in% rev_codes)
|
||||||
|
expect_true("Y01" %in% rev_codes)
|
||||||
|
expect_false("Y01" %in% spend_codes)
|
||||||
|
})
|
||||||
@@ -0,0 +1,112 @@
|
|||||||
|
# tests/testthat/test-fixture-vintage.R
|
||||||
|
#
|
||||||
|
# The bundled fixture is a slice of a real cog_pipeline publish tree, and
|
||||||
|
# every test in this package -- plus the whole cog-api suite -- runs against
|
||||||
|
# it. When the published corpus changes shape and the fixture does not, both
|
||||||
|
# suites stay green against a corpus that no longer exists (uscogdata#18).
|
||||||
|
#
|
||||||
|
# These tests pin the structural facts that distinguish the current published
|
||||||
|
# vintage from its predecessor, so a stale fixture fails loudly instead of
|
||||||
|
# passing quietly. They assert shape, never dollar values: re-running
|
||||||
|
# data-raw/regenerate_fixture_corpus.R against a newer publish tree should
|
||||||
|
# keep them green.
|
||||||
|
|
||||||
|
# Open a bare DuckDB connection on the fixture's parquet files. Deliberately
|
||||||
|
# not the package session: these assertions are about what the fixture
|
||||||
|
# CONTAINS, and routing them through the reader's own views would let a
|
||||||
|
# filter hide the very absence being checked.
|
||||||
|
fixture_query <- function(sql, ...) {
|
||||||
|
con <- DBI::dbConnect(duckdb::duckdb())
|
||||||
|
on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE)
|
||||||
|
path <- function(rel) {
|
||||||
|
sprintf("read_parquet(%s)",
|
||||||
|
DBI::dbQuoteString(con, file.path(fixture_corpus_path(), rel)))
|
||||||
|
}
|
||||||
|
DBI::dbGetQuery(con, do.call(sprintf, c(list(sql), lapply(c(...), path))))
|
||||||
|
}
|
||||||
|
|
||||||
|
test_that("fixture ships every metadata table the publish tree does", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
# representation/code_set are what make a sparse corpus interpretable; a
|
||||||
|
# fixture without them predates sparsification (cog_pipeline#64).
|
||||||
|
expected <- c(
|
||||||
|
"canonical_alias.parquet", "canonical_fips_xwalk.parquet",
|
||||||
|
"census_collection_coverage.parquet", "code_set.parquet",
|
||||||
|
"harmonization_map.parquet", "harmonization_recipes.parquet",
|
||||||
|
"lineage_events.parquet", "representation.parquet",
|
||||||
|
"series_breaks.parquet", "summary_categories.parquet"
|
||||||
|
)
|
||||||
|
on_disk <- basename(list.files(
|
||||||
|
file.path(fixture_corpus_path(), "data"), pattern = "\\.parquet$"
|
||||||
|
))
|
||||||
|
expect_true(all(expected %in% on_disk))
|
||||||
|
|
||||||
|
# The manifest must list them too -- consumers read the manifest, not ls().
|
||||||
|
in_manifest <- with_fixture_corpus(
|
||||||
|
basename(vapply(cog_manifest()$files$metadata, function(f) f$path, character(1)))
|
||||||
|
)
|
||||||
|
expect_true(all(expected %in% in_manifest))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("fixture carries the dense/sparse representation contract", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
rep <- fixture_query(
|
||||||
|
"SELECT year, representation, absence_means FROM %s
|
||||||
|
WHERE year IN (2011, 2012, 2019, 2020) ORDER BY year",
|
||||||
|
"data/representation.parquet"
|
||||||
|
)
|
||||||
|
expect_equal(nrow(rep), 4L)
|
||||||
|
expect_equal(rep$representation, c("dense_source", rep("sparse_source", 3L)))
|
||||||
|
expect_equal(rep$absence_means, c("census_zero", rep("not_reported", 3L)))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("the fixture's wide era is sparse, not zero-padded", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
# FY2011 is a dense_source year: the corpus publishes only the cells Census
|
||||||
|
# reported non-zero, and an absent cell means Census published $0. Before
|
||||||
|
# sparsification this partition was 2,864,212 rows, ~83% of them explicit
|
||||||
|
# zeros. A single explicit zero here means the fixture predates the change.
|
||||||
|
zeros_2011 <- fixture_query(
|
||||||
|
"SELECT COUNT(*) AS n FROM %s WHERE amt = 0",
|
||||||
|
"data/long/year=2011/part-0.parquet"
|
||||||
|
)$n
|
||||||
|
expect_equal(zeros_2011, 0L)
|
||||||
|
|
||||||
|
# The modern era is a different regime: a reported zero there is real data
|
||||||
|
# (the government filed $0), so zeros legitimately survive and must not be
|
||||||
|
# asserted away.
|
||||||
|
expect_gt(
|
||||||
|
fixture_query("SELECT COUNT(*) AS n FROM %s", "data/long/year=2012/part-0.parquet")$n,
|
||||||
|
0L
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("code_set covers every fixture year with the reader-spec columns", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
cs <- fixture_query(
|
||||||
|
"SELECT * FROM %s WHERE year IN (2011, 2012, 2019, 2020)",
|
||||||
|
"data/code_set.parquet"
|
||||||
|
)
|
||||||
|
expect_true(all(
|
||||||
|
c("code_set_id", "year", "type", "item_code", "is_aggregate", "n_units")
|
||||||
|
%in% names(cs)
|
||||||
|
))
|
||||||
|
expect_setequal(unique(cs$year), c(2011L, 2012L, 2019L, 2020L))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("every flow code carrying dollars has a category, J-prefix included", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
# The J (assistance/benefit) codes were uncategorised until the crosswalk
|
||||||
|
# completion shipped (cog_pipeline#60/#65, J19 held back until #64's
|
||||||
|
# duplication fix landed). Their absence is how a pre-crosswalk fixture
|
||||||
|
# gives itself away.
|
||||||
|
j <- fixture_query(
|
||||||
|
"SELECT item_code, category, category_type, spend_subtype FROM %s
|
||||||
|
WHERE LEFT(item_code, 1) = 'J' ORDER BY item_code",
|
||||||
|
"data/summary_categories.parquet"
|
||||||
|
)
|
||||||
|
expect_true("J19" %in% j$item_code)
|
||||||
|
expect_true(all(j$category_type == "expenditure"))
|
||||||
|
expect_true(all(j$spend_subtype == "assistance"))
|
||||||
|
expect_false(any(is.na(j$category)))
|
||||||
|
})
|
||||||
@@ -0,0 +1,58 @@
|
|||||||
|
# Madison walkthrough audit -- finding F-025. Tracked as uscogdata#16.
|
||||||
|
# See docs/walkthroughs/FINDINGS.md in cog_explorer.
|
||||||
|
#
|
||||||
|
# cog_gov_search()'s UTILITY mode interpolates `name` into
|
||||||
|
# regexp_matches(gov_name, <name>, 'i')
|
||||||
|
# unescaped (R/search.R:102), while BASKET mode in the same file already routes
|
||||||
|
# it through .escape_regex() (R/search.R:307) with the comment "so `name` is
|
||||||
|
# treated as a literal substring". Two failure modes result:
|
||||||
|
# correctness -- a real government cannot be found by its own exact name, and
|
||||||
|
# a single "." matches everything (HTTP 200 both ways via the API);
|
||||||
|
# robustness -- malformed regex reaches the engine and errors, which cog-api
|
||||||
|
# surfaces as a 500, reachable by typing a real name one
|
||||||
|
# character at a time.
|
||||||
|
#
|
||||||
|
# NOT asserted here: the finding's `q=St. Louis` example. Under correct literal
|
||||||
|
# matching that search still returns 0 rows, because the stored name is
|
||||||
|
# "ST LOUIS CITY" with no period -- it demonstrates today's over-matching
|
||||||
|
# semantics, not a row the fix makes findable.
|
||||||
|
|
||||||
|
test_that("cog_gov_search() matches name literally, not as an unescaped regex", {
|
||||||
|
testthat::skip("Blocked on uscogdata#16 (finding F-025)")
|
||||||
|
|
||||||
|
# -- correctness (1): a government must be findable by its own exact name ---
|
||||||
|
# FREDONIA (BRISCOE) CITY is real; today the parentheses are read as regex
|
||||||
|
# grouping, so its own complete name matches nothing.
|
||||||
|
fredonia <- cog_gov_search(name = "FREDONIA (BRISCOE) CITY")
|
||||||
|
expect_equal(nrow(fredonia), 1L)
|
||||||
|
expect_equal(fredonia$canonical_govid, "052117184386")
|
||||||
|
expect_equal(cog_gov_search(name = "FREDONIA (BRISCOE)")$canonical_govid,
|
||||||
|
"052117184386")
|
||||||
|
|
||||||
|
# -- correctness (2): a metacharacter must not become a wildcard ------------
|
||||||
|
# No Wisconsin city or village name contains a literal period -- established
|
||||||
|
# against the raw registry below, NOT through the verb under test. A literal
|
||||||
|
# search for "." must therefore return nothing; today it returns all 608.
|
||||||
|
con <- DBI::dbConnect(duckdb::duckdb())
|
||||||
|
on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE)
|
||||||
|
xwalk <- paste0(sub("/$", "", Sys.getenv("USCOGDATA_URL")),
|
||||||
|
"/data/canonical_fips_xwalk.parquet")
|
||||||
|
with_dot <- DBI::dbGetQuery(con, paste0(
|
||||||
|
"SELECT COUNT(*) n FROM read_parquet('", xwalk, "') ",
|
||||||
|
"WHERE fips_state = '55' AND govs_type = 2 AND gov_name LIKE '%.%'"))
|
||||||
|
expect_equal(as.integer(with_dot$n[[1]]), 0L)
|
||||||
|
|
||||||
|
expect_equal(nrow(cog_gov_search(name = ".", state = "WI", type = "city")), 0L)
|
||||||
|
expect_equal(nrow(cog_gov_search(name = "M.dison", state = "WI", type = "city")), 0L)
|
||||||
|
expect_equal(nrow(cog_gov_search(name = "Mad(i|o)son", state = "WI", type = "city")), 0L)
|
||||||
|
|
||||||
|
# A metacharacter-free name still resolves exactly as before.
|
||||||
|
expect_equal(nrow(cog_gov_search(name = "Madison", state = "WI", type = "city")), 1L)
|
||||||
|
|
||||||
|
# -- robustness: malformed pattern text returns no rows, and does not error --
|
||||||
|
# "[" alone, and "Athens-Clarke County (bal" -- an in-progress substring of
|
||||||
|
# ATHENS-CLARKE COUNTY (BALANCE), a real government -- both currently raise
|
||||||
|
# (DuckDB: "Invalid Input Error: missing ]").
|
||||||
|
expect_equal(nrow(cog_gov_search(name = "[")), 0L)
|
||||||
|
expect_equal(nrow(cog_gov_search(name = "Athens-Clarke County (bal")), 0L)
|
||||||
|
})
|
||||||
@@ -0,0 +1,45 @@
|
|||||||
|
# Madison walkthrough audit -- finding F-021. Tracked as uscogdata#14.
|
||||||
|
# See docs/walkthroughs/FINDINGS.md in cog_explorer.
|
||||||
|
#
|
||||||
|
# .peer_summary_rows() computes stats::quantile() separately INSIDE each
|
||||||
|
# (year, spend_subtype, category) cell. A summary_p50 row is therefore "the
|
||||||
|
# median peer's value in that one category", not "the value of the median
|
||||||
|
# peer's total". Summing those rows across categories -- the obvious move for a
|
||||||
|
# caller who wants one peer-median total line and reads only the column names --
|
||||||
|
# misstated a total-spending band by -32.7% to +251.0% across the 24 years the
|
||||||
|
# audit tested, with a sign flip at FY2012.
|
||||||
|
#
|
||||||
|
# The verb is not wrong and its documented use (faceting by role AND category)
|
||||||
|
# is unaffected, so the fix is documentation: one sentence in @return.
|
||||||
|
|
||||||
|
test_that("cog_peer_compare() documents that summary_* rows are per-category quantiles", {
|
||||||
|
testthat::skip("Blocked on uscogdata#14 (finding F-021)")
|
||||||
|
|
||||||
|
rd <- paste(readLines(testthat::test_path("..", "..", "man", "cog_peer_compare.Rd"),
|
||||||
|
warn = FALSE), collapse = " ")
|
||||||
|
|
||||||
|
# The @return section must say the quantile is computed within each cell...
|
||||||
|
expect_match(rd, "within each|per-category|per category", ignore.case = TRUE)
|
||||||
|
# ...and must warn that the rows are not additive across category.
|
||||||
|
expect_match(rd, "not additive|do(es)? not sum|cannot be summed", ignore.case = TRUE)
|
||||||
|
# ...naming the grouping explicitly.
|
||||||
|
expect_match(rd, "spend_subtype", fixed = TRUE)
|
||||||
|
|
||||||
|
# Pin the mechanism numerically so a future refactor that quietly changes the
|
||||||
|
# quantile grouping fails here rather than silently invalidating the sentence
|
||||||
|
# above. Fixture: Madison, 10 peers found at FY2020, category = NULL.
|
||||||
|
peers <- cog_find_peers("552025209777", year = 2020L, max_peers = 10L)
|
||||||
|
cmp <- cog_peer_compare(target_govid = "552025209777", peers = peers,
|
||||||
|
category = NULL, years = 2020L, per_capita = TRUE)
|
||||||
|
|
||||||
|
naive <- sum(cmp$amt_per_capita_nominal[cmp$role == "summary_p50"], na.rm = TRUE)
|
||||||
|
|
||||||
|
peer_rows <- cmp[cmp$role == "peer", ]
|
||||||
|
per_gov <- tapply(peer_rows$amt_per_capita_nominal, peer_rows$canonical_govid,
|
||||||
|
sum, na.rm = TRUE)
|
||||||
|
correct <- unname(stats::quantile(per_gov, 0.5, na.rm = TRUE))
|
||||||
|
|
||||||
|
expect_equal(round(naive), 6180) # summing the built-in summary rows
|
||||||
|
expect_equal(round(correct), 2043) # quantile of each peer's OWN total
|
||||||
|
expect_gt(naive / correct, 2) # a +200% misstatement on this cohort
|
||||||
|
})
|
||||||
@@ -0,0 +1,55 @@
|
|||||||
|
# Madison walkthrough audit -- finding F-014. Tracked as uscogdata#12.
|
||||||
|
# See docs/walkthroughs/FINDINGS.md in cog_explorer.
|
||||||
|
#
|
||||||
|
# cog_revenue()'s flow_prefixes = c("T","A","U","B","C","D") never returns
|
||||||
|
# item-code prefix X (Employee Retirement) or Y (other Insurance Trust). Per
|
||||||
|
# Census's standard identity, Total Revenue = General + Utility + Liquor Store +
|
||||||
|
# Insurance Trust Revenue, and Employee Retirement System contributions and
|
||||||
|
# earnings ARE the Insurance Trust Revenue component -- so prefix X sits inside
|
||||||
|
# a published Census revenue concept exactly the way I89 sits inside Census's
|
||||||
|
# Direct Expenditure concept (finding F-012).
|
||||||
|
#
|
||||||
|
# CAVEAT FOR WHOEVER PICKS THIS UP: the argument name below (`revenue_concept =
|
||||||
|
# "total"`) is this test's *proposal*, not a settled decision. The owner's
|
||||||
|
# 2026-07-28 resolution covers expenditure concepts only; no revenue-side
|
||||||
|
# naming has been ruled on. If the eventual argument is named differently,
|
||||||
|
# change the two calls here -- the asserted dollar invariants are what matter
|
||||||
|
# and are independent of the naming.
|
||||||
|
#
|
||||||
|
# Fixture reproducibility: Madison's own X-prefix revenue (FY1970-FY1986,
|
||||||
|
# $15,098,000 nominal, $0 thereafter) is outside the bundled fixture's year
|
||||||
|
# window (2011/2012/2019/2020), so the same invariant is asserted on Wisconsin
|
||||||
|
# state government FY2012, where the fixture carries nonzero X01/X05/X08.
|
||||||
|
|
||||||
|
test_that("cog_revenue() can return Census Total Revenue including Insurance Trust (prefix X)", {
|
||||||
|
testthat::skip("Blocked on uscogdata#12 (finding F-014)")
|
||||||
|
|
||||||
|
wi_state <- "550000227544" # WISCONSIN (state government)
|
||||||
|
|
||||||
|
# Revenue-shaped Employee Retirement codes, read from the RAW corpus rather
|
||||||
|
# than through cog_revenue(), which is the filter under test:
|
||||||
|
# X01 local employee contribution, X04/X05 contributions and transfers from
|
||||||
|
# other governments, X08 earnings on investments.
|
||||||
|
x_revenue <- wt_raw_amt(wi_state, 2012L, codes = c("X01", "X04", "X05", "X08"))
|
||||||
|
expect_equal(x_revenue, 2038800) # 615,835 + 0 + 560,382 + 862,583 ($1,000s)
|
||||||
|
|
||||||
|
general <- cog_revenue(govid = wi_state, years = 2012L)
|
||||||
|
expect_equal(sum(general$amt_nominal), 31338293000)
|
||||||
|
|
||||||
|
total <- cog_revenue(govid = wi_state, years = 2012L, revenue_concept = "total")
|
||||||
|
expect_equal(sum(total$amt_nominal) - sum(general$amt_nominal), x_revenue * 1000)
|
||||||
|
expect_equal(sum(total$amt_nominal), 33377093000)
|
||||||
|
expect_true(all(c("X01", "X05", "X08") %in% wt_codes_included(total)))
|
||||||
|
|
||||||
|
# Sibling codes under the SAME first letter must stay out: X11/X12 are
|
||||||
|
# benefit payments (an expenditure) and X21/X30/X47 are cash and securities
|
||||||
|
# holdings (a balance-sheet stock). This is the F-018 point restated on the
|
||||||
|
# revenue side -- the split has to come from the crosswalk's spend_type, not
|
||||||
|
# from the letter X.
|
||||||
|
expect_false(any(c("X11", "X12", "X21", "X30", "X47") %in% wt_codes_included(total)))
|
||||||
|
|
||||||
|
# Every returned row still resolves to a category. summary_categories has
|
||||||
|
# zero rows for prefix X today, so relaxing the prefix filter alone would
|
||||||
|
# produce category = NA rows -- see census_of_governments_finance_pipeline#60.
|
||||||
|
expect_false(any(is.na(total$category)))
|
||||||
|
})
|
||||||
@@ -244,24 +244,42 @@ test_that("basis defaults to 'harmonized' when not passed", {
|
|||||||
test_that("provenance carries basis + harmonization block with na_rows_excluded", {
|
test_that("provenance carries basis + harmonization block with na_rows_excluded", {
|
||||||
skip_if_no_corpus()
|
skip_if_no_corpus()
|
||||||
with_fixture_corpus({
|
with_fixture_corpus({
|
||||||
r <- cog_spending("121011212191", 2011:2012, "Corrections")
|
# FL state government. The harmonization block is scoped by government,
|
||||||
|
# year and flow prefix -- NOT by category -- so a Corrections query still
|
||||||
|
# counts every E/F/G-prefixed row the harmonized basis drops for having
|
||||||
|
# no harmonized_code. The three that apply here are E21/F21/G21
|
||||||
|
# (Education NEC, SB184-186, "discontinued_na", wide-era window ending
|
||||||
|
# FY2011); the other discontinued_na rulings live outside E/F/G.
|
||||||
|
# See docs/phase_r_harmonization_review.md § 1.3/1.4 and cog_pipeline
|
||||||
|
# data/harmonization_map.csv.
|
||||||
|
r <- cog_spending("120000226351", 2011:2012, "Corrections")
|
||||||
prov <- attr(r, "provenance")
|
prov <- attr(r, "provenance")
|
||||||
expect_equal(prov$basis, "harmonized")
|
expect_equal(prov$basis, "harmonized")
|
||||||
expect_true(prov$harmonization$applied)
|
expect_true(prov$harmonization$applied)
|
||||||
expect_true(prov$harmonization$na_rows_excluded >= 0L)
|
|
||||||
expect_true(prov$harmonization$na_amount_excluded >= 0)
|
|
||||||
# Data-verified for the v6 fixture (corpus 2026-07-22). The Task 18 map
|
|
||||||
# extension added E/F/G-prefix discontinued_na rulings the earlier pin's
|
|
||||||
# comment predated: E21/F21/G21 (Education NEC local, SB184-186,
|
|
||||||
# "trivial; explicit-NA, full wide-era window"). Broward's 2011 legacy
|
|
||||||
# partition zero-pads exactly those three codes, so this query now
|
|
||||||
# excludes 3 NA-harmonized rows -- all with amt = 0, hence the excluded
|
|
||||||
# AMOUNT stays exactly zero. (The other discontinued_na rulings -- S74,
|
|
||||||
# Z61, X04, X06, the debt-detail family, L24 -- remain outside the
|
|
||||||
# E/F/G/K prefixes.) See docs/phase_r_harmonization_review.md § 1.3/1.4
|
|
||||||
# and cog_pipeline data/harmonization_map.csv E21/F21/G21 rows.
|
|
||||||
expect_equal(prov$harmonization$na_rows_excluded, 3L)
|
expect_equal(prov$harmonization$na_rows_excluded, 3L)
|
||||||
expect_equal(prov$harmonization$na_amount_excluded, 0)
|
# $2,825,439 thousands of FY2011 E21 + F21 + G21, reported in full USD.
|
||||||
|
# Pinning a non-zero amount is the point: the earlier Broward anchor's
|
||||||
|
# three rows were all explicit zeros, so the AMOUNT accounting was
|
||||||
|
# asserted only against 0 and could not have caught a bug.
|
||||||
|
expect_equal(prov$harmonization$na_amount_excluded, 2825439 * 1000)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("sparsification removed the wide era's zero-pads from the exclusion count", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
# Broward County FY2011 used to carry E21/F21/G21 rows of exactly $0 --
|
||||||
|
# the wide era stored every government x every code, zeros included. The
|
||||||
|
# published corpus no longer does (SB194, cog_pipeline#64), so there is
|
||||||
|
# now nothing for the harmonized basis to exclude. Absence in a
|
||||||
|
# dense_source year means Census published $0; it does not mean the
|
||||||
|
# exclusion machinery stopped working, which the FL state anchor above
|
||||||
|
# proves independently.
|
||||||
|
r <- cog_spending("121011212191", 2011:2012, "Corrections")
|
||||||
|
h <- attr(r, "provenance")$harmonization
|
||||||
|
expect_true(h$applied)
|
||||||
|
expect_equal(h$na_rows_excluded, 0L)
|
||||||
|
expect_equal(h$na_amount_excluded, 0)
|
||||||
})
|
})
|
||||||
})
|
})
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user