Merge pull request 'fix: surface ALL-scoped series breaks in provenance (#19)' (#21) from fix/all-scoped-series-breaks-19 into main
R-CMD-check / check (push) Successful in 3m23s

This commit was merged in pull request #21.
This commit is contained in:
2026-07-30 10:33:24 -04:00
8 changed files with 183 additions and 10 deletions
+20
View File
@@ -1,5 +1,25 @@
# uscogdata 0.1.0 (development) # uscogdata 0.1.0 (development)
## Corpus-wide series breaks now reach users (`corpus_break_refs`)
* Four catalogued series breaks carry `fin_code = "ALL"` — caveats about the
corpus as a whole rather than about one item code. `series_break_refs` is
built by matching `fin_code` against the item codes in the result, and no
row's `item_code` is ever the literal `"ALL"`, so **none of them could ever
be surfaced**: `SB085` (dollar precision across the 1976/1977 boundary),
`SB087` (imputation exclusion from FY2002), `SB194` (the dense → sparse
representation change at FY2012) and `SB086` (the government id scheme
change at FY2017).
* Provenance gains `corpus_break_refs`, selected on the break-year window
alone and disjoint from `series_break_refs` by construction, so a consumer
can tell a whole-result caveat from a break in one series. `cog_explain()`
prints them under their own "Corpus-wide caveats" heading. cog-api passes
provenance through verbatim, so the field appears there without an API
change.
* `SB194` is the one that made this urgent: a query spanning FY2011 → FY2012
crosses the boundary where an absent cell stops meaning "Census published
`$0`" and starts meaning "not reported", and until now nothing said so.
## Bundled fixture regenerated against the sparsified corpus ## Bundled fixture regenerated against the sparsified corpus
* `inst/extdata/fixture_corpus/` now tracks the corpus published on * `inst/extdata/fixture_corpus/` now tracks the corpus published on
+8
View File
@@ -123,6 +123,14 @@ cog_explain <- function(result, format = c("print", "list")) {
cli::cli_ul(.series_break_story_lines(prov$series_break_refs)) cli::cli_ul(.series_break_story_lines(prov$series_break_refs))
} }
# Kept in a section of its own: these qualify the whole result, so folding
# them in with the per-code breaks above would invite reading them as a
# caveat about one series.
if (length(prov$corpus_break_refs) > 0L) {
cli::cli_h2("Corpus-wide caveats")
cli::cli_ul(.series_break_story_lines(prov$corpus_break_refs))
}
cli::cli_h2("Transformations") cli::cli_h2("Transformations")
uc <- prov$transformations$units_conversion uc <- prov$transformations$units_conversion
if (isTRUE(uc$applied)) { if (isTRUE(uc$applied)) {
+11 -1
View File
@@ -37,11 +37,20 @@
schema_version <- suppressWarnings(as.integer(manifest$schema_version %||% 0L)) schema_version <- suppressWarnings(as.integer(manifest$schema_version %||% 0L))
con <- .uscogdata_env$con con <- .uscogdata_env$con
break_refs <- if (!is.null(con) && DBI::dbIsValid(con)) { have_con <- !is.null(con) && DBI::dbIsValid(con)
break_refs <- if (have_con) {
.build_series_break_refs(con, codes_observed, years, schema_version) .build_series_break_refs(con, codes_observed, years, schema_version)
} else { } else {
character(0) character(0)
} }
# Corpus-wide caveats travel separately: they qualify the whole result
# rather than one series, and they do not depend on codes_observed (see
# .build_corpus_break_refs()).
corpus_refs <- if (have_con) {
.build_corpus_break_refs(con, years, schema_version)
} else {
character(0)
}
list( list(
verb = verb, verb = verb,
@@ -116,6 +125,7 @@
) )
), ),
series_break_refs = break_refs, series_break_refs = break_refs,
corpus_break_refs = corpus_refs,
manifest = list( manifest = list(
schema_version = as.integer(manifest$schema_version), schema_version = as.integer(manifest$schema_version),
pipeline_commit = manifest$pipeline_commit %||% NA_character_, pipeline_commit = manifest$pipeline_commit %||% NA_character_,
+33 -1
View File
@@ -14,9 +14,41 @@
sql <- sprintf( sql <- sprintf(
"SELECT DISTINCT break_id "SELECT DISTINCT break_id
FROM series_breaks_pq FROM series_breaks_pq
WHERE fin_code IN (%s) AND break_year BETWEEN %d AND %d WHERE fin_code IN (%s) AND fin_code <> 'ALL'
AND break_year BETWEEN %d AND %d
ORDER BY break_id", ORDER BY break_id",
.sql_lit_chr(codes_observed), min(as.integer(years)), max(as.integer(years)) .sql_lit_chr(codes_observed), min(as.integer(years)), max(as.integer(years))
) )
DBI::dbGetQuery(con, sql)$break_id DBI::dbGetQuery(con, sql)$break_id
} }
#' Corpus-wide caveats: catalogued breaks whose `fin_code` is the literal
#' `"ALL"` rather than an item code. They qualify the whole result, so they
#' cannot be matched the way `.build_series_break_refs()` matches -- no row's
#' `item_code` is ever `"ALL"`, which is exactly why they reached no user
#' before uscogdata#19. Selection is on the break_year window alone: which
#' codes a result happens to contain is irrelevant to a caveat about the
#' corpus.
#'
#' All four catalogued entries are *boundary* caveats (dollar precision
#' across 1976/1977, imputation exclusion from 2002, the dense -> sparse
#' representation change at 2012, the id scheme change at 2017), so the same
#' `break_year BETWEEN min(years) AND max(years)` rule the code-specific
#' path uses is the right one -- a request that never crosses the boundary
#' is not affected by it.
#'
#' Returned separately from `series_break_refs` so a consumer can tell a
#' whole-result caveat from a break in one series; the two are disjoint by
#' construction.
#' @noRd
.build_corpus_break_refs <- function(con, years, schema_version) {
if (schema_version < 5L || length(years) == 0L) return(character(0))
sql <- sprintf(
"SELECT DISTINCT break_id
FROM series_breaks_pq
WHERE fin_code = 'ALL' AND break_year BETWEEN %d AND %d
ORDER BY break_id",
min(as.integer(years)), max(as.integer(years))
)
DBI::dbGetQuery(con, sql)$break_id
}
+5
View File
@@ -33,6 +33,11 @@
"aggregate_fallback": { "type": ["object", "null"] }, "aggregate_fallback": { "type": ["object", "null"] },
"transformations":{ "type": "object" }, "transformations":{ "type": "object" },
"series_break_refs": { "type": "array", "items": { "type": "string" } }, "series_break_refs": { "type": "array", "items": { "type": "string" } },
"corpus_break_refs": {
"type": "array",
"items": { "type": "string" },
"description": "Ids of catalogued series breaks whose fin_code is the literal 'ALL' -- caveats about the corpus as a whole (dollar precision across 1976/1977, imputation exclusion from 2002, the dense -> sparse representation change at 2012, the government id scheme change at 2017) rather than about one item code. Selected on the break_year window alone, so they do not depend on which codes a result contains. Disjoint from series_break_refs by construction: an entry qualifies the whole result, not one series."
},
"manifest": { "type": "object" }, "manifest": { "type": "object" },
"sql_query": { "type": "string" } "sql_query": { "type": "string" }
} }
+94
View File
@@ -0,0 +1,94 @@
# tests/testthat/test-corpus-breaks.R
#
# uscogdata#19. Four catalogued series breaks carry fin_code = "ALL" -- they
# are caveats about the corpus itself rather than about one item code:
#
# SB085 1977 dollar precision across the 1976/1977 boundary
# SB087 2002 imputation exclusion FY2002-2006
# SB194 2012 dense -> sparse representation change
# SB086 2017 government ID scheme change
#
# .build_series_break_refs() matches `fin_code IN (<codes in the result>)`,
# and no row's item_code is ever the literal "ALL", so none of them could
# ever reach a user. They now travel in their own provenance field,
# `corpus_break_refs`, which keeps them distinguishable from the
# code-specific `series_break_refs` (an ALL caveat qualifies the whole
# result, not one series).
test_that("corpus_break_refs surfaces an ALL-scoped break the year range spans", {
skip_if_no_corpus()
with_fixture_corpus({
# SB194 sits at FY2012 -- the dense/sparse boundary. A query spanning
# 2011 -> 2012 straddles it, and this is the case cog_pipeline#64's
# DoD 4 intended to reach users.
r <- cog_spending("121011212191", 2011:2012, "Police")
prov <- attr(r, "provenance")
expect_true("SB194" %in% prov$corpus_break_refs)
})
})
test_that("corpus_break_refs stays empty when no ALL break falls in the range", {
skip_if_no_corpus()
with_fixture_corpus({
# 2019-2020 spans no catalogued corpus-wide break.
r <- cog_spending("121011212191", 2019:2020, "Police")
expect_equal(attr(r, "provenance")$corpus_break_refs, character(0))
})
})
test_that("corpus_break_refs and series_break_refs stay disjoint", {
skip_if_no_corpus()
with_fixture_corpus({
r <- cog_spending("121011212191", 2011:2012, "Police")
prov <- attr(r, "provenance")
expect_type(prov$series_break_refs, "character")
expect_type(prov$corpus_break_refs, "character")
# An ALL caveat must never masquerade as a break in a specific series.
expect_length(intersect(prov$series_break_refs, prov$corpus_break_refs), 0L)
expect_false("SB194" %in% prov$series_break_refs)
})
})
test_that(".build_corpus_break_refs matches on the break_year window alone", {
skip_if_no_corpus()
con <- cog_open()
on.exit(cog_close())
# SB085's boundary is 1976/1977, outside the fixture's partitions -- the
# series_breaks table is a full cross-vintage registry, so the matching
# logic is testable there even though no long partition covers it.
expect_true("SB085" %in% uscogdata:::.build_corpus_break_refs(
con, years = 1975:1980, schema_version = 6L
))
# ... and does not fire for a range that misses it, unlike a filter keyed
# on the era rather than the boundary.
expect_false("SB085" %in% uscogdata:::.build_corpus_break_refs(
con, years = 1978:1980, schema_version = 6L
))
# Unlike code-specific refs, these do not depend on which codes a result
# happens to contain -- that dependency is the whole defect.
expect_setequal(
uscogdata:::.build_corpus_break_refs(con, years = 2001:2003, schema_version = 6L),
"SB087"
)
# Gated on schema_version >= 5: series_breaks_pq is not registered below it.
expect_equal(
uscogdata:::.build_corpus_break_refs(con, years = 2011:2012, schema_version = 4L),
character(0)
)
})
test_that("cog_explain() prints corpus-wide caveats under their own heading", {
skip_if_no_corpus()
with_fixture_corpus({
r <- cog_spending("121011212191", 2011:2012, "Police")
out <- paste(c(
capture.output(cog_explain(r)),
capture.output(cog_explain(r), type = "message")
), collapse = "\n")
expect_match(out, "Corpus-wide caveats", fixed = TRUE)
expect_match(out, "SB194", fixed = TRUE)
})
})
+7 -5
View File
@@ -321,11 +321,13 @@ test_that("provenance$series_break_refs is a populated-when-applicable character
r <- cog_spending("121011212191", 2020L, "Corrections") r <- cog_spending("121011212191", 2020L, "Corrections")
refs <- attr(r, "provenance")$series_break_refs refs <- attr(r, "provenance")$series_break_refs
expect_type(refs, "character") expect_type(refs, "character")
# No catalogued series_breaks_pq row falls inside this fixture's # No catalogued code-specific series_breaks_pq row falls inside this
# 2011/2012/2019/2020 window for the codes this query touches (E04/G04) # fixture's 2011/2012/2019/2020 window for the codes this query touches
# -- data-verified; the mechanism itself is what's under test here, via # (E04/G04) -- data-verified; the mechanism itself is what's under test
# a query-shaped unit test in test-views.R since the fixture has no # here, via a query-shaped unit test in test-views.R since the fixture
# positive case to pin against. # has no positive case to pin against. Corpus-wide ("ALL") entries never
# appear in this field by construction -- they travel in
# corpus_break_refs; see test-corpus-breaks.R.
expect_equal(refs, character(0)) expect_equal(refs, character(0))
}) })
}) })
+5 -3
View File
@@ -177,11 +177,13 @@ test_that("inst/sql/24- and 25- IG views retain aggregates, COALESCE NULL harmon
}) })
test_that(".build_series_break_refs matches fin_code + break_year window", { test_that(".build_series_break_refs matches fin_code + break_year window", {
# No series_breaks_pq row falls inside the bundled fixture's 2011-2020 # No CODE-SPECIFIC series_breaks_pq row falls inside the bundled fixture's
# window (data-verified; see the "series_break_refs" test in # 2011-2020 window (data-verified; see the "series_break_refs" test in
# test-spending.R), so this proves the matching logic itself against the # test-spending.R), so this proves the matching logic itself against the
# live view + a synthetic year window that DOES hit a cataloged break # live view + a synthetic year window that DOES hit a cataloged break
# (SB075, fin_code E62, break_year 2005). # (SB075, fin_code E62, break_year 2005). The corpus-wide entries are a
# separate path with its own coverage -- SB194 does sit at 2012, inside
# the fixture window; see test-corpus-breaks.R.
skip_if_no_corpus() skip_if_no_corpus()
con <- cog_open() con <- cog_open()
on.exit(cog_close()) on.exit(cog_close())