diff --git a/NEWS.md b/NEWS.md index 6def2c7..eea3134 100644 --- a/NEWS.md +++ b/NEWS.md @@ -1,5 +1,25 @@ # uscogdata 0.1.0 (development) +## Corpus-wide series breaks now reach users (`corpus_break_refs`) + +* Four catalogued series breaks carry `fin_code = "ALL"` — caveats about the + corpus as a whole rather than about one item code. `series_break_refs` is + built by matching `fin_code` against the item codes in the result, and no + row's `item_code` is ever the literal `"ALL"`, so **none of them could ever + be surfaced**: `SB085` (dollar precision across the 1976/1977 boundary), + `SB087` (imputation exclusion from FY2002), `SB194` (the dense → sparse + representation change at FY2012) and `SB086` (the government id scheme + change at FY2017). +* Provenance gains `corpus_break_refs`, selected on the break-year window + alone and disjoint from `series_break_refs` by construction, so a consumer + can tell a whole-result caveat from a break in one series. `cog_explain()` + prints them under their own "Corpus-wide caveats" heading. cog-api passes + provenance through verbatim, so the field appears there without an API + change. +* `SB194` is the one that made this urgent: a query spanning FY2011 → FY2012 + crosses the boundary where an absent cell stops meaning "Census published + `$0`" and starts meaning "not reported", and until now nothing said so. + ## Bundled fixture regenerated against the sparsified corpus * `inst/extdata/fixture_corpus/` now tracks the corpus published on diff --git a/R/explain.R b/R/explain.R index f7dd5b9..aedbda4 100644 --- a/R/explain.R +++ b/R/explain.R @@ -123,6 +123,14 @@ cog_explain <- function(result, format = c("print", "list")) { cli::cli_ul(.series_break_story_lines(prov$series_break_refs)) } + # Kept in a section of its own: these qualify the whole result, so folding + # them in with the per-code breaks above would invite reading them as a + # caveat about one series. + if (length(prov$corpus_break_refs) > 0L) { + cli::cli_h2("Corpus-wide caveats") + cli::cli_ul(.series_break_story_lines(prov$corpus_break_refs)) + } + cli::cli_h2("Transformations") uc <- prov$transformations$units_conversion if (isTRUE(uc$applied)) { diff --git a/R/provenance.R b/R/provenance.R index e8c6d65..b521423 100644 --- a/R/provenance.R +++ b/R/provenance.R @@ -37,11 +37,20 @@ schema_version <- suppressWarnings(as.integer(manifest$schema_version %||% 0L)) con <- .uscogdata_env$con - break_refs <- if (!is.null(con) && DBI::dbIsValid(con)) { + have_con <- !is.null(con) && DBI::dbIsValid(con) + break_refs <- if (have_con) { .build_series_break_refs(con, codes_observed, years, schema_version) } else { character(0) } + # Corpus-wide caveats travel separately: they qualify the whole result + # rather than one series, and they do not depend on codes_observed (see + # .build_corpus_break_refs()). + corpus_refs <- if (have_con) { + .build_corpus_break_refs(con, years, schema_version) + } else { + character(0) + } list( verb = verb, @@ -116,6 +125,7 @@ ) ), series_break_refs = break_refs, + corpus_break_refs = corpus_refs, manifest = list( schema_version = as.integer(manifest$schema_version), pipeline_commit = manifest$pipeline_commit %||% NA_character_, diff --git a/R/series_breaks.R b/R/series_breaks.R index d4f0c1c..f7ff77c 100644 --- a/R/series_breaks.R +++ b/R/series_breaks.R @@ -14,9 +14,41 @@ sql <- sprintf( "SELECT DISTINCT break_id FROM series_breaks_pq - WHERE fin_code IN (%s) AND break_year BETWEEN %d AND %d + WHERE fin_code IN (%s) AND fin_code <> 'ALL' + AND break_year BETWEEN %d AND %d ORDER BY break_id", .sql_lit_chr(codes_observed), min(as.integer(years)), max(as.integer(years)) ) DBI::dbGetQuery(con, sql)$break_id } + +#' Corpus-wide caveats: catalogued breaks whose `fin_code` is the literal +#' `"ALL"` rather than an item code. They qualify the whole result, so they +#' cannot be matched the way `.build_series_break_refs()` matches -- no row's +#' `item_code` is ever `"ALL"`, which is exactly why they reached no user +#' before uscogdata#19. Selection is on the break_year window alone: which +#' codes a result happens to contain is irrelevant to a caveat about the +#' corpus. +#' +#' All four catalogued entries are *boundary* caveats (dollar precision +#' across 1976/1977, imputation exclusion from 2002, the dense -> sparse +#' representation change at 2012, the id scheme change at 2017), so the same +#' `break_year BETWEEN min(years) AND max(years)` rule the code-specific +#' path uses is the right one -- a request that never crosses the boundary +#' is not affected by it. +#' +#' Returned separately from `series_break_refs` so a consumer can tell a +#' whole-result caveat from a break in one series; the two are disjoint by +#' construction. +#' @noRd +.build_corpus_break_refs <- function(con, years, schema_version) { + if (schema_version < 5L || length(years) == 0L) return(character(0)) + sql <- sprintf( + "SELECT DISTINCT break_id + FROM series_breaks_pq + WHERE fin_code = 'ALL' AND break_year BETWEEN %d AND %d + ORDER BY break_id", + min(as.integer(years)), max(as.integer(years)) + ) + DBI::dbGetQuery(con, sql)$break_id +} diff --git a/inst/schemas/provenance-v1.json b/inst/schemas/provenance-v1.json index d83a57b..814b29b 100644 --- a/inst/schemas/provenance-v1.json +++ b/inst/schemas/provenance-v1.json @@ -33,6 +33,11 @@ "aggregate_fallback": { "type": ["object", "null"] }, "transformations":{ "type": "object" }, "series_break_refs": { "type": "array", "items": { "type": "string" } }, + "corpus_break_refs": { + "type": "array", + "items": { "type": "string" }, + "description": "Ids of catalogued series breaks whose fin_code is the literal 'ALL' -- caveats about the corpus as a whole (dollar precision across 1976/1977, imputation exclusion from 2002, the dense -> sparse representation change at 2012, the government id scheme change at 2017) rather than about one item code. Selected on the break_year window alone, so they do not depend on which codes a result contains. Disjoint from series_break_refs by construction: an entry qualifies the whole result, not one series." + }, "manifest": { "type": "object" }, "sql_query": { "type": "string" } } diff --git a/tests/testthat/test-corpus-breaks.R b/tests/testthat/test-corpus-breaks.R new file mode 100644 index 0000000..01aea84 --- /dev/null +++ b/tests/testthat/test-corpus-breaks.R @@ -0,0 +1,94 @@ +# tests/testthat/test-corpus-breaks.R +# +# uscogdata#19. Four catalogued series breaks carry fin_code = "ALL" -- they +# are caveats about the corpus itself rather than about one item code: +# +# SB085 1977 dollar precision across the 1976/1977 boundary +# SB087 2002 imputation exclusion FY2002-2006 +# SB194 2012 dense -> sparse representation change +# SB086 2017 government ID scheme change +# +# .build_series_break_refs() matches `fin_code IN ()`, +# and no row's item_code is ever the literal "ALL", so none of them could +# ever reach a user. They now travel in their own provenance field, +# `corpus_break_refs`, which keeps them distinguishable from the +# code-specific `series_break_refs` (an ALL caveat qualifies the whole +# result, not one series). + +test_that("corpus_break_refs surfaces an ALL-scoped break the year range spans", { + skip_if_no_corpus() + with_fixture_corpus({ + # SB194 sits at FY2012 -- the dense/sparse boundary. A query spanning + # 2011 -> 2012 straddles it, and this is the case cog_pipeline#64's + # DoD 4 intended to reach users. + r <- cog_spending("121011212191", 2011:2012, "Police") + prov <- attr(r, "provenance") + expect_true("SB194" %in% prov$corpus_break_refs) + }) +}) + +test_that("corpus_break_refs stays empty when no ALL break falls in the range", { + skip_if_no_corpus() + with_fixture_corpus({ + # 2019-2020 spans no catalogued corpus-wide break. + r <- cog_spending("121011212191", 2019:2020, "Police") + expect_equal(attr(r, "provenance")$corpus_break_refs, character(0)) + }) +}) + +test_that("corpus_break_refs and series_break_refs stay disjoint", { + skip_if_no_corpus() + with_fixture_corpus({ + r <- cog_spending("121011212191", 2011:2012, "Police") + prov <- attr(r, "provenance") + expect_type(prov$series_break_refs, "character") + expect_type(prov$corpus_break_refs, "character") + # An ALL caveat must never masquerade as a break in a specific series. + expect_length(intersect(prov$series_break_refs, prov$corpus_break_refs), 0L) + expect_false("SB194" %in% prov$series_break_refs) + }) +}) + +test_that(".build_corpus_break_refs matches on the break_year window alone", { + skip_if_no_corpus() + con <- cog_open() + on.exit(cog_close()) + + # SB085's boundary is 1976/1977, outside the fixture's partitions -- the + # series_breaks table is a full cross-vintage registry, so the matching + # logic is testable there even though no long partition covers it. + expect_true("SB085" %in% uscogdata:::.build_corpus_break_refs( + con, years = 1975:1980, schema_version = 6L + )) + # ... and does not fire for a range that misses it, unlike a filter keyed + # on the era rather than the boundary. + expect_false("SB085" %in% uscogdata:::.build_corpus_break_refs( + con, years = 1978:1980, schema_version = 6L + )) + + # Unlike code-specific refs, these do not depend on which codes a result + # happens to contain -- that dependency is the whole defect. + expect_setequal( + uscogdata:::.build_corpus_break_refs(con, years = 2001:2003, schema_version = 6L), + "SB087" + ) + + # Gated on schema_version >= 5: series_breaks_pq is not registered below it. + expect_equal( + uscogdata:::.build_corpus_break_refs(con, years = 2011:2012, schema_version = 4L), + character(0) + ) +}) + +test_that("cog_explain() prints corpus-wide caveats under their own heading", { + skip_if_no_corpus() + with_fixture_corpus({ + r <- cog_spending("121011212191", 2011:2012, "Police") + out <- paste(c( + capture.output(cog_explain(r)), + capture.output(cog_explain(r), type = "message") + ), collapse = "\n") + expect_match(out, "Corpus-wide caveats", fixed = TRUE) + expect_match(out, "SB194", fixed = TRUE) + }) +}) diff --git a/tests/testthat/test-spending.R b/tests/testthat/test-spending.R index 8897c50..30a9f91 100644 --- a/tests/testthat/test-spending.R +++ b/tests/testthat/test-spending.R @@ -321,11 +321,13 @@ test_that("provenance$series_break_refs is a populated-when-applicable character r <- cog_spending("121011212191", 2020L, "Corrections") refs <- attr(r, "provenance")$series_break_refs expect_type(refs, "character") - # No catalogued series_breaks_pq row falls inside this fixture's - # 2011/2012/2019/2020 window for the codes this query touches (E04/G04) - # -- data-verified; the mechanism itself is what's under test here, via - # a query-shaped unit test in test-views.R since the fixture has no - # positive case to pin against. + # No catalogued code-specific series_breaks_pq row falls inside this + # fixture's 2011/2012/2019/2020 window for the codes this query touches + # (E04/G04) -- data-verified; the mechanism itself is what's under test + # here, via a query-shaped unit test in test-views.R since the fixture + # has no positive case to pin against. Corpus-wide ("ALL") entries never + # appear in this field by construction -- they travel in + # corpus_break_refs; see test-corpus-breaks.R. expect_equal(refs, character(0)) }) }) diff --git a/tests/testthat/test-views.R b/tests/testthat/test-views.R index 3bbfb88..587a610 100644 --- a/tests/testthat/test-views.R +++ b/tests/testthat/test-views.R @@ -177,11 +177,13 @@ test_that("inst/sql/24- and 25- IG views retain aggregates, COALESCE NULL harmon }) test_that(".build_series_break_refs matches fin_code + break_year window", { - # No series_breaks_pq row falls inside the bundled fixture's 2011-2020 - # window (data-verified; see the "series_break_refs" test in + # No CODE-SPECIFIC series_breaks_pq row falls inside the bundled fixture's + # 2011-2020 window (data-verified; see the "series_break_refs" test in # test-spending.R), so this proves the matching logic itself against the # live view + a synthetic year window that DOES hit a cataloged break - # (SB075, fin_code E62, break_year 2005). + # (SB075, fin_code E62, break_year 2005). The corpus-wide entries are a + # separate path with its own coverage -- SB194 does sit at 2012, inside + # the fixture window; see test-corpus-breaks.R. skip_if_no_corpus() con <- cog_open() on.exit(cog_close())