.build_series_break_refs() matches `fin_code IN (<codes in the result>)`. No row's item_code is ever the literal "ALL", so the four corpus-wide entries could never match and reached no user: SB085 1977 dollar precision across the 1976/1977 boundary SB087 2002 imputation exclusion FY2002-2006 SB194 2012 dense -> sparse representation change SB086 2017 government id scheme change SB194 is why this matters now. cog_pipeline#64 DoD 4 was "series_breaks.csv carries an ALL @ 2012 entry describing the representation change, SO cog_explain() surfaces it". The entry shipped; the reader dropped it. A query spanning FY2011 -> FY2012 crosses the boundary where an absent cell stops meaning "Census published $0" and starts meaning "not reported", and nothing said so. Provenance gains `corpus_break_refs`, built by .build_corpus_break_refs() on the break_year window alone -- which codes a result happens to contain is irrelevant to a caveat about the corpus. A separate field rather than more entries in series_break_refs, because an ALL caveat qualifies the whole result and folding the two together invites reading it as a caveat about one series; .build_series_break_refs() now excludes 'ALL' explicitly so the two stay disjoint by construction. cog_explain() prints them under their own "Corpus-wide caveats" heading, and cog-api passes provenance through verbatim, so the field reaches the API with no change there. On the year rule: all four entries are BOUNDARY caveats -- their own join_advice speaks of crossing 1976/1977, of FY2002-2006, of absence not being comparable across FY2012, of pre- vs post-2017 ids -- so the same `break_year BETWEEN min(years) AND max(years)` rule the code-specific path uses is the right one, and matches the issue's DoD 1. The issue's DoD 3 also asks that a FY2011 query surface SB085; that cannot hold under DoD 1 and does not hold under any reading of SB085's text, whose boundary is 1976/1977. Tested with a range that actually spans it, and flagged on the issue. Stacked on fix/regen-fixture-corpus-18: SB194 does not exist in main's bundled fixture, which predates the break being catalogued. Suite: 606 pass / 0 fail / 6 skip (was 594/0/6). cog-api 357 / 0 / 8, unchanged.
137 lines
4.5 KiB
R
137 lines
4.5 KiB
R
# R/provenance.R
|
|
# Shared provenance construction. Matches inst/schemas/provenance-v1.json.
|
|
|
|
#' @noRd
|
|
.build_provenance <- function(verb, call, govid, years, category,
|
|
per_capita, adjust_to_year, result, sql,
|
|
subtype_col, basis = NA_character_,
|
|
basis_note = NA_character_,
|
|
expenditure_concept = "direct",
|
|
expenditure_concept_note = NA_character_,
|
|
expenditure_concept_direct_suppressed = FALSE,
|
|
harmonization = NULL, recipe = NULL,
|
|
suggestions = list()) {
|
|
manifest <- .uscogdata_env$manifest
|
|
|
|
codes <- result[["codes_included"]]
|
|
codes_observed <- if (length(codes) == 0L) {
|
|
character(0)
|
|
} else {
|
|
sorted <- sort(unique(unlist(strsplit(codes, ",", fixed = TRUE))))
|
|
sorted[nzchar(sorted)]
|
|
}
|
|
|
|
agg_flag <- result[["aggregate_fallback"]]
|
|
agg_applied <- isTRUE(any(agg_flag, na.rm = TRUE))
|
|
agg_years <- if (agg_applied) {
|
|
unique(as.integer(result$year[which(agg_flag)]))
|
|
} else {
|
|
integer(0)
|
|
}
|
|
|
|
gov_names <- if (nrow(result) == 0L) {
|
|
character(0)
|
|
} else {
|
|
unique(result$gov_name)
|
|
}
|
|
|
|
schema_version <- suppressWarnings(as.integer(manifest$schema_version %||% 0L))
|
|
con <- .uscogdata_env$con
|
|
have_con <- !is.null(con) && DBI::dbIsValid(con)
|
|
break_refs <- if (have_con) {
|
|
.build_series_break_refs(con, codes_observed, years, schema_version)
|
|
} else {
|
|
character(0)
|
|
}
|
|
# Corpus-wide caveats travel separately: they qualify the whole result
|
|
# rather than one series, and they do not depend on codes_observed (see
|
|
# .build_corpus_break_refs()).
|
|
corpus_refs <- if (have_con) {
|
|
.build_corpus_break_refs(con, years, schema_version)
|
|
} else {
|
|
character(0)
|
|
}
|
|
|
|
list(
|
|
verb = verb,
|
|
call = paste(deparse(call), collapse = " "),
|
|
target = list(
|
|
canonical_govid = as.character(govid),
|
|
gov_name = gov_names
|
|
),
|
|
years = as.integer(years),
|
|
category = category,
|
|
basis = basis,
|
|
basis_note = basis_note,
|
|
expenditure_concept = expenditure_concept,
|
|
expenditure_concept_note = expenditure_concept_note,
|
|
expenditure_concept_direct_suppressed = isTRUE(expenditure_concept_direct_suppressed),
|
|
harmonization = harmonization %||% list(
|
|
applied = FALSE, na_rows_excluded = 0L, na_amount_excluded = 0,
|
|
note = NA_character_
|
|
),
|
|
recipe = recipe,
|
|
suggestions = suggestions,
|
|
scope = list(
|
|
gov_types_included = as.integer(unlist(manifest$scope$gov_types_included)),
|
|
gov_types_excluded = as.integer(unlist(manifest$scope$gov_types_excluded)),
|
|
scope_note = manifest$scope$scope_note %||% ""
|
|
),
|
|
codes_summed = list(
|
|
observed = codes_observed,
|
|
subtype_column = subtype_col
|
|
),
|
|
aggregate_fallback = list(
|
|
applied = agg_applied,
|
|
years = agg_years
|
|
),
|
|
transformations = list(
|
|
units_conversion = list(
|
|
applied = TRUE,
|
|
source_unit = "$1,000s (raw Census)",
|
|
target_unit = "$USD",
|
|
multiplier = 1000L
|
|
),
|
|
per_capita = list(
|
|
applied = isTRUE(per_capita),
|
|
denominator_source = if (isTRUE(per_capita)) {
|
|
"Census F-33 population (per-year, from long.population)"
|
|
} else {
|
|
NA_character_
|
|
},
|
|
popyear_range = if (isTRUE(per_capita)) {
|
|
attr(result, ".popyear_range") %||% integer(0)
|
|
} else {
|
|
integer(0)
|
|
},
|
|
pop_source_counts = if (isTRUE(per_capita)) {
|
|
ps <- result[["pop_source"]]
|
|
if (is.null(ps) || length(ps) == 0L) {
|
|
list(census_f33 = 0L, unavailable = 0L)
|
|
} else {
|
|
list(
|
|
census_f33 = sum(ps == "census_f33", na.rm = TRUE),
|
|
unavailable = sum(ps == "unavailable", na.rm = TRUE)
|
|
)
|
|
}
|
|
} else {
|
|
NULL
|
|
}
|
|
),
|
|
inflation = list(
|
|
applied = !is.null(adjust_to_year),
|
|
base_year = if (is.null(adjust_to_year)) NA_integer_ else as.integer(adjust_to_year),
|
|
index = if (is.null(adjust_to_year)) NA_character_ else "CPI-U (BLS CPIAUCSL annual average, bundled)"
|
|
)
|
|
),
|
|
series_break_refs = break_refs,
|
|
corpus_break_refs = corpus_refs,
|
|
manifest = list(
|
|
schema_version = as.integer(manifest$schema_version),
|
|
pipeline_commit = manifest$pipeline_commit %||% NA_character_,
|
|
built_at = manifest$built_at %||% NA_character_
|
|
),
|
|
sql_query = sql
|
|
)
|
|
}
|