diff --git a/NEWS.md b/NEWS.md index eea3134..5ef31bb 100644 --- a/NEWS.md +++ b/NEWS.md @@ -1,5 +1,37 @@ # uscogdata 0.1.0 (development) +## `complete = TRUE`: absent cells, labelled with why they are absent + +* `cog_spending()` and `cog_revenue()` gain `complete`, defaulting to `FALSE` + (today's behaviour). With `complete = TRUE` the requested grid is filled + from the corpus's `code_set` table and every row carries a new + `value_source` column: + + | `value_source` | meaning | `amt_nominal` | + |---|---|---| + | `reported` | the corpus carries this cell | as published | + | `census_zero` | dense-source year (≤ FY2011), cell absent — Census published `$0` | `0` | + | `not_reported` | sparse-source year (≥ FY2012), cell absent — unknown | `NA` | + + The `NA` is deliberate and is the whole point: filling a modern absence + with `0` would invent data, which is precisely the error the corpus's + representation contract exists to prevent. +* This restores information the reader lost when the corpus was sparsified + (`SB194`, cog_pipeline#64) — a wide-era query whose cells were all `$0` + had begun returning nothing at all — and improves on what came before it, + since the pre-sparsification corpus could not distinguish a published zero + from an unreported cell either. +* The grid is scoped to each government's **own type**, so a county is never + filled with cells only a state can report. +* Needs a corpus published from 2026-07-29 onward (when `representation` and + `code_set` began shipping); aborts with class + `uscogdata_representation_unavailable` otherwise. Gated on the manifest + listing those tables rather than on `schema_version`, which was never + bumped for the change. Not available with `recipe` or + `expenditure_concept = "total"` — neither draws its cells from `code_set`. +* `provenance$completion` reports `applied`, `rows_filled`, and the per-year + `absence_means` rule; `cog_explain()` prints a "Completion" section. + ## Corpus-wide series breaks now reach users (`corpus_break_refs`) * Four catalogued series breaks carry `fin_code = "ALL"` — caveats about the diff --git a/R/complete.R b/R/complete.R new file mode 100644 index 0000000..b2c6cfc --- /dev/null +++ b/R/complete.R @@ -0,0 +1,148 @@ +# R/complete.R +# +# `complete = TRUE` on the money verbs. Fills the requested grid so that a +# cell the corpus does not carry still appears, labelled with WHY it is +# missing. +# +# The corpus stopped storing the wide era's explicit zeros +# (cog_pipeline#64, series break SB194), which made absence ambiguous: +# +# <= FY2011 dense_source absent => Census published $0 (census_zero) +# >= FY2012 sparse_source absent => not reported, unknown (not_reported) +# +# Before sparsification a wide-era query whose cells were all $0 came back as +# explicit $0 rows; afterwards it came back empty, with nothing to say which +# of the two meanings applied. This restores that -- and improves on it, +# because the pre-sparsification corpus could not distinguish the two either. +# +# `census_zero` fills carry `amt_nominal = 0`; `not_reported` fills carry NA. +# That difference is the entire point: writing 0 into a modern absence would +# invent data, which is the error the representation contract exists to stop. + +#' @noRd +.abort_complete_unsupported <- function(reason, alternative) { + cli::cli_abort(c( + "{.code complete = TRUE} is not supported for this query.", + x = reason, + i = alternative + ), class = "uscogdata_complete_unsupported") +} + +#' @noRd +.require_representation <- function(con, manifest) { + needed <- c("representation.parquet", "code_set.parquet") + missing <- needed[!vapply(needed, function(f) .corpus_has_table(manifest, f), + logical(1))] + if (length(missing) == 0L) return(invisible(TRUE)) + cli::cli_abort(c( + "This corpus does not publish the representation contract.", + x = "Missing: {.file {missing}}.", + i = "{.code complete = TRUE} needs those tables to know whether an absent cell means Census published $0 or means the government did not report.", + i = "They ship with corpora published from 2026-07-29 onward; re-point {.envvar USCOGDATA_URL} at a current corpus, or omit {.code complete}." + ), class = "uscogdata_representation_unavailable") +} + +#' The cells a government-year COULD carry: every code in force for that +#' government's own type, mapped through `summary_categories`, restricted to +#' the calling verb's flow prefixes and (when given) its category filter. +#' +#' Scoped by `govs_type` deliberately. Filling against the union of all types +#' would invent cells that the government can never report -- a county row for +#' "state IG transfer to school districts" -- and those inventions would then +#' be indistinguishable from real census zeros. +#' +#' `NOT cs.is_aggregate` mirrors `spending_long` / `revenue_long`, which drop +#' aggregate rows. Without it the grid would offer cells the verb structurally +#' never returns, so every one of them would fill as a phantom $0. +#' @noRd +.completion_grid_sql <- function(subtype_col, govid, years, category, + flow_prefixes) { + category_pred <- if (is.null(category)) { + "" + } else { + sprintf("AND c.category IN (%s)", .sql_lit_chr(category)) + } + sprintf( + "SELECT DISTINCT + cs.year, + x.canonical_govid, + x.gov_name, + c.%1$s AS subtype_value, + c.category, + r.absence_means + FROM code_set cs + JOIN canonical_fips_xwalk x ON x.govs_type = cs.type + JOIN summary_categories c ON c.item_code = cs.item_code + JOIN representation r ON r.year = cs.year + WHERE x.canonical_govid IN (%2$s) + AND cs.year IN (%3$s) + AND NOT cs.is_aggregate + AND LEFT(cs.item_code, 1) IN (%4$s) + AND c.category IS NOT NULL + AND c.%1$s IS NOT NULL + %5$s", + subtype_col, .sql_lit_chr(govid), + paste(as.integer(years), collapse = ","), + .sql_lit_chr(flow_prefixes), category_pred + ) +} + +#' Fill `result` out to the full grid, stamping `value_source` on every row. +#' +#' Returns the completed tibble with a `.completion` attribute carrying the +#' provenance block. Reported rows are passed through untouched -- filling +#' must never alter or drop what the corpus actually published. +#' @noRd +.complete_result <- function(result, con, subtype_col, govid, years, category, + flow_prefixes) { + grid <- tibble::as_tibble(DBI::dbGetQuery( + con, .completion_grid_sql(subtype_col, govid, years, category, flow_prefixes) + )) + + result$value_source <- rep("reported", nrow(result)) + if (nrow(grid) == 0L) { + attr(result, ".completion") <- list( + applied = TRUE, rows_filled = 0L, absence_means = list() + ) + return(result) + } + + names(grid)[names(grid) == "subtype_value"] <- subtype_col + key <- function(d) { + paste(d$year, d$canonical_govid, d[[subtype_col]], d$category, sep = "\r") + } + missing <- grid[!key(grid) %in% key(result), , drop = FALSE] + + if (nrow(missing) > 0L) { + filled <- tibble::tibble( + year = as.integer(missing$year), + canonical_govid = as.character(missing$canonical_govid), + gov_name = as.character(missing$gov_name), + category = as.character(missing$category), + # census_zero is a value Census published; not_reported is unknown and + # must stay NA. Collapsing the two to 0 is the defect, not the fill. + amt_nominal = ifelse(missing$absence_means == "census_zero", + 0, NA_real_), + codes_included = NA_character_, + aggregate_fallback = NA, + value_source = as.character(missing$absence_means) + ) + filled[[subtype_col]] <- as.character(missing[[subtype_col]]) + if ("notes" %in% names(result)) filled$notes <- NA_character_ + + result <- dplyr::bind_rows(result, filled) + result <- result[order(result$year, result$canonical_govid, + result[[subtype_col]], result$category), , + drop = FALSE] + } + + rules <- unique(grid[, c("year", "absence_means")]) + attr(result, ".completion") <- list( + applied = TRUE, + rows_filled = nrow(missing), + absence_means = stats::setNames( + as.list(as.character(rules$absence_means)), as.character(rules$year) + ) + ) + result +} diff --git a/R/explain.R b/R/explain.R index aedbda4..55d2c34 100644 --- a/R/explain.R +++ b/R/explain.R @@ -118,6 +118,24 @@ cog_explain <- function(result, format = c("print", "list")) { cli::cli_ul(sugg_lines) } + if (isTRUE(prov$completion$applied)) { + cli::cli_h2("Completion") + cli::cli_text( + "Filled {prov$completion$rows_filled} absent cell(s) from the corpus code set." + ) + rules <- prov$completion$absence_means + if (length(rules) > 0L) { + cli::cli_ul(vapply(names(rules), function(y) { + sprintf("%s: an absent cell means %s", y, + if (identical(rules[[y]], "census_zero")) { + "Census published $0 (filled as 0)" + } else { + "the government did not report (filled as NA, not 0)" + }) + }, character(1))) + } + } + if (length(prov$series_break_refs) > 0L) { cli::cli_h2("Series breaks") cli::cli_ul(.series_break_story_lines(prov$series_break_refs)) diff --git a/R/provenance.R b/R/provenance.R index b521423..7b46220 100644 --- a/R/provenance.R +++ b/R/provenance.R @@ -10,7 +10,8 @@ expenditure_concept_note = NA_character_, expenditure_concept_direct_suppressed = FALSE, harmonization = NULL, recipe = NULL, - suggestions = list()) { + suggestions = list(), + completion = NULL) { manifest <- .uscogdata_env$manifest codes <- result[["codes_included"]] @@ -126,6 +127,13 @@ ), series_break_refs = break_refs, corpus_break_refs = corpus_refs, + # What `complete = TRUE` filled, and the rule it filled by. Always + # present so a consumer can read `completion$applied` without testing + # for the key -- an absent block and applied = FALSE would otherwise be + # indistinguishable from an older reader version. + completion = completion %||% list( + applied = FALSE, rows_filled = 0L, absence_means = list() + ), manifest = list( schema_version = as.integer(manifest$schema_version), pipeline_commit = manifest$pipeline_commit %||% NA_character_, diff --git a/R/revenue.R b/R/revenue.R index ed71d08..eb11dd0 100644 --- a/R/revenue.R +++ b/R/revenue.R @@ -11,11 +11,13 @@ #' @return Tibble with columns `year`, `canonical_govid`, `gov_name`, #' `revenue_subtype`, `category`, `amt_nominal`, optional `amt_real`, #' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`, -#' optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`. +#' optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`, +#' and `value_source` when `complete = TRUE`. #' @export cog_revenue <- function(govid, years, category = NULL, per_capita = FALSE, adjust_to_year = NULL, - basis = c("harmonized", "raw"), recipe = NULL) { + basis = c("harmonized", "raw"), recipe = NULL, + complete = FALSE) { .verb_spendrev( verb = "cog_revenue", view_base = "revenue_annotated", @@ -28,6 +30,7 @@ cog_revenue <- function(govid, years, category = NULL, per_capita = per_capita, adjust_to_year = adjust_to_year, basis = basis, - recipe = recipe + recipe = recipe, + complete = complete ) } diff --git a/R/spending.R b/R/spending.R index 80a9064..7ab4b5f 100644 --- a/R/spending.R +++ b/R/spending.R @@ -70,16 +70,42 @@ #' `provenance$expenditure_concept_direct_suppressed` is `TRUE` -- the #' figure in those rows is the intergovernmental leg alone, not Direct + #' IG. +#' @param complete If `TRUE`, fill the requested grid so that a cell the +#' corpus does not carry still appears, labelled with **why** it is +#' missing, and add a `value_source` column to every row: +#' +#' * `"reported"` — the corpus carries this cell. +#' * `"census_zero"` — dense-source year (`<= FY2011`), cell absent: +#' Census published `$0`. `amt_nominal` is `0`. +#' * `"not_reported"` — sparse-source year (`>= FY2012`), cell absent: the +#' government did not report, and the value is unknown. `amt_nominal` is +#' `NA`, **not** `0` — writing a zero there would invent data. +#' +#' The grid comes from the corpus's `code_set` table, scoped to each +#' government's own type, so a county is never filled with cells only a +#' state can report. Reported rows are passed through untouched. +#' +#' Defaults to `FALSE` (the historical behaviour: absent cells simply do +#' not appear). Needs a corpus published from 2026-07-29 onward, which is +#' when `representation`/`code_set` began shipping; aborts with class +#' `uscogdata_representation_unavailable` otherwise. Not available with +#' `recipe` or with `expenditure_concept = "total"` (class +#' `uscogdata_complete_unsupported`) — neither draws its cells from +#' `code_set`. #' @return Tibble with columns `year`, `canonical_govid`, `gov_name`, #' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real`, #' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`, -#' optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`. -#' Carries a `provenance` attribute matching `inst/schemas/provenance-v1.json`. +#' optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`, +#' and `value_source` when `complete = TRUE`. +#' Carries a `provenance` attribute matching `inst/schemas/provenance-v1.json`, +#' whose `completion` block reports `applied`, `rows_filled`, and the +#' per-year `absence_means` rule that was applied. #' @export cog_spending <- function(govid, years, category = NULL, per_capita = FALSE, adjust_to_year = NULL, basis = c("harmonized", "raw"), recipe = NULL, - expenditure_concept = c("direct", "total")) { + expenditure_concept = c("direct", "total"), + complete = FALSE) { .verb_spendrev( verb = "cog_spending", view_base = "spending_annotated", @@ -93,7 +119,8 @@ cog_spending <- function(govid, years, category = NULL, adjust_to_year = adjust_to_year, basis = basis, recipe = recipe, - expenditure_concept = expenditure_concept + expenditure_concept = expenditure_concept, + complete = complete ) } @@ -118,7 +145,8 @@ cog_spending <- function(govid, years, category = NULL, govid, years, category, per_capita, adjust_to_year, basis = c("harmonized", "raw"), recipe = NULL, - expenditure_concept = c("direct", "total")) { + expenditure_concept = c("direct", "total"), + complete = FALSE) { basis_explicit <- length(basis) == 1L basis <- match.arg(basis, c("harmonized", "raw")) # match.arg() itself throws a base `simpleError`, not an rlang-classed @@ -165,12 +193,27 @@ cog_spending <- function(govid, years, category = NULL, ) } + complete <- isTRUE(complete) + if (complete && !is.null(recipe)) { + .abort_complete_unsupported( + "A recipe defines its own component codes and never goes through `summary_categories`, so there is no grid to fill from.", + "Query the recipe without `complete`, or use a category query with `complete = TRUE`." + ) + } + if (complete && identical(expenditure_concept, "total")) { + .abort_complete_unsupported( + "The intergovernmental leg deliberately keeps aggregate-flagged rows (see `inst/sql/24-ig_long.sql`), so its cells are not the ones `code_set` describes.", + "Use `expenditure_concept = \"direct\"` with `complete = TRUE`, or drop `complete`." + ) + } + years <- as.integer(years) if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year) con <- .ensure_session() manifest <- .uscogdata_env$manifest scope <- .check_govids_in_scope(govid) + if (complete) .require_representation(con, manifest) resolved <- .resolve_basis(basis, basis_explicit, manifest) @@ -201,6 +244,18 @@ cog_spending <- function(govid, years, category = NULL, result <- tibble::as_tibble(DBI::dbGetQuery(con, sql)) } + # Fill BEFORE per_capita / inflation so the added cells get the same + # treatment as reported ones: a census_zero stays $0 per capita and in real + # dollars, and a not_reported stays NA through both rather than becoming a + # spurious 0. + completion <- list(applied = FALSE, rows_filled = 0L, absence_means = list()) + if (complete) { + result <- .complete_result(result, con, subtype_col, govid, years, + category, flow_prefixes) + completion <- attr(result, ".completion") + attr(result, ".completion") <- NULL + } + if (per_capita) result <- .attach_per_capita(result, con, govid) if (!is.null(adjust_to_year)) { result <- .attach_real_dollars(result, adjust_to_year, per_capita) @@ -309,7 +364,8 @@ cog_spending <- function(govid, years, category = NULL, expenditure_concept_direct_suppressed = direct_suppressed_flag, harmonization = harmonization, recipe = recipe_block, - suggestions = suggestions + suggestions = suggestions, + completion = completion ) prov$scope$govids_found <- scope$found prov$scope$govids_missing <- scope$missing diff --git a/R/views.R b/R/views.R index c5febdd..172d6e9 100644 --- a/R/views.R +++ b/R/views.R @@ -32,6 +32,29 @@ "45-ig_annotated_harmonized.sql" ) +# The representation contract (cog_pipeline#64): two parquet tables that say +# what an ABSENT cell means in a given year. Gated on manifest PRESENCE, not +# on schema_version, because the sparsification that introduced them did not +# bump the version -- the pre-sparsification corpus this package shipped +# against until 2026-07-30 was already schema v6 and carried neither table. +# Keying off the version number would therefore register a view over a file +# that does not exist and fail at CREATE VIEW time on exactly the corpora this +# check exists to tolerate. +.representation_view_files <- c( + "36-representation.sql" = "representation.parquet", + "37-code_set.sql" = "code_set.parquet" +) + +#' Does the mounted corpus publish `file` (e.g. "code_set.parquet")? +#' Reads the manifest's metadata list rather than stat-ing the URL, so it +#' works identically for a local fixture and a remote share. +#' @noRd +.corpus_has_table <- function(manifest, file) { + paths <- vapply(manifest$files$metadata %||% list(), + function(f) as.character(f$path %||% ""), character(1)) + file %in% basename(paths) +} + #' Register DuckDB views from inst/sql/ SQL files #' @noRd .register_views <- function(con, url, manifest) { @@ -39,7 +62,10 @@ files <- sort(list.files(sql_dir, pattern = "\\.sql$", full.names = TRUE)) schema_version <- suppressWarnings(as.integer(manifest$schema_version %||% 0L)) for (f in files) { - if (basename(f) %in% .harmonization_view_files && schema_version < 5L) next + base <- basename(f) + if (base %in% .harmonization_view_files && schema_version < 5L) next + if (base %in% names(.representation_view_files) && + !.corpus_has_table(manifest, .representation_view_files[[base]])) next sql <- paste(readLines(f, warn = FALSE), collapse = "\n") sql <- gsub("\\{url\\}", url, sql, fixed = FALSE) DBI::dbExecute(con, sql) diff --git a/inst/schemas/provenance-v1.json b/inst/schemas/provenance-v1.json index 814b29b..6307533 100644 --- a/inst/schemas/provenance-v1.json +++ b/inst/schemas/provenance-v1.json @@ -33,6 +33,15 @@ "aggregate_fallback": { "type": ["object", "null"] }, "transformations":{ "type": "object" }, "series_break_refs": { "type": "array", "items": { "type": "string" } }, + "completion": { + "type": "object", + "description": "What `complete = TRUE` filled. `applied` is FALSE on an ordinary query. `rows_filled` counts cells added to the requested grid, and `absence_means` maps each requested year to the meaning of an absent cell there ('census_zero' in a dense_source year, 'not_reported' in a sparse_source one). Filled rows carry `value_source` in the result: 'reported', 'census_zero' (amount 0 -- Census published $0), or 'not_reported' (amount NA -- unknown).", + "properties": { + "applied": { "type": "boolean" }, + "rows_filled": { "type": "integer" }, + "absence_means": { "type": "object" } + } + }, "corpus_break_refs": { "type": "array", "items": { "type": "string" }, diff --git a/inst/sql/36-representation.sql b/inst/sql/36-representation.sql new file mode 100644 index 0000000..ac878c3 --- /dev/null +++ b/inst/sql/36-representation.sql @@ -0,0 +1,3 @@ +CREATE OR REPLACE VIEW representation AS +SELECT * +FROM read_parquet('{url}data/representation.parquet'); diff --git a/inst/sql/37-code_set.sql b/inst/sql/37-code_set.sql new file mode 100644 index 0000000..8905bc8 --- /dev/null +++ b/inst/sql/37-code_set.sql @@ -0,0 +1,3 @@ +CREATE OR REPLACE VIEW code_set AS +SELECT * +FROM read_parquet('{url}data/code_set.parquet'); diff --git a/man/cog_revenue.Rd b/man/cog_revenue.Rd index 843d48e..b018fef 100644 --- a/man/cog_revenue.Rd +++ b/man/cog_revenue.Rd @@ -11,7 +11,8 @@ cog_revenue( per_capita = FALSE, adjust_to_year = NULL, basis = c("harmonized", "raw"), - recipe = NULL + recipe = NULL, + complete = FALSE ) } \arguments{ @@ -55,12 +56,36 @@ argument is ignored and the result's provenance reports `basis = "recipe"` with an inert `harmonization` block (`applied = FALSE`, pointing at the `recipe` block instead) rather than a possibly-misleading `"harmonized"`/`"raw"` value.} + +\item{complete}{If `TRUE`, fill the requested grid so that a cell the + corpus does not carry still appears, labelled with **why** it is + missing, and add a `value_source` column to every row: + + * `"reported"` — the corpus carries this cell. + * `"census_zero"` — dense-source year (`<= FY2011`), cell absent: + Census published `$0`. `amt_nominal` is `0`. + * `"not_reported"` — sparse-source year (`>= FY2012`), cell absent: the + government did not report, and the value is unknown. `amt_nominal` is + `NA`, **not** `0` — writing a zero there would invent data. + + The grid comes from the corpus's `code_set` table, scoped to each + government's own type, so a county is never filled with cells only a + state can report. Reported rows are passed through untouched. + + Defaults to `FALSE` (the historical behaviour: absent cells simply do + not appear). Needs a corpus published from 2026-07-29 onward, which is + when `representation`/`code_set` began shipping; aborts with class + `uscogdata_representation_unavailable` otherwise. Not available with + `recipe` or with `expenditure_concept = "total"` (class + `uscogdata_complete_unsupported`) — neither draws its cells from + `code_set`.} } \value{ Tibble with columns `year`, `canonical_govid`, `gov_name`, `revenue_subtype`, `category`, `amt_nominal`, optional `amt_real`, optional `amt_per_capita_nominal`, optional `amt_per_capita_real`, - optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`. + optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`, + and `value_source` when `complete = TRUE`. } \description{ Mirror of [cog_spending()] for revenue categories. One row per diff --git a/man/cog_spending.Rd b/man/cog_spending.Rd index 45f8d11..9638efa 100644 --- a/man/cog_spending.Rd +++ b/man/cog_spending.Rd @@ -12,7 +12,8 @@ cog_spending( adjust_to_year = NULL, basis = c("harmonized", "raw"), recipe = NULL, - expenditure_concept = c("direct", "total") + expenditure_concept = c("direct", "total"), + complete = FALSE ) } \arguments{ @@ -58,40 +59,66 @@ FALSE`, pointing at the `recipe` block instead) rather than a possibly-misleading `"harmonized"`/`"raw"` value.} \item{expenditure_concept}{`"direct"` (default) returns only the -government's own direct spending (item codes `E`/`F`/`G`), unchanged -from prior releases. `"total"` additionally UNIONs in the -intergovernmental leg -- payments to local governments (`M` codes) and -to the state government (`L` codes, excluding the `L--` family-total -rollup) -- so results gain rows with `spend_subtype == -"intergovernmental"`. Requires the active corpus's `summary_categories` -to carry M/L rows (added by cog_pipeline PR #59); aborts with class -`uscogdata_ig_categories_unsupported` on an older corpus rather than -silently under-reporting. Mutually exclusive with `recipe` (a recipe -already defines its own component codes). **Do not sum `"total"` -results across levels of government** (e.g. state + county + city): -a state's `M12` payment to a school district is the same dollar the -district reports as its own direct `E12`, so summing both double-counts -it. This matters in particular with [cog_geographic_rollup()], which -sums across exactly that kind of multi-layer government set. + government's own direct spending (item codes `E`/`F`/`G`), unchanged + from prior releases. `"total"` additionally UNIONs in the + intergovernmental leg -- payments to local governments (`M` codes) and + to the state government (`L` codes, excluding the `L--` family-total + rollup) -- so results gain rows with `spend_subtype == + "intergovernmental"`. Requires the active corpus's `summary_categories` + to carry M/L rows (added by cog_pipeline PR #59); aborts with class + `uscogdata_ig_categories_unsupported` on an older corpus rather than + silently under-reporting. Mutually exclusive with `recipe` (a recipe + already defines its own component codes). **Do not sum `"total"` + results across levels of government** (e.g. state + county + city): + a state's `M12` payment to a school district is the same dollar the + district reports as its own direct `E12`, so summing both double-counts + it. This matters in particular with [cog_geographic_rollup()], which + sums across exactly that kind of multi-layer government set. -In the legacy wide era (<= FY2011), some functions are published ONLY -as an aggregate-flagged family total (e.g. Corrections' `E04`/`E05` -split), which the Direct leg excludes by construction but the IG leg -deliberately keeps (see `inst/sql/24-ig_long.sql`). For a `"total"` -query, any (year, category) where this leaves intergovernmental rows -with NO Direct counterpart is flagged: the affected rows' `notes` -name the harmonization recipe that recovers the missing Direct -component (when one exists), and -`provenance$expenditure_concept_direct_suppressed` is `TRUE` -- the -figure in those rows is the intergovernmental leg alone, not Direct + -IG.} + In the legacy wide era (<= FY2011), some functions are published ONLY + as an aggregate-flagged family total (e.g. Corrections' `E04`/`E05` + split), which the Direct leg excludes by construction but the IG leg + deliberately keeps (see `inst/sql/24-ig_long.sql`). For a `"total"` + query, any (year, category) where this leaves intergovernmental rows + with NO Direct counterpart is flagged: the affected rows' `notes` + name the harmonization recipe that recovers the missing Direct + component (when one exists), and + `provenance$expenditure_concept_direct_suppressed` is `TRUE` -- the + figure in those rows is the intergovernmental leg alone, not Direct + + IG.} + +\item{complete}{If `TRUE`, fill the requested grid so that a cell the + corpus does not carry still appears, labelled with **why** it is + missing, and add a `value_source` column to every row: + + * `"reported"` — the corpus carries this cell. + * `"census_zero"` — dense-source year (`<= FY2011`), cell absent: + Census published `$0`. `amt_nominal` is `0`. + * `"not_reported"` — sparse-source year (`>= FY2012`), cell absent: the + government did not report, and the value is unknown. `amt_nominal` is + `NA`, **not** `0` — writing a zero there would invent data. + + The grid comes from the corpus's `code_set` table, scoped to each + government's own type, so a county is never filled with cells only a + state can report. Reported rows are passed through untouched. + + Defaults to `FALSE` (the historical behaviour: absent cells simply do + not appear). Needs a corpus published from 2026-07-29 onward, which is + when `representation`/`code_set` began shipping; aborts with class + `uscogdata_representation_unavailable` otherwise. Not available with + `recipe` or with `expenditure_concept = "total"` (class + `uscogdata_complete_unsupported`) — neither draws its cells from + `code_set`.} } \value{ Tibble with columns `year`, `canonical_govid`, `gov_name`, `spend_subtype`, `category`, `amt_nominal`, optional `amt_real`, optional `amt_per_capita_nominal`, optional `amt_per_capita_real`, - optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`. - Carries a `provenance` attribute matching `inst/schemas/provenance-v1.json`. + optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`, + and `value_source` when `complete = TRUE`. + Carries a `provenance` attribute matching `inst/schemas/provenance-v1.json`, + whose `completion` block reports `applied`, `rows_filled`, and the + per-year `absence_means` rule that was applied. } \description{ One row per `(year, canonical_govid, spend_subtype, category)`. Amounts are diff --git a/tests/testthat/helper-fixture.R b/tests/testthat/helper-fixture.R index 47f0726..3ea3744 100644 --- a/tests/testthat/helper-fixture.R +++ b/tests/testthat/helper-fixture.R @@ -85,6 +85,42 @@ with_doctored_schema_version <- function(version, code) { force(code) } +# Copy the bundled fixture to a temp dir with representation.parquet and +# code_set.parquet removed (and dropped from the manifest's metadata list), +# then run `code` against it. Models a corpus published BEFORE sparsification: +# schema_version is left alone deliberately, because it was never bumped for +# that change -- the pre-sparsification fixture this package shipped until +# 2026-07-30 was schema v6 and carried neither table. Presence in the manifest +# is therefore the only honest signal, and this helper is what proves the +# package keys off it rather than off the version number. +with_corpus_missing_representation <- function(code) { + src <- fixture_corpus_path() + tmp <- withr::local_tempdir(.local_envir = parent.frame()) + file.copy(list.files(src, full.names = TRUE), tmp, recursive = TRUE) + + dropped <- c("representation.parquet", "code_set.parquet") + file.remove(file.path(tmp, "data", dropped)) + + manifest_path <- file.path(tmp, "manifest.json") + m <- jsonlite::fromJSON(manifest_path, simplifyVector = FALSE) + m$files$metadata <- Filter( + function(f) !basename(f$path) %in% dropped, m$files$metadata + ) + writeLines( + jsonlite::toJSON(m, auto_unbox = TRUE, pretty = TRUE, null = "null"), + manifest_path + ) + + old_url <- Sys.getenv("USCOGDATA_URL", unset = NA) + uscogdata:::cog_close() + Sys.setenv(USCOGDATA_URL = paste0(tmp, "/")) + on.exit({ + uscogdata:::cog_close() + if (is.na(old_url)) Sys.unsetenv("USCOGDATA_URL") else Sys.setenv(USCOGDATA_URL = old_url) + }, add = TRUE) + force(code) +} + # Copy the bundled fixture to a temp dir with summary_categories.parquet # rewritten to drop every M/L (intergovernmental) row, then run `code` # against it with a clean session (mirrors with_fixture_corpus()/ diff --git a/tests/testthat/test-complete.R b/tests/testthat/test-complete.R new file mode 100644 index 0000000..4a2c771 --- /dev/null +++ b/tests/testthat/test-complete.R @@ -0,0 +1,188 @@ +# tests/testthat/test-complete.R +# +# uscogdata#18. The published corpus no longer stores the wide era's explicit +# zeros (cog_pipeline#64, series break SB194), so absence means two different +# things: +# +# <= FY2011 (dense_source) : cell absent => Census published $0 +# >= FY2012 (sparse_source): cell absent => not reported, unknown +# +# `complete = TRUE` fills the requested grid from `code_set` and stamps every +# row's `value_source` so the two are distinguishable. Expected row sets here +# are built from the corpus parquet directly, never from the verb under test -- +# verifying what a filter does through that same filter proves nothing. + +# The (subtype, category) cells that SHOULD exist for one government-year: +# every code in force for that government's type, mapped through +# summary_categories, matching the verb's flow prefixes and excluding +# aggregate-flagged codes (which spending_long/revenue_long drop). +raw_expected_cells <- function(govid, year, prefixes, subtype_col) { + fx <- sub("/$", "", Sys.getenv("USCOGDATA_URL")) + q <- function(f) sprintf("read_parquet('%s/data/%s')", fx, f) + wt_raw_query(sprintf( + "SELECT DISTINCT c.%s AS subtype, c.category + FROM %s cs + JOIN %s x ON x.govs_type = cs.type + JOIN %s c ON c.item_code = cs.item_code + WHERE x.canonical_govid = '%s' + AND cs.year = %d + AND NOT cs.is_aggregate + AND LEFT(cs.item_code, 1) IN (%s) + AND c.category IS NOT NULL + AND c.%s IS NOT NULL", + subtype_col, q("code_set.parquet"), q("canonical_fips_xwalk.parquet"), + q("summary_categories.parquet"), govid, year, + paste0("'", prefixes, "'", collapse = ","), subtype_col + )) +} + +test_that("complete = FALSE is the default and changes nothing", { + skip_if_no_corpus() + with_fixture_corpus({ + plain <- cog_spending("121011212191", 2011L) + explicit <- cog_spending("121011212191", 2011L, complete = FALSE) + expect_equal(nrow(plain), nrow(explicit)) + expect_false("value_source" %in% names(plain)) + }) +}) + +test_that("complete = TRUE round-trips a dense-source year to the pre-sparsification cells", { + skip_if_no_corpus() + with_fixture_corpus({ + # FY2011 is dense_source: before sparsification this government carried a + # row for every code in force, most of them $0. complete = TRUE must + # reproduce that cell set exactly. + r <- cog_spending("121011212191", 2011L, complete = TRUE) + expected <- raw_expected_cells("121011212191", 2011L, + c("E", "F", "G"), "spend_subtype") + + key <- function(sub, cat) paste(sub, cat, sep = "|") + expect_setequal(key(r$spend_subtype, r$category), + key(expected$subtype, expected$category)) + expect_gt(nrow(expected), 0L) + + # Every filled cell in a dense-source year is a Census-published $0 -- + # never "unknown", which is what the modern era's absences mean. + expect_setequal(unique(r$value_source), c("reported", "census_zero")) + expect_true(all(r$amt_nominal[r$value_source == "census_zero"] == 0)) + expect_true(all(r$amt_nominal[r$value_source == "reported"] != 0)) + }) +}) + +test_that("complete = TRUE preserves the reported rows and their amounts exactly", { + skip_if_no_corpus() + with_fixture_corpus({ + plain <- cog_spending("121011212191", 2011L) + full <- cog_spending("121011212191", 2011L, complete = TRUE) + + # Filling adds rows; it must never alter or drop one. + expect_gt(nrow(full), nrow(plain)) + reported <- full[full$value_source == "reported", ] + expect_equal(nrow(reported), nrow(plain)) + expect_equal(sum(reported$amt_nominal), sum(plain$amt_nominal)) + # ... and the total is unchanged, because every added cell is $0. + expect_equal(sum(full$amt_nominal, na.rm = TRUE), sum(plain$amt_nominal)) + }) +}) + +test_that("a sparse-source year's absences are unknown, not zero", { + skip_if_no_corpus() + with_fixture_corpus({ + # FY2019 is sparse_source: an absent cell means the government did not + # report, which is NOT a zero. Filling those with 0 would invent data -- + # the exact error the representation contract exists to prevent. + r <- cog_spending("121011212191", 2019L, complete = TRUE) + filled <- r[r$value_source != "reported", ] + expect_gt(nrow(filled), 0L) + expect_true(all(filled$value_source == "not_reported")) + expect_true(all(is.na(filled$amt_nominal))) + expect_false(any(r$value_source == "census_zero")) + }) +}) + +test_that("the fill is scoped to each government's own type", { + skip_if_no_corpus() + with_fixture_corpus({ + # Filling against the union of all types would invent cells for codes a + # county can never report. Every filled category must be one that + # code_set puts in force for type 1 (county) specifically. + r <- cog_spending("121011212191", 2011L, complete = TRUE) + county_cells <- raw_expected_cells("121011212191", 2011L, + c("E", "F", "G"), "spend_subtype") + expect_true(all(r$category %in% county_cells$category)) + }) +}) + +test_that("complete = TRUE respects the category filter", { + skip_if_no_corpus() + with_fixture_corpus({ + r <- cog_spending("121011212191", 2011L, category = "Police", + complete = TRUE) + expect_true(all(r$category == "Police")) + expect_true("value_source" %in% names(r)) + }) +}) + +test_that("cog_revenue() completes on its own flow", { + skip_if_no_corpus() + with_fixture_corpus({ + r <- cog_revenue("121011212191", 2011L, complete = TRUE) + expected <- raw_expected_cells("121011212191", 2011L, + c("T", "A", "U", "B", "C", "D"), + "revenue_subtype") + key <- function(sub, cat) paste(sub, cat, sep = "|") + expect_setequal(key(r$revenue_subtype, r$category), + key(expected$subtype, expected$category)) + expect_setequal(unique(r$value_source), c("reported", "census_zero")) + }) +}) + +test_that("provenance records the completion and its absence rule", { + skip_if_no_corpus() + with_fixture_corpus({ + prov <- attr(cog_spending("121011212191", 2011L, complete = TRUE), + "provenance") + expect_true(prov$completion$applied) + expect_equal(prov$completion$absence_means$`2011`, "census_zero") + expect_gt(prov$completion$rows_filled, 0L) + + off <- attr(cog_spending("121011212191", 2011L), "provenance") + expect_false(off$completion$applied) + expect_equal(off$completion$rows_filled, 0L) + }) +}) + +test_that("complete = TRUE is refused where the fill would be guesswork", { + skip_if_no_corpus() + with_fixture_corpus({ + # A recipe defines its own component codes and does not go through + # summary_categories at all, so there is no grid to fill from. + expect_error( + cog_spending("121011212191", 2011L, recipe = "corrections_combined", + complete = TRUE), + class = "uscogdata_complete_unsupported" + ) + # The intergovernmental leg keeps aggregate rows by design + # (inst/sql/24-ig_long.sql), so its grid is not code_set's grid. + expect_error( + cog_spending("121011212191", 2011L, expenditure_concept = "total", + complete = TRUE), + class = "uscogdata_complete_unsupported" + ) + }) +}) + +test_that("complete = TRUE aborts on a corpus with no representation contract", { + skip_if_no_corpus() + # A corpus published before sparsification carries neither table, so there + # is nothing to fill from and no rule saying what an absence means. That + # must abort rather than guess. + with_corpus_missing_representation({ + expect_error( + cog_spending("121011212191", 2011L, complete = TRUE), + class = "uscogdata_representation_unavailable" + ) + # ... while an ordinary query on the same corpus still works. + expect_gt(nrow(cog_spending("121011212191", 2011L)), 0L) + }) +})