Merge pull request 'feat: complete = TRUE fills absent cells with their meaning (#18)' (#23) from feat/complete-argument-18 into main
R-CMD-check / check (push) Successful in 4m14s
R-CMD-check / check (push) Successful in 4m14s
Reviewed-on: #23
This commit was merged in pull request #23.
This commit is contained in:
@@ -1,5 +1,37 @@
|
|||||||
# uscogdata 0.1.0 (development)
|
# uscogdata 0.1.0 (development)
|
||||||
|
|
||||||
|
## `complete = TRUE`: absent cells, labelled with why they are absent
|
||||||
|
|
||||||
|
* `cog_spending()` and `cog_revenue()` gain `complete`, defaulting to `FALSE`
|
||||||
|
(today's behaviour). With `complete = TRUE` the requested grid is filled
|
||||||
|
from the corpus's `code_set` table and every row carries a new
|
||||||
|
`value_source` column:
|
||||||
|
|
||||||
|
| `value_source` | meaning | `amt_nominal` |
|
||||||
|
|---|---|---|
|
||||||
|
| `reported` | the corpus carries this cell | as published |
|
||||||
|
| `census_zero` | dense-source year (≤ FY2011), cell absent — Census published `$0` | `0` |
|
||||||
|
| `not_reported` | sparse-source year (≥ FY2012), cell absent — unknown | `NA` |
|
||||||
|
|
||||||
|
The `NA` is deliberate and is the whole point: filling a modern absence
|
||||||
|
with `0` would invent data, which is precisely the error the corpus's
|
||||||
|
representation contract exists to prevent.
|
||||||
|
* This restores information the reader lost when the corpus was sparsified
|
||||||
|
(`SB194`, cog_pipeline#64) — a wide-era query whose cells were all `$0`
|
||||||
|
had begun returning nothing at all — and improves on what came before it,
|
||||||
|
since the pre-sparsification corpus could not distinguish a published zero
|
||||||
|
from an unreported cell either.
|
||||||
|
* The grid is scoped to each government's **own type**, so a county is never
|
||||||
|
filled with cells only a state can report.
|
||||||
|
* Needs a corpus published from 2026-07-29 onward (when `representation` and
|
||||||
|
`code_set` began shipping); aborts with class
|
||||||
|
`uscogdata_representation_unavailable` otherwise. Gated on the manifest
|
||||||
|
listing those tables rather than on `schema_version`, which was never
|
||||||
|
bumped for the change. Not available with `recipe` or
|
||||||
|
`expenditure_concept = "total"` — neither draws its cells from `code_set`.
|
||||||
|
* `provenance$completion` reports `applied`, `rows_filled`, and the per-year
|
||||||
|
`absence_means` rule; `cog_explain()` prints a "Completion" section.
|
||||||
|
|
||||||
## Corpus-wide series breaks now reach users (`corpus_break_refs`)
|
## Corpus-wide series breaks now reach users (`corpus_break_refs`)
|
||||||
|
|
||||||
* Four catalogued series breaks carry `fin_code = "ALL"` — caveats about the
|
* Four catalogued series breaks carry `fin_code = "ALL"` — caveats about the
|
||||||
|
|||||||
+148
@@ -0,0 +1,148 @@
|
|||||||
|
# R/complete.R
|
||||||
|
#
|
||||||
|
# `complete = TRUE` on the money verbs. Fills the requested grid so that a
|
||||||
|
# cell the corpus does not carry still appears, labelled with WHY it is
|
||||||
|
# missing.
|
||||||
|
#
|
||||||
|
# The corpus stopped storing the wide era's explicit zeros
|
||||||
|
# (cog_pipeline#64, series break SB194), which made absence ambiguous:
|
||||||
|
#
|
||||||
|
# <= FY2011 dense_source absent => Census published $0 (census_zero)
|
||||||
|
# >= FY2012 sparse_source absent => not reported, unknown (not_reported)
|
||||||
|
#
|
||||||
|
# Before sparsification a wide-era query whose cells were all $0 came back as
|
||||||
|
# explicit $0 rows; afterwards it came back empty, with nothing to say which
|
||||||
|
# of the two meanings applied. This restores that -- and improves on it,
|
||||||
|
# because the pre-sparsification corpus could not distinguish the two either.
|
||||||
|
#
|
||||||
|
# `census_zero` fills carry `amt_nominal = 0`; `not_reported` fills carry NA.
|
||||||
|
# That difference is the entire point: writing 0 into a modern absence would
|
||||||
|
# invent data, which is the error the representation contract exists to stop.
|
||||||
|
|
||||||
|
#' @noRd
|
||||||
|
.abort_complete_unsupported <- function(reason, alternative) {
|
||||||
|
cli::cli_abort(c(
|
||||||
|
"{.code complete = TRUE} is not supported for this query.",
|
||||||
|
x = reason,
|
||||||
|
i = alternative
|
||||||
|
), class = "uscogdata_complete_unsupported")
|
||||||
|
}
|
||||||
|
|
||||||
|
#' @noRd
|
||||||
|
.require_representation <- function(con, manifest) {
|
||||||
|
needed <- c("representation.parquet", "code_set.parquet")
|
||||||
|
missing <- needed[!vapply(needed, function(f) .corpus_has_table(manifest, f),
|
||||||
|
logical(1))]
|
||||||
|
if (length(missing) == 0L) return(invisible(TRUE))
|
||||||
|
cli::cli_abort(c(
|
||||||
|
"This corpus does not publish the representation contract.",
|
||||||
|
x = "Missing: {.file {missing}}.",
|
||||||
|
i = "{.code complete = TRUE} needs those tables to know whether an absent cell means Census published $0 or means the government did not report.",
|
||||||
|
i = "They ship with corpora published from 2026-07-29 onward; re-point {.envvar USCOGDATA_URL} at a current corpus, or omit {.code complete}."
|
||||||
|
), class = "uscogdata_representation_unavailable")
|
||||||
|
}
|
||||||
|
|
||||||
|
#' The cells a government-year COULD carry: every code in force for that
|
||||||
|
#' government's own type, mapped through `summary_categories`, restricted to
|
||||||
|
#' the calling verb's flow prefixes and (when given) its category filter.
|
||||||
|
#'
|
||||||
|
#' Scoped by `govs_type` deliberately. Filling against the union of all types
|
||||||
|
#' would invent cells that the government can never report -- a county row for
|
||||||
|
#' "state IG transfer to school districts" -- and those inventions would then
|
||||||
|
#' be indistinguishable from real census zeros.
|
||||||
|
#'
|
||||||
|
#' `NOT cs.is_aggregate` mirrors `spending_long` / `revenue_long`, which drop
|
||||||
|
#' aggregate rows. Without it the grid would offer cells the verb structurally
|
||||||
|
#' never returns, so every one of them would fill as a phantom $0.
|
||||||
|
#' @noRd
|
||||||
|
.completion_grid_sql <- function(subtype_col, govid, years, category,
|
||||||
|
flow_prefixes) {
|
||||||
|
category_pred <- if (is.null(category)) {
|
||||||
|
""
|
||||||
|
} else {
|
||||||
|
sprintf("AND c.category IN (%s)", .sql_lit_chr(category))
|
||||||
|
}
|
||||||
|
sprintf(
|
||||||
|
"SELECT DISTINCT
|
||||||
|
cs.year,
|
||||||
|
x.canonical_govid,
|
||||||
|
x.gov_name,
|
||||||
|
c.%1$s AS subtype_value,
|
||||||
|
c.category,
|
||||||
|
r.absence_means
|
||||||
|
FROM code_set cs
|
||||||
|
JOIN canonical_fips_xwalk x ON x.govs_type = cs.type
|
||||||
|
JOIN summary_categories c ON c.item_code = cs.item_code
|
||||||
|
JOIN representation r ON r.year = cs.year
|
||||||
|
WHERE x.canonical_govid IN (%2$s)
|
||||||
|
AND cs.year IN (%3$s)
|
||||||
|
AND NOT cs.is_aggregate
|
||||||
|
AND LEFT(cs.item_code, 1) IN (%4$s)
|
||||||
|
AND c.category IS NOT NULL
|
||||||
|
AND c.%1$s IS NOT NULL
|
||||||
|
%5$s",
|
||||||
|
subtype_col, .sql_lit_chr(govid),
|
||||||
|
paste(as.integer(years), collapse = ","),
|
||||||
|
.sql_lit_chr(flow_prefixes), category_pred
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Fill `result` out to the full grid, stamping `value_source` on every row.
|
||||||
|
#'
|
||||||
|
#' Returns the completed tibble with a `.completion` attribute carrying the
|
||||||
|
#' provenance block. Reported rows are passed through untouched -- filling
|
||||||
|
#' must never alter or drop what the corpus actually published.
|
||||||
|
#' @noRd
|
||||||
|
.complete_result <- function(result, con, subtype_col, govid, years, category,
|
||||||
|
flow_prefixes) {
|
||||||
|
grid <- tibble::as_tibble(DBI::dbGetQuery(
|
||||||
|
con, .completion_grid_sql(subtype_col, govid, years, category, flow_prefixes)
|
||||||
|
))
|
||||||
|
|
||||||
|
result$value_source <- rep("reported", nrow(result))
|
||||||
|
if (nrow(grid) == 0L) {
|
||||||
|
attr(result, ".completion") <- list(
|
||||||
|
applied = TRUE, rows_filled = 0L, absence_means = list()
|
||||||
|
)
|
||||||
|
return(result)
|
||||||
|
}
|
||||||
|
|
||||||
|
names(grid)[names(grid) == "subtype_value"] <- subtype_col
|
||||||
|
key <- function(d) {
|
||||||
|
paste(d$year, d$canonical_govid, d[[subtype_col]], d$category, sep = "\r")
|
||||||
|
}
|
||||||
|
missing <- grid[!key(grid) %in% key(result), , drop = FALSE]
|
||||||
|
|
||||||
|
if (nrow(missing) > 0L) {
|
||||||
|
filled <- tibble::tibble(
|
||||||
|
year = as.integer(missing$year),
|
||||||
|
canonical_govid = as.character(missing$canonical_govid),
|
||||||
|
gov_name = as.character(missing$gov_name),
|
||||||
|
category = as.character(missing$category),
|
||||||
|
# census_zero is a value Census published; not_reported is unknown and
|
||||||
|
# must stay NA. Collapsing the two to 0 is the defect, not the fill.
|
||||||
|
amt_nominal = ifelse(missing$absence_means == "census_zero",
|
||||||
|
0, NA_real_),
|
||||||
|
codes_included = NA_character_,
|
||||||
|
aggregate_fallback = NA,
|
||||||
|
value_source = as.character(missing$absence_means)
|
||||||
|
)
|
||||||
|
filled[[subtype_col]] <- as.character(missing[[subtype_col]])
|
||||||
|
if ("notes" %in% names(result)) filled$notes <- NA_character_
|
||||||
|
|
||||||
|
result <- dplyr::bind_rows(result, filled)
|
||||||
|
result <- result[order(result$year, result$canonical_govid,
|
||||||
|
result[[subtype_col]], result$category), ,
|
||||||
|
drop = FALSE]
|
||||||
|
}
|
||||||
|
|
||||||
|
rules <- unique(grid[, c("year", "absence_means")])
|
||||||
|
attr(result, ".completion") <- list(
|
||||||
|
applied = TRUE,
|
||||||
|
rows_filled = nrow(missing),
|
||||||
|
absence_means = stats::setNames(
|
||||||
|
as.list(as.character(rules$absence_means)), as.character(rules$year)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
result
|
||||||
|
}
|
||||||
+18
@@ -118,6 +118,24 @@ cog_explain <- function(result, format = c("print", "list")) {
|
|||||||
cli::cli_ul(sugg_lines)
|
cli::cli_ul(sugg_lines)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (isTRUE(prov$completion$applied)) {
|
||||||
|
cli::cli_h2("Completion")
|
||||||
|
cli::cli_text(
|
||||||
|
"Filled {prov$completion$rows_filled} absent cell(s) from the corpus code set."
|
||||||
|
)
|
||||||
|
rules <- prov$completion$absence_means
|
||||||
|
if (length(rules) > 0L) {
|
||||||
|
cli::cli_ul(vapply(names(rules), function(y) {
|
||||||
|
sprintf("%s: an absent cell means %s", y,
|
||||||
|
if (identical(rules[[y]], "census_zero")) {
|
||||||
|
"Census published $0 (filled as 0)"
|
||||||
|
} else {
|
||||||
|
"the government did not report (filled as NA, not 0)"
|
||||||
|
})
|
||||||
|
}, character(1)))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
if (length(prov$series_break_refs) > 0L) {
|
if (length(prov$series_break_refs) > 0L) {
|
||||||
cli::cli_h2("Series breaks")
|
cli::cli_h2("Series breaks")
|
||||||
cli::cli_ul(.series_break_story_lines(prov$series_break_refs))
|
cli::cli_ul(.series_break_story_lines(prov$series_break_refs))
|
||||||
|
|||||||
+9
-1
@@ -10,7 +10,8 @@
|
|||||||
expenditure_concept_note = NA_character_,
|
expenditure_concept_note = NA_character_,
|
||||||
expenditure_concept_direct_suppressed = FALSE,
|
expenditure_concept_direct_suppressed = FALSE,
|
||||||
harmonization = NULL, recipe = NULL,
|
harmonization = NULL, recipe = NULL,
|
||||||
suggestions = list()) {
|
suggestions = list(),
|
||||||
|
completion = NULL) {
|
||||||
manifest <- .uscogdata_env$manifest
|
manifest <- .uscogdata_env$manifest
|
||||||
|
|
||||||
codes <- result[["codes_included"]]
|
codes <- result[["codes_included"]]
|
||||||
@@ -126,6 +127,13 @@
|
|||||||
),
|
),
|
||||||
series_break_refs = break_refs,
|
series_break_refs = break_refs,
|
||||||
corpus_break_refs = corpus_refs,
|
corpus_break_refs = corpus_refs,
|
||||||
|
# What `complete = TRUE` filled, and the rule it filled by. Always
|
||||||
|
# present so a consumer can read `completion$applied` without testing
|
||||||
|
# for the key -- an absent block and applied = FALSE would otherwise be
|
||||||
|
# indistinguishable from an older reader version.
|
||||||
|
completion = completion %||% list(
|
||||||
|
applied = FALSE, rows_filled = 0L, absence_means = list()
|
||||||
|
),
|
||||||
manifest = list(
|
manifest = list(
|
||||||
schema_version = as.integer(manifest$schema_version),
|
schema_version = as.integer(manifest$schema_version),
|
||||||
pipeline_commit = manifest$pipeline_commit %||% NA_character_,
|
pipeline_commit = manifest$pipeline_commit %||% NA_character_,
|
||||||
|
|||||||
+6
-3
@@ -11,11 +11,13 @@
|
|||||||
#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||||
#' `revenue_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
#' `revenue_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
||||||
#' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
#' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
||||||
#' optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`.
|
#' optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`,
|
||||||
|
#' and `value_source` when `complete = TRUE`.
|
||||||
#' @export
|
#' @export
|
||||||
cog_revenue <- function(govid, years, category = NULL,
|
cog_revenue <- function(govid, years, category = NULL,
|
||||||
per_capita = FALSE, adjust_to_year = NULL,
|
per_capita = FALSE, adjust_to_year = NULL,
|
||||||
basis = c("harmonized", "raw"), recipe = NULL) {
|
basis = c("harmonized", "raw"), recipe = NULL,
|
||||||
|
complete = FALSE) {
|
||||||
.verb_spendrev(
|
.verb_spendrev(
|
||||||
verb = "cog_revenue",
|
verb = "cog_revenue",
|
||||||
view_base = "revenue_annotated",
|
view_base = "revenue_annotated",
|
||||||
@@ -28,6 +30,7 @@ cog_revenue <- function(govid, years, category = NULL,
|
|||||||
per_capita = per_capita,
|
per_capita = per_capita,
|
||||||
adjust_to_year = adjust_to_year,
|
adjust_to_year = adjust_to_year,
|
||||||
basis = basis,
|
basis = basis,
|
||||||
recipe = recipe
|
recipe = recipe,
|
||||||
|
complete = complete
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|||||||
+62
-6
@@ -70,16 +70,42 @@
|
|||||||
#' `provenance$expenditure_concept_direct_suppressed` is `TRUE` -- the
|
#' `provenance$expenditure_concept_direct_suppressed` is `TRUE` -- the
|
||||||
#' figure in those rows is the intergovernmental leg alone, not Direct +
|
#' figure in those rows is the intergovernmental leg alone, not Direct +
|
||||||
#' IG.
|
#' IG.
|
||||||
|
#' @param complete If `TRUE`, fill the requested grid so that a cell the
|
||||||
|
#' corpus does not carry still appears, labelled with **why** it is
|
||||||
|
#' missing, and add a `value_source` column to every row:
|
||||||
|
#'
|
||||||
|
#' * `"reported"` — the corpus carries this cell.
|
||||||
|
#' * `"census_zero"` — dense-source year (`<= FY2011`), cell absent:
|
||||||
|
#' Census published `$0`. `amt_nominal` is `0`.
|
||||||
|
#' * `"not_reported"` — sparse-source year (`>= FY2012`), cell absent: the
|
||||||
|
#' government did not report, and the value is unknown. `amt_nominal` is
|
||||||
|
#' `NA`, **not** `0` — writing a zero there would invent data.
|
||||||
|
#'
|
||||||
|
#' The grid comes from the corpus's `code_set` table, scoped to each
|
||||||
|
#' government's own type, so a county is never filled with cells only a
|
||||||
|
#' state can report. Reported rows are passed through untouched.
|
||||||
|
#'
|
||||||
|
#' Defaults to `FALSE` (the historical behaviour: absent cells simply do
|
||||||
|
#' not appear). Needs a corpus published from 2026-07-29 onward, which is
|
||||||
|
#' when `representation`/`code_set` began shipping; aborts with class
|
||||||
|
#' `uscogdata_representation_unavailable` otherwise. Not available with
|
||||||
|
#' `recipe` or with `expenditure_concept = "total"` (class
|
||||||
|
#' `uscogdata_complete_unsupported`) — neither draws its cells from
|
||||||
|
#' `code_set`.
|
||||||
#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||||
#' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
#' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
||||||
#' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
#' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
||||||
#' optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`.
|
#' optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`,
|
||||||
#' Carries a `provenance` attribute matching `inst/schemas/provenance-v1.json`.
|
#' and `value_source` when `complete = TRUE`.
|
||||||
|
#' Carries a `provenance` attribute matching `inst/schemas/provenance-v1.json`,
|
||||||
|
#' whose `completion` block reports `applied`, `rows_filled`, and the
|
||||||
|
#' per-year `absence_means` rule that was applied.
|
||||||
#' @export
|
#' @export
|
||||||
cog_spending <- function(govid, years, category = NULL,
|
cog_spending <- function(govid, years, category = NULL,
|
||||||
per_capita = FALSE, adjust_to_year = NULL,
|
per_capita = FALSE, adjust_to_year = NULL,
|
||||||
basis = c("harmonized", "raw"), recipe = NULL,
|
basis = c("harmonized", "raw"), recipe = NULL,
|
||||||
expenditure_concept = c("direct", "total")) {
|
expenditure_concept = c("direct", "total"),
|
||||||
|
complete = FALSE) {
|
||||||
.verb_spendrev(
|
.verb_spendrev(
|
||||||
verb = "cog_spending",
|
verb = "cog_spending",
|
||||||
view_base = "spending_annotated",
|
view_base = "spending_annotated",
|
||||||
@@ -93,7 +119,8 @@ cog_spending <- function(govid, years, category = NULL,
|
|||||||
adjust_to_year = adjust_to_year,
|
adjust_to_year = adjust_to_year,
|
||||||
basis = basis,
|
basis = basis,
|
||||||
recipe = recipe,
|
recipe = recipe,
|
||||||
expenditure_concept = expenditure_concept
|
expenditure_concept = expenditure_concept,
|
||||||
|
complete = complete
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -118,7 +145,8 @@ cog_spending <- function(govid, years, category = NULL,
|
|||||||
govid, years, category,
|
govid, years, category,
|
||||||
per_capita, adjust_to_year,
|
per_capita, adjust_to_year,
|
||||||
basis = c("harmonized", "raw"), recipe = NULL,
|
basis = c("harmonized", "raw"), recipe = NULL,
|
||||||
expenditure_concept = c("direct", "total")) {
|
expenditure_concept = c("direct", "total"),
|
||||||
|
complete = FALSE) {
|
||||||
basis_explicit <- length(basis) == 1L
|
basis_explicit <- length(basis) == 1L
|
||||||
basis <- match.arg(basis, c("harmonized", "raw"))
|
basis <- match.arg(basis, c("harmonized", "raw"))
|
||||||
# match.arg() itself throws a base `simpleError`, not an rlang-classed
|
# match.arg() itself throws a base `simpleError`, not an rlang-classed
|
||||||
@@ -165,12 +193,27 @@ cog_spending <- function(govid, years, category = NULL,
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
complete <- isTRUE(complete)
|
||||||
|
if (complete && !is.null(recipe)) {
|
||||||
|
.abort_complete_unsupported(
|
||||||
|
"A recipe defines its own component codes and never goes through `summary_categories`, so there is no grid to fill from.",
|
||||||
|
"Query the recipe without `complete`, or use a category query with `complete = TRUE`."
|
||||||
|
)
|
||||||
|
}
|
||||||
|
if (complete && identical(expenditure_concept, "total")) {
|
||||||
|
.abort_complete_unsupported(
|
||||||
|
"The intergovernmental leg deliberately keeps aggregate-flagged rows (see `inst/sql/24-ig_long.sql`), so its cells are not the ones `code_set` describes.",
|
||||||
|
"Use `expenditure_concept = \"direct\"` with `complete = TRUE`, or drop `complete`."
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
years <- as.integer(years)
|
years <- as.integer(years)
|
||||||
if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year)
|
if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year)
|
||||||
|
|
||||||
con <- .ensure_session()
|
con <- .ensure_session()
|
||||||
manifest <- .uscogdata_env$manifest
|
manifest <- .uscogdata_env$manifest
|
||||||
scope <- .check_govids_in_scope(govid)
|
scope <- .check_govids_in_scope(govid)
|
||||||
|
if (complete) .require_representation(con, manifest)
|
||||||
|
|
||||||
resolved <- .resolve_basis(basis, basis_explicit, manifest)
|
resolved <- .resolve_basis(basis, basis_explicit, manifest)
|
||||||
|
|
||||||
@@ -201,6 +244,18 @@ cog_spending <- function(govid, years, category = NULL,
|
|||||||
result <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
result <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# Fill BEFORE per_capita / inflation so the added cells get the same
|
||||||
|
# treatment as reported ones: a census_zero stays $0 per capita and in real
|
||||||
|
# dollars, and a not_reported stays NA through both rather than becoming a
|
||||||
|
# spurious 0.
|
||||||
|
completion <- list(applied = FALSE, rows_filled = 0L, absence_means = list())
|
||||||
|
if (complete) {
|
||||||
|
result <- .complete_result(result, con, subtype_col, govid, years,
|
||||||
|
category, flow_prefixes)
|
||||||
|
completion <- attr(result, ".completion")
|
||||||
|
attr(result, ".completion") <- NULL
|
||||||
|
}
|
||||||
|
|
||||||
if (per_capita) result <- .attach_per_capita(result, con, govid)
|
if (per_capita) result <- .attach_per_capita(result, con, govid)
|
||||||
if (!is.null(adjust_to_year)) {
|
if (!is.null(adjust_to_year)) {
|
||||||
result <- .attach_real_dollars(result, adjust_to_year, per_capita)
|
result <- .attach_real_dollars(result, adjust_to_year, per_capita)
|
||||||
@@ -309,7 +364,8 @@ cog_spending <- function(govid, years, category = NULL,
|
|||||||
expenditure_concept_direct_suppressed = direct_suppressed_flag,
|
expenditure_concept_direct_suppressed = direct_suppressed_flag,
|
||||||
harmonization = harmonization,
|
harmonization = harmonization,
|
||||||
recipe = recipe_block,
|
recipe = recipe_block,
|
||||||
suggestions = suggestions
|
suggestions = suggestions,
|
||||||
|
completion = completion
|
||||||
)
|
)
|
||||||
prov$scope$govids_found <- scope$found
|
prov$scope$govids_found <- scope$found
|
||||||
prov$scope$govids_missing <- scope$missing
|
prov$scope$govids_missing <- scope$missing
|
||||||
|
|||||||
@@ -32,6 +32,29 @@
|
|||||||
"45-ig_annotated_harmonized.sql"
|
"45-ig_annotated_harmonized.sql"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# The representation contract (cog_pipeline#64): two parquet tables that say
|
||||||
|
# what an ABSENT cell means in a given year. Gated on manifest PRESENCE, not
|
||||||
|
# on schema_version, because the sparsification that introduced them did not
|
||||||
|
# bump the version -- the pre-sparsification corpus this package shipped
|
||||||
|
# against until 2026-07-30 was already schema v6 and carried neither table.
|
||||||
|
# Keying off the version number would therefore register a view over a file
|
||||||
|
# that does not exist and fail at CREATE VIEW time on exactly the corpora this
|
||||||
|
# check exists to tolerate.
|
||||||
|
.representation_view_files <- c(
|
||||||
|
"36-representation.sql" = "representation.parquet",
|
||||||
|
"37-code_set.sql" = "code_set.parquet"
|
||||||
|
)
|
||||||
|
|
||||||
|
#' Does the mounted corpus publish `file` (e.g. "code_set.parquet")?
|
||||||
|
#' Reads the manifest's metadata list rather than stat-ing the URL, so it
|
||||||
|
#' works identically for a local fixture and a remote share.
|
||||||
|
#' @noRd
|
||||||
|
.corpus_has_table <- function(manifest, file) {
|
||||||
|
paths <- vapply(manifest$files$metadata %||% list(),
|
||||||
|
function(f) as.character(f$path %||% ""), character(1))
|
||||||
|
file %in% basename(paths)
|
||||||
|
}
|
||||||
|
|
||||||
#' Register DuckDB views from inst/sql/ SQL files
|
#' Register DuckDB views from inst/sql/ SQL files
|
||||||
#' @noRd
|
#' @noRd
|
||||||
.register_views <- function(con, url, manifest) {
|
.register_views <- function(con, url, manifest) {
|
||||||
@@ -39,7 +62,10 @@
|
|||||||
files <- sort(list.files(sql_dir, pattern = "\\.sql$", full.names = TRUE))
|
files <- sort(list.files(sql_dir, pattern = "\\.sql$", full.names = TRUE))
|
||||||
schema_version <- suppressWarnings(as.integer(manifest$schema_version %||% 0L))
|
schema_version <- suppressWarnings(as.integer(manifest$schema_version %||% 0L))
|
||||||
for (f in files) {
|
for (f in files) {
|
||||||
if (basename(f) %in% .harmonization_view_files && schema_version < 5L) next
|
base <- basename(f)
|
||||||
|
if (base %in% .harmonization_view_files && schema_version < 5L) next
|
||||||
|
if (base %in% names(.representation_view_files) &&
|
||||||
|
!.corpus_has_table(manifest, .representation_view_files[[base]])) next
|
||||||
sql <- paste(readLines(f, warn = FALSE), collapse = "\n")
|
sql <- paste(readLines(f, warn = FALSE), collapse = "\n")
|
||||||
sql <- gsub("\\{url\\}", url, sql, fixed = FALSE)
|
sql <- gsub("\\{url\\}", url, sql, fixed = FALSE)
|
||||||
DBI::dbExecute(con, sql)
|
DBI::dbExecute(con, sql)
|
||||||
|
|||||||
@@ -33,6 +33,15 @@
|
|||||||
"aggregate_fallback": { "type": ["object", "null"] },
|
"aggregate_fallback": { "type": ["object", "null"] },
|
||||||
"transformations":{ "type": "object" },
|
"transformations":{ "type": "object" },
|
||||||
"series_break_refs": { "type": "array", "items": { "type": "string" } },
|
"series_break_refs": { "type": "array", "items": { "type": "string" } },
|
||||||
|
"completion": {
|
||||||
|
"type": "object",
|
||||||
|
"description": "What `complete = TRUE` filled. `applied` is FALSE on an ordinary query. `rows_filled` counts cells added to the requested grid, and `absence_means` maps each requested year to the meaning of an absent cell there ('census_zero' in a dense_source year, 'not_reported' in a sparse_source one). Filled rows carry `value_source` in the result: 'reported', 'census_zero' (amount 0 -- Census published $0), or 'not_reported' (amount NA -- unknown).",
|
||||||
|
"properties": {
|
||||||
|
"applied": { "type": "boolean" },
|
||||||
|
"rows_filled": { "type": "integer" },
|
||||||
|
"absence_means": { "type": "object" }
|
||||||
|
}
|
||||||
|
},
|
||||||
"corpus_break_refs": {
|
"corpus_break_refs": {
|
||||||
"type": "array",
|
"type": "array",
|
||||||
"items": { "type": "string" },
|
"items": { "type": "string" },
|
||||||
|
|||||||
@@ -0,0 +1,3 @@
|
|||||||
|
CREATE OR REPLACE VIEW representation AS
|
||||||
|
SELECT *
|
||||||
|
FROM read_parquet('{url}data/representation.parquet');
|
||||||
@@ -0,0 +1,3 @@
|
|||||||
|
CREATE OR REPLACE VIEW code_set AS
|
||||||
|
SELECT *
|
||||||
|
FROM read_parquet('{url}data/code_set.parquet');
|
||||||
+27
-2
@@ -11,7 +11,8 @@ cog_revenue(
|
|||||||
per_capita = FALSE,
|
per_capita = FALSE,
|
||||||
adjust_to_year = NULL,
|
adjust_to_year = NULL,
|
||||||
basis = c("harmonized", "raw"),
|
basis = c("harmonized", "raw"),
|
||||||
recipe = NULL
|
recipe = NULL,
|
||||||
|
complete = FALSE
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
\arguments{
|
\arguments{
|
||||||
@@ -55,12 +56,36 @@ argument is ignored and the result's provenance reports
|
|||||||
`basis = "recipe"` with an inert `harmonization` block (`applied =
|
`basis = "recipe"` with an inert `harmonization` block (`applied =
|
||||||
FALSE`, pointing at the `recipe` block instead) rather than a
|
FALSE`, pointing at the `recipe` block instead) rather than a
|
||||||
possibly-misleading `"harmonized"`/`"raw"` value.}
|
possibly-misleading `"harmonized"`/`"raw"` value.}
|
||||||
|
|
||||||
|
\item{complete}{If `TRUE`, fill the requested grid so that a cell the
|
||||||
|
corpus does not carry still appears, labelled with **why** it is
|
||||||
|
missing, and add a `value_source` column to every row:
|
||||||
|
|
||||||
|
* `"reported"` — the corpus carries this cell.
|
||||||
|
* `"census_zero"` — dense-source year (`<= FY2011`), cell absent:
|
||||||
|
Census published `$0`. `amt_nominal` is `0`.
|
||||||
|
* `"not_reported"` — sparse-source year (`>= FY2012`), cell absent: the
|
||||||
|
government did not report, and the value is unknown. `amt_nominal` is
|
||||||
|
`NA`, **not** `0` — writing a zero there would invent data.
|
||||||
|
|
||||||
|
The grid comes from the corpus's `code_set` table, scoped to each
|
||||||
|
government's own type, so a county is never filled with cells only a
|
||||||
|
state can report. Reported rows are passed through untouched.
|
||||||
|
|
||||||
|
Defaults to `FALSE` (the historical behaviour: absent cells simply do
|
||||||
|
not appear). Needs a corpus published from 2026-07-29 onward, which is
|
||||||
|
when `representation`/`code_set` began shipping; aborts with class
|
||||||
|
`uscogdata_representation_unavailable` otherwise. Not available with
|
||||||
|
`recipe` or with `expenditure_concept = "total"` (class
|
||||||
|
`uscogdata_complete_unsupported`) — neither draws its cells from
|
||||||
|
`code_set`.}
|
||||||
}
|
}
|
||||||
\value{
|
\value{
|
||||||
Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||||
`revenue_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
`revenue_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
||||||
optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
||||||
optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`.
|
optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`,
|
||||||
|
and `value_source` when `complete = TRUE`.
|
||||||
}
|
}
|
||||||
\description{
|
\description{
|
||||||
Mirror of [cog_spending()] for revenue categories. One row per
|
Mirror of [cog_spending()] for revenue categories. One row per
|
||||||
|
|||||||
+30
-3
@@ -12,7 +12,8 @@ cog_spending(
|
|||||||
adjust_to_year = NULL,
|
adjust_to_year = NULL,
|
||||||
basis = c("harmonized", "raw"),
|
basis = c("harmonized", "raw"),
|
||||||
recipe = NULL,
|
recipe = NULL,
|
||||||
expenditure_concept = c("direct", "total")
|
expenditure_concept = c("direct", "total"),
|
||||||
|
complete = FALSE
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
\arguments{
|
\arguments{
|
||||||
@@ -85,13 +86,39 @@ component (when one exists), and
|
|||||||
`provenance$expenditure_concept_direct_suppressed` is `TRUE` -- the
|
`provenance$expenditure_concept_direct_suppressed` is `TRUE` -- the
|
||||||
figure in those rows is the intergovernmental leg alone, not Direct +
|
figure in those rows is the intergovernmental leg alone, not Direct +
|
||||||
IG.}
|
IG.}
|
||||||
|
|
||||||
|
\item{complete}{If `TRUE`, fill the requested grid so that a cell the
|
||||||
|
corpus does not carry still appears, labelled with **why** it is
|
||||||
|
missing, and add a `value_source` column to every row:
|
||||||
|
|
||||||
|
* `"reported"` — the corpus carries this cell.
|
||||||
|
* `"census_zero"` — dense-source year (`<= FY2011`), cell absent:
|
||||||
|
Census published `$0`. `amt_nominal` is `0`.
|
||||||
|
* `"not_reported"` — sparse-source year (`>= FY2012`), cell absent: the
|
||||||
|
government did not report, and the value is unknown. `amt_nominal` is
|
||||||
|
`NA`, **not** `0` — writing a zero there would invent data.
|
||||||
|
|
||||||
|
The grid comes from the corpus's `code_set` table, scoped to each
|
||||||
|
government's own type, so a county is never filled with cells only a
|
||||||
|
state can report. Reported rows are passed through untouched.
|
||||||
|
|
||||||
|
Defaults to `FALSE` (the historical behaviour: absent cells simply do
|
||||||
|
not appear). Needs a corpus published from 2026-07-29 onward, which is
|
||||||
|
when `representation`/`code_set` began shipping; aborts with class
|
||||||
|
`uscogdata_representation_unavailable` otherwise. Not available with
|
||||||
|
`recipe` or with `expenditure_concept = "total"` (class
|
||||||
|
`uscogdata_complete_unsupported`) — neither draws its cells from
|
||||||
|
`code_set`.}
|
||||||
}
|
}
|
||||||
\value{
|
\value{
|
||||||
Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||||
`spend_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
`spend_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
||||||
optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
||||||
optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`.
|
optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`,
|
||||||
Carries a `provenance` attribute matching `inst/schemas/provenance-v1.json`.
|
and `value_source` when `complete = TRUE`.
|
||||||
|
Carries a `provenance` attribute matching `inst/schemas/provenance-v1.json`,
|
||||||
|
whose `completion` block reports `applied`, `rows_filled`, and the
|
||||||
|
per-year `absence_means` rule that was applied.
|
||||||
}
|
}
|
||||||
\description{
|
\description{
|
||||||
One row per `(year, canonical_govid, spend_subtype, category)`. Amounts are
|
One row per `(year, canonical_govid, spend_subtype, category)`. Amounts are
|
||||||
|
|||||||
@@ -85,6 +85,42 @@ with_doctored_schema_version <- function(version, code) {
|
|||||||
force(code)
|
force(code)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# Copy the bundled fixture to a temp dir with representation.parquet and
|
||||||
|
# code_set.parquet removed (and dropped from the manifest's metadata list),
|
||||||
|
# then run `code` against it. Models a corpus published BEFORE sparsification:
|
||||||
|
# schema_version is left alone deliberately, because it was never bumped for
|
||||||
|
# that change -- the pre-sparsification fixture this package shipped until
|
||||||
|
# 2026-07-30 was schema v6 and carried neither table. Presence in the manifest
|
||||||
|
# is therefore the only honest signal, and this helper is what proves the
|
||||||
|
# package keys off it rather than off the version number.
|
||||||
|
with_corpus_missing_representation <- function(code) {
|
||||||
|
src <- fixture_corpus_path()
|
||||||
|
tmp <- withr::local_tempdir(.local_envir = parent.frame())
|
||||||
|
file.copy(list.files(src, full.names = TRUE), tmp, recursive = TRUE)
|
||||||
|
|
||||||
|
dropped <- c("representation.parquet", "code_set.parquet")
|
||||||
|
file.remove(file.path(tmp, "data", dropped))
|
||||||
|
|
||||||
|
manifest_path <- file.path(tmp, "manifest.json")
|
||||||
|
m <- jsonlite::fromJSON(manifest_path, simplifyVector = FALSE)
|
||||||
|
m$files$metadata <- Filter(
|
||||||
|
function(f) !basename(f$path) %in% dropped, m$files$metadata
|
||||||
|
)
|
||||||
|
writeLines(
|
||||||
|
jsonlite::toJSON(m, auto_unbox = TRUE, pretty = TRUE, null = "null"),
|
||||||
|
manifest_path
|
||||||
|
)
|
||||||
|
|
||||||
|
old_url <- Sys.getenv("USCOGDATA_URL", unset = NA)
|
||||||
|
uscogdata:::cog_close()
|
||||||
|
Sys.setenv(USCOGDATA_URL = paste0(tmp, "/"))
|
||||||
|
on.exit({
|
||||||
|
uscogdata:::cog_close()
|
||||||
|
if (is.na(old_url)) Sys.unsetenv("USCOGDATA_URL") else Sys.setenv(USCOGDATA_URL = old_url)
|
||||||
|
}, add = TRUE)
|
||||||
|
force(code)
|
||||||
|
}
|
||||||
|
|
||||||
# Copy the bundled fixture to a temp dir with summary_categories.parquet
|
# Copy the bundled fixture to a temp dir with summary_categories.parquet
|
||||||
# rewritten to drop every M/L (intergovernmental) row, then run `code`
|
# rewritten to drop every M/L (intergovernmental) row, then run `code`
|
||||||
# against it with a clean session (mirrors with_fixture_corpus()/
|
# against it with a clean session (mirrors with_fixture_corpus()/
|
||||||
|
|||||||
@@ -0,0 +1,188 @@
|
|||||||
|
# tests/testthat/test-complete.R
|
||||||
|
#
|
||||||
|
# uscogdata#18. The published corpus no longer stores the wide era's explicit
|
||||||
|
# zeros (cog_pipeline#64, series break SB194), so absence means two different
|
||||||
|
# things:
|
||||||
|
#
|
||||||
|
# <= FY2011 (dense_source) : cell absent => Census published $0
|
||||||
|
# >= FY2012 (sparse_source): cell absent => not reported, unknown
|
||||||
|
#
|
||||||
|
# `complete = TRUE` fills the requested grid from `code_set` and stamps every
|
||||||
|
# row's `value_source` so the two are distinguishable. Expected row sets here
|
||||||
|
# are built from the corpus parquet directly, never from the verb under test --
|
||||||
|
# verifying what a filter does through that same filter proves nothing.
|
||||||
|
|
||||||
|
# The (subtype, category) cells that SHOULD exist for one government-year:
|
||||||
|
# every code in force for that government's type, mapped through
|
||||||
|
# summary_categories, matching the verb's flow prefixes and excluding
|
||||||
|
# aggregate-flagged codes (which spending_long/revenue_long drop).
|
||||||
|
raw_expected_cells <- function(govid, year, prefixes, subtype_col) {
|
||||||
|
fx <- sub("/$", "", Sys.getenv("USCOGDATA_URL"))
|
||||||
|
q <- function(f) sprintf("read_parquet('%s/data/%s')", fx, f)
|
||||||
|
wt_raw_query(sprintf(
|
||||||
|
"SELECT DISTINCT c.%s AS subtype, c.category
|
||||||
|
FROM %s cs
|
||||||
|
JOIN %s x ON x.govs_type = cs.type
|
||||||
|
JOIN %s c ON c.item_code = cs.item_code
|
||||||
|
WHERE x.canonical_govid = '%s'
|
||||||
|
AND cs.year = %d
|
||||||
|
AND NOT cs.is_aggregate
|
||||||
|
AND LEFT(cs.item_code, 1) IN (%s)
|
||||||
|
AND c.category IS NOT NULL
|
||||||
|
AND c.%s IS NOT NULL",
|
||||||
|
subtype_col, q("code_set.parquet"), q("canonical_fips_xwalk.parquet"),
|
||||||
|
q("summary_categories.parquet"), govid, year,
|
||||||
|
paste0("'", prefixes, "'", collapse = ","), subtype_col
|
||||||
|
))
|
||||||
|
}
|
||||||
|
|
||||||
|
test_that("complete = FALSE is the default and changes nothing", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
plain <- cog_spending("121011212191", 2011L)
|
||||||
|
explicit <- cog_spending("121011212191", 2011L, complete = FALSE)
|
||||||
|
expect_equal(nrow(plain), nrow(explicit))
|
||||||
|
expect_false("value_source" %in% names(plain))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("complete = TRUE round-trips a dense-source year to the pre-sparsification cells", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
# FY2011 is dense_source: before sparsification this government carried a
|
||||||
|
# row for every code in force, most of them $0. complete = TRUE must
|
||||||
|
# reproduce that cell set exactly.
|
||||||
|
r <- cog_spending("121011212191", 2011L, complete = TRUE)
|
||||||
|
expected <- raw_expected_cells("121011212191", 2011L,
|
||||||
|
c("E", "F", "G"), "spend_subtype")
|
||||||
|
|
||||||
|
key <- function(sub, cat) paste(sub, cat, sep = "|")
|
||||||
|
expect_setequal(key(r$spend_subtype, r$category),
|
||||||
|
key(expected$subtype, expected$category))
|
||||||
|
expect_gt(nrow(expected), 0L)
|
||||||
|
|
||||||
|
# Every filled cell in a dense-source year is a Census-published $0 --
|
||||||
|
# never "unknown", which is what the modern era's absences mean.
|
||||||
|
expect_setequal(unique(r$value_source), c("reported", "census_zero"))
|
||||||
|
expect_true(all(r$amt_nominal[r$value_source == "census_zero"] == 0))
|
||||||
|
expect_true(all(r$amt_nominal[r$value_source == "reported"] != 0))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("complete = TRUE preserves the reported rows and their amounts exactly", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
plain <- cog_spending("121011212191", 2011L)
|
||||||
|
full <- cog_spending("121011212191", 2011L, complete = TRUE)
|
||||||
|
|
||||||
|
# Filling adds rows; it must never alter or drop one.
|
||||||
|
expect_gt(nrow(full), nrow(plain))
|
||||||
|
reported <- full[full$value_source == "reported", ]
|
||||||
|
expect_equal(nrow(reported), nrow(plain))
|
||||||
|
expect_equal(sum(reported$amt_nominal), sum(plain$amt_nominal))
|
||||||
|
# ... and the total is unchanged, because every added cell is $0.
|
||||||
|
expect_equal(sum(full$amt_nominal, na.rm = TRUE), sum(plain$amt_nominal))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("a sparse-source year's absences are unknown, not zero", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
# FY2019 is sparse_source: an absent cell means the government did not
|
||||||
|
# report, which is NOT a zero. Filling those with 0 would invent data --
|
||||||
|
# the exact error the representation contract exists to prevent.
|
||||||
|
r <- cog_spending("121011212191", 2019L, complete = TRUE)
|
||||||
|
filled <- r[r$value_source != "reported", ]
|
||||||
|
expect_gt(nrow(filled), 0L)
|
||||||
|
expect_true(all(filled$value_source == "not_reported"))
|
||||||
|
expect_true(all(is.na(filled$amt_nominal)))
|
||||||
|
expect_false(any(r$value_source == "census_zero"))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("the fill is scoped to each government's own type", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
# Filling against the union of all types would invent cells for codes a
|
||||||
|
# county can never report. Every filled category must be one that
|
||||||
|
# code_set puts in force for type 1 (county) specifically.
|
||||||
|
r <- cog_spending("121011212191", 2011L, complete = TRUE)
|
||||||
|
county_cells <- raw_expected_cells("121011212191", 2011L,
|
||||||
|
c("E", "F", "G"), "spend_subtype")
|
||||||
|
expect_true(all(r$category %in% county_cells$category))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("complete = TRUE respects the category filter", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
r <- cog_spending("121011212191", 2011L, category = "Police",
|
||||||
|
complete = TRUE)
|
||||||
|
expect_true(all(r$category == "Police"))
|
||||||
|
expect_true("value_source" %in% names(r))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("cog_revenue() completes on its own flow", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
r <- cog_revenue("121011212191", 2011L, complete = TRUE)
|
||||||
|
expected <- raw_expected_cells("121011212191", 2011L,
|
||||||
|
c("T", "A", "U", "B", "C", "D"),
|
||||||
|
"revenue_subtype")
|
||||||
|
key <- function(sub, cat) paste(sub, cat, sep = "|")
|
||||||
|
expect_setequal(key(r$revenue_subtype, r$category),
|
||||||
|
key(expected$subtype, expected$category))
|
||||||
|
expect_setequal(unique(r$value_source), c("reported", "census_zero"))
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("provenance records the completion and its absence rule", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
prov <- attr(cog_spending("121011212191", 2011L, complete = TRUE),
|
||||||
|
"provenance")
|
||||||
|
expect_true(prov$completion$applied)
|
||||||
|
expect_equal(prov$completion$absence_means$`2011`, "census_zero")
|
||||||
|
expect_gt(prov$completion$rows_filled, 0L)
|
||||||
|
|
||||||
|
off <- attr(cog_spending("121011212191", 2011L), "provenance")
|
||||||
|
expect_false(off$completion$applied)
|
||||||
|
expect_equal(off$completion$rows_filled, 0L)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("complete = TRUE is refused where the fill would be guesswork", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
with_fixture_corpus({
|
||||||
|
# A recipe defines its own component codes and does not go through
|
||||||
|
# summary_categories at all, so there is no grid to fill from.
|
||||||
|
expect_error(
|
||||||
|
cog_spending("121011212191", 2011L, recipe = "corrections_combined",
|
||||||
|
complete = TRUE),
|
||||||
|
class = "uscogdata_complete_unsupported"
|
||||||
|
)
|
||||||
|
# The intergovernmental leg keeps aggregate rows by design
|
||||||
|
# (inst/sql/24-ig_long.sql), so its grid is not code_set's grid.
|
||||||
|
expect_error(
|
||||||
|
cog_spending("121011212191", 2011L, expenditure_concept = "total",
|
||||||
|
complete = TRUE),
|
||||||
|
class = "uscogdata_complete_unsupported"
|
||||||
|
)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("complete = TRUE aborts on a corpus with no representation contract", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
# A corpus published before sparsification carries neither table, so there
|
||||||
|
# is nothing to fill from and no rule saying what an absence means. That
|
||||||
|
# must abort rather than guess.
|
||||||
|
with_corpus_missing_representation({
|
||||||
|
expect_error(
|
||||||
|
cog_spending("121011212191", 2011L, complete = TRUE),
|
||||||
|
class = "uscogdata_representation_unavailable"
|
||||||
|
)
|
||||||
|
# ... while an ordinary query on the same corpus still works.
|
||||||
|
expect_gt(nrow(cog_spending("121011212191", 2011L)), 0L)
|
||||||
|
})
|
||||||
|
})
|
||||||
Reference in New Issue
Block a user