Compare commits
31
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d258cef8c5 | ||
|
|
e7d3a7a310 | ||
|
|
a4eb80d823 | ||
|
|
aba7ffbac2 | ||
|
|
c1c6b5a6ba | ||
|
|
c7260cb20c | ||
|
|
54ece11867 | ||
|
|
3bd9b1f011 | ||
|
|
24b2ff7d8c | ||
|
|
c28712f62f | ||
|
|
7913b0f664 | ||
|
|
887acf7e81 | ||
|
|
81fd1a5279 | ||
|
|
e2088458e1 | ||
|
|
fefd4fe969 | ||
|
|
7ed1da9b79 | ||
|
|
9240a18ea3 | ||
|
|
46fed3a241 | ||
|
|
c9d1a05d4f | ||
|
|
fcecd62a03 | ||
|
|
e581e7360c | ||
|
|
748ca4a56e | ||
|
|
fa40266d07 | ||
|
|
3583c05852
|
||
|
|
bd53230ae7 | ||
|
|
e813ffd3aa | ||
|
|
b0df1ec668 | ||
|
|
77f48047b1
|
||
|
|
4de915b557
|
||
|
|
7818cd2b1a
|
||
|
|
3b725770d2 |
@@ -16,3 +16,4 @@
|
||||
^Meta$
|
||||
^\.gitea$
|
||||
^CLAUDE\.md$
|
||||
^\.superpowers$
|
||||
|
||||
@@ -9,3 +9,6 @@ docs/
|
||||
/Meta/
|
||||
.DS_Store
|
||||
/.quarto/
|
||||
|
||||
# SDD working artifacts (ledger, briefs, review packages) — plans/ stays tracked
|
||||
.superpowers/sdd/
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
+1
-1
@@ -32,4 +32,4 @@ Config/testthat/edition: 3
|
||||
VignetteBuilder: knitr
|
||||
RoxygenNote: 7.3.3
|
||||
MinCorpusSchema: 4
|
||||
MaxCorpusSchema: 4
|
||||
MaxCorpusSchema: 5
|
||||
|
||||
@@ -10,5 +10,6 @@ export(cog_gov_search)
|
||||
export(cog_manifest)
|
||||
export(cog_mirror)
|
||||
export(cog_peer_compare)
|
||||
export(cog_recipes)
|
||||
export(cog_revenue)
|
||||
export(cog_spending)
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
# R/basis.R
|
||||
# basis= resolution (harmonized/raw, with v4/v5 dual-accept) and the
|
||||
# harmonization exclusion-count block attached to provenance.
|
||||
|
||||
#' Resolve the requested `basis` against the active corpus's schema_version.
|
||||
#'
|
||||
#' On a `schema_version >= 5` corpus, the requested basis is used as-is. On
|
||||
#' an older (`schema_version == 4`) corpus, which has no harmonization
|
||||
#' tables: a caller who left `basis` at its default (`"harmonized"`, so
|
||||
#' `explicit` is `FALSE`) silently gets `"raw"` back, with a note recorded
|
||||
#' for provenance; a caller who explicitly asked for
|
||||
#' `basis = "harmonized"` gets a hard abort instead of a silent downgrade.
|
||||
#'
|
||||
#' @param basis `"harmonized"` or `"raw"` (already resolved via `match.arg`).
|
||||
#' @param explicit `TRUE` if the caller passed `basis` explicitly (as
|
||||
#' opposed to relying on the default `c("harmonized", "raw")`).
|
||||
#' @param manifest The active session's parsed manifest list.
|
||||
#' @return List with `basis` (the resolved value) and `note` (character or
|
||||
#' `NA_character_`).
|
||||
#' @noRd
|
||||
.resolve_basis <- function(basis, explicit, manifest) {
|
||||
schema_version <- suppressWarnings(as.integer(manifest$schema_version %||% 0L))
|
||||
|
||||
if (schema_version >= 5L) {
|
||||
return(list(basis = basis, note = NA_character_))
|
||||
}
|
||||
|
||||
if (identical(basis, "harmonized") && explicit) {
|
||||
cli::cli_abort(c(
|
||||
"basis = \"harmonized\" requires corpus schema_version >= 5.",
|
||||
x = "Active corpus has schema_version {schema_version}.",
|
||||
i = "Use basis = \"raw\" (the default on this corpus), or point USCOGDATA_URL at a schema_version >= 5 corpus."
|
||||
), class = "uscogdata_basis_unsupported")
|
||||
}
|
||||
|
||||
list(
|
||||
basis = "raw",
|
||||
note = sprintf(
|
||||
"basis resolved to \"raw\": corpus schema_version %d < 5 (harmonization tables unavailable)",
|
||||
schema_version
|
||||
)
|
||||
)
|
||||
}
|
||||
|
||||
#' Count + sum item-level rows that basis="harmonized" excludes because they
|
||||
#' carry no harmonized_code (discontinued / not-yet-ruled codes) within the
|
||||
#' requested flow type (spending or revenue), govids, and years. Only
|
||||
#' meaningful when the resolved basis is "harmonized"; returns an
|
||||
#' applied = FALSE stub otherwise (raw basis never excludes rows this way).
|
||||
#' @noRd
|
||||
.build_harmonization_block <- function(con, govid, years, resolved, flow_prefixes) {
|
||||
if (!identical(resolved$basis, "harmonized")) {
|
||||
return(list(
|
||||
applied = FALSE,
|
||||
na_rows_excluded = 0L,
|
||||
na_amount_excluded = 0,
|
||||
note = resolved$note
|
||||
))
|
||||
}
|
||||
|
||||
sql <- sprintf(
|
||||
"SELECT COUNT(*) AS n, COALESCE(SUM(amt), 0) * 1000.0 AS amt
|
||||
FROM long
|
||||
WHERE canonical_govid IN (%s) AND year IN (%s)
|
||||
AND NOT is_aggregate AND harmonized_code IS NULL
|
||||
AND LEFT(item_code, 1) IN (%s)",
|
||||
.sql_lit_chr(govid), paste(as.integer(years), collapse = ","),
|
||||
.sql_lit_chr(flow_prefixes)
|
||||
)
|
||||
na <- DBI::dbGetQuery(con, sql)
|
||||
|
||||
list(
|
||||
applied = TRUE,
|
||||
na_rows_excluded = as.integer(na$n),
|
||||
na_amount_excluded = as.numeric(na$amt),
|
||||
note = resolved$note
|
||||
)
|
||||
}
|
||||
+24
-1
@@ -21,7 +21,30 @@
|
||||
.uscogdata_defaults[[key]]
|
||||
}
|
||||
|
||||
.resolve_url <- function() .cfg("url")
|
||||
#' Resolve the corpus URL, guaranteeing the trailing slash the package assumes.
|
||||
#'
|
||||
#' Every consumer builds locations by CONCATENATION -- `paste0(url,
|
||||
#' "manifest.json")` in manifest.R, `paste0(url, e$path)` in mirror.R, and the
|
||||
#' parquet glob in views.R -- and mirror.R:104 documents the invariant outright
|
||||
#' ('url ends in "/"'). Nothing enforced it, so a URL entered without the slash
|
||||
#' failed silently and misleadingly:
|
||||
#'
|
||||
#' HTTPS -> ".../downloadmanifest.json"; the host answers with an HTML 404
|
||||
#' page, which lands in the JSON parser as the lexical error
|
||||
#' reported in issue #3 -- pointing the user at "login page / wrong
|
||||
#' share" when the real cause was one missing character.
|
||||
#' local -> ".../corpusdata/long/**/*.parquet" and a DuckDB "No files found".
|
||||
#'
|
||||
#' Normalizing here fixes every consumer at once, rather than each call site
|
||||
#' re-deriving the same invariant. An empty setting is passed through
|
||||
#' untouched so manifest.R's "not configured" guard still fires instead of the
|
||||
#' value degrading into a bare "/" filesystem root.
|
||||
#' @noRd
|
||||
.resolve_url <- function() {
|
||||
url <- .cfg("url")
|
||||
if (is.null(url) || !nzchar(url) || grepl("/$", url)) return(url)
|
||||
paste0(url, "/")
|
||||
}
|
||||
|
||||
.resolve_cache_dir <- function() {
|
||||
v <- .cfg("cache_dir")
|
||||
|
||||
+79
@@ -51,6 +51,30 @@ cog_explain <- function(result, format = c("print", "list")) {
|
||||
cli::cli_text("Category: (all)")
|
||||
}
|
||||
|
||||
if (!is.null(prov$basis)) {
|
||||
note <- if (!is.null(prov$basis_note) && !is.na(prov$basis_note)) {
|
||||
sprintf(" (%s)", prov$basis_note)
|
||||
} else {
|
||||
""
|
||||
}
|
||||
cli::cli_text("Basis: {prov$basis}{note}")
|
||||
}
|
||||
|
||||
if (!is.null(prov$expenditure_concept)) {
|
||||
concept_note <- if (!is.null(prov$expenditure_concept_note) &&
|
||||
!is.na(prov$expenditure_concept_note)) {
|
||||
sprintf(" (%s)", prov$expenditure_concept_note)
|
||||
} else {
|
||||
""
|
||||
}
|
||||
cli::cli_text("Concept: {prov$expenditure_concept}{concept_note}")
|
||||
if (isTRUE(prov$expenditure_concept_direct_suppressed)) {
|
||||
cli::cli_alert_warning(
|
||||
"Direct leg unavailable for at least one requested (year, category) -- affected rows report intergovernmental dollars alone, not Direct + IG. See each row's notes."
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
cli::cli_h2("Codes observed")
|
||||
codes <- prov$codes_summed$observed
|
||||
if (length(codes) == 0L) {
|
||||
@@ -66,6 +90,39 @@ cog_explain <- function(result, format = c("print", "list")) {
|
||||
)
|
||||
}
|
||||
|
||||
h <- prov$harmonization
|
||||
if (!is.null(h) && isTRUE(h$applied)) {
|
||||
cli::cli_h2("Harmonization")
|
||||
cli::cli_text(
|
||||
"Excluded {h$na_rows_excluded} row(s) with no harmonized_code (${format(h$na_amount_excluded, big.mark = ',')})"
|
||||
)
|
||||
}
|
||||
|
||||
rc <- prov$recipe
|
||||
if (!is.null(rc)) {
|
||||
cli::cli_h2("Recipe")
|
||||
cli::cli_text("{rc$recipe_id}: {rc$label}")
|
||||
comp_lines <- vapply(rc$components, function(x) {
|
||||
sprintf("%s (%s, %s-%s, weight=%s)", x$component_code, x$gov_type_scope,
|
||||
x$year_min, x$year_max, x$weight)
|
||||
}, character(1))
|
||||
cli::cli_ul(comp_lines)
|
||||
}
|
||||
|
||||
if (length(prov$suggestions) > 0L) {
|
||||
cli::cli_h2("Suggestions")
|
||||
sugg_lines <- vapply(prov$suggestions, function(s) {
|
||||
sprintf("%s -- %s (years %s-%s): %s", s$recipe_id, s$label,
|
||||
s$available_years[1], s$available_years[2], s$hint)
|
||||
}, character(1))
|
||||
cli::cli_ul(sugg_lines)
|
||||
}
|
||||
|
||||
if (length(prov$series_break_refs) > 0L) {
|
||||
cli::cli_h2("Series breaks")
|
||||
cli::cli_ul(.series_break_story_lines(prov$series_break_refs))
|
||||
}
|
||||
|
||||
cli::cli_h2("Transformations")
|
||||
uc <- prov$transformations$units_conversion
|
||||
if (isTRUE(uc$applied)) {
|
||||
@@ -109,6 +166,28 @@ cog_explain <- function(result, format = c("print", "list")) {
|
||||
invisible(NULL)
|
||||
}
|
||||
|
||||
# One "break-story" line per referenced break_id: "SB109 (2005): <join_advice>".
|
||||
# Re-queries series_breaks_pq for the detail (break_year, join_advice) that
|
||||
# provenance$series_break_refs deliberately doesn't carry (the schema keeps
|
||||
# that field to a plain id array). Falls back to bare ids if no session is
|
||||
# available (e.g. explaining a result after cog_close()) rather than
|
||||
# erroring cog_explain() over a cosmetic detail.
|
||||
#' @noRd
|
||||
.series_break_story_lines <- function(break_ids) {
|
||||
con <- tryCatch(.ensure_session(), error = function(e) NULL)
|
||||
if (is.null(con) || !DBI::dbIsValid(con)) return(break_ids)
|
||||
detail <- tryCatch(
|
||||
DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT break_id, break_year, join_advice FROM series_breaks_pq
|
||||
WHERE break_id IN (%s) ORDER BY break_id",
|
||||
.sql_lit_chr(break_ids)
|
||||
)),
|
||||
error = function(e) NULL
|
||||
)
|
||||
if (is.null(detail) || nrow(detail) == 0L) return(break_ids)
|
||||
sprintf("%s (%s): %s", detail$break_id, detail$break_year, detail$join_advice)
|
||||
}
|
||||
|
||||
# Expand a 2-digit Census popyear (e.g. 19) to a 4-digit calendar year (2019).
|
||||
# F-33 metadata stores popyear as 2 digits; pivot at 70 to handle a future
|
||||
# corpus that ever spans pre-1970 vintages, though current scope is 2000+.
|
||||
|
||||
+13
-3
@@ -128,11 +128,21 @@
|
||||
}
|
||||
|
||||
#' @noRd
|
||||
.validate_schema <- function(manifest, expected_version) {
|
||||
if (manifest$schema_version != expected_version) {
|
||||
#' Schema v6 (FIPS geography harmonization, 2026-07-22) is accepted alongside
|
||||
#' 4/5. v6 renamed the long table's fips_state_code/fips_county_code to
|
||||
#' fips_state_asof/fips_county_asof and added cog_legacy_state/
|
||||
#' cog_legacy_county (26 -> 28 cols); this package references NONE of those
|
||||
#' columns, so no code change was needed. NOTE the SILENT semantic change for
|
||||
#' any consumer of the raw long table: long fips_state/fips_county are now
|
||||
#' PRESENT/harmonized geography (current county identity carried back to every
|
||||
#' year, matching canonical_fips_xwalk) rather than as-of-year; as-of-year
|
||||
#' moved to the *_asof columns. This package's own geography always came from
|
||||
#' the xwalk (already present-based), so behaviour is unchanged.
|
||||
.validate_schema <- function(manifest, supported = c(4L, 5L, 6L)) {
|
||||
if (!manifest$schema_version %in% supported) {
|
||||
cli::cli_abort(c(
|
||||
"Corpus schema version mismatch.",
|
||||
x = "Package expects schema_version = {expected_version}; corpus has {manifest$schema_version}.",
|
||||
x = "Package supports schema_version in {paste(supported, collapse = ', ')}; corpus has {manifest$schema_version}.",
|
||||
i = "Update uscogdata (install.packages or pak::pkg_install) or re-publish corpus."
|
||||
))
|
||||
}
|
||||
|
||||
@@ -143,6 +143,10 @@ cog_find_peers <- function(target_govid,
|
||||
#' @param per_capita Default `TRUE` — peer compare usually normalizes by
|
||||
#' population.
|
||||
#' @param adjust_to_year Integer base year for CPI-U conversion or `NULL`.
|
||||
#' @param expenditure_concept `"direct"` (default) or `"total"`. Currently only
|
||||
#' `"direct"` is accepted; the `"total"` option exists in [cog_spending()] for
|
||||
#' single-government queries but cannot be used here because combining Total
|
||||
#' across peer sets counts intergovernmental transfers twice.
|
||||
#' @return Tibble matching [cog_spending()]'s columns, plus a `role`
|
||||
#' column taking values `"target"`, `"peer"`, `"summary_p25"`,
|
||||
#' `"summary_p50"`, or `"summary_p75"`, `target_rank` (target's rank
|
||||
@@ -153,8 +157,13 @@ cog_find_peers <- function(target_govid,
|
||||
#' `cohort_year`, and `cohort_govids`.
|
||||
#' @export
|
||||
cog_peer_compare <- function(target_govid, peers, category, years,
|
||||
per_capita = TRUE, adjust_to_year = NULL) {
|
||||
per_capita = TRUE, adjust_to_year = NULL,
|
||||
expenditure_concept = c("direct", "total")) {
|
||||
call <- match.call()
|
||||
expenditure_concept <- match.arg(expenditure_concept)
|
||||
if (identical(expenditure_concept, "total")) {
|
||||
.abort_concept_not_aggregatable("cog_peer_compare")
|
||||
}
|
||||
if (!is.character(target_govid) || length(target_govid) != 1L) {
|
||||
cli::cli_abort("`target_govid` must be a length-1 character string.")
|
||||
}
|
||||
|
||||
+27
-2
@@ -4,7 +4,13 @@
|
||||
#' @noRd
|
||||
.build_provenance <- function(verb, call, govid, years, category,
|
||||
per_capita, adjust_to_year, result, sql,
|
||||
subtype_col) {
|
||||
subtype_col, basis = NA_character_,
|
||||
basis_note = NA_character_,
|
||||
expenditure_concept = "direct",
|
||||
expenditure_concept_note = NA_character_,
|
||||
expenditure_concept_direct_suppressed = FALSE,
|
||||
harmonization = NULL, recipe = NULL,
|
||||
suggestions = list()) {
|
||||
manifest <- .uscogdata_env$manifest
|
||||
|
||||
codes <- result[["codes_included"]]
|
||||
@@ -29,6 +35,14 @@
|
||||
unique(result$gov_name)
|
||||
}
|
||||
|
||||
schema_version <- suppressWarnings(as.integer(manifest$schema_version %||% 0L))
|
||||
con <- .uscogdata_env$con
|
||||
break_refs <- if (!is.null(con) && DBI::dbIsValid(con)) {
|
||||
.build_series_break_refs(con, codes_observed, years, schema_version)
|
||||
} else {
|
||||
character(0)
|
||||
}
|
||||
|
||||
list(
|
||||
verb = verb,
|
||||
call = paste(deparse(call), collapse = " "),
|
||||
@@ -38,6 +52,17 @@
|
||||
),
|
||||
years = as.integer(years),
|
||||
category = category,
|
||||
basis = basis,
|
||||
basis_note = basis_note,
|
||||
expenditure_concept = expenditure_concept,
|
||||
expenditure_concept_note = expenditure_concept_note,
|
||||
expenditure_concept_direct_suppressed = isTRUE(expenditure_concept_direct_suppressed),
|
||||
harmonization = harmonization %||% list(
|
||||
applied = FALSE, na_rows_excluded = 0L, na_amount_excluded = 0,
|
||||
note = NA_character_
|
||||
),
|
||||
recipe = recipe,
|
||||
suggestions = suggestions,
|
||||
scope = list(
|
||||
gov_types_included = as.integer(unlist(manifest$scope$gov_types_included)),
|
||||
gov_types_excluded = as.integer(unlist(manifest$scope$gov_types_excluded)),
|
||||
@@ -90,7 +115,7 @@
|
||||
index = if (is.null(adjust_to_year)) NA_character_ else "CPI-U (BLS CPIAUCSL annual average, bundled)"
|
||||
)
|
||||
),
|
||||
series_break_refs = character(0),
|
||||
series_break_refs = break_refs,
|
||||
manifest = list(
|
||||
schema_version = as.integer(manifest$schema_version),
|
||||
pipeline_commit = manifest$pipeline_commit %||% NA_character_,
|
||||
|
||||
+167
@@ -0,0 +1,167 @@
|
||||
# R/recipes.R
|
||||
# Harmonization recipes: multi-code, cross-vintage series built by summing a
|
||||
# fixed set of component item codes with per-component weights and
|
||||
# year/gov-type scoping (see the `harmonization_recipes` view, registered
|
||||
# from data/harmonization_recipes.parquet, schema_version >= 5 only).
|
||||
#
|
||||
# Recipes exist because some cross-vintage series can't be expressed as a
|
||||
# 1:1 harmonized_code mapping (basis = "harmonized"): the wide era (pre-2012)
|
||||
# publishes only a combined aggregate row for these families (e.g.
|
||||
# corrections functions 04+05), while the modern era splits them into leaf
|
||||
# codes. A recipe's generic join sums whichever of its component codes are
|
||||
# present for a given year, so the resulting series is continuous across
|
||||
# that format boundary.
|
||||
|
||||
#' List available harmonization recipes
|
||||
#'
|
||||
#' Recipes are multi-code cross-vintage series (see [cog_spending()]'s
|
||||
#' `recipe` argument) catalogued in the corpus's `harmonization_recipes`
|
||||
#' table. Use this to discover valid `recipe` ids.
|
||||
#'
|
||||
#' @param pattern Optional regex matched case-insensitively against
|
||||
#' `recipe_id` or `label`.
|
||||
#' @return Tibble with columns `recipe_id`, `label`, `n_components`,
|
||||
#' `year_min`, `year_max` (the min/max component year coverage), sorted by
|
||||
#' `recipe_id`.
|
||||
#' @export
|
||||
cog_recipes <- function(pattern = NULL) {
|
||||
if (!is.null(pattern) &&
|
||||
(!is.character(pattern) || length(pattern) != 1L)) {
|
||||
cli::cli_abort("`pattern` must be a length-1 character string or NULL.")
|
||||
}
|
||||
con <- .ensure_session()
|
||||
.require_schema_v5(con, .uscogdata_env$manifest, "cog_recipes()")
|
||||
|
||||
where <- if (is.null(pattern)) {
|
||||
""
|
||||
} else {
|
||||
sprintf(
|
||||
"WHERE regexp_matches(recipe_id, %1$s, 'i') OR regexp_matches(label, %1$s, 'i')",
|
||||
.sql_lit_chr(pattern)
|
||||
)
|
||||
}
|
||||
sql <- paste(
|
||||
"SELECT recipe_id, any_value(label) AS label,
|
||||
COUNT(*) AS n_components,
|
||||
MIN(year_min) AS year_min, MAX(year_max) AS year_max
|
||||
FROM harmonization_recipes",
|
||||
where,
|
||||
"GROUP BY recipe_id
|
||||
ORDER BY recipe_id"
|
||||
)
|
||||
out <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
out$year_min <- as.integer(out$year_min)
|
||||
out$year_max <- as.integer(out$year_max)
|
||||
out$n_components <- as.integer(out$n_components)
|
||||
out
|
||||
}
|
||||
|
||||
#' Abort unless the active corpus has schema_version >= 5.
|
||||
#' @noRd
|
||||
.require_schema_v5 <- function(con, manifest, what) {
|
||||
sv <- suppressWarnings(as.integer(manifest$schema_version %||% 0L))
|
||||
if (sv < 5L) {
|
||||
cli::cli_abort(c(
|
||||
sprintf("%s requires corpus schema_version >= 5.", what),
|
||||
x = "Active corpus has schema_version {sv}.",
|
||||
i = "Point USCOGDATA_URL at a schema_version >= 5 corpus to use harmonization recipes."
|
||||
), class = "uscogdata_schema_unsupported")
|
||||
}
|
||||
invisible(sv)
|
||||
}
|
||||
|
||||
#' Abort with the valid id list unless `recipe_id` exists in the catalog.
|
||||
#' @noRd
|
||||
.validate_recipe_id <- function(con, recipe_id) {
|
||||
ids <- DBI::dbGetQuery(
|
||||
con, "SELECT DISTINCT recipe_id FROM harmonization_recipes"
|
||||
)$recipe_id
|
||||
if (!recipe_id %in% ids) {
|
||||
cli::cli_abort(c(
|
||||
"Unknown recipe = {.val {recipe_id}}.",
|
||||
i = "Valid ids: {paste(sort(ids), collapse = ', ')}",
|
||||
i = "See cog_recipes() for labels and year coverage."
|
||||
), class = "uscogdata_unknown_recipe")
|
||||
}
|
||||
invisible(TRUE)
|
||||
}
|
||||
|
||||
#' Fetch the component rows for one recipe (label, component codes, scope,
|
||||
#' year ranges, weights) -- both for running the recipe and for the
|
||||
#' `recipe` provenance block.
|
||||
#' @noRd
|
||||
.recipe_components <- function(con, recipe_id) {
|
||||
sql <- sprintf(
|
||||
"SELECT recipe_id, label, component_code, gov_type_scope,
|
||||
year_min, year_max, weight, source_break_ids, notes
|
||||
FROM harmonization_recipes
|
||||
WHERE recipe_id = %s
|
||||
ORDER BY component_code",
|
||||
.sql_lit_chr(recipe_id)
|
||||
)
|
||||
tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
}
|
||||
|
||||
#' Run a recipe's generic join: sum `amt * weight` across whichever
|
||||
#' component codes are present for each (year, canonical_govid), scoped by
|
||||
#' gov_type_scope. Deliberately does NOT filter `NOT is_aggregate`: in the
|
||||
#' wide era (<= 2011) these families' component codes exist ONLY as
|
||||
#' aggregate rows (leaves first appear 2012), so excluding aggregates would
|
||||
#' zero out the wide-era half of every recipe. This is safe by corpus
|
||||
#' construction -- wide-era rows for these codes are aggregate-only, modern
|
||||
#' rows are leaf-only, and every component row is year-scoped via
|
||||
#' `year_min`/`year_max` -- so there is no double-counting. (Checkpoint
|
||||
#' review docs/phase_r_harmonization_review.md § 0.2.)
|
||||
#' @noRd
|
||||
.run_recipe <- function(con, recipe_id, govid, years) {
|
||||
sql <- sprintf(
|
||||
"SELECT l.year, l.canonical_govid,
|
||||
COALESCE(x.gov_name, l.gov_name) AS gov_name,
|
||||
SUM(l.amt * r.weight) * 1000.0 AS amt_nominal,
|
||||
string_agg(DISTINCT l.item_code, ',' ORDER BY l.item_code) AS codes_included
|
||||
FROM long l
|
||||
JOIN harmonization_recipes r
|
||||
ON l.item_code = r.component_code
|
||||
AND l.year BETWEEN r.year_min AND r.year_max
|
||||
AND (r.gov_type_scope = 'all'
|
||||
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
||||
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||
LEFT JOIN canonical_fips_xwalk x USING (canonical_govid)
|
||||
WHERE r.recipe_id = %1$s
|
||||
AND l.canonical_govid IN (%2$s)
|
||||
AND l.year IN (%3$s)
|
||||
GROUP BY 1, 2, 3
|
||||
ORDER BY 1, 2",
|
||||
.sql_lit_chr(recipe_id), .sql_lit_chr(govid),
|
||||
paste(as.integer(years), collapse = ",")
|
||||
)
|
||||
result <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
attr(result, "sql_query") <- sql
|
||||
result
|
||||
}
|
||||
|
||||
#' Shape a raw .run_recipe() result into the standard cog_spending()/
|
||||
#' cog_revenue() column layout: subtype = "recipe", category = the recipe's
|
||||
#' label, aggregate_fallback = FALSE (recipes resolve coverage gaps by
|
||||
#' construction, not by falling back to an aggregate row).
|
||||
#' @noRd
|
||||
.shape_recipe_result <- function(result, subtype_col, label) {
|
||||
sql_query <- attr(result, "sql_query")
|
||||
n <- nrow(result)
|
||||
result[[subtype_col]] <- rep("recipe", n)
|
||||
result$category <- rep(label, n)
|
||||
result$aggregate_fallback <- rep(FALSE, n)
|
||||
result <- result[, c(
|
||||
"year", "canonical_govid", "gov_name", subtype_col, "category",
|
||||
"amt_nominal", "codes_included", "aggregate_fallback"
|
||||
), drop = FALSE]
|
||||
attr(result, "sql_query") <- sql_query
|
||||
result
|
||||
}
|
||||
|
||||
#' Turn a small data.frame into a list-of-lists (one list per row), the
|
||||
#' shape used for the `recipe$components` provenance block.
|
||||
#' @noRd
|
||||
.df_to_row_list <- function(df) {
|
||||
lapply(seq_len(nrow(df)), function(i) as.list(df[i, , drop = FALSE]))
|
||||
}
|
||||
+7
-3
@@ -14,16 +14,20 @@
|
||||
#' optional `pop_source`, `codes_included`, `aggregate_fallback`, `notes`.
|
||||
#' @export
|
||||
cog_revenue <- function(govid, years, category = NULL,
|
||||
per_capita = FALSE, adjust_to_year = NULL) {
|
||||
per_capita = FALSE, adjust_to_year = NULL,
|
||||
basis = c("harmonized", "raw"), recipe = NULL) {
|
||||
.verb_spendrev(
|
||||
verb = "cog_revenue",
|
||||
view = "revenue_annotated",
|
||||
view_base = "revenue_annotated",
|
||||
subtype_col = "revenue_subtype",
|
||||
flow_prefixes = c("T", "A", "U", "B", "C", "D"),
|
||||
call = match.call(),
|
||||
govid = govid,
|
||||
years = years,
|
||||
category = category,
|
||||
per_capita = per_capita,
|
||||
adjust_to_year = adjust_to_year
|
||||
adjust_to_year = adjust_to_year,
|
||||
basis = basis,
|
||||
recipe = recipe
|
||||
)
|
||||
}
|
||||
|
||||
+12
-1
@@ -25,6 +25,12 @@
|
||||
#' population from `gov_population_yearly`. Govs with missing population
|
||||
#' are excluded from the result.
|
||||
#' @param adjust_to_year Integer base year for CPI-U conversion, or `NULL`.
|
||||
#' @param expenditure_concept `"direct"` (default) or `"total"`. Currently only
|
||||
#' `"direct"` is accepted; the `"total"` option exists in [cog_spending()] for
|
||||
#' single-government queries but cannot be used here because combining Total
|
||||
#' across multiple layers of government double-counts intergovernmental
|
||||
#' transfers (a state's payment to a school district is the same dollar the
|
||||
#' district reports as its own Direct spending).
|
||||
#' @return Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`,
|
||||
#' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real` /
|
||||
#' `amt_per_capita_nominal` / `amt_per_capita_real`, optional `pop_source`,
|
||||
@@ -33,8 +39,13 @@
|
||||
#' and `rollup$included_govids` / `rollup$excluded_govids`.
|
||||
#' @export
|
||||
cog_geographic_rollup <- function(govids, category, years,
|
||||
per_capita = FALSE, adjust_to_year = NULL) {
|
||||
per_capita = FALSE, adjust_to_year = NULL,
|
||||
expenditure_concept = c("direct", "total")) {
|
||||
call <- match.call()
|
||||
expenditure_concept <- match.arg(expenditure_concept)
|
||||
if (identical(expenditure_concept, "total")) {
|
||||
.abort_concept_not_aggregatable("cog_geographic_rollup")
|
||||
}
|
||||
.validate_rollup_layers(govids)
|
||||
|
||||
govids <- lapply(govids, .coerce_govid_input, arg = "govids[[layer]]")
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
# R/series_breaks.R
|
||||
# Populates prov$series_break_refs (schema in inst/schemas/provenance-v1.json
|
||||
# defines the field; it was always present but always empty pre-Phase-R2)
|
||||
# with the ids of any catalogued series break whose fin_code appears among
|
||||
# the result's observed item codes and whose break_year falls inside the
|
||||
# requested year span -- the "break warnings in the provenance envelope"
|
||||
# spec § 5 promises downstream consumers (cog-api passes provenance through
|
||||
# verbatim). schema_version >= 5 only: series_breaks_pq isn't registered on
|
||||
# an older corpus.
|
||||
|
||||
#' @noRd
|
||||
.build_series_break_refs <- function(con, codes_observed, years, schema_version) {
|
||||
if (schema_version < 5L || length(codes_observed) == 0L) return(character(0))
|
||||
sql <- sprintf(
|
||||
"SELECT DISTINCT break_id
|
||||
FROM series_breaks_pq
|
||||
WHERE fin_code IN (%s) AND break_year BETWEEN %d AND %d
|
||||
ORDER BY break_id",
|
||||
.sql_lit_chr(codes_observed), min(as.integer(years)), max(as.integer(years))
|
||||
)
|
||||
DBI::dbGetQuery(con, sql)$break_id
|
||||
}
|
||||
+1
-1
@@ -12,7 +12,7 @@ cog_open <- function(url = .resolve_url(),
|
||||
DBI::dbExecute(con, "INSTALL httpfs; LOAD httpfs;")
|
||||
|
||||
manifest <- .fetch_or_cache_manifest(url, cache_dir)
|
||||
.validate_schema(manifest, expected_version = 4L)
|
||||
.validate_schema(manifest, supported = c(4L, 5L, 6L))
|
||||
.validate_scope(manifest)
|
||||
|
||||
.register_views(con, url, manifest)
|
||||
|
||||
+453
-16
@@ -20,6 +20,56 @@
|
||||
#' in that year).
|
||||
#' @param adjust_to_year Integer base year for CPI-U real-dollar conversion,
|
||||
#' or `NULL` for nominal only.
|
||||
#' @param basis `"harmonized"` (default) sums item codes through the
|
||||
#' cross-vintage harmonization mapping (folding series-break-affected
|
||||
#' codes onto a comparable target and excluding aggregate / discontinued
|
||||
#' rows -- see the `harmonization` block in `cog_explain()`); `"raw"`
|
||||
#' reproduces the pre-Phase-R2 behavior (published item codes, no
|
||||
#' folding). On a corpus with `schema_version < 5` (no harmonization
|
||||
#' tables), `basis` silently resolves to `"raw"` when left at its default
|
||||
#' and the resolution is recorded in the provenance; explicitly passing
|
||||
#' `basis = "harmonized"` on such a corpus aborts. Ignored when `recipe`
|
||||
#' is set (see below).
|
||||
#' @param recipe Optional harmonization recipe id (see [cog_recipes()]) for
|
||||
#' multi-code cross-vintage series that a 1:1 harmonized_code mapping
|
||||
#' can't express (e.g. a wide-era aggregate that only splits into leaf
|
||||
#' codes in the modern era). Mutually exclusive with `category`. The
|
||||
#' result's subtype column reads `"recipe"` and `category` reads the
|
||||
#' recipe's label. Requires `schema_version >= 5`. A recipe query bypasses
|
||||
#' `basis` entirely (it joins `long` directly rather than going through
|
||||
#' the `*_annotated`/`*_annotated_harmonized` views), so the `basis`
|
||||
#' argument is ignored and the result's provenance reports
|
||||
#' `basis = "recipe"` with an inert `harmonization` block (`applied =
|
||||
#' FALSE`, pointing at the `recipe` block instead) rather than a
|
||||
#' possibly-misleading `"harmonized"`/`"raw"` value.
|
||||
#' @param expenditure_concept `"direct"` (default) returns only the
|
||||
#' government's own direct spending (item codes `E`/`F`/`G`), unchanged
|
||||
#' from prior releases. `"total"` additionally UNIONs in the
|
||||
#' intergovernmental leg -- payments to local governments (`M` codes) and
|
||||
#' to the state government (`L` codes, excluding the `L--` family-total
|
||||
#' rollup) -- so results gain rows with `spend_subtype ==
|
||||
#' "intergovernmental"`. Requires the active corpus's `summary_categories`
|
||||
#' to carry M/L rows (added by cog_pipeline PR #59); aborts with class
|
||||
#' `uscogdata_ig_categories_unsupported` on an older corpus rather than
|
||||
#' silently under-reporting. Mutually exclusive with `recipe` (a recipe
|
||||
#' already defines its own component codes). **Do not sum `"total"`
|
||||
#' results across levels of government** (e.g. state + county + city):
|
||||
#' a state's `M12` payment to a school district is the same dollar the
|
||||
#' district reports as its own direct `E12`, so summing both double-counts
|
||||
#' it. This matters in particular with [cog_geographic_rollup()], which
|
||||
#' sums across exactly that kind of multi-layer government set.
|
||||
#'
|
||||
#' In the legacy wide era (<= FY2011), some functions are published ONLY
|
||||
#' as an aggregate-flagged family total (e.g. Corrections' `E04`/`E05`
|
||||
#' split), which the Direct leg excludes by construction but the IG leg
|
||||
#' deliberately keeps (see `inst/sql/24-ig_long.sql`). For a `"total"`
|
||||
#' query, any (year, category) where this leaves intergovernmental rows
|
||||
#' with NO Direct counterpart is flagged: the affected rows' `notes`
|
||||
#' name the harmonization recipe that recovers the missing Direct
|
||||
#' component (when one exists), and
|
||||
#' `provenance$expenditure_concept_direct_suppressed` is `TRUE` -- the
|
||||
#' figure in those rows is the intergovernmental leg alone, not Direct +
|
||||
#' IG.
|
||||
#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||
#' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real`,
|
||||
#' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
|
||||
@@ -27,65 +77,253 @@
|
||||
#' Carries a `provenance` attribute matching `inst/schemas/provenance-v1.json`.
|
||||
#' @export
|
||||
cog_spending <- function(govid, years, category = NULL,
|
||||
per_capita = FALSE, adjust_to_year = NULL) {
|
||||
per_capita = FALSE, adjust_to_year = NULL,
|
||||
basis = c("harmonized", "raw"), recipe = NULL,
|
||||
expenditure_concept = c("direct", "total")) {
|
||||
.verb_spendrev(
|
||||
verb = "cog_spending",
|
||||
view = "spending_annotated",
|
||||
view_base = "spending_annotated",
|
||||
subtype_col = "spend_subtype",
|
||||
flow_prefixes = c("E", "F", "G"),
|
||||
call = match.call(),
|
||||
govid = govid,
|
||||
years = years,
|
||||
category = category,
|
||||
per_capita = per_capita,
|
||||
adjust_to_year = adjust_to_year
|
||||
adjust_to_year = adjust_to_year,
|
||||
basis = basis,
|
||||
recipe = recipe,
|
||||
expenditure_concept = expenditure_concept
|
||||
)
|
||||
}
|
||||
|
||||
#' @noRd
|
||||
.verb_spendrev <- function(verb, view, subtype_col, call,
|
||||
.abort_concept_not_aggregatable <- function(verb) {
|
||||
cli::cli_abort(c(
|
||||
"{.code expenditure_concept = \"total\"} cannot be used in {.fn {verb}}.",
|
||||
"*" = "Use {.code expenditure_concept = \"direct\"} (the default) for any \\
|
||||
comparison or sum that spans more than one government.",
|
||||
"i" = "Why: Census \"Total\" is a government's own Direct spending PLUS the \\
|
||||
money it hands to other governments. The receiving government reports \\
|
||||
that same dollar again as its own Direct when it actually spends it, \\
|
||||
so combining Total across governments double-counts intergovernmental \\
|
||||
transfers.",
|
||||
"i" = "For one government's own Total, use \\
|
||||
{.code cog_spending(expenditure_concept = \"total\")}."
|
||||
), class = "uscogdata_concept_not_aggregatable")
|
||||
}
|
||||
|
||||
#' @noRd
|
||||
.verb_spendrev <- function(verb, view_base, subtype_col, flow_prefixes, call,
|
||||
govid, years, category,
|
||||
per_capita, adjust_to_year) {
|
||||
per_capita, adjust_to_year,
|
||||
basis = c("harmonized", "raw"), recipe = NULL,
|
||||
expenditure_concept = c("direct", "total")) {
|
||||
basis_explicit <- length(basis) == 1L
|
||||
basis <- match.arg(basis, c("harmonized", "raw"))
|
||||
# match.arg() itself throws a base `simpleError`, not an rlang-classed
|
||||
# condition; wrap it so an invalid expenditure_concept aborts consistently
|
||||
# with the rest of this package's validation (cli::cli_abort -> rlang_error).
|
||||
expenditure_concept <- tryCatch(
|
||||
match.arg(expenditure_concept, c("direct", "total")),
|
||||
error = function(e) {
|
||||
cli::cli_abort(
|
||||
"`expenditure_concept` must be one of {.val direct} or {.val total}.",
|
||||
class = "uscogdata_invalid_expenditure_concept",
|
||||
parent = e
|
||||
)
|
||||
}
|
||||
)
|
||||
|
||||
govid <- .coerce_govid_input(govid, arg = "govid")
|
||||
.validate_verb_inputs(govid, years, category, per_capita, adjust_to_year)
|
||||
.validate_verb_inputs(govid, years, category, per_capita, adjust_to_year,
|
||||
recipe)
|
||||
|
||||
if (!is.null(recipe) && identical(expenditure_concept, "total")) {
|
||||
cli::cli_abort(c(
|
||||
"`recipe` and `expenditure_concept = \"total\"` are mutually exclusive.",
|
||||
i = "A recipe defines its own component codes; pass one or the other.",
|
||||
i = "For a recipe's intergovernmental counterpart, use the matching IG recipe (e.g. `corrections_ig_local_combined`)."
|
||||
), class = "uscogdata_recipe_concept_conflict")
|
||||
}
|
||||
|
||||
# .verb_spendrev() is shared with cog_revenue(), which never exposes
|
||||
# expenditure_concept and always resolves it to "direct" -- so nothing on
|
||||
# the public API can reach this today. But it's a cheap guard against a
|
||||
# future call (direct or via a modified cog_revenue()) that would UNION
|
||||
# the IG leg's expenditure M/L rows into a revenue result, which has no
|
||||
# matching IG view and no sensible meaning.
|
||||
if (identical(expenditure_concept, "total") &&
|
||||
!identical(view_base, "spending_annotated")) {
|
||||
cli::cli_abort(
|
||||
paste0(
|
||||
"`expenditure_concept = \"total\"` is only supported for spending ",
|
||||
"(view_base = \"spending_annotated\"); got view_base = ",
|
||||
"{.val {view_base}}."
|
||||
),
|
||||
class = "uscogdata_expenditure_concept_unsupported"
|
||||
)
|
||||
}
|
||||
|
||||
years <- as.integer(years)
|
||||
if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year)
|
||||
|
||||
con <- .ensure_session()
|
||||
manifest <- .uscogdata_env$manifest
|
||||
scope <- .check_govids_in_scope(govid)
|
||||
|
||||
sql <- .build_verb_sql(view, subtype_col, govid, years, category)
|
||||
resolved <- .resolve_basis(basis, basis_explicit, manifest)
|
||||
|
||||
recipe_block <- NULL
|
||||
category_for_prov <- category
|
||||
if (!is.null(recipe)) {
|
||||
.require_schema_v5(con, manifest, "recipe =")
|
||||
.validate_recipe_id(con, recipe)
|
||||
comps <- .recipe_components(con, recipe)
|
||||
recipe_label <- comps$label[[1]]
|
||||
result <- .run_recipe(con, recipe, govid, years)
|
||||
sql <- attr(result, "sql_query")
|
||||
result <- .shape_recipe_result(result, subtype_col, recipe_label)
|
||||
recipe_block <- list(
|
||||
recipe_id = recipe, label = recipe_label,
|
||||
components = .df_to_row_list(comps)
|
||||
)
|
||||
category_for_prov <- recipe_label
|
||||
} else {
|
||||
view <- .select_view(view_base, resolved$basis)
|
||||
ig_view <- if (identical(expenditure_concept, "total")) {
|
||||
.require_ig_categories(con)
|
||||
.select_ig_view(resolved$basis)
|
||||
} else {
|
||||
NULL
|
||||
}
|
||||
sql <- .build_verb_sql(view, subtype_col, govid, years, category, ig_view)
|
||||
result <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||
}
|
||||
|
||||
if (per_capita) result <- .attach_per_capita(result, con, govid)
|
||||
if (!is.null(adjust_to_year)) {
|
||||
result <- .attach_real_dollars(result, adjust_to_year, per_capita)
|
||||
}
|
||||
|
||||
result$notes <- .notes_column(result)
|
||||
# A recipe result doesn't go through spending_annotated(_harmonized) /
|
||||
# revenue_annotated(_harmonized) at all -- .run_recipe()'s generic join
|
||||
# reads `long` directly -- so `basis` and the `harmonization` exclusion
|
||||
# count (which is itself computed from `long`, independent of which view
|
||||
# a non-recipe query used) would describe a code path this result never
|
||||
# took. Rather than report a technically-still-computed but misleading
|
||||
# basis = "harmonized"/"raw" + harmonization$applied combo, recipe
|
||||
# results report basis = "recipe" and an explicit, inert harmonization
|
||||
# block pointing at the `recipe` block instead. Task 12 (cog-api) passes
|
||||
# provenance through verbatim, so this needs to be unambiguous rather
|
||||
# than technically-defensible-but-confusing.
|
||||
if (!is.null(recipe)) {
|
||||
basis_for_prov <- "recipe"
|
||||
basis_note_for_prov <- NA_character_
|
||||
harmonization <- list(
|
||||
applied = FALSE, na_rows_excluded = 0L, na_amount_excluded = 0,
|
||||
note = "basis/harmonization not applicable to recipe results; see the recipe block instead"
|
||||
)
|
||||
suggestions <- list()
|
||||
} else {
|
||||
basis_for_prov <- resolved$basis
|
||||
basis_note_for_prov <- resolved$note
|
||||
harmonization <- .build_harmonization_block(
|
||||
con, govid, years, resolved, flow_prefixes
|
||||
)
|
||||
# C1(a): gap detection must run against the Direct leg alone. `result`
|
||||
# can also carry UNION'd intergovernmental rows (expenditure_concept =
|
||||
# "total"), and the wide era (<= FY2011) routinely has legacy IG dollars
|
||||
# surviving (ig_long deliberately keeps aggregate rows) for a
|
||||
# (year, category) whose legacy Direct dollars were suppressed (spending_
|
||||
# long/spending_long_harmonized both filter NOT is_aggregate). Passing
|
||||
# the UNION'd result here would let a surviving IG row count as coverage
|
||||
# and silently cancel the recipe-hint suggestion that should fire.
|
||||
direct_leg_result <- if (identical(expenditure_concept, "total")) {
|
||||
result[!(result[[subtype_col]] %in% "intergovernmental"), , drop = FALSE]
|
||||
} else {
|
||||
result
|
||||
}
|
||||
suggestions <- .build_suggestions(con, govid, years, category,
|
||||
direct_leg_result,
|
||||
resolved$basis, flow_prefixes)
|
||||
}
|
||||
|
||||
# C1(b): when expenditure_concept = "total", flag any row where the IG
|
||||
# leg has dollars but the Direct leg has none for that same (year,
|
||||
# canonical_govid, category) AND a harmonization recipe actually recovers
|
||||
# the missing Direct dollars for that exact triple -- see
|
||||
# .detect_direct_suppressed() for why bare Direct-row absence alone is NOT
|
||||
# sufficient (the dominant real cause is a government that simply has no
|
||||
# direct spending in that category, which is correct, ordinary data). When
|
||||
# a covering recipe is found, both the row-level notes and the provenance
|
||||
# say so rather than pass silently as a plausible Total.
|
||||
direct_suppressed_info <- if (identical(expenditure_concept, "total")) {
|
||||
.detect_direct_suppressed(con, result, subtype_col)
|
||||
} else {
|
||||
list(flag = rep(FALSE, nrow(result)), notes = rep(NA_character_, nrow(result)))
|
||||
}
|
||||
direct_suppressed <- direct_suppressed_info$flag
|
||||
direct_suppressed_flag <- isTRUE(any(direct_suppressed))
|
||||
|
||||
result$notes <- .notes_column(result, direct_suppressed_info$notes)
|
||||
|
||||
# Determine expenditure_concept_note: only non-empty for "total", explains
|
||||
# how the IG leg was assembled from legacy-era aggregates. When the Direct
|
||||
# leg is suppressed for at least one requested (year, category), append an
|
||||
# explicit warning rather than let the base note's "Total = Direct + IG"
|
||||
# framing stand unqualified for rows where that arithmetic didn't happen.
|
||||
expenditure_concept_note_for_prov <- if (identical(expenditure_concept, "total")) {
|
||||
base_note <- "Total = Direct + intergovernmental (M to local govts + L to state govts). Legacy-era IG is assembled from aggregate-flagged rows, which are year-disjoint from their modern leaf components; the L-- family total is excluded."
|
||||
if (direct_suppressed_flag) {
|
||||
paste0(
|
||||
base_note,
|
||||
" NOTE: for at least one requested (year, category) the Direct leg ",
|
||||
"has NO rows in this corpus (a legacy aggregate-only family) -- the ",
|
||||
"affected result rows report the intergovernmental leg alone, not ",
|
||||
"Direct + IG. See `expenditure_concept_direct_suppressed` and each ",
|
||||
"affected row's `notes`."
|
||||
)
|
||||
} else {
|
||||
base_note
|
||||
}
|
||||
} else {
|
||||
NA_character_
|
||||
}
|
||||
|
||||
prov <- .build_provenance(
|
||||
verb = verb,
|
||||
call = call,
|
||||
govid = govid,
|
||||
years = years,
|
||||
category = category,
|
||||
category = category_for_prov,
|
||||
per_capita = per_capita,
|
||||
adjust_to_year = adjust_to_year,
|
||||
result = result,
|
||||
sql = sql,
|
||||
subtype_col = subtype_col
|
||||
subtype_col = subtype_col,
|
||||
basis = basis_for_prov,
|
||||
basis_note = basis_note_for_prov,
|
||||
expenditure_concept = expenditure_concept,
|
||||
expenditure_concept_note = expenditure_concept_note_for_prov,
|
||||
expenditure_concept_direct_suppressed = direct_suppressed_flag,
|
||||
harmonization = harmonization,
|
||||
recipe = recipe_block,
|
||||
suggestions = suggestions
|
||||
)
|
||||
prov$scope$govids_found <- scope$found
|
||||
prov$scope$govids_missing <- scope$missing
|
||||
attr(result, "provenance") <- prov
|
||||
attr(result, ".popyear_range") <- NULL
|
||||
|
||||
if (length(suggestions) > 0L) .inform_suggestions(suggestions)
|
||||
|
||||
result
|
||||
}
|
||||
|
||||
#' @noRd
|
||||
.validate_verb_inputs <- function(govid, years, category,
|
||||
per_capita, adjust_to_year) {
|
||||
per_capita, adjust_to_year, recipe = NULL) {
|
||||
if (!is.character(govid) || length(govid) == 0L) {
|
||||
cli::cli_abort("`govid` must be a non-empty character vector.")
|
||||
}
|
||||
@@ -104,6 +342,59 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
cli::cli_abort("`adjust_to_year` must be NULL or a length-1 integer.")
|
||||
}
|
||||
}
|
||||
if (!is.null(recipe)) {
|
||||
if (!is.character(recipe) || length(recipe) != 1L) {
|
||||
cli::cli_abort("`recipe` must be NULL or a length-1 character string.")
|
||||
}
|
||||
if (!is.null(category)) {
|
||||
cli::cli_abort(c(
|
||||
"`recipe` and `category` are mutually exclusive.",
|
||||
i = "Pass one or the other, not both."
|
||||
), class = "uscogdata_recipe_category_conflict")
|
||||
}
|
||||
}
|
||||
invisible(TRUE)
|
||||
}
|
||||
|
||||
#' @noRd
|
||||
.select_view <- function(view_base, basis) {
|
||||
if (identical(basis, "harmonized")) paste0(view_base, "_harmonized") else view_base
|
||||
}
|
||||
|
||||
#' @noRd
|
||||
.select_ig_view <- function(basis) {
|
||||
if (identical(basis, "harmonized")) "ig_annotated_harmonized" else "ig_annotated"
|
||||
}
|
||||
|
||||
#' Abort unless the active corpus's `summary_categories` actually carries
|
||||
#' intergovernmental (M/L) rows.
|
||||
#'
|
||||
#' The 66 M/L category rows arrived via cog_pipeline PR #59 with NO
|
||||
#' `schema_version` bump (`DESCRIPTION` still declares `MinCorpusSchema: 4`),
|
||||
#' so `schema_version` alone cannot gate `expenditure_concept = "total"` --
|
||||
#' a pre-#59 corpus can validly report schema_version 4, 5, or 6 and still
|
||||
#' have zero M/L rows in `summary_categories`. Against such a corpus,
|
||||
#' `ig_annotated`'s LEFT JOIN to `summary_categories` silently produces NA
|
||||
#' `category`/`spend_subtype` for every IG row: with a `category` filter
|
||||
#' this returns 0 rows (reads as "no intergovernmental spending" rather than
|
||||
#' "can't tell"), and with `category = NULL` every IG dollar collapses into
|
||||
#' one NA-subtype group that is invisible to the `spend_subtype ==
|
||||
#' "intergovernmental"` filter this package's own tests, roxygen, and
|
||||
#' vignette all rely on. Checking the data directly (rather than
|
||||
#' schema_version) is the only reliable gate.
|
||||
#' @noRd
|
||||
.require_ig_categories <- function(con, what = "expenditure_concept = \"total\"") {
|
||||
n <- DBI::dbGetQuery(con,
|
||||
"SELECT COUNT(*) AS n FROM summary_categories WHERE LEFT(item_code, 1) IN ('M', 'L')"
|
||||
)$n
|
||||
if (identical(as.integer(n), 0L)) {
|
||||
cli::cli_abort(c(
|
||||
sprintf("%s requires a corpus with intergovernmental category rows.", what),
|
||||
x = "The active corpus's `summary_categories` has no M/L (intergovernmental) rows.",
|
||||
i = "This corpus predates the intergovernmental category rows added by cog_pipeline PR #59.",
|
||||
i = "Point USCOGDATA_URL at a newer corpus that includes the M/L summary_categories rows."
|
||||
), class = "uscogdata_ig_categories_unsupported")
|
||||
}
|
||||
invisible(TRUE)
|
||||
}
|
||||
|
||||
@@ -114,7 +405,8 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
}
|
||||
|
||||
#' @noRd
|
||||
.build_verb_sql <- function(view, subtype_col, govid, years, category) {
|
||||
.build_verb_sql <- function(view, subtype_col, govid, years, category,
|
||||
ig_view = NULL) {
|
||||
govid_lit <- .sql_lit_chr(govid)
|
||||
years_lit <- paste(as.integer(years), collapse = ",")
|
||||
category_pred <- if (is.null(category)) {
|
||||
@@ -123,6 +415,26 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
sprintf("AND category IN (%s)", .sql_lit_chr(category))
|
||||
}
|
||||
|
||||
# expenditure_concept = "total" adds the intergovernmental leg. UNION ALL,
|
||||
# never UNION: the two legs are disjoint by item_code prefix (E/F/G vs M/L),
|
||||
# so de-duplication would be pure cost, and a silent row-drop if two
|
||||
# governments ever reported identical values.
|
||||
source_expr <- if (is.null(ig_view)) {
|
||||
view
|
||||
} else {
|
||||
sprintf("(SELECT * FROM %s UNION ALL SELECT * FROM %s)", view, ig_view)
|
||||
}
|
||||
|
||||
# bool_or(), not bool_and(): a no-op for the Direct/revenue legs (those
|
||||
# views filter NOT is_aggregate, so no row in any group is ever aggregate),
|
||||
# but load-bearing for the IG leg, which deliberately keeps aggregate rows
|
||||
# (see inst/sql/24-ig_long.sql). The wide era is dense -- every government
|
||||
# has a row for every code in a family, most of them $0 -- so a $0 leaf
|
||||
# commonly lands in the same (year, gov, subtype, category) group as the
|
||||
# real aggregate row. bool_and() would then read FALSE for that group even
|
||||
# though its dollars came entirely from an aggregate row, silently
|
||||
# suppressing the "Aggregate fallback applied" note on exactly the rows
|
||||
# this feature exists to surface.
|
||||
sprintf(
|
||||
"SELECT
|
||||
year,
|
||||
@@ -132,14 +444,14 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
category,
|
||||
SUM(amt) * 1000.0 AS amt_nominal,
|
||||
string_agg(DISTINCT item_code, ',' ORDER BY item_code) AS codes_included,
|
||||
bool_and(is_aggregate) AS aggregate_fallback
|
||||
bool_or(is_aggregate) AS aggregate_fallback
|
||||
FROM %2$s
|
||||
WHERE canonical_govid IN (%3$s)
|
||||
AND year IN (%4$s)
|
||||
%5$s
|
||||
GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, %1$s, category
|
||||
ORDER BY year, canonical_govid, %1$s, category",
|
||||
subtype_col, view, govid_lit, years_lit, category_pred
|
||||
subtype_col, source_expr, govid_lit, years_lit, category_pred
|
||||
)
|
||||
}
|
||||
|
||||
@@ -192,11 +504,131 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
result
|
||||
}
|
||||
|
||||
#' Detect rows where expenditure_concept = "total" is reporting the
|
||||
#' intergovernmental leg with NO Direct counterpart in the same (year,
|
||||
#' canonical_govid, category) group AND a harmonization recipe actually
|
||||
#' recovers the missing Direct dollars for that exact (year, canonical_govid,
|
||||
#' category) triple.
|
||||
#'
|
||||
#' Bare Direct-row absence is deliberately NOT sufficient on its own: the
|
||||
#' dominant real cause of "no Direct sibling row" is a government that simply
|
||||
#' has no direct spending in that category (e.g. a state that funds K-12
|
||||
#' entirely through school districts), which is correct, ordinary data, not
|
||||
#' suppression. Genuine suppression -- a legacy aggregate-only family whose
|
||||
#' Direct-leg basis query excludes it by construction (spending_long/
|
||||
#' spending_long_harmonized both filter NOT is_aggregate) -- always has a
|
||||
#' covering harmonization recipe, because that is exactly what the recipe
|
||||
#' catalog exists to recover (see R/suggestions.R and `cog_recipes()`). So
|
||||
#' checking "does a recipe actually cover this triple" cleanly separates the
|
||||
#' two cases instead of conflating them.
|
||||
#'
|
||||
#' Returns `list(flag, notes)`, both the same length as `result`: `flag` is
|
||||
#' `TRUE` only for the `spend_subtype == "intergovernmental"` row(s) in a
|
||||
#' suppressed group, and `notes` names the recovering recipe(s) for those
|
||||
#' rows (`NA` everywhere else).
|
||||
#' @noRd
|
||||
.notes_column <- function(result) {
|
||||
.detect_direct_suppressed <- function(con, result, subtype_col) {
|
||||
n <- nrow(result)
|
||||
empty_notes <- rep(NA_character_, n)
|
||||
if (n == 0L) return(list(flag = logical(0), notes = character(0)))
|
||||
is_ig <- result[[subtype_col]] %in% "intergovernmental"
|
||||
if (!any(is_ig)) return(list(flag = rep(FALSE, n), notes = empty_notes))
|
||||
|
||||
key <- paste(result$year, result$canonical_govid, result$category, sep = "\r")
|
||||
has_direct <- key %in% unique(key[!is_ig])
|
||||
candidate <- is_ig & !has_direct
|
||||
|
||||
flag <- rep(FALSE, n)
|
||||
notes <- empty_notes
|
||||
if (!any(candidate)) return(list(flag = flag, notes = notes))
|
||||
|
||||
idx <- which(candidate)
|
||||
rows <- unique(result[idx, c("year", "canonical_govid", "category")])
|
||||
covering <- .covering_recipes(con, rows)
|
||||
cov_key <- paste(covering$year, covering$canonical_govid, covering$category,
|
||||
sep = "\r")
|
||||
|
||||
for (i in idx) {
|
||||
k <- paste(result$year[i], result$canonical_govid[i], result$category[i],
|
||||
sep = "\r")
|
||||
m <- match(k, cov_key)
|
||||
if (is.na(m)) next
|
||||
ids <- covering$recipe_ids[[m]]
|
||||
if (length(ids) == 0L) next
|
||||
flag[i] <- TRUE
|
||||
notes[i] <- sprintf(
|
||||
"Direct component is unavailable through this basis for this year; recover it via recipe = '%s' (see cog_recipes()).",
|
||||
paste(sort(unique(ids)), collapse = "', '")
|
||||
)
|
||||
}
|
||||
list(flag = flag, notes = notes)
|
||||
}
|
||||
|
||||
#' For each (year, canonical_govid, category) triple potentially affected by
|
||||
#' a suppressed Direct leg, find the harmonization recipe(s) that (a) cover
|
||||
#' this `category` (share a component item_code via `summary_categories`,
|
||||
#' excluding any recipe that is itself entirely intergovernmental M/L -- the
|
||||
#' same exclusion `.build_suggestions()` applies, see I2) and (b) actually
|
||||
#' produce a `long` row for this exact (canonical_govid, year) via the same
|
||||
#' generic join `.run_recipe()` uses (component year_min/year_max +
|
||||
#' gov_type_scope, no is_aggregate filter -- a recipe's whole point is to
|
||||
#' recover data that's aggregate-only). Adds a list-column `recipe_ids`
|
||||
#' (possibly length-0) to `rows`.
|
||||
#' @noRd
|
||||
.covering_recipes <- function(con, rows) {
|
||||
rows$recipe_ids <- vector("list", nrow(rows))
|
||||
cats <- unique(rows$category[!is.na(rows$category)])
|
||||
if (length(cats) == 0L) return(rows)
|
||||
|
||||
cand <- DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT DISTINCT sc.category, r.recipe_id
|
||||
FROM harmonization_recipes r
|
||||
JOIN summary_categories sc ON sc.item_code = r.component_code
|
||||
WHERE sc.category IN (%s)
|
||||
AND r.recipe_id NOT IN (
|
||||
SELECT DISTINCT recipe_id FROM harmonization_recipes
|
||||
WHERE LEFT(component_code, 1) IN ('M', 'L')
|
||||
)",
|
||||
.sql_lit_chr(cats)
|
||||
))
|
||||
if (nrow(cand) == 0L) return(rows)
|
||||
|
||||
recipe_ids_all <- unique(cand$recipe_id)
|
||||
govids <- unique(rows$canonical_govid)
|
||||
years <- unique(rows$year)
|
||||
covered <- DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT DISTINCT r.recipe_id, l.canonical_govid, l.year
|
||||
FROM long l
|
||||
JOIN harmonization_recipes r
|
||||
ON l.item_code = r.component_code
|
||||
AND l.year BETWEEN r.year_min AND r.year_max
|
||||
AND (r.gov_type_scope = 'all'
|
||||
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
||||
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||
WHERE r.recipe_id IN (%s)
|
||||
AND l.canonical_govid IN (%s)
|
||||
AND l.year IN (%s)",
|
||||
.sql_lit_chr(recipe_ids_all), .sql_lit_chr(govids), paste(years, collapse = ",")
|
||||
))
|
||||
|
||||
for (i in seq_len(nrow(rows))) {
|
||||
cat_i <- rows$category[i]
|
||||
if (is.na(cat_i)) next
|
||||
cat_recipe_ids <- cand$recipe_id[cand$category == cat_i]
|
||||
if (length(cat_recipe_ids) == 0L) next
|
||||
sub <- covered[covered$canonical_govid == rows$canonical_govid[i] &
|
||||
covered$year == rows$year[i] &
|
||||
covered$recipe_id %in% cat_recipe_ids, ]
|
||||
rows$recipe_ids[[i]] <- sort(unique(sub$recipe_id))
|
||||
}
|
||||
rows
|
||||
}
|
||||
|
||||
#' @noRd
|
||||
.notes_column <- function(result, direct_suppressed_notes = NULL) {
|
||||
n <- nrow(result)
|
||||
if (n == 0L) return(character(0))
|
||||
parts <- vector("list", 2L)
|
||||
parts <- vector("list", 3L)
|
||||
agg <- result[["aggregate_fallback"]]
|
||||
parts[[1]] <- if (!is.null(agg)) {
|
||||
ifelse(agg %in% TRUE,
|
||||
@@ -213,6 +645,11 @@ cog_spending <- function(govid, years, category = NULL,
|
||||
} else {
|
||||
rep(NA_character_, n)
|
||||
}
|
||||
parts[[3]] <- if (!is.null(direct_suppressed_notes)) {
|
||||
direct_suppressed_notes
|
||||
} else {
|
||||
rep(NA_character_, n)
|
||||
}
|
||||
out <- character(n)
|
||||
for (i in seq_len(n)) {
|
||||
pieces <- vapply(parts, `[[`, character(1), i)
|
||||
|
||||
+251
@@ -0,0 +1,251 @@
|
||||
# R/suggestions.R
|
||||
# Recipe-component-driven signposting: when a basis = "harmonized" query for
|
||||
# a category comes back with a coverage gap in some requested years (the
|
||||
# result has no rows at all in that year) that a harmonization recipe would
|
||||
# actually fill for this government, surface that recipe as a suggestion.
|
||||
#
|
||||
# This is deliberately keyed off the recipe catalog's component codes, not
|
||||
# off harmonization_map rows: no live map row carries a non-blank
|
||||
# suggested_recipe_id (the corpus's wide era exposes split families like
|
||||
# corrections functions 04+05 ONLY as aggregate rows, which basis =
|
||||
# "harmonized" excludes by construction -- there's no NA ruling to hang a
|
||||
# suggestion off of, just a leaf-code absence a recipe happens to fill).
|
||||
# See docs/phase_r_harmonization_review.md § 0.3.
|
||||
#
|
||||
# Scope is deliberately narrow: signposting only runs when the caller
|
||||
# supplied a `category` (an un-scoped, all-categories query has no single
|
||||
# coverage question to answer) and only flags a recipe when the ACTUAL
|
||||
# result has zero rows in a requested year AND the candidate recipe's own
|
||||
# generic join (same join .run_recipe() uses, including its wide-era
|
||||
# aggregate rows) produces at least one row for this government in that
|
||||
# year. Checking presence per-government (not corpus-wide) avoids false
|
||||
# positives from ordinary reporting variance -- most governments don't use
|
||||
# every sibling code in a multi-code category every year, and that is not
|
||||
# a format-boundary gap worth signposting.
|
||||
#
|
||||
# C1(a): for expenditure_concept = "total" callers, `result` here must
|
||||
# already be the Direct-leg subset (the caller filters out
|
||||
# spend_subtype == "intergovernmental" rows before calling in). A gap year
|
||||
# is "the requested year has no Direct rows", never "no rows at all" --
|
||||
# an IG row surviving on a legacy aggregate that Direct excludes must not
|
||||
# read as coverage and cancel the very suggestion that would recover it.
|
||||
|
||||
#' Build the `prov$suggestions` list for a (non-recipe) basis = "harmonized"
|
||||
#' verb call: recipes whose generic join would fill a real gap in `result`.
|
||||
#'
|
||||
#' @param con Active DuckDB connection.
|
||||
#' @param govid Character vector of canonical_govid values (the verb's raw
|
||||
#' `govid`).
|
||||
#' @param years Integer vector of requested years.
|
||||
#' @param category `category` argument as passed to the verb (character
|
||||
#' vector or `NULL`; suggestions are only computed when non-NULL).
|
||||
#' @param result The verb's already-computed result tibble (post basis
|
||||
#' query, pre per_capita/adjust_to_year), pre-filtered to the Direct leg
|
||||
#' only when the caller's `expenditure_concept = "total"` (see C1(a)).
|
||||
#' @param basis The *resolved* basis (`"harmonized"` or `"raw"`).
|
||||
#' @param flow_prefixes The calling verb's own flow-type prefixes (e.g.
|
||||
#' `c("E", "F", "G")` for `cog_spending()`, `c("T", "A", "U", "B", "C",
|
||||
#' "D")` for `cog_revenue()` -- see `.verb_spendrev()`). Passed through to
|
||||
#' `.attach_ig_counterparts()` to keep the intergovernmental-counterpart
|
||||
#' lookup scoped to the calling verb's own flow family.
|
||||
#' @return List of `list(recipe_id, label, available_years, hint,
|
||||
#' ig_recipe_id)`, possibly empty.
|
||||
#' @noRd
|
||||
.build_suggestions <- function(con, govid, years, category, result, basis,
|
||||
flow_prefixes) {
|
||||
if (!identical(basis, "harmonized") || is.null(category)) return(list())
|
||||
|
||||
# Exclude any recipe that is ITSELF an intergovernmental (M/L) recipe --
|
||||
# i.e. every one of its own component codes is M/L-prefixed. Without this,
|
||||
# a category whose summary_categories rows span both a Direct family
|
||||
# (e.g. E04/E05, "Corrections") and its M/L counterpart (M04/M05, same
|
||||
# category since Task 1) makes the M/L recipe itself (e.g.
|
||||
# `corrections_ig_local_combined`) a raw top-level candidate for a plain
|
||||
# (Direct) cog_spending() call -- following that hint would silently
|
||||
# return intergovernmental dollars under `expenditure_concept = "direct"`
|
||||
# provenance. This is a stronger, unconditional exclusion than the
|
||||
# flow-prefix gate below/in `.attach_ig_counterparts()`: an M/L recipe
|
||||
# should never be suggested as a coverage-gap filler for EITHER verb, not
|
||||
# just kept from being named as the *counterpart* of another suggestion.
|
||||
candidates <- DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT DISTINCT recipe_id FROM harmonization_recipes
|
||||
WHERE component_code IN (
|
||||
SELECT DISTINCT item_code FROM summary_categories WHERE category IN (%s)
|
||||
)
|
||||
AND recipe_id NOT IN (
|
||||
SELECT DISTINCT recipe_id FROM harmonization_recipes
|
||||
WHERE LEFT(component_code, 1) IN ('M', 'L')
|
||||
)",
|
||||
.sql_lit_chr(category)
|
||||
))$recipe_id
|
||||
if (length(candidates) == 0L) return(list())
|
||||
|
||||
result_years <- if (is.null(result) || nrow(result) == 0L) {
|
||||
integer(0)
|
||||
} else {
|
||||
unique(as.integer(result$year))
|
||||
}
|
||||
gap_years <- setdiff(as.integer(years), result_years)
|
||||
if (length(gap_years) == 0L) return(list())
|
||||
|
||||
meta <- tibble::as_tibble(DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT recipe_id, any_value(label) AS label,
|
||||
MIN(year_min) AS year_min, MAX(year_max) AS year_max
|
||||
FROM harmonization_recipes
|
||||
WHERE recipe_id IN (%s)
|
||||
GROUP BY recipe_id",
|
||||
.sql_lit_chr(candidates)
|
||||
)))
|
||||
|
||||
# Which (recipe_id, year) pairs the recipe's own generic join actually
|
||||
# covers for this government, restricted to the gap years -- the same
|
||||
# join .run_recipe() uses (component year_min/year_max + gov_type_scope,
|
||||
# no is_aggregate filter), just checking existence instead of summing.
|
||||
covered <- DBI::dbGetQuery(con, sprintf(
|
||||
"SELECT DISTINCT r.recipe_id, l.year
|
||||
FROM long l
|
||||
JOIN harmonization_recipes r
|
||||
ON l.item_code = r.component_code
|
||||
AND l.year BETWEEN r.year_min AND r.year_max
|
||||
AND (r.gov_type_scope = 'all'
|
||||
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
||||
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||
WHERE r.recipe_id IN (%s)
|
||||
AND l.canonical_govid IN (%s)
|
||||
AND l.year IN (%s)",
|
||||
.sql_lit_chr(candidates), .sql_lit_chr(govid),
|
||||
paste(gap_years, collapse = ",")
|
||||
))
|
||||
|
||||
suggestions <- list()
|
||||
for (rid in candidates) {
|
||||
if (!rid %in% covered$recipe_id) next
|
||||
m <- meta[meta$recipe_id == rid, ]
|
||||
suggestions[[length(suggestions) + 1L]] <- list(
|
||||
recipe_id = rid,
|
||||
label = m$label[[1]],
|
||||
available_years = c(as.integer(m$year_min), as.integer(m$year_max)),
|
||||
hint = sprintf("re-run with recipe = '%s'", rid)
|
||||
)
|
||||
}
|
||||
.attach_ig_counterparts(con, suggestions, flow_prefixes)
|
||||
}
|
||||
|
||||
#' Attach `ig_recipe_id` to each suggestion: the intergovernmental-expenditure
|
||||
#' recipe (an M-to-local or L-to-state recipe) whose component codes cover
|
||||
#' exactly the same set of function suffixes as the firing recipe's own
|
||||
#' components, e.g. `corrections_combined`'s {E04, E05} -> suffixes {"04",
|
||||
#' "05"} matches `corrections_ig_local_combined`'s {M04, M05} -> the same
|
||||
#' {"04", "05"}. `NULL` when no such recipe exists, which also covers the
|
||||
#' case where the firing recipe already IS the IG recipe (self-matches are
|
||||
#' excluded, so an IG recipe never names itself as its own counterpart).
|
||||
#'
|
||||
#' Matching is deliberately an exact set match, not "any suffix in common":
|
||||
#' the two-digit suffix only means the same "function" across recipes that
|
||||
#' share the underlying Census functional-classification scheme (E/F/G/L/M
|
||||
#' all use "04"/"05" for corrections). M/L "combined other" codes (47/89/
|
||||
#' 91-94) reuse digits for an unrelated catch-all construct, so e.g.
|
||||
#' `general_gov_e89_wide`'s {E85, E89} -> {"85", "89"} must NOT match
|
||||
#' `ige_local_m89_wide`'s {"89", "91", "92", "93"} on the shared "89" alone.
|
||||
#' Checked by hand against the full harmonization_recipes catalog: only the
|
||||
#' corrections family (E/F/G/M, suffixes 04/05) has an exact-set match in
|
||||
#' this corpus.
|
||||
#'
|
||||
#' Exact-set suffix matching is NOT enough on its own, though: the same
|
||||
#' reused-digit problem exists ACROSS the revenue-side IG families too.
|
||||
#' `ig_local_d47_wide` (D47/D94, suffixes {"47","94"}) is an exact-set match
|
||||
#' for `ige_local_m47_wide` (M47/M94, same suffixes) even though one is
|
||||
#' intergovernmental REVENUE received from local governments and the other is
|
||||
#' intergovernmental EXPENDITURE paid to local governments -- unrelated flows
|
||||
#' that happen to reuse "47"/"94" for their own "transit/utilities" and
|
||||
#' "other/combined" catch-alls. `ig_federal_b47_wide`, `ig_state_c47_wide`,
|
||||
#' and their `*_89` siblings all collide the same way. None of this is
|
||||
#' reachable via `cog_revenue()` in the bundled fixture today (its B/C/D
|
||||
#' recipes never happen to have a covered gap year for any fixture govid),
|
||||
#' but it IS reachable via a mis-scoped `cog_spending()` call on a
|
||||
#' revenue-only category, e.g. `cog_spending(gov, category = "IG Federal")`
|
||||
#' fires `ig_federal_b47_wide`/`ig_federal_b89_wide` for real in the fixture
|
||||
#' -- so this is a live, not merely theoretical, gap.
|
||||
#'
|
||||
#' Two flow-family checks close this, both required (see
|
||||
#' `tests/testthat/test-expenditure-concept.R`, "revenue-flavored ... never
|
||||
#' receives an M/L counterpart" tests, for the pairwise verification):
|
||||
#' 1. `own_prefix %in% flow_prefixes`: the firing recipe's own component
|
||||
#' codes must belong to the calling verb's own flow family (the same
|
||||
#' `flow_prefixes` `.build_harmonization_block()` uses, see
|
||||
#' `R/basis.R`). This blocks a recipe surfaced through a mis-scoped
|
||||
#' category from ever reaching the M/L search, e.g. `cog_spending()`'s
|
||||
#' flow_prefixes are `c("E","F","G")`, which `ig_federal_b47_wide`'s own
|
||||
#' `"B"` is not part of.
|
||||
#' 2. `own_prefix %in% c("E","F","G")`: M/L only ever pairs with the
|
||||
#' DIRECT-expenditure family, never with revenue (`cog_revenue()`'s
|
||||
#' flow_prefixes already fold B/C/D in as ordinary revenue -- there is
|
||||
#' no separate "Total" bolt-on for revenue the way `expenditure_concept`
|
||||
#' adds one for spending) and never with ANOTHER M/L recipe (without
|
||||
#' this check, `ige_local_m47_wide` would wrongly match sibling
|
||||
#' `ige_state_l47_wide` on their shared {"47","94"} suffix set).
|
||||
#' Condition 1 alone does not catch this: under `cog_revenue()`,
|
||||
#' `ig_federal_b47_wide`'s own `"B"` IS inside revenue's own
|
||||
#' `flow_prefixes`, so only this second, family-specific check blocks
|
||||
#' the search.
|
||||
#' @noRd
|
||||
.attach_ig_counterparts <- function(con, suggestions, flow_prefixes) {
|
||||
if (length(suggestions) == 0L) return(suggestions)
|
||||
|
||||
comp <- DBI::dbGetQuery(con,
|
||||
"SELECT recipe_id, component_code FROM harmonization_recipes")
|
||||
comp$prefix <- substr(comp$component_code, 1L, 1L)
|
||||
comp$suffix <- substr(comp$component_code, 2L, nchar(comp$component_code))
|
||||
suffix_sets <- lapply(split(comp$suffix, comp$recipe_id), function(x) sort(unique(x)))
|
||||
prefix_sets <- lapply(split(comp$prefix, comp$recipe_id), function(x) sort(unique(x)))
|
||||
|
||||
ig_recipe_ids <- unique(comp$recipe_id[comp$prefix %in% c("M", "L")])
|
||||
|
||||
find_counterpart <- function(rid) {
|
||||
own_prefix <- prefix_sets[[rid]]
|
||||
own_suffix <- suffix_sets[[rid]]
|
||||
if (is.null(own_prefix) || is.null(own_suffix)) return(NULL)
|
||||
if (!all(own_prefix %in% flow_prefixes)) return(NULL)
|
||||
if (!all(own_prefix %in% c("E", "F", "G"))) return(NULL)
|
||||
for (cand in ig_recipe_ids) {
|
||||
if (identical(cand, rid)) next
|
||||
if (setequal(suffix_sets[[cand]], own_suffix)) return(cand)
|
||||
}
|
||||
NULL
|
||||
}
|
||||
|
||||
lapply(suggestions, function(s) {
|
||||
# `s$ig_recipe_id <- NULL` would DELETE the element rather than set it
|
||||
# (standard R list-assignment gotcha), leaving no-match entries missing
|
||||
# the key entirely instead of carrying it as NULL. Single-bracket
|
||||
# assignment with a wrapped list preserves a NULL-valued element so the
|
||||
# field is always present, per the brief's "NULL when there is none".
|
||||
s["ig_recipe_id"] <- list(find_counterpart(s$recipe_id))
|
||||
s
|
||||
})
|
||||
}
|
||||
|
||||
#' Emit the single cli::cli_inform() message summarizing all suggestions
|
||||
#' for a verb call (the brief's "one message", not one per suggestion).
|
||||
#' Bullet text is pre-formatted plain text (no cli/glue `{}` markup) since
|
||||
#' recipe ids/labels are untrusted-ish data values, not literal call-site
|
||||
#' expressions. When a suggestion has an `ig_recipe_id`, one indented
|
||||
#' continuation line is appended naming the intergovernmental counterpart
|
||||
#' recipe (embedded `\n` renders as a hanging-indent continuation of the
|
||||
#' same bullet under cli, not a new bullet).
|
||||
#' @noRd
|
||||
.inform_suggestions <- function(suggestions) {
|
||||
bullets <- vapply(suggestions, function(s) {
|
||||
bullet <- sprintf("%s (%d-%d): %s", s$recipe_id,
|
||||
s$available_years[1], s$available_years[2], s$hint)
|
||||
if (!is.null(s$ig_recipe_id)) {
|
||||
bullet <- paste0(bullet, sprintf(
|
||||
"\n intergovernmental counterpart: recipe = '%s'", s$ig_recipe_id))
|
||||
}
|
||||
bullet
|
||||
}, character(1))
|
||||
cli::cli_inform(c(
|
||||
i = "Coverage gap detected for the requested years; a harmonization recipe may fill it:",
|
||||
stats::setNames(bullets, rep("*", length(bullets)))
|
||||
))
|
||||
}
|
||||
@@ -1,11 +1,45 @@
|
||||
# R/views.R
|
||||
|
||||
# SQL files that cannot be registered unconditionally against a v4 corpus,
|
||||
# for one of two distinct reasons -- both fail at CREATE VIEW time (DuckDB
|
||||
# resolves a view's source schema eagerly, even though it defers execution),
|
||||
# so a v4 corpus can't tolerate either unconditionally:
|
||||
#
|
||||
# (a) Missing FILE. 33-/34-/35- read_parquet() a v5-only parquet table
|
||||
# (harmonization_map.parquet, harmonization_recipes.parquet,
|
||||
# series_breaks.parquet) that doesn't exist at all on a v4 corpus --
|
||||
# "IO Error: No files found".
|
||||
#
|
||||
# (b) Missing COLUMN. 22-/23-/25- reference `long.harmonized_code`, a
|
||||
# column that does not exist on a v4 corpus's `long` table (harmonized
|
||||
# space was introduced in schema v5) -- "Binder Error: Referenced
|
||||
# column harmonized_code not found". 42-/43-/45- are on this list only
|
||||
# because they SELECT s.* FROM the (a)/(b) views above, so they'd fail
|
||||
# to resolve their own source view if it weren't already skipped.
|
||||
#
|
||||
# Registration is therefore gated on manifest$schema_version >= 5 for all of
|
||||
# them; verb-level *usage* of the resulting views is separately gated by
|
||||
# .resolve_basis() / .require_schema_v5().
|
||||
.harmonization_view_files <- c(
|
||||
"22-spending_long_harmonized.sql",
|
||||
"23-revenue_long_harmonized.sql",
|
||||
"25-ig_long_harmonized.sql",
|
||||
"33-harmonization_map.sql",
|
||||
"34-harmonization_recipes.sql",
|
||||
"35-series_breaks_pq.sql",
|
||||
"42-spending_annotated_harmonized.sql",
|
||||
"43-revenue_annotated_harmonized.sql",
|
||||
"45-ig_annotated_harmonized.sql"
|
||||
)
|
||||
|
||||
#' Register DuckDB views from inst/sql/ SQL files
|
||||
#' @noRd
|
||||
.register_views <- function(con, url, manifest) {
|
||||
sql_dir <- system.file("sql", package = "uscogdata")
|
||||
files <- list.files(sql_dir, pattern = "\\.sql$", full.names = TRUE)
|
||||
files <- sort(list.files(sql_dir, pattern = "\\.sql$", full.names = TRUE))
|
||||
schema_version <- suppressWarnings(as.integer(manifest$schema_version %||% 0L))
|
||||
for (f in files) {
|
||||
if (basename(f) %in% .harmonization_view_files && schema_version < 5L) next
|
||||
sql <- paste(readLines(f, warn = FALSE), collapse = "\n")
|
||||
sql <- gsub("\\{url\\}", url, sql, fixed = FALSE)
|
||||
DBI::dbExecute(con, sql)
|
||||
|
||||
@@ -25,14 +25,32 @@ package implements.
|
||||
- `USCOGDATA_CACHE_DIR` — optional override for the manifest cache directory
|
||||
- `USCOGDATA_MANIFEST_TTL_SECS` — optional manifest re-fetch TTL (default 3600)
|
||||
|
||||
## Direct vs Total spending
|
||||
|
||||
`cog_spending(..., expenditure_concept = c("direct", "total"))` controls
|
||||
whose spending a result counts. `"direct"` (the default) is a government's
|
||||
own current operations, capital outlay, and other direct spending. `"total"`
|
||||
additionally adds in the intergovernmental legs — money it hands to other
|
||||
governments to spend on its behalf — which is meaningful for describing one
|
||||
government's own budget over time, but double-counts when summed across
|
||||
governments (a state's payment to a county is the same dollar the county
|
||||
reports as its own direct spending).
|
||||
|
||||
**Rule of thumb: any figure that spans more than one government uses
|
||||
`direct`.** `cog_geographic_rollup()` and `cog_peer_compare()` enforce this
|
||||
by refusing `expenditure_concept = "total"`. See
|
||||
`vignette("total-spending", package = "uscogdata")` for the full
|
||||
explanation with worked examples.
|
||||
|
||||
## Developer notes
|
||||
|
||||
### Testing
|
||||
|
||||
The package ships a bundled fixture corpus at `inst/extdata/fixture_corpus/` —
|
||||
a 3.6 MB two-year slice (2019 + 2020) of the full corpus covering all 50
|
||||
states. `tests/testthat/setup.R` automatically points `USCOGDATA_URL` at this
|
||||
fixture, so the full test suite runs offline with no network dependency:
|
||||
a 15 MB four-year slice (2011, 2012, 2019, 2020) of the full corpus covering
|
||||
all 50 states. `tests/testthat/setup.R` automatically points `USCOGDATA_URL`
|
||||
at this fixture, so the full test suite runs offline with no network
|
||||
dependency:
|
||||
|
||||
```r
|
||||
devtools::test() # uses bundled fixture, no credentials required
|
||||
|
||||
@@ -3,12 +3,20 @@
|
||||
# Regenerate inst/extdata/fixture_corpus/ from a cog_pipeline publish tree.
|
||||
#
|
||||
# What this does:
|
||||
# 1. Copies the year=2019 and year=2020 long partitions as-is (byte-for-
|
||||
# byte) from <publish_cache>/data/long/ into the fixture.
|
||||
# 1. Copies each requested year's long partition as-is (byte-for-byte)
|
||||
# from <publish_cache>/data/long/ into the fixture. Default years are
|
||||
# c(2011L, 2012L, 2019L, 2020L): 2011/2012 straddle the wide-aggregate
|
||||
# -> modern-leaf format boundary (the harmonization/recipe seam), and
|
||||
# 2019/2020 are the pre-existing per-capita/CPI regression anchors.
|
||||
# Each partition is a full year (all states/govs) as published, so
|
||||
# Broward County FL and every other previously-pinned government stay
|
||||
# covered without any per-gov slicing logic.
|
||||
# 2. Copies the full canonical_fips_xwalk.parquet, canonical_alias.parquet,
|
||||
# and summary_categories.parquet metadata tables as-is (these are small
|
||||
# cross-vintage registries, not partitioned by year, so the fixture
|
||||
# ships the complete tables rather than a year-scoped subset).
|
||||
# summary_categories.parquet, harmonization_map.parquet,
|
||||
# harmonization_recipes.parquet, and series_breaks.parquet metadata
|
||||
# tables as-is (these are small cross-vintage registries, not
|
||||
# partitioned by year, so the fixture ships the complete tables rather
|
||||
# than a year-scoped subset).
|
||||
# 3. Resyncs the four reference docs (data_dictionary.md,
|
||||
# reader-specification.md, README.md, series_breaks.md) from the
|
||||
# publish tree's docs/.
|
||||
@@ -35,7 +43,7 @@ regenerate_fixture_corpus <- function(
|
||||
"..", "cog_pipeline", "_targets", "publish_cache"
|
||||
),
|
||||
fixture_dir = file.path("inst", "extdata", "fixture_corpus"),
|
||||
fixture_years = c(2019L, 2020L)) {
|
||||
fixture_years = c(2011L, 2012L, 2019L, 2020L)) {
|
||||
stopifnot(
|
||||
requireNamespace("digest", quietly = TRUE),
|
||||
requireNamespace("jsonlite", quietly = TRUE),
|
||||
@@ -92,14 +100,18 @@ regenerate_fixture_corpus <- function(
|
||||
invisible(NULL)
|
||||
}
|
||||
|
||||
# Copy the full (not year-scoped) canonical_fips_xwalk, canonical_alias, and
|
||||
# summary_categories parquet tables.
|
||||
# Copy the full (not year-scoped) canonical_fips_xwalk, canonical_alias,
|
||||
# summary_categories, and (schema v5+) harmonization_map/
|
||||
# harmonization_recipes/series_breaks parquet tables.
|
||||
#' @noRd
|
||||
.copy_metadata_parquets <- function(publish_cache_dir, fixture_dir) {
|
||||
files <- c(
|
||||
"canonical_fips_xwalk.parquet",
|
||||
"canonical_alias.parquet",
|
||||
"summary_categories.parquet"
|
||||
"summary_categories.parquet",
|
||||
"harmonization_map.parquet",
|
||||
"harmonization_recipes.parquet",
|
||||
"series_breaks.parquet"
|
||||
)
|
||||
for (f in files) {
|
||||
src <- file.path(publish_cache_dir, "data", f)
|
||||
@@ -170,7 +182,10 @@ regenerate_fixture_corpus <- function(
|
||||
metadata_files <- c(
|
||||
"canonical_alias.parquet",
|
||||
"canonical_fips_xwalk.parquet",
|
||||
"summary_categories.parquet"
|
||||
"summary_categories.parquet",
|
||||
"harmonization_map.parquet",
|
||||
"harmonization_recipes.parquet",
|
||||
"series_breaks.parquet"
|
||||
)
|
||||
metadata <- lapply(metadata_files, function(f) {
|
||||
rel <- file.path("data", f)
|
||||
@@ -187,11 +202,15 @@ regenerate_fixture_corpus <- function(
|
||||
built_at = format(Sys.time(), "%Y-%m-%dT%H:%M:%SZ", tz = "UTC"),
|
||||
pipeline_commit = source_manifest$pipeline_commit,
|
||||
fixture_note = paste(
|
||||
"Two-year (2019-2020) fixture for uscogdata tests. Full corpus",
|
||||
"available via USCOGDATA_URL. Regenerated for Phase P",
|
||||
"(schema_version 4, uniformly 12-char canonical_govid) with the full",
|
||||
"canonical_fips_xwalk master and the new canonical_alias lookup",
|
||||
"table via data-raw/regenerate_fixture_corpus.R."
|
||||
"Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full",
|
||||
"corpus available via USCOGDATA_URL. Regenerated for Phase R2",
|
||||
"(schema_version 5, harmonization_map/harmonization_recipes/",
|
||||
"series_breaks parquet tables added). 2011/2012 straddle the",
|
||||
"wide-aggregate -> modern-leaf format boundary exercised by basis=",
|
||||
"\"harmonized\" and recipe= queries; 2019/2020 retain the prior",
|
||||
"per-capita/CPI regression anchors. Full canonical_fips_xwalk master",
|
||||
"and canonical_alias lookup table included via",
|
||||
"data-raw/regenerate_fixture_corpus.R."
|
||||
),
|
||||
data_vintage = source_manifest$data_vintage,
|
||||
scope = source_manifest$scope,
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+57
-15
@@ -1,11 +1,24 @@
|
||||
{
|
||||
"schema_version": 4,
|
||||
"built_at": "2026-07-13T23:35:09Z",
|
||||
"pipeline_commit": "a082b26",
|
||||
"fixture_note": "Two-year (2019-2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated for Phase P (schema_version 4, uniformly 12-char canonical_govid) with the full canonical_fips_xwalk master and the new canonical_alias lookup table via data-raw/regenerate_fixture_corpus.R.",
|
||||
"schema_version": 6,
|
||||
"built_at": "2026-07-27T13:04:05Z",
|
||||
"pipeline_commit": "6098baf",
|
||||
"fixture_note": "Four-year (2011, 2012, 2019, 2020) fixture for uscogdata tests. Full corpus available via USCOGDATA_URL. Regenerated for Phase R2 (schema_version 5, harmonization_map/harmonization_recipes/ series_breaks parquet tables added). 2011/2012 straddle the wide-aggregate -> modern-leaf format boundary exercised by basis= \"harmonized\" and recipe= queries; 2019/2020 retain the prior per-capita/CPI regression anchors. Full canonical_fips_xwalk master and canonical_alias lookup table included via data-raw/regenerate_fixture_corpus.R.",
|
||||
"data_vintage": {
|
||||
"census_source_downloaded": "unknown",
|
||||
"cpi_vintage": "FRED CPIAUCSL",
|
||||
"source_vintages": {
|
||||
"2012": "10162019",
|
||||
"2013": "10162019",
|
||||
"2014": "10162019",
|
||||
"2015": "10162019",
|
||||
"2016": "10162019",
|
||||
"2017": "06102021",
|
||||
"2018": "06102021",
|
||||
"2019": "06102021",
|
||||
"2020": "06122023",
|
||||
"2021": "06122023",
|
||||
"2022": "06052025",
|
||||
"2023": "06052025"
|
||||
},
|
||||
"registry_rows": 148,
|
||||
"acs_vintage": "ACS 2018-2022 5-year"
|
||||
},
|
||||
"scope": {
|
||||
@@ -14,42 +27,71 @@
|
||||
"scope_note": "v0.1 covers state, county, city/municipality, and township governments. Special districts (type 4) and school districts (type 5) are excluded pending validation in a future cycle."
|
||||
},
|
||||
"schema": {
|
||||
"long_column_count": 24,
|
||||
"long_columns": ["fips_state", "type", "fips_county", "govid", "gov_blank", "gov_name", "county_name", "fips_state_code", "fips_county_code", "fips_place_code", "population", "popyear", "enrollment", "enrollyear", "function_code", "sch_level_code", "fiscal_year_end", "srvy_year", "item_code", "amt", "srv_data", "impute_flag", "is_aggregate", "canonical_govid"],
|
||||
"long_column_count": 28,
|
||||
"long_columns": ["fips_state", "type", "fips_county", "govid", "gov_blank", "gov_name", "county_name", "fips_state_asof", "fips_county_asof", "cog_legacy_state", "cog_legacy_county", "fips_place_code", "population", "popyear", "enrollment", "enrollyear", "function_code", "sch_level_code", "fiscal_year_end", "srvy_year", "item_code", "amt", "srv_data", "impute_flag", "is_aggregate", "canonical_govid", "harmonized_code", "survey_weight"],
|
||||
"data_dictionary": "docs/data_dictionary.md"
|
||||
},
|
||||
"files": {
|
||||
"long_partitions": [
|
||||
{
|
||||
"year": 2011,
|
||||
"path": "data/long/year=2011/part-0.parquet",
|
||||
"sha256": "84302ab364dc9fc3b3fbbc3c3f8b826e3508b4d73ff7c42d094d3863cd1e37b5",
|
||||
"row_count": 2864212,
|
||||
"size_bytes": 3845911
|
||||
},
|
||||
{
|
||||
"year": 2012,
|
||||
"path": "data/long/year=2012/part-0.parquet",
|
||||
"sha256": "b82ac82d5e35f844b26c887445601f3748438c52c998ba4e403b025941a6f170",
|
||||
"row_count": 1163338,
|
||||
"size_bytes": 5929917
|
||||
},
|
||||
{
|
||||
"year": 2019,
|
||||
"path": "data/long/year=2019/part-0.parquet",
|
||||
"sha256": "c0a2bf0758af129d5dfddb6ff6665cc435ddee87fd6879e788fb56ed53ab22b8",
|
||||
"sha256": "5cbd4726dcc7d0dab5c2a05a64702e979533ae119ed0587073cd31c089e0d737",
|
||||
"row_count": 318139,
|
||||
"size_bytes": 1441404
|
||||
"size_bytes": 1719548
|
||||
},
|
||||
{
|
||||
"year": 2020,
|
||||
"path": "data/long/year=2020/part-0.parquet",
|
||||
"sha256": "92570b9d55ec3425d034db37838f91c3b8359d0454d3d98730a6016b62e4bb48",
|
||||
"sha256": "ee548fec80bf1beda844fe03916ac145f10dd34c45968407cc330ec260935f00",
|
||||
"row_count": 317500,
|
||||
"size_bytes": 1444011
|
||||
"size_bytes": 1722918
|
||||
}
|
||||
],
|
||||
"metadata": [
|
||||
{
|
||||
"path": "data/canonical_alias.parquet",
|
||||
"sha256": "feb8d01a640fb16c9a4b4ad66726b50b8fe8a1ce2a771bed8c1190fec51d5c8d",
|
||||
"sha256": "3f617051c23a99bea322889857f7106df0c92954564afeec181df7083ee6698e",
|
||||
"description": "canonical_alias.parquet"
|
||||
},
|
||||
{
|
||||
"path": "data/canonical_fips_xwalk.parquet",
|
||||
"sha256": "1ae47981531c7389f69eff3f7656045428564bfbe8032200eb9c039d32739a7e",
|
||||
"sha256": "f98742f941269dacf8f7de5c273aa4dd4e75017a5bb70c054da35852a95a8d46",
|
||||
"description": "canonical_fips_xwalk.parquet"
|
||||
},
|
||||
{
|
||||
"path": "data/summary_categories.parquet",
|
||||
"sha256": "dd59e7f58a022679ad43511c8c8e938b8dd4be81196bbeeee21e67bdcca2295b",
|
||||
"sha256": "0985b607f3f35a8dff62c0561261ab6922423b81d11c07b03bcb3e3461f85e33",
|
||||
"description": "summary_categories.parquet"
|
||||
},
|
||||
{
|
||||
"path": "data/harmonization_map.parquet",
|
||||
"sha256": "4cf32d0f817079ba4f28dc0ce65450d3247ebbf08d94c0c26c0d02af597bf812",
|
||||
"description": "harmonization_map.parquet"
|
||||
},
|
||||
{
|
||||
"path": "data/harmonization_recipes.parquet",
|
||||
"sha256": "1133e9a0b02f8f34f5f936e55c5ecd596bb8a55d8425dcce76767f0f3203581c",
|
||||
"description": "harmonization_recipes.parquet"
|
||||
},
|
||||
{
|
||||
"path": "data/series_breaks.parquet",
|
||||
"sha256": "b0b6794b6887a4f300079adfa10029c2a77109faa4952fbff1c5a270793cc02b",
|
||||
"description": "series_breaks.parquet"
|
||||
}
|
||||
]
|
||||
},
|
||||
|
||||
@@ -10,6 +10,24 @@
|
||||
"target": { "type": "object" },
|
||||
"years": { "type": "array", "items": { "type": "integer" } },
|
||||
"category": { "type": ["string", "array", "null"] },
|
||||
"basis": { "type": ["string", "null"] },
|
||||
"basis_note": { "type": ["string", "null"] },
|
||||
"expenditure_concept": {
|
||||
"type": "string",
|
||||
"enum": ["direct", "total"],
|
||||
"description": "Which spending concept produced this result. 'direct' is the government's own E/F/G spending; 'total' adds its intergovernmental payments (M to local governments, L to state governments). Only 'direct' is valid for results combined across governments."
|
||||
},
|
||||
"expenditure_concept_note": {
|
||||
"type": ["string", "null"],
|
||||
"description": "How the intergovernmental leg was assembled; null for 'direct'."
|
||||
},
|
||||
"expenditure_concept_direct_suppressed": {
|
||||
"type": "boolean",
|
||||
"description": "TRUE when expenditure_concept = 'total' and at least one requested (year, category) has intergovernmental rows but NO Direct rows in this corpus (typically a legacy aggregate-only family) -- those result rows report the intergovernmental leg alone, not Direct + IG. Always FALSE for expenditure_concept = 'direct'. See the affected rows' `notes` for the recovering recipe, if any."
|
||||
},
|
||||
"harmonization": { "type": "object" },
|
||||
"recipe": { "type": ["object", "null"] },
|
||||
"suggestions": { "type": "array" },
|
||||
"scope": { "type": "object" },
|
||||
"codes_summed": { "type": "object" },
|
||||
"aggregate_fallback": { "type": ["object", "null"] },
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
CREATE OR REPLACE VIEW spending_long AS
|
||||
SELECT *
|
||||
FROM long
|
||||
WHERE LEFT(item_code, 1) IN ('E', 'F', 'G', 'K')
|
||||
WHERE LEFT(item_code, 1) IN ('E', 'F', 'G')
|
||||
AND NOT is_aggregate;
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
CREATE OR REPLACE VIEW spending_long_harmonized AS
|
||||
SELECT * REPLACE (harmonized_code AS item_code)
|
||||
FROM long
|
||||
WHERE NOT is_aggregate
|
||||
AND harmonized_code IS NOT NULL
|
||||
AND LEFT(harmonized_code, 1) IN ('E', 'F', 'G');
|
||||
@@ -0,0 +1,6 @@
|
||||
CREATE OR REPLACE VIEW revenue_long_harmonized AS
|
||||
SELECT * REPLACE (harmonized_code AS item_code)
|
||||
FROM long
|
||||
WHERE NOT is_aggregate
|
||||
AND harmonized_code IS NOT NULL
|
||||
AND LEFT(harmonized_code, 1) IN ('T', 'A', 'U', 'B', 'C', 'D');
|
||||
@@ -0,0 +1,18 @@
|
||||
-- Intergovernmental expenditure rows (M = to local govts, L = to state govts).
|
||||
--
|
||||
-- Deliberately does NOT filter `NOT is_aggregate`, unlike spending_long. In the
|
||||
-- wide era (<= FY2011) the IG families M05/M12/M47/M89/L47/L89 are published
|
||||
-- ONLY as aggregate-flagged rows -- filtering them would hide ~70% of legacy IG
|
||||
-- dollars and make Total silently collapse to Direct. This is safe because the
|
||||
-- aggregate codes and their modern leaf components are strictly year-disjoint
|
||||
-- (M47 ends 2011 / M94 starts 2012; M89 is aggregate only <= 2011 and a leaf
|
||||
-- from 2012 alongside M91-93), so no row is ever counted twice. Same argument
|
||||
-- the pipeline's recipe joins use.
|
||||
--
|
||||
-- `L--` IS excluded: it is the IG-to-state FAMILY TOTAL and genuinely rolls up
|
||||
-- the L-NN codes, so including it would double-count.
|
||||
CREATE OR REPLACE VIEW ig_long AS
|
||||
SELECT *
|
||||
FROM long
|
||||
WHERE LEFT(item_code, 1) IN ('M', 'L')
|
||||
AND item_code NOT LIKE '%--';
|
||||
@@ -0,0 +1,15 @@
|
||||
-- Harmonized-basis IG rows. Uses COALESCE(harmonized_code, item_code) rather
|
||||
-- than harmonized_code alone: aggregate rows carry NO harmonized_code by
|
||||
-- construction (harmonized space is leaf-only), so a plain
|
||||
-- `harmonized_code IS NOT NULL` filter would drop every legacy IG aggregate --
|
||||
-- in the bundled fixture corpus (year 2011; 2012+ all carry a harmonized_code)
|
||||
-- that is $379,016,063k across 25,688 M rows and $2,277,458k across 19,266 L
|
||||
-- rows (`SELECT year, LEFT(item_code,1), SUM(amt), COUNT(*) FROM ig_long
|
||||
-- WHERE harmonized_code IS NULL GROUP BY 1, 2`). COALESCE keeps the one real
|
||||
-- IG collapse rule (M38 -> M36, SB012, year-disjoint 1967-2011 vs 2012+)
|
||||
-- while never dropping a row.
|
||||
CREATE OR REPLACE VIEW ig_long_harmonized AS
|
||||
SELECT * REPLACE (COALESCE(harmonized_code, item_code) AS item_code)
|
||||
FROM long
|
||||
WHERE LEFT(item_code, 1) IN ('M', 'L')
|
||||
AND item_code NOT LIKE '%--';
|
||||
@@ -0,0 +1,3 @@
|
||||
CREATE OR REPLACE VIEW harmonization_map AS
|
||||
SELECT *
|
||||
FROM read_parquet('{url}data/harmonization_map.parquet');
|
||||
@@ -0,0 +1,3 @@
|
||||
CREATE OR REPLACE VIEW harmonization_recipes AS
|
||||
SELECT *
|
||||
FROM read_parquet('{url}data/harmonization_recipes.parquet');
|
||||
@@ -0,0 +1,3 @@
|
||||
CREATE OR REPLACE VIEW series_breaks_pq AS
|
||||
SELECT *
|
||||
FROM read_parquet('{url}data/series_breaks.parquet');
|
||||
@@ -0,0 +1,16 @@
|
||||
CREATE OR REPLACE VIEW spending_annotated_harmonized AS
|
||||
SELECT
|
||||
s.*,
|
||||
x.gov_name AS xwalk_gov_name,
|
||||
x.govs_type,
|
||||
x.type_label,
|
||||
x.fips_state AS xwalk_fips_state,
|
||||
x.fips_county AS xwalk_fips_county,
|
||||
x.fips_place,
|
||||
x.population_acs,
|
||||
c.category,
|
||||
c.category_type,
|
||||
c.spend_subtype
|
||||
FROM spending_long_harmonized s
|
||||
LEFT JOIN canonical_fips_xwalk x USING (canonical_govid)
|
||||
LEFT JOIN summary_categories c USING (item_code);
|
||||
@@ -0,0 +1,16 @@
|
||||
CREATE OR REPLACE VIEW revenue_annotated_harmonized AS
|
||||
SELECT
|
||||
s.*,
|
||||
x.gov_name AS xwalk_gov_name,
|
||||
x.govs_type,
|
||||
x.type_label,
|
||||
x.fips_state AS xwalk_fips_state,
|
||||
x.fips_county AS xwalk_fips_county,
|
||||
x.fips_place,
|
||||
x.population_acs,
|
||||
c.category,
|
||||
c.category_type,
|
||||
c.revenue_subtype
|
||||
FROM revenue_long_harmonized s
|
||||
LEFT JOIN canonical_fips_xwalk x USING (canonical_govid)
|
||||
LEFT JOIN summary_categories c USING (item_code);
|
||||
@@ -0,0 +1,16 @@
|
||||
CREATE OR REPLACE VIEW ig_annotated AS
|
||||
SELECT
|
||||
s.*,
|
||||
x.gov_name AS xwalk_gov_name,
|
||||
x.govs_type,
|
||||
x.type_label,
|
||||
x.fips_state AS xwalk_fips_state,
|
||||
x.fips_county AS xwalk_fips_county,
|
||||
x.fips_place,
|
||||
x.population_acs,
|
||||
c.category,
|
||||
c.category_type,
|
||||
c.spend_subtype
|
||||
FROM ig_long s
|
||||
LEFT JOIN canonical_fips_xwalk x USING (canonical_govid)
|
||||
LEFT JOIN summary_categories c USING (item_code);
|
||||
@@ -0,0 +1,16 @@
|
||||
CREATE OR REPLACE VIEW ig_annotated_harmonized AS
|
||||
SELECT
|
||||
s.*,
|
||||
x.gov_name AS xwalk_gov_name,
|
||||
x.govs_type,
|
||||
x.type_label,
|
||||
x.fips_state AS xwalk_fips_state,
|
||||
x.fips_county AS xwalk_fips_county,
|
||||
x.fips_place,
|
||||
x.population_acs,
|
||||
c.category,
|
||||
c.category_type,
|
||||
c.spend_subtype
|
||||
FROM ig_long_harmonized s
|
||||
LEFT JOIN canonical_fips_xwalk x USING (canonical_govid)
|
||||
LEFT JOIN summary_categories c USING (item_code);
|
||||
@@ -9,7 +9,8 @@ cog_geographic_rollup(
|
||||
category,
|
||||
years,
|
||||
per_capita = FALSE,
|
||||
adjust_to_year = NULL
|
||||
adjust_to_year = NULL,
|
||||
expenditure_concept = c("direct", "total")
|
||||
)
|
||||
}
|
||||
\arguments{
|
||||
@@ -27,6 +28,13 @@ population from `gov_population_yearly`. Govs with missing population
|
||||
are excluded from the result.}
|
||||
|
||||
\item{adjust_to_year}{Integer base year for CPI-U conversion, or `NULL`.}
|
||||
|
||||
\item{expenditure_concept}{`"direct"` (default) or `"total"`. Currently only
|
||||
`"direct"` is accepted; the `"total"` option exists in [cog_spending()] for
|
||||
single-government queries but cannot be used here because combining Total
|
||||
across multiple layers of government double-counts intergovernmental
|
||||
transfers (a state's payment to a school district is the same dollar the
|
||||
district reports as its own Direct spending).}
|
||||
}
|
||||
\value{
|
||||
Tibble with columns `year`, `layer`, `canonical_govid`, `gov_name`,
|
||||
|
||||
@@ -10,7 +10,8 @@ cog_peer_compare(
|
||||
category,
|
||||
years,
|
||||
per_capita = TRUE,
|
||||
adjust_to_year = NULL
|
||||
adjust_to_year = NULL,
|
||||
expenditure_concept = c("direct", "total")
|
||||
)
|
||||
}
|
||||
\arguments{
|
||||
@@ -27,6 +28,11 @@ cog_peer_compare(
|
||||
population.}
|
||||
|
||||
\item{adjust_to_year}{Integer base year for CPI-U conversion or `NULL`.}
|
||||
|
||||
\item{expenditure_concept}{`"direct"` (default) or `"total"`. Currently only
|
||||
`"direct"` is accepted; the `"total"` option exists in [cog_spending()] for
|
||||
single-government queries but cannot be used here because combining Total
|
||||
across peer sets counts intergovernmental transfers twice.}
|
||||
}
|
||||
\value{
|
||||
Tibble matching [cog_spending()]'s columns, plus a `role`
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
% Generated by roxygen2: do not edit by hand
|
||||
% Please edit documentation in R/recipes.R
|
||||
\name{cog_recipes}
|
||||
\alias{cog_recipes}
|
||||
\title{List available harmonization recipes}
|
||||
\usage{
|
||||
cog_recipes(pattern = NULL)
|
||||
}
|
||||
\arguments{
|
||||
\item{pattern}{Optional regex matched case-insensitively against
|
||||
`recipe_id` or `label`.}
|
||||
}
|
||||
\value{
|
||||
Tibble with columns `recipe_id`, `label`, `n_components`,
|
||||
`year_min`, `year_max` (the min/max component year coverage), sorted by
|
||||
`recipe_id`.
|
||||
}
|
||||
\description{
|
||||
Recipes are multi-code cross-vintage series (see [cog_spending()]'s
|
||||
`recipe` argument) catalogued in the corpus's `harmonization_recipes`
|
||||
table. Use this to discover valid `recipe` ids.
|
||||
}
|
||||
+27
-1
@@ -9,7 +9,9 @@ cog_revenue(
|
||||
years,
|
||||
category = NULL,
|
||||
per_capita = FALSE,
|
||||
adjust_to_year = NULL
|
||||
adjust_to_year = NULL,
|
||||
basis = c("harmonized", "raw"),
|
||||
recipe = NULL
|
||||
)
|
||||
}
|
||||
\arguments{
|
||||
@@ -29,6 +31,30 @@ in that year).}
|
||||
|
||||
\item{adjust_to_year}{Integer base year for CPI-U real-dollar conversion,
|
||||
or `NULL` for nominal only.}
|
||||
|
||||
\item{basis}{`"harmonized"` (default) sums item codes through the
|
||||
cross-vintage harmonization mapping (folding series-break-affected
|
||||
codes onto a comparable target and excluding aggregate / discontinued
|
||||
rows -- see the `harmonization` block in `cog_explain()`); `"raw"`
|
||||
reproduces the pre-Phase-R2 behavior (published item codes, no
|
||||
folding). On a corpus with `schema_version < 5` (no harmonization
|
||||
tables), `basis` silently resolves to `"raw"` when left at its default
|
||||
and the resolution is recorded in the provenance; explicitly passing
|
||||
`basis = "harmonized"` on such a corpus aborts. Ignored when `recipe`
|
||||
is set (see below).}
|
||||
|
||||
\item{recipe}{Optional harmonization recipe id (see [cog_recipes()]) for
|
||||
multi-code cross-vintage series that a 1:1 harmonized_code mapping
|
||||
can't express (e.g. a wide-era aggregate that only splits into leaf
|
||||
codes in the modern era). Mutually exclusive with `category`. The
|
||||
result's subtype column reads `"recipe"` and `category` reads the
|
||||
recipe's label. Requires `schema_version >= 5`. A recipe query bypasses
|
||||
`basis` entirely (it joins `long` directly rather than going through
|
||||
the `*_annotated`/`*_annotated_harmonized` views), so the `basis`
|
||||
argument is ignored and the result's provenance reports
|
||||
`basis = "recipe"` with an inert `harmonization` block (`applied =
|
||||
FALSE`, pointing at the `recipe` block instead) rather than a
|
||||
possibly-misleading `"harmonized"`/`"raw"` value.}
|
||||
}
|
||||
\value{
|
||||
Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||
|
||||
+57
-1
@@ -9,7 +9,10 @@ cog_spending(
|
||||
years,
|
||||
category = NULL,
|
||||
per_capita = FALSE,
|
||||
adjust_to_year = NULL
|
||||
adjust_to_year = NULL,
|
||||
basis = c("harmonized", "raw"),
|
||||
recipe = NULL,
|
||||
expenditure_concept = c("direct", "total")
|
||||
)
|
||||
}
|
||||
\arguments{
|
||||
@@ -29,6 +32,59 @@ in that year).}
|
||||
|
||||
\item{adjust_to_year}{Integer base year for CPI-U real-dollar conversion,
|
||||
or `NULL` for nominal only.}
|
||||
|
||||
\item{basis}{`"harmonized"` (default) sums item codes through the
|
||||
cross-vintage harmonization mapping (folding series-break-affected
|
||||
codes onto a comparable target and excluding aggregate / discontinued
|
||||
rows -- see the `harmonization` block in `cog_explain()`); `"raw"`
|
||||
reproduces the pre-Phase-R2 behavior (published item codes, no
|
||||
folding). On a corpus with `schema_version < 5` (no harmonization
|
||||
tables), `basis` silently resolves to `"raw"` when left at its default
|
||||
and the resolution is recorded in the provenance; explicitly passing
|
||||
`basis = "harmonized"` on such a corpus aborts. Ignored when `recipe`
|
||||
is set (see below).}
|
||||
|
||||
\item{recipe}{Optional harmonization recipe id (see [cog_recipes()]) for
|
||||
multi-code cross-vintage series that a 1:1 harmonized_code mapping
|
||||
can't express (e.g. a wide-era aggregate that only splits into leaf
|
||||
codes in the modern era). Mutually exclusive with `category`. The
|
||||
result's subtype column reads `"recipe"` and `category` reads the
|
||||
recipe's label. Requires `schema_version >= 5`. A recipe query bypasses
|
||||
`basis` entirely (it joins `long` directly rather than going through
|
||||
the `*_annotated`/`*_annotated_harmonized` views), so the `basis`
|
||||
argument is ignored and the result's provenance reports
|
||||
`basis = "recipe"` with an inert `harmonization` block (`applied =
|
||||
FALSE`, pointing at the `recipe` block instead) rather than a
|
||||
possibly-misleading `"harmonized"`/`"raw"` value.}
|
||||
|
||||
\item{expenditure_concept}{`"direct"` (default) returns only the
|
||||
government's own direct spending (item codes `E`/`F`/`G`), unchanged
|
||||
from prior releases. `"total"` additionally UNIONs in the
|
||||
intergovernmental leg -- payments to local governments (`M` codes) and
|
||||
to the state government (`L` codes, excluding the `L--` family-total
|
||||
rollup) -- so results gain rows with `spend_subtype ==
|
||||
"intergovernmental"`. Requires the active corpus's `summary_categories`
|
||||
to carry M/L rows (added by cog_pipeline PR #59); aborts with class
|
||||
`uscogdata_ig_categories_unsupported` on an older corpus rather than
|
||||
silently under-reporting. Mutually exclusive with `recipe` (a recipe
|
||||
already defines its own component codes). **Do not sum `"total"`
|
||||
results across levels of government** (e.g. state + county + city):
|
||||
a state's `M12` payment to a school district is the same dollar the
|
||||
district reports as its own direct `E12`, so summing both double-counts
|
||||
it. This matters in particular with [cog_geographic_rollup()], which
|
||||
sums across exactly that kind of multi-layer government set.
|
||||
|
||||
In the legacy wide era (<= FY2011), some functions are published ONLY
|
||||
as an aggregate-flagged family total (e.g. Corrections' `E04`/`E05`
|
||||
split), which the Direct leg excludes by construction but the IG leg
|
||||
deliberately keeps (see `inst/sql/24-ig_long.sql`). For a `"total"`
|
||||
query, any (year, category) where this leaves intergovernmental rows
|
||||
with NO Direct counterpart is flagged: the affected rows' `notes`
|
||||
name the harmonization recipe that recovers the missing Direct
|
||||
component (when one exists), and
|
||||
`provenance$expenditure_concept_direct_suppressed` is `TRUE` -- the
|
||||
figure in those rows is the intergovernmental leg alone, not Direct +
|
||||
IG.}
|
||||
}
|
||||
\value{
|
||||
Tibble with columns `year`, `canonical_govid`, `gov_name`,
|
||||
|
||||
@@ -26,3 +26,68 @@ with_fixture_corpus <- function(code) {
|
||||
}, add = TRUE)
|
||||
force(code)
|
||||
}
|
||||
|
||||
# Copy the bundled fixture to a temp dir with manifest.json's schema_version
|
||||
# patched to `version`, then run `code` against it with a clean session
|
||||
# (mirrors with_fixture_corpus()). Used to exercise the v4/v5 dual-accept
|
||||
# path without a second physical fixture tree: a real v4 corpus has no
|
||||
# harmonization_map/harmonization_recipes/series_breaks parquet files, but
|
||||
# .register_views() only *reads* those when schema_version >= 5 (see
|
||||
# R/views.R), so a doctored copy of the (v5) bundled fixture with the
|
||||
# manifest's schema_version knocked down to 4 is a faithful stand-in.
|
||||
with_doctored_schema_version <- function(version, code) {
|
||||
src <- fixture_corpus_path()
|
||||
tmp <- withr::local_tempdir(.local_envir = parent.frame())
|
||||
file.copy(list.files(src, full.names = TRUE), tmp, recursive = TRUE)
|
||||
|
||||
manifest_path <- file.path(tmp, "manifest.json")
|
||||
m <- jsonlite::fromJSON(manifest_path, simplifyVector = FALSE)
|
||||
m$schema_version <- as.integer(version)
|
||||
writeLines(
|
||||
jsonlite::toJSON(m, auto_unbox = TRUE, pretty = TRUE, null = "null"),
|
||||
manifest_path
|
||||
)
|
||||
|
||||
old_url <- Sys.getenv("USCOGDATA_URL", unset = NA)
|
||||
uscogdata:::cog_close()
|
||||
Sys.setenv(USCOGDATA_URL = paste0(tmp, "/"))
|
||||
on.exit({
|
||||
uscogdata:::cog_close()
|
||||
if (is.na(old_url)) Sys.unsetenv("USCOGDATA_URL") else Sys.setenv(USCOGDATA_URL = old_url)
|
||||
}, add = TRUE)
|
||||
force(code)
|
||||
}
|
||||
|
||||
# Copy the bundled fixture to a temp dir with summary_categories.parquet
|
||||
# rewritten to drop every M/L (intergovernmental) row, then run `code`
|
||||
# against it with a clean session (mirrors with_fixture_corpus()/
|
||||
# with_doctored_schema_version()). Models a real pre-cog_pipeline-PR#59
|
||||
# corpus: the 66 M/L category rows shipped with NO schema_version bump (see
|
||||
# C2 in the expenditure-concept review), so schema_version is left
|
||||
# untouched here -- only the category data itself is rolled back.
|
||||
with_corpus_missing_ig_categories <- function(code) {
|
||||
src <- fixture_corpus_path()
|
||||
tmp <- withr::local_tempdir(.local_envir = parent.frame())
|
||||
file.copy(list.files(src, full.names = TRUE), tmp, recursive = TRUE)
|
||||
|
||||
cats_path <- file.path(tmp, "data", "summary_categories.parquet")
|
||||
filtered_path <- file.path(tmp, "data", "summary_categories_filtered.parquet")
|
||||
write_con <- DBI::dbConnect(duckdb::duckdb())
|
||||
on.exit(DBI::dbDisconnect(write_con, shutdown = TRUE), add = TRUE)
|
||||
DBI::dbExecute(write_con, sprintf(
|
||||
"COPY (SELECT * FROM read_parquet(%s) WHERE LEFT(item_code, 1) NOT IN ('M', 'L'))
|
||||
TO %s (FORMAT PARQUET)",
|
||||
uscogdata:::.sql_lit_chr(cats_path), uscogdata:::.sql_lit_chr(filtered_path)
|
||||
))
|
||||
file.remove(cats_path)
|
||||
file.rename(filtered_path, cats_path)
|
||||
|
||||
old_url <- Sys.getenv("USCOGDATA_URL", unset = NA)
|
||||
uscogdata:::cog_close()
|
||||
Sys.setenv(USCOGDATA_URL = paste0(tmp, "/"))
|
||||
on.exit({
|
||||
uscogdata:::cog_close()
|
||||
if (is.na(old_url)) Sys.unsetenv("USCOGDATA_URL") else Sys.setenv(USCOGDATA_URL = old_url)
|
||||
}, add = TRUE)
|
||||
force(code)
|
||||
}
|
||||
|
||||
@@ -15,7 +15,18 @@ test_that("cog_categories(type = 'spending') returns only expenditure rows", {
|
||||
skip_if_no_corpus()
|
||||
r <- cog_categories(type = "spending")
|
||||
expect_true(all(r$category_type == "expenditure"))
|
||||
expect_true(all(r$subtype %in% c("operations", "capital")))
|
||||
expect_true(all(r$subtype %in% c("operations", "capital", "intergovernmental")))
|
||||
})
|
||||
|
||||
test_that("cog_categories surfaces the intergovernmental spending subtype", {
|
||||
skip_if_no_corpus()
|
||||
r <- cog_categories(type = "spending")
|
||||
expect_true("intergovernmental" %in% r$subtype)
|
||||
# IG rows reuse the existing functional categories -- they add a subtype,
|
||||
# not new category values.
|
||||
ig_cats <- sort(unique(r$category[r$subtype == "intergovernmental"]))
|
||||
direct_cats <- sort(unique(r$category[r$subtype != "intergovernmental"]))
|
||||
expect_true(all(ig_cats %in% c(direct_cats, "Other Education")))
|
||||
})
|
||||
|
||||
test_that("cog_categories(type = 'revenue') returns only revenue rows", {
|
||||
|
||||
@@ -26,3 +26,41 @@ test_that(".resolve_cache_dir falls back to R_user_dir", {
|
||||
})
|
||||
})
|
||||
})
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Trailing-slash normalization (uscogdata #3 follow-up).
|
||||
#
|
||||
# EVERY consumer builds paths by concatenation: paste0(url, "manifest.json")
|
||||
# (manifest.R), paste0(url, e$path) (mirror.R), and the parquet glob in
|
||||
# views.R. mirror.R:104 even comments 'url ends in "/"' -- an assumption the
|
||||
# package documents and relies on but never enforced.
|
||||
#
|
||||
# A URL missing its trailing slash therefore fails SILENTLY and confusingly:
|
||||
# HTTPS -> ".../downloadmanifest.json" -> the host answers with an HTML 404
|
||||
# page -> the jsonlite lexical error that issue #3 reported;
|
||||
# local -> ".../corpusdata/long/**/*.parquet" -> DuckDB "No files found".
|
||||
# Neither message points at the real cause. Normalize once, at resolution.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
test_that(".resolve_url appends a missing trailing slash", {
|
||||
withr::local_envvar(USCOGDATA_URL = "https://example.org/s/TOKEN/download")
|
||||
expect_equal(.resolve_url(), "https://example.org/s/TOKEN/download/")
|
||||
})
|
||||
|
||||
test_that(".resolve_url leaves an existing trailing slash alone", {
|
||||
withr::local_envvar(USCOGDATA_URL = "https://example.org/s/TOKEN/download/")
|
||||
expect_equal(.resolve_url(), "https://example.org/s/TOKEN/download/")
|
||||
})
|
||||
|
||||
test_that(".resolve_url normalizes a local path without a trailing slash", {
|
||||
withr::local_envvar(USCOGDATA_URL = "/tmp/corpus")
|
||||
expect_equal(.resolve_url(), "/tmp/corpus/")
|
||||
})
|
||||
|
||||
test_that(".resolve_url does not invent a slash for an empty setting", {
|
||||
# An unset/empty URL must stay empty so the "not configured" guard in
|
||||
# manifest.R still fires, rather than degrading into a bare "/" root.
|
||||
withr::local_envvar(USCOGDATA_URL = "")
|
||||
withr::local_options(uscogdata.url = "")
|
||||
expect_equal(.resolve_url(), "")
|
||||
})
|
||||
|
||||
@@ -0,0 +1,536 @@
|
||||
test_that("the corpus contains no K-prefix rows, so the Direct leg omits K", {
|
||||
con <- .ensure_session()
|
||||
n <- DBI::dbGetQuery(con,
|
||||
"SELECT COUNT(*) AS n FROM long WHERE LEFT(item_code, 1) = 'K'")$n
|
||||
expect_equal(n, 0)
|
||||
|
||||
sql_files <- c("20-spending_long.sql", "22-spending_long_harmonized.sql")
|
||||
for (f in sql_files) {
|
||||
txt <- paste(readLines(system.file("sql", f, package = "uscogdata")),
|
||||
collapse = " ")
|
||||
expect_false(grepl("'K'", txt, fixed = TRUE),
|
||||
label = paste(f, "must not reference the inert K prefix"))
|
||||
}
|
||||
})
|
||||
|
||||
test_that("expenditure_concept defaults to direct and preserves today's numbers", {
|
||||
gov <- "010000226085" # Alabama state government
|
||||
base <- cog_spending(gov, years = 2019, category = "Police")
|
||||
expl <- cog_spending(gov, years = 2019, category = "Police",
|
||||
expenditure_concept = "direct")
|
||||
expect_equal(base$amt_nominal, expl$amt_nominal)
|
||||
expect_false("intergovernmental" %in% base$spend_subtype)
|
||||
})
|
||||
|
||||
test_that("expenditure_concept = 'total' adds an intergovernmental subtype", {
|
||||
gov <- "010000226085"
|
||||
d <- cog_spending(gov, years = 2019, category = "Police",
|
||||
expenditure_concept = "direct")
|
||||
t <- cog_spending(gov, years = 2019, category = "Police",
|
||||
expenditure_concept = "total")
|
||||
expect_true("intergovernmental" %in% t$spend_subtype)
|
||||
# Direct rows are untouched; Total only ever ADDS. Use %in% rather than
|
||||
# != : a category = NULL result can contain a NULL-subtype group (codes
|
||||
# with no summary_categories row, e.g. E16/E21/E85/F16/F85/G16/G21/G85),
|
||||
# and `NA != "intergovernmental"` is NA, not TRUE, which would silently
|
||||
# smuggle an all-NA phantom row into dt.
|
||||
dt <- t[!(t$spend_subtype %in% "intergovernmental"), ]
|
||||
expect_equal(sort(dt$amt_nominal), sort(d$amt_nominal))
|
||||
expect_gt(sum(t$amt_nominal), sum(d$amt_nominal))
|
||||
})
|
||||
|
||||
test_that("legacy-era Total does not collapse to Direct (the is_aggregate trap)", {
|
||||
# In the wide era the IG dollars live almost entirely on aggregate-flagged
|
||||
# rows. A Total leg that inherited the Direct leg's NOT is_aggregate filter
|
||||
# would silently return Total == Direct here.
|
||||
gov <- "010000226085"
|
||||
d <- cog_spending(gov, years = 2011, category = "Education K-12",
|
||||
expenditure_concept = "direct")
|
||||
t <- cog_spending(gov, years = 2011, category = "Education K-12",
|
||||
expenditure_concept = "total")
|
||||
expect_true("intergovernmental" %in% t$spend_subtype)
|
||||
ig <- sum(t$amt_nominal[t$spend_subtype == "intergovernmental"])
|
||||
expect_gt(ig, 0)
|
||||
expect_gt(sum(t$amt_nominal), sum(d$amt_nominal))
|
||||
})
|
||||
|
||||
test_that("the IG leg never includes the L-- family total", {
|
||||
con <- .ensure_session()
|
||||
codes <- DBI::dbGetQuery(con,
|
||||
"SELECT DISTINCT item_code FROM ig_long")$item_code
|
||||
expect_false(any(grepl("--$", codes)))
|
||||
expect_true(all(substr(codes, 1, 1) %in% c("M", "L")))
|
||||
})
|
||||
|
||||
test_that("expenditure_concept rejects unknown values", {
|
||||
expect_error(
|
||||
cog_spending("010000226085", years = 2019, expenditure_concept = "gross"),
|
||||
class = "rlang_error"
|
||||
)
|
||||
})
|
||||
|
||||
test_that("total composes with basis = 'raw' and basis = 'harmonized'", {
|
||||
gov <- "010000226085"
|
||||
h <- cog_spending(gov, years = 2011, category = "Education K-12",
|
||||
expenditure_concept = "total", basis = "harmonized")
|
||||
r <- cog_spending(gov, years = 2011, category = "Education K-12",
|
||||
expenditure_concept = "total", basis = "raw")
|
||||
ig_h <- sum(h$amt_nominal[h$spend_subtype == "intergovernmental"])
|
||||
ig_r <- sum(r$amt_nominal[r$spend_subtype == "intergovernmental"])
|
||||
# The only IG harmonization rule is M38 -> M36 (year-disjoint), so the IG
|
||||
# total must agree between bases even though the code labels may differ.
|
||||
expect_equal(ig_h, ig_r)
|
||||
})
|
||||
|
||||
test_that("recipe = and expenditure_concept = 'total' together aborts", {
|
||||
expect_error(
|
||||
cog_spending("121011212191", 2020L, recipe = "corrections_combined",
|
||||
expenditure_concept = "total"),
|
||||
class = "uscogdata_recipe_concept_conflict"
|
||||
)
|
||||
})
|
||||
|
||||
test_that("aggregate-sourced IG dollars are flagged aggregate_fallback = TRUE (bool_or, not bool_and)", {
|
||||
# Regression test: .build_verb_sql() originally used bool_and(is_aggregate)
|
||||
# for aggregate_fallback, which is correct for the Direct leg (a group can
|
||||
# never mix aggregate and non-aggregate rows there -- spending_long filters
|
||||
# NOT is_aggregate) but wrong for the IG leg. The wide era is dense -- every
|
||||
# government has a $0 row for every code in a family -- so a $0 leaf sits in
|
||||
# the same (year, gov, subtype, category) group as the real aggregate row
|
||||
# and flips bool_and() to FALSE. Measured: AL state 2011 had $5,740,775,000
|
||||
# of aggregate-sourced IG dollars (Corrections $31,358,000 + Education K-12
|
||||
# $5,152,385,000 + General Government $557,032,000) reporting
|
||||
# aggregate_fallback = FALSE under bool_and(), with the only TRUE row being
|
||||
# Transit Utilities at $0. bool_or() reports all of them correctly.
|
||||
gov <- "010000226085"
|
||||
t <- cog_spending(gov, years = 2011, category = "Education K-12",
|
||||
expenditure_concept = "total")
|
||||
ig <- t[t$spend_subtype == "intergovernmental", ]
|
||||
expect_equal(nrow(ig), 1L)
|
||||
expect_true(ig$aggregate_fallback)
|
||||
expect_true(nzchar(ig$notes))
|
||||
expect_match(ig$notes, "Aggregate fallback applied", fixed = TRUE)
|
||||
})
|
||||
|
||||
test_that("legacy aggregate IG codes are year-disjoint from their modern leaf components", {
|
||||
# The safety of ig_long's deliberate omission of `NOT is_aggregate` (see
|
||||
# inst/sql/24-ig_long.sql) rests entirely on each legacy code's AGGREGATE
|
||||
# instance being year-disjoint from the modern leaf codes it rolls up --
|
||||
# if a future corpus rebuild ever back-filled a leaf into a year where the
|
||||
# code is still flagged aggregate, `total` would silently double-count and
|
||||
# this suite would still pass. This test fails loudly if that ever
|
||||
# happens.
|
||||
#
|
||||
# Note the invariant is scoped to the AGGREGATE flag, not bare code
|
||||
# presence: M89/L89 do NOT disappear after the wide era the way M47/L47
|
||||
# do -- they continue past 2011 as their OWN independent leaf line item
|
||||
# (is_aggregate = FALSE) alongside M91-93/L91-93, which is fine because a
|
||||
# non-aggregate M89/L89 no longer represents a rollup of those codes.
|
||||
# (Verified in the fixture: M89/L89 are is_aggregate = TRUE only in 2011,
|
||||
# when M91-93/L91-93 don't exist yet; from 2012 on M89/L89 are
|
||||
# is_aggregate = FALSE leaves coexisting with M91-93/L91-93.)
|
||||
#
|
||||
# Pairs are the M/L-prefixed components (this package's ig_long only
|
||||
# covers M/L; other prefixes in the same rollup, e.g. N/O/P/Q/R, fall
|
||||
# outside its domain and are irrelevant here) enumerated in
|
||||
# cog_pipeline's data/wide_to_long_xwalk.csv `full_desc` column (read
|
||||
# once at authoring time, not at test time -- this test stays offline):
|
||||
# M47 "To local governments, total (includes N47, O47, P47, R47, and M94)"
|
||||
# M89 "To local governments, total (incl N89, O89, P89, R89, M91, M92, and M93)"
|
||||
# L47 "To state government (includes L94)"
|
||||
# L89 "To state government (includes L91, L92, and L93)"
|
||||
con <- .ensure_session()
|
||||
pairs <- list(
|
||||
list(aggregate = "M47", components = "M94"),
|
||||
list(aggregate = "M89", components = c("M91", "M92", "M93")),
|
||||
list(aggregate = "L47", components = "L94"),
|
||||
list(aggregate = "L89", components = c("L91", "L92", "L93"))
|
||||
)
|
||||
agg_years_by_code <- DBI::dbGetQuery(con,
|
||||
"SELECT DISTINCT year, item_code FROM ig_long WHERE is_aggregate")
|
||||
codes_by_year <- DBI::dbGetQuery(con, "SELECT DISTINCT year, item_code FROM ig_long")
|
||||
|
||||
for (p in pairs) {
|
||||
agg_years <- agg_years_by_code$year[agg_years_by_code$item_code == p$aggregate]
|
||||
for (yr in agg_years) {
|
||||
codes_yr <- codes_by_year$item_code[codes_by_year$year == yr]
|
||||
has_component <- any(p$components %in% codes_yr)
|
||||
expect_false(
|
||||
has_component,
|
||||
label = sprintf(
|
||||
"year %s has aggregate-flagged %s co-occurring with a modern component (%s)",
|
||||
yr, p$aggregate, paste(p$components, collapse = ",")
|
||||
)
|
||||
)
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
test_that(".verb_spendrev rejects expenditure_concept = 'total' for a non-spending view_base", {
|
||||
# cog_revenue() never exposes expenditure_concept and always resolves it
|
||||
# to the "direct" default, so there is no revenue codepath that reaches
|
||||
# this today -- but .verb_spendrev() is shared, and nothing else stops a
|
||||
# future caller from passing expenditure_concept = "total" alongside
|
||||
# view_base = "revenue_annotated", which would UNION expenditure M/L rows
|
||||
# into a revenue result. Exercise the internal helper directly.
|
||||
expect_error(
|
||||
uscogdata:::.verb_spendrev(
|
||||
verb = "cog_revenue_test", view_base = "revenue_annotated",
|
||||
subtype_col = "revenue_subtype",
|
||||
flow_prefixes = c("T", "A", "U", "B", "C", "D"),
|
||||
call = quote(cog_revenue_test()),
|
||||
govid = "010000226085", years = 2019L, category = NULL,
|
||||
per_capita = FALSE, adjust_to_year = NULL, basis = "raw",
|
||||
recipe = NULL, expenditure_concept = "total"
|
||||
),
|
||||
class = "uscogdata_expenditure_concept_unsupported"
|
||||
)
|
||||
})
|
||||
|
||||
test_that("cog_geographic_rollup refuses expenditure_concept = 'total'", {
|
||||
expect_error(
|
||||
cog_geographic_rollup(
|
||||
govids = list(state = "010000226085"),
|
||||
category = "Police", years = 2019,
|
||||
expenditure_concept = "total"
|
||||
),
|
||||
class = "uscogdata_concept_not_aggregatable"
|
||||
)
|
||||
})
|
||||
|
||||
test_that("cog_peer_compare refuses expenditure_concept = 'total'", {
|
||||
expect_error(
|
||||
cog_peer_compare(
|
||||
target_govid = "010000226085", peers = "010000226085",
|
||||
category = "Police", years = 2019,
|
||||
expenditure_concept = "total"
|
||||
),
|
||||
class = "uscogdata_concept_not_aggregatable"
|
||||
)
|
||||
})
|
||||
|
||||
test_that("the refusal message names the fix and the reason", {
|
||||
err <- tryCatch(
|
||||
cog_geographic_rollup(govids = list(state = "010000226085"),
|
||||
category = "Police", years = 2019,
|
||||
expenditure_concept = "total"),
|
||||
condition = function(e) e
|
||||
)
|
||||
msg <- paste(conditionMessage(err), collapse = " ")
|
||||
expect_match(msg, "direct")
|
||||
expect_match(msg, "double-count|double count")
|
||||
expect_match(msg, "cog_geographic_rollup")
|
||||
|
||||
# Test that cog_peer_compare's message names its own function
|
||||
err2 <- tryCatch(
|
||||
cog_peer_compare(target_govid = "010000226085", peers = "010000226085",
|
||||
category = "Police", years = 2019,
|
||||
expenditure_concept = "total"),
|
||||
condition = function(e) e
|
||||
)
|
||||
msg2 <- paste(conditionMessage(err2), collapse = " ")
|
||||
expect_match(msg2, "direct")
|
||||
expect_match(msg2, "double-count|double count")
|
||||
expect_match(msg2, "cog_peer_compare")
|
||||
})
|
||||
|
||||
test_that("both cross-government verbs still accept the direct default", {
|
||||
expect_no_error(
|
||||
cog_geographic_rollup(govids = list(state = "010000226085"),
|
||||
category = "Police", years = 2019)
|
||||
)
|
||||
expect_no_error(
|
||||
cog_peer_compare(target_govid = "010000226085", peers = "010000226085",
|
||||
category = "Police", years = 2019)
|
||||
)
|
||||
})
|
||||
|
||||
test_that("provenance always records the expenditure concept", {
|
||||
d <- cog_spending("010000226085", years = 2019, category = "Police")
|
||||
t <- cog_spending("010000226085", years = 2019, category = "Police",
|
||||
expenditure_concept = "total")
|
||||
expect_equal(attr(d, "provenance")$expenditure_concept, "direct")
|
||||
expect_equal(attr(t, "provenance")$expenditure_concept, "total")
|
||||
# The note explains the non-obvious part: how legacy IG was assembled.
|
||||
expect_true(nzchar(attr(t, "provenance")$expenditure_concept_note))
|
||||
expect_true(is.na(attr(d, "provenance")$expenditure_concept_note) ||
|
||||
!nzchar(attr(d, "provenance")$expenditure_concept_note))
|
||||
})
|
||||
|
||||
test_that("the provenance schema documents expenditure_concept", {
|
||||
sch <- jsonlite::fromJSON(
|
||||
system.file("schemas", "provenance-v1.json", package = "uscogdata"),
|
||||
simplifyVector = FALSE
|
||||
)
|
||||
expect_true("expenditure_concept" %in% names(sch$properties))
|
||||
})
|
||||
|
||||
test_that("a firing suggestion names the intergovernmental counterpart recipe", {
|
||||
# Corrections has no legacy leaf rows, so the coverage-gap suggestion fires;
|
||||
# corrections_ig_local_combined is its IG counterpart.
|
||||
r <- suppressMessages(
|
||||
cog_spending("010000226085", years = c(2005, 2011), category = "Corrections")
|
||||
)
|
||||
sugg <- attr(r, "provenance")$suggestions
|
||||
expect_gt(length(sugg), 0L)
|
||||
ids <- vapply(sugg, function(s) s$recipe_id %||% "", character(1))
|
||||
expect_true("corrections_combined" %in% ids)
|
||||
ig <- unlist(lapply(sugg, function(s) s$ig_recipe_id))
|
||||
expect_true("corrections_ig_local_combined" %in% ig)
|
||||
})
|
||||
|
||||
test_that("no suggestion fires for a healthy query", {
|
||||
r <- cog_spending("010000226085", years = 2019, category = "Police")
|
||||
expect_length(attr(r, "provenance")$suggestions, 0L)
|
||||
})
|
||||
|
||||
test_that("a mis-scoped cog_spending() call never attaches an M/L counterpart to a revenue-flavored recipe", {
|
||||
# "IG Federal" is a revenue-only category (summary_categories maps it to
|
||||
# B-prefixed component codes only; its recipes are ig_federal_b47_wide /
|
||||
# ig_federal_b89_wide). A cog_spending() call scoped to it returns zero
|
||||
# spending rows for every requested year -- there is no spending
|
||||
# component in this category at all -- so the coverage-gap machinery
|
||||
# fires for real (not hypothetically) even though this isn't the kind of
|
||||
# format-boundary gap the recipe catalog is meant to signpost. This is
|
||||
# exactly the live-corpus risk flagged in review: ig_federal_b47_wide's
|
||||
# own component codes (B47/B94, suffixes {"47","94"}) are an EXACT
|
||||
# suffix-set match for the expenditure recipe ige_local_m47_wide
|
||||
# (M47/M94, same suffixes) -- a coincidence of reused digits, not a real
|
||||
# Direct/Total pairing. The flow-family gate in
|
||||
# .attach_ig_counterparts() must keep ig_recipe_id NULL here.
|
||||
r <- suppressMessages(
|
||||
cog_spending("010000226085", years = c(2005, 2011), category = "IG Federal")
|
||||
)
|
||||
sugg <- attr(r, "provenance")$suggestions
|
||||
expect_gt(length(sugg), 0L)
|
||||
ids <- vapply(sugg, function(s) s$recipe_id %||% "", character(1))
|
||||
expect_true("ig_federal_b47_wide" %in% ids)
|
||||
ig <- unlist(lapply(sugg, function(s) s$ig_recipe_id))
|
||||
expect_length(ig, 0L)
|
||||
})
|
||||
|
||||
test_that("C1: 'total' on a legacy aggregate-only family reports the IG-only figure honestly, not as Direct + IG", {
|
||||
# AL state government, Corrections, 2011. Measured pre-fix: 'total'
|
||||
# returned $31,358,000 (the IG leg alone, on an aggregate-flagged M04/M05
|
||||
# row) with 0 suggestions (the surviving IG row made the gap-detection
|
||||
# machinery think the Direct leg was covered) and a note asserting
|
||||
# "Total = Direct + intergovernmental" with no caveat. True Direct (via
|
||||
# recipe = "corrections_combined") is $521,651,000 -- the IG-only figure
|
||||
# is ~6% of it.
|
||||
gov <- "010000226085"
|
||||
|
||||
d <- cog_spending(gov, years = 2011, category = "Corrections",
|
||||
expenditure_concept = "direct")
|
||||
expect_equal(nrow(d), 0L)
|
||||
|
||||
t <- suppressMessages(cog_spending(
|
||||
gov, years = 2011, category = "Corrections", expenditure_concept = "total"
|
||||
))
|
||||
expect_equal(nrow(t), 1L)
|
||||
expect_equal(t$spend_subtype, "intergovernmental")
|
||||
expect_equal(t$amt_nominal, 31358000)
|
||||
|
||||
r <- cog_spending(gov, years = 2011, recipe = "corrections_combined")
|
||||
expect_equal(r$amt_nominal, 521651000)
|
||||
|
||||
# C1(a): the recipe hints must fire for "total" exactly as they do for
|
||||
# "direct" -- the surviving IG row must not be mistaken for Direct
|
||||
# coverage.
|
||||
prov <- attr(t, "provenance")
|
||||
expect_gt(length(prov$suggestions), 0L)
|
||||
ids <- vapply(prov$suggestions, function(s) s$recipe_id %||% "", character(1))
|
||||
expect_true("corrections_combined" %in% ids)
|
||||
|
||||
# C1(b): the affected row's notes name a recovering recipe rather than
|
||||
# staying silent, and the provenance carries a flag a downstream consumer
|
||||
# (e.g. cog-api, which passes provenance through verbatim) can test.
|
||||
expect_true(nzchar(t$notes))
|
||||
expect_match(t$notes, "unavailable", fixed = TRUE)
|
||||
expect_match(t$notes, "corrections_combined", fixed = TRUE)
|
||||
expect_true(prov$expenditure_concept_direct_suppressed)
|
||||
|
||||
# The base "Total = Direct + IG" note must NOT stand unqualified when that
|
||||
# arithmetic didn't actually happen for this row.
|
||||
expect_match(prov$expenditure_concept_note, "NOTE", fixed = TRUE)
|
||||
expect_match(prov$expenditure_concept_note,
|
||||
"expenditure_concept_direct_suppressed", fixed = TRUE)
|
||||
})
|
||||
|
||||
test_that("C1(b): expenditure_concept_direct_suppressed is FALSE when the Direct leg is present", {
|
||||
d <- cog_spending("010000226085", years = 2019, category = "Police",
|
||||
expenditure_concept = "direct")
|
||||
t <- cog_spending("010000226085", years = 2019, category = "Police",
|
||||
expenditure_concept = "total")
|
||||
expect_false(isTRUE(attr(d, "provenance")$expenditure_concept_direct_suppressed))
|
||||
expect_false(isTRUE(attr(t, "provenance")$expenditure_concept_direct_suppressed))
|
||||
expect_false(any(nzchar(t$notes[t$spend_subtype == "intergovernmental"]) &
|
||||
grepl("unavailable", t$notes[t$spend_subtype == "intergovernmental"])))
|
||||
})
|
||||
|
||||
# M/I fix: .detect_direct_suppressed() was equating "no Direct sibling row"
|
||||
# with "Direct was suppressed", but the dominant real cause is a government
|
||||
# that simply has no direct spending in that category -- correct, ordinary
|
||||
# data. The fix gates the flag (and its row note) on a harmonization recipe
|
||||
# ACTUALLY covering that exact (year, canonical_govid, category) triple.
|
||||
|
||||
test_that("M/I: true positive, category supplied explicitly (unchanged behavior)", {
|
||||
al <- "010000226085"
|
||||
t_cat <- suppressMessages(cog_spending(
|
||||
al, years = 2011, category = "Corrections", expenditure_concept = "total"
|
||||
))
|
||||
expect_true(attr(t_cat, "provenance")$expenditure_concept_direct_suppressed)
|
||||
expect_match(t_cat$notes, "corrections_combined", fixed = TRUE)
|
||||
expect_match(t_cat$notes, "unavailable", fixed = TRUE)
|
||||
})
|
||||
|
||||
test_that("M/I: true positive, category = NULL now also names the recipe (was the fallback bug)", {
|
||||
# Root bug: .build_suggestions() short-circuits to list() when category is
|
||||
# NULL, so the note previously always hit its "no covering recipe found"
|
||||
# fallback here even though corrections_combined genuinely covers this row.
|
||||
al <- "010000226085"
|
||||
t_null <- suppressMessages(cog_spending(
|
||||
al, years = 2011, category = NULL, expenditure_concept = "total"
|
||||
))
|
||||
corr_row <- t_null[t_null$category %in% "Corrections", ]
|
||||
expect_equal(nrow(corr_row), 1L)
|
||||
expect_true(attr(t_null, "provenance")$expenditure_concept_direct_suppressed)
|
||||
expect_match(corr_row$notes, "corrections_combined", fixed = TRUE)
|
||||
expect_match(corr_row$notes, "unavailable", fixed = TRUE)
|
||||
expect_false(grepl("no covering recipe found", corr_row$notes, fixed = TRUE))
|
||||
})
|
||||
|
||||
test_that("M/I: false positive -- Virginia Education K-12 FY2019 total is NOT flagged", {
|
||||
# States fund K-12 through school districts, so the Direct leg (E12/F12/
|
||||
# G12) is genuinely, correctly zero -- not suppressed. Must not be flagged
|
||||
# and must carry no suppression note.
|
||||
va <- "510000227542"
|
||||
t_va <- suppressMessages(cog_spending(
|
||||
va, years = 2019, category = "Education K-12", expenditure_concept = "total"
|
||||
))
|
||||
expect_equal(nrow(t_va), 1L)
|
||||
expect_equal(t_va$spend_subtype, "intergovernmental")
|
||||
expect_equal(t_va$amt_nominal, 8028179000)
|
||||
expect_false(isTRUE(attr(t_va, "provenance")$expenditure_concept_direct_suppressed))
|
||||
expect_false(nzchar(t_va$notes) && grepl("unavailable", t_va$notes))
|
||||
})
|
||||
|
||||
test_that("M/I: false positive by construction -- 'Other Education' has no E/F/G code, never flagged", {
|
||||
# "Other Education" maps only to M21/L21 in summary_categories -- there is
|
||||
# no E/F/G code for it in this corpus at all, so no Direct-recovering
|
||||
# recipe can exist and it must never be flagged, in any fixture year.
|
||||
con <- uscogdata:::.ensure_session()
|
||||
years_all <- DBI::dbGetQuery(con, "SELECT DISTINCT year FROM long ORDER BY year")$year
|
||||
states <- DBI::dbGetQuery(con,
|
||||
"SELECT DISTINCT canonical_govid FROM long WHERE type = 0")$canonical_govid
|
||||
oe <- suppressMessages(cog_spending(
|
||||
states, years = years_all, category = "Other Education",
|
||||
expenditure_concept = "total"
|
||||
))
|
||||
expect_false(isTRUE(attr(oe, "provenance")$expenditure_concept_direct_suppressed))
|
||||
expect_false(any(nzchar(oe$notes) & grepl("unavailable", oe$notes)))
|
||||
})
|
||||
|
||||
test_that("M/I: a clean FY2019 category = NULL total query flags far fewer than the pre-fix 32/50 states", {
|
||||
con <- uscogdata:::.ensure_session()
|
||||
states <- DBI::dbGetQuery(con,
|
||||
"SELECT DISTINCT canonical_govid FROM long WHERE type = 0")$canonical_govid
|
||||
r <- suppressMessages(cog_spending(
|
||||
states, years = 2019, category = NULL, expenditure_concept = "total"
|
||||
))
|
||||
ig <- r[r$spend_subtype == "intergovernmental", ]
|
||||
flagged <- ig[nzchar(ig$notes) & grepl("unavailable", ig$notes), ]
|
||||
expect_lt(length(unique(flagged$canonical_govid)), 32L)
|
||||
# Every remaining flagged row must actually name a covering recipe --
|
||||
# never the old no-recipe-found fallback.
|
||||
expect_true(all(grepl("recipe = '", flagged$notes, fixed = TRUE)))
|
||||
expect_false(any(grepl("no covering recipe found", flagged$notes, fixed = TRUE)))
|
||||
})
|
||||
|
||||
test_that("C2: expenditure_concept = 'total' aborts on a corpus with no intergovernmental category rows", {
|
||||
with_corpus_missing_ig_categories({
|
||||
con <- uscogdata:::.ensure_session()
|
||||
n <- DBI::dbGetQuery(con,
|
||||
"SELECT COUNT(*) AS n FROM summary_categories WHERE LEFT(item_code, 1) IN ('M', 'L')"
|
||||
)$n
|
||||
expect_equal(n, 0)
|
||||
|
||||
err <- tryCatch(
|
||||
cog_spending("010000226085", years = 2019, category = "Police",
|
||||
expenditure_concept = "total"),
|
||||
condition = function(e) e
|
||||
)
|
||||
expect_s3_class(err, "uscogdata_ig_categories_unsupported")
|
||||
msg <- conditionMessage(err)
|
||||
expect_match(msg, "PR #59|predates", perl = TRUE)
|
||||
})
|
||||
|
||||
# 'direct' is unaffected on the same corpus -- the guard is scoped to
|
||||
# expenditure_concept = "total" only.
|
||||
with_corpus_missing_ig_categories({
|
||||
expect_no_error(
|
||||
cog_spending("010000226085", years = 2019, category = "Police",
|
||||
expenditure_concept = "direct")
|
||||
)
|
||||
})
|
||||
})
|
||||
|
||||
test_that("C2: expenditure_concept = 'total' still works on a corpus that DOES carry M/L category rows", {
|
||||
expect_no_error(
|
||||
cog_spending("010000226085", years = 2019, category = "Police",
|
||||
expenditure_concept = "total")
|
||||
)
|
||||
})
|
||||
|
||||
test_that("I2: an intergovernmental (M/L) recipe never appears as its own top-level suggestion", {
|
||||
# Task 1's M04/M05 category rows share the "Corrections" summary_categories
|
||||
# category with the Direct-flavored E04/E05, so `corrections_ig_local_
|
||||
# combined` (entirely M-prefixed) becomes a raw *candidate* in
|
||||
# .build_suggestions()'s component_code-driven query. Following a
|
||||
# "re-run with recipe = 'corrections_ig_local_combined'" hint on a plain
|
||||
# cog_spending() call would silently return intergovernmental dollars
|
||||
# under provenance$expenditure_concept = "direct". Task 6's gate
|
||||
# (.attach_ig_counterparts()) already protects the *counterpart* lookup;
|
||||
# this exercises that the candidate list itself is filtered too.
|
||||
r <- suppressMessages(
|
||||
cog_spending("010000226085", years = c(2005, 2011), category = "Corrections")
|
||||
)
|
||||
sugg <- attr(r, "provenance")$suggestions
|
||||
ids <- vapply(sugg, function(s) s$recipe_id %||% "", character(1))
|
||||
expect_true("corrections_combined" %in% ids)
|
||||
expect_false("corrections_ig_local_combined" %in% ids)
|
||||
})
|
||||
|
||||
test_that(".attach_ig_counterparts() never pairs a revenue-side recipe with its coincidental M/L suffix twin", {
|
||||
# Broader version of the case above, run at the matching-helper level
|
||||
# (the same level code review's pairwise enumeration was done at) rather
|
||||
# than end-to-end: the fixture has no (govid, year) combination where
|
||||
# cog_revenue() itself produces a covered gap for any B/C/D recipe, so an
|
||||
# end-to-end repro for THIS specific set of recipes isn't reachable
|
||||
# today. Each of these six recipes shares an exact suffix set with an
|
||||
# M/L expenditure recipe purely by reused-digit coincidence:
|
||||
# ig_federal_b47_wide {"47","94"} == ige_local_m47_wide / ige_state_l47_wide
|
||||
# ig_federal_b89_wide {"89","91","92","93"} == ige_local_m89_wide / ige_state_l89_wide
|
||||
# ig_state_c47_wide {"47","94"} == ige_local_m47_wide / ige_state_l47_wide
|
||||
# ig_state_c89_wide {"89","91","92","93"} == ige_local_m89_wide / ige_state_l89_wide
|
||||
# ig_local_d47_wide {"47","94"} == ige_local_m47_wide / ige_state_l47_wide
|
||||
# ig_local_d89_wide {"89","91","92","93"} == ige_local_m89_wide / ige_state_l89_wide
|
||||
# None of them may receive an ig_recipe_id under cog_revenue()'s own
|
||||
# flow_prefixes, since M/L only ever pairs with the direct-expenditure
|
||||
# (E/F/G) family.
|
||||
con <- uscogdata:::.ensure_session()
|
||||
fake_suggestion <- function(rid) {
|
||||
list(recipe_id = rid, label = "x", available_years = c(1967L, 2023L),
|
||||
hint = "h")
|
||||
}
|
||||
fake_suggestions <- lapply(
|
||||
c("ig_federal_b47_wide", "ig_federal_b89_wide",
|
||||
"ig_state_c47_wide", "ig_state_c89_wide",
|
||||
"ig_local_d47_wide", "ig_local_d89_wide"),
|
||||
fake_suggestion
|
||||
)
|
||||
out <- uscogdata:::.attach_ig_counterparts(
|
||||
con, fake_suggestions, c("T", "A", "U", "B", "C", "D")
|
||||
)
|
||||
ig <- unlist(lapply(out, function(s) s$ig_recipe_id))
|
||||
expect_length(ig, 0L)
|
||||
})
|
||||
@@ -31,6 +31,76 @@ test_that("cog_explain errors on non-verb input", {
|
||||
expect_error(cog_explain(df), "provenance")
|
||||
})
|
||||
|
||||
test_that("cog_explain prints basis + harmonization block", {
|
||||
skip_if_no_corpus()
|
||||
r <- cog_spending("121011212191", 2020L, "Corrections")
|
||||
txt <- paste(c(
|
||||
capture.output(cog_explain(r)),
|
||||
capture.output(cog_explain(r), type = "message")
|
||||
), collapse = "\n")
|
||||
expect_true(grepl("Basis: harmonized", txt))
|
||||
expect_true(grepl("Harmonization", txt))
|
||||
expect_true(grepl("Excluded 0 row", txt))
|
||||
})
|
||||
|
||||
test_that("cog_explain prints a Recipe section for recipe = results", {
|
||||
skip_if_no_corpus()
|
||||
r <- cog_spending("121011212191", c(2011L, 2012L), recipe = "corrections_combined")
|
||||
txt <- paste(c(
|
||||
capture.output(cog_explain(r)),
|
||||
capture.output(cog_explain(r), type = "message")
|
||||
), collapse = "\n")
|
||||
expect_true(grepl("Recipe", txt))
|
||||
expect_true(grepl("corrections_combined", txt))
|
||||
expect_true(grepl("E04", txt))
|
||||
expect_true(grepl("E05", txt))
|
||||
})
|
||||
|
||||
test_that("cog_explain prints a Suggestions section when the provenance has one", {
|
||||
skip_if_no_corpus()
|
||||
r <- suppressMessages(
|
||||
cog_spending("121011212191", c(2011L, 2012L), category = "Corrections")
|
||||
)
|
||||
txt <- paste(c(
|
||||
capture.output(cog_explain(r)),
|
||||
capture.output(cog_explain(r), type = "message")
|
||||
), collapse = "\n")
|
||||
expect_true(grepl("Suggestions", txt))
|
||||
expect_true(grepl("corrections_combined", txt))
|
||||
expect_true(grepl("re-run with recipe", txt))
|
||||
})
|
||||
|
||||
test_that("cog_explain prints the expenditure concept (I1)", {
|
||||
skip_if_no_corpus()
|
||||
d <- cog_spending("010000226085", years = 2019, category = "Police")
|
||||
t <- cog_spending("010000226085", years = 2019, category = "Police",
|
||||
expenditure_concept = "total")
|
||||
txt_d <- paste(c(
|
||||
capture.output(cog_explain(d)),
|
||||
capture.output(cog_explain(d), type = "message")
|
||||
), collapse = "\n")
|
||||
txt_t <- paste(c(
|
||||
capture.output(cog_explain(t)),
|
||||
capture.output(cog_explain(t), type = "message")
|
||||
), collapse = "\n")
|
||||
expect_true(grepl("Concept: direct", txt_d))
|
||||
expect_true(grepl("Concept: total", txt_t))
|
||||
})
|
||||
|
||||
test_that("cog_explain surfaces the C1(b) direct-suppressed flag as a warning", {
|
||||
skip_if_no_corpus()
|
||||
t <- suppressMessages(cog_spending(
|
||||
"010000226085", years = 2011, category = "Corrections",
|
||||
expenditure_concept = "total"
|
||||
))
|
||||
expect_true(attr(t, "provenance")$expenditure_concept_direct_suppressed)
|
||||
txt <- paste(c(
|
||||
capture.output(cog_explain(t)),
|
||||
capture.output(cog_explain(t), type = "message")
|
||||
), collapse = "\n")
|
||||
expect_true(grepl("Direct leg unavailable", txt))
|
||||
})
|
||||
|
||||
test_that("cog_explain prints denominator + popyear_range + counts", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
|
||||
@@ -107,6 +107,49 @@ test_that("cog_manifest returns the active session's parsed manifest", {
|
||||
expect_true(m$schema_version >= 4L)
|
||||
yrs <- vapply(m$files$long_partitions, function(p) as.integer(p$year),
|
||||
integer(1))
|
||||
expect_setequal(yrs, c(2019L, 2020L))
|
||||
expect_setequal(yrs, c(2011L, 2012L, 2019L, 2020L))
|
||||
})
|
||||
})
|
||||
|
||||
test_that(".validate_schema accepts schema_version 4, 5 and 6, rejects others", {
|
||||
expect_silent(uscogdata:::.validate_schema(list(schema_version = 4L)))
|
||||
expect_silent(uscogdata:::.validate_schema(list(schema_version = 5L)))
|
||||
# v6 = FIPS geography harmonization (2026-07-22): _code -> _asof rename +
|
||||
# cog_legacy_* columns (26 -> 28 cols). This package references none of the
|
||||
# renamed columns and its geography comes from the xwalk, so v6 is accepted
|
||||
# without behavioural change -- see .validate_schema()'s note.
|
||||
expect_silent(uscogdata:::.validate_schema(list(schema_version = 6L)))
|
||||
expect_error(
|
||||
uscogdata:::.validate_schema(list(schema_version = 3L)),
|
||||
"schema_version"
|
||||
)
|
||||
expect_error(
|
||||
uscogdata:::.validate_schema(list(schema_version = 7L)),
|
||||
"schema_version"
|
||||
)
|
||||
})
|
||||
|
||||
test_that("cog_open succeeds against a doctored schema_version 4 corpus (dual-accept)", {
|
||||
skip_if_no_corpus()
|
||||
with_doctored_schema_version(4L, {
|
||||
con <- cog_open()
|
||||
expect_true(DBI::dbIsValid(con))
|
||||
expect_equal(as.integer(cog_manifest()$schema_version), 4L)
|
||||
|
||||
# Core (pre-Phase-R2) views must still register on a v4 corpus.
|
||||
views <- DBI::dbGetQuery(con,
|
||||
"SELECT table_name FROM information_schema.tables
|
||||
WHERE table_schema = 'main' AND table_type = 'VIEW'"
|
||||
)$table_name
|
||||
expect_true(all(c("spending_annotated", "revenue_annotated") %in% views))
|
||||
|
||||
# Schema-v5-only harmonization views must NOT register on a v4 corpus:
|
||||
# their parquet sources don't exist there and DuckDB's read_parquet()
|
||||
# errors eagerly at CREATE VIEW time for a missing file/glob, so
|
||||
# .register_views() gates these on manifest$schema_version >= 5.
|
||||
expect_false(any(c(
|
||||
"spending_long_harmonized", "spending_annotated_harmonized",
|
||||
"harmonization_recipes", "harmonization_map", "series_breaks_pq"
|
||||
) %in% views))
|
||||
})
|
||||
})
|
||||
|
||||
@@ -0,0 +1,216 @@
|
||||
# tests/testthat/test-recipes.R
|
||||
#
|
||||
# cog_recipes(), recipe = in cog_spending()/cog_revenue(), and the
|
||||
# recipe-component-driven signposting in prov$suggestions (Phase R2 /
|
||||
# Task 11, schema_version 5).
|
||||
|
||||
test_that("cog_recipes lists the curated catalog including corrections_combined", {
|
||||
skip_if_no_corpus()
|
||||
r <- cog_recipes()
|
||||
expect_s3_class(r, "tbl_df")
|
||||
expect_equal(names(r), c("recipe_id", "label", "n_components", "year_min", "year_max"))
|
||||
expect_equal(nrow(r), 24L)
|
||||
expect_true("corrections_combined" %in% r$recipe_id)
|
||||
expect_true("t19_selective_sales_wide" %in% r$recipe_id)
|
||||
expect_true("ig_federal_b89_wide" %in% r$recipe_id)
|
||||
expect_true("rents_royalties_u4_wide" %in% r$recipe_id)
|
||||
expect_true("higher_ed_e18_wide" %in% r$recipe_id)
|
||||
expect_true("cash_securities_z77_wide" %in% r$recipe_id)
|
||||
# Superseded id from the pre-curation brief text must NOT be present.
|
||||
expect_false("corrections_judicial_combined" %in% r$recipe_id)
|
||||
})
|
||||
|
||||
test_that("cog_recipes(pattern=) filters by recipe_id or label", {
|
||||
skip_if_no_corpus()
|
||||
r <- cog_recipes("corrections")
|
||||
expect_true(nrow(r) >= 1L)
|
||||
expect_true(all(grepl("corrections", r$recipe_id, ignore.case = TRUE) |
|
||||
grepl("corrections", r$label, ignore.case = TRUE)))
|
||||
})
|
||||
|
||||
test_that("cog_recipes requires schema_version >= 5", {
|
||||
skip_if_no_corpus()
|
||||
with_doctored_schema_version(4L, {
|
||||
expect_error(cog_recipes(), class = "uscogdata_schema_unsupported")
|
||||
})
|
||||
})
|
||||
|
||||
# --- recipe = : generic join, no is_aggregate filter -----------------------
|
||||
|
||||
test_that("recipe = 'corrections_combined' is continuous across the 2011->2012 seam", {
|
||||
skip_if_no_corpus()
|
||||
r <- cog_spending("121011212191", years = c(2011L, 2012L),
|
||||
recipe = "corrections_combined")
|
||||
expect_equal(nrow(r), 2L)
|
||||
expect_true(all(c("year", "canonical_govid", "gov_name", "spend_subtype",
|
||||
"category", "amt_nominal", "codes_included",
|
||||
"aggregate_fallback", "notes") %in% names(r)))
|
||||
expect_equal(unique(r$spend_subtype), "recipe")
|
||||
expect_equal(unique(r$category), "Corrections (functions 04+05 combined)")
|
||||
expect_false(any(r$aggregate_fallback))
|
||||
|
||||
r2011 <- r$amt_nominal[r$year == 2011L]
|
||||
r2012 <- r$amt_nominal[r$year == 2012L]
|
||||
# 2011: E05 only exists as a wide-era AGGREGATE row (is_aggregate = TRUE)
|
||||
# for Broward -- data-verified $216,088,000. Since .run_recipe() does NOT
|
||||
# filter is_aggregate (amendment: the recipe join must not, because these
|
||||
# families exist ONLY as aggregate rows in the wide era), the recipe
|
||||
# correctly picks this up.
|
||||
expect_equal(r2011, 216088000)
|
||||
# 2012: modern E04 leaf ($213,056,000); Broward reports no E05 leaf that
|
||||
# year, so the recipe total equals E04 alone -- still continuous with the
|
||||
# 2011 aggregate, proving the wide-aggregate -> modern-leaf handoff.
|
||||
expect_equal(r2012, 213056000)
|
||||
expect_true(all(grepl("E04|E05", r$codes_included)))
|
||||
})
|
||||
|
||||
test_that("recipe result carries a recipe provenance block with component rows", {
|
||||
skip_if_no_corpus()
|
||||
r <- cog_spending("121011212191", years = c(2011L, 2012L),
|
||||
recipe = "corrections_combined")
|
||||
prov <- attr(r, "provenance")
|
||||
expect_equal(prov$basis, "recipe")
|
||||
expect_equal(prov$category, "Corrections (functions 04+05 combined)")
|
||||
expect_type(prov$recipe, "list")
|
||||
expect_equal(prov$recipe$recipe_id, "corrections_combined")
|
||||
expect_equal(prov$recipe$label, "Corrections (functions 04+05 combined)")
|
||||
expect_length(prov$recipe$components, 2L)
|
||||
comp_codes <- vapply(prov$recipe$components, function(x) x$component_code, character(1))
|
||||
expect_setequal(comp_codes, c("E04", "E05"))
|
||||
# A recipe query resolves its own coverage; it should never also carry
|
||||
# suggestions for itself.
|
||||
expect_length(prov$suggestions, 0L)
|
||||
})
|
||||
|
||||
test_that("recipe results report an unambiguous basis/harmonization, ignoring basis=", {
|
||||
skip_if_no_corpus()
|
||||
# A recipe query bypasses spending_annotated(_harmonized) entirely --
|
||||
# .run_recipe() joins `long` directly -- so `basis` must never read
|
||||
# "harmonized"/"raw" (which would describe a code path this query never
|
||||
# took) regardless of what the caller passed for `basis`. Task 12
|
||||
# consumes provenance verbatim, so this needs to be unambiguous.
|
||||
r_default <- cog_spending("121011212191", years = c(2011L, 2012L),
|
||||
recipe = "corrections_combined")
|
||||
r_raw <- cog_spending("121011212191", years = c(2011L, 2012L),
|
||||
recipe = "corrections_combined", basis = "raw")
|
||||
r_harm <- cog_spending("121011212191", years = c(2011L, 2012L),
|
||||
recipe = "corrections_combined", basis = "harmonized")
|
||||
|
||||
for (r in list(r_default, r_raw, r_harm)) {
|
||||
prov <- attr(r, "provenance")
|
||||
expect_equal(prov$basis, "recipe")
|
||||
expect_true(is.na(prov$basis_note))
|
||||
expect_false(prov$harmonization$applied)
|
||||
expect_equal(prov$harmonization$na_rows_excluded, 0L)
|
||||
expect_match(prov$harmonization$note, "recipe", ignore.case = TRUE)
|
||||
}
|
||||
|
||||
# basis= truly has zero effect on a recipe query's actual numbers.
|
||||
expect_equal(r_raw$amt_nominal, r_harm$amt_nominal)
|
||||
expect_equal(r_default$amt_nominal, r_raw$amt_nominal)
|
||||
})
|
||||
|
||||
test_that("recipe = 't19_selective_sales_wide' sums the local T11/T14 legs when present", {
|
||||
skip_if_no_corpus()
|
||||
# Westminster City, CA (canonical_govid 082001211654): T11 = 0 in 2011,
|
||||
# T11 = 568 (T14 = 0/absent) in 2012 -- a real, data-verified equality/
|
||||
# inequality pair inside the amended fixture window (2011-2012), standing
|
||||
# in for the brief's original 2004/2005 example (out of scope per the
|
||||
# amended fixture years; the underlying local-tax-split boundary is
|
||||
# nationally FY2005, but this government's own T11 reporting activates
|
||||
# within our 2011-2012 window).
|
||||
r <- cog_revenue("082001211654", years = c(2011L, 2012L),
|
||||
recipe = "t19_selective_sales_wide")
|
||||
|
||||
# Raw, single-code T19 total (not the "Other Taxes" category total, which
|
||||
# would also sum in T11/T14/T21/T23/T27/T29/T53/T99 -- queried directly to
|
||||
# isolate exactly the code the brief's equality/inequality check is about).
|
||||
con <- uscogdata:::.ensure_session()
|
||||
raw_t19 <- DBI::dbGetQuery(con, "
|
||||
SELECT year, SUM(amt) * 1000.0 AS amt
|
||||
FROM revenue_long
|
||||
WHERE canonical_govid = '082001211654' AND item_code = 'T19'
|
||||
AND year IN (2011, 2012)
|
||||
GROUP BY year ORDER BY year
|
||||
")
|
||||
raw_t19_2011 <- raw_t19$amt[raw_t19$year == 2011L]
|
||||
raw_t19_2012 <- raw_t19$amt[raw_t19$year == 2012L]
|
||||
expect_equal(raw_t19_2011, 2231000)
|
||||
expect_equal(raw_t19_2012, 2365000)
|
||||
|
||||
recipe_2011 <- r$amt_nominal[r$year == 2011L]
|
||||
recipe_2012 <- r$amt_nominal[r$year == 2012L]
|
||||
|
||||
expect_equal(recipe_2011, raw_t19_2011) # equality: no local T11/T14 yet
|
||||
expect_gt(recipe_2012, raw_t19_2012) # inequality: local T11 joins in
|
||||
expect_equal(recipe_2012, raw_t19_2012 + 568000)
|
||||
})
|
||||
|
||||
test_that("recipe = and category = together aborts", {
|
||||
skip_if_no_corpus()
|
||||
expect_error(
|
||||
cog_spending("121011212191", 2020L, category = "Corrections",
|
||||
recipe = "corrections_combined"),
|
||||
class = "uscogdata_recipe_category_conflict"
|
||||
)
|
||||
})
|
||||
|
||||
test_that("unknown recipe id aborts and lists valid ids", {
|
||||
skip_if_no_corpus()
|
||||
err <- tryCatch(
|
||||
cog_spending("121011212191", 2020L, recipe = "does_not_exist"),
|
||||
error = identity
|
||||
)
|
||||
expect_s3_class(err, "uscogdata_unknown_recipe")
|
||||
expect_match(conditionMessage(err), "corrections_combined")
|
||||
})
|
||||
|
||||
test_that("recipe = requires schema_version >= 5", {
|
||||
skip_if_no_corpus()
|
||||
with_doctored_schema_version(4L, {
|
||||
expect_error(
|
||||
cog_spending("121011212191", 2020L, recipe = "corrections_combined"),
|
||||
class = "uscogdata_schema_unsupported"
|
||||
)
|
||||
})
|
||||
})
|
||||
|
||||
# --- signposting -------------------------------------------------------
|
||||
|
||||
test_that("signposting suggests corrections_combined across the 2011->2012 gap", {
|
||||
skip_if_no_corpus()
|
||||
expect_message(
|
||||
r <- cog_spending("121011212191", years = c(2011L, 2012L),
|
||||
category = "Corrections"),
|
||||
"recipe"
|
||||
)
|
||||
prov <- attr(r, "provenance")
|
||||
expect_true(length(prov$suggestions) >= 1L)
|
||||
ids <- vapply(prov$suggestions, function(s) s$recipe_id, character(1))
|
||||
expect_true("corrections_combined" %in% ids)
|
||||
hit <- prov$suggestions[[which(ids == "corrections_combined")]]
|
||||
expect_equal(hit$hint, "re-run with recipe = 'corrections_combined'")
|
||||
expect_equal(hit$available_years, c(1967L, 2023L))
|
||||
})
|
||||
|
||||
test_that("no signposting when the result already has full year coverage", {
|
||||
skip_if_no_corpus()
|
||||
r <- cog_spending("121011212191", years = 2019:2020, category = "Corrections")
|
||||
prov <- attr(r, "provenance")
|
||||
expect_length(prov$suggestions, 0L)
|
||||
})
|
||||
|
||||
test_that("no signposting when category is NULL (unscoped query)", {
|
||||
skip_if_no_corpus()
|
||||
r <- cog_spending("121011212191", years = c(2011L, 2012L))
|
||||
prov <- attr(r, "provenance")
|
||||
expect_length(prov$suggestions, 0L)
|
||||
})
|
||||
|
||||
test_that("no signposting under basis = 'raw'", {
|
||||
skip_if_no_corpus()
|
||||
r <- cog_spending("121011212191", years = c(2011L, 2012L),
|
||||
category = "Corrections", basis = "raw")
|
||||
prov <- attr(r, "provenance")
|
||||
expect_length(prov$suggestions, 0L)
|
||||
})
|
||||
@@ -36,3 +36,20 @@ test_that("cog_revenue result has provenance attribute", {
|
||||
test_that("cog_revenue rejects invalid inputs", {
|
||||
expect_error(cog_revenue(list(), 2020L), "character|data frame")
|
||||
})
|
||||
|
||||
test_that("cog_revenue basis = 'harmonized' (default) matches 'raw' in this fixture window", {
|
||||
skip_if_no_corpus()
|
||||
r_raw <- cog_revenue("121011212191", 2019:2020, basis = "raw")
|
||||
r_harm <- cog_revenue("121011212191", 2019:2020, basis = "harmonized")
|
||||
expect_equal(attr(r_raw, "provenance")$basis, "raw")
|
||||
expect_equal(attr(r_harm, "provenance")$basis, "harmonized")
|
||||
expect_equal(sum(r_raw$amt_nominal), sum(r_harm$amt_nominal))
|
||||
})
|
||||
|
||||
test_that("cog_revenue provenance carries the harmonization block", {
|
||||
skip_if_no_corpus()
|
||||
r <- cog_revenue("121011212191", 2020L)
|
||||
h <- attr(r, "provenance")$harmonization
|
||||
expect_true(h$applied)
|
||||
expect_true(h$na_rows_excluded >= 0L)
|
||||
})
|
||||
|
||||
@@ -189,3 +189,134 @@ test_that("provenance records per-year denominator metadata", {
|
||||
expect_equal(length(pc$popyear_range), 2L)
|
||||
})
|
||||
})
|
||||
|
||||
# --- basis = "harmonized" / "raw" (Phase R2, schema v5) --------------------
|
||||
|
||||
test_that("basis = 'raw' reproduces the pre-harmonization Broward Police totals", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
r <- cog_spending("121011212191", years = 2019:2020, category = "Police",
|
||||
basis = "raw")
|
||||
# Regression pin captured against the schema v5 fixture (2026-07-18,
|
||||
# pipeline_commit ece9b32) before basis = "harmonized" existed as a
|
||||
# concept; these are the same totals the pre-Phase-R2 default query
|
||||
# returned (spending_annotated is untouched by the harmonized views).
|
||||
ops <- r$amt_nominal[r$year == 2019L & r$spend_subtype == "operations"]
|
||||
cap <- r$amt_nominal[r$year == 2020L & r$spend_subtype == "capital"]
|
||||
expect_equal(ops, 483560000)
|
||||
expect_equal(cap, 26693000)
|
||||
expect_equal(attr(r, "provenance")$basis, "raw")
|
||||
})
|
||||
})
|
||||
|
||||
test_that("basis = 'harmonized' (default) matches 'raw' when no harmonization rule applies", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
# Every `method = "collapse"` mapping in the curated harmonization_map
|
||||
# ends by FY2004 for codes inside the spending/revenue flow-type
|
||||
# prefixes (E/F/G/K, T/A/U/B/C/D); the one collapse extending to FY2011
|
||||
# (L38/M38 -> L36/M36) is intergovernmental-transfer (L/M prefix) codes
|
||||
# that were never part of spending_long/revenue_long to begin with. So
|
||||
# for the fixture's 2011-2020 window, basis = "harmonized" is a
|
||||
# data-verified no-op vs "raw" for in-scope codes -- this is the
|
||||
# positive-control counterpart to the synthetic REPLACE-mechanism test
|
||||
# in test-views.R, which proves the fold itself works when data exists.
|
||||
r_raw <- cog_spending("121011212191", c(2011L, 2012L, 2019L, 2020L),
|
||||
"Police", basis = "raw")
|
||||
r_harm <- cog_spending("121011212191", c(2011L, 2012L, 2019L, 2020L),
|
||||
"Police", basis = "harmonized")
|
||||
expect_equal(attr(r_harm, "provenance")$basis, "harmonized")
|
||||
expect_equal(
|
||||
r_harm$amt_nominal[order(r_harm$year, r_harm$spend_subtype)],
|
||||
r_raw$amt_nominal[order(r_raw$year, r_raw$spend_subtype)]
|
||||
)
|
||||
})
|
||||
})
|
||||
|
||||
test_that("basis defaults to 'harmonized' when not passed", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
r <- cog_spending("121011212191", 2020L, "Police")
|
||||
expect_equal(attr(r, "provenance")$basis, "harmonized")
|
||||
})
|
||||
})
|
||||
|
||||
test_that("provenance carries basis + harmonization block with na_rows_excluded", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
r <- cog_spending("121011212191", 2011:2012, "Corrections")
|
||||
prov <- attr(r, "provenance")
|
||||
expect_equal(prov$basis, "harmonized")
|
||||
expect_true(prov$harmonization$applied)
|
||||
expect_true(prov$harmonization$na_rows_excluded >= 0L)
|
||||
expect_true(prov$harmonization$na_amount_excluded >= 0)
|
||||
# Data-verified for the v6 fixture (corpus 2026-07-22). The Task 18 map
|
||||
# extension added E/F/G-prefix discontinued_na rulings the earlier pin's
|
||||
# comment predated: E21/F21/G21 (Education NEC local, SB184-186,
|
||||
# "trivial; explicit-NA, full wide-era window"). Broward's 2011 legacy
|
||||
# partition zero-pads exactly those three codes, so this query now
|
||||
# excludes 3 NA-harmonized rows -- all with amt = 0, hence the excluded
|
||||
# AMOUNT stays exactly zero. (The other discontinued_na rulings -- S74,
|
||||
# Z61, X04, X06, the debt-detail family, L24 -- remain outside the
|
||||
# E/F/G/K prefixes.) See docs/phase_r_harmonization_review.md § 1.3/1.4
|
||||
# and cog_pipeline data/harmonization_map.csv E21/F21/G21 rows.
|
||||
expect_equal(prov$harmonization$na_rows_excluded, 3L)
|
||||
expect_equal(prov$harmonization$na_amount_excluded, 0)
|
||||
})
|
||||
})
|
||||
|
||||
test_that("basis = 'raw' never populates the harmonization exclusion block", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
r <- cog_spending("121011212191", 2020L, "Corrections", basis = "raw")
|
||||
h <- attr(r, "provenance")$harmonization
|
||||
expect_false(h$applied)
|
||||
expect_equal(h$na_rows_excluded, 0L)
|
||||
})
|
||||
})
|
||||
|
||||
test_that("v4 corpus: basis silently resolves to raw (default) with a provenance note", {
|
||||
skip_if_no_corpus()
|
||||
with_doctored_schema_version(4L, {
|
||||
r <- cog_spending("121011212191", 2019L, "Police")
|
||||
prov <- attr(r, "provenance")
|
||||
expect_equal(prov$basis, "raw")
|
||||
expect_match(prov$basis_note, "raw", fixed = TRUE)
|
||||
expect_match(prov$basis_note, "schema_version", fixed = TRUE)
|
||||
expect_false(prov$harmonization$applied)
|
||||
})
|
||||
})
|
||||
|
||||
test_that("v4 corpus: explicit basis = 'harmonized' aborts", {
|
||||
skip_if_no_corpus()
|
||||
with_doctored_schema_version(4L, {
|
||||
expect_error(
|
||||
cog_spending("121011212191", 2019L, "Police", basis = "harmonized"),
|
||||
class = "uscogdata_basis_unsupported"
|
||||
)
|
||||
})
|
||||
})
|
||||
|
||||
test_that("provenance$series_break_refs is a populated-when-applicable character vector", {
|
||||
skip_if_no_corpus()
|
||||
with_fixture_corpus({
|
||||
r <- cog_spending("121011212191", 2020L, "Corrections")
|
||||
refs <- attr(r, "provenance")$series_break_refs
|
||||
expect_type(refs, "character")
|
||||
# No catalogued series_breaks_pq row falls inside this fixture's
|
||||
# 2011/2012/2019/2020 window for the codes this query touches (E04/G04)
|
||||
# -- data-verified; the mechanism itself is what's under test here, via
|
||||
# a query-shaped unit test in test-views.R since the fixture has no
|
||||
# positive case to pin against.
|
||||
expect_equal(refs, character(0))
|
||||
})
|
||||
})
|
||||
|
||||
test_that("v4 corpus: explicit basis = 'raw' still works", {
|
||||
skip_if_no_corpus()
|
||||
with_doctored_schema_version(4L, {
|
||||
r <- cog_spending("121011212191", 2019L, "Police", basis = "raw")
|
||||
expect_equal(attr(r, "provenance")$basis, "raw")
|
||||
expect_gt(nrow(r), 0L)
|
||||
})
|
||||
})
|
||||
|
||||
+315
-7
@@ -9,11 +9,317 @@ test_that("all expected views register on session open", {
|
||||
expected <- c(
|
||||
"long", "spending_long", "revenue_long",
|
||||
"canonical_fips_xwalk", "summary_categories",
|
||||
"spending_annotated", "revenue_annotated"
|
||||
"spending_annotated", "revenue_annotated",
|
||||
"ig_long", "ig_annotated",
|
||||
"ig_long_harmonized", "ig_annotated_harmonized"
|
||||
)
|
||||
expect_true(all(expected %in% views$table_name))
|
||||
})
|
||||
|
||||
test_that("inst/sql/22- and 23- harmonized views enforce every WHERE predicate (real SQL text, synthetic parquet)", {
|
||||
# spending_long_harmonized / revenue_long_harmonized are three-predicate
|
||||
# views:
|
||||
# SELECT * REPLACE (harmonized_code AS item_code)
|
||||
# FROM long
|
||||
# WHERE NOT is_aggregate
|
||||
# AND harmonized_code IS NOT NULL
|
||||
# AND LEFT(harmonized_code, 1) IN (<flow prefixes>)
|
||||
# None of the curated harmonization_map's `collapse` rulings land inside
|
||||
# the bundled fixture's 2011-2020 window for spending/revenue-prefixed
|
||||
# codes (see the "basis = 'harmonized' (default) matches 'raw'" test in
|
||||
# test-spending.R and docs/phase_r_harmonization_review.md § 0.2/§ 2), so
|
||||
# there is no real fixture row that exercises a nonzero fold or a
|
||||
# predicate-excluded row. Rather than re-implement the WHERE clause by
|
||||
# hand against an in-memory VALUES table (which would only prove the SQL
|
||||
# *pattern* works, not that the deployed inst/sql/22-/23- text actually
|
||||
# applies it), this test reads the real SQL files off disk, substitutes
|
||||
# {url} exactly as .register_views() does, and executes them -- plus
|
||||
# their 10-long.sql dependency -- against a synthetic hive-partitioned
|
||||
# parquet tree written to a temp dir. A regression in any predicate (e.g.
|
||||
# `NOT is_aggregate` dropped, the prefix list changed, the NULL guard
|
||||
# removed) would change which of the rows below survive.
|
||||
#
|
||||
# The synthetic parquet is written with DuckDB's own COPY ... TO (FORMAT
|
||||
# PARQUET) rather than the arrow package: this package has no arrow
|
||||
# dependency (CLAUDE.md "No arrow dependency -- DuckDB reads parquet
|
||||
# natively"), and DuckDB can round-trip its own parquet writer/reader
|
||||
# without adding one for tests either.
|
||||
skip_if_no_corpus()
|
||||
|
||||
tmp <- withr::local_tempdir()
|
||||
part_dir <- file.path(tmp, "data", "long", "year=2004")
|
||||
dir.create(part_dir, recursive = TRUE)
|
||||
part_path <- file.path(part_dir, "part-0.parquet")
|
||||
|
||||
write_con <- DBI::dbConnect(duckdb::duckdb())
|
||||
on.exit(DBI::dbDisconnect(write_con, shutdown = TRUE), add = TRUE)
|
||||
DBI::dbExecute(write_con, sprintf("
|
||||
COPY (
|
||||
SELECT * FROM (VALUES
|
||||
-- Spending (E/F/G/K) rows, exercised against spending_long_harmonized:
|
||||
('spend-A', 'E36', 100, false, 'E36'), -- control: passes every predicate as-is
|
||||
('spend-B', 'E38', 50, false, 'E36'), -- collapse-fold: passes every predicate, renamed to E36
|
||||
('spend-C', 'E05', 999999, true, 'E05'), -- excluded ONLY by `NOT is_aggregate`
|
||||
('spend-D', 'E99', 888888, false, NULL), -- excluded by `harmonized_code IS NOT NULL`
|
||||
-- 'S74' is outside BOTH flow families (E/F/G/K spending and
|
||||
-- T/A/U/B/C/D revenue -- it mirrors the real corpus's own
|
||||
-- non-flow-type codes like S74/Z61), so it can only leak into
|
||||
-- EITHER view via the E/F/G/K or T/A/U/B/C/D prefix filter, never
|
||||
-- both at once -- a prefix drawn from the other view's own family
|
||||
-- (e.g. a real T-code for the spending row) would incorrectly
|
||||
-- leak into the other view's assertion below and not discriminate
|
||||
-- the predicate under test.
|
||||
('spend-E', 'S74', 777777, false, 'S74'), -- excluded ONLY by the E/F/G/K prefix filter
|
||||
-- Revenue (T/A/U/B/C/D) rows, exercised against revenue_long_harmonized:
|
||||
('rev-A', 'U11', 200, false, 'U11'), -- control: passes every predicate as-is
|
||||
('rev-B', 'U10', 25, false, 'U11'), -- collapse-fold: passes every predicate, renamed to U11
|
||||
('rev-C', 'T29', 555555, true, 'T29'), -- excluded ONLY by `NOT is_aggregate`
|
||||
('rev-D', 'T88', 444444, false, NULL), -- excluded by `harmonized_code IS NOT NULL`
|
||||
('rev-E', 'Z61', 333333, false, 'Z61') -- excluded ONLY by the T/A/U/B/C/D prefix filter
|
||||
) AS t(canonical_govid, item_code, amt, is_aggregate, harmonized_code)
|
||||
) TO %s (FORMAT PARQUET)
|
||||
", uscogdata:::.sql_lit_chr(part_path)))
|
||||
|
||||
sql_dir <- system.file("sql", package = "uscogdata")
|
||||
.read_view_sql <- function(filename) {
|
||||
txt <- paste(readLines(file.path(sql_dir, filename), warn = FALSE), collapse = "\n")
|
||||
gsub("\\{url\\}", paste0(tmp, "/"), txt, fixed = FALSE)
|
||||
}
|
||||
|
||||
con <- DBI::dbConnect(duckdb::duckdb())
|
||||
on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE)
|
||||
DBI::dbExecute(con, .read_view_sql("10-long.sql"))
|
||||
DBI::dbExecute(con, .read_view_sql("22-spending_long_harmonized.sql"))
|
||||
DBI::dbExecute(con, .read_view_sql("23-revenue_long_harmonized.sql"))
|
||||
|
||||
spend <- DBI::dbGetQuery(con,
|
||||
"SELECT item_code, SUM(amt) AS amt FROM spending_long_harmonized
|
||||
GROUP BY item_code ORDER BY item_code"
|
||||
)
|
||||
# Exactly one surviving row: spend-C (aggregate), spend-D (NULL
|
||||
# harmonized_code), and spend-E (wrong prefix family) must all be gone,
|
||||
# and spend-A + spend-B must be folded together under E36.
|
||||
expect_equal(nrow(spend), 1L)
|
||||
expect_equal(spend$item_code, "E36")
|
||||
expect_equal(spend$amt, 150)
|
||||
|
||||
rev <- DBI::dbGetQuery(con,
|
||||
"SELECT item_code, SUM(amt) AS amt FROM revenue_long_harmonized
|
||||
GROUP BY item_code ORDER BY item_code"
|
||||
)
|
||||
expect_equal(nrow(rev), 1L)
|
||||
expect_equal(rev$item_code, "U11")
|
||||
expect_equal(rev$amt, 225)
|
||||
})
|
||||
|
||||
test_that("inst/sql/24- and 25- IG views retain aggregates, COALESCE NULL harmonized_code, and exclude the L-- family total (real SQL text, synthetic parquet)", {
|
||||
# ig_long / ig_long_harmonized have the subtlest predicates in the package:
|
||||
# a deliberately ABSENT `NOT is_aggregate` (unlike every other *_long view),
|
||||
# and COALESCE(harmonized_code, item_code) instead of a plain
|
||||
# `harmonized_code IS NOT NULL` filter. The only end-to-end guard on this
|
||||
# today is bound to AL state / 2011 / Education K-12, where M12 happens to
|
||||
# be the sole IG code present -- regenerate the fixture without that one
|
||||
# row and the guard would die silently while staying green. As with the
|
||||
# 22-/23- test above, this reads the real inst/sql/24-/25- text off disk
|
||||
# and executes it against a synthetic hive-partitioned parquet tree, so a
|
||||
# regression in either predicate changes which rows survive.
|
||||
skip_if_no_corpus()
|
||||
|
||||
tmp <- withr::local_tempdir()
|
||||
part_dir <- file.path(tmp, "data", "long", "year=2004")
|
||||
dir.create(part_dir, recursive = TRUE)
|
||||
part_path <- file.path(part_dir, "part-0.parquet")
|
||||
|
||||
write_con <- DBI::dbConnect(duckdb::duckdb())
|
||||
on.exit(DBI::dbDisconnect(write_con, shutdown = TRUE), add = TRUE)
|
||||
DBI::dbExecute(write_con, sprintf("
|
||||
COPY (
|
||||
SELECT * FROM (VALUES
|
||||
('ig-A', 'M04', 100, false, 'M04'), -- control: passes through as-is
|
||||
('ig-B', 'M38', 50, false, 'M36'), -- fold control: real SB012 rule, renamed to M36 under harmonized basis
|
||||
('ig-C', 'M47', 99999, true, NULL), -- legacy aggregate, NO harmonized_code: must survive BOTH views
|
||||
('ig-D', 'L--', 55555, false, 'L--'), -- family total: excluded from BOTH views
|
||||
('ig-E', 'T29', 44444, false, 'T29') -- wrong prefix (revenue, not M/L): excluded from BOTH views
|
||||
) AS t(canonical_govid, item_code, amt, is_aggregate, harmonized_code)
|
||||
) TO %s (FORMAT PARQUET)
|
||||
", uscogdata:::.sql_lit_chr(part_path)))
|
||||
|
||||
sql_dir <- system.file("sql", package = "uscogdata")
|
||||
.read_view_sql <- function(filename) {
|
||||
txt <- paste(readLines(file.path(sql_dir, filename), warn = FALSE), collapse = "\n")
|
||||
gsub("\\{url\\}", paste0(tmp, "/"), txt, fixed = FALSE)
|
||||
}
|
||||
|
||||
con <- DBI::dbConnect(duckdb::duckdb())
|
||||
on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE)
|
||||
DBI::dbExecute(con, .read_view_sql("10-long.sql"))
|
||||
DBI::dbExecute(con, .read_view_sql("24-ig_long.sql"))
|
||||
DBI::dbExecute(con, .read_view_sql("25-ig_long_harmonized.sql"))
|
||||
|
||||
raw <- DBI::dbGetQuery(con,
|
||||
"SELECT item_code, SUM(amt) AS amt FROM ig_long
|
||||
GROUP BY item_code ORDER BY item_code"
|
||||
)
|
||||
# L-- (family total) and T29 (wrong prefix) are gone; the aggregate row
|
||||
# M47 survives -- proof `NOT is_aggregate` is absent from ig_long.
|
||||
expect_equal(raw$item_code, c("M04", "M38", "M47"))
|
||||
expect_equal(raw$amt, c(100, 50, 99999))
|
||||
|
||||
harmonized <- DBI::dbGetQuery(con,
|
||||
"SELECT item_code, SUM(amt) AS amt FROM ig_long_harmonized
|
||||
GROUP BY item_code ORDER BY item_code"
|
||||
)
|
||||
# M38 folds to M36 (real harmonized_code present); M47 keeps its raw code
|
||||
# via COALESCE(NULL, 'M47') -- proof the aggregate row is NOT dropped by
|
||||
# a plain `harmonized_code IS NOT NULL` filter. L-- and T29 stay excluded.
|
||||
expect_equal(harmonized$item_code, c("M04", "M36", "M47"))
|
||||
expect_equal(harmonized$amt, c(100, 50, 99999))
|
||||
})
|
||||
|
||||
test_that(".build_series_break_refs matches fin_code + break_year window", {
|
||||
# No series_breaks_pq row falls inside the bundled fixture's 2011-2020
|
||||
# window (data-verified; see the "series_break_refs" test in
|
||||
# test-spending.R), so this proves the matching logic itself against the
|
||||
# live view + a synthetic year window that DOES hit a cataloged break
|
||||
# (SB075, fin_code E62, break_year 2005).
|
||||
skip_if_no_corpus()
|
||||
con <- cog_open()
|
||||
on.exit(cog_close())
|
||||
refs <- uscogdata:::.build_series_break_refs(
|
||||
con, codes_observed = c("E62", "E04"), years = c(2003L, 2006L),
|
||||
schema_version = 5L
|
||||
)
|
||||
expect_true("SB075" %in% refs)
|
||||
expect_true("SB071" %in% refs)
|
||||
|
||||
# Gated on schema_version >= 5 even when the codes/years would otherwise match.
|
||||
refs_v4 <- uscogdata:::.build_series_break_refs(
|
||||
con, codes_observed = c("E62"), years = c(2003L, 2006L), schema_version = 4L
|
||||
)
|
||||
expect_equal(refs_v4, character(0))
|
||||
})
|
||||
|
||||
test_that("schema v5 harmonization views register when the corpus supports them", {
|
||||
skip_if_no_corpus()
|
||||
con <- cog_open()
|
||||
on.exit(cog_close())
|
||||
manifest <- uscogdata:::.uscogdata_env$manifest
|
||||
skip_if(as.integer(manifest$schema_version) < 5L, "fixture is schema_version < 5")
|
||||
|
||||
views <- DBI::dbGetQuery(con,
|
||||
"SELECT table_name FROM information_schema.tables
|
||||
WHERE table_schema = 'main' AND table_type = 'VIEW'"
|
||||
)
|
||||
expected_v5 <- c(
|
||||
"spending_long_harmonized", "revenue_long_harmonized",
|
||||
"spending_annotated_harmonized", "revenue_annotated_harmonized",
|
||||
"harmonization_map", "harmonization_recipes", "series_breaks_pq"
|
||||
)
|
||||
expect_true(all(expected_v5 %in% views$table_name))
|
||||
})
|
||||
|
||||
test_that(".harmonization_view_files guard is necessary: registration against a v4-shaped corpus (no harmonized_code column at all) succeeds only because the harmonized views are skipped", {
|
||||
# with_doctored_schema_version() (used elsewhere in this suite) only
|
||||
# rewrites manifest.json's schema_version -- the underlying `long` parquet
|
||||
# is still the bundled v6 fixture, which DOES have a harmonized_code
|
||||
# column, so it only proves the skip *happens*, not that it is *required*.
|
||||
# This test builds a genuinely v4-shaped corpus: `long` has no
|
||||
# harmonized_code column at all, matching a real pre-Phase-R2 publish
|
||||
# tree, and then shows two things: (1) the real .register_views(), gated
|
||||
# on manifest$schema_version, registers cleanly against it; (2) the exact
|
||||
# SQL text of a gated file (25-ig_long_harmonized.sql), executed directly
|
||||
# against the same corpus without the gate, fails -- proving the gate is
|
||||
# load-bearing, not incidental.
|
||||
tmp <- withr::local_tempdir()
|
||||
part_dir <- file.path(tmp, "data", "long", "year=2004")
|
||||
dir.create(part_dir, recursive = TRUE)
|
||||
part_path <- file.path(part_dir, "part-0.parquet")
|
||||
|
||||
write_con <- DBI::dbConnect(duckdb::duckdb())
|
||||
on.exit(DBI::dbDisconnect(write_con, shutdown = TRUE), add = TRUE)
|
||||
DBI::dbExecute(write_con, sprintf("
|
||||
COPY (
|
||||
SELECT * FROM (VALUES
|
||||
('gov-1', 'E36', 100, false, 500000, 2020)
|
||||
) AS t(canonical_govid, item_code, amt, is_aggregate, population, popyear)
|
||||
) TO %s (FORMAT PARQUET)
|
||||
", uscogdata:::.sql_lit_chr(part_path)))
|
||||
|
||||
xwalk_path <- file.path(tmp, "data", "canonical_fips_xwalk.parquet")
|
||||
DBI::dbExecute(write_con, sprintf("
|
||||
COPY (
|
||||
SELECT * FROM (VALUES
|
||||
('gov-1', 'Test Gov', 1, 'County', '01', '001', NULL, 500000)
|
||||
) AS t(canonical_govid, gov_name, govs_type, type_label, fips_state,
|
||||
fips_county, fips_place, population_acs)
|
||||
) TO %s (FORMAT PARQUET)
|
||||
", uscogdata:::.sql_lit_chr(xwalk_path)))
|
||||
|
||||
cats_path <- file.path(tmp, "data", "summary_categories.parquet")
|
||||
DBI::dbExecute(write_con, sprintf("
|
||||
COPY (
|
||||
SELECT * FROM (VALUES
|
||||
('E36', 'Test Category', 'expenditure', 'direct', NULL)
|
||||
) AS t(item_code, category, category_type, spend_subtype, revenue_subtype)
|
||||
) TO %s (FORMAT PARQUET)
|
||||
", uscogdata:::.sql_lit_chr(cats_path)))
|
||||
|
||||
# Confirm the synthetic `long` genuinely lacks harmonized_code (not just
|
||||
# NULL values -- the column itself must be absent) before trusting the
|
||||
# rest of this test.
|
||||
cols <- DBI::dbGetQuery(write_con, sprintf(
|
||||
"DESCRIBE SELECT * FROM read_parquet(%s)", uscogdata:::.sql_lit_chr(part_path)
|
||||
))$column_name
|
||||
expect_false("harmonized_code" %in% cols)
|
||||
|
||||
url <- paste0(tmp, "/")
|
||||
|
||||
# (1) Full .register_views() against this v4-shaped corpus must succeed --
|
||||
# this is the behavior the guard exists to protect.
|
||||
con <- DBI::dbConnect(duckdb::duckdb())
|
||||
on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE)
|
||||
expect_no_error(
|
||||
uscogdata:::.register_views(con, url, manifest = list(schema_version = 4L))
|
||||
)
|
||||
views <- DBI::dbGetQuery(con,
|
||||
"SELECT table_name FROM information_schema.tables
|
||||
WHERE table_schema = 'main' AND table_type = 'VIEW'")$table_name
|
||||
expect_true(all(c("ig_long", "ig_annotated", "spending_annotated") %in% views))
|
||||
expect_false(any(c("ig_long_harmonized", "ig_annotated_harmonized",
|
||||
"spending_long_harmonized") %in% views))
|
||||
|
||||
# (2) Prove the gate is load-bearing: the exact SQL text of the skipped
|
||||
# file, executed directly (bypassing .register_views()'s schema_version
|
||||
# check) against the SAME corpus, fails because it references
|
||||
# long.harmonized_code, a column this corpus's `long` does not have.
|
||||
sql_dir <- system.file("sql", package = "uscogdata")
|
||||
.read_view_sql <- function(filename) {
|
||||
txt <- paste(readLines(file.path(sql_dir, filename), warn = FALSE), collapse = "\n")
|
||||
gsub("\\{url\\}", url, txt, fixed = FALSE)
|
||||
}
|
||||
con2 <- DBI::dbConnect(duckdb::duckdb())
|
||||
on.exit(DBI::dbDisconnect(con2, shutdown = TRUE), add = TRUE)
|
||||
DBI::dbExecute(con2, .read_view_sql("10-long.sql"))
|
||||
expect_error(DBI::dbExecute(con2, .read_view_sql("25-ig_long_harmonized.sql")))
|
||||
|
||||
# Reconciling this test with the C2 guard (expenditure-concept review):
|
||||
# `ig_annotated`/`spending_annotated` registering cleanly above proves
|
||||
# only that CREATE VIEW binds against a `summary_categories` with no M/L
|
||||
# rows at all (this synthetic corpus's own summary_categories has a
|
||||
# single E36 row, see the COPY above) -- a LEFT JOIN never fails to
|
||||
# resolve regardless of what the joined-to table contains. It does NOT
|
||||
# mean querying expenditure_concept = "total" against this shape is safe:
|
||||
# exactly this corpus (schema_version reported as supported, but
|
||||
# summary_categories predates the M/L rows cog_pipeline PR #59 added) is
|
||||
# what .require_ig_categories() exists to catch at the *verb* level,
|
||||
# since PR #59 shipped those rows with no schema_version bump. Confirm
|
||||
# the new runtime guard actually fires against this same `con`.
|
||||
expect_error(
|
||||
uscogdata:::.require_ig_categories(con),
|
||||
class = "uscogdata_ig_categories_unsupported"
|
||||
)
|
||||
})
|
||||
|
||||
test_that("spending_long filters to E/F/G/K prefixes and excludes aggregates", {
|
||||
skip_if_no_corpus()
|
||||
con <- cog_open()
|
||||
@@ -69,13 +375,15 @@ test_that("gov_population_yearly exposes one row per (year, canonical_govid)", {
|
||||
WHERE canonical_govid = '121011212191'
|
||||
ORDER BY year"
|
||||
)
|
||||
expect_setequal(df$year, c(2019L, 2020L))
|
||||
expect_equal(nrow(df), 2L)
|
||||
expect_setequal(df$year, c(2011L, 2012L, 2019L, 2020L))
|
||||
expect_equal(nrow(df), 4L)
|
||||
expect_true(all(!is.na(df$population)))
|
||||
# Hardcoded values are from the bundled fixture (regenerated 2026-07-11
|
||||
# against cog_pipeline publish tree, pipeline_commit 1a00925, Phase P
|
||||
# schema_version 4). Update if the fixture is rebuilt against a
|
||||
# different source vintage.
|
||||
# Hardcoded values are from the bundled fixture (regenerated 2026-07-18
|
||||
# against cog_pipeline publish tree, pipeline_commit ece9b32, Phase R2
|
||||
# schema_version 5, years 2011/2012/2019/2020). Update if the fixture is
|
||||
# rebuilt against a different source vintage.
|
||||
expect_equal(df$population[df$year == 2011L], 1759591L)
|
||||
expect_equal(df$population[df$year == 2012L], 1819773L)
|
||||
expect_equal(df$population[df$year == 2019L], 1935878L)
|
||||
expect_equal(df$population[df$year == 2020L], 1952778L)
|
||||
# Uniqueness on (year, canonical_govid) across the whole view.
|
||||
|
||||
@@ -0,0 +1,229 @@
|
||||
---
|
||||
title: "Total spending: Direct, Total, and when each is right"
|
||||
output: rmarkdown::html_vignette
|
||||
vignette: >
|
||||
%\VignetteIndexEntry{Total spending: Direct, Total, and when each is right}
|
||||
%\VignetteEngine{knitr::rmarkdown}
|
||||
%\VignetteEncoding{UTF-8}
|
||||
---
|
||||
|
||||
```{r setup, include = FALSE}
|
||||
knitr::opts_chunk$set(collapse = TRUE, comment = "#>")
|
||||
```
|
||||
|
||||
# Two questions that sound the same but aren't
|
||||
|
||||
"Total spending" means two different things depending on whether the question
|
||||
is about one government or several:
|
||||
|
||||
1. **"What did my county spend in total, a decade ago vs today?"** — one
|
||||
government, tracked over time. Either `direct` or `total` spending answers
|
||||
this correctly, as long as the same concept is used for both years.
|
||||
2. **"How do all the counties in my state compare, a decade ago vs today,
|
||||
against the neighboring state?"** — several governments, summed together.
|
||||
Here only `direct` gives the right answer; summing `total` across
|
||||
governments double-counts money that passes between them.
|
||||
|
||||
`cog_spending()`'s `expenditure_concept` argument (`"direct"` or `"total"`)
|
||||
controls which of these a query answers. This vignette walks through both
|
||||
questions with code that actually runs against the package's bundled fixture
|
||||
corpus, then explains why the second question refuses `"total"` outright.
|
||||
|
||||
```{r}
|
||||
library(uscogdata)
|
||||
|
||||
# Point at the bundled offline fixture (years 2011, 2012, 2019, 2020, all 50
|
||||
# states) so this vignette knits without network access. In real use,
|
||||
# USCOGDATA_URL is instead set to the published corpus URL -- see README.md.
|
||||
Sys.setenv(USCOGDATA_URL = paste0(
|
||||
system.file("extdata/fixture_corpus", package = "uscogdata"), "/"
|
||||
))
|
||||
```
|
||||
|
||||
The fixture doesn't carry 2017 or the present year, so the examples below use
|
||||
the closest years it does ship -- **2012 and 2020** -- in place of "2017 vs
|
||||
today" / "ten years ago vs today". Point `USCOGDATA_URL` at the published
|
||||
corpus and swap in real years; the mechanics are identical.
|
||||
|
||||
# Archetype 1: one government's own trend
|
||||
|
||||
For a single government, `total` is a legitimate way to describe "everything
|
||||
this government spent, including money it handed to other governments to
|
||||
spend on its behalf":
|
||||
|
||||
```{r}
|
||||
al_total <- cog_spending(
|
||||
"010000226085", # Alabama, the state government
|
||||
years = c(2012, 2020),
|
||||
category = "Highways",
|
||||
expenditure_concept = "total"
|
||||
)
|
||||
al_total
|
||||
```
|
||||
|
||||
The `intergovernmental` rows are what `"total"` adds on top of `"direct"`
|
||||
(`capital` + `operations`): Alabama's own payments out to counties and
|
||||
cities for highway work. Because this query only ever concerns Alabama,
|
||||
including that piece is safe -- there's no other government's number it
|
||||
could be double-counted against.
|
||||
|
||||
`"direct"` (the default) answers the same trend question just as validly:
|
||||
|
||||
```{r}
|
||||
al_direct <- cog_spending(
|
||||
"010000226085", years = c(2012, 2020), category = "Highways"
|
||||
# expenditure_concept = "direct" is the default; shown here for contrast
|
||||
)
|
||||
al_direct
|
||||
```
|
||||
|
||||
Both are internally consistent series. What breaks the comparison is
|
||||
**switching concepts between the two years being compared** -- e.g. `direct`
|
||||
for 2012 and `total` for 2020 -- which manufactures a trend that isn't
|
||||
really there. Pick one concept for a given question and hold it fixed across
|
||||
every year in the series.
|
||||
|
||||
# Archetype 2: a cross-government rollup
|
||||
|
||||
`cog_geographic_rollup()` sums spending across state/county/city layers for
|
||||
a place. Its default -- and, as shown below, its *only* accepted value for
|
||||
`expenditure_concept` -- is `"direct"`:
|
||||
|
||||
```{r}
|
||||
fl_rollup <- cog_geographic_rollup(
|
||||
govids = list(
|
||||
state = "120000226351", # Florida
|
||||
county = c("121011212191", "121099101897") # Broward + Palm Beach
|
||||
),
|
||||
category = "Highways",
|
||||
years = c(2012, 2020)
|
||||
)
|
||||
fl_rollup
|
||||
```
|
||||
|
||||
For the neighboring state, the comparison is a single government, so it's a
|
||||
plain `cog_spending()` call rather than a rollup:
|
||||
|
||||
```{r}
|
||||
ga_state <- cog_spending(
|
||||
"130000226087", years = c(2012, 2020), category = "Highways" # Georgia
|
||||
)
|
||||
ga_state
|
||||
```
|
||||
|
||||
Now the same rollup, but asking for `expenditure_concept = "total"`:
|
||||
|
||||
```{r, error = TRUE}
|
||||
cog_geographic_rollup(
|
||||
govids = list(state = "120000226351", county = "121011212191"),
|
||||
category = "Highways",
|
||||
years = 2020,
|
||||
expenditure_concept = "total"
|
||||
)
|
||||
```
|
||||
|
||||
`cog_geographic_rollup()` (and `cog_peer_compare()`, for the same reason)
|
||||
refuses `"total"` outright rather than silently returning an inflated
|
||||
number. The next section is why.
|
||||
|
||||
# The mechanism
|
||||
|
||||
Suppose Alabama gives a county $10M toward a highway project. That $10M
|
||||
shows up **twice** in the underlying corpus:
|
||||
|
||||
- Once on Alabama's own record, coded `M44` ("to local governments,
|
||||
Highways") -- Alabama's intergovernmental leg.
|
||||
- Again on the county's record, coded `E44` / `F44` ("Highways, current
|
||||
operations" / "capital outlay") -- the county's direct spending, because
|
||||
the county is the government that actually lets the contract and pays the
|
||||
paving crew.
|
||||
|
||||
`direct` (item codes `E`/`F`/`G`) only ever counts the second of those --
|
||||
the government that actually did the spending. `total` (Direct plus the
|
||||
`M`/`L` intergovernmental legs) counts the first one *as well*, which is
|
||||
exactly right for describing Alabama's own budget: Alabama's `total`
|
||||
genuinely includes the $10M it committed to highways, whether it built the
|
||||
road itself or paid the county to. But sum `total` across Alabama **and**
|
||||
the county, and that $10M is counted twice -- once as Alabama's payment out,
|
||||
once as the county's spending in -- reporting $20M of highway work for $10M
|
||||
actually spent.
|
||||
|
||||
This is exactly the shape of query `cog_geographic_rollup()` exists to run
|
||||
(summing across layers of government), so it refuses `"total"` rather than
|
||||
silently overstating every multi-layer figure it produces.
|
||||
|
||||
# How big is the risk in practice
|
||||
|
||||
Intergovernmental transfers aren't evenly distributed by government type.
|
||||
Measured against the bundled fixture corpus (all 50 states, each of its
|
||||
four years -- 2011, 2012, 2019, 2020), intergovernmental spending as a
|
||||
share of a government's own Direct spending is:
|
||||
|
||||
| Government type | Intergovernmental / Direct |
|
||||
|---|---|
|
||||
| State | 16.7%-48.4% (varies by year; 24.0% pooled across all four) |
|
||||
| County | 3.4%-5.1% (varies by year) |
|
||||
| City | 2.6%-3.1% (varies by year) |
|
||||
|
||||
So the Direct/Total choice matters overwhelmingly for **state** governments
|
||||
-- a state's Total genuinely differs from its Direct by a meaningful margin,
|
||||
while for a county or city the two are close. The state range is also far
|
||||
wider than a single flat figure would suggest: legacy wide-era years (2011:
|
||||
48.4%) carry proportionally more intergovernmental spending than the modern
|
||||
era (2019-2020: 16.7%-17.0%), so a state's Direct/Total gap can be nearly
|
||||
3x larger a decade earlier than it is today. That's also why the mistake
|
||||
this vignette warns about is easy to make unnoticed at the county/city level
|
||||
and costly at the state level: rolling up every government in a state using
|
||||
`total` instead of `direct` overstates the true figure -- measured at 7.6%
|
||||
for Alabama in FY2019, and 11.6% nationally.
|
||||
|
||||
# Why Total = Direct + M + L, not Direct + M
|
||||
|
||||
It's tempting to assume `total` only needs to add `M`. But `M` and `L` are
|
||||
both money the queried government itself pays **out** -- they're not two
|
||||
different accounts of a receiving government's revenue. `M` is what it
|
||||
pays to other **local** governments (e.g. a county paying a city for a
|
||||
shared paving contract); `L` is what it pays **up** to its **state**
|
||||
government (e.g. a county's contribution to a state-administered program).
|
||||
A local government's Total genuinely includes both legs, because both are
|
||||
its own spending, just routed to a different kind of recipient. On the
|
||||
bundled fixture corpus (all 50 states, 2011/2012/2019/2020), `L` is 0 for
|
||||
state governments (a state has no "payments to the state government" leg of
|
||||
its own) but is 43%-51% the size of `M` for counties (varies by year) and
|
||||
144%-189% the size of `M` for cities (varies by year; 166% pooled across
|
||||
all four) -- so a `total` that omitted `L` would silently undercount Total
|
||||
specifically for local governments, and for cities `L` is often the
|
||||
*larger* of the two legs.
|
||||
`cog_spending(expenditure_concept = "total")` includes both legs (excluding
|
||||
the `L--` family-total rollup row, which would double-count its own
|
||||
components).
|
||||
|
||||
# Composition rules
|
||||
|
||||
- `expenditure_concept` (whose spending counts -- Direct vs Direct plus
|
||||
intergovernmental) is **orthogonal** to `basis` (which vintage of the
|
||||
item-code space a query resolves against -- `"harmonized"` vs `"raw"`).
|
||||
They combine freely: `expenditure_concept = "total", basis = "raw"` is a
|
||||
valid, meaningful query, and so is every other pairing.
|
||||
- `expenditure_concept = "total"` is **mutually exclusive** with `recipe`: a
|
||||
recipe already defines its own component codes (some recipes have their
|
||||
own matching intergovernmental counterpart recipe instead -- see
|
||||
`cog_recipes()` and the "firing suggestion" notes surfaced in
|
||||
`cog_spending()`'s provenance), so layering a second, generic `total`
|
||||
union on top of a recipe query has no well-defined meaning. Passing both
|
||||
together aborts with an error naming the conflict.
|
||||
- `expenditure_concept` is a **spending-only** concept: `cog_revenue()`
|
||||
doesn't expose it (revenue's own intergovernmental codes are a different
|
||||
axis -- see `?cog_revenue`).
|
||||
|
||||
# Summary
|
||||
|
||||
- Comparing one government to itself over time: `"direct"` or `"total"`
|
||||
both work -- pick one and hold it fixed across every year compared.
|
||||
- Comparing or summing across governments -- counties within a state, a
|
||||
state against its neighbor, cities against counties: use `"direct"`.
|
||||
`cog_geographic_rollup()` and `cog_peer_compare()` enforce this by
|
||||
refusing `"total"`.
|
||||
- `"total"` = Direct (`E`/`F`/`G`) + intergovernmental (`M` to local
|
||||
governments + `L` to the state government, excluding the `L--`
|
||||
family-total row).
|
||||
Reference in New Issue
Block a user