feat: cog_spending + cog_revenue + cog_explain

Three core query verbs over the spending_annotated / revenue_annotated
DuckDB views. Each verb accepts vector govid, vector years, optional
category filter, per_capita flag, and adjust_to_year for CPI-U
real-dollar conversion (bundled index).

Amounts are returned in full USD (SUM(amt) * 1000) so callers can
freely rescale to millions/billions. The $1,000s -> $USD conversion
is recorded in provenance$transformations$units_conversion.

Every result carries an attr(., "provenance") list matching
inst/schemas/provenance-v1.json. cog_explain() prints the structured
form via cli or returns the raw list for MCP/JSON consumers.

Also: .fetch_or_cache_manifest() now handles local fixture paths so
tests can point USCOGDATA_FIXTURE_URL at the pipeline publish_cache/
without a working HTTP server.

Tests: 80 pass / 0 fail. devtools::check() 0E/0W/2N (both notes
pre-existing / environmental).
This commit is contained in:
2026-04-24 09:49:09 -04:00
parent cea7a241c5
commit c682e6547d
12 changed files with 672 additions and 1 deletions
+100
View File
@@ -0,0 +1,100 @@
# R/explain.R
#' Explain a verb result's provenance
#'
#' Prints the structured provenance attached to a tibble returned by any
#' `cog_*` verb, or returns it as a list for downstream use (MCP tools,
#' dashboards, JSON export).
#'
#' @param result A tibble returned by a `cog_*` verb.
#' @param format `"print"` (default) for a human-readable cli summary;
#' returns `result` invisibly for chaining. `"list"` returns the raw
#' provenance list (identical to `attr(result, "provenance")`).
#' @return Either `result` (invisibly) or the provenance list.
#' @export
cog_explain <- function(result, format = c("print", "list")) {
format <- match.arg(format)
prov <- attr(result, "provenance")
if (is.null(prov)) {
cli::cli_abort(c(
"No `provenance` attribute on result.",
i = "Pass a tibble returned by a cog_* verb (e.g. cog_spending())."
))
}
if (format == "list") return(prov)
.print_provenance(prov)
invisible(result)
}
#' @noRd
.print_provenance <- function(prov) {
cli::cli_h1("{prov$verb}()")
tgt_ids <- paste(prov$target$canonical_govid, collapse = ", ")
tgt_names <- if (length(prov$target$gov_name) == 0L) {
"(no rows returned)"
} else {
paste(prov$target$gov_name, collapse = ", ")
}
cli::cli_text("Target: {tgt_names} [canonical_govid: {tgt_ids}]")
yrs <- prov$years
cli::cli_text(if (length(yrs) == 1L) {
"Year: {yrs}"
} else {
"Years: {min(yrs)}-{max(yrs)} ({length(yrs)} years)"
})
if (!is.null(prov$category)) {
cli::cli_text("Category: {paste(prov$category, collapse = ', ')}")
} else {
cli::cli_text("Category: (all)")
}
cli::cli_h2("Codes observed")
codes <- prov$codes_summed$observed
if (length(codes) == 0L) {
cli::cli_alert_info("No item codes matched.")
} else {
cli::cli_ul(codes)
}
if (isTRUE(prov$aggregate_fallback$applied)) {
cli::cli_h2("Aggregate fallback")
cli::cli_alert_warning(
"Aggregate fallback used for years: {paste(prov$aggregate_fallback$years, collapse = ', ')}"
)
}
cli::cli_h2("Transformations")
uc <- prov$transformations$units_conversion
if (isTRUE(uc$applied)) {
cli::cli_text("Units: {uc$source_unit} -> {uc$target_unit} (x{uc$multiplier})")
}
pc <- prov$transformations$per_capita
if (isTRUE(pc$applied)) {
cli::cli_text("Per-capita denominator: {pc$denominator_source}")
}
infl <- prov$transformations$inflation
if (isTRUE(infl$applied)) {
cli::cli_text("Inflation: {infl$index}, base year {infl$base_year}")
}
cli::cli_h2("Scope")
cli::cli_text(
"Included gov types: {paste(prov$scope$gov_types_included, collapse = ', ')}"
)
cli::cli_text(
"Excluded gov types: {paste(prov$scope$gov_types_excluded, collapse = ', ')}"
)
if (nzchar(prov$scope$scope_note %||% "")) {
cli::cli_text("Note: {prov$scope$scope_note}")
}
cli::cli_h2("Data vintage")
cli::cli_text(
"Manifest schema v{prov$manifest$schema_version}, pipeline {prov$manifest$pipeline_commit}, built {prov$manifest$built_at}"
)
invisible(NULL)
}
+17 -1
View File
@@ -1,8 +1,19 @@
# R/manifest.R
#' Fetch manifest.json from URL, cache locally, validate TTL.
#' Fetch manifest.json from URL (or read from a local fixture path),
#' cache locally, validate TTL.
#' @noRd
.fetch_or_cache_manifest <- function(url, cache_dir) {
# Local fixture path: read manifest directly; skip cache/TTL plumbing so
# tests pick up regenerated manifests immediately.
if (.is_local_path(url)) {
local_manifest <- file.path(url, "manifest.json")
if (!file.exists(local_manifest)) {
cli::cli_abort("Local fixture has no manifest.json at {local_manifest}")
}
return(jsonlite::fromJSON(local_manifest, simplifyVector = FALSE))
}
cache_path <- file.path(cache_dir, "manifest.json")
ttl <- as.integer(.cfg("manifest_ttl_secs"))
@@ -19,6 +30,11 @@
jsonlite::fromJSON(cache_path, simplifyVector = FALSE)
}
#' @noRd
.is_local_path <- function(url) {
!grepl("^[a-zA-Z][a-zA-Z0-9+.-]*://", url)
}
#' @noRd
.validate_schema <- function(manifest, expected_version) {
if (manifest$schema_version != expected_version) {
+83
View File
@@ -0,0 +1,83 @@
# R/provenance.R
# Shared provenance construction. Matches inst/schemas/provenance-v1.json.
#' @noRd
.build_provenance <- function(verb, call, govid, years, category,
per_capita, adjust_to_year, result, sql,
subtype_col) {
manifest <- .uscogdata_env$manifest
codes <- result[["codes_included"]]
codes_observed <- if (length(codes) == 0L) {
character(0)
} else {
sorted <- sort(unique(unlist(strsplit(codes, ",", fixed = TRUE))))
sorted[nzchar(sorted)]
}
agg_flag <- result[["aggregate_fallback"]]
agg_applied <- isTRUE(any(agg_flag, na.rm = TRUE))
agg_years <- if (agg_applied) {
unique(as.integer(result$year[which(agg_flag)]))
} else {
integer(0)
}
gov_names <- if (nrow(result) == 0L) {
character(0)
} else {
unique(result$gov_name)
}
list(
verb = verb,
call = paste(deparse(call), collapse = " "),
target = list(
canonical_govid = as.character(govid),
gov_name = gov_names
),
years = as.integer(years),
category = category,
scope = list(
gov_types_included = as.integer(unlist(manifest$scope$gov_types_included)),
gov_types_excluded = as.integer(unlist(manifest$scope$gov_types_excluded)),
scope_note = manifest$scope$scope_note %||% ""
),
codes_summed = list(
observed = codes_observed,
subtype_column = subtype_col
),
aggregate_fallback = list(
applied = agg_applied,
years = agg_years
),
transformations = list(
units_conversion = list(
applied = TRUE,
source_unit = "$1,000s (raw Census)",
target_unit = "$USD",
multiplier = 1000L
),
per_capita = list(
applied = isTRUE(per_capita),
denominator_source = if (isTRUE(per_capita)) {
"ACS 2018-2022 B01003_001 (population_acs from canonical_fips_xwalk)"
} else {
NA_character_
}
),
inflation = list(
applied = !is.null(adjust_to_year),
base_year = if (is.null(adjust_to_year)) NA_integer_ else as.integer(adjust_to_year),
index = if (is.null(adjust_to_year)) NA_character_ else "CPI-U (BLS CPIAUCSL annual average, bundled)"
)
),
series_break_refs = character(0),
manifest = list(
schema_version = as.integer(manifest$schema_version),
pipeline_commit = manifest$pipeline_commit %||% NA_character_,
built_at = manifest$built_at %||% NA_character_
),
sql_query = sql
)
}
+29
View File
@@ -0,0 +1,29 @@
# R/revenue.R
#' Summarized revenue by category
#'
#' Mirror of [cog_spending()] for revenue categories. One row per
#' `(year, canonical_govid, revenue_subtype, category)`. Amounts are returned
#' in **full U.S. dollars** (raw Census values are in $1,000s; this verb
#' multiplies by 1000 and records the conversion in `provenance`).
#'
#' @inheritParams cog_spending
#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`,
#' `revenue_subtype`, `category`, `amt_nominal`, optional `amt_real`,
#' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
#' `codes_included`, `aggregate_fallback`, `notes`.
#' @export
cog_revenue <- function(govid, years, category = NULL,
per_capita = FALSE, adjust_to_year = NULL) {
.verb_spendrev(
verb = "cog_revenue",
view = "revenue_annotated",
subtype_col = "revenue_subtype",
call = match.call(),
govid = govid,
years = years,
category = category,
per_capita = per_capita,
adjust_to_year = adjust_to_year
)
}
+181
View File
@@ -0,0 +1,181 @@
# R/spending.R
#' Summarized spending by category
#'
#' One row per `(year, canonical_govid, spend_subtype, category)`. Amounts are
#' returned in **full U.S. dollars** (the raw corpus stores them in $1,000s;
#' this verb multiplies by 1000 so downstream code can freely rescale to
#' millions/billions). The conversion is recorded in the provenance attribute
#' under `transformations$units_conversion`.
#'
#' @param govid Character vector of `canonical_govid` values.
#' @param years Integer vector of years.
#' @param category Character vector of category names (from
#' `summary_categories.category`), or `NULL` for all categories.
#' @param per_capita If `TRUE`, adds `amt_per_capita_nominal` (and
#' `amt_per_capita_real` when `adjust_to_year` is set) using
#' `population_acs` from the canonical xwalk.
#' @param adjust_to_year Integer base year for CPI-U real-dollar conversion,
#' or `NULL` for nominal only.
#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`,
#' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real`,
#' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
#' `codes_included`, `aggregate_fallback`, `notes`. Carries a `provenance`
#' attribute matching `inst/schemas/provenance-v1.json`.
#' @export
cog_spending <- function(govid, years, category = NULL,
per_capita = FALSE, adjust_to_year = NULL) {
.verb_spendrev(
verb = "cog_spending",
view = "spending_annotated",
subtype_col = "spend_subtype",
call = match.call(),
govid = govid,
years = years,
category = category,
per_capita = per_capita,
adjust_to_year = adjust_to_year
)
}
#' @noRd
.verb_spendrev <- function(verb, view, subtype_col, call,
govid, years, category,
per_capita, adjust_to_year) {
.validate_verb_inputs(govid, years, category, per_capita, adjust_to_year)
govid <- as.character(govid)
years <- as.integer(years)
if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year)
con <- .ensure_session()
sql <- .build_verb_sql(view, subtype_col, govid, years, category)
result <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
if (per_capita) result <- .attach_per_capita(result, con, govid)
if (!is.null(adjust_to_year)) {
result <- .attach_real_dollars(result, adjust_to_year, per_capita)
}
result$notes <- .notes_column(result)
attr(result, "provenance") <- .build_provenance(
verb = verb,
call = call,
govid = govid,
years = years,
category = category,
per_capita = per_capita,
adjust_to_year = adjust_to_year,
result = result,
sql = sql,
subtype_col = subtype_col
)
result
}
#' @noRd
.validate_verb_inputs <- function(govid, years, category,
per_capita, adjust_to_year) {
if (!is.character(govid) || length(govid) == 0L) {
cli::cli_abort("`govid` must be a non-empty character vector.")
}
if (!(is.integer(years) || is.numeric(years)) || length(years) == 0L) {
cli::cli_abort("`years` must be a non-empty integer vector.")
}
if (!is.null(category) && !is.character(category)) {
cli::cli_abort("`category` must be character or NULL.")
}
if (!is.logical(per_capita) || length(per_capita) != 1L) {
cli::cli_abort("`per_capita` must be a length-1 logical.")
}
if (!is.null(adjust_to_year)) {
if (!(is.integer(adjust_to_year) || is.numeric(adjust_to_year)) ||
length(adjust_to_year) != 1L) {
cli::cli_abort("`adjust_to_year` must be NULL or a length-1 integer.")
}
}
invisible(TRUE)
}
#' @noRd
.sql_lit_chr <- function(x) {
safe <- gsub("'", "''", x, fixed = TRUE)
paste0("'", safe, "'", collapse = ",")
}
#' @noRd
.build_verb_sql <- function(view, subtype_col, govid, years, category) {
govid_lit <- .sql_lit_chr(govid)
years_lit <- paste(as.integer(years), collapse = ",")
category_pred <- if (is.null(category)) {
""
} else {
sprintf("AND category IN (%s)", .sql_lit_chr(category))
}
sprintf(
"SELECT
year,
canonical_govid,
COALESCE(xwalk_gov_name, gov_name) AS gov_name,
%1$s,
category,
SUM(amt) * 1000.0 AS amt_nominal,
string_agg(DISTINCT item_code, ',' ORDER BY item_code) AS codes_included,
bool_and(is_aggregate) AS aggregate_fallback
FROM %2$s
WHERE canonical_govid IN (%3$s)
AND year IN (%4$s)
%5$s
GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, %1$s, category
ORDER BY year, canonical_govid, %1$s, category",
subtype_col, view, govid_lit, years_lit, category_pred
)
}
#' @noRd
.attach_per_capita <- function(result, con, govid) {
if (nrow(result) == 0L) {
result$amt_per_capita_nominal <- numeric(0)
return(result)
}
sql <- sprintf(
"SELECT canonical_govid, population_acs
FROM canonical_fips_xwalk
WHERE canonical_govid IN (%s)",
.sql_lit_chr(govid)
)
pops <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
result <- dplyr::left_join(result, pops, by = "canonical_govid")
result$amt_per_capita_nominal <- result$amt_nominal / result$population_acs
result$population_acs <- NULL
result
}
#' @noRd
.attach_real_dollars <- function(result, adjust_to_year, per_capita) {
if (nrow(result) == 0L) {
result$amt_real <- numeric(0)
if (per_capita) result$amt_per_capita_real <- numeric(0)
return(result)
}
result$amt_real <- .inflate(result$amt_nominal, result$year, adjust_to_year)
if (per_capita && "amt_per_capita_nominal" %in% names(result)) {
result$amt_per_capita_real <- .inflate(
result$amt_per_capita_nominal, result$year, adjust_to_year
)
}
result
}
#' @noRd
.notes_column <- function(result) {
if (nrow(result) == 0L) return(character(0))
ifelse(
isTRUE(result$aggregate_fallback) | result$aggregate_fallback %in% TRUE,
"Aggregate fallback applied; see cog_explain()",
""
)
}