Compare commits
13
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8bf9c4ccc1
|
||
|
|
77074621d8
|
||
|
|
4b749205a5
|
||
|
|
f77adb6c83
|
||
|
|
230f3401c4
|
||
|
|
7522b48a08
|
||
|
|
693f8d81a6
|
||
|
|
db35fa9058
|
||
|
|
cabe2e2799
|
||
|
|
6cd219a291 | ||
|
|
e067a5930f
|
||
|
|
342debaefa
|
||
|
|
b59b79b2d5 |
@@ -11,6 +11,21 @@ jobs:
|
|||||||
steps:
|
steps:
|
||||||
- name: Install system libraries and Node.js (required by actions/checkout)
|
- name: Install system libraries and Node.js (required by actions/checkout)
|
||||||
run: |
|
run: |
|
||||||
|
# Switch apt to HTTPS mirrors. Measured from this runner on
|
||||||
|
# 2026-08-04: the SAME index file takes 20.1s over http:// and 3.1s
|
||||||
|
# over https://. apt fetches many indexes serially, so http:// does
|
||||||
|
# not read as "slow" -- it reads as a hang (zero bytes in
|
||||||
|
# /var/cache/apt/archives after 3+ minutes, apt's http workers parked
|
||||||
|
# in S state). rocker/r-ver:4.4 already ships ca-certificates and
|
||||||
|
# apt 2.8.3 has the https method built in, so nothing needs to be
|
||||||
|
# installed over http first to bootstrap this.
|
||||||
|
# `|| true` because the step runs under `sh -e`: on an image whose
|
||||||
|
# sources live in the other location, the missing-file sed must not
|
||||||
|
# kill the job.
|
||||||
|
sed -i -E 's#http://(archive|security)\.ubuntu\.com#https://\1.ubuntu.com#g' \
|
||||||
|
/etc/apt/sources.list.d/ubuntu.sources 2>/dev/null || true
|
||||||
|
sed -i -E 's#http://(archive|security)\.ubuntu\.com#https://\1.ubuntu.com#g' \
|
||||||
|
/etc/apt/sources.list 2>/dev/null || true
|
||||||
apt-get update -qq
|
apt-get update -qq
|
||||||
apt-get install -y --no-install-recommends \
|
apt-get install -y --no-install-recommends \
|
||||||
nodejs git \
|
nodejs git \
|
||||||
|
|||||||
@@ -1,5 +1,38 @@
|
|||||||
# uscogdata 0.1.0 (development)
|
# uscogdata 0.1.0 (development)
|
||||||
|
|
||||||
|
## Signposting now catches partially-suppressed categories
|
||||||
|
|
||||||
|
* A coverage suggestion used to fire only when a category returned **no rows
|
||||||
|
at all** in a requested year. That missed the more dangerous case: a
|
||||||
|
category that still returns rows while silently dropping component codes
|
||||||
|
the wide era publishes only as aggregates (#9). `cog_spending(category =
|
||||||
|
"Public Welfare")` for FY2011 returned a plausible figure that omitted
|
||||||
|
`E67`/`E68` entirely -- for Los Angeles County, $2,075,461,000 of a true
|
||||||
|
$5,261,404,000, a 39% understatement, with `provenance$suggestions` empty.
|
||||||
|
* Suggestions now also fire on **partial** coverage, and every suggestion
|
||||||
|
carries `trigger` (`"empty_year"` or `"suppressed_component"`),
|
||||||
|
`suppressed_amount`, `suppressed_years` and `suppressed_codes`, so a caller
|
||||||
|
can see how much is missing and decide whether to re-run with the recipe.
|
||||||
|
* `cog_revenue()` gets the same fix through the shared verb path. Alaska's
|
||||||
|
FY2011 `Miscellaneous Revenue` reported $943,842,000 while dropping
|
||||||
|
$1,899,995,000 of aggregate-published `U4-` rents and royalties.
|
||||||
|
* The trigger stays recipe-driven, so it only fires where a harmonization
|
||||||
|
recipe actually exists to name the fix. `higher_ed_e18_wide` and
|
||||||
|
`general_gov_e89_wide` stay silent in every year measured on the bundled
|
||||||
|
fixture, because their components are ordinary classified leaves even
|
||||||
|
pre-2012.
|
||||||
|
* The `suppressed_component` trigger (and any `suppressed_amount`/
|
||||||
|
`suppressed_codes` an `empty_year` fire also carries) is scoped to the
|
||||||
|
calling verb's own flow family: `cog_spending()` only ever measures E/F/G
|
||||||
|
component dollars, `cog_revenue()` only T/A/U/B/C/D. A component from the
|
||||||
|
OTHER flow family reports `suppressed_amount = 0` rather than a fabricated
|
||||||
|
claim. The `empty_year` trigger itself is not flow-scoped -- a category
|
||||||
|
belonging to the other flow (e.g. `cog_spending(category = "IG Local")`)
|
||||||
|
still returns zero rows and can still fire, in any year including modern
|
||||||
|
ones, naming the recipe whose own generic join finds real data for this
|
||||||
|
government. That is a mis-scoped query, not a corpus-format gap, so its
|
||||||
|
`suppressed_amount` is correctly 0.
|
||||||
|
|
||||||
## New: `cog_balances()` for cash-and-security holdings
|
## New: `cog_balances()` for cash-and-security holdings
|
||||||
|
|
||||||
* New `cog_balances()` exposes the 14 cash-and-security holding codes
|
* New `cog_balances()` exposes the 14 cash-and-security holding codes
|
||||||
|
|||||||
+8
-1
@@ -125,8 +125,15 @@ cog_explain <- function(result, format = c("print", "list")) {
|
|||||||
if (length(prov$suggestions) > 0L) {
|
if (length(prov$suggestions) > 0L) {
|
||||||
cli::cli_h2("Suggestions")
|
cli::cli_h2("Suggestions")
|
||||||
sugg_lines <- vapply(prov$suggestions, function(s) {
|
sugg_lines <- vapply(prov$suggestions, function(s) {
|
||||||
sprintf("%s -- %s (years %s-%s): %s", s$recipe_id, s$label,
|
line <- sprintf("%s -- %s (years %s-%s): %s", s$recipe_id, s$label,
|
||||||
s$available_years[1], s$available_years[2], s$hint)
|
s$available_years[1], s$available_years[2], s$hint)
|
||||||
|
if (isTRUE(s$suppressed_amount > 0)) {
|
||||||
|
line <- paste0(line, sprintf(" [$%s excluded from %s: %s]",
|
||||||
|
formatC(s$suppressed_amount, format = "f", digits = 0, big.mark = ","),
|
||||||
|
paste0("FY", s$suppressed_years, collapse = ", "),
|
||||||
|
paste(s$suppressed_codes, collapse = ", ")))
|
||||||
|
}
|
||||||
|
line
|
||||||
}, character(1))
|
}, character(1))
|
||||||
cli::cli_ul(sugg_lines)
|
cli::cli_ul(sugg_lines)
|
||||||
}
|
}
|
||||||
|
|||||||
+14
-1
@@ -395,7 +395,8 @@ cog_spending <- function(govid, years, category = NULL,
|
|||||||
}
|
}
|
||||||
suggestions <- .build_suggestions(con, govid, years, category,
|
suggestions <- .build_suggestions(con, govid, years, category,
|
||||||
direct_leg_result,
|
direct_leg_result,
|
||||||
resolved$basis, flow_prefixes)
|
resolved$basis, flow_prefixes,
|
||||||
|
.select_long_view(view_base, resolved$basis))
|
||||||
}
|
}
|
||||||
|
|
||||||
# C1(b): when expenditure_concept = "total", flag any row where the IG
|
# C1(b): when expenditure_concept = "total", flag any row where the IG
|
||||||
@@ -512,6 +513,18 @@ cog_spending <- function(govid, years, category = NULL,
|
|||||||
if (identical(basis, "harmonized")) paste0(view_base, "_harmonized") else view_base
|
if (identical(basis, "harmonized")) paste0(view_base, "_harmonized") else view_base
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#' The `*_long`/`*_long_harmonized` view behind an annotated view base --
|
||||||
|
#' `"spending_annotated"` -> `"spending_long_harmonized"`. `.build_suggestions()`
|
||||||
|
#' anti-joins the LONG view rather than the annotated one: they have identical
|
||||||
|
#' row membership (the annotated views are the long views plus LEFT JOINs, see
|
||||||
|
#' inst/sql/42-spending_annotated_harmonized.sql), but the long view is the
|
||||||
|
#' one that actually owns the `NOT is_aggregate` + crosswalk-membership rule
|
||||||
|
#' the suppression test is asking about.
|
||||||
|
#' @noRd
|
||||||
|
.select_long_view <- function(view_base, basis) {
|
||||||
|
.select_view(sub("_annotated$", "_long", view_base), basis)
|
||||||
|
}
|
||||||
|
|
||||||
#' @noRd
|
#' @noRd
|
||||||
.select_ig_view <- function(basis) {
|
.select_ig_view <- function(basis) {
|
||||||
if (identical(basis, "harmonized")) "ig_annotated_harmonized" else "ig_annotated"
|
if (identical(basis, "harmonized")) "ig_annotated_harmonized" else "ig_annotated"
|
||||||
|
|||||||
+96
-30
@@ -1,8 +1,16 @@
|
|||||||
# R/suggestions.R
|
# R/suggestions.R
|
||||||
# Recipe-component-driven signposting: when a basis = "harmonized" query for
|
# Recipe-component-driven signposting. When a basis = "harmonized" query for
|
||||||
# a category comes back with a coverage gap in some requested years (the
|
# a category comes back incomplete in some requested year -- and a
|
||||||
# result has no rows at all in that year) that a harmonization recipe would
|
# harmonization recipe would actually fill it for this government -- surface
|
||||||
# actually fill for this government, surface that recipe as a suggestion.
|
# that recipe as a suggestion. "Incomplete" has two forms, and a recipe
|
||||||
|
# qualifies on either:
|
||||||
|
# 1. empty_year -- the result has no rows at all in that year.
|
||||||
|
# 2. suppressed_component -- the result HAS rows, but a component code
|
||||||
|
# carries dollars the verb's own long view structurally excludes
|
||||||
|
# (aggregate-published, or absent from summary_categories). This is
|
||||||
|
# uscogdata#9: Public Welfare kept returning E74/E79 rows while dropping
|
||||||
|
# aggregate-only E67/E68, so form 1 never fired and the caller got a
|
||||||
|
# number a third too low with no signpost at all.
|
||||||
#
|
#
|
||||||
# This is deliberately keyed off the recipe catalog's component codes, not
|
# This is deliberately keyed off the recipe catalog's component codes, not
|
||||||
# off harmonization_map rows: no live map row carries a non-blank
|
# off harmonization_map rows: no live map row carries a non-blank
|
||||||
@@ -48,11 +56,15 @@
|
|||||||
#' "D")` for `cog_revenue()` -- see `.verb_spendrev()`). Passed through to
|
#' "D")` for `cog_revenue()` -- see `.verb_spendrev()`). Passed through to
|
||||||
#' `.attach_ig_counterparts()` to keep the intergovernmental-counterpart
|
#' `.attach_ig_counterparts()` to keep the intergovernmental-counterpart
|
||||||
#' lookup scoped to the calling verb's own flow family.
|
#' lookup scoped to the calling verb's own flow family.
|
||||||
|
#' @param long_view Name of the verb's own long view (from
|
||||||
|
#' `.select_long_view()`), passed through to `.suppressed_components()` to
|
||||||
|
#' measure the second qualifying path (uscogdata#9).
|
||||||
#' @return List of `list(recipe_id, label, available_years, hint,
|
#' @return List of `list(recipe_id, label, available_years, hint,
|
||||||
#' ig_recipe_id)`, possibly empty.
|
#' ig_recipe_id, trigger, suppressed_amount, suppressed_years,
|
||||||
|
#' suppressed_codes)`, possibly empty.
|
||||||
#' @noRd
|
#' @noRd
|
||||||
.build_suggestions <- function(con, govid, years, category, result, basis,
|
.build_suggestions <- function(con, govid, years, category, result, basis,
|
||||||
flow_prefixes) {
|
flow_prefixes, long_view) {
|
||||||
if (!identical(basis, "harmonized") || is.null(category)) return(list())
|
if (!identical(basis, "harmonized") || is.null(category)) return(list())
|
||||||
|
|
||||||
# Exclude any recipe that is ITSELF an intergovernmental (M/L) recipe --
|
# Exclude any recipe that is ITSELF an intergovernmental (M/L) recipe --
|
||||||
@@ -86,7 +98,28 @@
|
|||||||
unique(as.integer(result$year))
|
unique(as.integer(result$year))
|
||||||
}
|
}
|
||||||
gap_years <- setdiff(as.integer(years), result_years)
|
gap_years <- setdiff(as.integer(years), result_years)
|
||||||
if (length(gap_years) == 0L) return(list())
|
|
||||||
|
# Path 2 (uscogdata#9): component dollars this government holds that the
|
||||||
|
# verb's own view structurally excludes. Measured across ALL requested
|
||||||
|
# years, not just gap years -- the whole point is that a year with rows can
|
||||||
|
# still be missing dollars. Scoped to the calling verb's own flow_prefixes
|
||||||
|
# (I1) -- see `.suppressed_components()`'s own roxygen for why.
|
||||||
|
#
|
||||||
|
# This runs unconditionally whenever there are candidates -- an earlier
|
||||||
|
# revision of this fix wave tried a free, in-memory pre-check
|
||||||
|
# (`.needs_suppression_query()`) to skip the round trip on an already-
|
||||||
|
# covered path, but a scoped re-review measured it against the fixture and
|
||||||
|
# found it didn't pay for itself (it skipped ~3% of healthy calls, ~0% of
|
||||||
|
# the multi-govid batch shape it was meant to help, at a net cost increase
|
||||||
|
# once its own always-run metadata query was counted) while adding an
|
||||||
|
# untested exactness invariant -- that `result$codes_included` and this
|
||||||
|
# anti-join share the harmonized `item_code` space -- whose silent
|
||||||
|
# violation would kill signposting, the exact failure class uscogdata#9
|
||||||
|
# exists to prevent. Owner's call: keep this simple; a batch-aware
|
||||||
|
# optimization, if one is worth building, is a separate issue.
|
||||||
|
supp <- .suppressed_components(con, candidates, govid, years, long_view, flow_prefixes)
|
||||||
|
|
||||||
|
if (length(gap_years) == 0L && nrow(supp) == 0L) return(list())
|
||||||
|
|
||||||
meta <- tibble::as_tibble(DBI::dbGetQuery(con, sprintf(
|
meta <- tibble::as_tibble(DBI::dbGetQuery(con, sprintf(
|
||||||
"SELECT recipe_id, any_value(label) AS label,
|
"SELECT recipe_id, any_value(label) AS label,
|
||||||
@@ -97,35 +130,55 @@
|
|||||||
.sql_lit_chr(candidates)
|
.sql_lit_chr(candidates)
|
||||||
)))
|
)))
|
||||||
|
|
||||||
# Which (recipe_id, year) pairs the recipe's own generic join actually
|
# Path 1 (unchanged): (recipe, year) pairs the recipe's own generic join
|
||||||
# covers for this government, restricted to the gap years -- the same
|
# covers for this government, restricted to the gap years.
|
||||||
# join .run_recipe() uses (component year_min/year_max + gov_type_scope,
|
covered <- if (length(gap_years) == 0L) {
|
||||||
# no is_aggregate filter), just checking existence instead of summing.
|
data.frame(recipe_id = character(0), year = integer(0))
|
||||||
covered <- DBI::dbGetQuery(con, sprintf(
|
} else {
|
||||||
"SELECT DISTINCT r.recipe_id, l.year
|
DBI::dbGetQuery(con, sprintf(
|
||||||
FROM long l
|
"SELECT DISTINCT r.recipe_id, l.year
|
||||||
JOIN harmonization_recipes r
|
FROM long l
|
||||||
ON l.item_code = r.component_code
|
JOIN harmonization_recipes r
|
||||||
AND l.year BETWEEN r.year_min AND r.year_max
|
ON l.item_code = r.component_code
|
||||||
AND (r.gov_type_scope = 'all'
|
AND l.year BETWEEN r.year_min AND r.year_max
|
||||||
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
AND (r.gov_type_scope = 'all'
|
||||||
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
||||||
WHERE r.recipe_id IN (%s)
|
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||||
AND l.canonical_govid IN (%s)
|
WHERE r.recipe_id IN (%s)
|
||||||
AND l.year IN (%s)",
|
AND l.canonical_govid IN (%s)
|
||||||
.sql_lit_chr(candidates), .sql_lit_chr(govid),
|
AND l.year IN (%s)",
|
||||||
paste(gap_years, collapse = ",")
|
.sql_lit_chr(candidates), .sql_lit_chr(govid),
|
||||||
))
|
paste(gap_years, collapse = ",")
|
||||||
|
))
|
||||||
|
}
|
||||||
|
|
||||||
suggestions <- list()
|
suggestions <- list()
|
||||||
for (rid in candidates) {
|
for (rid in candidates) {
|
||||||
if (!rid %in% covered$recipe_id) next
|
empty_hit <- rid %in% covered$recipe_id
|
||||||
|
s_rows <- supp[supp$recipe_id == rid, , drop = FALSE]
|
||||||
|
supp_hit <- nrow(s_rows) > 0L
|
||||||
|
if (!empty_hit && !supp_hit) next
|
||||||
m <- meta[meta$recipe_id == rid, ]
|
m <- meta[meta$recipe_id == rid, ]
|
||||||
suggestions[[length(suggestions) + 1L]] <- list(
|
suggestions[[length(suggestions) + 1L]] <- list(
|
||||||
recipe_id = rid,
|
recipe_id = rid,
|
||||||
label = m$label[[1]],
|
label = m$label[[1]],
|
||||||
available_years = c(as.integer(m$year_min), as.integer(m$year_max)),
|
available_years = c(as.integer(m$year_min), as.integer(m$year_max)),
|
||||||
hint = sprintf("re-run with recipe = '%s'", rid)
|
hint = sprintf("re-run with recipe = '%s'", rid),
|
||||||
|
# An empty year is the stronger claim -- the category returned nothing
|
||||||
|
# at all -- so it wins when both paths qualify. The suppressed_* fields
|
||||||
|
# are still populated, so an empty_year fire also reports its dollars.
|
||||||
|
trigger = if (empty_hit) "empty_year" else "suppressed_component",
|
||||||
|
suppressed_amount = if (supp_hit) sum(s_rows$suppressed_amount) else 0,
|
||||||
|
suppressed_years = if (supp_hit) {
|
||||||
|
sort(unique(as.integer(s_rows$year)))
|
||||||
|
} else {
|
||||||
|
integer(0)
|
||||||
|
},
|
||||||
|
suppressed_codes = if (supp_hit) {
|
||||||
|
sort(unique(unlist(strsplit(s_rows$suppressed_codes, ",", fixed = TRUE))))
|
||||||
|
} else {
|
||||||
|
character(0)
|
||||||
|
}
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
.attach_ig_counterparts(con, suggestions, flow_prefixes)
|
.attach_ig_counterparts(con, suggestions, flow_prefixes)
|
||||||
@@ -232,12 +285,25 @@
|
|||||||
#' expressions. When a suggestion has an `ig_recipe_id`, one indented
|
#' expressions. When a suggestion has an `ig_recipe_id`, one indented
|
||||||
#' continuation line is appended naming the intergovernmental counterpart
|
#' continuation line is appended naming the intergovernmental counterpart
|
||||||
#' recipe (embedded `\n` renders as a hanging-indent continuation of the
|
#' recipe (embedded `\n` renders as a hanging-indent continuation of the
|
||||||
#' same bullet under cli, not a new bullet).
|
#' same bullet under cli, not a new bullet). Same treatment for
|
||||||
|
#' `suppressed_amount` (uscogdata#9): only present when dollars were
|
||||||
|
#' actually measured as excluded (an `empty_year` fire can carry them too --
|
||||||
|
#' see `.build_suggestions()` -- so this keys off the amount, not `trigger`).
|
||||||
#' @noRd
|
#' @noRd
|
||||||
.inform_suggestions <- function(suggestions) {
|
.inform_suggestions <- function(suggestions) {
|
||||||
bullets <- vapply(suggestions, function(s) {
|
bullets <- vapply(suggestions, function(s) {
|
||||||
bullet <- sprintf("%s (%d-%d): %s", s$recipe_id,
|
bullet <- sprintf("%s (%d-%d): %s", s$recipe_id,
|
||||||
s$available_years[1], s$available_years[2], s$hint)
|
s$available_years[1], s$available_years[2], s$hint)
|
||||||
|
# Only present when dollars were actually measured as excluded. An
|
||||||
|
# empty_year fire can carry them too -- the year had no rows AND the
|
||||||
|
# component was suppressed -- which is strictly more informative.
|
||||||
|
if (isTRUE(s$suppressed_amount > 0)) {
|
||||||
|
bullet <- paste0(bullet, sprintf(
|
||||||
|
"\n $%s excluded from %s (%s), published as an aggregate or outside the crosswalk",
|
||||||
|
formatC(s$suppressed_amount, format = "f", digits = 0, big.mark = ","),
|
||||||
|
paste0("FY", s$suppressed_years, collapse = ", "),
|
||||||
|
paste(s$suppressed_codes, collapse = ", ")))
|
||||||
|
}
|
||||||
if (!is.null(s$ig_recipe_id)) {
|
if (!is.null(s$ig_recipe_id)) {
|
||||||
bullet <- paste0(bullet, sprintf(
|
bullet <- paste0(bullet, sprintf(
|
||||||
"\n intergovernmental counterpart: recipe = '%s'", s$ig_recipe_id))
|
"\n intergovernmental counterpart: recipe = '%s'", s$ig_recipe_id))
|
||||||
@@ -245,7 +311,7 @@
|
|||||||
bullet
|
bullet
|
||||||
}, character(1))
|
}, character(1))
|
||||||
cli::cli_inform(c(
|
cli::cli_inform(c(
|
||||||
i = "Coverage gap detected for the requested years; a harmonization recipe may fill it:",
|
i = "Incomplete coverage for the requested years; a harmonization recipe may fill it:",
|
||||||
stats::setNames(bullets, rep("*", length(bullets)))
|
stats::setNames(bullets, rep("*", length(bullets)))
|
||||||
))
|
))
|
||||||
}
|
}
|
||||||
|
|||||||
+115
@@ -0,0 +1,115 @@
|
|||||||
|
# R/suppression.R
|
||||||
|
# Split out of R/suggestions.R (2026-08-05) to keep files under the project's
|
||||||
|
# 400-line limit. Owns the second qualifying path for coverage signposting
|
||||||
|
# (uscogdata#9): measuring, per government, the component dollars the
|
||||||
|
# calling verb's own long view structurally excludes (aggregate-published,
|
||||||
|
# or absent from summary_categories). See R/suggestions.R for the
|
||||||
|
# orchestrator (`.build_suggestions()`) that calls this and the full
|
||||||
|
# uscogdata#9 background.
|
||||||
|
|
||||||
|
#' Measure, per (recipe, year), the component dollars this government holds
|
||||||
|
#' that the calling verb's own long view structurally excludes.
|
||||||
|
#'
|
||||||
|
#' This is the second qualifying path for a suggestion (uscogdata#9). The
|
||||||
|
#' first -- row absence -- only fires when a category returns NOTHING in a
|
||||||
|
#' requested year, which is how Corrections behaves in the wide era. Public
|
||||||
|
#' Welfare is the failure mode it misses: E74/E75/E77/E79 still return rows,
|
||||||
|
#' so there is no absence to detect, while E67/E68 (aggregate-flagged 1967-
|
||||||
|
#' 2011, and absent from `summary_categories` entirely) are dropped. The
|
||||||
|
#' caller gets a plausible number a third too low, silently.
|
||||||
|
#'
|
||||||
|
#' "Structurally excluded" is decided by anti-joining the verb's REAL long
|
||||||
|
#' view rather than restating its WHERE clause, so this stays correct if
|
||||||
|
#' `spending_long_harmonized` / `revenue_long_harmonized` ever change. That
|
||||||
|
#' anti-join is keyed on `item_code`, which is sound only because
|
||||||
|
#' harmonization never renames a recipe component -- asserted by the "no
|
||||||
|
#' recipe component is ever renamed by harmonization" test in
|
||||||
|
#' tests/testthat/test-recipes.R.
|
||||||
|
#'
|
||||||
|
#' Note what this deliberately does NOT count as suppressed: a component
|
||||||
|
#' excluded from the RESULT for scoping reasons -- because it belongs to a
|
||||||
|
#' different `category`, or because `expenditure_concept` narrowed the
|
||||||
|
#' subtypes -- is still present in the view, so it never fires. Suggesting a
|
||||||
|
#' recipe is a coverage fix, not a category redefinition.
|
||||||
|
#'
|
||||||
|
#' `flow_prefixes` (uscogdata#9 review, finding I1) restricts the measured
|
||||||
|
#' components to the CALLING VERB's own flow family (`c("E","F","G")` for
|
||||||
|
#' spending, `c("T","A","U","B","C","D")` for revenue). Without this, a
|
||||||
|
#' candidate recipe belonging to the OTHER flow family is always absent from
|
||||||
|
#' this verb's view (by construction -- `cog_revenue()`'s view never carries
|
||||||
|
#' an E-coded row) and so was always reported as "suppressed", fabricating a
|
||||||
|
#' dollar claim across flow families (`cog_revenue(category = "Corrections")`
|
||||||
|
#' claimed $3.63B excluded that `cog_spending()` reports and fully accounts
|
||||||
|
#' for). Filtering on `LEFT(r.component_code, 1)` also drops M/L-prefixed
|
||||||
|
#' components from measurement under `cog_spending()` (`flow_prefixes` never
|
||||||
|
#' includes "M"/"L") -- harmless today, because a recipe's own M/L components
|
||||||
|
#' (e.g. `corrections_ig_local_combined`'s M04/M05) are present in the view
|
||||||
|
#' in every year they exist and so never fired as suppressed anyway, but
|
||||||
|
#' worth recording since this filter is now the thing relied on to prevent
|
||||||
|
#' it.
|
||||||
|
#'
|
||||||
|
#' @param con Active DuckDB connection.
|
||||||
|
#' @param candidates Character vector of recipe ids to measure.
|
||||||
|
#' @param govid Character vector of canonical_govid values.
|
||||||
|
#' @param years Integer vector of requested years.
|
||||||
|
#' @param long_view Name of the verb's long view, from `.select_long_view()`.
|
||||||
|
#' @param flow_prefixes The calling verb's own flow-type prefixes (see
|
||||||
|
#' `.build_suggestions()`). Only recipe components whose first character is
|
||||||
|
#' in this set are measured.
|
||||||
|
#' @return Tibble of `recipe_id`, `year`, `suppressed_amount` (full US
|
||||||
|
#' dollars), `suppressed_codes` (comma-joined, sorted). Zero rows when
|
||||||
|
#' nothing is suppressed.
|
||||||
|
#' @noRd
|
||||||
|
.suppressed_components <- function(con, candidates, govid, years, long_view,
|
||||||
|
flow_prefixes) {
|
||||||
|
empty <- tibble::tibble(
|
||||||
|
recipe_id = character(0), year = numeric(0),
|
||||||
|
suppressed_amount = numeric(0), suppressed_codes = character(0)
|
||||||
|
)
|
||||||
|
if (length(candidates) == 0L) return(empty)
|
||||||
|
|
||||||
|
# long_view is interpolated as a SQL IDENTIFIER, not a literal, so it can
|
||||||
|
# never be quoted safely. It is always internally derived from a fixed
|
||||||
|
# view_base, so an off-allowlist value is a programming error, not input.
|
||||||
|
if (!long_view %in% c("spending_long", "spending_long_harmonized",
|
||||||
|
"revenue_long", "revenue_long_harmonized")) {
|
||||||
|
cli::cli_abort(
|
||||||
|
"Internal error: unexpected `long_view` {.val {long_view}}.",
|
||||||
|
class = "uscogdata_internal_error"
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
sql <- sprintf(
|
||||||
|
"SELECT r.recipe_id,
|
||||||
|
l.year,
|
||||||
|
SUM(l.amt) * 1000.0 AS suppressed_amount,
|
||||||
|
string_agg(DISTINCT l.item_code, ',' ORDER BY l.item_code)
|
||||||
|
AS suppressed_codes
|
||||||
|
FROM long l
|
||||||
|
JOIN harmonization_recipes r
|
||||||
|
ON l.item_code = r.component_code
|
||||||
|
AND l.year BETWEEN r.year_min AND r.year_max
|
||||||
|
AND (r.gov_type_scope = 'all'
|
||||||
|
OR (r.gov_type_scope = 'state' AND l.type = 0)
|
||||||
|
OR (r.gov_type_scope = 'local' AND l.type BETWEEN 1 AND 3))
|
||||||
|
WHERE r.recipe_id IN (%1$s)
|
||||||
|
AND l.canonical_govid IN (%2$s)
|
||||||
|
AND l.year IN (%3$s)
|
||||||
|
AND l.amt <> 0
|
||||||
|
AND LEFT(r.component_code, 1) IN (%5$s)
|
||||||
|
AND NOT EXISTS (
|
||||||
|
SELECT 1 FROM %4$s v
|
||||||
|
WHERE v.canonical_govid = l.canonical_govid
|
||||||
|
AND v.year = l.year
|
||||||
|
AND v.item_code = l.item_code
|
||||||
|
AND v.year IN (%3$s) -- restated: enables partition pruning (I3a)
|
||||||
|
AND v.canonical_govid IN (%2$s) -- restated: pushes the govid filter (I3a)
|
||||||
|
)
|
||||||
|
GROUP BY 1, 2
|
||||||
|
ORDER BY 1, 2",
|
||||||
|
.sql_lit_chr(candidates), .sql_lit_chr(govid),
|
||||||
|
paste(as.integer(years), collapse = ","), long_view,
|
||||||
|
.sql_lit_chr(flow_prefixes)
|
||||||
|
)
|
||||||
|
tibble::as_tibble(DBI::dbGetQuery(con, sql))
|
||||||
|
}
|
||||||
@@ -33,7 +33,48 @@
|
|||||||
},
|
},
|
||||||
"harmonization": { "type": "object" },
|
"harmonization": { "type": "object" },
|
||||||
"recipe": { "type": ["object", "null"] },
|
"recipe": { "type": ["object", "null"] },
|
||||||
"suggestions": { "type": "array" },
|
"suggestions": {
|
||||||
|
"type": "array",
|
||||||
|
"description": "Harmonization recipes that would fill incomplete coverage in the requested years for this government. Empty on a healthy query, on an un-scoped (category = NULL) query, on basis = 'raw', and on a recipe = query (which resolves its own coverage).",
|
||||||
|
"items": {
|
||||||
|
"type": "object",
|
||||||
|
"required": ["recipe_id", "label", "available_years", "hint", "ig_recipe_id",
|
||||||
|
"trigger", "suppressed_amount", "suppressed_years", "suppressed_codes"],
|
||||||
|
"properties": {
|
||||||
|
"recipe_id": { "type": "string" },
|
||||||
|
"label": { "type": "string" },
|
||||||
|
"available_years": {
|
||||||
|
"type": "array",
|
||||||
|
"items": { "type": "integer" },
|
||||||
|
"description": "[year_min, year_max] of the recipe's component coverage."
|
||||||
|
},
|
||||||
|
"hint": { "type": "string" },
|
||||||
|
"ig_recipe_id": {
|
||||||
|
"type": ["string", "null"],
|
||||||
|
"description": "The intergovernmental (M/L) counterpart recipe covering the same function suffixes, or null. Never set for revenue recipes."
|
||||||
|
},
|
||||||
|
"trigger": {
|
||||||
|
"type": "string",
|
||||||
|
"enum": ["empty_year", "suppressed_component"],
|
||||||
|
"description": "Why this fired. 'empty_year': the result has no rows at all in a requested year. 'suppressed_component': the result HAS rows, but a component code carries dollars this government reports in the requested years that the verb's underlying long view structurally excludes -- aggregate-published, carrying no harmonized code, or absent from summary_categories. This is NOT the same thing as 'excluded from the result': a component present in the view under a different category (a scoping choice, e.g. a different `category` or a narrower `expenditure_concept`) contributes 0 and never fires. 'empty_year' wins when both apply, being the stronger claim; the suppressed_* fields are populated either way, using the same underlying-view measurement, and can be 0 even on an 'empty_year' fire."
|
||||||
|
},
|
||||||
|
"suppressed_amount": {
|
||||||
|
"type": "number",
|
||||||
|
"description": "Full US dollars this government reports, in the recipe's component codes, in the requested years, that the verb's underlying long view structurally excludes (aggregate-published, carrying no harmonized code, or absent from summary_categories) -- summed across those years. This is NOT the same quantity as 'what the result excludes': a component present in the view under a different category or a narrower `expenditure_concept` is scoped out on purpose, counts as 0 here, and is not suppression. 0 does not always mean full coverage -- see 'trigger' and 'empty_year'. May be negative where Census publishes a negative `amt` for the excluded rows."
|
||||||
|
},
|
||||||
|
"suppressed_years": {
|
||||||
|
"type": "array",
|
||||||
|
"items": { "type": "integer" },
|
||||||
|
"description": "The requested years contributing to suppressed_amount."
|
||||||
|
},
|
||||||
|
"suppressed_codes": {
|
||||||
|
"type": "array",
|
||||||
|
"items": { "type": "string" },
|
||||||
|
"description": "The excluded component item codes, sorted."
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
"scope": { "type": "object" },
|
"scope": { "type": "object" },
|
||||||
"codes_summed": { "type": "object" },
|
"codes_summed": { "type": "object" },
|
||||||
"aggregate_fallback": { "type": ["object", "null"] },
|
"aggregate_fallback": { "type": ["object", "null"] },
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -111,7 +111,7 @@ test_that("cog_manifest returns the active session's parsed manifest", {
|
|||||||
})
|
})
|
||||||
})
|
})
|
||||||
|
|
||||||
test_that(".validate_schema accepts schema_version 4, 5 and 6, rejects others", {
|
test_that(".validate_schema accepts schema_version 4 through 7, rejects others", {
|
||||||
expect_silent(uscogdata:::.validate_schema(list(schema_version = 4L)))
|
expect_silent(uscogdata:::.validate_schema(list(schema_version = 4L)))
|
||||||
expect_silent(uscogdata:::.validate_schema(list(schema_version = 5L)))
|
expect_silent(uscogdata:::.validate_schema(list(schema_version = 5L)))
|
||||||
# v6 = FIPS geography harmonization (2026-07-22): _code -> _asof rename +
|
# v6 = FIPS geography harmonization (2026-07-22): _code -> _asof rename +
|
||||||
@@ -119,12 +119,22 @@ test_that(".validate_schema accepts schema_version 4, 5 and 6, rejects others",
|
|||||||
# renamed columns and its geography comes from the xwalk, so v6 is accepted
|
# renamed columns and its geography comes from the xwalk, so v6 is accepted
|
||||||
# without behavioural change -- see .validate_schema()'s note.
|
# without behavioural change -- see .validate_schema()'s note.
|
||||||
expect_silent(uscogdata:::.validate_schema(list(schema_version = 6L)))
|
expect_silent(uscogdata:::.validate_schema(list(schema_version = 6L)))
|
||||||
|
# v7 = `data_year` APPENDED as column 29 (cog_pipeline #80, 2026-08-03), the
|
||||||
|
# most recent fiscal year contributing to a collapsed key. Appended, never
|
||||||
|
# inserted: canonical_govid stays at position 26, so nothing this package
|
||||||
|
# reads shifts. Verified against the real v7 corpus before widening the
|
||||||
|
# allow-list -- cog_spending()/cog_balances() return correctly for FY2024 AND
|
||||||
|
# for FY2012, so the new column is inert here.
|
||||||
|
expect_silent(uscogdata:::.validate_schema(list(schema_version = 7L)))
|
||||||
expect_error(
|
expect_error(
|
||||||
uscogdata:::.validate_schema(list(schema_version = 3L)),
|
uscogdata:::.validate_schema(list(schema_version = 3L)),
|
||||||
"schema_version"
|
"schema_version"
|
||||||
)
|
)
|
||||||
|
# The upper bound still has to be ENFORCED, not just moved. Without this the
|
||||||
|
# test would no longer prove that an unknown future schema is refused, and a
|
||||||
|
# v8 corpus with a genuinely breaking change would sail through.
|
||||||
expect_error(
|
expect_error(
|
||||||
uscogdata:::.validate_schema(list(schema_version = 7L)),
|
uscogdata:::.validate_schema(list(schema_version = 8L)),
|
||||||
"schema_version"
|
"schema_version"
|
||||||
)
|
)
|
||||||
})
|
})
|
||||||
|
|||||||
@@ -214,3 +214,255 @@ test_that("no signposting under basis = 'raw'", {
|
|||||||
prov <- attr(r, "provenance")
|
prov <- attr(r, "provenance")
|
||||||
expect_length(prov$suggestions, 0L)
|
expect_length(prov$suggestions, 0L)
|
||||||
})
|
})
|
||||||
|
|
||||||
|
# --- uscogdata#9: partial-coverage signposting ------------------------------
|
||||||
|
|
||||||
|
test_that("no recipe component is ever renamed by harmonization", {
|
||||||
|
# The suppression trigger anti-joins the verb's long view on item_code.
|
||||||
|
# That is only sound because harmonization never rewrites a recipe
|
||||||
|
# component's code -- every component whose harmonized_code differs has
|
||||||
|
# harmonized_code IS NULL (and is aggregate-flagged). If this ever fails,
|
||||||
|
# .suppressed_components() would report reachable dollars as suppressed.
|
||||||
|
skip_if_no_corpus()
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
n <- DBI::dbGetQuery(con,
|
||||||
|
"SELECT COUNT(*) AS renamed FROM long
|
||||||
|
WHERE item_code IN (SELECT DISTINCT component_code FROM harmonization_recipes)
|
||||||
|
AND harmonized_code IS NOT NULL
|
||||||
|
AND harmonized_code <> item_code")$renamed
|
||||||
|
expect_equal(as.integer(n), 0L)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".select_long_view maps annotated view bases to their long views", {
|
||||||
|
expect_equal(
|
||||||
|
uscogdata:::.select_long_view("spending_annotated", "harmonized"),
|
||||||
|
"spending_long_harmonized")
|
||||||
|
expect_equal(
|
||||||
|
uscogdata:::.select_long_view("revenue_annotated", "harmonized"),
|
||||||
|
"revenue_long_harmonized")
|
||||||
|
expect_equal(
|
||||||
|
uscogdata:::.select_long_view("spending_annotated", "raw"),
|
||||||
|
"spending_long")
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".suppressed_components measures the E67/E68 dollars Public Welfare drops", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
s <- uscogdata:::.suppressed_components(
|
||||||
|
con,
|
||||||
|
candidates = c("welfare_cash_e67_wide", "welfare_cash_e68_wide"),
|
||||||
|
govid = "061037123085", years = 2011L,
|
||||||
|
long_view = "spending_long_harmonized",
|
||||||
|
flow_prefixes = c("E", "F", "G"))
|
||||||
|
|
||||||
|
expect_s3_class(s, "tbl_df")
|
||||||
|
expect_equal(nrow(s), 2L)
|
||||||
|
s <- s[order(s$recipe_id), ]
|
||||||
|
expect_equal(s$recipe_id, c("welfare_cash_e67_wide", "welfare_cash_e68_wide"))
|
||||||
|
expect_equal(s$suppressed_amount, c(1803872000, 271589000))
|
||||||
|
expect_equal(s$suppressed_codes, c("E67", "E68"))
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".suppressed_components finds nothing in a modern year", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
s <- uscogdata:::.suppressed_components(
|
||||||
|
con,
|
||||||
|
candidates = c("welfare_cash_e67_wide", "welfare_cash_e68_wide"),
|
||||||
|
govid = "061037123085", years = 2019L,
|
||||||
|
long_view = "spending_long_harmonized",
|
||||||
|
flow_prefixes = c("E", "F", "G"))
|
||||||
|
expect_equal(nrow(s), 0L)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".suppressed_components rejects a long_view outside the allowlist", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
expect_error(
|
||||||
|
uscogdata:::.suppressed_components(
|
||||||
|
con, candidates = "welfare_cash_e67_wide", govid = "061037123085",
|
||||||
|
years = 2011L, long_view = "long; DROP TABLE x",
|
||||||
|
flow_prefixes = c("E", "F", "G")),
|
||||||
|
class = "uscogdata_internal_error")
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that(".suppressed_components never measures a component from the other flow family (I1)", {
|
||||||
|
# uscogdata#9 review, finding I1: without the flow_prefixes filter, a
|
||||||
|
# candidate recipe entirely outside the calling verb's own flow family is
|
||||||
|
# ALWAYS absent from that verb's view (by construction), so it was always
|
||||||
|
# reported as "suppressed" -- fabricating a dollar claim. E67/E68 are
|
||||||
|
# Public Welfare EXPENDITURE codes; scoping the measurement to revenue's
|
||||||
|
# own flow_prefixes must find nothing for them.
|
||||||
|
skip_if_no_corpus()
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
s <- uscogdata:::.suppressed_components(
|
||||||
|
con,
|
||||||
|
candidates = c("welfare_cash_e67_wide", "welfare_cash_e68_wide"),
|
||||||
|
govid = "061037123085", years = 2011L,
|
||||||
|
long_view = "revenue_long_harmonized",
|
||||||
|
flow_prefixes = c("T", "A", "U", "B", "C", "D"))
|
||||||
|
expect_equal(nrow(s), 0L)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("uscogdata#9: Public Welfare signposts its suppressed E67/E68 dollars", {
|
||||||
|
# The bug: E74/E79 return rows for FY2011, so there is no row-absence gap,
|
||||||
|
# so nothing fired -- while E67 ($1,803,872,000) and E68 ($271,589,000) were
|
||||||
|
# dropped for being aggregate-published. LA County reports $3,185,943,000
|
||||||
|
# and omits $2,075,461,000, a 39% understatement, silently.
|
||||||
|
skip_if_no_corpus()
|
||||||
|
r <- suppressMessages(
|
||||||
|
cog_spending("061037123085", years = 2011L, category = "Public Welfare"))
|
||||||
|
sugg <- attr(r, "provenance")$suggestions
|
||||||
|
|
||||||
|
expect_length(sugg, 2L)
|
||||||
|
ids <- vapply(sugg, function(s) s$recipe_id, character(1))
|
||||||
|
expect_setequal(ids, c("welfare_cash_e67_wide", "welfare_cash_e68_wide"))
|
||||||
|
|
||||||
|
e67 <- sugg[[which(ids == "welfare_cash_e67_wide")]]
|
||||||
|
expect_equal(e67$trigger, "suppressed_component")
|
||||||
|
expect_equal(e67$suppressed_amount, 1803872000)
|
||||||
|
expect_equal(e67$suppressed_years, 2011L)
|
||||||
|
expect_equal(e67$suppressed_codes, "E67")
|
||||||
|
expect_equal(e67$hint, "re-run with recipe = 'welfare_cash_e67_wide'")
|
||||||
|
|
||||||
|
e68 <- sugg[[which(ids == "welfare_cash_e68_wide")]]
|
||||||
|
expect_equal(e68$trigger, "suppressed_component")
|
||||||
|
expect_equal(e68$suppressed_amount, 271589000)
|
||||||
|
expect_equal(e68$suppressed_codes, "E68")
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("uscogdata#9: an empty_year fire keeps its trigger and gains the dollars", {
|
||||||
|
# Corrections is the case that already worked: zero rows in FY2011, so the
|
||||||
|
# row-absence path fires. It must keep firing, keep trigger = "empty_year",
|
||||||
|
# keep its IG counterpart -- and now also report what was suppressed.
|
||||||
|
skip_if_no_corpus()
|
||||||
|
r <- suppressMessages(
|
||||||
|
cog_spending("061037123085", years = 2011L, category = "Corrections"))
|
||||||
|
sugg <- attr(r, "provenance")$suggestions
|
||||||
|
|
||||||
|
expect_length(sugg, 3L)
|
||||||
|
ids <- vapply(sugg, function(s) s$recipe_id, character(1))
|
||||||
|
expect_setequal(ids, c("corrections_combined", "corrections_capital_combined",
|
||||||
|
"corrections_other_capital_combined"))
|
||||||
|
expect_true(all(vapply(sugg, function(s) s$trigger, character(1)) == "empty_year"))
|
||||||
|
|
||||||
|
cc <- sugg[[which(ids == "corrections_combined")]]
|
||||||
|
expect_equal(cc$suppressed_amount, 1371460000)
|
||||||
|
expect_equal(cc$suppressed_codes, "E05")
|
||||||
|
expect_equal(cc$ig_recipe_id, "corrections_ig_local_combined")
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("uscogdata#9: the revenue verb inherits the same trigger", {
|
||||||
|
# Alaska state FY2011 Miscellaneous Revenue reports $943,842,000 from
|
||||||
|
# U11/U20/U30 while dropping $1,899,995,000 of aggregate-published `U4-`
|
||||||
|
# rents and royalties -- the omission is LARGER than the reported figure.
|
||||||
|
skip_if_no_corpus()
|
||||||
|
r <- suppressMessages(
|
||||||
|
cog_revenue("020000227749", years = 2011L,
|
||||||
|
category = "Miscellaneous Revenue"))
|
||||||
|
sugg <- attr(r, "provenance")$suggestions
|
||||||
|
|
||||||
|
expect_length(sugg, 1L)
|
||||||
|
expect_equal(sugg[[1]]$recipe_id, "rents_royalties_u4_wide")
|
||||||
|
expect_equal(sugg[[1]]$trigger, "suppressed_component")
|
||||||
|
expect_equal(sugg[[1]]$suppressed_amount, 1899995000)
|
||||||
|
expect_equal(sugg[[1]]$suppressed_codes, "U4-")
|
||||||
|
# A revenue recipe must never be handed an M/L expenditure counterpart.
|
||||||
|
expect_null(sugg[[1]]$ig_recipe_id)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("I1: cog_revenue never fabricates suppressed dollars for an expenditure-only recipe", {
|
||||||
|
# uscogdata#9 review, finding I1: Corrections is an expenditure-only
|
||||||
|
# category (E04/E05). cog_revenue() naturally returns zero rows for it, so
|
||||||
|
# corrections_combined still fires as an empty_year suggestion (its own
|
||||||
|
# generic join finds real E04/E05 data for this government) -- but before
|
||||||
|
# the flow_prefixes fix, .suppressed_components() measured E04/E05 against
|
||||||
|
# cog_revenue()'s OWN view (which can never contain an E-coded row by
|
||||||
|
# construction) and reported the full $3,631,945,000 as "suppressed",
|
||||||
|
# when cog_spending() for the same gov/years/category actually returns
|
||||||
|
# $3,691,029,000 -- nothing was suppressed at all.
|
||||||
|
skip_if_no_corpus()
|
||||||
|
r <- suppressMessages(
|
||||||
|
cog_revenue("061037123085", years = 2019:2020, category = "Corrections"))
|
||||||
|
sugg <- attr(r, "provenance")$suggestions
|
||||||
|
ids <- vapply(sugg, function(s) s$recipe_id, character(1))
|
||||||
|
expect_true("corrections_combined" %in% ids)
|
||||||
|
|
||||||
|
hit <- sugg[[which(ids == "corrections_combined")]]
|
||||||
|
expect_equal(hit$suppressed_amount, 0)
|
||||||
|
expect_equal(hit$suppressed_years, integer(0))
|
||||||
|
expect_equal(hit$suppressed_codes, character(0))
|
||||||
|
|
||||||
|
# And cog_spending() for the identical gov/years/category is unaffected --
|
||||||
|
# it actually finds the E04/E05 dollars the buggy measurement claimed were
|
||||||
|
# excluded.
|
||||||
|
sp <- suppressMessages(
|
||||||
|
cog_spending("061037123085", years = 2019:2020, category = "Corrections"))
|
||||||
|
expect_equal(sum(sp$amt_nominal), 3691029000)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("uscogdata#9: no partial-coverage fire in a modern year", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
r <- cog_spending("061037123085", years = 2019L, category = "Public Welfare")
|
||||||
|
expect_length(attr(r, "provenance")$suggestions, 0L)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("uscogdata#9: leaf-and-classified wide-era families never fire", {
|
||||||
|
# higher_ed_e18_wide and general_gov_e89_wide are the control group: their
|
||||||
|
# components (E16/E18, E85/E89) are ordinary classified leaves even in the
|
||||||
|
# wide era, so widening the trigger must leave them silent. This is the
|
||||||
|
# measurement that refutes "it would fire on every category in every legacy
|
||||||
|
# year" -- corpus-wide on the fixture, these two produce zero suppressed rows.
|
||||||
|
skip_if_no_corpus()
|
||||||
|
con <- uscogdata:::.ensure_session()
|
||||||
|
n <- DBI::dbGetQuery(con,
|
||||||
|
"SELECT COUNT(*) AS n
|
||||||
|
FROM long l
|
||||||
|
JOIN harmonization_recipes r
|
||||||
|
ON l.item_code = r.component_code
|
||||||
|
AND l.year BETWEEN r.year_min AND r.year_max
|
||||||
|
WHERE r.recipe_id IN ('higher_ed_e18_wide', 'general_gov_e89_wide')
|
||||||
|
AND l.amt <> 0
|
||||||
|
AND NOT EXISTS (
|
||||||
|
SELECT 1 FROM spending_long_harmonized v
|
||||||
|
WHERE v.canonical_govid = l.canonical_govid
|
||||||
|
AND v.year = l.year AND v.item_code = l.item_code)")$n
|
||||||
|
expect_equal(as.integer(n), 0L)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("uscogdata#9: the cli message reports the suppressed dollars", {
|
||||||
|
skip_if_no_corpus()
|
||||||
|
expect_message(
|
||||||
|
cog_spending("061037123085", years = 2011L, category = "Public Welfare"),
|
||||||
|
"1,803,872,000", fixed = TRUE)
|
||||||
|
expect_message(
|
||||||
|
cog_spending("061037123085", years = 2011L, category = "Public Welfare"),
|
||||||
|
"FY2011", fixed = TRUE)
|
||||||
|
expect_message(
|
||||||
|
cog_spending("061037123085", years = 2011L, category = "Public Welfare"),
|
||||||
|
"E67", fixed = TRUE)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("uscogdata#9: cog_explain() reports the suppressed dollars", {
|
||||||
|
# cog_explain()'s whole "print" output -- including the Suggestions
|
||||||
|
# section built from cli::cli_ul() -- is emitted on the message stream
|
||||||
|
# (verified empirically 2026-08-04: capture.output(..., type = "output")
|
||||||
|
# returns character(0) for this call; testthat::capture_messages() is what
|
||||||
|
# actually carries it), so that is the stream this test captures.
|
||||||
|
skip_if_no_corpus()
|
||||||
|
r <- suppressMessages(
|
||||||
|
cog_spending("061037123085", years = 2011L, category = "Public Welfare"))
|
||||||
|
out <- paste(testthat::capture_messages(cog_explain(r)), collapse = "")
|
||||||
|
expect_match(out, "271,589,000", fixed = TRUE)
|
||||||
|
})
|
||||||
|
|
||||||
|
test_that("the provenance schema documents the suggestion trigger fields", {
|
||||||
|
sch <- jsonlite::fromJSON(
|
||||||
|
system.file("schemas", "provenance-v1.json", package = "uscogdata"),
|
||||||
|
simplifyVector = FALSE)
|
||||||
|
props <- sch$properties$suggestions$items$properties
|
||||||
|
expect_true(all(c("trigger", "suppressed_amount", "suppressed_years",
|
||||||
|
"suppressed_codes") %in% names(props)))
|
||||||
|
expect_setequal(unlist(props$trigger$enum),
|
||||||
|
c("empty_year", "suppressed_component"))
|
||||||
|
})
|
||||||
|
|||||||
Reference in New Issue
Block a user