expenditure_concept = direct|total in cog_spending(), refused in the cross-government verbs #10

Merged
jared merged 20 commits from feat/expenditure-concept into main 2026-07-27 13:13:02 -04:00
8 changed files with 208 additions and 8 deletions
Showing only changes of commit fefd4fe969 - Show all commits
+59 -6
View File
@@ -42,6 +42,14 @@
#' `basis = "recipe"` with an inert `harmonization` block (`applied =
#' FALSE`, pointing at the `recipe` block instead) rather than a
#' possibly-misleading `"harmonized"`/`"raw"` value.
#' @param expenditure_concept `"direct"` (default) returns only the
#' government's own direct spending (item codes `E`/`F`/`G`), unchanged
#' from prior releases. `"total"` additionally UNIONs in the
#' intergovernmental leg -- payments to local governments (`M` codes) and
#' to the state government (`L` codes, excluding the `L--` family-total
#' rollup) -- so results gain rows with `spend_subtype ==
#' "intergovernmental"`. Mutually exclusive with `recipe` (a recipe
#' already defines its own component codes).
#' @return Tibble with columns `year`, `canonical_govid`, `gov_name`,
#' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real`,
#' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`,
@@ -50,7 +58,8 @@
#' @export
cog_spending <- function(govid, years, category = NULL,
per_capita = FALSE, adjust_to_year = NULL,
basis = c("harmonized", "raw"), recipe = NULL) {
basis = c("harmonized", "raw"), recipe = NULL,
expenditure_concept = c("direct", "total")) {
.verb_spendrev(
verb = "cog_spending",
view_base = "spending_annotated",
@@ -63,7 +72,8 @@ cog_spending <- function(govid, years, category = NULL,
per_capita = per_capita,
adjust_to_year = adjust_to_year,
basis = basis,
recipe = recipe
recipe = recipe,
expenditure_concept = expenditure_concept
)
}
@@ -71,14 +81,36 @@ cog_spending <- function(govid, years, category = NULL,
.verb_spendrev <- function(verb, view_base, subtype_col, flow_prefixes, call,
govid, years, category,
per_capita, adjust_to_year,
basis = c("harmonized", "raw"), recipe = NULL) {
basis = c("harmonized", "raw"), recipe = NULL,
expenditure_concept = c("direct", "total")) {
basis_explicit <- length(basis) == 1L
basis <- match.arg(basis, c("harmonized", "raw"))
# match.arg() itself throws a base `simpleError`, not an rlang-classed
# condition; wrap it so an invalid expenditure_concept aborts consistently
# with the rest of this package's validation (cli::cli_abort -> rlang_error).
expenditure_concept <- tryCatch(
match.arg(expenditure_concept, c("direct", "total")),
error = function(e) {
cli::cli_abort(
"`expenditure_concept` must be one of {.val direct} or {.val total}.",
class = "uscogdata_invalid_expenditure_concept",
parent = e
)
}
)
govid <- .coerce_govid_input(govid, arg = "govid")
.validate_verb_inputs(govid, years, category, per_capita, adjust_to_year,
recipe)
if (!is.null(recipe) && identical(expenditure_concept, "total")) {
cli::cli_abort(c(
"`recipe` and `expenditure_concept = \"total\"` are mutually exclusive.",
i = "A recipe defines its own component codes; pass one or the other.",
i = "For a recipe's intergovernmental counterpart, use the matching IG recipe (e.g. `corrections_ig_local_combined`)."
), class = "uscogdata_recipe_concept_conflict")
}
years <- as.integer(years)
if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year)
@@ -105,7 +137,12 @@ cog_spending <- function(govid, years, category = NULL,
category_for_prov <- recipe_label
} else {
view <- .select_view(view_base, resolved$basis)
sql <- .build_verb_sql(view, subtype_col, govid, years, category)
ig_view <- if (identical(expenditure_concept, "total")) {
.select_ig_view(resolved$basis)
} else {
NULL
}
sql <- .build_verb_sql(view, subtype_col, govid, years, category, ig_view)
result <- tibble::as_tibble(DBI::dbGetQuery(con, sql))
}
@@ -211,6 +248,11 @@ cog_spending <- function(govid, years, category = NULL,
if (identical(basis, "harmonized")) paste0(view_base, "_harmonized") else view_base
}
#' @noRd
.select_ig_view <- function(basis) {
if (identical(basis, "harmonized")) "ig_annotated_harmonized" else "ig_annotated"
}
#' @noRd
.sql_lit_chr <- function(x) {
safe <- gsub("'", "''", x, fixed = TRUE)
@@ -218,7 +260,8 @@ cog_spending <- function(govid, years, category = NULL,
}
#' @noRd
.build_verb_sql <- function(view, subtype_col, govid, years, category) {
.build_verb_sql <- function(view, subtype_col, govid, years, category,
ig_view = NULL) {
govid_lit <- .sql_lit_chr(govid)
years_lit <- paste(as.integer(years), collapse = ",")
category_pred <- if (is.null(category)) {
@@ -227,6 +270,16 @@ cog_spending <- function(govid, years, category = NULL,
sprintf("AND category IN (%s)", .sql_lit_chr(category))
}
# expenditure_concept = "total" adds the intergovernmental leg. UNION ALL,
# never UNION: the two legs are disjoint by item_code prefix (E/F/G vs M/L),
# so de-duplication would be pure cost, and a silent row-drop if two
# governments ever reported identical values.
source_expr <- if (is.null(ig_view)) {
view
} else {
sprintf("(SELECT * FROM %s UNION ALL SELECT * FROM %s)", view, ig_view)
}
sprintf(
"SELECT
year,
@@ -243,7 +296,7 @@ cog_spending <- function(govid, years, category = NULL,
%5$s
GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, %1$s, category
ORDER BY year, canonical_govid, %1$s, category",
subtype_col, view, govid_lit, years_lit, category_pred
subtype_col, source_expr, govid_lit, years_lit, category_pred
)
}
+3 -1
View File
@@ -13,11 +13,13 @@
.harmonization_view_files <- c(
"22-spending_long_harmonized.sql",
"23-revenue_long_harmonized.sql",
"25-ig_long_harmonized.sql",
"33-harmonization_map.sql",
"34-harmonization_recipes.sql",
"35-series_breaks_pq.sql",
"42-spending_annotated_harmonized.sql",
"43-revenue_annotated_harmonized.sql"
"43-revenue_annotated_harmonized.sql",
"45-ig_annotated_harmonized.sql"
)
#' Register DuckDB views from inst/sql/ SQL files
+18
View File
@@ -0,0 +1,18 @@
-- Intergovernmental expenditure rows (M = to local govts, L = to state govts).
--
-- Deliberately does NOT filter `NOT is_aggregate`, unlike spending_long. In the
-- wide era (<= FY2011) the IG families M05/M12/M47/M89/L47/L89 are published
-- ONLY as aggregate-flagged rows -- filtering them would hide ~70% of legacy IG
-- dollars and make Total silently collapse to Direct. This is safe because the
-- aggregate codes and their modern leaf components are strictly year-disjoint
-- (M47 ends 2011 / M94 starts 2012; M89 is aggregate only <= 2011 and a leaf
-- from 2012 alongside M91-93), so no row is ever counted twice. Same argument
-- the pipeline's recipe joins use.
--
-- `L--` IS excluded: it is the IG-to-state FAMILY TOTAL and genuinely rolls up
-- the L-NN codes, so including it would double-count.
CREATE OR REPLACE VIEW ig_long AS
SELECT *
FROM long
WHERE LEFT(item_code, 1) IN ('M', 'L')
AND item_code NOT LIKE '%--';
+12
View File
@@ -0,0 +1,12 @@
-- Harmonized-basis IG rows. Uses COALESCE(harmonized_code, item_code) rather
-- than harmonized_code alone: aggregate rows carry NO harmonized_code by
-- construction (harmonized space is leaf-only), so a plain
-- `harmonized_code IS NOT NULL` filter would drop every legacy IG aggregate --
-- 6.4e9 of M and 3.1e8 of L in corpus units. COALESCE keeps the one real IG
-- collapse rule (M38 -> M36, SB012, year-disjoint 1967-2011 vs 2012+) while
-- never dropping a row.
CREATE OR REPLACE VIEW ig_long_harmonized AS
SELECT * REPLACE (COALESCE(harmonized_code, item_code) AS item_code)
FROM long
WHERE LEFT(item_code, 1) IN ('M', 'L')
AND item_code NOT LIKE '%--';
+16
View File
@@ -0,0 +1,16 @@
CREATE OR REPLACE VIEW ig_annotated AS
SELECT
s.*,
x.gov_name AS xwalk_gov_name,
x.govs_type,
x.type_label,
x.fips_state AS xwalk_fips_state,
x.fips_county AS xwalk_fips_county,
x.fips_place,
x.population_acs,
c.category,
c.category_type,
c.spend_subtype
FROM ig_long s
LEFT JOIN canonical_fips_xwalk x USING (canonical_govid)
LEFT JOIN summary_categories c USING (item_code);
+16
View File
@@ -0,0 +1,16 @@
CREATE OR REPLACE VIEW ig_annotated_harmonized AS
SELECT
s.*,
x.gov_name AS xwalk_gov_name,
x.govs_type,
x.type_label,
x.fips_state AS xwalk_fips_state,
x.fips_county AS xwalk_fips_county,
x.fips_place,
x.population_acs,
c.category,
c.category_type,
c.spend_subtype
FROM ig_long_harmonized s
LEFT JOIN canonical_fips_xwalk x USING (canonical_govid)
LEFT JOIN summary_categories c USING (item_code);
+11 -1
View File
@@ -11,7 +11,8 @@ cog_spending(
per_capita = FALSE,
adjust_to_year = NULL,
basis = c("harmonized", "raw"),
recipe = NULL
recipe = NULL,
expenditure_concept = c("direct", "total")
)
}
\arguments{
@@ -55,6 +56,15 @@ argument is ignored and the result's provenance reports
`basis = "recipe"` with an inert `harmonization` block (`applied =
FALSE`, pointing at the `recipe` block instead) rather than a
possibly-misleading `"harmonized"`/`"raw"` value.}
\item{expenditure_concept}{`"direct"` (default) returns only the
government's own direct spending (item codes `E`/`F`/`G`), unchanged
from prior releases. `"total"` additionally UNIONs in the
intergovernmental leg -- payments to local governments (`M` codes) and
to the state government (`L` codes, excluding the `L--` family-total
rollup) -- so results gain rows with `spend_subtype ==
"intergovernmental"`. Mutually exclusive with `recipe` (a recipe
already defines its own component codes).}
}
\value{
Tibble with columns `year`, `canonical_govid`, `gov_name`,
+73
View File
@@ -12,3 +12,76 @@ test_that("the corpus contains no K-prefix rows, so the Direct leg omits K", {
label = paste(f, "must not reference the inert K prefix"))
}
})
test_that("expenditure_concept defaults to direct and preserves today's numbers", {
gov <- "010000226085" # Alabama state government
base <- cog_spending(gov, years = 2019, category = "Police")
expl <- cog_spending(gov, years = 2019, category = "Police",
expenditure_concept = "direct")
expect_equal(base$amt_nominal, expl$amt_nominal)
expect_false("intergovernmental" %in% base$spend_subtype)
})
test_that("expenditure_concept = 'total' adds an intergovernmental subtype", {
gov <- "010000226085"
d <- cog_spending(gov, years = 2019, category = "Police",
expenditure_concept = "direct")
t <- cog_spending(gov, years = 2019, category = "Police",
expenditure_concept = "total")
expect_true("intergovernmental" %in% t$spend_subtype)
# Direct rows are untouched; Total only ever ADDS.
dt <- t[t$spend_subtype != "intergovernmental", ]
expect_equal(sort(dt$amt_nominal), sort(d$amt_nominal))
expect_gt(sum(t$amt_nominal), sum(d$amt_nominal))
})
test_that("legacy-era Total does not collapse to Direct (the is_aggregate trap)", {
# In the wide era the IG dollars live almost entirely on aggregate-flagged
# rows. A Total leg that inherited the Direct leg's NOT is_aggregate filter
# would silently return Total == Direct here.
gov <- "010000226085"
d <- cog_spending(gov, years = 2011, category = "Education K-12",
expenditure_concept = "direct")
t <- cog_spending(gov, years = 2011, category = "Education K-12",
expenditure_concept = "total")
expect_true("intergovernmental" %in% t$spend_subtype)
ig <- sum(t$amt_nominal[t$spend_subtype == "intergovernmental"])
expect_gt(ig, 0)
expect_gt(sum(t$amt_nominal), sum(d$amt_nominal))
})
test_that("the IG leg never includes the L-- family total", {
con <- .ensure_session()
codes <- DBI::dbGetQuery(con,
"SELECT DISTINCT item_code FROM ig_long")$item_code
expect_false(any(grepl("--$", codes)))
expect_true(all(substr(codes, 1, 1) %in% c("M", "L")))
})
test_that("expenditure_concept rejects unknown values", {
expect_error(
cog_spending("010000226085", years = 2019, expenditure_concept = "gross"),
class = "rlang_error"
)
})
test_that("total composes with basis = 'raw' and basis = 'harmonized'", {
gov <- "010000226085"
h <- cog_spending(gov, years = 2011, category = "Education K-12",
expenditure_concept = "total", basis = "harmonized")
r <- cog_spending(gov, years = 2011, category = "Education K-12",
expenditure_concept = "total", basis = "raw")
ig_h <- sum(h$amt_nominal[h$spend_subtype == "intergovernmental"])
ig_r <- sum(r$amt_nominal[r$spend_subtype == "intergovernmental"])
# The only IG harmonization rule is M38 -> M36 (year-disjoint), so the IG
# total must agree between bases even though the code labels may differ.
expect_equal(ig_h, ig_r)
})
test_that("recipe = and expenditure_concept = 'total' together aborts", {
expect_error(
cog_spending("121011212191", 2020L, recipe = "corrections_combined",
expenditure_concept = "total"),
class = "uscogdata_recipe_concept_conflict"
)
})