diff --git a/R/spending.R b/R/spending.R index f662e68..eccb958 100644 --- a/R/spending.R +++ b/R/spending.R @@ -42,6 +42,14 @@ #' `basis = "recipe"` with an inert `harmonization` block (`applied = #' FALSE`, pointing at the `recipe` block instead) rather than a #' possibly-misleading `"harmonized"`/`"raw"` value. +#' @param expenditure_concept `"direct"` (default) returns only the +#' government's own direct spending (item codes `E`/`F`/`G`), unchanged +#' from prior releases. `"total"` additionally UNIONs in the +#' intergovernmental leg -- payments to local governments (`M` codes) and +#' to the state government (`L` codes, excluding the `L--` family-total +#' rollup) -- so results gain rows with `spend_subtype == +#' "intergovernmental"`. Mutually exclusive with `recipe` (a recipe +#' already defines its own component codes). #' @return Tibble with columns `year`, `canonical_govid`, `gov_name`, #' `spend_subtype`, `category`, `amt_nominal`, optional `amt_real`, #' optional `amt_per_capita_nominal`, optional `amt_per_capita_real`, @@ -50,7 +58,8 @@ #' @export cog_spending <- function(govid, years, category = NULL, per_capita = FALSE, adjust_to_year = NULL, - basis = c("harmonized", "raw"), recipe = NULL) { + basis = c("harmonized", "raw"), recipe = NULL, + expenditure_concept = c("direct", "total")) { .verb_spendrev( verb = "cog_spending", view_base = "spending_annotated", @@ -63,7 +72,8 @@ cog_spending <- function(govid, years, category = NULL, per_capita = per_capita, adjust_to_year = adjust_to_year, basis = basis, - recipe = recipe + recipe = recipe, + expenditure_concept = expenditure_concept ) } @@ -71,14 +81,36 @@ cog_spending <- function(govid, years, category = NULL, .verb_spendrev <- function(verb, view_base, subtype_col, flow_prefixes, call, govid, years, category, per_capita, adjust_to_year, - basis = c("harmonized", "raw"), recipe = NULL) { + basis = c("harmonized", "raw"), recipe = NULL, + expenditure_concept = c("direct", "total")) { basis_explicit <- length(basis) == 1L basis <- match.arg(basis, c("harmonized", "raw")) + # match.arg() itself throws a base `simpleError`, not an rlang-classed + # condition; wrap it so an invalid expenditure_concept aborts consistently + # with the rest of this package's validation (cli::cli_abort -> rlang_error). + expenditure_concept <- tryCatch( + match.arg(expenditure_concept, c("direct", "total")), + error = function(e) { + cli::cli_abort( + "`expenditure_concept` must be one of {.val direct} or {.val total}.", + class = "uscogdata_invalid_expenditure_concept", + parent = e + ) + } + ) govid <- .coerce_govid_input(govid, arg = "govid") .validate_verb_inputs(govid, years, category, per_capita, adjust_to_year, recipe) + if (!is.null(recipe) && identical(expenditure_concept, "total")) { + cli::cli_abort(c( + "`recipe` and `expenditure_concept = \"total\"` are mutually exclusive.", + i = "A recipe defines its own component codes; pass one or the other.", + i = "For a recipe's intergovernmental counterpart, use the matching IG recipe (e.g. `corrections_ig_local_combined`)." + ), class = "uscogdata_recipe_concept_conflict") + } + years <- as.integer(years) if (!is.null(adjust_to_year)) adjust_to_year <- as.integer(adjust_to_year) @@ -105,7 +137,12 @@ cog_spending <- function(govid, years, category = NULL, category_for_prov <- recipe_label } else { view <- .select_view(view_base, resolved$basis) - sql <- .build_verb_sql(view, subtype_col, govid, years, category) + ig_view <- if (identical(expenditure_concept, "total")) { + .select_ig_view(resolved$basis) + } else { + NULL + } + sql <- .build_verb_sql(view, subtype_col, govid, years, category, ig_view) result <- tibble::as_tibble(DBI::dbGetQuery(con, sql)) } @@ -211,6 +248,11 @@ cog_spending <- function(govid, years, category = NULL, if (identical(basis, "harmonized")) paste0(view_base, "_harmonized") else view_base } +#' @noRd +.select_ig_view <- function(basis) { + if (identical(basis, "harmonized")) "ig_annotated_harmonized" else "ig_annotated" +} + #' @noRd .sql_lit_chr <- function(x) { safe <- gsub("'", "''", x, fixed = TRUE) @@ -218,7 +260,8 @@ cog_spending <- function(govid, years, category = NULL, } #' @noRd -.build_verb_sql <- function(view, subtype_col, govid, years, category) { +.build_verb_sql <- function(view, subtype_col, govid, years, category, + ig_view = NULL) { govid_lit <- .sql_lit_chr(govid) years_lit <- paste(as.integer(years), collapse = ",") category_pred <- if (is.null(category)) { @@ -227,6 +270,16 @@ cog_spending <- function(govid, years, category = NULL, sprintf("AND category IN (%s)", .sql_lit_chr(category)) } + # expenditure_concept = "total" adds the intergovernmental leg. UNION ALL, + # never UNION: the two legs are disjoint by item_code prefix (E/F/G vs M/L), + # so de-duplication would be pure cost, and a silent row-drop if two + # governments ever reported identical values. + source_expr <- if (is.null(ig_view)) { + view + } else { + sprintf("(SELECT * FROM %s UNION ALL SELECT * FROM %s)", view, ig_view) + } + sprintf( "SELECT year, @@ -243,7 +296,7 @@ cog_spending <- function(govid, years, category = NULL, %5$s GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, %1$s, category ORDER BY year, canonical_govid, %1$s, category", - subtype_col, view, govid_lit, years_lit, category_pred + subtype_col, source_expr, govid_lit, years_lit, category_pred ) } diff --git a/R/views.R b/R/views.R index a5139aa..6a4bdf4 100644 --- a/R/views.R +++ b/R/views.R @@ -13,11 +13,13 @@ .harmonization_view_files <- c( "22-spending_long_harmonized.sql", "23-revenue_long_harmonized.sql", + "25-ig_long_harmonized.sql", "33-harmonization_map.sql", "34-harmonization_recipes.sql", "35-series_breaks_pq.sql", "42-spending_annotated_harmonized.sql", - "43-revenue_annotated_harmonized.sql" + "43-revenue_annotated_harmonized.sql", + "45-ig_annotated_harmonized.sql" ) #' Register DuckDB views from inst/sql/ SQL files diff --git a/inst/sql/24-ig_long.sql b/inst/sql/24-ig_long.sql new file mode 100644 index 0000000..ca29324 --- /dev/null +++ b/inst/sql/24-ig_long.sql @@ -0,0 +1,18 @@ +-- Intergovernmental expenditure rows (M = to local govts, L = to state govts). +-- +-- Deliberately does NOT filter `NOT is_aggregate`, unlike spending_long. In the +-- wide era (<= FY2011) the IG families M05/M12/M47/M89/L47/L89 are published +-- ONLY as aggregate-flagged rows -- filtering them would hide ~70% of legacy IG +-- dollars and make Total silently collapse to Direct. This is safe because the +-- aggregate codes and their modern leaf components are strictly year-disjoint +-- (M47 ends 2011 / M94 starts 2012; M89 is aggregate only <= 2011 and a leaf +-- from 2012 alongside M91-93), so no row is ever counted twice. Same argument +-- the pipeline's recipe joins use. +-- +-- `L--` IS excluded: it is the IG-to-state FAMILY TOTAL and genuinely rolls up +-- the L-NN codes, so including it would double-count. +CREATE OR REPLACE VIEW ig_long AS +SELECT * +FROM long +WHERE LEFT(item_code, 1) IN ('M', 'L') + AND item_code NOT LIKE '%--'; diff --git a/inst/sql/25-ig_long_harmonized.sql b/inst/sql/25-ig_long_harmonized.sql new file mode 100644 index 0000000..08504b2 --- /dev/null +++ b/inst/sql/25-ig_long_harmonized.sql @@ -0,0 +1,12 @@ +-- Harmonized-basis IG rows. Uses COALESCE(harmonized_code, item_code) rather +-- than harmonized_code alone: aggregate rows carry NO harmonized_code by +-- construction (harmonized space is leaf-only), so a plain +-- `harmonized_code IS NOT NULL` filter would drop every legacy IG aggregate -- +-- 6.4e9 of M and 3.1e8 of L in corpus units. COALESCE keeps the one real IG +-- collapse rule (M38 -> M36, SB012, year-disjoint 1967-2011 vs 2012+) while +-- never dropping a row. +CREATE OR REPLACE VIEW ig_long_harmonized AS +SELECT * REPLACE (COALESCE(harmonized_code, item_code) AS item_code) +FROM long +WHERE LEFT(item_code, 1) IN ('M', 'L') + AND item_code NOT LIKE '%--'; diff --git a/inst/sql/44-ig_annotated.sql b/inst/sql/44-ig_annotated.sql new file mode 100644 index 0000000..ead46bd --- /dev/null +++ b/inst/sql/44-ig_annotated.sql @@ -0,0 +1,16 @@ +CREATE OR REPLACE VIEW ig_annotated AS +SELECT + s.*, + x.gov_name AS xwalk_gov_name, + x.govs_type, + x.type_label, + x.fips_state AS xwalk_fips_state, + x.fips_county AS xwalk_fips_county, + x.fips_place, + x.population_acs, + c.category, + c.category_type, + c.spend_subtype +FROM ig_long s +LEFT JOIN canonical_fips_xwalk x USING (canonical_govid) +LEFT JOIN summary_categories c USING (item_code); diff --git a/inst/sql/45-ig_annotated_harmonized.sql b/inst/sql/45-ig_annotated_harmonized.sql new file mode 100644 index 0000000..68df811 --- /dev/null +++ b/inst/sql/45-ig_annotated_harmonized.sql @@ -0,0 +1,16 @@ +CREATE OR REPLACE VIEW ig_annotated_harmonized AS +SELECT + s.*, + x.gov_name AS xwalk_gov_name, + x.govs_type, + x.type_label, + x.fips_state AS xwalk_fips_state, + x.fips_county AS xwalk_fips_county, + x.fips_place, + x.population_acs, + c.category, + c.category_type, + c.spend_subtype +FROM ig_long_harmonized s +LEFT JOIN canonical_fips_xwalk x USING (canonical_govid) +LEFT JOIN summary_categories c USING (item_code); diff --git a/man/cog_spending.Rd b/man/cog_spending.Rd index f1016ec..ce9edbe 100644 --- a/man/cog_spending.Rd +++ b/man/cog_spending.Rd @@ -11,7 +11,8 @@ cog_spending( per_capita = FALSE, adjust_to_year = NULL, basis = c("harmonized", "raw"), - recipe = NULL + recipe = NULL, + expenditure_concept = c("direct", "total") ) } \arguments{ @@ -55,6 +56,15 @@ argument is ignored and the result's provenance reports `basis = "recipe"` with an inert `harmonization` block (`applied = FALSE`, pointing at the `recipe` block instead) rather than a possibly-misleading `"harmonized"`/`"raw"` value.} + +\item{expenditure_concept}{`"direct"` (default) returns only the +government's own direct spending (item codes `E`/`F`/`G`), unchanged +from prior releases. `"total"` additionally UNIONs in the +intergovernmental leg -- payments to local governments (`M` codes) and +to the state government (`L` codes, excluding the `L--` family-total +rollup) -- so results gain rows with `spend_subtype == +"intergovernmental"`. Mutually exclusive with `recipe` (a recipe +already defines its own component codes).} } \value{ Tibble with columns `year`, `canonical_govid`, `gov_name`, diff --git a/tests/testthat/test-expenditure-concept.R b/tests/testthat/test-expenditure-concept.R index 8ae87c0..cf5f86f 100644 --- a/tests/testthat/test-expenditure-concept.R +++ b/tests/testthat/test-expenditure-concept.R @@ -12,3 +12,76 @@ test_that("the corpus contains no K-prefix rows, so the Direct leg omits K", { label = paste(f, "must not reference the inert K prefix")) } }) + +test_that("expenditure_concept defaults to direct and preserves today's numbers", { + gov <- "010000226085" # Alabama state government + base <- cog_spending(gov, years = 2019, category = "Police") + expl <- cog_spending(gov, years = 2019, category = "Police", + expenditure_concept = "direct") + expect_equal(base$amt_nominal, expl$amt_nominal) + expect_false("intergovernmental" %in% base$spend_subtype) +}) + +test_that("expenditure_concept = 'total' adds an intergovernmental subtype", { + gov <- "010000226085" + d <- cog_spending(gov, years = 2019, category = "Police", + expenditure_concept = "direct") + t <- cog_spending(gov, years = 2019, category = "Police", + expenditure_concept = "total") + expect_true("intergovernmental" %in% t$spend_subtype) + # Direct rows are untouched; Total only ever ADDS. + dt <- t[t$spend_subtype != "intergovernmental", ] + expect_equal(sort(dt$amt_nominal), sort(d$amt_nominal)) + expect_gt(sum(t$amt_nominal), sum(d$amt_nominal)) +}) + +test_that("legacy-era Total does not collapse to Direct (the is_aggregate trap)", { + # In the wide era the IG dollars live almost entirely on aggregate-flagged + # rows. A Total leg that inherited the Direct leg's NOT is_aggregate filter + # would silently return Total == Direct here. + gov <- "010000226085" + d <- cog_spending(gov, years = 2011, category = "Education K-12", + expenditure_concept = "direct") + t <- cog_spending(gov, years = 2011, category = "Education K-12", + expenditure_concept = "total") + expect_true("intergovernmental" %in% t$spend_subtype) + ig <- sum(t$amt_nominal[t$spend_subtype == "intergovernmental"]) + expect_gt(ig, 0) + expect_gt(sum(t$amt_nominal), sum(d$amt_nominal)) +}) + +test_that("the IG leg never includes the L-- family total", { + con <- .ensure_session() + codes <- DBI::dbGetQuery(con, + "SELECT DISTINCT item_code FROM ig_long")$item_code + expect_false(any(grepl("--$", codes))) + expect_true(all(substr(codes, 1, 1) %in% c("M", "L"))) +}) + +test_that("expenditure_concept rejects unknown values", { + expect_error( + cog_spending("010000226085", years = 2019, expenditure_concept = "gross"), + class = "rlang_error" + ) +}) + +test_that("total composes with basis = 'raw' and basis = 'harmonized'", { + gov <- "010000226085" + h <- cog_spending(gov, years = 2011, category = "Education K-12", + expenditure_concept = "total", basis = "harmonized") + r <- cog_spending(gov, years = 2011, category = "Education K-12", + expenditure_concept = "total", basis = "raw") + ig_h <- sum(h$amt_nominal[h$spend_subtype == "intergovernmental"]) + ig_r <- sum(r$amt_nominal[r$spend_subtype == "intergovernmental"]) + # The only IG harmonization rule is M38 -> M36 (year-disjoint), so the IG + # total must agree between bases even though the code labels may differ. + expect_equal(ig_h, ig_r) +}) + +test_that("recipe = and expenditure_concept = 'total' together aborts", { + expect_error( + cog_spending("121011212191", 2020L, recipe = "corrections_combined", + expenditure_concept = "total"), + class = "uscogdata_recipe_concept_conflict" + ) +})