diff --git a/R/basis.R b/R/basis.R index 8fa1284..dd60a67 100644 --- a/R/basis.R +++ b/R/basis.R @@ -44,11 +44,18 @@ #' Count + sum item-level rows that basis="harmonized" excludes because they #' carry no harmonized_code (discontinued / not-yet-ruled codes) within the -#' requested flow type (spending or revenue), govids, and years. Only -#' meaningful when the resolved basis is "harmonized"; returns an -#' applied = FALSE stub otherwise (raw basis never excludes rows this way). +#' calling verb's crosswalk scope (`subtype_col` values in `subtype_scope` -- +#' the same subtype-membership classification the verb SQL uses, never +#' item-code prefixes), govids, and years. Only meaningful when the resolved +#' basis is "harmonized"; returns an applied = FALSE stub otherwise (raw +#' basis never excludes rows this way). +#' +#' The intergovernmental leg is deliberately outside this count even for +#' expenditure_concept = "total": ig_long_harmonized COALESCEs rather than +#' drops NULL-harmonized rows, so harmonization never excludes an IG row. #' @noRd -.build_harmonization_block <- function(con, govid, years, resolved, flow_prefixes) { +.build_harmonization_block <- function(con, govid, years, resolved, + subtype_col, subtype_scope) { if (!identical(resolved$basis, "harmonized")) { return(list( applied = FALSE, @@ -63,9 +70,11 @@ FROM long WHERE canonical_govid IN (%s) AND year IN (%s) AND NOT is_aggregate AND harmonized_code IS NULL - AND LEFT(item_code, 1) IN (%s)", + AND item_code IN ( + SELECT item_code FROM summary_categories WHERE %s IN (%s) + )", .sql_lit_chr(govid), paste(as.integer(years), collapse = ","), - .sql_lit_chr(flow_prefixes) + subtype_col, .sql_lit_chr(subtype_scope) ) na <- DBI::dbGetQuery(con, sql) diff --git a/R/complete.R b/R/complete.R index b2c6cfc..a5cfc77 100644 --- a/R/complete.R +++ b/R/complete.R @@ -44,7 +44,9 @@ #' The cells a government-year COULD carry: every code in force for that #' government's own type, mapped through `summary_categories`, restricted to -#' the calling verb's flow prefixes and (when given) its category filter. +#' the calling verb's crosswalk subtype scope (the same subtype-membership +#' classification the verb SQL itself uses -- e.g. the `primary` concept's +#' operations/capital/assistance) and (when given) its category filter. #' #' Scoped by `govs_type` deliberately. Filling against the union of all types #' would invent cells that the government can never report -- a county row for @@ -56,7 +58,7 @@ #' never returns, so every one of them would fill as a phantom $0. #' @noRd .completion_grid_sql <- function(subtype_col, govid, years, category, - flow_prefixes) { + subtype_scope) { category_pred <- if (is.null(category)) { "" } else { @@ -77,13 +79,12 @@ WHERE x.canonical_govid IN (%2$s) AND cs.year IN (%3$s) AND NOT cs.is_aggregate - AND LEFT(cs.item_code, 1) IN (%4$s) AND c.category IS NOT NULL - AND c.%1$s IS NOT NULL + AND c.%1$s IN (%4$s) %5$s", subtype_col, .sql_lit_chr(govid), paste(as.integer(years), collapse = ","), - .sql_lit_chr(flow_prefixes), category_pred + .sql_lit_chr(subtype_scope), category_pred ) } @@ -94,9 +95,9 @@ #' must never alter or drop what the corpus actually published. #' @noRd .complete_result <- function(result, con, subtype_col, govid, years, category, - flow_prefixes) { + subtype_scope) { grid <- tibble::as_tibble(DBI::dbGetQuery( - con, .completion_grid_sql(subtype_col, govid, years, category, flow_prefixes) + con, .completion_grid_sql(subtype_col, govid, years, category, subtype_scope) )) result$value_source <- rep("reported", nrow(result)) diff --git a/R/peers.R b/R/peers.R index 3b71f12..292453e 100644 --- a/R/peers.R +++ b/R/peers.R @@ -176,10 +176,11 @@ cog_find_peers <- function(target_govid, #' @param per_capita Default `TRUE` — peer compare usually normalizes by #' population. #' @param adjust_to_year Integer base year for CPI-U conversion or `NULL`. -#' @param expenditure_concept `"direct"` (default) or `"total"`. Currently only -#' `"direct"` is accepted; the `"total"` option exists in [cog_spending()] for -#' single-government queries but cannot be used here because combining Total -#' across peer sets counts intergovernmental transfers twice. +#' @param expenditure_concept `"primary"` (default), `"direct"`, or +#' `"total"` -- see [cog_spending()] for the three concepts. `"total"` is +#' refused here because combining Total across peer sets counts +#' intergovernmental transfers twice; `"primary"` and `"direct"` combine +#' safely. #' @param coverage How to handle the Census of Governments survey cycle, #' which is a **complete census only in years ending in 2 and 7** -- every #' other year is a sample, and the sample varies enormously (on the bundled @@ -242,7 +243,7 @@ cog_find_peers <- function(target_govid, #' @export cog_peer_compare <- function(target_govid, peers, category, years, per_capita = TRUE, adjust_to_year = NULL, - expenditure_concept = c("direct", "total"), + expenditure_concept = c("primary", "direct", "total"), coverage = c("all", "census", "consistent")) { call <- match.call() expenditure_concept <- match.arg(expenditure_concept) @@ -271,7 +272,8 @@ cog_peer_compare <- function(target_govid, peers, category, years, years <- .apply_census_years(years, coverage, "cog_peer_compare") - r <- cog_spending(all_govids, years, category, per_capita, adjust_to_year) + r <- cog_spending(all_govids, years, category, per_capita, adjust_to_year, + expenditure_concept = expenditure_concept) r$role <- ifelse(r$canonical_govid == target_govid, "target", "peer") # The target is exempt from balancing: it is the subject of the comparison, diff --git a/R/revenue.R b/R/revenue.R index eb11dd0..49f7365 100644 --- a/R/revenue.R +++ b/R/revenue.R @@ -18,6 +18,10 @@ cog_revenue <- function(govid, years, category = NULL, per_capita = FALSE, adjust_to_year = NULL, basis = c("harmonized", "raw"), recipe = NULL, complete = FALSE) { + # flow_prefixes no longer classifies rows (crosswalk revenue_subtype + # membership does -- General Revenue, i.e. everything except + # insurance_trust) -- it only scopes the recipe-suggestion machinery to + # this verb's recipe families (see R/suggestions.R). .verb_spendrev( verb = "cog_revenue", view_base = "revenue_annotated", diff --git a/R/rollup.R b/R/rollup.R index 7534595..3b7d5c1 100644 --- a/R/rollup.R +++ b/R/rollup.R @@ -25,12 +25,12 @@ #' population from `gov_population_yearly`. Govs with missing population #' are excluded from the result. #' @param adjust_to_year Integer base year for CPI-U conversion, or `NULL`. -#' @param expenditure_concept `"direct"` (default) or `"total"`. Currently only -#' `"direct"` is accepted; the `"total"` option exists in [cog_spending()] for -#' single-government queries but cannot be used here because combining Total -#' across multiple layers of government double-counts intergovernmental -#' transfers (a state's payment to a school district is the same dollar the -#' district reports as its own Direct spending). +#' @param expenditure_concept `"primary"` (default), `"direct"`, or +#' `"total"` -- see [cog_spending()] for the three concepts. `"total"` is +#' refused here because combining Total across multiple layers of +#' government double-counts intergovernmental transfers (a state's payment +#' to a school district is the same dollar the district reports as its own +#' Direct spending); `"primary"` and `"direct"` combine safely. #' @param coverage How to handle the Census of Governments survey cycle, #' which is a **complete census only in years ending in 2 and 7** -- every #' other year is a sample, and the sample varies enormously (on the bundled @@ -59,7 +59,7 @@ #' @export cog_geographic_rollup <- function(govids, category, years, per_capita = FALSE, adjust_to_year = NULL, - expenditure_concept = c("direct", "total"), + expenditure_concept = c("primary", "direct", "total"), coverage = c("all", "census", "consistent")) { call <- match.call() expenditure_concept <- match.arg(expenditure_concept) @@ -85,7 +85,8 @@ cog_geographic_rollup <- function(govids, category, years, # to discard them would also let them into the coverage table. years <- .apply_census_years(years, coverage, "cog_geographic_rollup") - r <- cog_spending(all_govids, years, category, per_capita, adjust_to_year) + r <- cog_spending(all_govids, years, category, per_capita, adjust_to_year, + expenditure_concept = expenditure_concept) r <- dplyr::left_join(r, layer_map, by = "canonical_govid", relationship = "many-to-many") r$scope_note <- .rollup_scope_note(r$layer) diff --git a/R/spending.R b/R/spending.R index 7ab4b5f..c56196f 100644 --- a/R/spending.R +++ b/R/spending.R @@ -1,5 +1,40 @@ # R/spending.R +# The three expenditure concepts (uscogdata#11), as sets of the crosswalk's +# `spend_subtype` values. Classification is crosswalk membership, never +# item-code first letters: prefix Y alone spans revenue (Y01/Y02), +# expenditure (Y05/Y06) and balance codes, so no first-letter allowlist can +# route it (finding F-018). +# +# primary = operations + capital + assistance (the default) +# direct = primary + interest + insurance_benefits (Census Direct Expenditure) +# total = direct + intergovernmental (via the ig_* views) +# +# Census manual section 5.2.2.1: Direct Expenditure is ALL expenditure other +# than intergovernmental -- including payments to retirees, i.e. insurance +# trust benefits. Verified against Census's own published FY2020 state +# aggregates (20statetypepu.txt): `total` reproduces the published +# expenditure sum to the dollar; omitting insurance benefits understates +# California's Direct by 10.9%. +.spend_subtypes_primary <- c("operations", "capital", "assistance") +.spend_subtypes_direct <- c(.spend_subtypes_primary, "interest", "insurance_benefits") + +#' @noRd +.expenditure_concept_subtypes <- function(concept) { + switch(concept, + primary = .spend_subtypes_primary, + # "total" = the direct subtypes here PLUS the intergovernmental leg, + # which travels through the ig_* views rather than this scope (see + # .build_verb_sql()). + direct = , + total = .spend_subtypes_direct + ) +} + +# cog_revenue()'s single concept (until uscogdata#12 adds more): Census +# General Revenue -- every crosswalk revenue subtype except insurance_trust. +.revenue_subtypes_general <- c("own_source", "federal", "state", "local_aid") + #' Summarized spending by category #' #' One row per `(year, canonical_govid, spend_subtype, category)`. Amounts are @@ -42,22 +77,34 @@ #' `basis = "recipe"` with an inert `harmonization` block (`applied = #' FALSE`, pointing at the `recipe` block instead) rather than a #' possibly-misleading `"harmonized"`/`"raw"` value. -#' @param expenditure_concept `"direct"` (default) returns only the -#' government's own direct spending (item codes `E`/`F`/`G`), unchanged -#' from prior releases. `"total"` additionally UNIONs in the -#' intergovernmental leg -- payments to local governments (`M` codes) and -#' to the state government (`L` codes, excluding the `L--` family-total -#' rollup) -- so results gain rows with `spend_subtype == -#' "intergovernmental"`. Requires the active corpus's `summary_categories` -#' to carry M/L rows (added by cog_pipeline PR #59); aborts with class -#' `uscogdata_ig_categories_unsupported` on an older corpus rather than -#' silently under-reporting. Mutually exclusive with `recipe` (a recipe -#' already defines its own component codes). **Do not sum `"total"` -#' results across levels of government** (e.g. state + county + city): -#' a state's `M12` payment to a school district is the same dollar the -#' district reports as its own direct `E12`, so summing both double-counts -#' it. This matters in particular with [cog_geographic_rollup()], which -#' sums across exactly that kind of multi-layer government set. +#' @param expenditure_concept Which spending concept to return. Concepts are +#' defined as sets of the crosswalk's `spend_subtype` values -- never as +#' item-code first letters, which cannot classify correctly (prefix `Y` +#' alone spans revenue, expenditure, and balance codes): +#' +#' * `"primary"` (default) -- the government's own service provision: +#' `operations` + `capital` + `assistance` subtypes. +#' * `"direct"` -- Census's published Direct Expenditure: `primary` plus +#' `interest` (interest on debt) and `insurance_benefits` (insurance +#' trust benefit payments, e.g. pensions -- Census manual section +#' 5.2.2.1 includes payments to retirees in Direct). +#' * `"total"` -- `direct` plus the intergovernmental leg: payments to +#' local governments (`M` codes), to the state government (`L` codes, +#' excluding the `L--` family-total rollup), and state payments to +#' school systems (`Q11`/`Q12`/`Q18`), so results gain rows with +#' `spend_subtype == "intergovernmental"`. Requires the active corpus's +#' `summary_categories` to carry M/L rows (added by cog_pipeline PR +#' #59); aborts with class `uscogdata_ig_categories_unsupported` on an +#' older corpus rather than silently under-reporting. Mutually +#' exclusive with `recipe` (a recipe already defines its own component +#' codes). +#' +#' **Do not sum `"total"` results across levels of government** (e.g. +#' state + county + city): a state's `M12` payment to a school district is +#' the same dollar the district reports as its own direct `E12`, so +#' summing both double-counts it. This matters in particular with +#' [cog_geographic_rollup()], which sums across exactly that kind of +#' multi-layer government set. #' #' In the legacy wide era (<= FY2011), some functions are published ONLY #' as an aggregate-flagged family total (e.g. Corrections' `E04`/`E05` @@ -104,8 +151,12 @@ cog_spending <- function(govid, years, category = NULL, per_capita = FALSE, adjust_to_year = NULL, basis = c("harmonized", "raw"), recipe = NULL, - expenditure_concept = c("direct", "total"), + expenditure_concept = c("primary", "direct", "total"), complete = FALSE) { + # flow_prefixes no longer classifies rows (crosswalk subtype membership + # does, per expenditure_concept) -- it only scopes the recipe-suggestion + # machinery to this verb's recipe families (see R/suggestions.R; the + # catalog only has E/F/G-component direct-expenditure recipes). .verb_spendrev( verb = "cog_spending", view_base = "spending_annotated", @@ -128,8 +179,9 @@ cog_spending <- function(govid, years, category = NULL, .abort_concept_not_aggregatable <- function(verb) { cli::cli_abort(c( "{.code expenditure_concept = \"total\"} cannot be used in {.fn {verb}}.", - "*" = "Use {.code expenditure_concept = \"direct\"} (the default) for any \\ - comparison or sum that spans more than one government.", + "*" = "Use {.code expenditure_concept = \"primary\"} (the default) or \\ + {.code \"direct\"} for any comparison or sum that spans more than \\ + one government.", "i" = "Why: Census \"Total\" is a government's own Direct spending PLUS the \\ money it hands to other governments. The receiving government reports \\ that same dollar again as its own Direct when it actually spends it, \\ @@ -145,7 +197,7 @@ cog_spending <- function(govid, years, category = NULL, govid, years, category, per_capita, adjust_to_year, basis = c("harmonized", "raw"), recipe = NULL, - expenditure_concept = c("direct", "total"), + expenditure_concept = c("primary", "direct", "total"), complete = FALSE) { basis_explicit <- length(basis) == 1L basis <- match.arg(basis, c("harmonized", "raw")) @@ -153,16 +205,28 @@ cog_spending <- function(govid, years, category = NULL, # condition; wrap it so an invalid expenditure_concept aborts consistently # with the rest of this package's validation (cli::cli_abort -> rlang_error). expenditure_concept <- tryCatch( - match.arg(expenditure_concept, c("direct", "total")), + match.arg(expenditure_concept, c("primary", "direct", "total")), error = function(e) { cli::cli_abort( - "`expenditure_concept` must be one of {.val direct} or {.val total}.", + "`expenditure_concept` must be one of {.val primary}, {.val direct}, or {.val total}.", class = "uscogdata_invalid_expenditure_concept", parent = e ) } ) + # The concept's subtype scope. Every code path below -- the verb SQL, the + # harmonization exclusion count, and the complete = TRUE grid -- is scoped + # by crosswalk subtype membership, never by item-code prefix. For revenue + # there is a single concept today (General Revenue; uscogdata#12 will add + # more). "total"'s extra intergovernmental leg travels through the ig_* + # views, not through this scope. + subtype_scope <- if (identical(subtype_col, "spend_subtype")) { + .expenditure_concept_subtypes(expenditure_concept) + } else { + .revenue_subtypes_general + } + govid <- .coerce_govid_input(govid, arg = "govid") .validate_verb_inputs(govid, years, category, per_capita, adjust_to_year, recipe) @@ -176,10 +240,10 @@ cog_spending <- function(govid, years, category = NULL, } # .verb_spendrev() is shared with cog_revenue(), which never exposes - # expenditure_concept and always resolves it to "direct" -- so nothing on - # the public API can reach this today. But it's a cheap guard against a + # expenditure_concept and always resolves it to the default -- so nothing + # on the public API can reach this today. But it's a cheap guard against a # future call (direct or via a modified cog_revenue()) that would UNION - # the IG leg's expenditure M/L rows into a revenue result, which has no + # the IG leg's expenditure M/L/Q rows into a revenue result, which has no # matching IG view and no sensible meaning. if (identical(expenditure_concept, "total") && !identical(view_base, "spending_annotated")) { @@ -240,7 +304,8 @@ cog_spending <- function(govid, years, category = NULL, } else { NULL } - sql <- .build_verb_sql(view, subtype_col, govid, years, category, ig_view) + sql <- .build_verb_sql(view, subtype_col, govid, years, category, ig_view, + subtype_scope) result <- tibble::as_tibble(DBI::dbGetQuery(con, sql)) } @@ -251,7 +316,7 @@ cog_spending <- function(govid, years, category = NULL, completion <- list(applied = FALSE, rows_filled = 0L, absence_means = list()) if (complete) { result <- .complete_result(result, con, subtype_col, govid, years, - category, flow_prefixes) + category, subtype_scope) completion <- attr(result, ".completion") attr(result, ".completion") <- NULL } @@ -284,7 +349,7 @@ cog_spending <- function(govid, years, category = NULL, basis_for_prov <- resolved$basis basis_note_for_prov <- resolved$note harmonization <- .build_harmonization_block( - con, govid, years, resolved, flow_prefixes + con, govid, years, resolved, subtype_col, subtype_scope ) # C1(a): gap detection must run against the Direct leg alone. `result` # can also carry UNION'd intergovernmental rows (expenditure_concept = @@ -462,7 +527,7 @@ cog_spending <- function(govid, years, category = NULL, #' @noRd .build_verb_sql <- function(view, subtype_col, govid, years, category, - ig_view = NULL) { + ig_view = NULL, subtype_scope = NULL) { govid_lit <- .sql_lit_chr(govid) years_lit <- paste(as.integer(years), collapse = ",") category_pred <- if (is.null(category)) { @@ -471,9 +536,22 @@ cog_spending <- function(govid, years, category = NULL, sprintf("AND category IN (%s)", .sql_lit_chr(category)) } + # The concept's subtype allowlist (see .expenditure_concept_subtypes()). + # The base views carry every subtype of their flow (spending_annotated has + # all five non-IG expenditure subtypes); the concept narrows here. For + # "total", the IG leg's rows are 'intergovernmental', so that value joins + # the allowlist exactly when ig_view is present. + subtype_pred <- if (is.null(subtype_scope)) { + "" + } else { + scope <- if (is.null(ig_view)) subtype_scope else c(subtype_scope, "intergovernmental") + sprintf("AND %s IN (%s)", subtype_col, .sql_lit_chr(scope)) + } + # expenditure_concept = "total" adds the intergovernmental leg. UNION ALL, - # never UNION: the two legs are disjoint by item_code prefix (E/F/G vs M/L), - # so de-duplication would be pure cost, and a silent row-drop if two + # never UNION: the two legs are disjoint by crosswalk subtype (the direct + # view excludes 'intergovernmental'; the IG view is only that), so + # de-duplication would be pure cost, and a silent row-drop if two # governments ever reported identical values. source_expr <- if (is.null(ig_view)) { view @@ -505,9 +583,10 @@ cog_spending <- function(govid, years, category = NULL, WHERE canonical_govid IN (%3$s) AND year IN (%4$s) %5$s + %6$s GROUP BY year, canonical_govid, gov_name, xwalk_gov_name, %1$s, category ORDER BY year, canonical_govid, %1$s, category", - subtype_col, source_expr, govid_lit, years_lit, category_pred + subtype_col, source_expr, govid_lit, years_lit, category_pred, subtype_pred ) } diff --git a/README.md b/README.md index a568cd6..8d2f2d9 100644 --- a/README.md +++ b/README.md @@ -46,20 +46,28 @@ than obviously wrong. - `USCOGDATA_CACHE_DIR` — optional override for the manifest cache directory - `USCOGDATA_MANIFEST_TTL_SECS` — optional manifest re-fetch TTL (default 3600) -## Direct vs Total spending +## Primary vs Direct vs Total spending -`cog_spending(..., expenditure_concept = c("direct", "total"))` controls -whose spending a result counts. `"direct"` (the default) is a government's -own current operations, capital outlay, and other direct spending. `"total"` -additionally adds in the intergovernmental legs — money it hands to other -governments to spend on its behalf — which is meaningful for describing one -government's own budget over time, but double-counts when summed across -governments (a state's payment to a county is the same dollar the county -reports as its own direct spending). +`cog_spending(..., expenditure_concept = c("primary", "direct", "total"))` +controls whose spending a result counts. Concepts are defined as sets of the +crosswalk's `spend_subtype` values — never item-code first letters, which +cannot classify correctly (the letter `Y` alone spans revenue, expenditure, +and balance codes): + +- `"primary"` (the default) is the government's own service provision: + current operations, capital outlay, and assistance payments. +- `"direct"` is Census's published Direct Expenditure: `primary` plus + interest on debt and insurance trust benefit payments (e.g. pensions). +- `"total"` additionally adds the intergovernmental leg — money handed to + other governments to spend (`M`/`L` codes plus `Q11`/`Q12`/`Q18` state + payments to school systems) — which is meaningful for describing one + government's own budget over time, but double-counts when summed across + governments (a state's payment to a county is the same dollar the county + reports as its own direct spending). **Rule of thumb: any figure that spans more than one government uses -`direct`.** `cog_geographic_rollup()` and `cog_peer_compare()` enforce this -by refusing `expenditure_concept = "total"`. See +`primary` or `direct`.** `cog_geographic_rollup()` and `cog_peer_compare()` +enforce this by refusing `expenditure_concept = "total"`. See `vignette("total-spending", package = "uscogdata")` for the full explanation with worked examples. diff --git a/inst/schemas/provenance-v1.json b/inst/schemas/provenance-v1.json index 6307533..9f3973b 100644 --- a/inst/schemas/provenance-v1.json +++ b/inst/schemas/provenance-v1.json @@ -14,16 +14,16 @@ "basis_note": { "type": ["string", "null"] }, "expenditure_concept": { "type": "string", - "enum": ["direct", "total"], - "description": "Which spending concept produced this result. 'direct' is the government's own E/F/G spending; 'total' adds its intergovernmental payments (M to local governments, L to state governments). Only 'direct' is valid for results combined across governments." + "enum": ["primary", "direct", "total"], + "description": "Which spending concept produced this result, defined as crosswalk spend_subtype sets (never item-code prefixes). 'primary' (the default) is the government's own service provision: operations + capital + assistance. 'direct' adds interest on debt and insurance trust benefit payments (Census's published Direct Expenditure). 'total' adds intergovernmental payments (M to local governments, L to state government, Q11/Q12/Q18 to school systems). Only 'primary' and 'direct' are valid for results combined across governments." }, "expenditure_concept_note": { "type": ["string", "null"], - "description": "How the intergovernmental leg was assembled; null for 'direct'." + "description": "How the intergovernmental leg was assembled; null for 'primary' and 'direct'." }, "expenditure_concept_direct_suppressed": { "type": "boolean", - "description": "TRUE when expenditure_concept = 'total' and at least one requested (year, category) has intergovernmental rows but NO Direct rows in this corpus (typically a legacy aggregate-only family) -- those result rows report the intergovernmental leg alone, not Direct + IG. Always FALSE for expenditure_concept = 'direct'. See the affected rows' `notes` for the recovering recipe, if any." + "description": "TRUE when expenditure_concept = 'total' and at least one requested (year, category) has intergovernmental rows but NO Direct rows in this corpus (typically a legacy aggregate-only family) -- those result rows report the intergovernmental leg alone, not Direct + IG. Always FALSE for expenditure_concept = 'primary' or 'direct'. See the affected rows' `notes` for the recovering recipe, if any." }, "harmonization": { "type": "object" }, "recipe": { "type": ["object", "null"] }, diff --git a/inst/sql/11-summary_categories.sql b/inst/sql/11-summary_categories.sql new file mode 100644 index 0000000..d7c4bd0 --- /dev/null +++ b/inst/sql/11-summary_categories.sql @@ -0,0 +1,7 @@ +-- Category crosswalk. Numbered 11 (not with the other reference tables at +-- 30+) because the flow views (20-25) classify by MEMBERSHIP in this table +-- and DuckDB binds a view's sources eagerly at CREATE VIEW time, so it must +-- already exist when they register. +CREATE OR REPLACE VIEW summary_categories AS +SELECT * +FROM read_parquet('{url}data/summary_categories.parquet'); diff --git a/inst/sql/20-spending_long.sql b/inst/sql/20-spending_long.sql index 23c4d82..0fbdaf7 100644 --- a/inst/sql/20-spending_long.sql +++ b/inst/sql/20-spending_long.sql @@ -1,5 +1,22 @@ +-- Direct-side expenditure rows, classified by crosswalk MEMBERSHIP +-- (summary_categories.category_type = 'expenditure'), never by item-code +-- first letter: prefix Y alone spans revenue (Y01/Y02), expenditure +-- (Y05/Y06) and balance codes, so no first-letter allowlist can route it +-- (uscogdata#11, finding F-018). Which subtypes a query actually returns is +-- decided per expenditure_concept in R (.verb_spendrev); this view carries +-- every non-intergovernmental expenditure subtype: operations, capital, +-- assistance, interest, insurance_benefits. +-- +-- The intergovernmental subtype (M/L/Q codes) is deliberately carved out +-- into ig_long: its legacy-era rows are published ONLY as aggregate-flagged +-- rows, so it cannot live behind this view's NOT is_aggregate filter (see +-- 24-ig_long.sql). CREATE OR REPLACE VIEW spending_long AS SELECT * FROM long -WHERE LEFT(item_code, 1) IN ('E', 'F', 'G') +WHERE item_code IN ( + SELECT item_code FROM summary_categories + WHERE category_type = 'expenditure' + AND spend_subtype <> 'intergovernmental' + ) AND NOT is_aggregate; diff --git a/inst/sql/21-revenue_long.sql b/inst/sql/21-revenue_long.sql index 013a701..61e6c88 100644 --- a/inst/sql/21-revenue_long.sql +++ b/inst/sql/21-revenue_long.sql @@ -1,5 +1,15 @@ +-- Revenue rows, classified by crosswalk MEMBERSHIP rather than item-code +-- first letter (see 20-spending_long.sql for why prefixes cannot work). +-- Scope is Census General Revenue: every crosswalk revenue subtype EXCEPT +-- insurance_trust (Y01/Y02/Y04/Y11/Y12/Y51/Y52). Owner ruling 2026-07-30: +-- the default revenue concept stays general; surfacing insurance-trust +-- revenue through an explicit concept argument is uscogdata#12. CREATE OR REPLACE VIEW revenue_long AS SELECT * FROM long -WHERE LEFT(item_code, 1) IN ('T', 'A', 'U', 'B', 'C', 'D') +WHERE item_code IN ( + SELECT item_code FROM summary_categories + WHERE category_type = 'revenue' + AND revenue_subtype <> 'insurance_trust' + ) AND NOT is_aggregate; diff --git a/inst/sql/22-spending_long_harmonized.sql b/inst/sql/22-spending_long_harmonized.sql index 96baa58..d388c4b 100644 --- a/inst/sql/22-spending_long_harmonized.sql +++ b/inst/sql/22-spending_long_harmonized.sql @@ -1,6 +1,16 @@ +-- Harmonized-basis twin of 20-spending_long.sql: same crosswalk-membership +-- classification, applied to harmonized_code (the code the row is folded +-- onto) rather than the published item_code. Safe because the harmonized +-- space is leaf-only and every harmonized_code in the corpus is a +-- summary_categories member (verified at fixture regen; a code the +-- crosswalk cannot classify would be silently dropped here). CREATE OR REPLACE VIEW spending_long_harmonized AS SELECT * REPLACE (harmonized_code AS item_code) FROM long WHERE NOT is_aggregate AND harmonized_code IS NOT NULL - AND LEFT(harmonized_code, 1) IN ('E', 'F', 'G'); + AND harmonized_code IN ( + SELECT item_code FROM summary_categories + WHERE category_type = 'expenditure' + AND spend_subtype <> 'intergovernmental' + ); diff --git a/inst/sql/23-revenue_long_harmonized.sql b/inst/sql/23-revenue_long_harmonized.sql index 69a4aa1..3a5492c 100644 --- a/inst/sql/23-revenue_long_harmonized.sql +++ b/inst/sql/23-revenue_long_harmonized.sql @@ -1,6 +1,13 @@ +-- Harmonized-basis twin of 21-revenue_long.sql: same crosswalk-membership +-- classification (General Revenue = revenue minus insurance_trust), applied +-- to harmonized_code rather than the published item_code. CREATE OR REPLACE VIEW revenue_long_harmonized AS SELECT * REPLACE (harmonized_code AS item_code) FROM long WHERE NOT is_aggregate AND harmonized_code IS NOT NULL - AND LEFT(harmonized_code, 1) IN ('T', 'A', 'U', 'B', 'C', 'D'); + AND harmonized_code IN ( + SELECT item_code FROM summary_categories + WHERE category_type = 'revenue' + AND revenue_subtype <> 'insurance_trust' + ); diff --git a/inst/sql/24-ig_long.sql b/inst/sql/24-ig_long.sql index ca29324..cd451d8 100644 --- a/inst/sql/24-ig_long.sql +++ b/inst/sql/24-ig_long.sql @@ -1,4 +1,6 @@ --- Intergovernmental expenditure rows (M = to local govts, L = to state govts). +-- Intergovernmental expenditure rows: crosswalk spend_subtype = +-- 'intergovernmental' (M = to local govts, L = to state govts, Q11/Q12/Q18 +-- = state payments to school systems -- uscogdata#11, finding F-017). -- -- Deliberately does NOT filter `NOT is_aggregate`, unlike spending_long. In the -- wide era (<= FY2011) the IG families M05/M12/M47/M89/L47/L89 are published @@ -9,10 +11,15 @@ -- from 2012 alongside M91-93), so no row is ever counted twice. Same argument -- the pipeline's recipe joins use. -- --- `L--` IS excluded: it is the IG-to-state FAMILY TOTAL and genuinely rolls up --- the L-NN codes, so including it would double-count. +-- `L--` stays excluded: it is the IG-to-state FAMILY TOTAL and genuinely +-- rolls up the L-NN codes, so including it would double-count. The crosswalk +-- deliberately carries no `--` family-total codes, so membership excludes it +-- (guarded by "the IG leg never includes the L-- family total" in +-- tests/testthat/test-expenditure-concept.R). CREATE OR REPLACE VIEW ig_long AS SELECT * FROM long -WHERE LEFT(item_code, 1) IN ('M', 'L') - AND item_code NOT LIKE '%--'; +WHERE item_code IN ( + SELECT item_code FROM summary_categories + WHERE spend_subtype = 'intergovernmental' + ); diff --git a/inst/sql/25-ig_long_harmonized.sql b/inst/sql/25-ig_long_harmonized.sql index 5187265..62e5da0 100644 --- a/inst/sql/25-ig_long_harmonized.sql +++ b/inst/sql/25-ig_long_harmonized.sql @@ -8,8 +8,15 @@ -- WHERE harmonized_code IS NULL GROUP BY 1, 2`). COALESCE keeps the one real -- IG collapse rule (M38 -> M36, SB012, year-disjoint 1967-2011 vs 2012+) -- while never dropping a row. +-- +-- Membership is checked on the published item_code (mirroring 24-ig_long.sql) +-- rather than the COALESCEd code: every IG harmonization target (M36) is +-- itself an IG crosswalk member, so the two are equivalent, and item_code is +-- the column that exists on every row. CREATE OR REPLACE VIEW ig_long_harmonized AS SELECT * REPLACE (COALESCE(harmonized_code, item_code) AS item_code) FROM long -WHERE LEFT(item_code, 1) IN ('M', 'L') - AND item_code NOT LIKE '%--'; +WHERE item_code IN ( + SELECT item_code FROM summary_categories + WHERE spend_subtype = 'intergovernmental' + ); diff --git a/inst/sql/31-summary_categories.sql b/inst/sql/31-summary_categories.sql deleted file mode 100644 index 0fa10df..0000000 --- a/inst/sql/31-summary_categories.sql +++ /dev/null @@ -1,3 +0,0 @@ -CREATE OR REPLACE VIEW summary_categories AS -SELECT * -FROM read_parquet('{url}data/summary_categories.parquet'); diff --git a/man/cog_geographic_rollup.Rd b/man/cog_geographic_rollup.Rd index 14ffc2f..497fde8 100644 --- a/man/cog_geographic_rollup.Rd +++ b/man/cog_geographic_rollup.Rd @@ -10,7 +10,7 @@ cog_geographic_rollup( years, per_capita = FALSE, adjust_to_year = NULL, - expenditure_concept = c("direct", "total"), + expenditure_concept = c("primary", "direct", "total"), coverage = c("all", "census", "consistent") ) } @@ -30,12 +30,12 @@ are excluded from the result.} \item{adjust_to_year}{Integer base year for CPI-U conversion, or `NULL`.} -\item{expenditure_concept}{`"direct"` (default) or `"total"`. Currently only -`"direct"` is accepted; the `"total"` option exists in [cog_spending()] for -single-government queries but cannot be used here because combining Total -across multiple layers of government double-counts intergovernmental -transfers (a state's payment to a school district is the same dollar the -district reports as its own Direct spending).} +\item{expenditure_concept}{`"primary"` (default), `"direct"`, or +`"total"` -- see [cog_spending()] for the three concepts. `"total"` is +refused here because combining Total across multiple layers of +government double-counts intergovernmental transfers (a state's payment +to a school district is the same dollar the district reports as its own +Direct spending); `"primary"` and `"direct"` combine safely.} \item{coverage}{How to handle the Census of Governments survey cycle, which is a **complete census only in years ending in 2 and 7** -- every diff --git a/man/cog_peer_compare.Rd b/man/cog_peer_compare.Rd index a7824f6..bd8e21a 100644 --- a/man/cog_peer_compare.Rd +++ b/man/cog_peer_compare.Rd @@ -11,7 +11,7 @@ cog_peer_compare( years, per_capita = TRUE, adjust_to_year = NULL, - expenditure_concept = c("direct", "total"), + expenditure_concept = c("primary", "direct", "total"), coverage = c("all", "census", "consistent") ) } @@ -30,10 +30,11 @@ population.} \item{adjust_to_year}{Integer base year for CPI-U conversion or `NULL`.} -\item{expenditure_concept}{`"direct"` (default) or `"total"`. Currently only -`"direct"` is accepted; the `"total"` option exists in [cog_spending()] for -single-government queries but cannot be used here because combining Total -across peer sets counts intergovernmental transfers twice.} +\item{expenditure_concept}{`"primary"` (default), `"direct"`, or +`"total"` -- see [cog_spending()] for the three concepts. `"total"` is +refused here because combining Total across peer sets counts +intergovernmental transfers twice; `"primary"` and `"direct"` combine +safely.} \item{coverage}{How to handle the Census of Governments survey cycle, which is a **complete census only in years ending in 2 and 7** -- every diff --git a/man/cog_spending.Rd b/man/cog_spending.Rd index 9638efa..7a958b9 100644 --- a/man/cog_spending.Rd +++ b/man/cog_spending.Rd @@ -12,7 +12,7 @@ cog_spending( adjust_to_year = NULL, basis = c("harmonized", "raw"), recipe = NULL, - expenditure_concept = c("direct", "total"), + expenditure_concept = c("primary", "direct", "total"), complete = FALSE ) } @@ -58,22 +58,34 @@ argument is ignored and the result's provenance reports FALSE`, pointing at the `recipe` block instead) rather than a possibly-misleading `"harmonized"`/`"raw"` value.} -\item{expenditure_concept}{`"direct"` (default) returns only the - government's own direct spending (item codes `E`/`F`/`G`), unchanged - from prior releases. `"total"` additionally UNIONs in the - intergovernmental leg -- payments to local governments (`M` codes) and - to the state government (`L` codes, excluding the `L--` family-total - rollup) -- so results gain rows with `spend_subtype == - "intergovernmental"`. Requires the active corpus's `summary_categories` - to carry M/L rows (added by cog_pipeline PR #59); aborts with class - `uscogdata_ig_categories_unsupported` on an older corpus rather than - silently under-reporting. Mutually exclusive with `recipe` (a recipe - already defines its own component codes). **Do not sum `"total"` - results across levels of government** (e.g. state + county + city): - a state's `M12` payment to a school district is the same dollar the - district reports as its own direct `E12`, so summing both double-counts - it. This matters in particular with [cog_geographic_rollup()], which - sums across exactly that kind of multi-layer government set. +\item{expenditure_concept}{Which spending concept to return. Concepts are + defined as sets of the crosswalk's `spend_subtype` values -- never as + item-code first letters, which cannot classify correctly (prefix `Y` + alone spans revenue, expenditure, and balance codes): + + * `"primary"` (default) -- the government's own service provision: + `operations` + `capital` + `assistance` subtypes. + * `"direct"` -- Census's published Direct Expenditure: `primary` plus + `interest` (interest on debt) and `insurance_benefits` (insurance + trust benefit payments, e.g. pensions -- Census manual section + 5.2.2.1 includes payments to retirees in Direct). + * `"total"` -- `direct` plus the intergovernmental leg: payments to + local governments (`M` codes), to the state government (`L` codes, + excluding the `L--` family-total rollup), and state payments to + school systems (`Q11`/`Q12`/`Q18`), so results gain rows with + `spend_subtype == "intergovernmental"`. Requires the active corpus's + `summary_categories` to carry M/L rows (added by cog_pipeline PR + #59); aborts with class `uscogdata_ig_categories_unsupported` on an + older corpus rather than silently under-reporting. Mutually + exclusive with `recipe` (a recipe already defines its own component + codes). + + **Do not sum `"total"` results across levels of government** (e.g. + state + county + city): a state's `M12` payment to a school district is + the same dollar the district reports as its own direct `E12`, so + summing both double-counts it. This matters in particular with + [cog_geographic_rollup()], which sums across exactly that kind of + multi-layer government set. In the legacy wide era (<= FY2011), some functions are published ONLY as an aggregate-flagged family total (e.g. Corrections' `E04`/`E05` diff --git a/tests/testthat/test-complete.R b/tests/testthat/test-complete.R index 4a2c771..316c99e 100644 --- a/tests/testthat/test-complete.R +++ b/tests/testthat/test-complete.R @@ -14,9 +14,11 @@ # The (subtype, category) cells that SHOULD exist for one government-year: # every code in force for that government's type, mapped through -# summary_categories, matching the verb's flow prefixes and excluding -# aggregate-flagged codes (which spending_long/revenue_long drop). -raw_expected_cells <- function(govid, year, prefixes, subtype_col) { +# summary_categories, matching the verb's crosswalk subtype scope (the +# default concept, `primary`, is operations/capital/assistance -- see +# uscogdata#11) and excluding aggregate-flagged codes (which +# spending_long/revenue_long drop). +raw_expected_cells <- function(govid, year, subtypes, subtype_col) { fx <- sub("/$", "", Sys.getenv("USCOGDATA_URL")) q <- function(f) sprintf("read_parquet('%s/data/%s')", fx, f) wt_raw_query(sprintf( @@ -27,15 +29,18 @@ raw_expected_cells <- function(govid, year, prefixes, subtype_col) { WHERE x.canonical_govid = '%s' AND cs.year = %d AND NOT cs.is_aggregate - AND LEFT(cs.item_code, 1) IN (%s) AND c.category IS NOT NULL - AND c.%s IS NOT NULL", + AND c.%s IN (%s)", subtype_col, q("code_set.parquet"), q("canonical_fips_xwalk.parquet"), q("summary_categories.parquet"), govid, year, - paste0("'", prefixes, "'", collapse = ","), subtype_col + subtype_col, paste0("'", subtypes, "'", collapse = ",") )) } +# The default expenditure concept's subtype scope, mirrored from +# R/spending.R's .spend_subtypes_primary. +primary_subtypes <- c("operations", "capital", "assistance") + test_that("complete = FALSE is the default and changes nothing", { skip_if_no_corpus() with_fixture_corpus({ @@ -54,7 +59,7 @@ test_that("complete = TRUE round-trips a dense-source year to the pre-sparsifica # reproduce that cell set exactly. r <- cog_spending("121011212191", 2011L, complete = TRUE) expected <- raw_expected_cells("121011212191", 2011L, - c("E", "F", "G"), "spend_subtype") + primary_subtypes, "spend_subtype") key <- function(sub, cat) paste(sub, cat, sep = "|") expect_setequal(key(r$spend_subtype, r$category), @@ -108,7 +113,7 @@ test_that("the fill is scoped to each government's own type", { # code_set puts in force for type 1 (county) specifically. r <- cog_spending("121011212191", 2011L, complete = TRUE) county_cells <- raw_expected_cells("121011212191", 2011L, - c("E", "F", "G"), "spend_subtype") + primary_subtypes, "spend_subtype") expect_true(all(r$category %in% county_cells$category)) }) }) @@ -128,7 +133,7 @@ test_that("cog_revenue() completes on its own flow", { with_fixture_corpus({ r <- cog_revenue("121011212191", 2011L, complete = TRUE) expected <- raw_expected_cells("121011212191", 2011L, - c("T", "A", "U", "B", "C", "D"), + c("own_source", "federal", "state", "local_aid"), "revenue_subtype") key <- function(sub, cat) paste(sub, cat, sep = "|") expect_setequal(key(r$revenue_subtype, r$category), diff --git a/tests/testthat/test-expenditure-concept.R b/tests/testthat/test-expenditure-concept.R index 4151921..0afd23c 100644 --- a/tests/testthat/test-expenditure-concept.R +++ b/tests/testthat/test-expenditure-concept.R @@ -13,9 +13,13 @@ test_that("the corpus contains no K-prefix rows, so the Direct leg omits K", { } }) -test_that("expenditure_concept defaults to direct and preserves today's numbers", { +test_that("expenditure_concept defaults to primary; direct matches it on a pure operations/capital category", { gov <- "010000226085" # Alabama state government base <- cog_spending(gov, years = 2019, category = "Police") + expect_equal(attr(base, "provenance")$expenditure_concept, "primary") + # Police maps only to operations/capital codes (E62/F62/G62), so the + # direct concept's extra subtypes (interest, insurance_benefits) cannot + # contribute and the two concepts must agree exactly here. expl <- cog_spending(gov, years = 2019, category = "Police", expenditure_concept = "direct") expect_equal(base$amt_nominal, expl$amt_nominal) @@ -59,7 +63,9 @@ test_that("the IG leg never includes the L-- family total", { codes <- DBI::dbGetQuery(con, "SELECT DISTINCT item_code FROM ig_long")$item_code expect_false(any(grepl("--$", codes))) - expect_true(all(substr(codes, 1, 1) %in% c("M", "L"))) + # Q joined the IG family with the crosswalk-membership rewrite + # (uscogdata#11 / F-017: Q11/Q12/Q18 are state payments to school systems). + expect_true(all(substr(codes, 1, 1) %in% c("M", "L", "Q"))) }) test_that("expenditure_concept rejects unknown values", { @@ -246,9 +252,12 @@ test_that("both cross-government verbs still accept the direct default", { }) test_that("provenance always records the expenditure concept", { - d <- cog_spending("010000226085", years = 2019, category = "Police") + p <- cog_spending("010000226085", years = 2019, category = "Police") + d <- cog_spending("010000226085", years = 2019, category = "Police", + expenditure_concept = "direct") t <- cog_spending("010000226085", years = 2019, category = "Police", expenditure_concept = "total") + expect_equal(attr(p, "provenance")$expenditure_concept, "primary") expect_equal(attr(d, "provenance")$expenditure_concept, "direct") expect_equal(attr(t, "provenance")$expenditure_concept, "total") # The note explains the non-obvious part: how legacy IG was assembled. diff --git a/tests/testthat/test-expenditure-concepts.R b/tests/testthat/test-expenditure-concepts.R index 1349e2c..927f123 100644 --- a/tests/testthat/test-expenditure-concepts.R +++ b/tests/testthat/test-expenditure-concepts.R @@ -19,8 +19,6 @@ # also check the FY2022 numbers above. test_that("expenditure concepts classify on spend_type, not item-code prefix", { - testthat::skip("Blocked on uscogdata#11 (findings F-012, F-017, F-018)") - mad <- "552025209777" # MADISON CITY, WI wi_state <- "550000227544" # WISCONSIN (state government) @@ -61,15 +59,54 @@ test_that("expenditure concepts classify on spend_type, not item-code prefix", { # -- F-018: prefix Y splits revenue from expenditure, by spend_type --------- # Y01/Y02 are Insurance Trust revenue; Y05/Y06 are Insurance Trust benefit - # payments. All four share the first letter `Y` and the spend_type - # "Insurance Trust", so this pair of assertions is the concrete proof that - # classification is no longer keyed on the first letter. + # payments. All four share the first letter `Y`, so no first-letter allowlist + # can route them. The proof that classification is crosswalk-keyed: + # Y05 lands in `total` spending (insurance_benefits is inside `direct`), + # while Y01 -- same prefix -- is classified `revenue` by the crosswalk and + # therefore can never appear in a spending result. + # + # Per the owner's 2026-07-30 ruling (#11 DoD item 4 vs #12), cog_revenue()'s + # DEFAULT stays Census General Revenue and so excludes insurance-trust + # revenue; Y01's revenue-side classification is asserted against the + # crosswalk itself, not the default call. Surfacing Y01 through an explicit + # revenue concept argument is uscogdata#12. wi_revenue <- cog_revenue(govid = wi_state, years = 2019L) spend_codes <- wt_codes_included(wi_total) rev_codes <- wt_codes_included(wi_revenue) expect_true("Y05" %in% spend_codes) expect_false("Y05" %in% rev_codes) - expect_true("Y01" %in% rev_codes) expect_false("Y01" %in% spend_codes) + expect_false("Y01" %in% rev_codes) # default = general revenue (#12 ruling) + + con <- uscogdata:::.ensure_session() + y_class <- DBI::dbGetQuery(con, + "SELECT item_code, category_type, spend_subtype, revenue_subtype + FROM summary_categories WHERE item_code IN ('Y01', 'Y05')") + expect_equal(y_class$category_type[y_class$item_code == "Y01"], "revenue") + expect_equal(y_class$revenue_subtype[y_class$item_code == "Y01"], "insurance_trust") + expect_equal(y_class$category_type[y_class$item_code == "Y05"], "expenditure") + expect_equal(y_class$spend_subtype[y_class$item_code == "Y05"], "insurance_benefits") +}) + +test_that("no balance code or category ever reaches a spending or revenue result (uscogdata#25)", { + # Stocks are not flows. The crosswalk's balance codes (W/X/Y/Z fund + # balances) share first letters with flow codes, so this could never be + # guaranteed under prefix classification; under crosswalk membership it + # falls out structurally -- asserted here at the verb level, on a + # government-year the fixture gives real balance rows (Wisconsin carries + # Y07/Y08/Y21/Y61-type balances in FY2019). + wi_state <- "550000227544" + con <- uscogdata:::.ensure_session() + balance <- DBI::dbGetQuery(con, + "SELECT item_code, category FROM summary_categories WHERE category_type = 'balance'") + expect_gt(nrow(balance), 0L) + + spend <- cog_spending(wi_state, 2019L, expenditure_concept = "total") + rev <- cog_revenue(wi_state, 2019L) + + expect_false(any(spend$category %in% balance$category)) + expect_false(any(rev$category %in% balance$category)) + expect_length(intersect(wt_codes_included(spend), balance$item_code), 0L) + expect_length(intersect(wt_codes_included(rev), balance$item_code), 0L) }) diff --git a/tests/testthat/test-explain.R b/tests/testthat/test-explain.R index fc8f7d3..bbcfc41 100644 --- a/tests/testthat/test-explain.R +++ b/tests/testthat/test-explain.R @@ -83,7 +83,7 @@ test_that("cog_explain prints the expenditure concept (I1)", { capture.output(cog_explain(t)), capture.output(cog_explain(t), type = "message") ), collapse = "\n") - expect_true(grepl("Concept: direct", txt_d)) + expect_true(grepl("Concept: primary", txt_d)) expect_true(grepl("Concept: total", txt_t)) }) diff --git a/tests/testthat/test-views.R b/tests/testthat/test-views.R index 587a610..1153247 100644 --- a/tests/testthat/test-views.R +++ b/tests/testthat/test-views.R @@ -36,8 +36,8 @@ test_that("inst/sql/22- and 23- harmonized views enforce every WHERE predicate ( # {url} exactly as .register_views() does, and executes them -- plus # their 10-long.sql dependency -- against a synthetic hive-partitioned # parquet tree written to a temp dir. A regression in any predicate (e.g. - # `NOT is_aggregate` dropped, the prefix list changed, the NULL guard - # removed) would change which of the rows below survive. + # `NOT is_aggregate` dropped, the crosswalk-membership subquery changed, + # the NULL guard removed) would change which of the rows below survive. # # The synthetic parquet is written with DuckDB's own COPY ... TO (FORMAT # PARQUET) rather than the arrow package: this package has no arrow @@ -61,25 +61,45 @@ test_that("inst/sql/22- and 23- harmonized views enforce every WHERE predicate ( ('spend-B', 'E38', 50, false, 'E36'), -- collapse-fold: passes every predicate, renamed to E36 ('spend-C', 'E05', 999999, true, 'E05'), -- excluded ONLY by `NOT is_aggregate` ('spend-D', 'E99', 888888, false, NULL), -- excluded by `harmonized_code IS NOT NULL` - -- 'S74' is outside BOTH flow families (E/F/G/K spending and - -- T/A/U/B/C/D revenue -- it mirrors the real corpus's own - -- non-flow-type codes like S74/Z61), so it can only leak into - -- EITHER view via the E/F/G/K or T/A/U/B/C/D prefix filter, never - -- both at once -- a prefix drawn from the other view's own family - -- (e.g. a real T-code for the spending row) would incorrectly - -- leak into the other view's assertion below and not discriminate - -- the predicate under test. - ('spend-E', 'S74', 777777, false, 'S74'), -- excluded ONLY by the E/F/G/K prefix filter - -- Revenue (T/A/U/B/C/D) rows, exercised against revenue_long_harmonized: + -- 'S74' and 'Z61' are classified `balance` in the synthetic + -- crosswalk below (mirroring the real corpus's own non-flow codes), + -- so each is excluded from its view ONLY by the crosswalk-membership + -- subquery -- the mechanism that replaced the prefix allowlists + -- (uscogdata#11) and keeps balance stocks out of both flows + -- (uscogdata#25). + ('spend-E', 'S74', 777777, false, 'S74'), -- excluded ONLY by crosswalk membership (balance) + -- Revenue rows, exercised against revenue_long_harmonized: ('rev-A', 'U11', 200, false, 'U11'), -- control: passes every predicate as-is ('rev-B', 'U10', 25, false, 'U11'), -- collapse-fold: passes every predicate, renamed to U11 ('rev-C', 'T29', 555555, true, 'T29'), -- excluded ONLY by `NOT is_aggregate` ('rev-D', 'T88', 444444, false, NULL), -- excluded by `harmonized_code IS NOT NULL` - ('rev-E', 'Z61', 333333, false, 'Z61') -- excluded ONLY by the T/A/U/B/C/D prefix filter + ('rev-E', 'Z61', 333333, false, 'Z61') -- excluded ONLY by crosswalk membership (balance) ) AS t(canonical_govid, item_code, amt, is_aggregate, harmonized_code) ) TO %s (FORMAT PARQUET) ", uscogdata:::.sql_lit_chr(part_path))) + # The flow views classify by membership in summary_categories, so the + # synthetic corpus needs one too. Every flow code above is a member of its + # own flow (so is_aggregate / NULL-harmonized exclusions stay the SOLE + # excluder for those rows); S74/Z61 are members but classified balance, so + # membership itself is what excludes them. + DBI::dbExecute(write_con, sprintf(" + COPY ( + SELECT * FROM (VALUES + ('E36', 'Water Utilities', 'expenditure', 'operations', NULL), + ('E38', 'Water Utilities', 'expenditure', 'operations', NULL), + ('E05', 'Corrections', 'expenditure', 'operations', NULL), + ('E99', 'Other', 'expenditure', 'operations', NULL), + ('S74', 'Fund Balances', 'balance', NULL, NULL), + ('U11', 'Interest Earnings','revenue', NULL, 'own_source'), + ('U10', 'Interest Earnings','revenue', NULL, 'own_source'), + ('T29', 'Other Taxes', 'revenue', NULL, 'own_source'), + ('T88', 'Other Taxes', 'revenue', NULL, 'own_source'), + ('Z61', 'Fund Balances', 'balance', NULL, NULL) + ) AS t(item_code, category, category_type, spend_subtype, revenue_subtype) + ) TO %s (FORMAT PARQUET) + ", uscogdata:::.sql_lit_chr(file.path(tmp, "data", "summary_categories.parquet")))) + sql_dir <- system.file("sql", package = "uscogdata") .read_view_sql <- function(filename) { txt <- paste(readLines(file.path(sql_dir, filename), warn = FALSE), collapse = "\n") @@ -89,6 +109,7 @@ test_that("inst/sql/22- and 23- harmonized views enforce every WHERE predicate ( con <- DBI::dbConnect(duckdb::duckdb()) on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE) DBI::dbExecute(con, .read_view_sql("10-long.sql")) + DBI::dbExecute(con, .read_view_sql("11-summary_categories.sql")) DBI::dbExecute(con, .read_view_sql("22-spending_long_harmonized.sql")) DBI::dbExecute(con, .read_view_sql("23-revenue_long_harmonized.sql")) @@ -97,8 +118,8 @@ test_that("inst/sql/22- and 23- harmonized views enforce every WHERE predicate ( GROUP BY item_code ORDER BY item_code" ) # Exactly one surviving row: spend-C (aggregate), spend-D (NULL - # harmonized_code), and spend-E (wrong prefix family) must all be gone, - # and spend-A + spend-B must be folded together under E36. + # harmonized_code), and spend-E (balance, not an expenditure member) must + # all be gone, and spend-A + spend-B must be folded together under E36. expect_equal(nrow(spend), 1L) expect_equal(spend$item_code, "E36") expect_equal(spend$amt, 150) @@ -138,12 +159,27 @@ test_that("inst/sql/24- and 25- IG views retain aggregates, COALESCE NULL harmon ('ig-A', 'M04', 100, false, 'M04'), -- control: passes through as-is ('ig-B', 'M38', 50, false, 'M36'), -- fold control: real SB012 rule, renamed to M36 under harmonized basis ('ig-C', 'M47', 99999, true, NULL), -- legacy aggregate, NO harmonized_code: must survive BOTH views - ('ig-D', 'L--', 55555, false, 'L--'), -- family total: excluded from BOTH views - ('ig-E', 'T29', 44444, false, 'T29') -- wrong prefix (revenue, not M/L): excluded from BOTH views + ('ig-D', 'L--', 55555, false, 'L--'), -- family total: deliberately NOT a crosswalk member, excluded from BOTH views + ('ig-E', 'T29', 44444, false, 'T29') -- revenue member, not intergovernmental: excluded from BOTH views ) AS t(canonical_govid, item_code, amt, is_aggregate, harmonized_code) ) TO %s (FORMAT PARQUET) ", uscogdata:::.sql_lit_chr(part_path))) + # The IG views classify by summary_categories membership + # (spend_subtype = 'intergovernmental'). L-- is deliberately absent -- + # exactly as it is from the real crosswalk -- which is what excludes it. + DBI::dbExecute(write_con, sprintf(" + COPY ( + SELECT * FROM (VALUES + ('M04', 'Corrections', 'expenditure', 'intergovernmental', NULL), + ('M38', 'Health', 'expenditure', 'intergovernmental', NULL), + ('M36', 'Health', 'expenditure', 'intergovernmental', NULL), + ('M47', 'IG Other', 'expenditure', 'intergovernmental', NULL), + ('T29', 'Other Taxes', 'revenue', NULL, 'own_source') + ) AS t(item_code, category, category_type, spend_subtype, revenue_subtype) + ) TO %s (FORMAT PARQUET) + ", uscogdata:::.sql_lit_chr(file.path(tmp, "data", "summary_categories.parquet")))) + sql_dir <- system.file("sql", package = "uscogdata") .read_view_sql <- function(filename) { txt <- paste(readLines(file.path(sql_dir, filename), warn = FALSE), collapse = "\n") @@ -153,6 +189,7 @@ test_that("inst/sql/24- and 25- IG views retain aggregates, COALESCE NULL harmon con <- DBI::dbConnect(duckdb::duckdb()) on.exit(DBI::dbDisconnect(con, shutdown = TRUE), add = TRUE) DBI::dbExecute(con, .read_view_sql("10-long.sql")) + DBI::dbExecute(con, .read_view_sql("11-summary_categories.sql")) DBI::dbExecute(con, .read_view_sql("24-ig_long.sql")) DBI::dbExecute(con, .read_view_sql("25-ig_long_harmonized.sql")) @@ -160,8 +197,9 @@ test_that("inst/sql/24- and 25- IG views retain aggregates, COALESCE NULL harmon "SELECT item_code, SUM(amt) AS amt FROM ig_long GROUP BY item_code ORDER BY item_code" ) - # L-- (family total) and T29 (wrong prefix) are gone; the aggregate row - # M47 survives -- proof `NOT is_aggregate` is absent from ig_long. + # L-- (family total, not a member) and T29 (revenue, not IG) are gone; the + # aggregate row M47 survives -- proof `NOT is_aggregate` is absent from + # ig_long. expect_equal(raw$item_code, c("M04", "M38", "M47")) expect_equal(raw$amt, c(100, 50, 99999)) @@ -322,14 +360,32 @@ test_that(".harmonization_view_files guard is necessary: registration against a ) }) -test_that("spending_long filters to E/F/G/K prefixes and excludes aggregates", { +test_that("spending_long carries exactly the non-IG expenditure crosswalk codes and excludes aggregates", { skip_if_no_corpus() con <- cog_open() on.exit(cog_close()) - prefixes <- DBI::dbGetQuery(con, - "SELECT DISTINCT LEFT(item_code, 1) AS pfx FROM spending_long" - )$pfx - expect_true(all(prefixes %in% c("E", "F", "G", "K"))) + + # Classification is crosswalk membership, not prefixes (uscogdata#11): + # every row's code must classify as expenditure and never as + # intergovernmental (which lives in ig_long). + stray <- DBI::dbGetQuery(con, + "SELECT DISTINCT s.item_code + FROM spending_long s + LEFT JOIN summary_categories c USING (item_code) + WHERE c.category_type IS DISTINCT FROM 'expenditure' + OR c.spend_subtype = 'intergovernmental'" + )$item_code + expect_length(stray, 0L) + + # Balance codes are stocks, not flows -- they must never appear in a + # spending result (uscogdata#25). Prefix filtering could not guarantee + # this (W/X/Y/Z balance codes share letters with flow codes). + balance_n <- DBI::dbGetQuery(con, + "SELECT count(*) AS n FROM spending_long WHERE item_code IN ( + SELECT item_code FROM summary_categories WHERE category_type = 'balance' + )" + )$n + expect_equal(balance_n, 0) agg_count <- DBI::dbGetQuery(con, "SELECT count(*) AS n FROM spending_long WHERE is_aggregate" @@ -337,14 +393,29 @@ test_that("spending_long filters to E/F/G/K prefixes and excludes aggregates", { expect_equal(agg_count, 0) }) -test_that("revenue_long filters to T/A/U/B/C/D prefixes and excludes aggregates", { +test_that("revenue_long carries exactly the general-revenue crosswalk codes and excludes aggregates", { skip_if_no_corpus() con <- cog_open() on.exit(cog_close()) - prefixes <- DBI::dbGetQuery(con, - "SELECT DISTINCT LEFT(item_code, 1) AS pfx FROM revenue_long" - )$pfx - expect_true(all(prefixes %in% c("T", "A", "U", "B", "C", "D"))) + + # General Revenue scope: revenue crosswalk members minus insurance_trust + # (owner ruling 2026-07-30; an explicit wider concept is uscogdata#12). + stray <- DBI::dbGetQuery(con, + "SELECT DISTINCT s.item_code + FROM revenue_long s + LEFT JOIN summary_categories c USING (item_code) + WHERE c.category_type IS DISTINCT FROM 'revenue' + OR c.revenue_subtype = 'insurance_trust'" + )$item_code + expect_length(stray, 0L) + + # No balance stock ever appears in a revenue result (uscogdata#25). + balance_n <- DBI::dbGetQuery(con, + "SELECT count(*) AS n FROM revenue_long WHERE item_code IN ( + SELECT item_code FROM summary_categories WHERE category_type = 'balance' + )" + )$n + expect_equal(balance_n, 0) agg_count <- DBI::dbGetQuery(con, "SELECT count(*) AS n FROM revenue_long WHERE is_aggregate" diff --git a/vignettes/total-spending.Rmd b/vignettes/total-spending.Rmd index 71291a8..78a8319 100644 --- a/vignettes/total-spending.Rmd +++ b/vignettes/total-spending.Rmd @@ -1,8 +1,8 @@ --- -title: "Total spending: Direct, Total, and when each is right" +title: "Total spending: Primary, Direct, Total, and when each is right" output: rmarkdown::html_vignette vignette: > - %\VignetteIndexEntry{Total spending: Direct, Total, and when each is right} + %\VignetteIndexEntry{Total spending: Primary, Direct, Total, and when each is right} %\VignetteEngine{knitr::rmarkdown} %\VignetteEncoding{UTF-8} --- @@ -17,17 +17,28 @@ knitr::opts_chunk$set(collapse = TRUE, comment = "#>") is about one government or several: 1. **"What did my county spend in total, a decade ago vs today?"** — one - government, tracked over time. Either `direct` or `total` spending answers - this correctly, as long as the same concept is used for both years. + government, tracked over time. Any concept answers this correctly, as + long as the same concept is used for both years. 2. **"How do all the counties in my state compare, a decade ago vs today, against the neighboring state?"** — several governments, summed together. - Here only `direct` gives the right answer; summing `total` across - governments double-counts money that passes between them. + Here only a non-intergovernmental concept (`primary` or `direct`) gives + the right answer; summing `total` across governments double-counts money + that passes between them. -`cog_spending()`'s `expenditure_concept` argument (`"direct"` or `"total"`) -controls which of these a query answers. This vignette walks through both -questions with code that actually runs against the package's bundled fixture -corpus, then explains why the second question refuses `"total"` outright. +`cog_spending()`'s `expenditure_concept` argument controls which of these a +query answers, via three nested concepts defined as sets of the crosswalk's +`spend_subtype` values (never item-code first letters — the letter `Y` alone +spans revenue, expenditure, and balance codes): + +- `"primary"` (the default) — the government's own service provision: + `operations` + `capital` + `assistance`. +- `"direct"` — Census's published Direct Expenditure: `primary` plus + `interest` on debt and `insurance_benefits` (e.g. pension payments). +- `"total"` — `direct` plus the `intergovernmental` leg. + +This vignette walks through both questions with code that actually runs +against the package's bundled fixture corpus, then explains why the second +question refuses `"total"` outright. Before any of the numbers below: every amount column here — `amt_nominal`, `amt_real`, and their `amt_per_capita_*` counterparts — is in **full US @@ -68,20 +79,23 @@ al_total <- cog_spending( al_total ``` -The `intergovernmental` rows are what `"total"` adds on top of `"direct"` -(`capital` + `operations`): Alabama's own payments out to counties and -cities for highway work. Because this query only ever concerns Alabama, -including that piece is safe -- there's no other government's number it -could be double-counted against. +The `intergovernmental` rows are what `"total"` adds on top of the +non-intergovernmental subtypes (here `capital` + `operations`): Alabama's +own payments out to counties and cities for highway work. Because this +query only ever concerns Alabama, including that piece is safe -- there's +no other government's number it could be double-counted against. -`"direct"` (the default) answers the same trend question just as validly: +`"primary"` (the default) answers the same trend question just as validly +(for Highways, which maps only to operations/capital codes, `"primary"` and +`"direct"` coincide -- there is no highway-specific interest or insurance +benefit to add): ```{r} -al_direct <- cog_spending( +al_primary <- cog_spending( "010000226085", years = c(2012, 2020), category = "Highways" - # expenditure_concept = "direct" is the default; shown here for contrast + # expenditure_concept = "primary" is the default; shown here for contrast ) -al_direct +al_primary ``` Both are internally consistent series. What breaks the comparison is @@ -93,8 +107,8 @@ every year in the series. # Archetype 2: a cross-government rollup `cog_geographic_rollup()` sums spending across state/county/city layers for -a place. Its default -- and, as shown below, its *only* accepted value for -`expenditure_concept` -- is `"direct"`: +a place. Its default is `"primary"`, and (as shown below) it accepts only +the non-intergovernmental concepts, `"primary"` and `"direct"`: ```{r} fl_rollup <- cog_geographic_rollup( @@ -145,15 +159,15 @@ shows up **twice** in the underlying corpus: the county is the government that actually lets the contract and pays the paving crew. -`direct` (item codes `E`/`F`/`G`) only ever counts the second of those -- -the government that actually did the spending. `total` (Direct plus the -`M`/`L` intergovernmental legs) counts the first one *as well*, which is -exactly right for describing Alabama's own budget: Alabama's `total` -genuinely includes the $10M it committed to highways, whether it built the -road itself or paid the county to. But sum `total` across Alabama **and** -the county, and that $10M is counted twice -- once as Alabama's payment out, -once as the county's spending in -- reporting $20M of highway work for $10M -actually spent. +`primary` and `direct` (the crosswalk's non-intergovernmental expenditure +subtypes) only ever count the second of those -- the government that +actually did the spending. `total` (Direct plus the intergovernmental leg) +counts the first one *as well*, which is exactly right for describing +Alabama's own budget: Alabama's `total` genuinely includes the $10M it +committed to highways, whether it built the road itself or paid the county +to. But sum `total` across Alabama **and** the county, and that $10M is +counted twice -- once as Alabama's payment out, once as the county's +spending in -- reporting $20M of highway work for $10M actually spent. This is exactly the shape of query `cog_geographic_rollup()` exists to run (summing across layers of government), so it refuses `"total"` rather than @@ -168,48 +182,51 @@ share of a government's own Direct spending is: | Government type | Intergovernmental / Direct | |---|---| -| State | 16.7%-48.4% (varies by year; 24.0% pooled across all four) | -| County | 3.4%-5.1% (varies by year) | -| City | 2.6%-3.1% (varies by year) | +| State | 33.1%-40.5% (varies by year; 36.2% pooled across all four) | +| County | 3.3%-4.8% (varies by year) | +| City | 2.4%-2.9% (varies by year) | -So the Direct/Total choice matters overwhelmingly for **state** governments --- a state's Total genuinely differs from its Direct by a meaningful margin, -while for a county or city the two are close. The state range is also far -wider than a single flat figure would suggest: legacy wide-era years (2011: -48.4%) carry proportionally more intergovernmental spending than the modern -era (2019-2020: 16.7%-17.0%), so a state's Direct/Total gap can be nearly -3x larger a decade earlier than it is today. That's also why the mistake -this vignette warns about is easy to make unnoticed at the county/city level -and costly at the state level: rolling up every government in a state using -`total` instead of `direct` overstates the true figure -- measured at 7.6% -for Alabama in FY2019, and 11.6% nationally. +So the Direct/Total choice matters overwhelmingly for **state** +governments -- a state's Total genuinely differs from its Direct by more +than a third, while for a county or city the two are close. (The state +share is much larger than pre-#11 measurements suggested, because the +intergovernmental leg now correctly includes the `Q11`/`Q12`/`Q18` state +payments to school systems -- for most states the single largest transfer +they make.) That's also why the mistake this vignette warns about is easy +to make unnoticed at the county/city level and costly at the state level: +rolling up every government using `total` instead of `primary`/`direct` +overstates the FY2019 figure by 24.1% for Alabama and 23.2% nationally. -# Why Total = Direct + M + L, not Direct + M +# Why Total = Direct + M + L + Q, not Direct + M -It's tempting to assume `total` only needs to add `M`. But `M` and `L` are -both money the queried government itself pays **out** -- they're not two -different accounts of a receiving government's revenue. `M` is what it -pays to other **local** governments (e.g. a county paying a city for a -shared paving contract); `L` is what it pays **up** to its **state** -government (e.g. a county's contribution to a state-administered program). -A local government's Total genuinely includes both legs, because both are -its own spending, just routed to a different kind of recipient. On the -bundled fixture corpus (all 50 states, 2011/2012/2019/2020), `L` is 0 for -state governments (a state has no "payments to the state government" leg of -its own) but is 43%-51% the size of `M` for counties (varies by year) and -144%-189% the size of `M` for cities (varies by year; 166% pooled across -all four) -- so a `total` that omitted `L` would silently undercount Total -specifically for local governments, and for cities `L` is often the -*larger* of the two legs. -`cog_spending(expenditure_concept = "total")` includes both legs (excluding -the `L--` family-total rollup row, which would double-count its own -components). +It's tempting to assume `total` only needs to add `M`. But the +intergovernmental leg has three families, all money the queried government +itself pays **out** -- they're not different accounts of a receiving +government's revenue. `M` is what it pays to other **local** governments +(e.g. a county paying a city for a shared paving contract); `L` is what it +pays **up** to its **state** government (e.g. a county's contribution to a +state-administered program); and `Q11`/`Q12`/`Q18` are a state's payments +to **school systems** (K-12 and higher-ed aid -- for most states the +single largest transfer they make, and the piece the pre-#11 prefix +allowlist silently dropped, finding F-017). A government's Total genuinely +includes every leg it pays, because each is its own spending, just routed +to a different kind of recipient. On the bundled fixture corpus (all 50 +states, 2011/2012/2019/2020), `L` is 0 for state governments (a state has +no "payments to the state government" leg of its own) but is 43%-51% the +size of `M` for counties (varies by year) and 144%-189% the size of `M` +for cities (varies by year; 166% pooled across all four) -- so a `total` +that omitted `L` would silently undercount Total specifically for local +governments, and for cities `L` is often the *larger* of the two legs. +`cog_spending(expenditure_concept = "total")` includes every leg +(excluding the `L--` family-total rollup row, which would double-count its +own components). # Composition rules -- `expenditure_concept` (whose spending counts -- Direct vs Direct plus - intergovernmental) is **orthogonal** to `basis` (which vintage of the - item-code space a query resolves against -- `"harmonized"` vs `"raw"`). +- `expenditure_concept` (whose spending counts -- Primary, Direct, or + Direct plus intergovernmental) is **orthogonal** to `basis` (which + vintage of the item-code space a query resolves against -- + `"harmonized"` vs `"raw"`). They combine freely: `expenditure_concept = "total", basis = "raw"` is a valid, meaningful query, and so is every other pairing. - `expenditure_concept = "total"` is **mutually exclusive** with `recipe`: a @@ -225,12 +242,15 @@ components). # Summary -- Comparing one government to itself over time: `"direct"` or `"total"` - both work -- pick one and hold it fixed across every year compared. +- Comparing one government to itself over time: any concept works -- pick + one and hold it fixed across every year compared. - Comparing or summing across governments -- counties within a state, a - state against its neighbor, cities against counties: use `"direct"`. - `cog_geographic_rollup()` and `cog_peer_compare()` enforce this by - refusing `"total"`. -- `"total"` = Direct (`E`/`F`/`G`) + intergovernmental (`M` to local - governments + `L` to the state government, excluding the `L--` - family-total row). + state against its neighbor, cities against counties: use `"primary"` + (the default) or `"direct"`. `cog_geographic_rollup()` and + `cog_peer_compare()` enforce this by refusing `"total"`. +- `"primary"` = operations + capital + assistance. `"direct"` = primary + + interest on debt + insurance trust benefits (Census's published Direct + Expenditure). `"total"` = direct + intergovernmental (`M` to local + governments, `L` to the state government excluding the `L--` + family-total row, and `Q11`/`Q12`/`Q18` state payments to school + systems).